Qt
Internal/Contributor docs for the Qt SDK. Note: These are NOT official API docs; those are found at https://doc.qt.io/
Loading...
Searching...
No Matches
qstring.cpp
Go to the documentation of this file.
1// Copyright (C) 2021 The Qt Company Ltd.
2// Copyright (C) 2022 Intel Corporation.
3// Copyright (C) 2019 Mail.ru Group.
4// SPDX-License-Identifier: LicenseRef-Qt-Commercial OR LGPL-3.0-only OR GPL-2.0-only OR GPL-3.0-only
5// Qt-Security score:critical reason:data-parser
6
7#include "qstringlist.h"
8#if QT_CONFIG(regularexpression)
9#include "qregularexpression.h"
10#endif
12#include <private/qstringconverter_p.h>
13#include <private/qtools_p.h>
15#include "private/qsimd_p.h"
16#include <qnumeric.h>
17#include <qdatastream.h>
18#include <qlist.h>
19#include "qlocale.h"
20#include "qlocale_p.h"
21#include "qspan.h"
22#include "qstringbuilder.h"
23#include "qstringmatcher.h"
25#include "qdebug.h"
26#include "qendian.h"
27#include "qcollator.h"
28#include "qttypetraits.h"
29
30#ifdef Q_OS_DARWIN
31#include <private/qcore_mac_p.h>
32#endif
33
34#include <private/qfunctions_p.h>
35
36#include <limits.h>
37#include <string.h>
38#include <stdlib.h>
39#include <stdio.h>
40#include <stdarg.h>
41#include <wchar.h>
42
43#include "qchar.cpp"
48#include "qthreadstorage.h"
49
50#include <algorithm>
51#include <functional>
52
53#ifdef Q_OS_WIN
54# include <qt_windows.h>
55# if !defined(QT_BOOTSTRAPPED) && (defined(QT_NO_CAST_FROM_ASCII) || defined(QT_NO_CAST_TO_ASCII))
56// MSVC requires this, but let's apply it to MinGW compilers too, just in case
57# error "This file cannot be compiled with QT_NO_CAST_{TO,FROM}_ASCII, "
58 "otherwise some QString functions will not get exported."
59# endif
60#endif
61
62#ifdef truncate
63# undef truncate
64#endif
65
66#define REHASH(a)
67 if (sl_minus_1 < sizeof(sl_minus_1) * CHAR_BIT)
68 hashHaystack -= decltype(hashHaystack)(a) << sl_minus_1;
69 hashHaystack <<= 1
70
72
73using namespace Qt::StringLiterals;
74using namespace QtMiscUtils;
75
76const char16_t QString::_empty = 0;
77
78// in qstringmatcher.cpp
79qsizetype qFindStringBoyerMoore(QStringView haystack, qsizetype from, QStringView needle, Qt::CaseSensitivity cs);
80
81namespace {
82enum StringComparisonMode {
83 CompareStringsForEquality,
84 CompareStringsForOrdering
85};
86
87template <typename Pointer>
88char32_t foldCaseHelper(Pointer ch, Pointer start) = delete;
89
90template <>
91char32_t foldCaseHelper<const QChar*>(const QChar* ch, const QChar* start)
92{
93 return foldCase(reinterpret_cast<const char16_t*>(ch),
94 reinterpret_cast<const char16_t*>(start));
95}
96
97template <>
98char32_t foldCaseHelper<const char*>(const char* ch, const char*)
99{
100 return foldCase(char16_t(uchar(*ch)));
101}
102
103template <typename T>
104char16_t valueTypeToUtf16(T t) = delete;
105
106template <>
107char16_t valueTypeToUtf16<QChar>(QChar t)
108{
109 return t.unicode();
110}
111
112template <>
113char16_t valueTypeToUtf16<char>(char t)
114{
115 return char16_t{uchar(t)};
116}
117
118template <typename T>
119static inline bool foldAndCompare(const T a, const T b)
120{
121 return foldCase(a) == b;
122}
123
124/*!
125 \internal
126
127 Returns the index position of the first occurrence of the
128 character \a ch in the string given by \a str and \a len,
129 searching forward from index
130 position \a from. Returns -1 if \a ch could not be found.
131*/
132template <typename Haystack>
133static inline qsizetype qLastIndexOf(Haystack haystack, QChar needle,
134 qsizetype from, Qt::CaseSensitivity cs) noexcept
135{
136 if (haystack.size() == 0)
137 return -1;
138 if (from < 0)
139 from += haystack.size();
140 else if (std::size_t(from) > std::size_t(haystack.size()))
141 from = haystack.size() - 1;
142 if (from >= 0) {
143 char16_t c = needle.unicode();
144 const auto b = haystack.data();
145 auto n = b + from;
146 if (cs == Qt::CaseSensitive) {
147 for (; n >= b; --n)
148 if (valueTypeToUtf16(*n) == c)
149 return n - b;
150 } else {
151 c = foldCase(c);
152 for (; n >= b; --n)
153 if (foldCase(valueTypeToUtf16(*n)) == c)
154 return n - b;
155 }
156 }
157 return -1;
158}
159template <> qsizetype
160qLastIndexOf(QString, QChar, qsizetype, Qt::CaseSensitivity) noexcept = delete; // unwanted, would detach
161
162template<typename Haystack, typename Needle>
163static qsizetype qLastIndexOf(Haystack haystack0, qsizetype from,
164 Needle needle0, Qt::CaseSensitivity cs) noexcept
165{
166 const qsizetype sl = needle0.size();
167 if (sl == 1)
168 return qLastIndexOf(haystack0, needle0.front(), from, cs);
169
170 const qsizetype l = haystack0.size();
171 if (from < 0)
172 from += l;
173 if (from == l && sl == 0)
174 return from;
175 const qsizetype delta = l - sl;
176 if (std::size_t(from) > std::size_t(l) || delta < 0)
177 return -1;
178 if (from > delta)
179 from = delta;
180
181 auto sv = [sl](const typename Haystack::value_type *v) { return Haystack(v, sl); };
182
183 auto haystack = haystack0.data();
184 const auto needle = needle0.data();
185 const auto *end = haystack;
186 haystack += from;
187 const qregisteruint sl_minus_1 = sl ? sl - 1 : 0;
188 const auto *n = needle + sl_minus_1;
189 const auto *h = haystack + sl_minus_1;
190 qregisteruint hashNeedle = 0, hashHaystack = 0;
191
192 if (cs == Qt::CaseSensitive) {
193 for (qsizetype idx = 0; idx < sl; ++idx) {
194 hashNeedle = (hashNeedle << 1) + valueTypeToUtf16(*(n - idx));
195 hashHaystack = (hashHaystack << 1) + valueTypeToUtf16(*(h - idx));
196 }
197 hashHaystack -= valueTypeToUtf16(*haystack);
198
199 while (haystack >= end) {
200 hashHaystack += valueTypeToUtf16(*haystack);
201 if (hashHaystack == hashNeedle
202 && QtPrivate::compareStrings(needle0, sv(haystack), Qt::CaseSensitive) == 0)
203 return haystack - end;
204 --haystack;
205 REHASH(valueTypeToUtf16(haystack[sl]));
206 }
207 } else {
208 for (qsizetype idx = 0; idx < sl; ++idx) {
209 hashNeedle = (hashNeedle << 1) + foldCaseHelper(n - idx, needle);
210 hashHaystack = (hashHaystack << 1) + foldCaseHelper(h - idx, end);
211 }
212 hashHaystack -= foldCaseHelper(haystack, end);
213
214 while (haystack >= end) {
215 hashHaystack += foldCaseHelper(haystack, end);
216 if (hashHaystack == hashNeedle
217 && QtPrivate::compareStrings(sv(haystack), needle0, Qt::CaseInsensitive) == 0)
218 return haystack - end;
219 --haystack;
220 REHASH(foldCaseHelper(haystack + sl, end));
221 }
222 }
223 return -1;
224}
225
226template <typename Haystack, typename Needle>
227bool qt_starts_with_impl(Haystack haystack, Needle needle, Qt::CaseSensitivity cs) noexcept
228{
229 if (haystack.isNull())
230 return needle.isNull();
231 const auto haystackLen = haystack.size();
232 const auto needleLen = needle.size();
233 if (haystackLen == 0)
234 return needleLen == 0;
235 if (needleLen > haystackLen)
236 return false;
237
238 return QtPrivate::compareStrings(haystack.first(needleLen), needle, cs) == 0;
239}
240
241template <typename Haystack, typename Needle>
242bool qt_ends_with_impl(Haystack haystack, Needle needle, Qt::CaseSensitivity cs) noexcept
243{
244 if (haystack.isNull())
245 return needle.isNull();
246 const auto haystackLen = haystack.size();
247 const auto needleLen = needle.size();
248 if (haystackLen == 0)
249 return needleLen == 0;
250 if (haystackLen < needleLen)
251 return false;
252
253 return QtPrivate::compareStrings(haystack.last(needleLen), needle, cs) == 0;
254}
255
256template <typename T>
257static void append_helper(QString &self, T view)
258{
259 const auto strData = view.data();
260 const qsizetype strSize = view.size();
261 auto &d = self.data_ptr();
262 if (strData && strSize > 0) {
263 // the number of UTF-8 code units is always at a minimum equal to the number
264 // of equivalent UTF-16 code units
265 d.detachAndGrow(QArrayData::GrowsAtEnd, strSize, nullptr, nullptr);
266 Q_CHECK_PTR(d.data());
267 Q_ASSERT(strSize <= d.freeSpaceAtEnd());
268
269 auto dst = std::next(d.data(), d.size);
270 if constexpr (std::is_same_v<T, QUtf8StringView>) {
271 dst = QUtf8::convertToUnicode(dst, view);
272 } else if constexpr (std::is_same_v<T, QLatin1StringView>) {
273 QLatin1::convertToUnicode(dst, view);
274 dst += strSize;
275 } else {
276 static_assert(QtPrivate::type_dependent_false<T>(),
277 "Can only operate on UTF-8 and Latin-1");
278 }
279 self.resize(std::distance(d.begin(), dst));
280 } else if (d.isNull() && !view.isNull()) { // special case
281 self = QLatin1StringView("");
282 }
283}
284
285template <uint MaxCount> struct UnrollTailLoop
286{
287 template <typename RetType, typename Functor1, typename Functor2, typename Number>
288 static inline RetType exec(Number count, RetType returnIfExited, Functor1 loopCheck, Functor2 returnIfFailed, Number i = 0)
289 {
290 /* equivalent to:
291 * while (count--) {
292 * if (loopCheck(i))
293 * return returnIfFailed(i);
294 * }
295 * return returnIfExited;
296 */
297
298 if (!count)
299 return returnIfExited;
300
301 bool check = loopCheck(i);
302 if (check)
303 return returnIfFailed(i);
304
305 return UnrollTailLoop<MaxCount - 1>::exec(count - 1, returnIfExited, loopCheck, returnIfFailed, i + 1);
306 }
307
308 template <typename Functor, typename Number>
309 static inline void exec(Number count, Functor code)
310 {
311 /* equivalent to:
312 * for (Number i = 0; i < count; ++i)
313 * code(i);
314 */
315 exec(count, 0, [=](Number i) -> bool { code(i); return false; }, [](Number) { return 0; });
316 }
317};
318template <> template <typename RetType, typename Functor1, typename Functor2, typename Number>
319inline RetType UnrollTailLoop<0>::exec(Number, RetType returnIfExited, Functor1, Functor2, Number)
320{
321 return returnIfExited;
322}
323} // unnamed namespace
324
325/*
326 * Note on the use of SIMD in qstring.cpp:
327 *
328 * Several operations with strings are improved with the use of SIMD code,
329 * since they are repetitive. For MIPS, we have hand-written assembly code
330 * outside of qstring.cpp targeting MIPS DSP and MIPS DSPr2. For ARM and for
331 * x86, we can only use intrinsics and therefore everything is contained in
332 * qstring.cpp. We need to use intrinsics only for those platforms due to the
333 * different compilers and toolchains used, which have different syntax for
334 * assembly sources.
335 *
336 * ** SSE notes: **
337 *
338 * Whenever multiple alternatives are equivalent or near so, we prefer the one
339 * using instructions from SSE2, since SSE2 is guaranteed to be enabled for all
340 * 64-bit builds and we enable it for 32-bit builds by default. Use of higher
341 * SSE versions should be done when there is a clear performance benefit and
342 * requires fallback code to SSE2, if it exists.
343 *
344 * Performance measurement in the past shows that most strings are short in
345 * size and, therefore, do not benefit from alignment prologues. That is,
346 * trying to find a 16-byte-aligned boundary to operate on is often more
347 * expensive than executing the unaligned operation directly. In addition, note
348 * that the QString private data is designed so that the data is stored on
349 * 16-byte boundaries if the system malloc() returns 16-byte aligned pointers
350 * on its own (64-bit glibc on Linux does; 32-bit glibc on Linux returns them
351 * 50% of the time), so skipping the alignment prologue is actually optimizing
352 * for the common case.
353 */
354
355#if defined(__mips_dsp)
356// From qstring_mips_dsp_asm.S
357extern "C" void qt_fromlatin1_mips_asm_unroll4 (char16_t*, const char*, uint);
358extern "C" void qt_fromlatin1_mips_asm_unroll8 (char16_t*, const char*, uint);
359extern "C" void qt_toLatin1_mips_dsp_asm(uchar *dst, const char16_t *src, int length);
360#endif
361
362#if defined(__SSE2__) && defined(Q_CC_GNU)
363// We may overrun the buffer, but that's a false positive:
364// this won't crash nor produce incorrect results
365# define ATTRIBUTE_NO_SANITIZE __attribute__((__no_sanitize_address__, __no_sanitize_thread__))
366#else
367# define ATTRIBUTE_NO_SANITIZE
368#endif
369
370#ifdef __SSE2__
371static constexpr bool UseSse4_1 = bool(qCompilerCpuFeatures & CpuFeatureSSE4_1);
372static constexpr bool UseAvx2 = UseSse4_1 &&
373 (qCompilerCpuFeatures & CpuFeatureArchHaswell) == CpuFeatureArchHaswell;
374
375[[maybe_unused]]
376Q_ALWAYS_INLINE static __m128i mm_load8_zero_extend(const void *ptr)
377{
378 const __m128i *dataptr = static_cast<const __m128i *>(ptr);
379 if constexpr (UseSse4_1) {
380 // use a MOVQ followed by PMOVZXBW
381 // if AVX2 is present, these should combine into a single VPMOVZXBW instruction
382 __m128i data = _mm_loadl_epi64(dataptr);
383 return _mm_cvtepu8_epi16(data);
384 }
385
386 // use MOVQ followed by PUNPCKLBW
387 __m128i data = _mm_loadl_epi64(dataptr);
388 return _mm_unpacklo_epi8(data, _mm_setzero_si128());
389}
390
391[[maybe_unused]] ATTRIBUTE_NO_SANITIZE
392static qsizetype qustrlen_sse2(const char16_t *str) noexcept
393{
394 // find the 16-byte alignment immediately prior or equal to str
395 quintptr misalignment = quintptr(str) & 0xf;
396 Q_ASSERT((misalignment & 1) == 0);
397 const char16_t *ptr = str - (misalignment / 2);
398
399 // load 16 bytes and see if we have a null
400 // (aligned loads can never segfault)
401 const __m128i zeroes = _mm_setzero_si128();
402 __m128i data = _mm_load_si128(reinterpret_cast<const __m128i *>(ptr));
403 __m128i comparison = _mm_cmpeq_epi16(data, zeroes);
404 uint mask = _mm_movemask_epi8(comparison);
405
406 // ignore the result prior to the beginning of str
407 mask >>= misalignment;
408
409 // Have we found something in the first block? Need to handle it now
410 // because of the left shift above.
411 if (mask)
412 return qCountTrailingZeroBits(mask) / sizeof(char16_t);
413
414 constexpr qsizetype Step = sizeof(__m128i) / sizeof(char16_t);
415 qsizetype size = Step - misalignment / sizeof(char16_t);
416
417 size -= Step;
418 do {
419 size += Step;
420 data = _mm_load_si128(reinterpret_cast<const __m128i *>(str + size));
421
422 comparison = _mm_cmpeq_epi16(data, zeroes);
423 mask = _mm_movemask_epi8(comparison);
424 } while (mask == 0);
425
426 // found a null
427 return size + qCountTrailingZeroBits(mask) / sizeof(char16_t);
428}
429
430// Scans from \a ptr to \a end until \a maskval is non-zero. Returns true if
431// the no non-zero was found. Returns false and updates \a ptr to point to the
432// first 16-bit word that has any bit set (note: if the input is 8-bit, \a ptr
433// may be updated to one byte short).
434static bool simdTestMask(const char *&ptr, const char *end, quint32 maskval)
435{
436 auto updatePtr = [&](uint result) {
437 // found a character matching the mask
438 uint idx = qCountTrailingZeroBits(~result);
439 ptr += idx;
440 return false;
441 };
442
443 if constexpr (UseSse4_1) {
444# ifndef Q_OS_QNX // compiler fails in the code below
445 __m128i mask;
446 auto updatePtrSimd = [&](__m128i data) -> bool {
447 __m128i masked = _mm_and_si128(mask, data);
448 __m128i comparison = _mm_cmpeq_epi16(masked, _mm_setzero_si128());
449 uint result = _mm_movemask_epi8(comparison);
450 return updatePtr(result);
451 };
452
453 if constexpr (UseAvx2) {
454 // AVX2 implementation: test 32 bytes at a time
455 const __m256i mask256 = _mm256_broadcastd_epi32(_mm_cvtsi32_si128(maskval));
456 while (ptr + 32 <= end) {
457 __m256i data = _mm256_loadu_si256(reinterpret_cast<const __m256i *>(ptr));
458 if (!_mm256_testz_si256(mask256, data)) {
459 // found a character matching the mask
460 __m256i masked256 = _mm256_and_si256(mask256, data);
461 __m256i comparison256 = _mm256_cmpeq_epi16(masked256, _mm256_setzero_si256());
462 return updatePtr(_mm256_movemask_epi8(comparison256));
463 }
464 ptr += 32;
465 }
466
467 mask = _mm256_castsi256_si128(mask256);
468 } else {
469 // SSE 4.1 implementation: test 32 bytes at a time (two 16-byte
470 // comparisons, unrolled)
471 mask = _mm_set1_epi32(maskval);
472 while (ptr + 32 <= end) {
473 __m128i data1 = _mm_loadu_si128(reinterpret_cast<const __m128i *>(ptr));
474 __m128i data2 = _mm_loadu_si128(reinterpret_cast<const __m128i *>(ptr + 16));
475 if (!_mm_testz_si128(mask, data1))
476 return updatePtrSimd(data1);
477
478 ptr += 16;
479 if (!_mm_testz_si128(mask, data2))
480 return updatePtrSimd(data2);
481 ptr += 16;
482 }
483 }
484
485 // AVX2 and SSE4.1: final 16-byte comparison
486 if (ptr + 16 <= end) {
487 __m128i data1 = _mm_loadu_si128(reinterpret_cast<const __m128i *>(ptr));
488 if (!_mm_testz_si128(mask, data1))
489 return updatePtrSimd(data1);
490 ptr += 16;
491 }
492
493 // and final 8-byte comparison
494 if (ptr + 8 <= end) {
495 __m128i data1 = _mm_loadl_epi64(reinterpret_cast<const __m128i *>(ptr));
496 if (!_mm_testz_si128(mask, data1))
497 return updatePtrSimd(data1);
498 ptr += 8;
499 }
500
501 return true;
502# endif // QNX
503 }
504
505 // SSE2 implementation: test 16 bytes at a time.
506 const __m128i mask = _mm_set1_epi32(maskval);
507 while (ptr + 16 <= end) {
508 __m128i data = _mm_loadu_si128(reinterpret_cast<const __m128i *>(ptr));
509 __m128i masked = _mm_and_si128(mask, data);
510 __m128i comparison = _mm_cmpeq_epi16(masked, _mm_setzero_si128());
511 quint16 result = _mm_movemask_epi8(comparison);
512 if (result != 0xffff)
513 return updatePtr(result);
514 ptr += 16;
515 }
516
517 // and one 8-byte comparison
518 if (ptr + 8 <= end) {
519 __m128i data = _mm_loadl_epi64(reinterpret_cast<const __m128i *>(ptr));
520 __m128i masked = _mm_and_si128(mask, data);
521 __m128i comparison = _mm_cmpeq_epi16(masked, _mm_setzero_si128());
522 quint8 result = _mm_movemask_epi8(comparison);
523 if (result != 0xff)
524 return updatePtr(result);
525 ptr += 8;
526 }
527
528 return true;
529}
530
531template <StringComparisonMode Mode, typename Char> [[maybe_unused]]
532static int ucstrncmp_sse2(const char16_t *a, const Char *b, size_t l)
533{
534 static_assert(std::is_unsigned_v<Char>);
535
536 // Using the PMOVMSKB instruction, we get two bits for each UTF-16 character
537 // we compare. This lambda helps extract the code unit.
538 static const auto codeUnitAt = [](const auto *n, qptrdiff idx) -> int {
539 constexpr int Stride = 2;
540 // this is the same as:
541 // return n[idx / Stride];
542 // but using pointer arithmetic to avoid the compiler dividing by two
543 // and multiplying by two in the case of char16_t (we know idx is even,
544 // but the compiler does not). This is not UB.
545
546 auto ptr = reinterpret_cast<const uchar *>(n);
547 ptr += idx / (Stride / sizeof(*n));
548 return *reinterpret_cast<decltype(n)>(ptr);
549 };
550 auto difference = [a, b](uint mask, qptrdiff offset) {
551 if (Mode == CompareStringsForEquality)
552 return 1;
553 uint idx = qCountTrailingZeroBits(mask);
554 return codeUnitAt(a + offset, idx) - codeUnitAt(b + offset, idx);
555 };
556
557 static const auto load8Chars = [](const auto *ptr) {
558 if (sizeof(*ptr) == 2)
559 return _mm_loadu_si128(reinterpret_cast<const __m128i *>(ptr));
560 __m128i chunk = _mm_loadl_epi64(reinterpret_cast<const __m128i *>(ptr));
561 return _mm_unpacklo_epi8(chunk, _mm_setzero_si128());
562 };
563 static const auto load4Chars = [](const auto *ptr) {
564 if (sizeof(*ptr) == 2)
565 return _mm_loadl_epi64(reinterpret_cast<const __m128i *>(ptr));
566 __m128i chunk = _mm_cvtsi32_si128(qFromUnaligned<quint32>(ptr));
567 return _mm_unpacklo_epi8(chunk, _mm_setzero_si128());
568 };
569
570 // we're going to read a[0..15] and b[0..15] (32 bytes)
571 auto processChunk16Chars = [a, b](qptrdiff offset) -> uint {
572 if constexpr (UseAvx2) {
573 __m256i a_data = _mm256_loadu_si256(reinterpret_cast<const __m256i *>(a + offset));
574 __m256i b_data;
575 if (sizeof(Char) == 1) {
576 // expand to UTF-16 via zero-extension
577 __m128i chunk = _mm_loadu_si128(reinterpret_cast<const __m128i *>(b + offset));
578 b_data = _mm256_cvtepu8_epi16(chunk);
579 } else {
580 b_data = _mm256_loadu_si256(reinterpret_cast<const __m256i *>(b + offset));
581 }
582 __m256i result = _mm256_cmpeq_epi16(a_data, b_data);
583 return _mm256_movemask_epi8(result);
584 }
585
586 __m128i a_data1 = load8Chars(a + offset);
587 __m128i a_data2 = load8Chars(a + offset + 8);
588 __m128i b_data1, b_data2;
589 if (sizeof(Char) == 1) {
590 // expand to UTF-16 via unpacking
591 __m128i b_data = _mm_loadu_si128(reinterpret_cast<const __m128i *>(b + offset));
592 b_data1 = _mm_unpacklo_epi8(b_data, _mm_setzero_si128());
593 b_data2 = _mm_unpackhi_epi8(b_data, _mm_setzero_si128());
594 } else {
595 b_data1 = load8Chars(b + offset);
596 b_data2 = load8Chars(b + offset + 8);
597 }
598 __m128i result1 = _mm_cmpeq_epi16(a_data1, b_data1);
599 __m128i result2 = _mm_cmpeq_epi16(a_data2, b_data2);
600 return _mm_movemask_epi8(result1) | _mm_movemask_epi8(result2) << 16;
601 };
602
603 if (l >= sizeof(__m256i) / sizeof(char16_t)) {
604 qptrdiff offset = 0;
605 for ( ; l >= offset + sizeof(__m256i) / sizeof(char16_t); offset += sizeof(__m256i) / sizeof(char16_t)) {
606 uint mask = ~processChunk16Chars(offset);
607 if (mask)
608 return difference(mask, offset);
609 }
610
611 // maybe overlap the last 32 bytes
612 if (size_t(offset) < l) {
613 offset = l - sizeof(__m256i) / sizeof(char16_t);
614 uint mask = ~processChunk16Chars(offset);
615 return mask ? difference(mask, offset) : 0;
616 }
617 } else if (l >= 4) {
618 __m128i a_data1, b_data1;
619 __m128i a_data2, b_data2;
620 int width;
621 if (l >= 8) {
622 width = 8;
623 a_data1 = load8Chars(a);
624 b_data1 = load8Chars(b);
625 a_data2 = load8Chars(a + l - width);
626 b_data2 = load8Chars(b + l - width);
627 } else {
628 // we're going to read a[0..3] and b[0..3] (8 bytes)
629 width = 4;
630 a_data1 = load4Chars(a);
631 b_data1 = load4Chars(b);
632 a_data2 = load4Chars(a + l - width);
633 b_data2 = load4Chars(b + l - width);
634 }
635
636 __m128i result = _mm_cmpeq_epi16(a_data1, b_data1);
637 ushort mask = ~_mm_movemask_epi8(result);
638 if (mask)
639 return difference(mask, 0);
640
641 result = _mm_cmpeq_epi16(a_data2, b_data2);
642 mask = ~_mm_movemask_epi8(result);
643 if (mask)
644 return difference(mask, l - width);
645 } else {
646 // reset l
647 l &= 3;
648
649 const auto lambda = [=](size_t i) -> int {
650 return a[i] - b[i];
651 };
652 return UnrollTailLoop<3>::exec(l, 0, lambda, lambda);
653 }
654 return 0;
655}
656#endif
657
658Q_NEVER_INLINE
659qsizetype QtPrivate::qustrlen(const char16_t *str) noexcept
660{
661#if defined(__SSE2__) && !(defined(__SANITIZE_ADDRESS__) || __has_feature(address_sanitizer)) && !(defined(__SANITIZE_THREAD__) || __has_feature(thread_sanitizer))
662 return qustrlen_sse2(str);
663#endif
664
665 if (sizeof(wchar_t) == sizeof(char16_t))
666 return wcslen(reinterpret_cast<const wchar_t *>(str));
667
668 qsizetype result = 0;
669 while (*str++)
670 ++result;
671 return result;
672}
673
674qsizetype QtPrivate::qustrnlen(const char16_t *str, qsizetype maxlen) noexcept
675{
676 return qustrchr({ str, maxlen }, u'\0') - str;
677}
678
679/*!
680 * \internal
681 *
682 * Searches for character \a c in the string \a str and returns a pointer to
683 * it. Unlike strchr() and wcschr() (but like glibc's strchrnul()), if the
684 * character is not found, this function returns a pointer to the end of the
685 * string -- that is, \c{str.end()}.
686 */
688const char16_t *QtPrivate::qustrchr(QStringView str, char16_t c) noexcept
689{
690 const char16_t *n = str.utf16();
691 const char16_t *e = n + str.size();
692
693#ifdef __SSE2__
694 bool loops = true;
695 // Using the PMOVMSKB instruction, we get two bits for each character
696 // we compare.
697 __m128i mch;
698 if constexpr (UseAvx2) {
699 // we're going to read n[0..15] (32 bytes)
700 __m256i mch256 = _mm256_set1_epi32(c | (c << 16));
701 for (const char16_t *next = n + 16; next <= e; n = next, next += 16) {
702 __m256i data = _mm256_loadu_si256(reinterpret_cast<const __m256i *>(n));
703 __m256i result = _mm256_cmpeq_epi16(data, mch256);
704 uint mask = uint(_mm256_movemask_epi8(result));
705 if (mask) {
706 uint idx = qCountTrailingZeroBits(mask);
707 return n + idx / 2;
708 }
709 }
710 loops = false;
711 mch = _mm256_castsi256_si128(mch256);
712 } else {
713 mch = _mm_set1_epi32(c | (c << 16));
714 }
715
716 auto hasMatch = [mch, &n](__m128i data, ushort validityMask) {
717 __m128i result = _mm_cmpeq_epi16(data, mch);
718 uint mask = uint(_mm_movemask_epi8(result));
719 if ((mask & validityMask) == 0)
720 return false;
721 uint idx = qCountTrailingZeroBits(mask);
722 n += idx / 2;
723 return true;
724 };
725
726 // we're going to read n[0..7] (16 bytes)
727 for (const char16_t *next = n + 8; next <= e; n = next, next += 8) {
728 __m128i data = _mm_loadu_si128(reinterpret_cast<const __m128i *>(n));
729 if (hasMatch(data, 0xffff))
730 return n;
731
732 if (!loops) {
733 n += 8;
734 break;
735 }
736 }
737
738# if !defined(__OPTIMIZE_SIZE__)
739 // we're going to read n[0..3] (8 bytes)
740 if (e - n > 3) {
741 __m128i data = _mm_loadl_epi64(reinterpret_cast<const __m128i *>(n));
742 if (hasMatch(data, 0xff))
743 return n;
744
745 n += 4;
746 }
747
748 return UnrollTailLoop<3>::exec(e - n, e,
749 [=](qsizetype i) { return n[i] == c; },
750 [=](qsizetype i) { return n + i; });
751# endif
752#elif defined(__ARM_NEON__)
753 const uint16x8_t vmask = qvsetq_n_u16(1, 1 << 1, 1 << 2, 1 << 3, 1 << 4, 1 << 5, 1 << 6, 1 << 7);
754 const uint16x8_t ch_vec = vdupq_n_u16(c);
755 for (const char16_t *next = n + 8; next <= e; n = next, next += 8) {
756 uint16x8_t data = vld1q_u16(reinterpret_cast<const uint16_t *>(n));
757 uint mask = vaddvq_u16(vandq_u16(vceqq_u16(data, ch_vec), vmask));
758 if (ushort(mask)) {
759 // found a match
760 return n + qCountTrailingZeroBits(mask);
761 }
762 }
763#endif // aarch64
764
765 return std::find(n, e, c);
766}
767
768/*!
769 * \internal
770 *
771 * Searches case-insensitively for character \a c in the string \a str and
772 * returns a pointer to it. Iif the character is not found, this function
773 * returns a pointer to the end of the string -- that is, \c{str.end()}.
774 */
776const char16_t *QtPrivate::qustrcasechr(QStringView str, char16_t c) noexcept
777{
778 const QChar *n = str.begin();
779 const QChar *e = str.end();
780 c = foldCase(c);
781 auto it = std::find_if(n, e, [c](auto ch) { return foldAndCompare(ch, QChar(c)); });
782 return reinterpret_cast<const char16_t *>(it);
783}
784
785// Note: ptr on output may be off by one and point to a preceding US-ASCII
786// character. Usually harmless.
787bool qt_is_ascii(const char *&ptr, const char *end) noexcept
788{
789#if defined(__SSE2__)
790 // Testing for the high bit can be done efficiently with just PMOVMSKB
791 bool loops = true;
792 if constexpr (UseAvx2) {
793 while (ptr + 32 <= end) {
794 __m256i data = _mm256_loadu_si256(reinterpret_cast<const __m256i *>(ptr));
795 quint32 mask = _mm256_movemask_epi8(data);
796 if (mask) {
797 uint idx = qCountTrailingZeroBits(mask);
798 ptr += idx;
799 return false;
800 }
801 ptr += 32;
802 }
803 loops = false;
804 }
805
806 while (ptr + 16 <= end) {
807 __m128i data = _mm_loadu_si128(reinterpret_cast<const __m128i *>(ptr));
808 quint32 mask = _mm_movemask_epi8(data);
809 if (mask) {
810 uint idx = qCountTrailingZeroBits(mask);
811 ptr += idx;
812 return false;
813 }
814 ptr += 16;
815
816 if (!loops)
817 break;
818 }
819 if (ptr + 8 <= end) {
820 __m128i data = _mm_loadl_epi64(reinterpret_cast<const __m128i *>(ptr));
821 quint8 mask = _mm_movemask_epi8(data);
822 if (mask) {
823 uint idx = qCountTrailingZeroBits(mask);
824 ptr += idx;
825 return false;
826 }
827 ptr += 8;
828 }
829#endif
830
831 while (ptr + 4 <= end) {
832 quint32 data = qFromUnaligned<quint32>(ptr);
833 if (data &= 0x80808080U) {
834 uint idx = QSysInfo::ByteOrder == QSysInfo::BigEndian
835 ? qCountLeadingZeroBits(data)
836 : qCountTrailingZeroBits(data);
837 ptr += idx / 8;
838 return false;
839 }
840 ptr += 4;
841 }
842
843 while (ptr != end) {
844 if (quint8(*ptr) & 0x80)
845 return false;
846 ++ptr;
847 }
848 return true;
849}
850
851bool QtPrivate::isAscii(QLatin1StringView s) noexcept
852{
853 const char *ptr = s.begin();
854 const char *end = s.end();
855
856 return qt_is_ascii(ptr, end);
857}
858
859static bool isAscii_helper(const char16_t *&ptr, const char16_t *end)
860{
861#ifdef __SSE2__
862 const char *ptr8 = reinterpret_cast<const char *>(ptr);
863 const char *end8 = reinterpret_cast<const char *>(end);
864 bool ok = simdTestMask(ptr8, end8, 0xff80ff80);
865 ptr = reinterpret_cast<const char16_t *>(ptr8);
866 if (!ok)
867 return false;
868#endif
869
870 while (ptr != end) {
871 if (*ptr & 0xff80)
872 return false;
873 ++ptr;
874 }
875 return true;
876}
877
878bool QtPrivate::isAscii(QStringView s) noexcept
879{
880 const char16_t *ptr = s.utf16();
881 const char16_t *end = ptr + s.size();
882
883 return isAscii_helper(ptr, end);
884}
885
886bool QtPrivate::isLatin1(QStringView s) noexcept
887{
888 const char16_t *ptr = s.utf16();
889 const char16_t *end = ptr + s.size();
890
891#ifdef __SSE2__
892 const char *ptr8 = reinterpret_cast<const char *>(ptr);
893 const char *end8 = reinterpret_cast<const char *>(end);
894 if (!simdTestMask(ptr8, end8, 0xff00ff00))
895 return false;
896 ptr = reinterpret_cast<const char16_t *>(ptr8);
897#endif
898
899 while (ptr != end) {
900 if (*ptr++ > 0xff)
901 return false;
902 }
903 return true;
904}
905
906bool QtPrivate::isValidUtf16(QStringView s) noexcept
907{
908 constexpr char32_t InvalidCodePoint = UINT_MAX;
909
910 QStringIterator i(s);
911 while (i.hasNext()) {
912 const char32_t c = i.next(InvalidCodePoint);
913 if (c == InvalidCodePoint)
914 return false;
915 }
916
917 return true;
918}
919
920// conversion between Latin 1 and UTF-16
921Q_CORE_EXPORT void qt_from_latin1(char16_t *dst, const char *str, size_t size) noexcept
922{
923 /* SIMD:
924 * Unpacking with SSE has been shown to improve performance on recent CPUs
925 * The same method gives no improvement with NEON. On Aarch64, clang will do the vectorization
926 * itself in exactly the same way as one would do it with intrinsics.
927 */
928#if defined(__SSE2__)
929 // we're going to read str[offset..offset+15] (16 bytes)
930 const __m128i nullMask = _mm_setzero_si128();
931 auto processOneChunk = [=](qptrdiff offset) {
932 const __m128i chunk = _mm_loadu_si128((const __m128i*)(str + offset)); // load
933 if constexpr (UseAvx2) {
934 // zero extend to an YMM register
935 const __m256i extended = _mm256_cvtepu8_epi16(chunk);
936
937 // store
938 _mm256_storeu_si256((__m256i*)(dst + offset), extended);
939 } else {
940 // unpack the first 8 bytes, padding with zeros
941 const __m128i firstHalf = _mm_unpacklo_epi8(chunk, nullMask);
942 _mm_storeu_si128((__m128i*)(dst + offset), firstHalf); // store
943
944 // unpack the last 8 bytes, padding with zeros
945 const __m128i secondHalf = _mm_unpackhi_epi8 (chunk, nullMask);
946 _mm_storeu_si128((__m128i*)(dst + offset + 8), secondHalf); // store
947 }
948 };
949
950 const char *e = str + size;
951 if (size >= sizeof(__m128i)) {
952 qptrdiff offset = 0;
953 for ( ; str + offset + sizeof(__m128i) <= e; offset += sizeof(__m128i))
954 processOneChunk(offset);
955 if (str + offset < e)
956 processOneChunk(size - sizeof(__m128i));
957 return;
958 }
959
960# if !defined(__OPTIMIZE_SIZE__)
961 if (size >= 4) {
962 // two overlapped loads & stores, of either 64-bit or of 32-bit
963 if (size >= 8) {
964 const __m128i unpacked1 = mm_load8_zero_extend(str);
965 const __m128i unpacked2 = mm_load8_zero_extend(str + size - 8);
966 _mm_storeu_si128(reinterpret_cast<__m128i *>(dst), unpacked1);
967 _mm_storeu_si128(reinterpret_cast<__m128i *>(dst + size - 8), unpacked2);
968 } else {
969 const __m128i chunk1 = _mm_cvtsi32_si128(qFromUnaligned<quint32>(str));
970 const __m128i chunk2 = _mm_cvtsi32_si128(qFromUnaligned<quint32>(str + size - 4));
971 const __m128i unpacked1 = _mm_unpacklo_epi8(chunk1, nullMask);
972 const __m128i unpacked2 = _mm_unpacklo_epi8(chunk2, nullMask);
973 _mm_storel_epi64(reinterpret_cast<__m128i *>(dst), unpacked1);
974 _mm_storel_epi64(reinterpret_cast<__m128i *>(dst + size - 4), unpacked2);
975 }
976 return;
977 } else {
978 size = size % 4;
979 return UnrollTailLoop<3>::exec(qsizetype(size), [=](qsizetype i) { dst[i] = uchar(str[i]); });
980 }
981# endif
982#endif
983#if defined(__mips_dsp)
984 static_assert(sizeof(qsizetype) == sizeof(int),
985 "oops, the assembler implementation needs to be called in a loop");
986 if (size > 20)
987 qt_fromlatin1_mips_asm_unroll8(dst, str, size);
988 else
989 qt_fromlatin1_mips_asm_unroll4(dst, str, size);
990#else
991 while (size--)
992 *dst++ = (uchar)*str++;
993#endif
994}
995
996static QVarLengthArray<char16_t> qt_from_latin1_to_qvla(QLatin1StringView str)
997{
998 const qsizetype len = str.size();
999 QVarLengthArray<char16_t> arr(len);
1000 qt_from_latin1(arr.data(), str.data(), len);
1001 return arr;
1002}
1003
1004template <bool Checked>
1005static void qt_to_latin1_internal(uchar *dst, const char16_t *src, qsizetype length)
1006{
1007#if defined(__SSE2__)
1008 auto questionMark256 = []() {
1009 if constexpr (UseAvx2)
1010 return _mm256_broadcastw_epi16(_mm_cvtsi32_si128('?'));
1011 else
1012 return 0;
1013 }();
1014 auto outOfRange256 = []() {
1015 if constexpr (UseAvx2)
1016 return _mm256_broadcastw_epi16(_mm_cvtsi32_si128(0x100));
1017 else
1018 return 0;
1019 }();
1020 __m128i questionMark, outOfRange;
1021 if constexpr (UseAvx2) {
1022 questionMark = _mm256_castsi256_si128(questionMark256);
1023 outOfRange = _mm256_castsi256_si128(outOfRange256);
1024 } else {
1025 questionMark = _mm_set1_epi16('?');
1026 outOfRange = _mm_set1_epi16(0x100);
1027 }
1028
1029 auto mergeQuestionMarks = [=](__m128i chunk) {
1030 if (!Checked)
1031 return chunk;
1032
1033 // SSE has no compare instruction for unsigned comparison.
1034 if constexpr (UseSse4_1) {
1035 // We use an unsigned uc = qMin(uc, 0x100) and then compare for equality.
1036 chunk = _mm_min_epu16(chunk, outOfRange);
1037 const __m128i offLimitMask = _mm_cmpeq_epi16(chunk, outOfRange);
1038 chunk = _mm_blendv_epi8(chunk, questionMark, offLimitMask);
1039 return chunk;
1040 }
1041 // The variables must be shiffted + 0x8000 to be compared
1042 const __m128i signedBitOffset = _mm_set1_epi16(short(0x8000));
1043 const __m128i thresholdMask = _mm_set1_epi16(short(0xff + 0x8000));
1044
1045 const __m128i signedChunk = _mm_add_epi16(chunk, signedBitOffset);
1046 const __m128i offLimitMask = _mm_cmpgt_epi16(signedChunk, thresholdMask);
1047
1048 // offLimitQuestionMark contains '?' for each 16 bits that was off-limit
1049 // the 16 bits that were correct contains zeros
1050 const __m128i offLimitQuestionMark = _mm_and_si128(offLimitMask, questionMark);
1051
1052 // correctBytes contains the bytes that were in limit
1053 // the 16 bits that were off limits contains zeros
1054 const __m128i correctBytes = _mm_andnot_si128(offLimitMask, chunk);
1055
1056 // merge offLimitQuestionMark and correctBytes to have the result
1057 chunk = _mm_or_si128(correctBytes, offLimitQuestionMark);
1058
1059 Q_UNUSED(outOfRange);
1060 return chunk;
1061 };
1062
1063 // we're going to read to src[offset..offset+15] (16 bytes)
1064 auto loadChunkAt = [=](qptrdiff offset) {
1065 __m128i chunk1, chunk2;
1066 if constexpr (UseAvx2) {
1067 __m256i chunk = _mm256_loadu_si256(reinterpret_cast<const __m256i *>(src + offset));
1068 if (Checked) {
1069 // See mergeQuestionMarks lambda above for details
1070 chunk = _mm256_min_epu16(chunk, outOfRange256);
1071 const __m256i offLimitMask = _mm256_cmpeq_epi16(chunk, outOfRange256);
1072 chunk = _mm256_blendv_epi8(chunk, questionMark256, offLimitMask);
1073 }
1074
1075 chunk2 = _mm256_extracti128_si256(chunk, 1);
1076 chunk1 = _mm256_castsi256_si128(chunk);
1077 } else {
1078 chunk1 = _mm_loadu_si128((const __m128i*)(src + offset)); // load
1079 chunk1 = mergeQuestionMarks(chunk1);
1080
1081 chunk2 = _mm_loadu_si128((const __m128i*)(src + offset + 8)); // load
1082 chunk2 = mergeQuestionMarks(chunk2);
1083 }
1084
1085 // pack the two vector to 16 x 8bits elements
1086 return _mm_packus_epi16(chunk1, chunk2);
1087 };
1088
1089 if (size_t(length) >= sizeof(__m128i)) {
1090 // because of possible overlapping, we won't process the last chunk in the loop
1091 qptrdiff offset = 0;
1092 for ( ; offset + 2 * sizeof(__m128i) < size_t(length); offset += sizeof(__m128i))
1093 _mm_storeu_si128(reinterpret_cast<__m128i *>(dst + offset), loadChunkAt(offset));
1094
1095 // overlapped conversion of the last full chunk and the tail
1096 __m128i last1 = loadChunkAt(offset);
1097 __m128i last2 = loadChunkAt(length - sizeof(__m128i));
1098 _mm_storeu_si128(reinterpret_cast<__m128i *>(dst + offset), last1);
1099 _mm_storeu_si128(reinterpret_cast<__m128i *>(dst + length - sizeof(__m128i)), last2);
1100 return;
1101 }
1102
1103# if !defined(__OPTIMIZE_SIZE__)
1104 if (length >= 4) {
1105 // this code is fine even for in-place conversion because we load both
1106 // before any store
1107 if (length >= 8) {
1108 __m128i chunk1 = _mm_loadu_si128(reinterpret_cast<const __m128i *>(src));
1109 __m128i chunk2 = _mm_loadu_si128(reinterpret_cast<const __m128i *>(src + length - 8));
1110 chunk1 = mergeQuestionMarks(chunk1);
1111 chunk2 = mergeQuestionMarks(chunk2);
1112
1113 // pack, where the upper half is ignored
1114 const __m128i result1 = _mm_packus_epi16(chunk1, chunk1);
1115 const __m128i result2 = _mm_packus_epi16(chunk2, chunk2);
1116 _mm_storel_epi64(reinterpret_cast<__m128i *>(dst), result1);
1117 _mm_storel_epi64(reinterpret_cast<__m128i *>(dst + length - 8), result2);
1118 } else {
1119 __m128i chunk1 = _mm_loadl_epi64(reinterpret_cast<const __m128i *>(src));
1120 __m128i chunk2 = _mm_loadl_epi64(reinterpret_cast<const __m128i *>(src + length - 4));
1121 chunk1 = mergeQuestionMarks(chunk1);
1122 chunk2 = mergeQuestionMarks(chunk2);
1123
1124 // pack, we'll zero the upper three quarters
1125 const __m128i result1 = _mm_packus_epi16(chunk1, chunk1);
1126 const __m128i result2 = _mm_packus_epi16(chunk2, chunk2);
1127 qToUnaligned(_mm_cvtsi128_si32(result1), dst);
1128 qToUnaligned(_mm_cvtsi128_si32(result2), dst + length - 4);
1129 }
1130 return;
1131 }
1132
1133 length = length % 4;
1134 return UnrollTailLoop<3>::exec(length, [=](qsizetype i) {
1135 if (Checked)
1136 dst[i] = (src[i]>0xff) ? '?' : (uchar) src[i];
1137 else
1138 dst[i] = src[i];
1139 });
1140# else
1141 length = length % 16;
1142# endif // optimize size
1143#elif defined(__ARM_NEON__)
1144 // Refer to the documentation of the SSE2 implementation.
1145 // This uses exactly the same method as for SSE except:
1146 // 1) neon has unsigned comparison
1147 // 2) packing is done to 64 bits (8 x 8bits component).
1148 if (length >= 16) {
1149 const qsizetype chunkCount = length >> 3; // divided by 8
1150 const uint16x8_t questionMark = vdupq_n_u16('?'); // set
1151 const uint16x8_t thresholdMask = vdupq_n_u16(0xff); // set
1152 for (qsizetype i = 0; i < chunkCount; ++i) {
1153 uint16x8_t chunk = vld1q_u16((uint16_t *)src); // load
1154 src += 8;
1155
1156 if (Checked) {
1157 const uint16x8_t offLimitMask = vcgtq_u16(chunk, thresholdMask); // chunk > thresholdMask
1158 const uint16x8_t offLimitQuestionMark = vandq_u16(offLimitMask, questionMark); // offLimitMask & questionMark
1159 const uint16x8_t correctBytes = vbicq_u16(chunk, offLimitMask); // !offLimitMask & chunk
1160 chunk = vorrq_u16(correctBytes, offLimitQuestionMark); // correctBytes | offLimitQuestionMark
1161 }
1162 const uint8x8_t result = vmovn_u16(chunk); // narrowing move->packing
1163 vst1_u8(dst, result); // store
1164 dst += 8;
1165 }
1166 length = length % 8;
1167 }
1168#endif
1169#if defined(__mips_dsp)
1170 static_assert(sizeof(qsizetype) == sizeof(int),
1171 "oops, the assembler implementation needs to be called in a loop");
1172 qt_toLatin1_mips_dsp_asm(dst, src, length);
1173#else
1174 while (length--) {
1175 if (Checked)
1176 *dst++ = (*src>0xff) ? '?' : (uchar) *src;
1177 else
1178 *dst++ = *src;
1179 ++src;
1180 }
1181#endif
1182}
1183
1184void qt_to_latin1(uchar *dst, const char16_t *src, qsizetype length)
1185{
1186 qt_to_latin1_internal<true>(dst, src, length);
1187}
1188
1189void qt_to_latin1_unchecked(uchar *dst, const char16_t *src, qsizetype length)
1190{
1191 qt_to_latin1_internal<false>(dst, src, length);
1192}
1193
1194// Unicode case-insensitive comparison (argument order matches QStringView)
1195Q_NEVER_INLINE static int ucstricmp(qsizetype alen, const char16_t *a, qsizetype blen, const char16_t *b)
1196{
1197 if (a == b)
1198 return qt_lencmp(alen, blen);
1199
1200 qsizetype l = qMin(alen, blen);
1201 qsizetype i;
1202 for (i = 0; i < l; ++i) {
1203// qDebug() << Qt::hex << alast << blast;
1204// qDebug() << Qt::hex << "*a=" << *a << "alast=" << alast << "folded=" << foldCase (*a, alast);
1205// qDebug() << Qt::hex << "*b=" << *b << "blast=" << blast << "folded=" << foldCase (*b, blast);
1206 int diff = foldCase(a + i, a) - foldCase(b + i, b);
1207 if ((diff))
1208 return diff;
1209 }
1210 if (i == alen) {
1211 if (i == blen)
1212 return 0;
1213 return -1;
1214 }
1215 return 1;
1216}
1217
1218// Case-insensitive comparison between a QStringView and a QLatin1StringView
1219// (argument order matches those types)
1220Q_NEVER_INLINE static int ucstricmp(qsizetype alen, const char16_t *a, qsizetype blen, const char *b)
1221{
1222 qsizetype l = qMin(alen, blen);
1223 qsizetype i;
1224 for (i = 0; i < l; ++i) {
1225 int diff = foldCase(a[i]) - foldCase(char16_t{uchar(b[i])});
1226 if ((diff))
1227 return diff;
1228 }
1229 if (i == alen) {
1230 if (i == blen)
1231 return 0;
1232 return -1;
1233 }
1234 return 1;
1235}
1236
1237// Case-insensitive comparison between a Unicode string and a UTF-8 string
1238Q_NEVER_INLINE static int ucstricmp8(const char *utf8, const char *utf8end, const QChar *utf16, const QChar *utf16end)
1239{
1240 auto src1 = reinterpret_cast<const qchar8_t *>(utf8);
1241 auto end1 = reinterpret_cast<const qchar8_t *>(utf8end);
1242 QStringIterator src2(utf16, utf16end);
1243
1244 while (src1 < end1 && src2.hasNext()) {
1245 char32_t uc1 = QChar::toCaseFolded(QUtf8Functions::nextUcs4FromUtf8(src1, end1));
1246 char32_t uc2 = QChar::toCaseFolded(src2.next());
1247 int diff = uc1 - uc2; // can't underflow
1248 if (diff)
1249 return diff;
1250 }
1251
1252 // the shorter string sorts first
1253 return (end1 > src1) - int(src2.hasNext());
1254}
1255
1256#if defined(__mips_dsp)
1257// From qstring_mips_dsp_asm.S
1258extern "C" int qt_ucstrncmp_mips_dsp_asm(const char16_t *a,
1259 const char16_t *b,
1260 unsigned len);
1261#endif
1262
1263// Unicode case-sensitive compare two same-sized strings
1264template <StringComparisonMode Mode>
1265static int ucstrncmp(const char16_t *a, const char16_t *b, size_t l)
1266{
1267 // This function isn't memcmp() because that can return the wrong sorting
1268 // result in little-endian architectures: 0x00ff must sort before 0x0100,
1269 // but the bytes in memory are FF 00 and 00 01.
1270
1271#ifndef __OPTIMIZE_SIZE__
1272# if defined(__mips_dsp)
1273 static_assert(sizeof(uint) == sizeof(size_t));
1274 if (l >= 8) {
1275 return qt_ucstrncmp_mips_dsp_asm(a, b, l);
1276 }
1277# elif defined(__SSE2__)
1278 return ucstrncmp_sse2<Mode>(a, b, l);
1279# elif defined(__ARM_NEON__)
1280 if (l >= 8) {
1281 const char16_t *end = a + l;
1282 const uint16x8_t mask = qvsetq_n_u16( 1, 1 << 1, 1 << 2, 1 << 3, 1 << 4, 1 << 5, 1 << 6, 1 << 7 );
1283 while (end - a > 7) {
1284 uint16x8_t da = vld1q_u16(reinterpret_cast<const uint16_t *>(a));
1285 uint16x8_t db = vld1q_u16(reinterpret_cast<const uint16_t *>(b));
1286
1287 uint8_t r = ~(uint8_t)vaddvq_u16(vandq_u16(vceqq_u16(da, db), mask));
1288 if (r) {
1289 // found a different QChar
1290 if (Mode == CompareStringsForEquality)
1291 return 1;
1292 uint idx = qCountTrailingZeroBits(r);
1293 return a[idx] - b[idx];
1294 }
1295 a += 8;
1296 b += 8;
1297 }
1298 l &= 7;
1299 }
1300 const auto lambda = [=](size_t i) -> int {
1301 return a[i] - b[i];
1302 };
1303 return UnrollTailLoop<7>::exec(l, 0, lambda, lambda);
1304# endif // MIPS DSP or __SSE2__ or __ARM_NEON__
1305#endif // __OPTIMIZE_SIZE__
1306
1307 if (Mode == CompareStringsForEquality || QSysInfo::ByteOrder == QSysInfo::BigEndian)
1308 return memcmp(a, b, l * sizeof(char16_t));
1309
1310 for (size_t i = 0; i < l; ++i) {
1311 if (int diff = a[i] - b[i])
1312 return diff;
1313 }
1314 return 0;
1315}
1316
1317template <StringComparisonMode Mode>
1318static int ucstrncmp(const char16_t *a, const char *b, size_t l)
1319{
1320 const uchar *c = reinterpret_cast<const uchar *>(b);
1321 const char16_t *uc = a;
1322 const char16_t *e = uc + l;
1323
1324#if defined(__SSE2__) && !defined(__OPTIMIZE_SIZE__)
1325 return ucstrncmp_sse2<Mode>(uc, c, l);
1326#endif
1327
1328 while (uc < e) {
1329 int diff = *uc - *c;
1330 if (diff)
1331 return diff;
1332 uc++, c++;
1333 }
1334
1335 return 0;
1336}
1337
1338// Unicode case-sensitive equality
1339template <typename Char2>
1340static bool ucstreq(const char16_t *a, size_t alen, const Char2 *b)
1341{
1342 return ucstrncmp<CompareStringsForEquality>(a, b, alen) == 0;
1343}
1344
1345// Unicode case-sensitive comparison
1346template <typename Char2>
1347static int ucstrcmp(const char16_t *a, size_t alen, const Char2 *b, size_t blen)
1348{
1349 const size_t l = qMin(alen, blen);
1350 int cmp = ucstrncmp<CompareStringsForOrdering>(a, b, l);
1351 return cmp ? cmp : qt_lencmp(alen, blen);
1352}
1353
1355
1356static int latin1nicmp(const char *lhsChar, qsizetype lSize, const char *rhsChar, qsizetype rSize)
1357{
1358 // We're called with QLatin1StringView's .data() and .size():
1359 Q_ASSERT(lSize >= 0 && rSize >= 0);
1360 if (!lSize)
1361 return rSize ? -1 : 0;
1362 if (!rSize)
1363 return 1;
1364 const qsizetype size = std::min(lSize, rSize);
1365
1366 Q_ASSERT(lhsChar && rhsChar); // since both lSize and rSize are positive
1367 for (qsizetype i = 0; i < size; i++) {
1368 if (int res = CaseInsensitiveL1::difference(lhsChar[i], rhsChar[i]))
1369 return res;
1370 }
1371 return qt_lencmp(lSize, rSize);
1372}
1373
1374bool QtPrivate::equalStrings(QStringView lhs, QStringView rhs) noexcept
1375{
1376 Q_ASSERT(lhs.size() == rhs.size());
1377 return ucstreq(lhs.utf16(), lhs.size(), rhs.utf16());
1378}
1379
1380bool QtPrivate::equalStrings(QStringView lhs, QLatin1StringView rhs) noexcept
1381{
1382 Q_ASSERT(lhs.size() == rhs.size());
1383 return ucstreq(lhs.utf16(), lhs.size(), rhs.latin1());
1384}
1385
1386bool QtPrivate::equalStrings(QLatin1StringView lhs, QStringView rhs) noexcept
1387{
1388 return QtPrivate::equalStrings(rhs, lhs);
1389}
1390
1391bool QtPrivate::equalStrings(QLatin1StringView lhs, QLatin1StringView rhs) noexcept
1392{
1393 Q_ASSERT(lhs.size() == rhs.size());
1394 return (!lhs.size() || memcmp(lhs.data(), rhs.data(), lhs.size()) == 0);
1395}
1396
1397bool QtPrivate::equalStrings(QBasicUtf8StringView<false> lhs, QStringView rhs) noexcept
1398{
1399 return QUtf8::compareUtf8(lhs, rhs) == 0;
1400}
1401
1402bool QtPrivate::equalStrings(QStringView lhs, QBasicUtf8StringView<false> rhs) noexcept
1403{
1404 return QtPrivate::equalStrings(rhs, lhs);
1405}
1406
1407bool QtPrivate::equalStrings(QLatin1StringView lhs, QBasicUtf8StringView<false> rhs) noexcept
1408{
1409 return QUtf8::compareUtf8(QByteArrayView(rhs), lhs) == 0;
1410}
1411
1412bool QtPrivate::equalStrings(QBasicUtf8StringView<false> lhs, QLatin1StringView rhs) noexcept
1413{
1414 return QtPrivate::equalStrings(rhs, lhs);
1415}
1416
1417bool QtPrivate::equalStrings(QBasicUtf8StringView<false> lhs, QBasicUtf8StringView<false> rhs) noexcept
1418{
1419#if QT_VERSION >= QT_VERSION_CHECK(7, 0, 0) || defined(QT_BOOTSTRAPPED) || defined(QT_STATIC)
1420 Q_ASSERT(lhs.size() == rhs.size());
1421#else
1422 // operator== didn't enforce size prior to Qt 6.2
1423 if (lhs.size() != rhs.size())
1424 return false;
1425#endif
1426 return (!lhs.size() || memcmp(lhs.data(), rhs.data(), lhs.size()) == 0);
1427}
1428
1429bool QAnyStringView::equal(QAnyStringView lhs, QAnyStringView rhs) noexcept
1430{
1431 if (lhs.size() != rhs.size() && lhs.isUtf8() == rhs.isUtf8())
1432 return false;
1433 return lhs.visit([rhs](auto lhs) {
1434 return rhs.visit([lhs](auto rhs) {
1435 return QtPrivate::equalStrings(lhs, rhs);
1436 });
1437 });
1438}
1439
1440/*!
1441 \relates QStringView
1442 \internal
1443 \since 5.10
1444
1445 Returns an integer that compares to 0 as \a lhs compares to \a rhs.
1446
1447 \include qstring.qdocinc {search-comparison-case-sensitivity} {comparison}
1448
1449 Case-sensitive comparison is based exclusively on the numeric Unicode values
1450 of the characters and is very fast, but is not what a human would expect.
1451 Consider sorting user-visible strings with QString::localeAwareCompare().
1452
1453 \sa {Comparing Strings}
1454*/
1455int QtPrivate::compareStrings(QStringView lhs, QStringView rhs, Qt::CaseSensitivity cs) noexcept
1456{
1457 if (cs == Qt::CaseSensitive)
1458 return ucstrcmp(lhs.utf16(), lhs.size(), rhs.utf16(), rhs.size());
1459 return ucstricmp(lhs.size(), lhs.utf16(), rhs.size(), rhs.utf16());
1460}
1461
1462/*!
1463 \relates QStringView
1464 \internal
1465 \since 5.10
1466 \overload
1467
1468 Returns an integer that compares to 0 as \a lhs compares to \a rhs.
1469
1470 \include qstring.qdocinc {search-comparison-case-sensitivity} {comparison}
1471
1472 Case-sensitive comparison is based exclusively on the numeric Unicode values
1473 of the characters and is very fast, but is not what a human would expect.
1474 Consider sorting user-visible strings with QString::localeAwareCompare().
1475
1476 \sa {Comparing Strings}
1477*/
1478int QtPrivate::compareStrings(QStringView lhs, QLatin1StringView rhs, Qt::CaseSensitivity cs) noexcept
1479{
1480 if (cs == Qt::CaseSensitive)
1481 return ucstrcmp(lhs.utf16(), lhs.size(), rhs.latin1(), rhs.size());
1482 return ucstricmp(lhs.size(), lhs.utf16(), rhs.size(), rhs.latin1());
1483}
1484
1485/*!
1486 \relates QStringView
1487 \internal
1488 \since 6.0
1489 \overload
1490*/
1491int QtPrivate::compareStrings(QStringView lhs, QBasicUtf8StringView<false> rhs, Qt::CaseSensitivity cs) noexcept
1492{
1493 return -compareStrings(rhs, lhs, cs);
1494}
1495
1496/*!
1497 \relates QStringView
1498 \internal
1499 \since 5.10
1500 \overload
1501*/
1502int QtPrivate::compareStrings(QLatin1StringView lhs, QStringView rhs, Qt::CaseSensitivity cs) noexcept
1503{
1504 return -compareStrings(rhs, lhs, cs);
1505}
1506
1507/*!
1508 \relates QStringView
1509 \internal
1510 \since 5.10
1511 \overload
1512
1513 Returns an integer that compares to 0 as \a lhs compares to \a rhs.
1514
1515 \include qstring.qdocinc {search-comparison-case-sensitivity} {comparison}
1516
1517 Case-sensitive comparison is based exclusively on the numeric Latin-1 values
1518 of the characters and is very fast, but is not what a human would expect.
1519 Consider sorting user-visible strings with QString::localeAwareCompare().
1520
1521 \sa {Comparing Strings}
1522*/
1523int QtPrivate::compareStrings(QLatin1StringView lhs, QLatin1StringView rhs, Qt::CaseSensitivity cs) noexcept
1524{
1525 if (lhs.isEmpty())
1526 return qt_lencmp(qsizetype(0), rhs.size());
1527 if (rhs.isEmpty())
1528 return qt_lencmp(lhs.size(), qsizetype(0));
1529 if (cs == Qt::CaseInsensitive)
1530 return latin1nicmp(lhs.data(), lhs.size(), rhs.data(), rhs.size());
1531 const auto l = std::min(lhs.size(), rhs.size());
1532 int r = memcmp(lhs.data(), rhs.data(), l);
1533 return r ? r : qt_lencmp(lhs.size(), rhs.size());
1534}
1535
1536/*!
1537 \relates QStringView
1538 \internal
1539 \since 6.0
1540 \overload
1541*/
1542int QtPrivate::compareStrings(QLatin1StringView lhs, QBasicUtf8StringView<false> rhs, Qt::CaseSensitivity cs) noexcept
1543{
1544 return -QUtf8::compareUtf8(QByteArrayView(rhs), lhs, cs);
1545}
1546
1547/*!
1548 \relates QStringView
1549 \internal
1550 \since 6.0
1551 \overload
1552*/
1553int QtPrivate::compareStrings(QBasicUtf8StringView<false> lhs, QStringView rhs, Qt::CaseSensitivity cs) noexcept
1554{
1555 if (cs == Qt::CaseSensitive)
1556 return QUtf8::compareUtf8(lhs, rhs);
1557 return ucstricmp8(lhs.begin(), lhs.end(), rhs.begin(), rhs.end());
1558}
1559
1560/*!
1561 \relates QStringView
1562 \internal
1563 \since 6.0
1564 \overload
1565*/
1566int QtPrivate::compareStrings(QBasicUtf8StringView<false> lhs, QLatin1StringView rhs, Qt::CaseSensitivity cs) noexcept
1567{
1568 return -compareStrings(rhs, lhs, cs);
1569}
1570
1571/*!
1572 \relates QStringView
1573 \internal
1574 \since 6.0
1575 \overload
1576*/
1577int QtPrivate::compareStrings(QBasicUtf8StringView<false> lhs, QBasicUtf8StringView<false> rhs, Qt::CaseSensitivity cs) noexcept
1578{
1579 return QUtf8::compareUtf8(QByteArrayView(lhs), QByteArrayView(rhs), cs);
1580}
1581
1582int QAnyStringView::compare(QAnyStringView lhs, QAnyStringView rhs, Qt::CaseSensitivity cs) noexcept
1583{
1584 return lhs.visit([rhs, cs](auto lhs) {
1585 return rhs.visit([lhs, cs](auto rhs) {
1586 return QtPrivate::compareStrings(lhs, rhs, cs);
1587 });
1588 });
1589}
1590
1591// ### Qt 7: do not allow anything but ASCII digits
1592// in arg()'s replacements.
1593#if QT_VERSION < QT_VERSION_CHECK(7, 0, 0) && !defined(QT_BOOTSTRAPPED)
1594static bool supportUnicodeDigitValuesInArg()
1595{
1596 static const bool result = []() {
1597 static const char supportUnicodeDigitValuesEnvVar[]
1598 = "QT_USE_UNICODE_DIGIT_VALUES_IN_STRING_ARG";
1599
1600 if (qEnvironmentVariableIsSet(supportUnicodeDigitValuesEnvVar))
1601 return qEnvironmentVariableIntValue(supportUnicodeDigitValuesEnvVar) != 0;
1602
1603#if QT_VERSION < QT_VERSION_CHECK(6, 6, 0) // keep it in sync with the test
1604 return true;
1605#else
1606 return false;
1607#endif
1608 }();
1609
1610 return result;
1611}
1612#endif
1613
1614static int qArgDigitValue(QChar ch) noexcept
1615{
1616#if QT_VERSION < QT_VERSION_CHECK(7, 0, 0) && !defined(QT_BOOTSTRAPPED)
1617 if (supportUnicodeDigitValuesInArg())
1618 return ch.digitValue();
1619#endif
1620 if (ch >= u'0' && ch <= u'9')
1621 return int(ch.unicode() - u'0');
1622 return -1;
1623}
1624
1625#if QT_CONFIG(regularexpression)
1626Q_DECL_COLD_FUNCTION
1627static void qtWarnAboutInvalidRegularExpression(const QRegularExpression &re, const char *cls, const char *method)
1628{
1629 extern void qtWarnAboutInvalidRegularExpression(const QString &pattern, const char *cls, const char *method);
1630 qtWarnAboutInvalidRegularExpression(re.pattern(), cls, method);
1631}
1632#endif
1633
1634/*!
1635 \macro QT_RESTRICTED_CAST_FROM_ASCII
1636 \relates QString
1637
1638 Disables most automatic conversions from source literals and 8-bit data
1639 to unicode QStrings, but allows the use of
1640 the \c{QChar(char)} and \c{QString(const char (&ch)[N]} constructors,
1641 and the \c{QString::operator=(const char (&ch)[N])} assignment operator.
1642 This gives most of the type-safety benefits of \l QT_NO_CAST_FROM_ASCII
1643 but does not require user code to wrap character and string literals
1644 with QLatin1Char, QLatin1StringView or similar.
1645
1646 Using this macro together with source strings outside the 7-bit range,
1647 non-literals, or literals with embedded NUL characters is undefined.
1648
1649 \sa QT_NO_CAST_FROM_ASCII, QT_NO_CAST_TO_ASCII
1650*/
1651
1652/*!
1653 \macro QT_NO_CAST_FROM_ASCII
1654 \relates QString
1655 \relates QChar
1656
1657 Disables automatic conversions from 8-bit strings (\c{char *}) to Unicode
1658 QStrings, as well as from 8-bit \c{char} types (\c{char} and
1659 \c{unsigned char}) to QChar.
1660
1661 \sa QT_NO_CAST_TO_ASCII, QT_RESTRICTED_CAST_FROM_ASCII,
1662 QT_NO_CAST_FROM_BYTEARRAY
1663*/
1664
1665/*!
1666 \macro QT_NO_CAST_TO_ASCII
1667 \relates QString
1668
1669 Disables automatic conversion from QString to 8-bit strings (\c{char *}).
1670
1671 \sa QT_NO_CAST_FROM_ASCII, QT_RESTRICTED_CAST_FROM_ASCII,
1672 QT_NO_CAST_FROM_BYTEARRAY
1673*/
1674
1675/*!
1676 \macro QT_ASCII_CAST_WARNINGS
1677 \internal
1678 \relates QString
1679
1680 This macro can be defined to force a warning whenever a function is
1681 called that automatically converts between unicode and 8-bit encodings.
1682
1683 Note: This only works for compilers that support warnings for
1684 deprecated API.
1685
1686 \sa QT_NO_CAST_TO_ASCII, QT_NO_CAST_FROM_ASCII, QT_RESTRICTED_CAST_FROM_ASCII
1687*/
1688
1689/*!
1690 \class QString
1691 \inmodule QtCore
1692 \reentrant
1693
1694 \brief The QString class provides a Unicode character string.
1695
1696 \ingroup tools
1697 \ingroup shared
1698 \ingroup string-processing
1699
1700 \compares strong
1701 \compareswith strong QChar QLatin1StringView {const char16_t *} \
1702 QStringView QUtf8StringView
1703 \endcompareswith
1704 \compareswith strong QByteArray QByteArrayView {const char *}
1705 When comparing with byte arrays, their content is interpreted as UTF-8.
1706 \endcompareswith
1707
1708 QString stores a string of 16-bit \l{QChar}s, where each QChar
1709 corresponds to one UTF-16 code unit. (Unicode characters
1710 with code values above 65535 are stored using surrogate pairs,
1711 that is, two consecutive \l{QChar}s.)
1712
1713 \l{Unicode} is an international standard that supports most of the
1714 writing systems in use today. It is a superset of US-ASCII (ANSI
1715 X3.4-1986) and Latin-1 (ISO 8859-1), and all the US-ASCII/Latin-1
1716 characters are available at the same code positions.
1717
1718 Behind the scenes, QString uses \l{implicit sharing}
1719 (copy-on-write) to reduce memory usage and to avoid the needless
1720 copying of data. This also helps reduce the inherent overhead of
1721 storing 16-bit characters instead of 8-bit characters.
1722
1723 In addition to QString, Qt also provides the QByteArray class to
1724 store raw bytes and traditional 8-bit '\\0'-terminated strings.
1725 For most purposes, QString is the class you want to use. It is
1726 used throughout the Qt API, and the Unicode support ensures that
1727 your applications are easy to translate if you want to expand
1728 your application's market at some point. Two prominent cases
1729 where QByteArray is appropriate are when you need to store raw
1730 binary data, and when memory conservation is critical (like in
1731 embedded systems).
1732
1733 \section1 Initializing a string
1734
1735 One way to initialize a QString is to pass a \c{const char
1736 *} to its constructor. For example, the following code creates a
1737 QString of size 5 containing the data "Hello":
1738
1739 \snippet qstring/main.cpp 0
1740
1741 QString converts the \c{const char *} data into Unicode using the
1742 fromUtf8() function.
1743
1744 In all of the QString functions that take \c{const char *}
1745 parameters, the \c{const char *} is interpreted as a classic
1746 C-style \c{'\\0'}-terminated string. Except where the function's
1747 name overtly indicates some other encoding, such \c{const char *}
1748 parameters are assumed to be encoded in UTF-8.
1749
1750 Since Qt 6.4, it is also possible to initialize QStrings using
1751 the \l {Qt::Literals::StringLiterals::operator""_s()} and
1752 \l {Qt::Literals::StringLiterals::operator""_L1()} literal
1753 operators. In many cases, using the literals results in
1754 \l{More efficient string construction}{more efficient string construction}.
1755
1756
1757 You can also provide string data as an array of \l{QChar}s:
1758
1759 \snippet qstring/main.cpp 1
1760
1761 QString makes a deep copy of the QChar data, so you can modify it
1762 later without experiencing side effects. You can avoid taking a
1763 deep copy of the character data by using QStringView or
1764 QString::fromRawData() instead.
1765
1766 Another approach is to set the size of the string using resize()
1767 and to initialize the data character per character. QString uses
1768 0-based indexes, just like C++ arrays. To access the character at
1769 a particular index position, you can use \l operator[](). On
1770 non-\c{const} strings, \l operator[]() returns a reference to a
1771 character that can be used on the left side of an assignment. For
1772 example:
1773
1774 \snippet qstring/main.cpp 2
1775
1776 For read-only access, an alternative syntax is to use the at()
1777 function:
1778
1779 \snippet qstring/main.cpp 3
1780
1781 The at() function can be faster than \l operator[]() because it
1782 never causes a \l{deep copy} to occur. Alternatively, use the
1783 first(), last(), or sliced() functions to extract several characters
1784 at a time.
1785
1786 A QString can embed '\\0' characters (QChar::Null). The size()
1787 function always returns the size of the whole string, including
1788 embedded '\\0' characters.
1789
1790 After a call to the resize() function, newly allocated characters
1791 have undefined values. To set all the characters in the string to
1792 a particular value, use the fill() function.
1793
1794 QString provides dozens of overloads designed to simplify string
1795 usage. For example, if you want to compare a QString with a string
1796 literal, you can write code like this and it will work as expected:
1797
1798 \snippet qstring/main.cpp 4
1799
1800 You can also pass string literals to functions that take QStrings
1801 as arguments, invoking the QString(const char *)
1802 constructor. Similarly, you can pass a QString to a function that
1803 takes a \c{const char *} argument using the \l qPrintable() macro,
1804 which returns the given QString as a \c{const char *}. This is
1805 equivalent to calling toLocal8Bit().\l{QByteArray::}{constData()}
1806 on the QString.
1807
1808 \section1 Manipulating string data
1809
1810 QString provides the following basic functions for modifying the
1811 character data: append(), prepend(), insert(), replace(), and
1812 remove(). For example:
1813
1814 \snippet qstring/main.cpp 5
1815
1816 In the above example, the replace() function's first two arguments are the
1817 position from which to start replacing and the number of characters that
1818 should be replaced.
1819
1820 When data-modifying functions increase the size of the string,
1821 QString may reallocate the memory in which it holds its data. When
1822 this happens, QString expands by more than it immediately needs so as
1823 to have space for further expansion without reallocation until the size
1824 of the string has significantly increased.
1825
1826 The insert(), remove(), and, when replacing a sub-string with one of
1827 different size, replace() functions can be slow (\l{linear time}) for
1828 large strings because they require moving many characters in the string
1829 by at least one position in memory.
1830
1831 If you are building a QString gradually and know in advance
1832 approximately how many characters the QString will contain, you
1833 can call reserve(), asking QString to preallocate a certain amount
1834 of memory. You can also call capacity() to find out how much
1835 memory the QString actually has allocated.
1836
1837 QString provides \l{STL-style iterators} (QString::const_iterator and
1838 QString::iterator). In practice, iterators are handy when working with
1839 generic algorithms provided by the C++ standard library.
1840
1841 \note Iterators over a QString, and references to individual characters
1842 within one, cannot be relied on to remain valid when any non-\c{const}
1843 method of the QString is called. Accessing such an iterator or reference
1844 after the call to a non-\c{const} method leads to undefined behavior. When
1845 stability for iterator-like functionality is required, you should use
1846 indexes instead of iterators, as they are not tied to QString's internal
1847 state and thus do not get invalidated.
1848
1849 \note Due to \l{implicit sharing}, the first non-\c{const} operator or
1850 function used on a given QString may cause it to internally perform a deep
1851 copy of its data. This invalidates all iterators over the string and
1852 references to individual characters within it. Do not call non-const
1853 functions while keeping iterators. Accessing an iterator or reference
1854 after it has been invalidated leads to undefined behavior. See the
1855 \l{Implicit sharing iterator problem} section for more information.
1856
1857 A frequent requirement is to remove or simplify the spacing between
1858 visible characters in a string. The characters that make up that spacing
1859 are those for which \l {QChar::}{isSpace()} returns \c true, such as
1860 the simple space \c{' '}, the horizontal tab \c{'\\t'} and the newline \c{'\\n'}.
1861 To obtain a copy of a string leaving out any spacing from its start and end,
1862 use \l trimmed(). To also replace each sequence of spacing characters within
1863 the string with a simple space, \c{' '}, use \l simplified().
1864
1865 If you want to find all occurrences of a particular character or
1866 substring in a QString, use the indexOf() or lastIndexOf()
1867 functions.The former searches forward, the latter searches backward.
1868 Either can be told an index position from which to start their search.
1869 Each returns the index position of the character or substring if they
1870 find it; otherwise, they return -1. For example, here is a typical loop
1871 that finds all occurrences of a particular substring:
1872
1873 \snippet qstring/main.cpp 6
1874
1875 QString provides many functions for converting numbers into
1876 strings and strings into numbers. See the arg() functions, the
1877 setNum() functions, the number() static functions, and the
1878 toInt(), toDouble(), and similar functions.
1879
1880 To get an uppercase or lowercase version of a string, use toUpper() or
1881 toLower().
1882
1883 Lists of strings are handled by the QStringList class. You can
1884 split a string into a list of strings using the split() function,
1885 and join a list of strings into a single string with an optional
1886 separator using QStringList::join(). You can obtain a filtered list
1887 from a string list by selecting the entries in it that contain a
1888 particular substring or match a particular QRegularExpression.
1889 See QStringList::filter() for details.
1890
1891 \section1 Querying string data
1892
1893 To see if a QString starts or ends with a particular substring, use
1894 startsWith() or endsWith(). To check whether a QString contains a
1895 specific character or substring, use the contains() function. To
1896 find out how many times a particular character or substring occurs
1897 in a string, use count().
1898
1899 To obtain a pointer to the actual character data, call data() or
1900 constData(). These functions return a pointer to the beginning of
1901 the QChar data. The pointer is guaranteed to remain valid until a
1902 non-\c{const} function is called on the QString.
1903
1904 \section2 Comparing strings
1905
1906 QStrings can be compared using overloaded operators such as \l
1907 operator<(), \l operator<=(), \l operator==(), \l operator>=(),
1908 and so on. The comparison is based exclusively on the lexicographical
1909 order of the two strings, seen as sequences of UTF-16 code units.
1910 It is very fast but is not what a human would expect; the
1911 QString::localeAwareCompare() function is usually a better choice for
1912 sorting user-interface strings, when such a comparison is available.
1913
1914 When Qt is linked with the ICU library (which it usually is), its
1915 locale-aware sorting is used. Otherwise, platform-specific solutions
1916 are used:
1917 \list
1918 \li On Windows, localeAwareCompare() uses the current user locale,
1919 as set in the \uicontrol{regional} and \uicontrol{language}
1920 options portion of \uicontrol{Control Panel}.
1921 \li On \macos and iOS, \l localeAwareCompare() compares according
1922 to the \uicontrol{Order for sorted lists} setting in the
1923 \uicontrol{International preferences} panel.
1924 \li On other Unix-like systems, the comparison falls back to the
1925 system library's \c strcoll().
1926 \endlist
1927
1928 \section1 Converting between encoded string data and QString
1929
1930 QString provides the following functions that return a
1931 \c{const char *} version of the string as QByteArray: toUtf8(),
1932 toLatin1(), and toLocal8Bit().
1933
1934 \list
1935 \li toLatin1() returns a Latin-1 (ISO 8859-1) encoded 8-bit string.
1936 \li toUtf8() returns a UTF-8 encoded 8-bit string. UTF-8 is a
1937 superset of US-ASCII (ANSI X3.4-1986) that supports the entire
1938 Unicode character set through multibyte sequences.
1939 \li toLocal8Bit() returns an 8-bit string using the system's local
1940 encoding. This is the same as toUtf8() on Unix systems.
1941 \endlist
1942
1943 To convert from one of these encodings, QString provides
1944 fromLatin1(), fromUtf8(), and fromLocal8Bit(). Other
1945 encodings are supported through the QStringEncoder and QStringDecoder
1946 classes.
1947
1948 As mentioned above, QString provides a lot of functions and
1949 operators that make it easy to interoperate with \c{const char *}
1950 strings. But this functionality is a double-edged sword: It makes
1951 QString more convenient to use if all strings are US-ASCII or
1952 Latin-1, but there is always the risk that an implicit conversion
1953 from or to \c{const char *} is done using the wrong 8-bit
1954 encoding. To minimize these risks, you can turn off these implicit
1955 conversions by defining some of the following preprocessor symbols:
1956
1957 \list
1958 \li \l QT_NO_CAST_FROM_ASCII disables automatic conversions from
1959 C string literals and pointers to Unicode.
1960 \li \l QT_RESTRICTED_CAST_FROM_ASCII allows automatic conversions
1961 from C characters and character arrays but disables automatic
1962 conversions from character pointers to Unicode.
1963 \li \l QT_NO_CAST_TO_ASCII disables automatic conversion from QString
1964 to C strings.
1965 \endlist
1966
1967 You then need to explicitly call fromUtf8(), fromLatin1(),
1968 or fromLocal8Bit() to construct a QString from an
1969 8-bit string, or use the lightweight QLatin1StringView class. For
1970 example:
1971
1972 \snippet code/src_corelib_text_qstring.cpp 1
1973
1974 Similarly, you must call toLatin1(), toUtf8(), or
1975 toLocal8Bit() explicitly to convert the QString to an 8-bit
1976 string.
1977
1978 \table 100 %
1979 \header
1980 \li Note for C Programmers
1981
1982 \row
1983 \li
1984 Due to C++'s type system and the fact that QString is
1985 \l{implicitly shared}, QStrings may be treated like \c{int}s or
1986 other basic types. For example:
1987
1988 \snippet qstring/main.cpp 7
1989
1990 The \c result variable is a normal variable allocated on the
1991 stack. When \c return is called, and because we're returning by
1992 value, the copy constructor is called and a copy of the string is
1993 returned. No actual copying takes place thanks to the implicit
1994 sharing.
1995
1996 \endtable
1997
1998 \section1 Distinction between null and empty strings
1999
2000 For historical reasons, QString distinguishes between null
2001 and empty strings. A \e null string is a string that is
2002 initialized using QString's default constructor or by passing
2003 \nullptr to the constructor. An \e empty string is any
2004 string with size 0. A null string is always empty, but an empty
2005 string isn't necessarily null:
2006
2007 \snippet qstring/main.cpp 8
2008
2009 All functions except isNull() treat null strings the same as empty
2010 strings. For example, toUtf8().\l{QByteArray::}{constData()} returns a valid pointer
2011 (not \nullptr) to a '\\0' character for a null string. We
2012 recommend that you always use the isEmpty() function and avoid isNull().
2013
2014 \section1 Number formats
2015
2016 When a QString::arg() \c{'%'} format specifier includes the \c{'L'} locale
2017 qualifier, and the base is ten (its default), the default locale is
2018 used. This can be set using \l{QLocale::setDefault()}. For more refined
2019 control of localized string representations of numbers, see
2020 QLocale::toString(). All other number formatting done by QString follows the
2021 C locale's representation of numbers.
2022
2023 When QString::arg() applies left-padding to numbers, the fill character
2024 \c{'0'} is treated specially. If the number is negative, its minus sign
2025 appears before the zero-padding. If the field is localized, the
2026 locale-appropriate zero character is used in place of \c{'0'}. For
2027 floating-point numbers, this special treatment only applies if the number is
2028 finite.
2029
2030 \section2 Floating-point formats
2031
2032 In member functions (for example, arg() and number()) that format floating-point
2033 numbers (\c float or \c double) as strings, the representation used can be
2034 controlled by a choice of \e format and \e precision, whose meanings are as
2035 for \l {QLocale::toString(double, char, int)}.
2036
2037 If the selected \e format includes an exponent, localized forms follow the
2038 locale's convention on digits in the exponent. For non-localized formatting,
2039 the exponent shows its sign and includes at least two digits, left-padding
2040 with zero if needed.
2041
2042 \section1 More efficient string construction
2043
2044 Many strings are known at compile time. The QString constructor from
2045 C++ string literals will copy the contents of the string,
2046 treating the contents as UTF-8. This requires memory allocation and
2047 re-encoding string data, operations that will happen at runtime.
2048 If the string data is known at compile time, you can use the QStringLiteral
2049 macro or similarly \c{operator""_s} to create QString's payload at compile
2050 time instead.
2051
2052 Using the QString \c{'+'} operator, it is easy to construct a
2053 complex string from multiple substrings. You will often write code
2054 like this:
2055
2056 \snippet qstring/stringbuilder.cpp 0
2057
2058 There is nothing wrong with either of these string constructions,
2059 but there are a few hidden inefficiencies:
2060
2061 First, repeated use of the \c{'+'} operator may lead to
2062 multiple memory allocations. When concatenating \e{n} substrings,
2063 where \e{n > 2}, there can be as many as \e{n - 1} calls to the
2064 memory allocator.
2065
2066 These allocations can be optimized by an internal class
2067 \c{QStringBuilder}. This class is marked
2068 internal and does not appear in the documentation, because you
2069 aren't meant to instantiate it in your code. Its use will be
2070 automatic, as described below.
2071
2072 \c{QStringBuilder} uses expression templates and reimplements the
2073 \c{'%'} operator so that when you use \c{'%'} for string
2074 concatenation instead of \c{'+'}, multiple substring
2075 concatenations will be postponed until the final result is about
2076 to be assigned to a QString. At this point, the amount of memory
2077 required for the final result is known. The memory allocator is
2078 then called \e{once} to get the required space, and the substrings
2079 are copied into it one by one.
2080
2081 Additional efficiency is gained by inlining and reducing reference
2082 counting (the QString created from a \c{QStringBuilder}
2083 has a ref count of 1, whereas QString::append() needs an extra
2084 test).
2085
2086 There are two ways you can access this improved method of string
2087 construction. The straightforward way is to include
2088 \c{QStringBuilder} wherever you want to use it and use the
2089 \c{'%'} operator instead of \c{'+'} when concatenating strings:
2090
2091 \snippet qstring/stringbuilder.cpp 5
2092
2093 A more global approach, which is more convenient but not entirely
2094 source-compatible, is to define \c QT_USE_QSTRINGBUILDER (by adding
2095 it to the compiler flags) at build time. This will make concatenating
2096 strings with \c{'+'} work the same way as \c{QStringBuilder's} \c{'%'}.
2097
2098 \note Using automatic type deduction (for example, by using the \c
2099 auto keyword) with the result of string concatenation when QStringBuilder
2100 is enabled will show that the concatenation is indeed an object of a
2101 QStringBuilder specialization:
2102
2103 \snippet qstring/stringbuilder.cpp 6
2104
2105 This does not cause any harm, as QStringBuilder will implicitly convert to
2106 QString when required. If this is undesirable, then one should specify
2107 the necessary types instead of having the compiler deduce them:
2108
2109 \snippet qstring/stringbuilder.cpp 7
2110
2111 \section1 Maximum size and out-of-memory conditions
2112
2113 The maximum size of QString depends on the architecture. Most 64-bit
2114 systems can allocate more than 2 GB of memory, with a typical limit
2115 of 2^63 bytes. The actual value also depends on the overhead required for
2116 managing the data block. As a result, you can expect a maximum size
2117 of 2 GB minus overhead on 32-bit platforms and 2^63 bytes minus overhead
2118 on 64-bit platforms. The number of elements that can be stored in a
2119 QString is this maximum size divided by the size of QChar.
2120
2121 When memory allocation fails, QString throws a \c std::bad_alloc
2122 exception if the application was compiled with exception support.
2123 Out-of-memory conditions in Qt containers are the only cases where Qt
2124 will throw exceptions. If exceptions are disabled, then running out of
2125 memory is undefined behavior.
2126
2127 \note Target operating systems may impose limits on how much memory an
2128 application can allocate, in total, or on the size of individual allocations.
2129 This may further restrict the size of string a QString can hold.
2130 Mitigating or controlling the behavior these limits cause is beyond the
2131 scope of the Qt API.
2132
2133 \sa {Which string class to use?}, fromRawData(), QChar, QStringView,
2134 QLatin1StringView, QByteArray
2135*/
2136
2137/*! \typedef QString::ConstIterator
2138
2139 Qt-style synonym for QString::const_iterator.
2140*/
2141
2142/*! \typedef QString::Iterator
2143
2144 Qt-style synonym for QString::iterator.
2145*/
2146
2147/*! \typedef QString::const_iterator
2148
2149 \sa QString::iterator
2150*/
2151
2152/*! \typedef QString::iterator
2153
2154 \sa QString::const_iterator
2155*/
2156
2157/*! \typedef QString::const_reverse_iterator
2158 \since 5.6
2159
2160 \sa QString::reverse_iterator, QString::const_iterator
2161*/
2162
2163/*! \typedef QString::reverse_iterator
2164 \since 5.6
2165
2166 \sa QString::const_reverse_iterator, QString::iterator
2167*/
2168
2169/*!
2170 \typedef QString::size_type
2171*/
2172
2173/*!
2174 \typedef QString::difference_type
2175*/
2176
2177/*!
2178 \typedef QString::const_reference
2179*/
2180/*!
2181 \typedef QString::reference
2182*/
2183
2184/*!
2185 \typedef QString::const_pointer
2186
2187 The QString::const_pointer typedef provides an STL-style
2188 const pointer to a QString element (QChar).
2189*/
2190/*!
2191 \typedef QString::pointer
2192
2193 The QString::pointer typedef provides an STL-style
2194 pointer to a QString element (QChar).
2195*/
2196
2197/*!
2198 \typedef QString::value_type
2199*/
2200
2201/*! \fn QString::iterator QString::begin()
2202
2203 Returns an \l{STL-style iterators}{STL-style iterator} pointing to the
2204 first character in the string.
2205
2206//! [iterator-invalidation-func-desc]
2207 \warning The returned iterator is invalidated on detachment or when the
2208 QString is modified.
2209//! [iterator-invalidation-func-desc]
2210
2211 \sa constBegin(), end()
2212*/
2213
2214/*! \fn QString::const_iterator QString::begin() const
2215
2216 \overload begin()
2217*/
2218
2219/*! \fn QString::const_iterator QString::cbegin() const
2220 \since 5.0
2221
2222 Returns a const \l{STL-style iterators}{STL-style iterator} pointing to the
2223 first character in the string.
2224
2225 \include qstring.cpp iterator-invalidation-func-desc
2226
2227 \sa begin(), cend()
2228*/
2229
2230/*! \fn QString::const_iterator QString::constBegin() const
2231
2232 Returns a const \l{STL-style iterators}{STL-style iterator} pointing to the
2233 first character in the string.
2234
2235 \include qstring.cpp iterator-invalidation-func-desc
2236
2237 \sa begin(), constEnd()
2238*/
2239
2240/*! \fn QString::iterator QString::end()
2241
2242 Returns an \l{STL-style iterators}{STL-style iterator} pointing just after
2243 the last character in the string.
2244
2245 \include qstring.cpp iterator-invalidation-func-desc
2246
2247 \sa begin(), constEnd()
2248*/
2249
2250/*! \fn QString::const_iterator QString::end() const
2251
2252 \overload end()
2253*/
2254
2255/*! \fn QString::const_iterator QString::cend() const
2256 \since 5.0
2257
2258 Returns a const \l{STL-style iterators}{STL-style iterator} pointing just
2259 after the last character in the string.
2260
2261 \include qstring.cpp iterator-invalidation-func-desc
2262
2263 \sa cbegin(), end()
2264*/
2265
2266/*! \fn QString::const_iterator QString::constEnd() const
2267
2268 Returns a const \l{STL-style iterators}{STL-style iterator} pointing just
2269 after the last character in the string.
2270
2271 \include qstring.cpp iterator-invalidation-func-desc
2272
2273 \sa constBegin(), end()
2274*/
2275
2276/*! \fn QString::reverse_iterator QString::rbegin()
2277 \since 5.6
2278
2279 Returns a \l{STL-style iterators}{STL-style} reverse iterator pointing to
2280 the first character in the string, in reverse order.
2281
2282 \include qstring.cpp iterator-invalidation-func-desc
2283
2284 \sa begin(), crbegin(), rend()
2285*/
2286
2287/*! \fn QString::const_reverse_iterator QString::rbegin() const
2288 \since 5.6
2289 \overload
2290*/
2291
2292/*! \fn QString::const_reverse_iterator QString::crbegin() const
2293 \since 5.6
2294
2295 Returns a const \l{STL-style iterators}{STL-style} reverse iterator
2296 pointing to the first character in the string, in reverse order.
2297
2298 \include qstring.cpp iterator-invalidation-func-desc
2299
2300 \sa begin(), rbegin(), rend()
2301*/
2302
2303/*! \fn QString::reverse_iterator QString::rend()
2304 \since 5.6
2305
2306 Returns a \l{STL-style iterators}{STL-style} reverse iterator pointing just
2307 after the last character in the string, in reverse order.
2308
2309 \include qstring.cpp iterator-invalidation-func-desc
2310
2311 \sa end(), crend(), rbegin()
2312*/
2313
2314/*! \fn QString::const_reverse_iterator QString::rend() const
2315 \since 5.6
2316 \overload
2317*/
2318
2319/*! \fn QString::const_reverse_iterator QString::crend() const
2320 \since 5.6
2321
2322 Returns a const \l{STL-style iterators}{STL-style} reverse iterator
2323 pointing just after the last character in the string, in reverse order.
2324
2325 \include qstring.cpp iterator-invalidation-func-desc
2326
2327 \sa end(), rend(), rbegin()
2328*/
2329
2330/*!
2331 \fn QString::QString()
2332
2333 Constructs a null string. Null strings are also considered empty.
2334
2335 \sa isEmpty(), isNull(), {Distinction Between Null and Empty Strings}
2336*/
2337
2338/*!
2339 \fn QString::QString(QString &&other)
2340
2341 Move-constructs a QString instance, making it point at the same
2342 object that \a other was pointing to.
2343
2344 \since 5.2
2345*/
2346
2347/*! \fn QString::QString(const char *str)
2348
2349 Constructs a string initialized with the 8-bit string \a str. The
2350 given const char pointer is converted to Unicode using the
2351 fromUtf8() function.
2352
2353 You can disable this constructor by defining
2354 \l QT_NO_CAST_FROM_ASCII when you compile your applications. This
2355 can be useful if you want to ensure that all user-visible strings
2356 go through QObject::tr(), for example.
2357
2358 \note Defining \l QT_RESTRICTED_CAST_FROM_ASCII also disables
2359 this constructor, but enables a \c{QString(const char (&ch)[N])}
2360 constructor instead. Using non-literal input, or input with
2361 embedded NUL characters, or non-7-bit characters is undefined
2362 in this case.
2363
2364 \sa fromLatin1(), fromLocal8Bit(), fromUtf8()
2365*/
2366
2367/*! \fn QString::QString(const char8_t *str)
2368
2369 Constructs a string initialized with the UTF-8 string \a str. The
2370 given const char8_t pointer is converted to Unicode using the
2371 fromUtf8() function.
2372
2373 \since 6.1
2374 \sa fromLatin1(), fromLocal8Bit(), fromUtf8()
2375*/
2376
2377/*!
2378 \fn QString::QString(QStringView sv)
2379
2380 Constructs a string initialized with the string view's data.
2381
2382 The QString will be null if and only if \a sv is null.
2383
2384 \since 6.8
2385
2386 \sa fromUtf16()
2387*/
2388
2389/*
2390//! [from-std-string]
2391Returns a copy of the \a str string. The given string is assumed to be
2392encoded in \1, and is converted to QString using the \2 function.
2393//! [from-std-string]
2394*/
2395
2396/*! \fn QString QString::fromStdString(const std::string &str)
2397
2398 \include qstring.cpp {from-std-string} {UTF-8} {fromUtf8()}
2399
2400 \sa fromLatin1(), fromLocal8Bit(), fromUtf8(), QByteArray::fromStdString()
2401*/
2402
2403/*! \fn QString QString::fromStdWString(const std::wstring &str)
2404
2405 Returns a copy of the \a str string. The given string is assumed
2406 to be encoded in utf16 if the size of wchar_t is 2 bytes (e.g. on
2407 windows) and ucs4 if the size of wchar_t is 4 bytes (most Unix
2408 systems).
2409
2410 \sa fromUtf16(), fromLatin1(), fromLocal8Bit(), fromUtf8(), fromUcs4(),
2411 fromStdU16String(), fromStdU32String()
2412*/
2413
2414/*! \fn QString QString::fromWCharArray(const wchar_t *string, qsizetype size)
2415 \since 4.2
2416
2417 Reads the first \a size code units of the \c wchar_t array to whose start
2418 \a string points, converting them to Unicode and returning the result as
2419 a QString. The encoding used by \c wchar_t is assumed to be UTF-32 if the
2420 type's size is four bytes or UTF-16 if its size is two bytes.
2421
2422 If \a size is -1 (default), the \a string must be '\\0'-terminated.
2423
2424 \sa fromUtf16(), fromLatin1(), fromLocal8Bit(), fromUtf8(), fromUcs4(),
2425 fromStdWString()
2426*/
2427
2428/*! \fn std::wstring QString::toStdWString() const
2429
2430 Returns a std::wstring object with the data contained in this
2431 QString. The std::wstring is encoded in UTF-16 on platforms where
2432 wchar_t is 2 bytes wide (for example, Windows) and in UTF-32 on platforms
2433 where wchar_t is 4 bytes wide (most Unix systems).
2434
2435 This method is mostly useful to pass a QString to a function
2436 that accepts a std::wstring object.
2437
2438 \sa utf16(), toLatin1(), toUtf8(), toLocal8Bit(), toStdU16String(),
2439 toStdU32String()
2440*/
2441
2442qsizetype QString::toUcs4_helper(const char16_t *uc, qsizetype length, char32_t *out)
2443{
2444 qsizetype count = 0;
2445
2446 QStringIterator i(QStringView(uc, length));
2447 while (i.hasNext())
2448 out[count++] = i.next();
2449
2450 return count;
2451}
2452
2453/*! \fn qsizetype QString::toWCharArray(wchar_t *array) const
2454 \since 4.2
2455
2456 Fills the \a array with the data contained in this QString object.
2457 The array is encoded in UTF-16 on platforms where
2458 wchar_t is 2 bytes wide (e.g. windows) and in UTF-32 on platforms
2459 where wchar_t is 4 bytes wide (most Unix systems).
2460
2461 \a array has to be allocated by the caller and contain enough space to
2462 hold the complete string (allocating the array with the same length as the
2463 string is always sufficient).
2464
2465 This function returns the actual length of the string in \a array.
2466
2467 \note This function does not append a null character to the array.
2468
2469 \sa utf16(), toUcs4(), toLatin1(), toUtf8(), toLocal8Bit(), toStdWString(),
2470 QStringView::toWCharArray()
2471*/
2472
2473/*! \fn QString::QString(const QString &other)
2474
2475 Constructs a copy of \a other.
2476
2477 This operation takes \l{constant time}, because QString is
2478 \l{implicitly shared}. This makes returning a QString from a
2479 function very fast. If a shared instance is modified, it will be
2480 copied (copy-on-write), and that takes \l{linear time}.
2481
2482 \sa operator=()
2483*/
2484
2485/*!
2486 Constructs a string initialized with the first \a size characters
2487 of the QChar array \a unicode.
2488
2489 If \a unicode is 0, a null string is constructed.
2490
2491 If \a size is negative, \a unicode is assumed to point to a '\\0'-terminated
2492 array and its length is determined dynamically. The terminating
2493 null character is not considered part of the string.
2494
2495 QString makes a deep copy of the string data. The unicode data is copied as
2496 is and the Byte Order Mark is preserved if present.
2497
2498 \sa fromRawData()
2499*/
2500QString::QString(const QChar *unicode, qsizetype size)
2501{
2502 if (!unicode) {
2503 d.clear();
2504 } else {
2505 if (size < 0)
2506 size = QtPrivate::qustrlen(reinterpret_cast<const char16_t *>(unicode));
2507 if (!size) {
2508 d = DataPointer::fromRawData(&_empty, 0);
2509 } else {
2510 d = DataPointer(size, size);
2511 Q_CHECK_PTR(d.data());
2512 memcpy(d.data(), unicode, size * sizeof(QChar));
2513 d.data()[size] = '\0';
2514 }
2515 }
2516}
2517
2518/*!
2519 Constructs a string of the given \a size with every character set
2520 to \a ch.
2521
2522 \sa fill()
2523*/
2524QString::QString(qsizetype size, QChar ch)
2525{
2526 if (size <= 0) {
2527 d = DataPointer::fromRawData(&_empty, 0);
2528 } else {
2529 d = DataPointer(size, size);
2530 Q_CHECK_PTR(d.data());
2531 d.data()[size] = '\0';
2532 char16_t *b = d.data();
2533 char16_t *e = d.data() + size;
2534 const char16_t value = ch.unicode();
2535 std::fill(b, e, value);
2536 }
2537}
2538
2539/*! \fn QString::QString(qsizetype size, Qt::Initialization)
2540 \internal
2541
2542 Constructs a string of the given \a size without initializing the
2543 characters. This is only used in \c QStringBuilder::toString().
2544*/
2545QString::QString(qsizetype size, Qt::Initialization)
2546{
2547 if (size <= 0) {
2548 d = DataPointer::fromRawData(&_empty, 0);
2549 } else {
2550 d = DataPointer(size, size);
2551 Q_CHECK_PTR(d.data());
2552 d.data()[size] = '\0';
2553 }
2554}
2555
2556/*! \fn QString::QString(QLatin1StringView str)
2557
2558 Constructs a copy of the Latin-1 string viewed by \a str.
2559
2560 \sa fromLatin1()
2561*/
2562
2563/*!
2564 Constructs a string of size 1 containing the character \a ch.
2565*/
2566QString::QString(QChar ch)
2567{
2568 d = DataPointer(1, 1);
2569 Q_CHECK_PTR(d.data());
2570 d.data()[0] = ch.unicode();
2571 d.data()[1] = '\0';
2572}
2573
2574/*! \fn QString::QString(const QByteArray &ba)
2575
2576 Constructs a string initialized with the byte array \a ba. The
2577 given byte array is converted to Unicode using fromUtf8().
2578
2579 You can disable this constructor by defining
2580 \l QT_NO_CAST_FROM_ASCII when you compile your applications. This
2581 can be useful if you want to ensure that all user-visible strings
2582 go through QObject::tr(), for example.
2583
2584 \note Any null ('\\0') bytes in the byte array will be included in this
2585 string, converted to Unicode null characters (U+0000). This behavior is
2586 different from Qt 5.x.
2587
2588 \sa fromLatin1(), fromLocal8Bit(), fromUtf8()
2589*/
2590
2591/*! \fn QString::QString(const Null &)
2592 \internal
2593*/
2594
2595/*! \fn QString::QString(QStringPrivate)
2596 \internal
2597*/
2598
2599/*! \fn QString &QString::operator=(const QString::Null &)
2600 \internal
2601*/
2602
2603/*!
2604 \fn QString::~QString()
2605
2606 Destroys the string.
2607*/
2608
2609
2610/*! \fn void QString::swap(QString &other)
2611 \since 4.8
2612 \memberswap{string}
2613*/
2614
2615/*! \fn void QString::detach()
2616
2617 Ensures that this string's data is no longer
2618 \l{Implicit Sharing}{shared} with other instances.
2619*/
2620
2621/*! \fn bool QString::isDetached() const
2622
2623 \internal
2624*/
2625
2626/*! \fn bool QString::isSharedWith(const QString &other) const
2627
2628 \internal
2629*/
2630
2631/*! \fn QString::operator std::u16string_view() const
2632 \target qstring-operator-std-u16string_view
2633 \since 6.7
2634
2635 Converts this QString object to a \c{std::u16string_view} object.
2636*/
2637
2638static bool needsReallocate(const QString &str, qsizetype newSize)
2639{
2640 const auto capacityAtEnd = str.capacity() - str.data_ptr().freeSpaceAtBegin();
2641 return newSize > capacityAtEnd;
2642}
2643
2644/*!
2645 Sets the size of the string to \a size characters.
2646
2647 If \a size is greater than the current size, the string is
2648 extended to make it \a size characters long with the extra
2649 characters added to the end. The new characters are uninitialized.
2650
2651 If \a size is less than the current size, characters beyond position
2652 \a size are excluded from the string.
2653
2654 \note While resize() will grow the capacity if needed, it never shrinks
2655 capacity. To shed excess capacity, use squeeze().
2656
2657 Example:
2658
2659 \snippet qstring/main.cpp 45
2660
2661 If you want to append a certain number of identical characters to
2662 the string, use the \l {QString::}{resize(qsizetype, QChar)} overload.
2663
2664 If you want to expand the string so that it reaches a certain
2665 width and fill the new positions with a particular character, use
2666 the leftJustified() function:
2667
2668 If \a size is negative, it is equivalent to passing zero.
2669
2670 \snippet qstring/main.cpp 47
2671
2672 \sa truncate(), reserve(), squeeze()
2673*/
2674
2675void QString::resize(qsizetype size)
2676{
2677 if (size < 0)
2678 size = 0;
2679
2680 if (d.needsDetach() || needsReallocate(*this, size))
2681 reallocData(size, QArrayData::Grow);
2682 d.size = size;
2683 if (d.allocatedCapacity())
2684 d.data()[size] = u'\0';
2685}
2686
2687/*!
2688 \overload
2689 \since 5.7
2690
2691 Unlike \l {QString::}{resize(qsizetype)}, this overload
2692 initializes the new characters to \a fillChar:
2693
2694 \snippet qstring/main.cpp 46
2695*/
2696
2697void QString::resize(qsizetype newSize, QChar fillChar)
2698{
2699 const qsizetype oldSize = size();
2700 resize(newSize);
2701 const qsizetype difference = size() - oldSize;
2702 if (difference > 0)
2703 std::fill_n(d.data() + oldSize, difference, fillChar.unicode());
2704}
2705
2706
2707/*!
2708 \since 6.8
2709
2710 Sets the size of the string to \a size characters. If the size of
2711 the string grows, the new characters are uninitialized.
2712
2713 The behavior is identical to \c{resize(size)}.
2714
2715 \sa resize()
2716*/
2717
2718void QString::resizeForOverwrite(qsizetype size)
2719{
2720 resize(size);
2721}
2722
2723
2724/*! \fn qsizetype QString::capacity() const
2725
2726 Returns the maximum number of characters that can be stored in
2727 the string without forcing a reallocation.
2728
2729 The sole purpose of this function is to provide a means of fine
2730 tuning QString's memory usage. In general, you will rarely ever
2731 need to call this function. If you want to know how many
2732 characters are in the string, call size().
2733
2734 \note a statically allocated string will report a capacity of 0,
2735 even if it's not empty.
2736
2737 \note The free space position in the allocated memory block is undefined. In
2738 other words, one should not assume that the free memory is always located
2739 after the initialized elements.
2740
2741 \sa reserve(), squeeze()
2742*/
2743
2744/*!
2745 \fn void QString::reserve(qsizetype size)
2746
2747 Ensures the string has space for at least \a size characters.
2748
2749 If you know in advance how large a string will be, you can call this
2750 function to save repeated reallocation while building it.
2751 This can improve performance when building a string incrementally.
2752 A long sequence of operations that add to a string may trigger several
2753 reallocations, the last of which may leave you with significantly more
2754 space than you need. This is less efficient than doing a single
2755 allocation of the right size at the start.
2756
2757 If in doubt about how much space shall be needed, it is usually better to
2758 use an upper bound as \a size, or a high estimate of the most likely size,
2759 if a strict upper bound would be much bigger than this. If \a size is an
2760 underestimate, the string will grow as needed once the reserved size is
2761 exceeded, which may lead to a larger allocation than your best
2762 overestimate would have and will slow the operation that triggers it.
2763
2764 \warning reserve() reserves memory but does not change the size of the
2765 string. Accessing data beyond the end of the string is undefined behavior.
2766 If you need to access memory beyond the current end of the string,
2767 use resize().
2768
2769 This function is useful for code that needs to build up a long
2770 string and wants to avoid repeated reallocation. In this example,
2771 we want to add to the string until some condition is \c true, and
2772 we're fairly sure that size is large enough to make a call to
2773 reserve() worthwhile:
2774
2775 \snippet qstring/main.cpp 44
2776
2777 \sa squeeze(), capacity(), resize()
2778*/
2779
2780/*!
2781 \fn void QString::squeeze()
2782
2783 Releases any memory not required to store the character data.
2784
2785 The sole purpose of this function is to provide a means of fine
2786 tuning QString's memory usage. In general, you will rarely ever
2787 need to call this function.
2788
2789 \sa reserve(), capacity()
2790*/
2791
2792void QString::reallocData(qsizetype alloc, QArrayData::AllocationOption option)
2793{
2794 if (!alloc) {
2795 d = DataPointer::fromRawData(&_empty, 0);
2796 return;
2797 }
2798
2799 // don't use reallocate path when reducing capacity and there's free space
2800 // at the beginning: might shift data pointer outside of allocated space
2801 const bool cannotUseReallocate = d.freeSpaceAtBegin() > 0;
2802
2803 if (d.needsDetach() || cannotUseReallocate) {
2804 DataPointer dd(alloc, qMin(alloc, d.size), option);
2805 Q_CHECK_PTR(dd.data());
2806 if (dd.size > 0)
2807 ::memcpy(dd.data(), d.data(), dd.size * sizeof(QChar));
2808 dd.data()[dd.size] = 0;
2809 d.swap(dd);
2810 } else {
2811 d->reallocate(alloc, option);
2812 }
2813}
2814
2815void QString::reallocGrowData(qsizetype n)
2816{
2817 if (!n) // expected to always allocate
2818 n = 1;
2819
2820 if (d.needsDetach()) {
2821 DataPointer dd(DataPointer::allocateGrow(d, n, QArrayData::GrowsAtEnd));
2822 Q_CHECK_PTR(dd.data());
2823 dd->copyAppend(d.data(), d.data() + d.size);
2824 dd.data()[dd.size] = 0;
2825 d.swap(dd);
2826 } else {
2827 d->reallocate(d.constAllocatedCapacity() + n, QArrayData::Grow);
2828 }
2829}
2830
2831/*! \fn void QString::clear()
2832
2833 Clears the contents of the string and makes it null.
2834
2835 \sa resize(), isNull()
2836*/
2837
2838/*! \fn QString &QString::operator=(const QString &other)
2839
2840 Assigns \a other to this string and returns a reference to this
2841 string.
2842*/
2843
2844QString &QString::operator=(const QString &other) noexcept
2845{
2846 d = other.d;
2847 return *this;
2848}
2849
2850/*!
2851 \fn QString &QString::operator=(QString &&other)
2852
2853 Move-assigns \a other to this QString instance.
2854
2855 \since 5.2
2856*/
2857
2858/*! \fn QString &QString::operator=(QLatin1StringView str)
2859
2860 \overload operator=()
2861
2862 Assigns the Latin-1 string viewed by \a str to this string.
2863*/
2864QString &QString::operator=(QLatin1StringView other)
2865{
2866 const qsizetype capacityAtEnd = capacity() - d.freeSpaceAtBegin();
2867 if (isDetached() && other.size() <= capacityAtEnd) { // assumes d.alloc == 0 -> !isDetached() (sharedNull)
2868 d.size = other.size();
2869 d.data()[other.size()] = 0;
2870 qt_from_latin1(d.data(), other.latin1(), other.size());
2871 } else {
2872 *this = fromLatin1(other.latin1(), other.size());
2873 }
2874 return *this;
2875}
2876
2877/*! \fn QString &QString::operator=(const QByteArray &ba)
2878
2879 \overload operator=()
2880
2881 Assigns \a ba to this string. The byte array is converted to Unicode
2882 using the fromUtf8() function.
2883
2884 You can disable this operator by defining
2885 \l QT_NO_CAST_FROM_ASCII when you compile your applications. This
2886 can be useful if you want to ensure that all user-visible strings
2887 go through QObject::tr(), for example.
2888*/
2889
2890/*! \fn QString &QString::operator=(const char *str)
2891
2892 \overload operator=()
2893
2894 Assigns \a str to this string. The const char pointer is converted
2895 to Unicode using the fromUtf8() function.
2896
2897 You can disable this operator by defining \l QT_NO_CAST_FROM_ASCII
2898 or \l QT_RESTRICTED_CAST_FROM_ASCII when you compile your applications.
2899 This can be useful if you want to ensure that all user-visible strings
2900 go through QObject::tr(), for example.
2901*/
2902
2903/*!
2904 \overload operator=()
2905
2906 Sets the string to contain the single character \a ch.
2907*/
2908QString &QString::operator=(QChar ch)
2909{
2910 return assign(1, ch);
2911}
2912
2913/*!
2914 \fn QString& QString::insert(qsizetype position, const QString &str)
2915
2916 Inserts the string \a str at the given index \a position and
2917 returns a reference to this string.
2918
2919 Example:
2920
2921 \snippet qstring/main.cpp 26
2922
2923//! [string-grow-at-insertion]
2924 This string grows to accommodate the insertion. If \a position is beyond
2925 the end of the string, space characters are appended to the string to reach
2926 this \a position, followed by \a str.
2927//! [string-grow-at-insertion]
2928
2929 \sa append(), prepend(), replace(), remove()
2930*/
2931
2932/*!
2933 \fn QString& QString::insert(qsizetype position, QStringView str)
2934 \since 6.0
2935 \overload insert()
2936
2937 Inserts the string view \a str at the given index \a position and
2938 returns a reference to this string.
2939
2940 \include qstring.cpp string-grow-at-insertion
2941*/
2942
2943
2944/*!
2945 \fn QString& QString::insert(qsizetype position, const char *str)
2946 \since 5.5
2947 \overload insert()
2948
2949 Inserts the C string \a str at the given index \a position and
2950 returns a reference to this string.
2951
2952 \include qstring.cpp string-grow-at-insertion
2953
2954 This function is not available when \l QT_NO_CAST_FROM_ASCII is
2955 defined.
2956*/
2957
2958/*!
2959 \fn QString& QString::insert(qsizetype position, const QByteArray &str)
2960 \since 5.5
2961 \overload insert()
2962
2963 Interprets the contents of \a str as UTF-8, inserts the Unicode string
2964 it encodes at the given index \a position and returns a reference to
2965 this string.
2966
2967 \include qstring.cpp string-grow-at-insertion
2968
2969 This function is not available when \l QT_NO_CAST_FROM_ASCII is
2970 defined.
2971*/
2972
2973/*! \internal
2974 T is a view or a container on/of QChar, char16_t, or char
2975*/
2976template <typename T>
2977static void insert_helper(QString &str, qsizetype i, const T &toInsert)
2978{
2979 auto &str_d = str.data_ptr();
2980 qsizetype difference = 0;
2981 if (Q_UNLIKELY(i > str_d.size))
2982 difference = i - str_d.size;
2983 const qsizetype oldSize = str_d.size;
2984 const qsizetype insert_size = toInsert.size();
2985 const qsizetype newSize = str_d.size + difference + insert_size;
2986 const auto side = i == 0 ? QArrayData::GrowsAtBeginning : QArrayData::GrowsAtEnd;
2987
2988 if (str_d.needsDetach() || needsReallocate(str, newSize)) {
2989 const auto cbegin = str.cbegin();
2990 const auto cend = str.cend();
2991 const auto insert_start = difference == 0 ? std::next(cbegin, i) : cend;
2992 QString other;
2993 // Using detachAndGrow() so that prepend optimization works and QStringBuilder
2994 // unittests pass
2995 other.data_ptr().detachAndGrow(side, newSize, nullptr, nullptr);
2996 other.append(QStringView(cbegin, insert_start));
2997 other.resize(i, u' ');
2998 other.append(toInsert);
2999 other.append(QStringView(insert_start, cend));
3000 str.swap(other);
3001 return;
3002 }
3003
3004 str_d.detachAndGrow(side, difference + insert_size, nullptr, nullptr);
3005 Q_CHECK_PTR(str_d.data());
3006 str.resize(newSize);
3007
3008 auto begin = str_d.begin();
3009 auto old_end = std::next(begin, oldSize);
3010 std::fill_n(old_end, difference, u' ');
3011 auto insert_start = std::next(begin, i);
3012 if (difference == 0)
3013 std::move_backward(insert_start, old_end, str_d.end());
3014
3015 using Char = std::remove_cv_t<typename T::value_type>;
3016 if constexpr(std::is_same_v<Char, QChar>)
3017 std::copy_n(reinterpret_cast<const char16_t *>(toInsert.data()), insert_size, insert_start);
3018 else if constexpr (std::is_same_v<Char, char16_t>)
3019 std::copy_n(toInsert.data(), insert_size, insert_start);
3020 else if constexpr (std::is_same_v<Char, char>)
3021 qt_from_latin1(insert_start, toInsert.data(), insert_size);
3022}
3023
3024/*!
3025 \fn QString &QString::insert(qsizetype position, QLatin1StringView str)
3026 \overload insert()
3027
3028 Inserts the Latin-1 string viewed by \a str at the given index \a position.
3029
3030 \include qstring.cpp string-grow-at-insertion
3031*/
3032QString &QString::insert(qsizetype i, QLatin1StringView str)
3033{
3034 const char *s = str.latin1();
3035 if (i < 0 || !s || !(*s))
3036 return *this;
3037
3038 insert_helper(*this, i, str);
3039 return *this;
3040}
3041
3042/*!
3043 \fn QString &QString::insert(qsizetype position, QUtf8StringView str)
3044 \overload insert()
3045 \since 6.5
3046
3047 Inserts the UTF-8 string view \a str at the given index \a position.
3048
3049 \note Inserting variable-width UTF-8-encoded string data is conceptually slower
3050 than inserting fixed-width string data such as UTF-16 (QStringView) or Latin-1
3051 (QLatin1StringView) and should thus be used sparingly.
3052
3053 \include qstring.cpp string-grow-at-insertion
3054*/
3055QString &QString::insert(qsizetype i, QUtf8StringView s)
3056{
3057 auto insert_size = s.size();
3058 if (i < 0 || insert_size <= 0)
3059 return *this;
3060
3061 qsizetype difference = 0;
3062 if (Q_UNLIKELY(i > d.size))
3063 difference = i - d.size;
3064
3065 const qsizetype newSize = d.size + difference + insert_size;
3066
3067 if (d.needsDetach() || needsReallocate(*this, newSize)) {
3068 const auto cbegin = this->cbegin();
3069 const auto insert_start = difference == 0 ? std::next(cbegin, i) : cend();
3070 QString other;
3071 other.reserve(newSize);
3072 other.append(QStringView(cbegin, insert_start));
3073 if (difference > 0)
3074 other.resize(i, u' ');
3075 other.append(s);
3076 other.append(QStringView(insert_start, cend()));
3077 swap(other);
3078 return *this;
3079 }
3080
3081 if (i >= d.size) {
3082 d.detachAndGrow(QArrayData::GrowsAtEnd, difference + insert_size, nullptr, nullptr);
3083 Q_CHECK_PTR(d.data());
3084
3085 if (difference > 0)
3086 resize(i, u' ');
3087 append(s);
3088 } else {
3089 // Optimal insertion of Utf8 data is at the end, anywhere else could
3090 // potentially lead to moving characters twice if Utf8 data size
3091 // (variable-width) is less than the equivalent Utf16 data size
3092 QVarLengthArray<char16_t> buffer(insert_size); // ### optimize (QTBUG-108546)
3093 char16_t *b = QUtf8::convertToUnicode(buffer.data(), s);
3094 insert_helper(*this, i, QStringView(buffer.data(), b));
3095 }
3096
3097 return *this;
3098}
3099
3100/*!
3101 \fn QString& QString::insert(qsizetype position, const QChar *unicode, qsizetype size)
3102 \overload insert()
3103
3104 Inserts the first \a size characters of the QChar array \a unicode
3105 at the given index \a position in the string.
3106
3107 This string grows to accommodate the insertion. If \a position is beyond
3108 the end of the string, space characters are appended to the string to reach
3109 this \a position, followed by \a size characters of the QChar array
3110 \a unicode.
3111*/
3112QString& QString::insert(qsizetype i, const QChar *unicode, qsizetype size)
3113{
3114 if (i < 0 || size <= 0)
3115 return *this;
3116
3117 // In case when data points into "this"
3118 if (!d.needsDetach() && QtPrivate::q_points_into_range(unicode, *this)) {
3119 QVarLengthArray copy(unicode, unicode + size);
3120 insert(i, copy.data(), size);
3121 } else {
3122 insert_helper(*this, i, QStringView(unicode, size));
3123 }
3124
3125 return *this;
3126}
3127
3128/*!
3129 \fn QString& QString::insert(qsizetype position, QChar ch)
3130 \overload insert()
3131
3132 Inserts \a ch at the given index \a position in the string.
3133
3134 This string grows to accommodate the insertion. If \a position is beyond
3135 the end of the string, space characters are appended to the string to reach
3136 this \a position, followed by \a ch.
3137*/
3138
3139QString& QString::insert(qsizetype i, QChar ch)
3140{
3141 if (i < 0)
3142 i += d.size;
3143 return insert(i, &ch, 1);
3144}
3145
3146/*!
3147 Appends the string \a str onto the end of this string.
3148
3149 Example:
3150
3151 \snippet qstring/main.cpp 9
3152
3153 This is the same as using the insert() function:
3154
3155 \snippet qstring/main.cpp 10
3156
3157 The append() function is typically very fast (\l{constant time}),
3158 because QString preallocates extra space at the end of the string
3159 data so it can grow without reallocating the entire string each
3160 time.
3161
3162 \sa operator+=(), prepend(), insert()
3163*/
3164QString &QString::append(const QString &str)
3165{
3166 if (!str.isNull()) {
3167 if (isNull()) {
3168 if (Q_UNLIKELY(!str.d.isMutable()))
3169 assign(str); // fromRawData, so we do a deep copy
3170 else
3171 operator=(str);
3172 } else if (str.size()) {
3173 append(str.constData(), str.size());
3174 }
3175 }
3176 return *this;
3177}
3178
3179/*!
3180 \fn QString &QString::append(QStringView v)
3181 \overload append()
3182 \since 6.0
3183
3184 Appends the given string view \a v to this string and returns the result.
3185*/
3186
3187/*!
3188 \overload append()
3189 \since 5.0
3190
3191 Appends \a len characters from the QChar array \a str to this string.
3192*/
3193QString &QString::append(const QChar *str, qsizetype len)
3194{
3195 if (str && len > 0) {
3196 static_assert(sizeof(QChar) == sizeof(char16_t), "Unexpected difference in sizes");
3197 // the following should be safe as QChar uses char16_t as underlying data
3198 const char16_t *char16String = reinterpret_cast<const char16_t *>(str);
3199 d->growAppend(char16String, char16String + len);
3200 d.data()[d.size] = u'\0';
3201 }
3202 return *this;
3203}
3204
3205/*!
3206 \overload append()
3207
3208 Appends the Latin-1 string viewed by \a str to this string.
3209*/
3210QString &QString::append(QLatin1StringView str)
3211{
3212 append_helper(*this, str);
3213 return *this;
3214}
3215
3216/*!
3217 \overload append()
3218 \since 6.5
3219
3220 Appends the UTF-8 string view \a str to this string.
3221*/
3222QString &QString::append(QUtf8StringView str)
3223{
3224 append_helper(*this, str);
3225 return *this;
3226}
3227
3228/*! \fn QString &QString::append(const QByteArray &ba)
3229
3230 \overload append()
3231
3232 Appends the byte array \a ba to this string. The given byte array
3233 is converted to Unicode using the fromUtf8() function.
3234
3235 You can disable this function by defining \l QT_NO_CAST_FROM_ASCII
3236 when you compile your applications. This can be useful if you want
3237 to ensure that all user-visible strings go through QObject::tr(),
3238 for example.
3239*/
3240
3241/*! \fn QString &QString::append(const char *str)
3242
3243 \overload append()
3244
3245 Appends the string \a str to this string. The given const char
3246 pointer is converted to Unicode using the fromUtf8() function.
3247
3248 You can disable this function by defining \l QT_NO_CAST_FROM_ASCII
3249 when you compile your applications. This can be useful if you want
3250 to ensure that all user-visible strings go through QObject::tr(),
3251 for example.
3252*/
3253
3254/*!
3255 \overload append()
3256
3257 Appends the character \a ch to this string.
3258*/
3259QString &QString::append(QChar ch)
3260{
3261 d.detachAndGrow(QArrayData::GrowsAtEnd, 1, nullptr, nullptr);
3262 d->copyAppend(1, ch.unicode());
3263 d.data()[d.size] = '\0';
3264 return *this;
3265}
3266
3267/*! \fn QString &QString::prepend(const QString &str)
3268
3269 Prepends the string \a str to the beginning of this string and
3270 returns a reference to this string.
3271
3272 This operation is typically very fast (\l{constant time}), because
3273 QString preallocates extra space at the beginning of the string data,
3274 so it can grow without reallocating the entire string each time.
3275
3276 Example:
3277
3278 \snippet qstring/main.cpp 36
3279
3280 \sa append(), insert()
3281*/
3282
3283/*! \fn QString &QString::prepend(QLatin1StringView str)
3284
3285 \overload prepend()
3286
3287 Prepends the Latin-1 string viewed by \a str to this string.
3288*/
3289
3290/*! \fn QString &QString::prepend(QUtf8StringView str)
3291 \since 6.5
3292 \overload prepend()
3293
3294 Prepends the UTF-8 string view \a str to this string.
3295*/
3296
3297/*! \fn QString &QString::prepend(const QChar *str, qsizetype len)
3298 \since 5.5
3299 \overload prepend()
3300
3301 Prepends \a len characters from the QChar array \a str to this string and
3302 returns a reference to this string.
3303*/
3304
3305/*! \fn QString &QString::prepend(QStringView str)
3306 \since 6.0
3307 \overload prepend()
3308
3309 Prepends the string view \a str to the beginning of this string and
3310 returns a reference to this string.
3311*/
3312
3313/*! \fn QString &QString::prepend(const QByteArray &ba)
3314
3315 \overload prepend()
3316
3317 Prepends the byte array \a ba to this string. The byte array is
3318 converted to Unicode using the fromUtf8() function.
3319
3320 You can disable this function by defining
3321 \l QT_NO_CAST_FROM_ASCII when you compile your applications. This
3322 can be useful if you want to ensure that all user-visible strings
3323 go through QObject::tr(), for example.
3324*/
3325
3326/*! \fn QString &QString::prepend(const char *str)
3327
3328 \overload prepend()
3329
3330 Prepends the string \a str to this string. The const char pointer
3331 is converted to Unicode using the fromUtf8() function.
3332
3333 You can disable this function by defining
3334 \l QT_NO_CAST_FROM_ASCII when you compile your applications. This
3335 can be useful if you want to ensure that all user-visible strings
3336 go through QObject::tr(), for example.
3337*/
3338
3339/*! \fn QString &QString::prepend(QChar ch)
3340
3341 \overload prepend()
3342
3343 Prepends the character \a ch to this string.
3344*/
3345
3346/*!
3347 \fn QString &QString::assign(QAnyStringView v)
3348 \since 6.6
3349
3350 Replaces the contents of this string with a copy of \a v and returns a
3351 reference to this string.
3352
3353 The size of this string will be equal to the size of \a v, converted to
3354 UTF-16 as if by \c{v.toString()}. Unlike QAnyStringView::toString(), however,
3355 this function only allocates memory if the estimated size exceeds the capacity
3356 of this string or this string is shared.
3357
3358 \sa QAnyStringView::toString()
3359*/
3360
3361/*!
3362 \fn QString &QString::assign(qsizetype n, QChar c)
3363 \since 6.6
3364
3365 Replaces the contents of this string with \a n copies of \a c and
3366 returns a reference to this string.
3367
3368 The size of this string will be equal to \a n, which has to be non-negative.
3369
3370 This function will only allocate memory if \a n exceeds the capacity of this
3371 string or this string is shared.
3372
3373 \sa fill()
3374*/
3375
3376/*!
3377 \fn template <typename InputIterator, QString::if_compatible_iterator<InputIterator>> QString &QString::assign(InputIterator first, InputIterator last)
3378 \since 6.6
3379
3380 Replaces the contents of this string with a copy of the elements in the
3381 iterator range [\a first, \a last) and returns a reference to this string.
3382
3383 The size of this string will be equal to the decoded length of the elements
3384 in the range [\a first, \a last), which need not be the same as the length of
3385 the range itself, because this function transparently recodes the input
3386 character set to UTF-16.
3387
3388 This function will only allocate memory if the number of elements in the
3389 range, or, for non-UTF-16-encoded input, the maximum possible size of the
3390 resulting string, exceeds the capacity of this string, or if this string is
3391 shared.
3392
3393 \note The behavior is undefined if either argument is an iterator into *this or
3394 [\a first, \a last) is not a valid range.
3395
3396 \constraints
3397 \c InputIterator meets the requirements of a
3398 \l {https://en.cppreference.com/w/cpp/named_req/InputIterator} {LegacyInputIterator}
3399 and the \c{value_type} of \c InputIterator is one of the following character types:
3400 \list
3401 \li QChar
3402 \li QLatin1Char
3403 \li \c {char}
3404 \li \c {unsigned char}
3405 \li \c {signed char}
3406 \li \c {char8_t}
3407 \li \c char16_t
3408 \li (on platforms, such as Windows, where it is a 16-bit type) \c wchar_t
3409 \li \c char32_t
3410 \endlist
3411*/
3412
3413QString &QString::assign(QAnyStringView s)
3414{
3415 if (s.size() <= capacity() && isDetached()) {
3416 const auto offset = d.freeSpaceAtBegin();
3417 if (offset)
3418 d.setBegin(d.begin() - offset);
3419 resize(0);
3420 s.visit([this](auto input) {
3421 this->append(input);
3422 });
3423 } else {
3424 *this = s.toString();
3425 }
3426 return *this;
3427}
3428
3429#ifndef QT_BOOTSTRAPPED
3430QString &QString::assign_helper(const char32_t *data, qsizetype len)
3431{
3432 // worst case: each char32_t requires a surrogate pair, so
3433 const auto requiredCapacity = len * 2;
3434 if (requiredCapacity <= capacity() && isDetached()) {
3435 const auto offset = d.freeSpaceAtBegin();
3436 if (offset)
3437 d.setBegin(d.begin() - offset);
3438 auto begin = reinterpret_cast<QChar *>(d.begin());
3439 auto ba = QByteArrayView(reinterpret_cast<const std::byte*>(data), len * sizeof(char32_t));
3440 QStringConverter::State state;
3441 const auto end = QUtf32::convertToUnicode(begin, ba, &state, DetectEndianness);
3442 d.size = end - begin;
3443 d.data()[d.size] = u'\0';
3444 } else {
3445 *this = QString::fromUcs4(data, len);
3446 }
3447 return *this;
3448}
3449#endif
3450
3451/*!
3452 \fn QString &QString::remove(qsizetype position, qsizetype n)
3453
3454 Removes \a n characters from the string, starting at the given \a
3455 position index, and returns a reference to the string.
3456
3457 If the specified \a position index is within the string, but \a
3458 position + \a n is beyond the end of the string, the string is
3459 truncated at the specified \a position.
3460
3461 If \a n is <= 0 nothing is changed.
3462
3463 \snippet qstring/main.cpp 37
3464
3465//! [shrinking-erase]
3466 Element removal will preserve the string's capacity and not reduce the
3467 amount of allocated memory. To shed extra capacity and free as much memory
3468 as possible, call squeeze() after the last change to the string's size.
3469//! [shrinking-erase]
3470
3471 \sa insert(), replace()
3472*/
3473QString &QString::remove(qsizetype pos, qsizetype len)
3474{
3475 if (pos < 0) // count from end of string
3476 pos += size();
3477
3478 if (size_t(pos) >= size_t(size()) || len <= 0)
3479 return *this;
3480
3481 len = std::min(len, size() - pos);
3482
3483 if (!d.isShared()) {
3484 d->erase(d.begin() + pos, len);
3485 d.data()[d.size] = u'\0';
3486 } else {
3487 // TODO: either reserve "size()", which is bigger than needed, or
3488 // modify the shrinking-erase docs of this method (since the size
3489 // of "copy" won't have any extra capacity any more)
3490 const qsizetype sz = size() - len;
3491 QString copy{sz, Qt::Uninitialized};
3492 auto begin = d.begin();
3493 auto toRemove_start = d.begin() + pos;
3494 copy.d->copyRanges({{begin, toRemove_start},
3495 {toRemove_start + len, d.end()}});
3496 swap(copy);
3497 }
3498 return *this;
3499}
3500
3501template<typename T>
3502static void removeStringImpl(QString &s, const T &needle, Qt::CaseSensitivity cs)
3503{
3504 const auto needleSize = needle.size();
3505 if (!needleSize)
3506 return;
3507
3508 // avoid detach if nothing to do:
3509 qsizetype i = s.indexOf(needle, 0, cs);
3510 if (i < 0)
3511 return;
3512
3513 QString::DataPointer &dptr = s.data_ptr();
3514 auto begin = dptr.begin();
3515 auto end = dptr.end();
3516
3517 auto copyFunc = [&](auto &dst) {
3518 auto src = begin + i + needleSize;
3519 while (src < end) {
3520 i = s.indexOf(needle, std::distance(begin, src), cs);
3521 auto hit = i == -1 ? end : begin + i;
3522 dst = std::copy(src, hit, dst);
3523 src = hit + needleSize;
3524 }
3525 return dst;
3526 };
3527
3528 if (!dptr.needsDetach()) {
3529 auto dst = begin + i;
3530 dst = copyFunc(dst);
3531 s.truncate(std::distance(begin, dst));
3532 } else {
3533 QString copy{s.size(), Qt::Uninitialized};
3534 auto copy_begin = copy.begin();
3535 auto dst = std::copy(begin, begin + i, copy_begin); // Chunk before the first hit
3536 dst = copyFunc(dst);
3537 copy.resize(std::distance(copy_begin, dst));
3538 s.swap(copy);
3539 }
3540}
3541
3542/*!
3543 Removes every occurrence of the given \a str string in this
3544 string, and returns a reference to this string.
3545
3546 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
3547
3548 This is the same as \c replace(str, "", cs).
3549
3550 \include qstring.cpp shrinking-erase
3551
3552 \sa replace()
3553*/
3554QString &QString::remove(const QString &str, Qt::CaseSensitivity cs)
3555{
3556 const auto s = str.d.data();
3557 if (QtPrivate::q_points_into_range(s, d))
3558 removeStringImpl(*this, QStringView{QVarLengthArray(s, s + str.size())}, cs);
3559 else
3560 removeStringImpl(*this, qToStringViewIgnoringNull(str), cs);
3561 return *this;
3562}
3563
3564/*!
3565 \since 5.11
3566 \overload
3567
3568 Removes every occurrence of the given Latin-1 string viewed by \a str
3569 from this string, and returns a reference to this string.
3570
3571 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
3572
3573 This is the same as \c replace(str, "", cs).
3574
3575 \include qstring.cpp shrinking-erase
3576
3577 \sa replace()
3578*/
3579QString &QString::remove(QLatin1StringView str, Qt::CaseSensitivity cs)
3580{
3581 removeStringImpl(*this, str, cs);
3582 return *this;
3583}
3584
3585/*!
3586 \fn QString &QString::removeAt(qsizetype pos)
3587
3588 \since 6.5
3589
3590 Removes the character at index \a pos. If \a pos is out of bounds
3591 (i.e. \a pos >= size()), this function does nothing.
3592
3593 \sa remove()
3594*/
3595
3596/*!
3597 \fn QString &QString::removeFirst()
3598
3599 \since 6.5
3600
3601 Removes the first character in this string. If the string is empty,
3602 this function does nothing.
3603
3604 \sa remove()
3605*/
3606
3607/*!
3608 \fn QString &QString::removeLast()
3609
3610 \since 6.5
3611
3612 Removes the last character in this string. If the string is empty,
3613 this function does nothing.
3614
3615 \sa remove()
3616*/
3617
3618/*!
3619 Removes every occurrence of the character \a ch in this string, and
3620 returns a reference to this string.
3621
3622 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
3623
3624 Example:
3625
3626 \snippet qstring/main.cpp 38
3627
3628 This is the same as \c replace(ch, "", cs).
3629
3630 \include qstring.cpp shrinking-erase
3631
3632 \sa replace()
3633*/
3634QString &QString::remove(QChar ch, Qt::CaseSensitivity cs)
3635{
3636 const qsizetype idx = indexOf(ch, 0, cs);
3637 if (idx == -1)
3638 return *this;
3639
3640 const bool isCase = cs == Qt::CaseSensitive;
3641 ch = isCase ? ch : ch.toCaseFolded();
3642 auto match = [ch, isCase](QChar x) {
3643 return ch == (isCase ? x : x.toCaseFolded());
3644 };
3645
3646
3647 auto begin = d.begin();
3648 auto first_match = begin + idx;
3649 auto end = d.end();
3650 if (!d.isShared()) {
3651 auto it = std::remove_if(first_match, end, match);
3652 d->erase(it, std::distance(it, end));
3653 d.data()[d.size] = u'\0';
3654 } else {
3655 // Instead of detaching, create a new string and copy all characters except for
3656 // the ones we're removing
3657 // TODO: size() is more than the needed since "copy" would be shorter
3658 QString copy{size(), Qt::Uninitialized};
3659 auto dst = copy.d.begin();
3660 auto it = std::copy(begin, first_match, dst); // Chunk before idx
3661 it = std::remove_copy_if(first_match + 1, end, it, match);
3662 copy.d.size = std::distance(dst, it);
3663 copy.d.data()[copy.d.size] = u'\0';
3664 *this = std::move(copy);
3665 }
3666 return *this;
3667}
3668
3669/*!
3670 \fn QString &QString::remove(const QRegularExpression &re)
3671 \since 5.0
3672
3673 Removes every occurrence of the regular expression \a re in the
3674 string, and returns a reference to the string. For example:
3675
3676 \snippet qstring/main.cpp 96
3677
3678 \include qstring.cpp shrinking-erase
3679
3680 \sa indexOf(), lastIndexOf(), replace()
3681*/
3682
3683/*!
3684 \fn template <typename Predicate> QString &QString::removeIf(Predicate pred)
3685 \since 6.1
3686
3687 Removes all elements for which the predicate \a pred returns true
3688 from the string. Returns a reference to the string.
3689
3690 \sa remove()
3691*/
3692
3693static void replace_helper(QString &str, QSpan<qsizetype> indices, qsizetype blen, QStringView after)
3694{
3695 const qsizetype oldSize = str.data_ptr().size;
3696 const qsizetype adjust = indices.size() * (after.size() - blen);
3697 const qsizetype newSize = oldSize + adjust;
3698 using A = QStringAlgorithms<QString>;
3699 if (str.data_ptr().needsDetach() || needsReallocate(str, newSize)) {
3700 A::replace_helper(str, blen, after, indices);
3701 return;
3702 }
3703
3704 if (QtPrivate::q_points_into_range(after.begin(), str)) {
3705 // Copy after if it lies inside our own d.b area (which we could
3706 // possibly invalidate via a realloc or modify by replacement)
3707 A::replace_helper(str, blen, QVarLengthArray(after.begin(), after.end()), indices);
3708 } else {
3709 A::replace_helper(str, blen, after, indices);
3710 }
3711}
3712
3713/*!
3714 \fn QString &QString::replace(qsizetype position, qsizetype n, const QString &after)
3715
3716 Replaces \a n characters beginning at index \a position with
3717 the string \a after and returns a reference to this string.
3718
3719 \note If the specified \a position index is within the string,
3720 but \a position + \a n goes outside the strings range,
3721 then \a n will be adjusted to stop at the end of the string.
3722
3723 Example:
3724
3725 \snippet qstring/main.cpp 40
3726
3727 \sa insert(), remove()
3728*/
3729QString &QString::replace(qsizetype pos, qsizetype len, const QString &after)
3730{
3731 return replace(pos, len, after.constData(), after.size());
3732}
3733
3734/*!
3735 \fn QString &QString::replace(qsizetype position, qsizetype n, const QChar *after, qsizetype alen)
3736 \overload replace()
3737 Replaces \a n characters beginning at index \a position with the
3738 first \a alen characters of the QChar array \a after and returns a
3739 reference to this string.
3740
3741 \a n must not be negative.
3742*/
3743QString &QString::replace(qsizetype pos, qsizetype len, const QChar *after, qsizetype alen)
3744{
3745 Q_PRE(len >= 0);
3746
3747 if (size_t(pos) > size_t(this->size()))
3748 return *this;
3749 if (len > this->size() - pos)
3750 len = this->size() - pos;
3751
3752 qsizetype indices[] = {pos};
3753 replace_helper(*this, indices, len, QStringView{after, alen});
3754 return *this;
3755}
3756
3757/*!
3758 \fn QString &QString::replace(qsizetype position, qsizetype n, QChar after)
3759 \overload replace()
3760
3761 Replaces \a n characters beginning at index \a position with the
3762 character \a after and returns a reference to this string.
3763*/
3764QString &QString::replace(qsizetype pos, qsizetype len, QChar after)
3765{
3766 return replace(pos, len, &after, 1);
3767}
3768
3769/*!
3770 \overload replace()
3771 Replaces every occurrence of the string \a before with the string \a
3772 after and returns a reference to this string.
3773
3774 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
3775
3776 Example:
3777
3778 \snippet qstring/main.cpp 41
3779
3780 \note The replacement text is not rescanned after it is inserted.
3781
3782 Example:
3783
3784 \snippet qstring/main.cpp 86
3785
3786//! [empty-before-arg-in-replace]
3787 \note If you use an empty \a before argument, the \a after argument will be
3788 inserted \e {before and after} each character of the string.
3789//! [empty-before-arg-in-replace]
3790
3791*/
3792QString &QString::replace(const QString &before, const QString &after, Qt::CaseSensitivity cs)
3793{
3794 return replace(before.constData(), before.size(), after.constData(), after.size(), cs);
3795}
3796
3797/*!
3798 \since 4.5
3799 \overload replace()
3800
3801 Replaces each occurrence in this string of the first \a blen
3802 characters of \a before with the first \a alen characters of \a
3803 after and returns a reference to this string.
3804
3805 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
3806
3807 \note If \a before points to an \e empty string (that is, \a blen == 0),
3808 the string pointed to by \a after will be inserted \e {before and after}
3809 each character in this string.
3810*/
3811QString &QString::replace(const QChar *before, qsizetype blen,
3812 const QChar *after, qsizetype alen,
3813 Qt::CaseSensitivity cs)
3814{
3815 if (isEmpty()) {
3816 if (blen)
3817 return *this;
3818 } else {
3819 if (cs == Qt::CaseSensitive && before == after && blen == alen)
3820 return *this;
3821 }
3822 if (alen == 0 && blen == 0)
3823 return *this;
3824 if (alen == 1 && blen == 1)
3825 return replace(*before, *after, cs);
3826
3827 QStringMatcher matcher(before, blen, cs);
3828
3829 qsizetype index = 0;
3830
3831 QVarLengthArray<qsizetype> indices;
3832 while ((index = matcher.indexIn(*this, index)) != -1) {
3833 indices.push_back(index);
3834 if (blen) // Step over before:
3835 index += blen;
3836 else // Only count one instance of empty between any two characters:
3837 index++;
3838 }
3839 if (indices.isEmpty())
3840 return *this;
3841
3842 replace_helper(*this, indices, blen, QStringView{after, alen});
3843 return *this;
3844}
3845
3846/*!
3847 \overload replace()
3848 Replaces every occurrence of the character \a ch in the string with
3849 \a after and returns a reference to this string.
3850
3851 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
3852*/
3853QString& QString::replace(QChar ch, const QString &after, Qt::CaseSensitivity cs)
3854{
3855 if (after.size() == 0)
3856 return remove(ch, cs);
3857
3858 if (after.size() == 1)
3859 return replace(ch, after.front(), cs);
3860
3861 if (size() == 0)
3862 return *this;
3863
3864 const char16_t cc = (cs == Qt::CaseSensitive ? ch.unicode() : ch.toCaseFolded().unicode());
3865
3866 QVarLengthArray<qsizetype> indices;
3867 if (cs == Qt::CaseSensitive) {
3868 const char16_t *begin = d.begin();
3869 const char16_t *end = d.end();
3870 QStringView view(begin, end);
3871 const char16_t *hit = nullptr;
3872 while ((hit = QtPrivate::qustrchr(view, cc)) != end) {
3873 indices.push_back(std::distance(begin, hit));
3874 view = QStringView(std::next(hit), end);
3875 }
3876 } else {
3877 for (qsizetype i = 0; i < d.size; ++i)
3878 if (QChar::toCaseFolded(d.data()[i]) == cc)
3879 indices.push_back(i);
3880 }
3881 if (indices.isEmpty())
3882 return *this;
3883
3884 replace_helper(*this, indices, 1, after);
3885 return *this;
3886}
3887
3888/*!
3889 \overload replace()
3890 Replaces every occurrence of the character \a before with the
3891 character \a after and returns a reference to this string.
3892
3893 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
3894*/
3895QString& QString::replace(QChar before, QChar after, Qt::CaseSensitivity cs)
3896{
3897 const qsizetype idx = indexOf(before, 0, cs);
3898 if (idx == -1)
3899 return *this;
3900
3901 const char16_t achar = after.unicode();
3902 char16_t bchar = before.unicode();
3903
3904 auto matchesCIS = [](char16_t beforeChar) {
3905 return [beforeChar](char16_t ch) { return foldAndCompare(ch, beforeChar); };
3906 };
3907
3908 auto hit = d.begin() + idx;
3909 if (!d.needsDetach()) {
3910 *hit++ = achar;
3911 if (cs == Qt::CaseSensitive) {
3912 std::replace(hit, d.end(), bchar, achar);
3913 } else {
3914 bchar = foldCase(bchar);
3915 std::replace_if(hit, d.end(), matchesCIS(bchar), achar);
3916 }
3917 } else {
3918 QString other{ d.size, Qt::Uninitialized };
3919 auto dest = std::copy(d.begin(), hit, other.d.begin());
3920 *dest++ = achar;
3921 ++hit;
3922 if (cs == Qt::CaseSensitive) {
3923 std::replace_copy(hit, d.end(), dest, bchar, achar);
3924 } else {
3925 bchar = foldCase(bchar);
3926 std::replace_copy_if(hit, d.end(), dest, matchesCIS(bchar), achar);
3927 }
3928
3929 swap(other);
3930 }
3931 return *this;
3932}
3933
3934/*!
3935 \since 4.5
3936 \overload replace()
3937
3938 Replaces every occurrence in this string of the Latin-1 string viewed
3939 by \a before with the Latin-1 string viewed by \a after, and returns a
3940 reference to this string.
3941
3942 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
3943
3944 \note The text is not rescanned after a replacement.
3945
3946 \include qstring.cpp empty-before-arg-in-replace
3947*/
3948QString &QString::replace(QLatin1StringView before, QLatin1StringView after, Qt::CaseSensitivity cs)
3949{
3950 const qsizetype alen = after.size();
3951 const qsizetype blen = before.size();
3952 if (blen == 1 && alen == 1)
3953 return replace(before.front(), after.front(), cs);
3954
3955 QVarLengthArray<char16_t> a = qt_from_latin1_to_qvla(after);
3956 QVarLengthArray<char16_t> b = qt_from_latin1_to_qvla(before);
3957 return replace((const QChar *)b.data(), blen, (const QChar *)a.data(), alen, cs);
3958}
3959
3960/*!
3961 \since 4.5
3962 \overload replace()
3963
3964 Replaces every occurrence in this string of the Latin-1 string viewed
3965 by \a before with the string \a after, and returns a reference to this
3966 string.
3967
3968 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
3969
3970 \note The text is not rescanned after a replacement.
3971
3972 \include qstring.cpp empty-before-arg-in-replace
3973*/
3974QString &QString::replace(QLatin1StringView before, const QString &after, Qt::CaseSensitivity cs)
3975{
3976 const qsizetype blen = before.size();
3977 if (blen == 1 && after.size() == 1)
3978 return replace(before.front(), after.front(), cs);
3979
3980 QVarLengthArray<char16_t> b = qt_from_latin1_to_qvla(before);
3981 return replace((const QChar *)b.data(), blen, after.constData(), after.d.size, cs);
3982}
3983
3984/*!
3985 \since 4.5
3986 \overload replace()
3987
3988 Replaces every occurrence of the string \a before with the string \a
3989 after and returns a reference to this string.
3990
3991 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
3992
3993 \note The text is not rescanned after a replacement.
3994
3995 \include qstring.cpp empty-before-arg-in-replace
3996*/
3997QString &QString::replace(const QString &before, QLatin1StringView after, Qt::CaseSensitivity cs)
3998{
3999 const qsizetype alen = after.size();
4000 if (before.size() == 1 && alen == 1)
4001 return replace(before.front(), after.front(), cs);
4002
4003 QVarLengthArray<char16_t> a = qt_from_latin1_to_qvla(after);
4004 return replace(before.constData(), before.d.size, (const QChar *)a.data(), alen, cs);
4005}
4006
4007/*!
4008 \since 4.5
4009 \overload replace()
4010
4011 Replaces every occurrence of the character \a c with the string \a
4012 after and returns a reference to this string.
4013
4014 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
4015
4016 \note The text is not rescanned after a replacement.
4017*/
4018QString &QString::replace(QChar c, QLatin1StringView after, Qt::CaseSensitivity cs)
4019{
4020 const qsizetype alen = after.size();
4021 if (alen == 1)
4022 return replace(c, after.front(), cs);
4023
4024 QVarLengthArray<char16_t> a = qt_from_latin1_to_qvla(after);
4025 return replace(&c, 1, (const QChar *)a.data(), alen, cs);
4026}
4027
4028/*!
4029 \fn bool QString::operator==(const QString &lhs, const QString &rhs)
4030 \overload operator==()
4031
4032 Returns \c true if string \a lhs is equal to string \a rhs; otherwise
4033 returns \c false.
4034
4035 \include qstring.cpp compare-isNull-vs-isEmpty
4036
4037 \sa {Comparing Strings}
4038*/
4039
4040/*!
4041 \fn bool QString::operator==(const QString &lhs, const QLatin1StringView &rhs)
4042
4043 \overload operator==()
4044
4045 Returns \c true if \a lhs is equal to \a rhs; otherwise
4046 returns \c false.
4047*/
4048
4049/*!
4050 \fn bool QString::operator==(const QLatin1StringView &lhs, const QString &rhs)
4051
4052 \overload operator==()
4053
4054 Returns \c true if \a lhs is equal to \a rhs; otherwise
4055 returns \c false.
4056*/
4057
4058/*! \fn bool QString::operator==(const QString &lhs, const QByteArray &rhs)
4059
4060 \overload operator==()
4061
4062 The \a rhs byte array is converted to a QUtf8StringView.
4063
4064 You can disable this operator by defining
4065 \l QT_NO_CAST_FROM_ASCII when you compile your applications. This
4066 can be useful if you want to ensure that all user-visible strings
4067 go through QObject::tr(), for example.
4068
4069 Returns \c true if string \a lhs is lexically equal to \a rhs.
4070 Otherwise returns \c false.
4071*/
4072
4073/*! \fn bool QString::operator==(const QString &lhs, const char * const &rhs)
4074
4075 \overload operator==()
4076
4077 The \a rhs const char pointer is converted to a QUtf8StringView.
4078
4079 You can disable this operator by defining
4080 \l QT_NO_CAST_FROM_ASCII when you compile your applications. This
4081 can be useful if you want to ensure that all user-visible strings
4082 go through QObject::tr(), for example.
4083*/
4084
4085/*!
4086 \fn bool QString::operator<(const QString &lhs, const QString &rhs)
4087
4088 \overload operator<()
4089
4090 Returns \c true if string \a lhs is lexically less than string
4091 \a rhs; otherwise returns \c false.
4092
4093 \sa {Comparing Strings}
4094*/
4095
4096/*!
4097 \fn bool QString::operator<(const QString &lhs, const QLatin1StringView &rhs)
4098
4099 \overload operator<()
4100
4101 Returns \c true if \a lhs is lexically less than \a rhs;
4102 otherwise returns \c false.
4103*/
4104
4105/*!
4106 \fn bool QString::operator<(const QLatin1StringView &lhs, const QString &rhs)
4107
4108 \overload operator<()
4109
4110 Returns \c true if \a lhs is lexically less than \a rhs;
4111 otherwise returns \c false.
4112*/
4113
4114/*! \fn bool QString::operator<(const QString &lhs, const QByteArray &rhs)
4115
4116 \overload operator<()
4117
4118 The \a rhs byte array is converted to a QUtf8StringView.
4119 If any NUL characters ('\\0') are embedded in the byte array, they will be
4120 included in the transformation.
4121
4122 You can disable this operator
4123 \l QT_NO_CAST_FROM_ASCII when you compile your applications. This
4124 can be useful if you want to ensure that all user-visible strings
4125 go through QObject::tr(), for example.
4126*/
4127
4128/*! \fn bool QString::operator<(const QString &lhs, const char * const &rhs)
4129
4130 Returns \c true if string \a lhs is lexically less than string \a rhs.
4131 Otherwise returns \c false.
4132
4133 \overload operator<()
4134
4135 The \a rhs const char pointer is converted to a QUtf8StringView.
4136
4137 You can disable this operator by defining
4138 \l QT_NO_CAST_FROM_ASCII when you compile your applications. This
4139 can be useful if you want to ensure that all user-visible strings
4140 go through QObject::tr(), for example.
4141*/
4142
4143/*! \fn bool QString::operator<=(const QString &lhs, const QString &rhs)
4144
4145 Returns \c true if string \a lhs is lexically less than or equal to
4146 string \a rhs; otherwise returns \c false.
4147
4148 \sa {Comparing Strings}
4149*/
4150
4151/*!
4152 \fn bool QString::operator<=(const QString &lhs, const QLatin1StringView &rhs)
4153
4154 \overload operator<=()
4155
4156 Returns \c true if \a lhs is lexically less than or equal to \a rhs;
4157 otherwise returns \c false.
4158*/
4159
4160/*!
4161 \fn bool QString::operator<=(const QLatin1StringView &lhs, const QString &rhs)
4162
4163 \overload operator<=()
4164
4165 Returns \c true if \a lhs is lexically less than or equal to \a rhs;
4166 otherwise returns \c false.
4167*/
4168
4169/*! \fn bool QString::operator<=(const QString &lhs, const QByteArray &rhs)
4170
4171 \overload operator<=()
4172
4173 The \a rhs byte array is converted to a QUtf8StringView.
4174 If any NUL characters ('\\0') are embedded in the byte array, they will be
4175 included in the transformation.
4176
4177 You can disable this operator by defining
4178 \l QT_NO_CAST_FROM_ASCII when you compile your applications. This
4179 can be useful if you want to ensure that all user-visible strings
4180 go through QObject::tr(), for example.
4181*/
4182
4183/*! \fn bool QString::operator<=(const QString &lhs, const char * const &rhs)
4184
4185 \overload operator<=()
4186
4187 The \a rhs const char pointer is converted to a QUtf8StringView.
4188
4189 You can disable this operator by defining
4190 \l QT_NO_CAST_FROM_ASCII when you compile your applications. This
4191 can be useful if you want to ensure that all user-visible strings
4192 go through QObject::tr(), for example.
4193*/
4194
4195/*! \fn bool QString::operator>(const QString &lhs, const QString &rhs)
4196
4197 Returns \c true if string \a lhs is lexically greater than string \a rhs;
4198 otherwise returns \c false.
4199
4200 \sa {Comparing Strings}
4201*/
4202
4203/*!
4204 \fn bool QString::operator>(const QString &lhs, const QLatin1StringView &rhs)
4205
4206 \overload operator>()
4207
4208 Returns \c true if \a lhs is lexically greater than \a rhs;
4209 otherwise returns \c false.
4210*/
4211
4212/*!
4213 \fn bool QString::operator>(const QLatin1StringView &lhs, const QString &rhs)
4214
4215 \overload operator>()
4216
4217 Returns \c true if \a lhs is lexically greater than \a rhs;
4218 otherwise returns \c false.
4219*/
4220
4221/*! \fn bool QString::operator>(const QString &lhs, const QByteArray &rhs)
4222
4223 \overload operator>()
4224
4225 The \a rhs byte array is converted to a QUtf8StringView.
4226 If any NUL characters ('\\0') are embedded in the byte array, they will be
4227 included in the transformation.
4228
4229 You can disable this operator by defining
4230 \l QT_NO_CAST_FROM_ASCII when you compile your applications. This
4231 can be useful if you want to ensure that all user-visible strings
4232 go through QObject::tr(), for example.
4233*/
4234
4235/*! \fn bool QString::operator>(const QString &lhs, const char * const &rhs)
4236
4237 \overload operator>()
4238
4239 The \a rhs const char pointer is converted to a QUtf8StringView.
4240
4241 You can disable this operator by defining \l QT_NO_CAST_FROM_ASCII
4242 when you compile your applications. This can be useful if you want
4243 to ensure that all user-visible strings go through QObject::tr(),
4244 for example.
4245*/
4246
4247/*! \fn bool QString::operator>=(const QString &lhs, const QString &rhs)
4248
4249 Returns \c true if string \a lhs is lexically greater than or equal to
4250 string \a rhs; otherwise returns \c false.
4251
4252 \sa {Comparing Strings}
4253*/
4254
4255/*!
4256 \fn bool QString::operator>=(const QString &lhs, const QLatin1StringView &rhs)
4257
4258 \overload operator>=()
4259
4260 Returns \c true if \a lhs is lexically greater than or equal to \a rhs;
4261 otherwise returns \c false.
4262*/
4263
4264/*!
4265 \fn bool QString::operator>=(const QLatin1StringView &lhs, const QString &rhs)
4266
4267 \overload operator>=()
4268
4269 Returns \c true if \a lhs is lexically greater than or equal to \a rhs;
4270 otherwise returns \c false.
4271*/
4272
4273/*! \fn bool QString::operator>=(const QString &lhs, const QByteArray &rhs)
4274
4275 \overload operator>=()
4276
4277 The \a rhs byte array is converted to a QUtf8StringView.
4278 If any NUL characters ('\\0') are embedded in the byte array, they will be
4279 included in the transformation.
4280
4281 You can disable this operator by defining \l QT_NO_CAST_FROM_ASCII
4282 when you compile your applications. This can be useful if you want
4283 to ensure that all user-visible strings go through QObject::tr(),
4284 for example.
4285*/
4286
4287/*! \fn bool QString::operator>=(const QString &lhs, const char * const &rhs)
4288
4289 \overload operator>=()
4290
4291 The \a rhs const char pointer is converted to a QUtf8StringView.
4292
4293 You can disable this operator by defining \l QT_NO_CAST_FROM_ASCII
4294 when you compile your applications. This can be useful if you want
4295 to ensure that all user-visible strings go through QObject::tr(),
4296 for example.
4297*/
4298
4299/*! \fn bool QString::operator!=(const QString &lhs, const QString &rhs)
4300
4301 Returns \c true if string \a lhs is not equal to string \a rhs;
4302 otherwise returns \c false.
4303
4304 \sa {Comparing Strings}
4305*/
4306
4307/*! \fn bool QString::operator!=(const QString &lhs, const QLatin1StringView &rhs)
4308
4309 Returns \c true if string \a lhs is not equal to string \a rhs.
4310 Otherwise returns \c false.
4311
4312 \overload operator!=()
4313*/
4314
4315/*! \fn bool QString::operator!=(const QString &lhs, const QByteArray &rhs)
4316
4317 \overload operator!=()
4318
4319 The \a rhs byte array is converted to a QUtf8StringView.
4320 If any NUL characters ('\\0') are embedded in the byte array, they will be
4321 included in the transformation.
4322
4323 You can disable this operator by defining \l QT_NO_CAST_FROM_ASCII
4324 when you compile your applications. This can be useful if you want
4325 to ensure that all user-visible strings go through QObject::tr(),
4326 for example.
4327*/
4328
4329/*! \fn bool QString::operator!=(const QString &lhs, const char * const &rhs)
4330
4331 \overload operator!=()
4332
4333 The \a rhs const char pointer is converted to a QUtf8StringView.
4334
4335 You can disable this operator by defining
4336 \l QT_NO_CAST_FROM_ASCII when you compile your applications. This
4337 can be useful if you want to ensure that all user-visible strings
4338 go through QObject::tr(), for example.
4339*/
4340
4341/*! \fn bool QString::operator==(const QByteArray &lhs, const QString &rhs)
4342
4343 Returns \c true if byte array \a lhs is equal to the UTF-8 encoding of
4344 \a rhs; otherwise returns \c false.
4345
4346 The comparison is case sensitive.
4347
4348 You can disable this operator by defining \c
4349 QT_NO_CAST_FROM_ASCII when you compile your applications. You
4350 then need to call QString::fromUtf8(), QString::fromLatin1(),
4351 or QString::fromLocal8Bit() explicitly if you want to convert the byte
4352 array to a QString before doing the comparison.
4353*/
4354
4355/*! \fn bool QString::operator!=(const QByteArray &lhs, const QString &rhs)
4356
4357 Returns \c true if byte array \a lhs is not equal to the UTF-8 encoding of
4358 \a rhs; otherwise returns \c false.
4359
4360 The comparison is case sensitive.
4361
4362 You can disable this operator by defining \c
4363 QT_NO_CAST_FROM_ASCII when you compile your applications. You
4364 then need to call QString::fromUtf8(), QString::fromLatin1(),
4365 or QString::fromLocal8Bit() explicitly if you want to convert the byte
4366 array to a QString before doing the comparison.
4367*/
4368
4369/*! \fn bool QString::operator<(const QByteArray &lhs, const QString &rhs)
4370
4371 Returns \c true if byte array \a lhs is lexically less than the UTF-8 encoding
4372 of \a rhs; otherwise returns \c false.
4373
4374 The comparison is case sensitive.
4375
4376 You can disable this operator by defining \c
4377 QT_NO_CAST_FROM_ASCII when you compile your applications. You
4378 then need to call QString::fromUtf8(), QString::fromLatin1(),
4379 or QString::fromLocal8Bit() explicitly if you want to convert the byte
4380 array to a QString before doing the comparison.
4381*/
4382
4383/*! \fn bool QString::operator>(const QByteArray &lhs, const QString &rhs)
4384
4385 Returns \c true if byte array \a lhs is lexically greater than the UTF-8
4386 encoding of \a rhs; otherwise returns \c false.
4387
4388 The comparison is case sensitive.
4389
4390 You can disable this operator by defining \c
4391 QT_NO_CAST_FROM_ASCII when you compile your applications. You
4392 then need to call QString::fromUtf8(), QString::fromLatin1(),
4393 or QString::fromLocal8Bit() explicitly if you want to convert the byte
4394 array to a QString before doing the comparison.
4395*/
4396
4397/*! \fn bool QString::operator<=(const QByteArray &lhs, const QString &rhs)
4398
4399 Returns \c true if byte array \a lhs is lexically less than or equal to the
4400 UTF-8 encoding of \a rhs; otherwise returns \c false.
4401
4402 The comparison is case sensitive.
4403
4404 You can disable this operator by defining \c
4405 QT_NO_CAST_FROM_ASCII when you compile your applications. You
4406 then need to call QString::fromUtf8(), QString::fromLatin1(),
4407 or QString::fromLocal8Bit() explicitly if you want to convert the byte
4408 array to a QString before doing the comparison.
4409*/
4410
4411/*! \fn bool QString::operator>=(const QByteArray &lhs, const QString &rhs)
4412
4413 Returns \c true if byte array \a lhs is greater than or equal to the UTF-8
4414 encoding of \a rhs; otherwise returns \c false.
4415
4416 The comparison is case sensitive.
4417
4418 You can disable this operator by defining \c
4419 QT_NO_CAST_FROM_ASCII when you compile your applications. You
4420 then need to call QString::fromUtf8(), QString::fromLatin1(),
4421 or QString::fromLocal8Bit() explicitly if you want to convert the byte
4422 array to a QString before doing the comparison.
4423*/
4424
4425/*!
4426 \fn qsizetype QString::indexOf(const QString &str, qsizetype from, Qt::CaseSensitivity cs) const
4427 \include qstring.qdocinc {qstring-first-index-of} {string} {str}
4428
4429 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
4430
4431 Example:
4432
4433 \snippet qstring/main.cpp 24
4434
4435 \include qstring.qdocinc negative-index-start-search-from-end
4436
4437 \sa lastIndexOf(), contains(), count()
4438*/
4439
4440/*!
4441 \fn qsizetype QString::indexOf(QStringView str, qsizetype from, Qt::CaseSensitivity cs) const
4442 \since 5.14
4443 \overload indexOf()
4444
4445 \include qstring.qdocinc {qstring-first-index-of} {string view} {str}
4446
4447 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
4448
4449 \include qstring.qdocinc negative-index-start-search-from-end
4450
4451 \sa QStringView::indexOf(), lastIndexOf(), contains(), count()
4452*/
4453
4454/*!
4455 \fn qsizetype QString::indexOf(QLatin1StringView str, qsizetype from, Qt::CaseSensitivity cs) const
4456 \since 4.5
4457
4458 \include {qstring.qdocinc} {qstring-first-index-of} {Latin-1 string viewed by} {str}
4459
4460 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
4461
4462 Example:
4463
4464 \snippet qstring/main.cpp 24
4465
4466 \include qstring.qdocinc negative-index-start-search-from-end
4467
4468 \sa lastIndexOf(), contains(), count()
4469*/
4470
4471/*!
4472 \fn qsizetype QString::indexOf(QChar ch, qsizetype from, Qt::CaseSensitivity cs) const
4473 \overload indexOf()
4474
4475 \include qstring.qdocinc {qstring-first-index-of} {character} {ch}
4476*/
4477
4478/*!
4479 \fn qsizetype QString::lastIndexOf(const QString &str, qsizetype from, Qt::CaseSensitivity cs) const
4480 \include qstring.qdocinc {qstring-last-index-of} {string} {str}
4481
4482 \include qstring.qdocinc negative-index-start-search-from-end
4483
4484 Returns -1 if \a str is not found.
4485
4486 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
4487
4488 Example:
4489
4490 \snippet qstring/main.cpp 29
4491
4492 \note When searching for a 0-length \a str, the match at the end of
4493 the data is excluded from the search by a negative \a from, even
4494 though \c{-1} is normally thought of as searching from the end of the
4495 string: the match at the end is \e after the last character, so it is
4496 excluded. To include such a final empty match, either give a positive
4497 value for \a from or omit the \a from parameter entirely.
4498
4499 \sa indexOf(), contains(), count()
4500*/
4501
4502/*!
4503 \fn qsizetype QString::lastIndexOf(const QString &str, Qt::CaseSensitivity cs = Qt::CaseSensitive) const
4504 \since 6.2
4505 \overload lastIndexOf()
4506
4507 Returns the index position of the last occurrence of the string \a
4508 str in this string. Returns -1 if \a str is not found.
4509
4510 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
4511
4512 Example:
4513
4514 \snippet qstring/main.cpp 29
4515
4516 \sa indexOf(), contains(), count()
4517*/
4518
4519
4520/*!
4521 \fn qsizetype QString::lastIndexOf(QLatin1StringView str, qsizetype from, Qt::CaseSensitivity cs) const
4522 \since 4.5
4523 \overload lastIndexOf()
4524
4525 \include qstring.qdocinc {qstring-last-index-of} {Latin-1 string viewed by} {str}
4526
4527 \include qstring.qdocinc negative-index-start-search-from-end
4528
4529 Returns -1 if \a str is not found.
4530
4531 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
4532
4533 Example:
4534
4535 \snippet qstring/main.cpp 29
4536
4537 \note When searching for a 0-length \a str, the match at the end of
4538 the data is excluded from the search by a negative \a from, even
4539 though \c{-1} is normally thought of as searching from the end of the
4540 string: the match at the end is \e after the last character, so it is
4541 excluded. To include such a final empty match, either give a positive
4542 value for \a from or omit the \a from parameter entirely.
4543
4544 \sa indexOf(), contains(), count()
4545*/
4546
4547/*!
4548 \fn qsizetype QString::lastIndexOf(QLatin1StringView str, Qt::CaseSensitivity cs = Qt::CaseSensitive) const
4549 \since 6.2
4550 \overload lastIndexOf()
4551
4552 Returns the index position of the last occurrence of the string \a
4553 str in this string. Returns -1 if \a str is not found.
4554
4555 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
4556
4557 Example:
4558
4559 \snippet qstring/main.cpp 29
4560
4561 \sa indexOf(), contains(), count()
4562*/
4563
4564/*!
4565 \fn qsizetype QString::lastIndexOf(QChar ch, qsizetype from, Qt::CaseSensitivity cs) const
4566 \overload lastIndexOf()
4567
4568 \include qstring.qdocinc {qstring-last-index-of} {character} {ch}
4569*/
4570
4571/*!
4572 \fn QString::lastIndexOf(QChar ch, Qt::CaseSensitivity) const
4573 \since 6.3
4574 \overload lastIndexOf()
4575*/
4576
4577/*!
4578 \fn qsizetype QString::lastIndexOf(QStringView str, qsizetype from, Qt::CaseSensitivity cs) const
4579 \since 5.14
4580 \overload lastIndexOf()
4581
4582 \include qstring.qdocinc {qstring-last-index-of} {string view} {str}
4583
4584 \include qstring.qdocinc negative-index-start-search-from-end
4585
4586 Returns -1 if \a str is not found.
4587
4588 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
4589
4590 \note When searching for a 0-length \a str, the match at the end of
4591 the data is excluded from the search by a negative \a from, even
4592 though \c{-1} is normally thought of as searching from the end of the
4593 string: the match at the end is \e after the last character, so it is
4594 excluded. To include such a final empty match, either give a positive
4595 value for \a from or omit the \a from parameter entirely.
4596
4597 \sa indexOf(), contains(), count()
4598*/
4599
4600/*!
4601 \fn qsizetype QString::lastIndexOf(QStringView str, Qt::CaseSensitivity cs = Qt::CaseSensitive) const
4602 \since 6.2
4603 \overload lastIndexOf()
4604
4605 Returns the index position of the last occurrence of the string view \a
4606 str in this string. Returns -1 if \a str is not found.
4607
4608 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
4609
4610 \sa indexOf(), contains(), count()
4611*/
4612
4613#if QT_CONFIG(regularexpression)
4614struct QStringCapture
4615{
4616 qsizetype pos;
4617 qsizetype len;
4618 int no;
4619};
4620Q_DECLARE_TYPEINFO(QStringCapture, Q_PRIMITIVE_TYPE);
4621
4622/*!
4623 \overload replace()
4624 \since 5.0
4625
4626 Replaces every occurrence of the regular expression \a re in the
4627 string with \a after. Returns a reference to the string. For
4628 example:
4629
4630 \snippet qstring/main.cpp 87
4631
4632 For regular expressions containing capturing groups,
4633 occurrences of \b{\\1}, \b{\\2}, ..., in \a after are replaced
4634 with the string captured by the corresponding capturing group.
4635
4636 \snippet qstring/main.cpp 88
4637
4638 \sa indexOf(), lastIndexOf(), remove(), QRegularExpression, QRegularExpressionMatch
4639*/
4640QString &QString::replace(const QRegularExpression &re, const QString &after)
4641{
4642 if (!re.isValid()) {
4643 qtWarnAboutInvalidRegularExpression(re, "QString", "replace");
4644 return *this;
4645 }
4646
4647 const QString copy(*this);
4648 QRegularExpressionMatchIterator iterator = re.globalMatch(copy);
4649 if (!iterator.hasNext()) // no matches at all
4650 return *this;
4651
4652 reallocData(d.size, QArrayData::KeepSize);
4653
4654 qsizetype numCaptures = re.captureCount();
4655
4656 // 1. build the backreferences list, holding where the backreferences
4657 // are in the replacement string
4658 QVarLengthArray<QStringCapture> backReferences;
4659 const qsizetype al = after.size();
4660 const QChar *ac = after.unicode();
4661
4662 for (qsizetype i = 0; i < al - 1; i++) {
4663 if (ac[i] == u'\\') {
4664 int no = ac[i + 1].digitValue();
4665 if (no > 0 && no <= numCaptures) {
4666 QStringCapture backReference;
4667 backReference.pos = i;
4668 backReference.len = 2;
4669
4670 if (i < al - 2) {
4671 int secondDigit = ac[i + 2].digitValue();
4672 if (secondDigit != -1 && ((no * 10) + secondDigit) <= numCaptures) {
4673 no = (no * 10) + secondDigit;
4674 ++backReference.len;
4675 }
4676 }
4677
4678 backReference.no = no;
4679 backReferences.append(backReference);
4680 }
4681 }
4682 }
4683
4684 // 2. iterate on the matches. For every match, copy in chunks
4685 // - the part before the match
4686 // - the after string, with the proper replacements for the backreferences
4687
4688 qsizetype newLength = 0; // length of the new string, with all the replacements
4689 qsizetype lastEnd = 0;
4690 QVarLengthArray<QStringView> chunks;
4691 const QStringView copyView{ copy }, afterView{ after };
4692 while (iterator.hasNext()) {
4693 QRegularExpressionMatch match = iterator.next();
4694 qsizetype len;
4695 // add the part before the match
4696 len = match.capturedStart() - lastEnd;
4697 if (len > 0) {
4698 chunks << copyView.mid(lastEnd, len);
4699 newLength += len;
4700 }
4701
4702 lastEnd = 0;
4703 // add the after string, with replacements for the backreferences
4704 for (const QStringCapture &backReference : std::as_const(backReferences)) {
4705 // part of "after" before the backreference
4706 len = backReference.pos - lastEnd;
4707 if (len > 0) {
4708 chunks << afterView.mid(lastEnd, len);
4709 newLength += len;
4710 }
4711
4712 // backreference itself
4713 len = match.capturedLength(backReference.no);
4714 if (len > 0) {
4715 chunks << copyView.mid(match.capturedStart(backReference.no), len);
4716 newLength += len;
4717 }
4718
4719 lastEnd = backReference.pos + backReference.len;
4720 }
4721
4722 // add the last part of the after string
4723 len = afterView.size() - lastEnd;
4724 if (len > 0) {
4725 chunks << afterView.mid(lastEnd, len);
4726 newLength += len;
4727 }
4728
4729 lastEnd = match.capturedEnd();
4730 }
4731
4732 // 3. trailing string after the last match
4733 if (copyView.size() > lastEnd) {
4734 chunks << copyView.mid(lastEnd);
4735 newLength += copyView.size() - lastEnd;
4736 }
4737
4738 // 4. assemble the chunks together
4739 resize(newLength);
4740 qsizetype i = 0;
4741 QChar *uc = data();
4742 for (const QStringView &chunk : std::as_const(chunks)) {
4743 qsizetype len = chunk.size();
4744 memcpy(uc + i, chunk.constData(), len * sizeof(QChar));
4745 i += len;
4746 }
4747
4748 return *this;
4749}
4750#endif // QT_CONFIG(regularexpression)
4751
4752/*!
4753 Returns the number of (potentially overlapping) occurrences of
4754 the string \a str in this string.
4755
4756 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
4757
4758 \sa contains(), indexOf()
4759*/
4760
4761qsizetype QString::count(const QString &str, Qt::CaseSensitivity cs) const
4762{
4763 return QtPrivate::count(QStringView(unicode(), size()), QStringView(str.unicode(), str.size()), cs);
4764}
4765
4766/*!
4767 \overload count()
4768
4769 Returns the number of occurrences of character \a ch in the string.
4770
4771 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
4772
4773 \sa contains(), indexOf()
4774*/
4775
4776qsizetype QString::count(QChar ch, Qt::CaseSensitivity cs) const
4777{
4778 return QtPrivate::count(QStringView(unicode(), size()), ch, cs);
4779}
4780
4781/*!
4782 \since 6.0
4783 \overload count()
4784 Returns the number of (potentially overlapping) occurrences of the
4785 string view \a str in this string.
4786
4787 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
4788
4789 \sa contains(), indexOf()
4790*/
4791qsizetype QString::count(QStringView str, Qt::CaseSensitivity cs) const
4792{
4793 return QtPrivate::count(*this, str, cs);
4794}
4795
4796/*! \fn bool QString::contains(const QString &str, Qt::CaseSensitivity cs = Qt::CaseSensitive) const
4797
4798 Returns \c true if this string contains an occurrence of the string
4799 \a str; otherwise returns \c false.
4800
4801 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
4802
4803 Example:
4804 \snippet qstring/main.cpp 17
4805
4806 \sa indexOf(), count()
4807*/
4808
4809/*! \fn bool QString::contains(QLatin1StringView str, Qt::CaseSensitivity cs = Qt::CaseSensitive) const
4810 \since 5.3
4811
4812 \overload contains()
4813
4814 Returns \c true if this string contains an occurrence of the latin-1 string
4815 \a str; otherwise returns \c false.
4816*/
4817
4818/*! \fn bool QString::contains(QChar ch, Qt::CaseSensitivity cs = Qt::CaseSensitive) const
4819
4820 \overload contains()
4821
4822 Returns \c true if this string contains an occurrence of the
4823 character \a ch; otherwise returns \c false.
4824*/
4825
4826/*! \fn bool QString::contains(QStringView str, Qt::CaseSensitivity cs = Qt::CaseSensitive) const
4827 \since 5.14
4828 \overload contains()
4829
4830 Returns \c true if this string contains an occurrence of the string view
4831 \a str; otherwise returns \c false.
4832
4833 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
4834
4835 \sa indexOf(), count()
4836*/
4837
4838#if QT_CONFIG(regularexpression)
4839/*!
4840 \since 5.5
4841
4842 Returns the index position of the first match of the regular
4843 expression \a re in the string, searching forward from index
4844 position \a from. Returns -1 if \a re didn't match anywhere.
4845
4846 If the match is successful and \a rmatch is not \nullptr, it also
4847 writes the results of the match into the QRegularExpressionMatch object
4848 pointed to by \a rmatch.
4849
4850 Example:
4851
4852 \snippet qstring/main.cpp 93
4853*/
4854qsizetype QString::indexOf(const QRegularExpression &re, qsizetype from, QRegularExpressionMatch *rmatch) const
4855{
4856 return QtPrivate::indexOf(QStringView(*this), this, re, from, rmatch);
4857}
4858
4859/*!
4860 \since 5.5
4861
4862 Returns the index position of the last match of the regular
4863 expression \a re in the string, which starts before the index
4864 position \a from.
4865
4866 \include qstring.qdocinc negative-index-start-search-from-end
4867
4868 Returns -1 if \a re didn't match anywhere.
4869
4870 If the match is successful and \a rmatch is not \nullptr, it also
4871 writes the results of the match into the QRegularExpressionMatch object
4872 pointed to by \a rmatch.
4873
4874 Example:
4875
4876 \snippet qstring/main.cpp 94
4877
4878 \note Due to how the regular expression matching algorithm works,
4879 this function will actually match repeatedly from the beginning of
4880 the string until the position \a from is reached.
4881
4882 \note When searching for a regular expression \a re that may match
4883 0 characters, the match at the end of the data is excluded from the
4884 search by a negative \a from, even though \c{-1} is normally
4885 thought of as searching from the end of the string: the match at
4886 the end is \e after the last character, so it is excluded. To
4887 include such a final empty match, either give a positive value for
4888 \a from or omit the \a from parameter entirely.
4889*/
4890qsizetype QString::lastIndexOf(const QRegularExpression &re, qsizetype from, QRegularExpressionMatch *rmatch) const
4891{
4892 return QtPrivate::lastIndexOf(QStringView(*this), this, re, from, rmatch);
4893}
4894
4895/*!
4896 \fn qsizetype QString::lastIndexOf(const QRegularExpression &re, QRegularExpressionMatch *rmatch = nullptr) const
4897 \since 6.2
4898 \overload lastIndexOf()
4899
4900 Returns the index position of the last match of the regular
4901 expression \a re in the string. Returns -1 if \a re didn't match anywhere.
4902
4903 If the match is successful and \a rmatch is not \nullptr, it also
4904 writes the results of the match into the QRegularExpressionMatch object
4905 pointed to by \a rmatch.
4906
4907 Example:
4908
4909 \snippet qstring/main.cpp 94
4910
4911 \note Due to how the regular expression matching algorithm works,
4912 this function will actually match repeatedly from the beginning of
4913 the string until the end of the string is reached.
4914*/
4915
4916/*!
4917 \since 5.1
4918
4919 Returns \c true if the regular expression \a re matches somewhere in this
4920 string; otherwise returns \c false.
4921
4922 If the match is successful and \a rmatch is not \nullptr, it also
4923 writes the results of the match into the QRegularExpressionMatch object
4924 pointed to by \a rmatch.
4925
4926 \sa QRegularExpression::match()
4927*/
4928
4929bool QString::contains(const QRegularExpression &re, QRegularExpressionMatch *rmatch) const
4930{
4931 return QtPrivate::contains(QStringView(*this), this, re, rmatch);
4932}
4933
4934/*!
4935 \overload count()
4936 \since 5.0
4937
4938 Returns the number of times the regular expression \a re matches
4939 in the string.
4940
4941 For historical reasons, this function counts overlapping matches,
4942 so in the example below, there are four instances of "ana" or
4943 "ama":
4944
4945 \snippet qstring/main.cpp 95
4946
4947 This behavior is different from simply iterating over the matches
4948 in the string using QRegularExpressionMatchIterator.
4949
4950 \sa QRegularExpression::globalMatch()
4951*/
4952qsizetype QString::count(const QRegularExpression &re) const
4953{
4954 return QtPrivate::count(QStringView(*this), re);
4955}
4956#endif // QT_CONFIG(regularexpression)
4957
4958#if QT_DEPRECATED_SINCE(6, 4)
4959/*! \fn qsizetype QString::count() const
4960 \deprecated [6.4] Use size() or length() instead.
4961 \overload count()
4962
4963 Same as size().
4964*/
4965#endif
4966
4967/*!
4968 \enum QString::SectionFlag
4969
4970 This enum specifies flags that can be used to affect various
4971 aspects of the section() function's behavior with respect to
4972 separators and empty fields.
4973
4974 \value SectionDefault Empty fields are counted, leading and
4975 trailing separators are not included, and the separator is
4976 compared case sensitively.
4977
4978 \value SectionSkipEmpty Treat empty fields as if they don't exist,
4979 i.e. they are not considered as far as \e start and \e end are
4980 concerned.
4981
4982 \value SectionIncludeLeadingSep Include the leading separator (if
4983 any) in the result string.
4984
4985 \value SectionIncludeTrailingSep Include the trailing separator
4986 (if any) in the result string.
4987
4988 \value SectionCaseInsensitiveSeps Compare the separator
4989 case-insensitively.
4990
4991 \sa section()
4992*/
4993
4994/*!
4995 \fn QString QString::section(QChar sep, qsizetype start, qsizetype end = -1, SectionFlags flags) const
4996
4997 This function returns a section of the string.
4998
4999 This string is treated as a sequence of fields separated by the
5000 character, \a sep. The returned string consists of the fields from
5001 position \a start to position \a end inclusive. If \a end is not
5002 specified, all fields from position \a start to the end of the
5003 string are included. Fields are numbered 0, 1, 2, etc., counting
5004 from the left, and -1, -2, etc., counting from right to left.
5005
5006 The \a flags argument can be used to affect some aspects of the
5007 function's behavior, e.g. whether to be case sensitive, whether
5008 to skip empty fields and how to deal with leading and trailing
5009 separators; see \l{SectionFlags}.
5010
5011 \snippet qstring/main.cpp 52
5012
5013 If \a start or \a end is negative, we count fields from the right
5014 of the string, the right-most field being -1, the one from
5015 right-most field being -2, and so on.
5016
5017 \snippet qstring/main.cpp 53
5018
5019 \sa split()
5020*/
5021
5022/*!
5023 \overload section()
5024
5025 \snippet qstring/main.cpp 51
5026 \snippet qstring/main.cpp 54
5027
5028 \sa split()
5029*/
5030
5031QString QString::section(const QString &sep, qsizetype start, qsizetype end, SectionFlags flags) const
5032{
5033 const QList<QStringView> sections = QStringView{ *this }.split(
5034 sep, Qt::KeepEmptyParts, (flags & SectionCaseInsensitiveSeps) ? Qt::CaseInsensitive : Qt::CaseSensitive);
5035 const qsizetype sectionsSize = sections.size();
5036 if (!(flags & SectionSkipEmpty)) {
5037 if (start < 0)
5038 start += sectionsSize;
5039 if (end < 0)
5040 end += sectionsSize;
5041 } else {
5042 qsizetype skip = 0;
5043 for (qsizetype k = 0; k < sectionsSize; ++k) {
5044 if (sections.at(k).isEmpty())
5045 skip++;
5046 }
5047 if (start < 0)
5048 start += sectionsSize - skip;
5049 if (end < 0)
5050 end += sectionsSize - skip;
5051 }
5052 if (start >= sectionsSize || end < 0 || start > end)
5053 return QString();
5054
5055 QString ret;
5056 qsizetype first_i = start, last_i = end;
5057 for (qsizetype x = 0, i = 0; x <= end && i < sectionsSize; ++i) {
5058 const QStringView &section = sections.at(i);
5059 const bool empty = section.isEmpty();
5060 if (x >= start) {
5061 if (x == start)
5062 first_i = i;
5063 if (x == end)
5064 last_i = i;
5065 if (x > start && i > 0)
5066 ret += sep;
5067 ret += section;
5068 }
5069 if (!empty || !(flags & SectionSkipEmpty))
5070 x++;
5071 }
5072 if ((flags & SectionIncludeLeadingSep) && first_i > 0)
5073 ret.prepend(sep);
5074 if ((flags & SectionIncludeTrailingSep) && last_i < sectionsSize - 1)
5075 ret += sep;
5076 return ret;
5077}
5078
5079#if QT_CONFIG(regularexpression)
5080struct qt_section_chunk
5081{
5082 qsizetype length;
5083 QStringView string;
5084};
5085Q_DECLARE_TYPEINFO(qt_section_chunk, Q_RELOCATABLE_TYPE);
5086
5087static QString extractSections(QSpan<qt_section_chunk> sections, qsizetype start, qsizetype end,
5088 QString::SectionFlags flags)
5089{
5090 const qsizetype sectionsSize = sections.size();
5091
5092 if (!(flags & QString::SectionSkipEmpty)) {
5093 if (start < 0)
5094 start += sectionsSize;
5095 if (end < 0)
5096 end += sectionsSize;
5097 } else {
5098 qsizetype skip = 0;
5099 for (qsizetype k = 0; k < sectionsSize; ++k) {
5100 const qt_section_chunk &section = sections[k];
5101 if (section.length == section.string.size())
5102 skip++;
5103 }
5104 if (start < 0)
5105 start += sectionsSize - skip;
5106 if (end < 0)
5107 end += sectionsSize - skip;
5108 }
5109 if (start >= sectionsSize || end < 0 || start > end)
5110 return QString();
5111
5112 QString ret;
5113 qsizetype x = 0;
5114 qsizetype first_i = start, last_i = end;
5115 for (qsizetype i = 0; x <= end && i < sectionsSize; ++i) {
5116 const qt_section_chunk &section = sections[i];
5117 const bool empty = (section.length == section.string.size());
5118 if (x >= start) {
5119 if (x == start)
5120 first_i = i;
5121 if (x == end)
5122 last_i = i;
5123 if (x != start)
5124 ret += section.string;
5125 else
5126 ret += section.string.mid(section.length);
5127 }
5128 if (!empty || !(flags & QString::SectionSkipEmpty))
5129 x++;
5130 }
5131
5132 if ((flags & QString::SectionIncludeLeadingSep) && first_i >= 0) {
5133 const qt_section_chunk &section = sections[first_i];
5134 ret.prepend(section.string.left(section.length));
5135 }
5136
5137 if ((flags & QString::SectionIncludeTrailingSep)
5138 && last_i < sectionsSize - 1) {
5139 const qt_section_chunk &section = sections[last_i + 1];
5140 ret += section.string.left(section.length);
5141 }
5142
5143 return ret;
5144}
5145
5146/*!
5147 \overload section()
5148 \since 5.0
5149
5150 This string is treated as a sequence of fields separated by the
5151 regular expression, \a re.
5152
5153 \snippet qstring/main.cpp 89
5154
5155 \warning Using this QRegularExpression version is much more expensive than
5156 the overloaded string and character versions.
5157
5158 \sa split(), simplified()
5159*/
5160QString QString::section(const QRegularExpression &re, qsizetype start, qsizetype end, SectionFlags flags) const
5161{
5162 if (!re.isValid()) {
5163 qtWarnAboutInvalidRegularExpression(re, "QString", "section");
5164 return QString();
5165 }
5166
5167 const QChar *uc = unicode();
5168 if (!uc)
5169 return QString();
5170
5171 QRegularExpression sep(re);
5172 if (flags & SectionCaseInsensitiveSeps)
5173 sep.setPatternOptions(sep.patternOptions() | QRegularExpression::CaseInsensitiveOption);
5174
5175 QVarLengthArray<qt_section_chunk> sections;
5176 qsizetype n = size(), m = 0, last_m = 0, last_len = 0;
5177 QRegularExpressionMatchIterator iterator = sep.globalMatch(*this);
5178 while (iterator.hasNext()) {
5179 QRegularExpressionMatch match = iterator.next();
5180 m = match.capturedStart();
5181 sections.append(qt_section_chunk{last_len, QStringView{*this}.sliced(last_m, m - last_m)});
5182 last_m = m;
5183 last_len = match.capturedLength();
5184 }
5185 sections.append(qt_section_chunk{last_len, QStringView{*this}.sliced(last_m, n - last_m)});
5186
5187 return extractSections(sections, start, end, flags);
5188}
5189#endif // QT_CONFIG(regularexpression)
5190
5191/*!
5192 \fn QString QString::left(qsizetype n) const &
5193 \fn QString QString::left(qsizetype n) &&
5194
5195 Returns a substring that contains the \a n leftmost characters of
5196 this string (that is, from the beginning of this string up to, but not
5197 including, the element at index position \a n).
5198
5199 If you know that \a n cannot be out of bounds, use first() instead in new
5200 code, because it is faster.
5201
5202 The entire string is returned if \a n is greater than or equal
5203 to size(), or less than zero.
5204
5205 \sa first(), last(), startsWith(), chopped(), chop(), truncate()
5206*/
5207
5208/*!
5209 \fn QString QString::right(qsizetype n) const &
5210 \fn QString QString::right(qsizetype n) &&
5211
5212 Returns a substring that contains the \a n rightmost characters
5213 of the string.
5214
5215 If you know that \a n cannot be out of bounds, use last() instead in new
5216 code, because it is faster.
5217
5218 The entire string is returned if \a n is greater than or equal
5219 to size(), or less than zero.
5220
5221 \sa endsWith(), last(), first(), sliced(), chopped(), chop(), truncate(), slice()
5222*/
5223
5224/*!
5225 \fn QString QString::mid(qsizetype position, qsizetype n) const &
5226 \fn QString QString::mid(qsizetype position, qsizetype n) &&
5227
5228 Returns a string that contains \a n characters of this string, starting
5229 at the specified \a position index up to, but not including, the element
5230 at index position \tt {\a position + \a n}.
5231
5232 If you know that \a position and \a n cannot be out of bounds, use sliced()
5233 instead in new code, because it is faster.
5234
5235 Returns a null string if the \a position index exceeds the
5236 length of the string. If there are less than \a n characters
5237 available in the string starting at the given \a position, or if
5238 \a n is -1 (default), the function returns all characters that
5239 are available from the specified \a position.
5240
5241 \sa first(), last(), sliced(), chopped(), chop(), truncate(), slice()
5242*/
5243QString QString::mid(qsizetype position, qsizetype n) const &
5244{
5245 qsizetype p = position;
5246 qsizetype l = n;
5247 using namespace QtPrivate;
5248 switch (QContainerImplHelper::mid(size(), &p, &l)) {
5249 case QContainerImplHelper::Null:
5250 return QString();
5251 case QContainerImplHelper::Empty:
5252 return QString(DataPointer::fromRawData(&_empty, 0));
5253 case QContainerImplHelper::Full:
5254 return *this;
5255 case QContainerImplHelper::Subset:
5256 return sliced(p, l);
5257 }
5258 Q_UNREACHABLE_RETURN(QString());
5259}
5260
5261QString QString::mid(qsizetype position, qsizetype n) &&
5262{
5263 qsizetype p = position;
5264 qsizetype l = n;
5265 using namespace QtPrivate;
5266 switch (QContainerImplHelper::mid(size(), &p, &l)) {
5267 case QContainerImplHelper::Null:
5268 return QString();
5269 case QContainerImplHelper::Empty:
5270 resize(0); // keep capacity if we've reserve()d
5271 [[fallthrough]];
5272 case QContainerImplHelper::Full:
5273 return std::move(*this);
5274 case QContainerImplHelper::Subset:
5275 return std::move(*this).sliced(p, l);
5276 }
5277 Q_UNREACHABLE_RETURN(QString());
5278}
5279
5280/*!
5281 \fn QString QString::first(qsizetype n) const &
5282 \fn QString QString::first(qsizetype n) &&
5283 \since 6.0
5284
5285 Returns a string that contains the first \a n characters of this string,
5286 (that is, from the beginning of this string up to, but not including,
5287 the element at index position \a n).
5288
5289 \note The behavior is undefined when \a n < 0 or \a n > size().
5290
5291 \snippet qstring/main.cpp 31
5292
5293 \sa last(), sliced(), startsWith(), chopped(), chop(), truncate(), slice()
5294*/
5295
5296/*!
5297 \fn QString QString::last(qsizetype n) const &
5298 \fn QString QString::last(qsizetype n) &&
5299 \since 6.0
5300
5301 Returns the string that contains the last \a n characters of this string.
5302
5303 \note The behavior is undefined when \a n < 0 or \a n > size().
5304
5305 \snippet qstring/main.cpp 48
5306
5307 \sa first(), sliced(), endsWith(), chopped(), chop(), truncate(), slice()
5308*/
5309
5310/*!
5311 \fn QString QString::sliced(qsizetype pos, qsizetype n) const &
5312 \fn QString QString::sliced(qsizetype pos, qsizetype n) &&
5313 \since 6.0
5314
5315 Returns a string that contains \a n characters of this string, starting
5316 at position \a pos up to, but not including, the element at index position
5317 \tt {\a pos + \a n}.
5318
5319 \note The behavior is undefined when \a pos < 0, \a n < 0,
5320 or \a pos + \a n > size().
5321
5322 \snippet qstring/main.cpp 34
5323
5324 \sa first(), last(), chopped(), chop(), truncate(), slice()
5325*/
5326QString QString::sliced_helper(QString &str, qsizetype pos, qsizetype n)
5327{
5328 if (n == 0)
5329 return QString(DataPointer::fromRawData(&_empty, 0));
5330 DataPointer d = std::move(str.d).sliced(pos, n);
5331 d.data()[n] = 0;
5332 return QString(std::move(d));
5333}
5334
5335/*!
5336 \fn QString QString::sliced(qsizetype pos) const &
5337 \fn QString QString::sliced(qsizetype pos) &&
5338 \since 6.0
5339 \overload
5340
5341 Returns a string that contains the portion of this string starting at
5342 position \a pos and extending to its end.
5343
5344 \note The behavior is undefined when \a pos < 0 or \a pos > size().
5345
5346 \sa first(), last(), chopped(), chop(), truncate(), slice()
5347*/
5348
5349/*!
5350 \fn QString &QString::slice(qsizetype pos, qsizetype n)
5351 \since 6.8
5352
5353 Modifies this string to start at position \a pos, up to, but not including,
5354 the character (code point) at index position \tt {\a pos + \a n}; and
5355 returns a reference to this string.
5356
5357 \note The behavior is undefined if \a pos < 0, \a n < 0,
5358 or \a pos + \a n > size().
5359
5360 \snippet qstring/main.cpp slice97
5361
5362 \sa sliced(), first(), last(), chopped(), chop(), truncate()
5363*/
5364
5365/*!
5366 \fn QString &QString::slice(qsizetype pos)
5367 \since 6.8
5368 \overload
5369
5370 Modifies this string to start at position \a pos and extending to its end,
5371 and returns a reference to this string.
5372
5373 \note The behavior is undefined if \a pos < 0 or \a pos > size().
5374
5375 \sa sliced(), first(), last(), chopped(), chop(), truncate()
5376*/
5377
5378/*!
5379 \fn QString QString::chopped(qsizetype len) const &
5380 \fn QString QString::chopped(qsizetype len) &&
5381 \since 5.10
5382
5383 Returns a string that contains the size() - \a len leftmost characters
5384 of this string.
5385
5386 \note The behavior is undefined if \a len is negative or greater than size().
5387
5388 \sa endsWith(), first(), last(), sliced(), chop(), truncate(), slice()
5389*/
5390
5391/*!
5392 Returns \c true if the string starts with \a s; otherwise returns
5393 \c false.
5394
5395 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
5396
5397 \snippet qstring/main.cpp 65
5398
5399 \sa endsWith()
5400*/
5401bool QString::startsWith(const QString& s, Qt::CaseSensitivity cs) const
5402{
5403 return qt_starts_with_impl(QStringView(*this), QStringView(s), cs);
5404}
5405
5406/*!
5407 \overload startsWith()
5408 */
5409bool QString::startsWith(QLatin1StringView s, Qt::CaseSensitivity cs) const
5410{
5411 return qt_starts_with_impl(QStringView(*this), s, cs);
5412}
5413
5414/*!
5415 \overload startsWith()
5416
5417 Returns \c true if the string starts with \a c; otherwise returns
5418 \c false.
5419*/
5420bool QString::startsWith(QChar c, Qt::CaseSensitivity cs) const
5421{
5422 if (!size())
5423 return false;
5424 if (cs == Qt::CaseSensitive)
5425 return at(0) == c;
5426 return foldCase(at(0)) == foldCase(c);
5427}
5428
5429/*!
5430 \fn bool QString::startsWith(QStringView str, Qt::CaseSensitivity cs) const
5431 \since 5.10
5432 \overload
5433
5434 Returns \c true if the string starts with the string view \a str;
5435 otherwise returns \c false.
5436
5437 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
5438
5439 \sa endsWith()
5440*/
5441
5442/*!
5443 Returns \c true if the string ends with \a s; otherwise returns
5444 \c false.
5445
5446 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
5447
5448 \snippet qstring/main.cpp 20
5449
5450 \sa startsWith()
5451*/
5452bool QString::endsWith(const QString &s, Qt::CaseSensitivity cs) const
5453{
5454 return qt_ends_with_impl(QStringView(*this), QStringView(s), cs);
5455}
5456
5457/*!
5458 \fn bool QString::endsWith(QStringView str, Qt::CaseSensitivity cs) const
5459 \since 5.10
5460 \overload endsWith()
5461 Returns \c true if the string ends with the string view \a str;
5462 otherwise returns \c false.
5463
5464 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
5465
5466 \sa startsWith()
5467*/
5468
5469/*!
5470 \overload endsWith()
5471*/
5472bool QString::endsWith(QLatin1StringView s, Qt::CaseSensitivity cs) const
5473{
5474 return qt_ends_with_impl(QStringView(*this), s, cs);
5475}
5476
5477/*!
5478 Returns \c true if the string ends with \a c; otherwise returns
5479 \c false.
5480
5481 \overload endsWith()
5482 */
5483bool QString::endsWith(QChar c, Qt::CaseSensitivity cs) const
5484{
5485 if (!size())
5486 return false;
5487 if (cs == Qt::CaseSensitive)
5488 return at(size() - 1) == c;
5489 return foldCase(at(size() - 1)) == foldCase(c);
5490}
5491
5492static bool checkCase(QStringView s, QUnicodeTables::Case c) noexcept
5493{
5494 QStringIterator it(s);
5495 while (it.hasNext()) {
5496 const char32_t uc = it.next();
5497 if (caseConversion(uc)[c].diff)
5498 return false;
5499 }
5500 return true;
5501}
5502
5503bool QtPrivate::isLower(QStringView s) noexcept
5504{
5505 return checkCase(s, QUnicodeTables::LowerCase);
5506}
5507
5508bool QtPrivate::isUpper(QStringView s) noexcept
5509{
5510 return checkCase(s, QUnicodeTables::UpperCase);
5511}
5512
5513/*!
5514 Returns \c true if the string is uppercase, that is, it's identical
5515 to its toUpper() folding.
5516
5517 Note that this does \e not mean that the string does not contain
5518 lowercase letters (some lowercase letters do not have a uppercase
5519 folding; they are left unchanged by toUpper()).
5520 For more information, refer to the Unicode standard, section 3.13.
5521
5522 \since 5.12
5523
5524 \sa QChar::toUpper(), isLower()
5525*/
5526bool QString::isUpper() const
5527{
5528 return QtPrivate::isUpper(qToStringViewIgnoringNull(*this));
5529}
5530
5531/*!
5532 Returns \c true if the string is lowercase, that is, it's identical
5533 to its toLower() folding.
5534
5535 Note that this does \e not mean that the string does not contain
5536 uppercase letters (some uppercase letters do not have a lowercase
5537 folding; they are left unchanged by toLower()).
5538 For more information, refer to the Unicode standard, section 3.13.
5539
5540 \since 5.12
5541
5542 \sa QChar::toLower(), isUpper()
5543 */
5544bool QString::isLower() const
5545{
5546 return QtPrivate::isLower(qToStringViewIgnoringNull(*this));
5547}
5548
5549static QByteArray qt_convert_to_latin1(QStringView string);
5550
5551QByteArray QString::toLatin1_helper(const QString &string)
5552{
5553 return qt_convert_to_latin1(string);
5554}
5555
5556/*!
5557 \since 6.0
5558 \internal
5559 \relates QAnyStringView
5560
5561 Returns a UTF-16 representation of \a string as a QString.
5562
5563 \sa QString::toLatin1(), QStringView::toLatin1(), QtPrivate::convertToUtf8(),
5564 QtPrivate::convertToLocal8Bit(), QtPrivate::convertToUcs4()
5565*/
5566QString QtPrivate::convertToQString(QAnyStringView string)
5567{
5568 return string.visit([] (auto string) { return string.toString(); });
5569}
5570
5571/*!
5572 \since 5.10
5573 \internal
5574 \relates QStringView
5575
5576 Returns a Latin-1 representation of \a string as a QByteArray.
5577
5578 The behavior is undefined if \a string contains non-Latin1 characters.
5579
5580 \sa QString::toLatin1(), QStringView::toLatin1(), QtPrivate::convertToUtf8(),
5581 QtPrivate::convertToLocal8Bit(), QtPrivate::convertToUcs4()
5582*/
5584{
5585 return qt_convert_to_latin1(string);
5586}
5587
5588Q_NEVER_INLINE
5589static QByteArray qt_convert_to_latin1(QStringView string)
5590{
5591 if (Q_UNLIKELY(string.isNull()))
5592 return QByteArray();
5593
5594 QByteArray ba(string.size(), Qt::Uninitialized);
5595
5596 // since we own the only copy, we're going to const_cast the constData;
5597 // that avoids an unnecessary call to detach() and expansion code that will never get used
5598 qt_to_latin1(reinterpret_cast<uchar *>(const_cast<char *>(ba.constData())),
5599 string.utf16(), string.size());
5600 return ba;
5601}
5602
5603QByteArray QString::toLatin1_helper_inplace(QString &s)
5604{
5605 if (!s.isDetached())
5606 return qt_convert_to_latin1(s);
5607
5608 // We can return our own buffer to the caller.
5609 // Conversion to Latin-1 always shrinks the buffer by half.
5610 // This relies on the fact that we use QArrayData for everything behind the scenes
5611
5612 // First, do the in-place conversion. Since isDetached() == true, the data
5613 // was allocated by QArrayData, so the null terminator must be there.
5614 qsizetype length = s.size();
5615 char16_t *sdata = s.d.data();
5616 Q_ASSERT(sdata[length] == u'\0');
5617 qt_to_latin1(reinterpret_cast<uchar *>(sdata), sdata, length + 1);
5618
5619 // Move the internals over to the byte array.
5620 // Kids, avert your eyes. Don't try this at home.
5621 auto ba_d = std::move(s.d).reinterpreted<char>();
5622
5623 // Some sanity checks
5624 Q_ASSERT(ba_d.d->allocatedCapacity() >= ba_d.size);
5625 Q_ASSERT(s.isNull());
5626 Q_ASSERT(s.isEmpty());
5627 Q_ASSERT(s.constData() == QString().constData());
5628
5629 return QByteArray(std::move(ba_d));
5630}
5631
5632/*!
5633 \since 6.9
5634 \internal
5635 \relates QLatin1StringView
5636
5637 Returns a UTF-8 representation of \a string as a QByteArray.
5638*/
5639QByteArray QtPrivate::convertToUtf8(QLatin1StringView string)
5640{
5641 if (Q_UNLIKELY(string.isNull()))
5642 return QByteArray();
5643
5644 // create a QByteArray with the worst case scenario size
5645 QByteArray ba(string.size() * 2, Qt::Uninitialized);
5646 const qsizetype sz = QUtf8::convertFromLatin1(ba.data(), string) - ba.data();
5647 ba.truncate(sz);
5648
5649 return ba;
5650}
5651
5652// QLatin1 methods that use helpers from qstring.cpp
5653char16_t *QLatin1::convertToUnicode(char16_t *out, QLatin1StringView in) noexcept
5654{
5655 const qsizetype len = in.size();
5656 qt_from_latin1(out, in.data(), len);
5657 return std::next(out, len);
5658}
5659
5660char *QLatin1::convertFromUnicode(char *out, QStringView in) noexcept
5661{
5662 const qsizetype len = in.size();
5663 qt_to_latin1(reinterpret_cast<uchar *>(out), in.utf16(), len);
5664 return out + len;
5665}
5666
5667/*!
5668 \fn QByteArray QString::toLatin1() const
5669
5670 Returns a Latin-1 representation of the string as a QByteArray.
5671
5672 The returned byte array is undefined if the string contains non-Latin1
5673 characters. Those characters may be suppressed or replaced with a
5674 question mark.
5675
5676 \sa fromLatin1(), toUtf8(), toLocal8Bit(), QStringEncoder
5677*/
5678
5679static QByteArray qt_convert_to_local_8bit(QStringView string);
5680
5681/*!
5682 \fn QByteArray QString::toLocal8Bit() const
5683
5684 Returns the local 8-bit representation of the string as a
5685 QByteArray.
5686
5687 \include qstring.qdocinc {qstring-local-8-bit-equivalent} {toUtf8}
5688
5689 If this string contains any characters that cannot be encoded in the
5690 local 8-bit encoding, the returned byte array is undefined. Those
5691 characters may be suppressed or replaced by another.
5692
5693 \sa fromLocal8Bit(), toLatin1(), toUtf8(), QStringEncoder
5694*/
5695
5696QByteArray QString::toLocal8Bit_helper(const QChar *data, qsizetype size)
5697{
5698 return qt_convert_to_local_8bit(QStringView(data, size));
5699}
5700
5701static QByteArray qt_convert_to_local_8bit(QStringView string)
5702{
5703 if (string.isNull())
5704 return QByteArray();
5705 QStringEncoder fromUtf16(QStringEncoder::System, QStringEncoder::Flag::Stateless);
5706 return fromUtf16(string);
5707}
5708
5709/*!
5710 \since 5.10
5711 \internal
5712 \relates QStringView
5713
5714 Returns a local 8-bit representation of \a string as a QByteArray.
5715
5716 On Unix systems this is equivalent to toUtf8(), on Windows the systems
5717 current code page is being used.
5718
5719 The behavior is undefined if \a string contains characters not
5720 supported by the locale's 8-bit encoding.
5721
5722 \sa QString::toLocal8Bit(), QStringView::toLocal8Bit()
5723*/
5725{
5726 return qt_convert_to_local_8bit(string);
5727}
5728
5729static QByteArray qt_convert_to_utf8(QStringView str);
5730
5731/*!
5732 \fn QByteArray QString::toUtf8() const
5733
5734 Returns a UTF-8 representation of the string as a QByteArray.
5735
5736 UTF-8 is a Unicode codec and can represent all characters in a Unicode
5737 string like QString.
5738
5739 \sa fromUtf8(), toLatin1(), toLocal8Bit(), QStringEncoder
5740*/
5741
5742QByteArray QString::toUtf8_helper(const QString &str)
5743{
5744 return qt_convert_to_utf8(str);
5745}
5746
5747static QByteArray qt_convert_to_utf8(QStringView str)
5748{
5749 if (str.isNull())
5750 return QByteArray();
5751
5752 return QUtf8::convertFromUnicode(str);
5753}
5754
5755/*!
5756 \since 5.10
5757 \internal
5758 \relates QStringView
5759
5760 Returns a UTF-8 representation of \a string as a QByteArray.
5761
5762 UTF-8 is a Unicode codec and can represent all characters in a Unicode
5763 string like QStringView.
5764
5765 \sa QString::toUtf8(), QStringView::toUtf8()
5766*/
5768{
5769 return qt_convert_to_utf8(string);
5770}
5771
5772static QList<uint> qt_convert_to_ucs4(QStringView string);
5773
5774/*!
5775 \since 4.2
5776
5777 Returns a UCS-4/UTF-32 representation of the string as a QList<uint>.
5778
5779 UTF-32 is a Unicode codec and therefore it is lossless. All characters from
5780 this string will be encoded in UTF-32. Any invalid sequence of code units in
5781 this string is replaced by the Unicode replacement character
5782 (QChar::ReplacementCharacter, which corresponds to \c{U+FFFD}).
5783
5784 The returned list is not 0-terminated.
5785
5786 \sa fromUtf8(), toUtf8(), toLatin1(), toLocal8Bit(), QStringEncoder,
5787 fromUcs4(), toWCharArray()
5788*/
5789QList<uint> QString::toUcs4() const
5790{
5791 return qt_convert_to_ucs4(*this);
5792}
5793
5794static QList<uint> qt_convert_to_ucs4(QStringView string)
5795{
5796 QList<uint> v(string.size());
5797 uint *a = const_cast<uint*>(v.constData());
5798 QStringIterator it(string);
5799 while (it.hasNext())
5800 *a++ = it.next();
5801 v.resize(a - v.constData());
5802 return v;
5803}
5804
5805/*!
5806 \since 5.10
5807 \internal
5808 \relates QStringView
5809
5810 Returns a UCS-4/UTF-32 representation of \a string as a QList<uint>.
5811
5812 UTF-32 is a Unicode codec and therefore it is lossless. All characters from
5813 this string will be encoded in UTF-32. Any invalid sequence of code units in
5814 this string is replaced by the Unicode replacement character
5815 (QChar::ReplacementCharacter, which corresponds to \c{U+FFFD}).
5816
5817 The returned list is not 0-terminated.
5818
5819 \sa QString::toUcs4(), QStringView::toUcs4(), QtPrivate::convertToLatin1(),
5820 QtPrivate::convertToLocal8Bit(), QtPrivate::convertToUtf8()
5821*/
5822QList<uint> QtPrivate::convertToUcs4(QStringView string)
5823{
5824 return qt_convert_to_ucs4(string);
5825}
5826
5827/*!
5828 \fn QString QString::fromLatin1(QByteArrayView str)
5829 \overload
5830 \since 6.0
5831
5832 Returns a QString initialized with the Latin-1 string \a str.
5833
5834 \note: any null ('\\0') bytes in the byte array will be included in this
5835 string, converted to Unicode null characters (U+0000).
5836*/
5837QString QString::fromLatin1(QByteArrayView ba)
5838{
5839 DataPointer d;
5840 if (!ba.data()) {
5841 // nothing to do
5842 } else if (ba.size() == 0) {
5843 d = DataPointer::fromRawData(&_empty, 0);
5844 } else {
5845 d = DataPointer(ba.size(), ba.size());
5846 Q_CHECK_PTR(d.data());
5847 d.data()[ba.size()] = '\0';
5848 char16_t *dst = d.data();
5849
5850 qt_from_latin1(dst, ba.data(), size_t(ba.size()));
5851 }
5852 return QString(std::move(d));
5853}
5854
5855/*!
5856 \fn QString QString::fromLatin1(const char *str, qsizetype size)
5857 Returns a QString initialized with the first \a size characters
5858 of the Latin-1 string \a str.
5859
5860 If \a size is \c{-1}, \c{strlen(str)} is used instead.
5861
5862 \sa toLatin1(), fromUtf8(), fromLocal8Bit()
5863*/
5864
5865/*!
5866 \fn QString QString::fromLatin1(const QByteArray &str)
5867 \overload
5868 \since 5.0
5869
5870 Returns a QString initialized with the Latin-1 string \a str.
5871
5872 \note: any null ('\\0') bytes in the byte array will be included in this
5873 string, converted to Unicode null characters (U+0000). This behavior is
5874 different from Qt 5.x.
5875*/
5876
5877/*!
5878 \fn QString QString::fromLocal8Bit(const char *str, qsizetype size)
5879 Returns a QString initialized with the first \a size characters
5880 of the 8-bit string \a str.
5881
5882 If \a size is \c{-1}, \c{strlen(str)} is used instead.
5883
5884 \include qstring.qdocinc {qstring-local-8-bit-equivalent} {fromUtf8}
5885
5886 \sa toLocal8Bit(), fromLatin1(), fromUtf8()
5887*/
5888
5889/*!
5890 \fn QString QString::fromLocal8Bit(const QByteArray &str)
5891 \overload
5892 \since 5.0
5893
5894 Returns a QString initialized with the 8-bit string \a str.
5895
5896 \include qstring.qdocinc {qstring-local-8-bit-equivalent} {fromUtf8}
5897
5898 \note: any null ('\\0') bytes in the byte array will be included in this
5899 string, converted to Unicode null characters (U+0000). This behavior is
5900 different from Qt 5.x.
5901*/
5902
5903/*!
5904 \fn QString QString::fromLocal8Bit(QByteArrayView str)
5905 \overload
5906 \since 6.0
5907
5908 Returns a QString initialized with the 8-bit string \a str.
5909
5910 \include qstring.qdocinc {qstring-local-8-bit-equivalent} {fromUtf8}
5911
5912 \note: any null ('\\0') bytes in the byte array will be included in this
5913 string, converted to Unicode null characters (U+0000).
5914*/
5915QString QString::fromLocal8Bit(QByteArrayView ba)
5916{
5917 if (ba.isNull())
5918 return QString();
5919 if (ba.isEmpty())
5920 return QString(DataPointer::fromRawData(&_empty, 0));
5921 QStringDecoder toUtf16(QStringDecoder::System, QStringDecoder::Flag::Stateless);
5922 return toUtf16(ba);
5923}
5924
5925/*! \fn QString QString::fromUtf8(const char *str, qsizetype size)
5926 Returns a QString initialized with the first \a size bytes
5927 of the UTF-8 string \a str.
5928
5929 If \a size is \c{-1}, \c{strlen(str)} is used instead.
5930
5931 UTF-8 is a Unicode codec and can represent all characters in a Unicode
5932 string like QString. However, invalid sequences are possible with UTF-8
5933 and, if any such are found, they will be replaced with one or more
5934 "replacement characters", or suppressed. These include non-Unicode
5935 sequences, non-characters, overlong sequences or surrogate codepoints
5936 encoded into UTF-8.
5937
5938 This function can be used to process incoming data incrementally as long as
5939 all UTF-8 characters are terminated within the incoming data. Any
5940 unterminated characters at the end of the string will be replaced or
5941 suppressed. In order to do stateful decoding, please use \l QStringDecoder.
5942
5943 \sa toUtf8(), fromLatin1(), fromLocal8Bit()
5944*/
5945
5946/*!
5947 \fn QString QString::fromUtf8(const char8_t *str)
5948 \overload
5949 \since 6.1
5950
5951 This overload is only available when compiling in C++20 mode.
5952*/
5953
5954/*!
5955 \fn QString QString::fromUtf8(const char8_t *str, qsizetype size)
5956 \overload
5957 \since 6.0
5958
5959 This overload is only available when compiling in C++20 mode.
5960*/
5961
5962/*!
5963 \fn QString QString::fromUtf8(const QByteArray &str)
5964 \overload
5965 \since 5.0
5966
5967 Returns a QString initialized with the UTF-8 string \a str.
5968
5969 \note: any null ('\\0') bytes in the byte array will be included in this
5970 string, converted to Unicode null characters (U+0000). This behavior is
5971 different from Qt 5.x.
5972*/
5973
5974/*!
5975 \fn QString QString::fromUtf8(QByteArrayView str)
5976 \overload
5977 \since 6.0
5978
5979 Returns a QString initialized with the UTF-8 string \a str.
5980
5981 \note: any null ('\\0') bytes in the byte array will be included in this
5982 string, converted to Unicode null characters (U+0000).
5983*/
5984QString QString::fromUtf8(QByteArrayView ba)
5985{
5986 if (ba.isNull())
5987 return QString();
5988 if (ba.isEmpty())
5989 return QString(DataPointer::fromRawData(&_empty, 0));
5990 return QUtf8::convertToUnicode(ba);
5991}
5992
5993#ifndef QT_BOOTSTRAPPED
5994/*!
5995 \since 5.3
5996 Returns a QString initialized with the first \a size characters
5997 of the Unicode string \a unicode (ISO-10646-UTF-16 encoded).
5998
5999 If \a size is -1 (default), \a unicode must be '\\0'-terminated.
6000
6001 This function checks for a Byte Order Mark (BOM). If it is missing,
6002 host byte order is assumed.
6003
6004 This function is slow compared to the other Unicode conversions.
6005 Use QString(const QChar *, qsizetype) or QString(const QChar *) if possible.
6006
6007 QString makes a deep copy of the Unicode data.
6008
6009 \sa utf16(), setUtf16(), fromStdU16String()
6010*/
6011QString QString::fromUtf16(const char16_t *unicode, qsizetype size)
6012{
6013 if (!unicode)
6014 return QString();
6015 if (size < 0)
6016 size = QtPrivate::qustrlen(unicode);
6017 QStringDecoder toUtf16(QStringDecoder::Utf16, QStringDecoder::Flag::Stateless);
6018 return toUtf16(QByteArrayView(reinterpret_cast<const char *>(unicode), size * 2));
6019}
6020
6021/*!
6022 \fn QString QString::fromUtf16(const ushort *str, qsizetype size)
6023 \deprecated [6.0] Use the \c char16_t overload instead.
6024*/
6025
6026/*!
6027 \fn QString QString::fromUcs4(const uint *str, qsizetype size)
6028 \since 4.2
6029 \deprecated [6.0] Use the \c char32_t overload instead.
6030*/
6031
6032/*!
6033 \since 5.3
6034
6035 Returns a QString initialized with the first \a size characters
6036 of the Unicode string \a unicode (encoded as UTF-32).
6037
6038 If \a size is -1 (default), \a unicode must be '\\0'-terminated.
6039
6040 \sa toUcs4(), fromUtf16(), utf16(), setUtf16(), fromWCharArray(),
6041 fromStdU32String()
6042*/
6043QString QString::fromUcs4(const char32_t *unicode, qsizetype size)
6044{
6045 if (!unicode)
6046 return QString();
6047 if (size < 0) {
6048 if constexpr (sizeof(char32_t) == sizeof(wchar_t))
6049 size = wcslen(reinterpret_cast<const wchar_t *>(unicode));
6050 else
6051 size = std::char_traits<char32_t>::length(unicode);
6052 }
6053 QStringDecoder toUtf16(QStringDecoder::Utf32, QStringDecoder::Flag::Stateless);
6054 return toUtf16(QByteArrayView(reinterpret_cast<const char *>(unicode), size * 4));
6055}
6056#endif // !QT_BOOTSTRAPPED
6057
6058/*!
6059 Resizes the string to \a size characters and copies \a unicode
6060 into the string.
6061
6062 If \a unicode is \nullptr, nothing is copied, but the string is still
6063 resized to \a size.
6064
6065 \sa unicode(), setUtf16()
6066*/
6067QString& QString::setUnicode(const QChar *unicode, qsizetype size)
6068{
6069 resize(size);
6070 if (unicode && size)
6071 memcpy(d.data(), unicode, size * sizeof(QChar));
6072 return *this;
6073}
6074
6075/*!
6076 \fn QString::setUnicode(const char16_t *unicode, qsizetype size)
6077 \overload
6078 \since 6.9
6079
6080 \sa unicode(), setUtf16()
6081*/
6082
6083/*!
6084 \fn QString::setUtf16(const char16_t *unicode, qsizetype size)
6085 \since 6.9
6086
6087 Resizes the string to \a size characters and copies \a unicode
6088 into the string.
6089
6090 If \a unicode is \nullptr, nothing is copied, but the string is still
6091 resized to \a size.
6092
6093 Note that unlike fromUtf16(), this function does not consider BOMs and
6094 possibly differing byte ordering.
6095
6096 \sa utf16(), setUnicode()
6097*/
6098
6099/*!
6100 \fn QString &QString::setUtf16(const ushort *unicode, qsizetype size)
6101 \obsolete Use the \c char16_t overload instead.
6102*/
6103
6104/*!
6105 \fn QString QString::simplified() const
6106
6107 Returns a string that has whitespace removed from the start
6108 and the end, and that has each sequence of internal whitespace
6109 replaced with a single space.
6110
6111 Whitespace means any character for which QChar::isSpace() returns
6112 \c true. This includes the ASCII characters '\\t', '\\n', '\\v',
6113 '\\f', '\\r', and ' '.
6114
6115 Example:
6116
6117 \snippet qstring/main.cpp 57
6118
6119 \sa trimmed()
6120*/
6121QString QString::simplified_helper(const QString &str)
6122{
6123 return QStringAlgorithms<const QString>::simplified_helper(str);
6124}
6125
6126QString QString::simplified_helper(QString &str)
6127{
6128 return QStringAlgorithms<QString>::simplified_helper(str);
6129}
6130
6131namespace {
6132 template <typename StringView>
6133 StringView qt_trimmed(StringView s) noexcept
6134 {
6135 const auto [begin, end] = QStringAlgorithms<const StringView>::trimmed_helper_positions(s);
6136 return StringView{begin, end};
6137 }
6138}
6139
6140/*!
6141 \fn QStringView QtPrivate::trimmed(QStringView s)
6142 \fn QLatin1StringView QtPrivate::trimmed(QLatin1StringView s)
6143 \internal
6144 \relates QStringView
6145 \since 5.10
6146
6147 Returns \a s with whitespace removed from the start and the end.
6148
6149 Whitespace means any character for which QChar::isSpace() returns
6150 \c true. This includes the ASCII characters '\\t', '\\n', '\\v',
6151 '\\f', '\\r', and ' '.
6152
6153 \sa QString::trimmed(), QStringView::trimmed(), QLatin1StringView::trimmed()
6154*/
6155QStringView QtPrivate::trimmed(QStringView s) noexcept
6156{
6157 return qt_trimmed(s);
6158}
6159
6160QLatin1StringView QtPrivate::trimmed(QLatin1StringView s) noexcept
6161{
6162 return qt_trimmed(s);
6163}
6164
6165/*!
6166 \fn QString QString::trimmed() const
6167
6168 Returns a string that has whitespace removed from the start and
6169 the end.
6170
6171 Whitespace means any character for which QChar::isSpace() returns
6172 \c true. This includes the ASCII characters '\\t', '\\n', '\\v',
6173 '\\f', '\\r', and ' '.
6174
6175 Example:
6176
6177 \snippet qstring/main.cpp 82
6178
6179 Unlike simplified(), trimmed() leaves internal whitespace alone.
6180
6181 \sa simplified()
6182*/
6183QString QString::trimmed_helper(const QString &str)
6184{
6185 return QStringAlgorithms<const QString>::trimmed_helper(str);
6186}
6187
6188QString QString::trimmed_helper(QString &str)
6189{
6190 return QStringAlgorithms<QString>::trimmed_helper(str);
6191}
6192
6193/*! \fn const QChar QString::at(qsizetype position) const
6194
6195 Returns the character at the given index \a position in the
6196 string.
6197
6198 The \a position must be a valid index position in the string
6199 (i.e., 0 <= \a position < size()).
6200
6201 \sa operator[]()
6202*/
6203
6204/*!
6205 \fn QChar &QString::operator[](qsizetype position)
6206
6207 Returns the character at the specified \a position in the string as a
6208 modifiable reference.
6209
6210 Example:
6211
6212 \snippet qstring/main.cpp 85
6213
6214 \sa at()
6215*/
6216
6217/*!
6218 \fn const QChar QString::operator[](qsizetype position) const
6219
6220 \overload operator[]()
6221*/
6222
6223/*!
6224 \fn QChar QString::front() const
6225 \since 5.10
6226
6227 Returns the first character in the string.
6228 Same as \c{at(0)}.
6229
6230 This function is provided for STL compatibility.
6231
6232 \warning Calling this function on an empty string constitutes
6233 undefined behavior.
6234
6235 \sa back(), at(), operator[]()
6236*/
6237
6238/*!
6239 \fn QChar QString::back() const
6240 \since 5.10
6241
6242 Returns the last character in the string.
6243 Same as \c{at(size() - 1)}.
6244
6245 This function is provided for STL compatibility.
6246
6247 \warning Calling this function on an empty string constitutes
6248 undefined behavior.
6249
6250 \sa front(), at(), operator[]()
6251*/
6252
6253/*!
6254 \fn QChar &QString::front()
6255 \since 5.10
6256
6257 Returns a reference to the first character in the string.
6258 Same as \c{operator[](0)}.
6259
6260 This function is provided for STL compatibility.
6261
6262 \warning Calling this function on an empty string constitutes
6263 undefined behavior.
6264
6265 \sa back(), at(), operator[]()
6266*/
6267
6268/*!
6269 \fn QChar &QString::back()
6270 \since 5.10
6271
6272 Returns a reference to the last character in the string.
6273 Same as \c{operator[](size() - 1)}.
6274
6275 This function is provided for STL compatibility.
6276
6277 \warning Calling this function on an empty string constitutes
6278 undefined behavior.
6279
6280 \sa front(), at(), operator[]()
6281*/
6282
6283/*!
6284 \fn void QString::truncate(qsizetype position)
6285
6286 Truncates the string starting from, and including, the element at index
6287 \a position.
6288
6289 If the specified \a position index is beyond the end of the
6290 string, nothing happens.
6291
6292 Example:
6293
6294 \snippet qstring/main.cpp 83
6295
6296 If \a position is negative, it is equivalent to passing zero.
6297
6298 \sa chop(), resize(), first(), QStringView::truncate()
6299*/
6300
6301void QString::truncate(qsizetype pos)
6302{
6303 if (pos < size())
6304 resize(pos);
6305}
6306
6307
6308/*!
6309 Removes \a n characters from the end of the string.
6310
6311 If \a n is greater than or equal to size(), the result is an
6312 empty string; if \a n is negative, it is equivalent to passing zero.
6313
6314 Example:
6315 \snippet qstring/main.cpp 15
6316
6317 If you want to remove characters from the \e beginning of the
6318 string, use remove() instead.
6319
6320 \sa truncate(), resize(), remove(), QStringView::chop()
6321*/
6322void QString::chop(qsizetype n)
6323{
6324 if (n > 0)
6325 resize(d.size - n);
6326}
6327
6328/*!
6329 Sets every character in the string to character \a ch. If \a size
6330 is different from -1 (default), the string is resized to \a
6331 size beforehand.
6332
6333 Example:
6334
6335 \snippet qstring/main.cpp 21
6336
6337 \sa resize()
6338*/
6339
6340QString& QString::fill(QChar ch, qsizetype size)
6341{
6342 resize(size < 0 ? d.size : size);
6343 if (d.size)
6344 std::fill(d.data(), d.data() + d.size, ch.unicode());
6345 return *this;
6346}
6347
6348/*!
6349 \fn qsizetype QString::length() const
6350
6351 Returns the number of characters in this string. Equivalent to
6352 size().
6353
6354 \sa resize()
6355*/
6356
6357/*!
6358 \fn qsizetype QString::size() const
6359
6360 Returns the number of characters in this string.
6361
6362 The last character in the string is at position size() - 1.
6363
6364 Example:
6365 \snippet qstring/main.cpp 58
6366
6367 \sa isEmpty(), resize()
6368*/
6369
6370/*!
6371 \fn qsizetype QString::max_size() const
6372 \fn qsizetype QString::maxSize()
6373 \since 6.8
6374
6375 It returns the maximum number of elements that the string can
6376 theoretically hold. In practice, the number can be much smaller,
6377 limited by the amount of memory available to the system.
6378*/
6379
6380/*! \fn bool QString::isNull() const
6381
6382 Returns \c true if this string is null; otherwise returns \c false.
6383
6384 Example:
6385
6386 \snippet qstring/main.cpp 28
6387
6388 Qt makes a distinction between null strings and empty strings for
6389 historical reasons. For most applications, what matters is
6390 whether or not a string contains any data, and this can be
6391 determined using the isEmpty() function.
6392
6393 \sa isEmpty()
6394*/
6395
6396/*! \fn bool QString::isEmpty() const
6397
6398 Returns \c true if the string has no characters; otherwise returns
6399 \c false.
6400
6401 Example:
6402
6403 \snippet qstring/main.cpp 27
6404
6405 \sa size()
6406*/
6407
6408/*! \fn QString &QString::operator+=(const QString &other)
6409
6410 Appends the string \a other onto the end of this string and
6411 returns a reference to this string.
6412
6413 Example:
6414
6415 \snippet qstring/main.cpp 84
6416
6417 This operation is typically very fast (\l{constant time}),
6418 because QString preallocates extra space at the end of the string
6419 data so it can grow without reallocating the entire string each
6420 time.
6421
6422 \sa append(), prepend()
6423*/
6424
6425/*! \fn QString &QString::operator+=(QLatin1StringView str)
6426
6427 \overload operator+=()
6428
6429 Appends the Latin-1 string viewed by \a str to this string.
6430*/
6431
6432/*! \fn QString &QString::operator+=(QUtf8StringView str)
6433 \since 6.5
6434 \overload operator+=()
6435
6436 Appends the UTF-8 string view \a str to this string.
6437*/
6438
6439/*! \fn QString &QString::operator+=(const QByteArray &ba)
6440
6441 \overload operator+=()
6442
6443 Appends the byte array \a ba to this string. The byte array is converted
6444 to Unicode using the fromUtf8() function. If any NUL characters ('\\0')
6445 are embedded in the \a ba byte array, they will be included in the
6446 transformation.
6447
6448 You can disable this function by defining
6449 \l QT_NO_CAST_FROM_ASCII when you compile your applications. This
6450 can be useful if you want to ensure that all user-visible strings
6451 go through QObject::tr(), for example.
6452*/
6453
6454/*! \fn QString &QString::operator+=(const char *str)
6455
6456 \overload operator+=()
6457
6458 Appends the string \a str to this string. The const char pointer
6459 is converted to Unicode using the fromUtf8() function.
6460
6461 You can disable this function by defining \l QT_NO_CAST_FROM_ASCII
6462 when you compile your applications. This can be useful if you want
6463 to ensure that all user-visible strings go through QObject::tr(),
6464 for example.
6465*/
6466
6467/*! \fn QString &QString::operator+=(QStringView str)
6468 \since 6.0
6469 \overload operator+=()
6470
6471 Appends the string view \a str to this string.
6472*/
6473
6474/*! \fn QString &QString::operator+=(QChar ch)
6475
6476 \overload operator+=()
6477
6478 Appends the character \a ch to the string.
6479*/
6480
6481/*!
6482 \fn bool QString::operator==(const char * const &lhs, const QString &rhs)
6483
6484 \overload operator==()
6485
6486 Returns \c true if \a lhs is equal to \a rhs; otherwise returns \c false.
6487 Note that no string is equal to \a lhs being 0.
6488
6489 Equivalent to \c {lhs != 0 && compare(lhs, rhs) == 0}.
6490*/
6491
6492/*!
6493 \fn bool QString::operator!=(const char * const &lhs, const QString &rhs)
6494
6495 Returns \c true if \a lhs is not equal to \a rhs; otherwise returns
6496 \c false.
6497
6498 For \a lhs != 0, this is equivalent to \c {compare(} \a lhs, \a rhs
6499 \c {) != 0}. Note that no string is equal to \a lhs being 0.
6500*/
6501
6502/*!
6503 \fn bool QString::operator<(const char * const &lhs, const QString &rhs)
6504
6505 Returns \c true if \a lhs is lexically less than \a rhs; otherwise
6506 returns \c false. For \a lhs != 0, this is equivalent to \c
6507 {compare(lhs, rhs) < 0}.
6508
6509 \sa {Comparing Strings}
6510*/
6511
6512/*!
6513 \fn bool QString::operator<=(const char * const &lhs, const QString &rhs)
6514
6515 Returns \c true if \a lhs is lexically less than or equal to \a rhs;
6516 otherwise returns \c false. For \a lhs != 0, this is equivalent to \c
6517 {compare(lhs, rhs) <= 0}.
6518
6519 \sa {Comparing Strings}
6520*/
6521
6522/*!
6523 \fn bool QString::operator>(const char * const &lhs, const QString &rhs)
6524
6525 Returns \c true if \a lhs is lexically greater than \a rhs; otherwise
6526 returns \c false. Equivalent to \c {compare(lhs, rhs) > 0}.
6527
6528 \sa {Comparing Strings}
6529*/
6530
6531/*!
6532 \fn bool QString::operator>=(const char * const &lhs, const QString &rhs)
6533
6534 Returns \c true if \a lhs is lexically greater than or equal to \a rhs;
6535 otherwise returns \c false. For \a lhs != 0, this is equivalent to \c
6536 {compare(lhs, rhs) >= 0}.
6537
6538 \sa {Comparing Strings}
6539*/
6540
6541/*!
6542 \fn QString operator+(const QString &s1, const QString &s2)
6543 \fn QString operator+(QString &&s1, const QString &s2)
6544 \relates QString
6545
6546 Returns a string which is the result of concatenating \a s1 and \a
6547 s2.
6548*/
6549
6550/*!
6551 \fn QString operator+(const QString &s1, const char *s2)
6552 \relates QString
6553
6554 Returns a string which is the result of concatenating \a s1 and \a
6555 s2 (\a s2 is converted to Unicode using the QString::fromUtf8()
6556 function).
6557
6558 \sa QString::fromUtf8()
6559*/
6560
6561/*!
6562 \fn QString operator+(const char *s1, const QString &s2)
6563 \relates QString
6564
6565 Returns a string which is the result of concatenating \a s1 and \a
6566 s2 (\a s1 is converted to Unicode using the QString::fromUtf8()
6567 function).
6568
6569 \sa QString::fromUtf8()
6570*/
6571
6572/*!
6573 \fn QString operator+(QStringView lhs, const QString &rhs)
6574 \fn QString operator+(const QString &lhs, QStringView rhs)
6575
6576 \relates QString
6577 \since 6.9
6578
6579 Returns a string that is the result of concatenating \a lhs and \a rhs.
6580*/
6581
6582/*!
6583 \fn int QString::compare(const QString &s1, const QString &s2, Qt::CaseSensitivity cs)
6584 \since 4.2
6585
6586 Compares the string \a s1 with the string \a s2 and returns a negative integer
6587 if \a s1 is less than \a s2, a positive integer if it is greater than \a s2,
6588 and zero if they are equal.
6589
6590 \include qstring.qdocinc {search-comparison-case-sensitivity} {comparison}
6591
6592 Case sensitive comparison is based exclusively on the numeric
6593 Unicode values of the characters and is very fast, but is not what
6594 a human would expect. Consider sorting user-visible strings with
6595 localeAwareCompare().
6596
6597 \snippet qstring/main.cpp 16
6598
6599//! [compare-isNull-vs-isEmpty]
6600 \note This function treats null strings the same as empty strings,
6601 for more details see \l {Distinction Between Null and Empty Strings}.
6602//! [compare-isNull-vs-isEmpty]
6603
6604 \sa operator==(), operator<(), operator>(), {Comparing Strings}
6605*/
6606
6607/*!
6608 \fn int QString::compare(const QString &s1, QLatin1StringView s2, Qt::CaseSensitivity cs)
6609 \since 4.2
6610 \overload compare()
6611
6612 Performs a comparison of \a s1 and \a s2, using the case
6613 sensitivity setting \a cs.
6614*/
6615
6616/*!
6617 \fn int QString::compare(QLatin1StringView s1, const QString &s2, Qt::CaseSensitivity cs = Qt::CaseSensitive)
6618
6619 \since 4.2
6620 \overload compare()
6621
6622 Performs a comparison of \a s1 and \a s2, using the case
6623 sensitivity setting \a cs.
6624*/
6625
6626/*!
6627 \fn int QString::compare(QStringView s, Qt::CaseSensitivity cs = Qt::CaseSensitive) const
6628
6629 \since 5.12
6630 \overload compare()
6631
6632 Performs a comparison of this with \a s, using the case
6633 sensitivity setting \a cs.
6634*/
6635
6636/*!
6637 \fn int QString::compare(QChar ch, Qt::CaseSensitivity cs = Qt::CaseSensitive) const
6638
6639 \since 5.14
6640 \overload compare()
6641
6642 Performs a comparison of this with \a ch, using the case
6643 sensitivity setting \a cs.
6644*/
6645
6646/*!
6647 \overload compare()
6648 \since 4.2
6649
6650 Lexically compares this string with the string \a other and returns
6651 a negative integer if this string is less than \a other, a positive
6652 integer if it is greater than \a other, and zero if they are equal.
6653
6654 Same as compare(*this, \a other, \a cs).
6655*/
6656int QString::compare(const QString &other, Qt::CaseSensitivity cs) const noexcept
6657{
6658 return QtPrivate::compareStrings(*this, other, cs);
6659}
6660
6661/*!
6662 \internal
6663 \since 4.5
6664*/
6665int QString::compare_helper(const QChar *data1, qsizetype length1, const QChar *data2, qsizetype length2,
6666 Qt::CaseSensitivity cs) noexcept
6667{
6668 Q_ASSERT(length1 >= 0);
6669 Q_ASSERT(length2 >= 0);
6670 Q_ASSERT(data1 || length1 == 0);
6671 Q_ASSERT(data2 || length2 == 0);
6672 return QtPrivate::compareStrings(QStringView(data1, length1), QStringView(data2, length2), cs);
6673}
6674
6675/*!
6676 \overload compare()
6677 \since 4.2
6678
6679 Same as compare(*this, \a other, \a cs).
6680*/
6681int QString::compare(QLatin1StringView other, Qt::CaseSensitivity cs) const noexcept
6682{
6683 return QtPrivate::compareStrings(*this, other, cs);
6684}
6685
6686/*!
6687 \internal
6688 \since 5.0
6689*/
6690int QString::compare_helper(const QChar *data1, qsizetype length1, const char *data2, qsizetype length2,
6691 Qt::CaseSensitivity cs) noexcept
6692{
6693 Q_ASSERT(length1 >= 0);
6694 Q_ASSERT(data1 || length1 == 0);
6695 if (!data2)
6696 return qt_lencmp(length1, 0);
6697 if (Q_UNLIKELY(length2 < 0))
6698 length2 = qsizetype(strlen(data2));
6699 return QtPrivate::compareStrings(QStringView(data1, length1),
6700 QUtf8StringView(data2, length2), cs);
6701}
6702
6703/*!
6704 \fn int QString::compare(const QString &s1, QStringView s2, Qt::CaseSensitivity cs = Qt::CaseSensitive)
6705 \overload compare()
6706*/
6707
6708/*!
6709 \fn int QString::compare(QStringView s1, const QString &s2, Qt::CaseSensitivity cs = Qt::CaseSensitive)
6710 \overload compare()
6711*/
6712
6713bool comparesEqual(const QByteArrayView &lhs, const QChar &rhs) noexcept
6714{
6715 return QtPrivate::equalStrings(QUtf8StringView(lhs), QStringView(&rhs, 1));
6716}
6717
6718Qt::strong_ordering compareThreeWay(const QByteArrayView &lhs, const QChar &rhs) noexcept
6719{
6720 const int res = QtPrivate::compareStrings(QUtf8StringView(lhs), QStringView(&rhs, 1));
6721 return Qt::compareThreeWay(res, 0);
6722}
6723
6724bool comparesEqual(const QByteArrayView &lhs, char16_t rhs) noexcept
6725{
6726 return QtPrivate::equalStrings(QUtf8StringView(lhs), QStringView(&rhs, 1));
6727}
6728
6729Qt::strong_ordering compareThreeWay(const QByteArrayView &lhs, char16_t rhs) noexcept
6730{
6731 const int res = QtPrivate::compareStrings(QUtf8StringView(lhs), QStringView(&rhs, 1));
6732 return Qt::compareThreeWay(res, 0);
6733}
6734
6735bool comparesEqual(const QByteArray &lhs, const QChar &rhs) noexcept
6736{
6737 return QtPrivate::equalStrings(QUtf8StringView(lhs), QStringView(&rhs, 1));
6738}
6739
6740Qt::strong_ordering compareThreeWay(const QByteArray &lhs, const QChar &rhs) noexcept
6741{
6742 const int res = QtPrivate::compareStrings(QUtf8StringView(lhs), QStringView(&rhs, 1));
6743 return Qt::compareThreeWay(res, 0);
6744}
6745
6746bool comparesEqual(const QByteArray &lhs, char16_t rhs) noexcept
6747{
6748 return QtPrivate::equalStrings(QUtf8StringView(lhs), QStringView(&rhs, 1));
6749}
6750
6751Qt::strong_ordering compareThreeWay(const QByteArray &lhs, char16_t rhs) noexcept
6752{
6753 const int res = QtPrivate::compareStrings(QUtf8StringView(lhs), QStringView(&rhs, 1));
6754 return Qt::compareThreeWay(res, 0);
6755}
6756
6757/*!
6758 \internal
6759 \since 6.8
6760*/
6761bool QT_FASTCALL QChar::equal_helper(QChar lhs, const char *rhs) noexcept
6762{
6763 return QtPrivate::equalStrings(QStringView(&lhs, 1), QUtf8StringView(rhs));
6764}
6765
6766int QT_FASTCALL QChar::compare_helper(QChar lhs, const char *rhs) noexcept
6767{
6768 return QtPrivate::compareStrings(QStringView(&lhs, 1), QUtf8StringView(rhs));
6769}
6770
6771/*!
6772 \internal
6773 \since 6.8
6774*/
6775bool QStringView::equal_helper(QStringView sv, const char *data, qsizetype len)
6776{
6777 Q_ASSERT(len >= 0);
6778 Q_ASSERT(data || len == 0);
6779 return QtPrivate::equalStrings(sv, QUtf8StringView(data, len));
6780}
6781
6782/*!
6783 \internal
6784 \since 6.8
6785*/
6786int QStringView::compare_helper(QStringView sv, const char *data, qsizetype len)
6787{
6788 Q_ASSERT(len >= 0);
6789 Q_ASSERT(data || len == 0);
6790 return QtPrivate::compareStrings(sv, QUtf8StringView(data, len));
6791}
6792
6793/*!
6794 \internal
6795 \since 6.8
6796*/
6797bool QLatin1StringView::equal_helper(QLatin1StringView s1, const char *s2, qsizetype len) noexcept
6798{
6799 // because qlatin1stringview.h can't include qutf8stringview.h
6800 Q_ASSERT(len >= 0);
6801 Q_ASSERT(s2 || len == 0);
6802 return QtPrivate::equalStrings(s1, QUtf8StringView(s2, len));
6803}
6804
6805/*!
6806 \internal
6807 \since 6.6
6808*/
6809int QLatin1StringView::compare_helper(const QLatin1StringView &s1, const char *s2, qsizetype len) noexcept
6810{
6811 // because qlatin1stringview.h can't include qutf8stringview.h
6812 Q_ASSERT(len >= 0);
6813 Q_ASSERT(s2 || len == 0);
6814 return QtPrivate::compareStrings(s1, QUtf8StringView(s2, len));
6815}
6816
6817/*!
6818 \internal
6819 \since 4.5
6820*/
6821int QLatin1StringView::compare_helper(const QChar *data1, qsizetype length1, QLatin1StringView s2,
6822 Qt::CaseSensitivity cs) noexcept
6823{
6824 Q_ASSERT(length1 >= 0);
6825 Q_ASSERT(data1 || length1 == 0);
6826 return QtPrivate::compareStrings(QStringView(data1, length1), s2, cs);
6827}
6828
6829/*!
6830 \fn int QString::localeAwareCompare(const QString & s1, const QString & s2)
6831
6832 Compares \a s1 with \a s2 and returns an integer less than, equal
6833 to, or greater than zero if \a s1 is less than, equal to, or
6834 greater than \a s2.
6835
6836 The comparison is performed in a locale- and also
6837 platform-dependent manner. Use this function to present sorted
6838 lists of strings to the user.
6839
6840 \sa compare(), QLocale, {Comparing Strings}
6841*/
6842
6843/*!
6844 \fn int QString::localeAwareCompare(QStringView other) const
6845 \since 6.0
6846 \overload localeAwareCompare()
6847
6848 Compares this string with the \a other string and returns an
6849 integer less than, equal to, or greater than zero if this string
6850 is less than, equal to, or greater than the \a other string.
6851
6852 The comparison is performed in a locale- and also
6853 platform-dependent manner. Use this function to present sorted
6854 lists of strings to the user.
6855
6856 Same as \c {localeAwareCompare(*this, other)}.
6857
6858 \sa {Comparing Strings}
6859*/
6860
6861/*!
6862 \fn int QString::localeAwareCompare(QStringView s1, QStringView s2)
6863 \since 6.0
6864 \overload localeAwareCompare()
6865
6866 Compares \a s1 with \a s2 and returns an integer less than, equal
6867 to, or greater than zero if \a s1 is less than, equal to, or
6868 greater than \a s2.
6869
6870 The comparison is performed in a locale- and also
6871 platform-dependent manner. Use this function to present sorted
6872 lists of strings to the user.
6873
6874 \sa {Comparing Strings}
6875*/
6876
6877
6878#if !defined(CSTR_LESS_THAN)
6879#define CSTR_LESS_THAN 1
6880#define CSTR_EQUAL 2
6881#define CSTR_GREATER_THAN 3
6882#endif
6883
6884/*!
6885 \overload localeAwareCompare()
6886
6887 Compares this string with the \a other string and returns an
6888 integer less than, equal to, or greater than zero if this string
6889 is less than, equal to, or greater than the \a other string.
6890
6891 The comparison is performed in a locale- and also
6892 platform-dependent manner. Use this function to present sorted
6893 lists of strings to the user.
6894
6895 Same as \c {localeAwareCompare(*this, other)}.
6896
6897 \sa {Comparing Strings}
6898*/
6899int QString::localeAwareCompare(const QString &other) const
6900{
6901 return localeAwareCompare_helper(constData(), size(), other.constData(), other.size());
6902}
6903
6904/*!
6905 \internal
6906 \since 4.5
6907*/
6908int QString::localeAwareCompare_helper(const QChar *data1, qsizetype length1,
6909 const QChar *data2, qsizetype length2)
6910{
6911 Q_ASSERT(length1 >= 0);
6912 Q_ASSERT(data1 || length1 == 0);
6913 Q_ASSERT(length2 >= 0);
6914 Q_ASSERT(data2 || length2 == 0);
6915
6916 // do the right thing for null and empty
6917 if (length1 == 0 || length2 == 0)
6918 return QtPrivate::compareStrings(QStringView(data1, length1), QStringView(data2, length2),
6919 Qt::CaseSensitive);
6920
6921#if QT_CONFIG(icu) || defined(Q_OS_ANDROID)
6922 return QCollator::defaultCompare(QStringView(data1, length1), QStringView(data2, length2));
6923#else
6924 const QString lhs = QString::fromRawData(data1, length1).normalized(QString::NormalizationForm_C);
6925 const QString rhs = QString::fromRawData(data2, length2).normalized(QString::NormalizationForm_C);
6926# if defined(Q_OS_WIN)
6927 int res = CompareStringEx(LOCALE_NAME_USER_DEFAULT, 0, (LPWSTR)lhs.constData(), lhs.length(), (LPWSTR)rhs.constData(), rhs.length(), NULL, NULL, 0);
6928
6929 switch (res) {
6930 case CSTR_LESS_THAN:
6931 return -1;
6932 case CSTR_GREATER_THAN:
6933 return 1;
6934 default:
6935 return 0;
6936 }
6937# elif defined (Q_OS_DARWIN)
6938 // Use CFStringCompare for comparing strings on Mac. This makes Qt order
6939 // strings the same way as native applications do, and also respects
6940 // the "Order for sorted lists" setting in the International preferences
6941 // panel.
6942 const CFStringRef thisString =
6943 CFStringCreateWithCharactersNoCopy(kCFAllocatorDefault,
6944 reinterpret_cast<const UniChar *>(lhs.constData()), lhs.length(), kCFAllocatorNull);
6945 const CFStringRef otherString =
6946 CFStringCreateWithCharactersNoCopy(kCFAllocatorDefault,
6947 reinterpret_cast<const UniChar *>(rhs.constData()), rhs.length(), kCFAllocatorNull);
6948
6949 const int result = CFStringCompare(thisString, otherString, kCFCompareLocalized);
6950 CFRelease(thisString);
6951 CFRelease(otherString);
6952 return result;
6953# elif defined(Q_OS_UNIX)
6954 // declared in <string.h> (no better than QtPrivate::compareStrings() on Android, sadly)
6955 return strcoll(lhs.toLocal8Bit().constData(), rhs.toLocal8Bit().constData());
6956# else
6957# error "This case shouldn't happen"
6958 return QtPrivate::compareStrings(lhs, rhs, Qt::CaseSensitive);
6959# endif
6960#endif // !QT_CONFIG(icu)
6961}
6962
6963
6964/*!
6965 \fn const QChar *QString::unicode() const
6966
6967 Returns a Unicode representation of the string.
6968 The result remains valid until the string is modified.
6969
6970 \note The returned string may not be '\\0'-terminated.
6971 Use size() to determine the length of the array.
6972
6973 \sa utf16(), fromRawData()
6974*/
6975
6976/*!
6977 \fn const ushort *QString::utf16() const
6978 \obsolete [6.11] Use nullTerminate() and cast data() to \c{const char16_t *}
6979
6980 Returns the QString as a '\\0\'-terminated array of unsigned
6981 shorts. The result remains valid until the string is modified.
6982
6983 The returned string is in host byte order.
6984
6985 \sa unicode()
6986*/
6987
6988const ushort *QString::utf16() const
6989{
6990 if (!d.isMutable()) {
6991 // ensure '\0'-termination for ::fromRawData strings
6992 const_cast<QString*>(this)->reallocData(d.size, QArrayData::KeepSize);
6993 }
6994 return reinterpret_cast<const ushort *>(d.data());
6995}
6996
6997/*!
6998 \fn QString &QString::nullTerminate()
6999 \since 6.10
7000
7001 If this string data isn't null-terminated, this method will make a deep
7002 copy of the data and make it null-terminated.
7003
7004 A QString is null-terminated by default, however in some cases (e.g.
7005 when using fromRawData()), the string data doesn't necessarily end
7006 with a \c {\0} character, which could be a problem when calling methods
7007 that expect a null-terminated string.
7008
7009 \sa nullTerminated(), fromRawData(), setRawData()
7010*/
7011QString &QString::nullTerminate()
7012{
7013 // ensure '\0'-termination for ::fromRawData strings
7014 if (!d.isMutable())
7015 *this = QString{constData(), size()};
7016 return *this;
7017}
7018
7019/*!
7020 \fn QString QString::nullTerminated() const &
7021 \fn QString QString::nullTerminated() &&
7022 \since 6.10
7023
7024 Returns a copy of this string that is always null-terminated.
7025
7026 \sa nullTerminate(), fromRawData(), setRawData()
7027*/
7028QString QString::nullTerminated() const &
7029{
7030 // ensure '\0'-termination for ::fromRawData strings
7031 if (!d.isMutable())
7032 return QString{constData(), size()};
7033 return *this;
7034}
7035
7036QString QString::nullTerminated() &&
7037{
7038 nullTerminate();
7039 return std::move(*this);
7040}
7041
7042/*!
7043 Returns a string of size \a width that contains this string
7044 padded by the \a fill character.
7045
7046 If \a truncate is \c false and the size() of the string is more than
7047 \a width, then the returned string is a copy of the string.
7048
7049 \snippet qstring/main.cpp 32
7050
7051 If \a truncate is \c true and the size() of the string is more than
7052 \a width, then any characters in a copy of the string after
7053 position \a width are removed, and the copy is returned.
7054
7055 \snippet qstring/main.cpp 33
7056
7057 \sa rightJustified()
7058*/
7059
7060QString QString::leftJustified(qsizetype width, QChar fill, bool truncate) const
7061{
7062 QString result;
7063 qsizetype len = size();
7064 qsizetype padlen = width - len;
7065 if (padlen > 0) {
7066 result.resize(len+padlen);
7067 if (len)
7068 memcpy(result.d.data(), d.data(), sizeof(QChar)*len);
7069 QChar *uc = (QChar*)result.d.data() + len;
7070 while (padlen--)
7071 * uc++ = fill;
7072 } else {
7073 if (truncate)
7074 result = left(width);
7075 else
7076 result = *this;
7077 }
7078 return result;
7079}
7080
7081/*!
7082 Returns a string of size() \a width that contains the \a fill
7083 character followed by the string. For example:
7084
7085 \snippet qstring/main.cpp 49
7086
7087 If \a truncate is \c false and the size() of the string is more than
7088 \a width, then the returned string is a copy of the string.
7089
7090 If \a truncate is true and the size() of the string is more than
7091 \a width, then the resulting string is truncated at position \a
7092 width.
7093
7094 \snippet qstring/main.cpp 50
7095
7096 \sa leftJustified()
7097*/
7098
7099QString QString::rightJustified(qsizetype width, QChar fill, bool truncate) const
7100{
7101 QString result;
7102 qsizetype len = size();
7103 qsizetype padlen = width - len;
7104 if (padlen > 0) {
7105 result.resize(len+padlen);
7106 QChar *uc = (QChar*)result.d.data();
7107 while (padlen--)
7108 * uc++ = fill;
7109 if (len)
7110 memcpy(static_cast<void *>(uc), static_cast<const void *>(d.data()), sizeof(QChar)*len);
7111 } else {
7112 if (truncate)
7113 result = left(width);
7114 else
7115 result = *this;
7116 }
7117 return result;
7118}
7119
7120/*!
7121 \fn QString QString::toLower() const
7122
7123 Returns a lowercase copy of the string.
7124
7125 \snippet qstring/main.cpp 75
7126
7127 The case conversion will always happen in the 'C' locale. For
7128 locale-dependent case folding use QLocale::toLower()
7129
7130 \sa toUpper(), QLocale::toLower()
7131*/
7132
7133namespace QUnicodeTables {
7134/*
7135 \internal
7136 Converts the \a str string starting from the position pointed to by the \a
7137 it iterator, using the Unicode case traits \c Traits, and returns the
7138 result. The input string must not be empty (the convertCase function below
7139 guarantees that).
7140
7141 The string type \c{T} is also a template and is either \c{const QString} or
7142 \c{QString}. This function can do both copy-conversion and in-place
7143 conversion depending on the state of the \a str parameter:
7144 \list
7145 \li \c{T} is \c{const QString}: copy-convert
7146 \li \c{T} is \c{QString} and its refcount != 1: copy-convert
7147 \li \c{T} is \c{QString} and its refcount == 1: in-place convert
7148 \endlist
7149
7150 In copy-convert mode, the local variable \c{s} is detached from the input
7151 \a str. In the in-place convert mode, \a str is in moved-from state and
7152 \c{s} contains the only copy of the string, without reallocation (thus,
7153 \a it is still valid).
7154
7155 There is one pathological case left: when the in-place conversion needs to
7156 reallocate memory to grow the buffer. In that case, we need to adjust the \a
7157 it pointer.
7158 */
7159template <typename T>
7160Q_NEVER_INLINE
7162{
7163 Q_ASSERT(!str.isEmpty());
7164 QString s = std::move(str); // will copy if T is const QString
7165 QChar *pp = s.begin() + it.index(); // will detach if necessary
7166
7167 do {
7168 const auto folded = fullConvertCase(it.next(), which);
7169 if (Q_UNLIKELY(folded.size() > 1)) {
7170 if (folded.chars[0] == *pp && folded.size() == 2) {
7171 // special case: only second actually changed (e.g. surrogate pairs),
7172 // avoid slow case
7173 ++pp;
7174 *pp++ = folded.chars[1];
7175 } else {
7176 // slow path: the string is growing
7177 qsizetype inpos = it.index() - 1;
7179
7180 s.replace(outpos, 1, reinterpret_cast<const QChar *>(folded.data()), folded.size());
7181 pp = const_cast<QChar *>(s.constBegin()) + outpos + folded.size();
7182
7183 // Adjust the input iterator if we are performing an in-place conversion
7184 if constexpr (!std::is_const<T>::value)
7186 }
7187 } else {
7188 *pp++ = folded.chars[0];
7189 }
7190 } while (it.hasNext());
7191
7192 return s;
7193}
7194
7195template <typename T>
7196static QString convertCase(T &str, QUnicodeTables::Case which)
7197{
7198 const QChar *p = str.constBegin();
7199 const QChar *e = p + str.size();
7200
7201 // this avoids out of bounds check in the loop
7202 while (e != p && e[-1].isHighSurrogate())
7203 --e;
7204
7205 QStringIterator it(p, e);
7206 while (it.hasNext()) {
7207 const char32_t uc = it.next();
7208 if (caseConversion(uc)[which].diff) {
7209 it.recede();
7210 return detachAndConvertCase(str, it, which);
7211 }
7212 }
7213 return std::move(str);
7214}
7215} // namespace QUnicodeTables
7216
7217QString QString::toLower_helper(const QString &str)
7218{
7219 return QUnicodeTables::convertCase(str, QUnicodeTables::LowerCase);
7220}
7221
7222QString QString::toLower_helper(QString &str)
7223{
7224 return QUnicodeTables::convertCase(str, QUnicodeTables::LowerCase);
7225}
7226
7227/*!
7228 \fn QString QString::toCaseFolded() const
7229
7230 Returns the case folded equivalent of the string. For most Unicode
7231 characters this is the same as toLower().
7232*/
7233
7234QString QString::toCaseFolded_helper(const QString &str)
7235{
7236 return QUnicodeTables::convertCase(str, QUnicodeTables::CaseFold);
7237}
7238
7239QString QString::toCaseFolded_helper(QString &str)
7240{
7241 return QUnicodeTables::convertCase(str, QUnicodeTables::CaseFold);
7242}
7243
7244/*!
7245 \fn QString QString::toUpper() const
7246
7247 Returns an uppercase copy of the string.
7248
7249 \snippet qstring/main.cpp 81
7250
7251 The case conversion will always happen in the 'C' locale. For
7252 locale-dependent case folding use QLocale::toUpper().
7253
7254 \note In some cases the uppercase form of a string may be longer than the
7255 original.
7256
7257 \note Since 2024, the German language officially prefers to uppercase ß
7258 (U+00DF LATIN SMALL LETTER SHARP S) as ẞ (U+1E9E LATIN CAPITAL LETTER SHARP S).
7259 Qt's implementation follows Unicode, which still mandates the use of "SS".
7260 If you need to implement the new German rules, you need to manually do
7261 \c{replace(u'ß', u'ẞ')} \e{before} calling this function.
7262
7263 \sa toLower(), QLocale::toLower()
7264*/
7265
7266QString QString::toUpper_helper(const QString &str)
7267{
7268 return QUnicodeTables::convertCase(str, QUnicodeTables::UpperCase);
7269}
7270
7271QString QString::toUpper_helper(QString &str)
7272{
7273 return QUnicodeTables::convertCase(str, QUnicodeTables::UpperCase);
7274}
7275
7276/*!
7277 \since 5.5
7278
7279 Safely builds a formatted string from the format string \a cformat
7280 and an arbitrary list of arguments.
7281
7282 The format string supports the conversion specifiers, length modifiers,
7283 and flags provided by printf() in the standard C++ library. The \a cformat
7284 string and \c{%s} arguments must be UTF-8 encoded.
7285
7286 \note The \c{%lc} escape sequence expects a unicode character of type
7287 \c char16_t (as returned by QChar::unicode()), or \c ushort.
7288 The \c{%ls} escape sequence expects a pointer to a zero-terminated array
7289 of unicode characters of type \c char16_t, or \c ushort (as returned by
7290 QString::utf16()). This is at odds with the printf() in the standard C++
7291 library, which defines \c {%lc} to print a wchar_t and \c{%ls} to print
7292 a \c{wchar_t*}, and might also produce compiler warnings on platforms
7293 where the size of \c {wchar_t} is not 16 bits.
7294
7295 \warning We do not recommend using QString::asprintf() in new Qt
7296 code. Instead, consider using QTextStream or arg(), both of
7297 which support Unicode strings seamlessly and are type-safe.
7298 Here is an example that uses QTextStream:
7299
7300 \snippet qstring/main.cpp 64
7301
7302 For \l {QObject::tr()}{translations}, especially if the strings
7303 contains more than one escape sequence, you should consider using
7304 the arg() function instead. This allows the order of the
7305 replacements to be controlled by the translator.
7306
7307 \sa arg()
7308*/
7309
7310QString QString::asprintf(const char *cformat, ...)
7311{
7312 va_list ap;
7313 va_start(ap, cformat);
7314 QString s = vasprintf(cformat, ap);
7315 va_end(ap);
7316 return s;
7317}
7318
7319static void append_utf8(QString &qs, const char *cs, qsizetype len)
7320{
7321 const qsizetype oldSize = qs.size();
7322 qs.resize(oldSize + len);
7323 const QChar *newEnd = QUtf8::convertToUnicode(qs.data() + oldSize, QByteArrayView(cs, len));
7324 qs.resize(newEnd - qs.constData());
7325}
7326
7327static uint parse_flag_characters(const char * &c) noexcept
7328{
7329 uint flags = QLocaleData::ZeroPadExponent;
7330 while (true) {
7331 switch (*c) {
7332 case '#':
7335 break;
7336 case '0': flags |= QLocaleData::ZeroPadded; break;
7337 case '-': flags |= QLocaleData::LeftAdjusted; break;
7338 case ' ': flags |= QLocaleData::BlankBeforePositive; break;
7339 case '+': flags |= QLocaleData::AlwaysShowSign; break;
7340 case '\'': flags |= QLocaleData::GroupDigits; break;
7341 default: return flags;
7342 }
7343 ++c;
7344 }
7345}
7346
7347static int parse_field_width(const char *&c, qsizetype size)
7348{
7349 Q_ASSERT(isAsciiDigit(*c));
7350 const char *const stop = c + size;
7351
7352 // can't be negative - started with a digit
7353 // contains at least one digit
7354 auto [result, used] = qstrntoull(c, size, 10);
7355 c += used;
7356 if (used <= 0)
7357 return false;
7358 // preserve Qt 5.5 behavior of consuming all digits, no matter how many
7359 while (c < stop && isAsciiDigit(*c))
7360 ++c;
7361 return result < qulonglong(std::numeric_limits<int>::max()) ? int(result) : 0;
7362}
7363
7365
7366static inline bool can_consume(const char * &c, char ch) noexcept
7367{
7368 if (*c == ch) {
7369 ++c;
7370 return true;
7371 }
7372 return false;
7373}
7374
7375static LengthMod parse_length_modifier(const char * &c) noexcept
7376{
7377 switch (*c++) {
7378 case 'h': return can_consume(c, 'h') ? lm_hh : lm_h;
7379 case 'l': return can_consume(c, 'l') ? lm_ll : lm_l;
7380 case 'L': return lm_L;
7381 case 'j': return lm_j;
7382 case 'z':
7383 case 'Z': return lm_z;
7384 case 't': return lm_t;
7385 }
7386 --c; // don't consume *c - it wasn't a flag
7387 return lm_none;
7388}
7389
7390/*!
7391 \fn QString QString::vasprintf(const char *cformat, va_list ap)
7392 \since 5.5
7393
7394 Equivalent method to asprintf(), but takes a va_list \a ap
7395 instead a list of variable arguments. See the asprintf()
7396 documentation for an explanation of \a cformat.
7397
7398 This method does not call the va_end macro, the caller
7399 is responsible to call va_end on \a ap.
7400
7401 \sa asprintf()
7402*/
7403
7404QString QString::vasprintf(const char *cformat, va_list ap)
7405{
7406 if (!cformat || !*cformat) {
7407 // Qt 1.x compat
7408 return fromLatin1("");
7409 }
7410
7411 // Parse cformat
7412
7413 QString result;
7414 const char *c = cformat;
7415 const char *formatEnd = cformat + qstrlen(cformat);
7416 for (;;) {
7417 // Copy non-escape chars to result
7418 const char *cb = c;
7419 while (*c != '\0' && *c != '%')
7420 c++;
7421 append_utf8(result, cb, qsizetype(c - cb));
7422
7423 if (*c == '\0')
7424 break;
7425
7426 // Found '%'
7427 const char *escape_start = c;
7428 ++c;
7429
7430 if (*c == '\0') {
7431 result.append(u'%'); // a % at the end of the string - treat as non-escape text
7432 break;
7433 }
7434 if (*c == '%') {
7435 result.append(u'%'); // %%
7436 ++c;
7437 continue;
7438 }
7439
7440 uint flags = parse_flag_characters(c);
7441
7442 if (*c == '\0') {
7443 result.append(QLatin1StringView(escape_start)); // incomplete escape, treat as non-escape text
7444 break;
7445 }
7446
7447 // Parse field width
7448 int width = -1; // -1 means unspecified
7449 if (isAsciiDigit(*c)) {
7450 width = parse_field_width(c, formatEnd - c);
7451 } else if (*c == '*') { // can't parse this in another function, not portably, at least
7452 width = va_arg(ap, int);
7453 if (width < 0)
7454 width = -1; // treat all negative numbers as unspecified
7455 ++c;
7456 }
7457
7458 if (*c == '\0') {
7459 result.append(QLatin1StringView(escape_start)); // incomplete escape, treat as non-escape text
7460 break;
7461 }
7462
7463 // Parse precision
7464 int precision = -1; // -1 means unspecified
7465 if (*c == '.') {
7466 ++c;
7467 precision = 0;
7468 if (isAsciiDigit(*c)) {
7469 precision = parse_field_width(c, formatEnd - c);
7470 } else if (*c == '*') { // can't parse this in another function, not portably, at least
7471 precision = va_arg(ap, int);
7472 if (precision < 0)
7473 precision = -1; // treat all negative numbers as unspecified
7474 ++c;
7475 }
7476 }
7477
7478 if (*c == '\0') {
7479 result.append(QLatin1StringView(escape_start)); // incomplete escape, treat as non-escape text
7480 break;
7481 }
7482
7483 const LengthMod length_mod = parse_length_modifier(c);
7484
7485 if (*c == '\0') {
7486 result.append(QLatin1StringView(escape_start)); // incomplete escape, treat as non-escape text
7487 break;
7488 }
7489
7490 // Parse the conversion specifier and do the conversion
7491 QString subst;
7492 switch (*c) {
7493 case 'd':
7494 case 'i': {
7495 qint64 i;
7496 switch (length_mod) {
7497 case lm_none: i = va_arg(ap, int); break;
7498 case lm_hh: i = va_arg(ap, int); break;
7499 case lm_h: i = va_arg(ap, int); break;
7500 case lm_l: i = va_arg(ap, long int); break;
7501 case lm_ll: i = va_arg(ap, qint64); break;
7502 case lm_j: i = va_arg(ap, long int); break;
7503
7504 /* ptrdiff_t actually, but it should be the same for us */
7505 case lm_z: i = va_arg(ap, qsizetype); break;
7506 case lm_t: i = va_arg(ap, qsizetype); break;
7507 default: i = 0; break;
7508 }
7509 subst = QLocaleData::c()->longLongToString(i, precision, 10, width, flags);
7510 ++c;
7511 break;
7512 }
7513 case 'o':
7514 case 'u':
7515 case 'x':
7516 case 'X': {
7517 quint64 u;
7518 switch (length_mod) {
7519 case lm_none: u = va_arg(ap, uint); break;
7520 case lm_hh: u = va_arg(ap, uint); break;
7521 case lm_h: u = va_arg(ap, uint); break;
7522 case lm_l: u = va_arg(ap, ulong); break;
7523 case lm_ll: u = va_arg(ap, quint64); break;
7524 case lm_t: u = va_arg(ap, size_t); break;
7525 case lm_z: u = va_arg(ap, size_t); break;
7526 default: u = 0; break;
7527 }
7528
7529 if (isAsciiUpper(*c))
7530 flags |= QLocaleData::CapitalEorX;
7531
7532 int base = 10;
7533 switch (QtMiscUtils::toAsciiLower(*c)) {
7534 case 'o':
7535 base = 8; break;
7536 case 'u':
7537 base = 10; break;
7538 case 'x':
7539 base = 16; break;
7540 default: break;
7541 }
7542 subst = QLocaleData::c()->unsLongLongToString(u, precision, base, width, flags);
7543 ++c;
7544 break;
7545 }
7546 case 'E':
7547 case 'e':
7548 case 'F':
7549 case 'f':
7550 case 'G':
7551 case 'g':
7552 case 'A':
7553 case 'a': {
7554 double d;
7555 if (length_mod == lm_L)
7556 d = va_arg(ap, long double); // not supported - converted to a double
7557 else
7558 d = va_arg(ap, double);
7559
7560 if (isAsciiUpper(*c))
7561 flags |= QLocaleData::CapitalEorX;
7562
7563 QLocaleData::DoubleForm form = QLocaleData::DFDecimal;
7564 switch (QtMiscUtils::toAsciiLower(*c)) {
7565 case 'e': form = QLocaleData::DFExponent; break;
7566 case 'a': // not supported - decimal form used instead
7567 case 'f': form = QLocaleData::DFDecimal; break;
7568 case 'g': form = QLocaleData::DFSignificantDigits; break;
7569 default: break;
7570 }
7571 subst = QLocaleData::c()->doubleToString(d, precision, form, width, flags);
7572 ++c;
7573 break;
7574 }
7575 case 'c': {
7576 if (length_mod == lm_l)
7577 subst = QChar::fromUcs2(va_arg(ap, int));
7578 else
7579 subst = QLatin1Char((uchar) va_arg(ap, int));
7580 ++c;
7581 break;
7582 }
7583 case 's': {
7584 if (length_mod == lm_l) {
7585 const char16_t *buff = va_arg(ap, const char16_t*);
7586 const auto *ch = buff;
7587 while (precision != 0 && *ch != 0) {
7588 ++ch;
7589 --precision;
7590 }
7591 subst.setUtf16(buff, ch - buff);
7592 } else if (precision == -1) {
7593 subst = QString::fromUtf8(va_arg(ap, const char*));
7594 } else {
7595 const char *buff = va_arg(ap, const char*);
7596 subst = QString::fromUtf8(buff, qstrnlen(buff, precision));
7597 }
7598 ++c;
7599 break;
7600 }
7601 case 'p': {
7602 void *arg = va_arg(ap, void*);
7603 const quint64 i = reinterpret_cast<quintptr>(arg);
7604 flags |= QLocaleData::ShowBase;
7605 subst = QLocaleData::c()->unsLongLongToString(i, precision, 16, width, flags);
7606 ++c;
7607 break;
7608 }
7609 case 'n':
7610 switch (length_mod) {
7611 case lm_hh: {
7612 signed char *n = va_arg(ap, signed char*);
7613 *n = result.size();
7614 break;
7615 }
7616 case lm_h: {
7617 short int *n = va_arg(ap, short int*);
7618 *n = result.size();
7619 break;
7620 }
7621 case lm_l: {
7622 long int *n = va_arg(ap, long int*);
7623 *n = result.size();
7624 break;
7625 }
7626 case lm_ll: {
7627 qint64 *n = va_arg(ap, qint64*);
7628 *n = result.size();
7629 break;
7630 }
7631 default: {
7632 int *n = va_arg(ap, int*);
7633 *n = int(result.size());
7634 break;
7635 }
7636 }
7637 ++c;
7638 break;
7639
7640 default: // bad escape, treat as non-escape text
7641 for (const char *cc = escape_start; cc != c; ++cc)
7642 result.append(QLatin1Char(*cc));
7643 continue;
7644 }
7645
7646 if (flags & QLocaleData::LeftAdjusted)
7647 result.append(subst.leftJustified(width));
7648 else
7649 result.append(subst.rightJustified(width));
7650 }
7651
7652 return result;
7653}
7654
7655/*!
7656 \fn QString::toLongLong(bool *ok, int base) const
7657
7658 Returns the string converted to a \c{long long} using base \a
7659 base, which is 10 by default and must be between 2 and 36, or 0.
7660 Returns 0 if the conversion fails.
7661
7662 If \a ok is not \nullptr, failure is reported by setting *\a{ok}
7663 to \c false, and success by setting *\a{ok} to \c true.
7664
7665 If \a base is 0, the C language convention is used: if the string begins
7666 with "0x", base 16 is used; otherwise, if the string begins with "0b", base
7667 2 is used; otherwise, if the string begins with "0", base 8 is used;
7668 otherwise, base 10 is used.
7669
7670 The string conversion will always happen in the 'C' locale. For
7671 locale-dependent conversion use QLocale::toLongLong()
7672
7673 Example:
7674
7675 \snippet qstring/main.cpp 74
7676
7677 This function ignores leading and trailing whitespace.
7678
7679 \note Support for the "0b" prefix was added in Qt 6.4.
7680
7681 \sa number(), toULongLong(), toInt(), QLocale::toLongLong()
7682*/
7683
7684template <typename Int>
7685static Int toIntegral(QStringView string, bool *ok, int base)
7686{
7687#if defined(QT_CHECK_RANGE)
7688 if (base != 0 && (base < 2 || base > 36)) {
7689 qWarning("QString::toIntegral: Invalid base (%d)", base);
7690 base = 10;
7691 }
7692#endif
7693
7694 QVarLengthArray<uchar> latin1(string.size());
7695 qt_to_latin1(latin1.data(), string.utf16(), string.size());
7696 QSimpleParsedNumber<Int> r;
7697 if constexpr (std::is_signed_v<Int>)
7698 r = QLocaleData::bytearrayToLongLong(latin1, base);
7699 else
7700 r = QLocaleData::bytearrayToUnsLongLong(latin1, base);
7701 if (ok)
7702 *ok = r.ok();
7703 return r.result;
7704}
7705
7706qlonglong QString::toIntegral_helper(QStringView string, bool *ok, int base)
7707{
7708 return toIntegral<qlonglong>(string, ok, base);
7709}
7710
7711/*!
7712 \fn QString::toULongLong(bool *ok, int base) const
7713
7714 Returns the string converted to an \c{unsigned long long} using base \a
7715 base, which is 10 by default and must be between 2 and 36, or 0.
7716 Returns 0 if the conversion fails.
7717
7718 If \a ok is not \nullptr, failure is reported by setting *\a{ok}
7719 to \c false, and success by setting *\a{ok} to \c true.
7720
7721 If \a base is 0, the C language convention is used: if the string begins
7722 with "0x", base 16 is used; otherwise, if the string begins with "0b", base
7723 2 is used; otherwise, if the string begins with "0", base 8 is used;
7724 otherwise, base 10 is used.
7725
7726 The string conversion will always happen in the 'C' locale. For
7727 locale-dependent conversion use QLocale::toULongLong()
7728
7729 Example:
7730
7731 \snippet qstring/main.cpp 79
7732
7733 This function ignores leading and trailing whitespace.
7734
7735 \note Support for the "0b" prefix was added in Qt 6.4.
7736
7737 \sa number(), toLongLong(), QLocale::toULongLong()
7738*/
7739
7740qulonglong QString::toIntegral_helper(QStringView string, bool *ok, uint base)
7741{
7742 return toIntegral<qulonglong>(string, ok, base);
7743}
7744
7745/*!
7746 \fn long QString::toLong(bool *ok, int base) const
7747
7748 Returns the string converted to a \c long using base \a
7749 base, which is 10 by default and must be between 2 and 36, or 0.
7750 Returns 0 if the conversion fails.
7751
7752 If \a ok is not \nullptr, failure is reported by setting *\a{ok}
7753 to \c false, and success by setting *\a{ok} to \c true.
7754
7755 If \a base is 0, the C language convention is used: if the string begins
7756 with "0x", base 16 is used; otherwise, if the string begins with "0b", base
7757 2 is used; otherwise, if the string begins with "0", base 8 is used;
7758 otherwise, base 10 is used.
7759
7760 The string conversion will always happen in the 'C' locale. For
7761 locale-dependent conversion use QLocale::toLongLong()
7762
7763 Example:
7764
7765 \snippet qstring/main.cpp 73
7766
7767 This function ignores leading and trailing whitespace.
7768
7769 \note Support for the "0b" prefix was added in Qt 6.4.
7770
7771 \sa number(), toULong(), toInt(), QLocale::toInt()
7772*/
7773
7774/*!
7775 \fn ulong QString::toULong(bool *ok, int base) const
7776
7777 Returns the string converted to an \c{unsigned long} using base \a
7778 base, which is 10 by default and must be between 2 and 36, or 0.
7779 Returns 0 if the conversion fails.
7780
7781 If \a ok is not \nullptr, failure is reported by setting *\a{ok}
7782 to \c false, and success by setting *\a{ok} to \c true.
7783
7784 If \a base is 0, the C language convention is used: if the string begins
7785 with "0x", base 16 is used; otherwise, if the string begins with "0b", base
7786 2 is used; otherwise, if the string begins with "0", base 8 is used;
7787 otherwise, base 10 is used.
7788
7789 The string conversion will always happen in the 'C' locale. For
7790 locale-dependent conversion use QLocale::toULongLong()
7791
7792 Example:
7793
7794 \snippet qstring/main.cpp 78
7795
7796 This function ignores leading and trailing whitespace.
7797
7798 \note Support for the "0b" prefix was added in Qt 6.4.
7799
7800 \sa number(), QLocale::toUInt()
7801*/
7802
7803/*!
7804 \fn int QString::toInt(bool *ok, int base) const
7805 Returns the string converted to an \c int using base \a
7806 base, which is 10 by default and must be between 2 and 36, or 0.
7807 Returns 0 if the conversion fails.
7808
7809 If \a ok is not \nullptr, failure is reported by setting *\a{ok}
7810 to \c false, and success by setting *\a{ok} to \c true.
7811
7812 If \a base is 0, the C language convention is used: if the string begins
7813 with "0x", base 16 is used; otherwise, if the string begins with "0b", base
7814 2 is used; otherwise, if the string begins with "0", base 8 is used;
7815 otherwise, base 10 is used.
7816
7817 The string conversion will always happen in the 'C' locale. For
7818 locale-dependent conversion use QLocale::toInt()
7819
7820 Example:
7821
7822 \snippet qstring/main.cpp 72
7823
7824 This function ignores leading and trailing whitespace.
7825
7826 \note Support for the "0b" prefix was added in Qt 6.4.
7827
7828 \sa number(), toUInt(), toDouble(), QLocale::toInt()
7829*/
7830
7831/*!
7832 \fn uint QString::toUInt(bool *ok, int base) const
7833 Returns the string converted to an \c{unsigned int} using base \a
7834 base, which is 10 by default and must be between 2 and 36, or 0.
7835 Returns 0 if the conversion fails.
7836
7837 If \a ok is not \nullptr, failure is reported by setting *\a{ok}
7838 to \c false, and success by setting *\a{ok} to \c true.
7839
7840 If \a base is 0, the C language convention is used: if the string begins
7841 with "0x", base 16 is used; otherwise, if the string begins with "0b", base
7842 2 is used; otherwise, if the string begins with "0", base 8 is used;
7843 otherwise, base 10 is used.
7844
7845 The string conversion will always happen in the 'C' locale. For
7846 locale-dependent conversion use QLocale::toUInt()
7847
7848 Example:
7849
7850 \snippet qstring/main.cpp 77
7851
7852 This function ignores leading and trailing whitespace.
7853
7854 \note Support for the "0b" prefix was added in Qt 6.4.
7855
7856 \sa number(), toInt(), QLocale::toUInt()
7857*/
7858
7859/*!
7860 \fn short QString::toShort(bool *ok, int base) const
7861
7862 Returns the string converted to a \c short using base \a
7863 base, which is 10 by default and must be between 2 and 36, or 0.
7864 Returns 0 if the conversion fails.
7865
7866 If \a ok is not \nullptr, failure is reported by setting *\a{ok}
7867 to \c false, and success by setting *\a{ok} to \c true.
7868
7869 If \a base is 0, the C language convention is used: if the string begins
7870 with "0x", base 16 is used; otherwise, if the string begins with "0b", base
7871 2 is used; otherwise, if the string begins with "0", base 8 is used;
7872 otherwise, base 10 is used.
7873
7874 The string conversion will always happen in the 'C' locale. For
7875 locale-dependent conversion use QLocale::toShort()
7876
7877 Example:
7878
7879 \snippet qstring/main.cpp 76
7880
7881 This function ignores leading and trailing whitespace.
7882
7883 \note Support for the "0b" prefix was added in Qt 6.4.
7884
7885 \sa number(), toUShort(), toInt(), QLocale::toShort()
7886*/
7887
7888/*!
7889 \fn ushort QString::toUShort(bool *ok, int base) const
7890
7891 Returns the string converted to an \c{unsigned short} using base \a
7892 base, which is 10 by default and must be between 2 and 36, or 0.
7893 Returns 0 if the conversion fails.
7894
7895 If \a ok is not \nullptr, failure is reported by setting *\a{ok}
7896 to \c false, and success by setting *\a{ok} to \c true.
7897
7898 If \a base is 0, the C language convention is used: if the string begins
7899 with "0x", base 16 is used; otherwise, if the string begins with "0b", base
7900 2 is used; otherwise, if the string begins with "0", base 8 is used;
7901 otherwise, base 10 is used.
7902
7903 The string conversion will always happen in the 'C' locale. For
7904 locale-dependent conversion use QLocale::toUShort()
7905
7906 Example:
7907
7908 \snippet qstring/main.cpp 80
7909
7910 This function ignores leading and trailing whitespace.
7911
7912 \note Support for the "0b" prefix was added in Qt 6.4.
7913
7914 \sa number(), toShort(), QLocale::toUShort()
7915*/
7916
7917/*!
7918 Returns the string converted to a \c double value.
7919
7920 Returns an infinity if the conversion overflows or 0.0 if the
7921 conversion fails for other reasons (e.g. underflow).
7922
7923 If \a ok is not \nullptr, failure is reported by setting *\a{ok}
7924 to \c false, and success by setting *\a{ok} to \c true.
7925
7926 \snippet qstring/main.cpp 66
7927
7928 \warning The QString content may only contain valid numerical characters
7929 which includes the plus/minus sign, the character e used in scientific
7930 notation, and the decimal point. Including the unit or additional characters
7931 leads to a conversion error.
7932
7933 \snippet qstring/main.cpp 67
7934
7935 The string conversion will always happen in the 'C' locale. For
7936 locale-dependent conversion use QLocale::toDouble()
7937
7938 \snippet qstring/main.cpp 68
7939
7940 For historical reasons, this function does not handle
7941 thousands group separators. If you need to convert such numbers,
7942 use QLocale::toDouble().
7943
7944 \snippet qstring/main.cpp 69
7945
7946 This function ignores leading and trailing whitespace.
7947
7948 \sa number(), QLocale::setDefault(), QLocale::toDouble(), trimmed()
7949*/
7950
7951double QString::toDouble(bool *ok) const
7952{
7953 return QStringView(*this).toDouble(ok);
7954}
7955
7956double QStringView::toDouble(bool *ok) const
7957{
7958 QStringView string = qt_trimmed(*this);
7959 QVarLengthArray<uchar> latin1(string.size());
7960 qt_to_latin1(latin1.data(), string.utf16(), string.size());
7961 auto r = qt_asciiToDouble(reinterpret_cast<const char *>(latin1.data()), string.size());
7962 if (ok != nullptr)
7963 *ok = r.ok();
7964 return r.result;
7965}
7966
7967/*!
7968 Returns the string converted to a \c float value.
7969
7970 Returns an infinity if the conversion overflows or 0.0 if the
7971 conversion fails for other reasons (e.g. underflow).
7972
7973 If \a ok is not \nullptr, failure is reported by setting *\a{ok}
7974 to \c false, and success by setting *\a{ok} to \c true.
7975
7976 \warning The QString content may only contain valid numerical characters
7977 which includes the plus/minus sign, the character e used in scientific
7978 notation, and the decimal point. Including the unit or additional characters
7979 leads to a conversion error.
7980
7981 The string conversion will always happen in the 'C' locale. For
7982 locale-dependent conversion use QLocale::toFloat()
7983
7984 For historical reasons, this function does not handle
7985 thousands group separators. If you need to convert such numbers,
7986 use QLocale::toFloat().
7987
7988 Example:
7989
7990 \snippet qstring/main.cpp 71
7991
7992 This function ignores leading and trailing whitespace.
7993
7994 \sa number(), toDouble(), toInt(), QLocale::toFloat(), trimmed()
7995*/
7996
7997float QString::toFloat(bool *ok) const
7998{
7999 return QLocaleData::convertDoubleToFloat(toDouble(ok), ok);
8000}
8001
8002float QStringView::toFloat(bool *ok) const
8003{
8004 return QLocaleData::convertDoubleToFloat(toDouble(ok), ok);
8005}
8006
8007/*! \fn QString &QString::setNum(int n, int base)
8008
8009 Sets the string to the printed value of \a n in the specified \a
8010 base, and returns a reference to the string.
8011
8012 The base is 10 by default and must be between 2 and 36.
8013
8014 \snippet qstring/main.cpp 56
8015
8016 The formatting always uses QLocale::C, i.e., English/UnitedStates.
8017 To get a localized string representation of a number, use
8018 QLocale::toString() with the appropriate locale.
8019
8020 \sa number()
8021*/
8022
8023/*! \fn QString &QString::setNum(uint n, int base)
8024
8025 \overload
8026*/
8027
8028/*! \fn QString &QString::setNum(long n, int base)
8029
8030 \overload
8031*/
8032
8033/*! \fn QString &QString::setNum(ulong n, int base)
8034
8035 \overload
8036*/
8037
8038/*!
8039 \overload
8040*/
8041QString &QString::setNum(qlonglong n, int base)
8042{
8043 return *this = number(n, base);
8044}
8045
8046/*!
8047 \overload
8048*/
8049QString &QString::setNum(qulonglong n, int base)
8050{
8051 return *this = number(n, base);
8052}
8053
8054/*! \fn QString &QString::setNum(short n, int base)
8055
8056 \overload
8057*/
8058
8059/*! \fn QString &QString::setNum(ushort n, int base)
8060
8061 \overload
8062*/
8063
8064/*!
8065 \overload
8066
8067 Sets the string to the printed value of \a n, formatted according to the
8068 given \a format and \a precision, and returns a reference to the string.
8069
8070 \sa number(), QLocale::FloatingPointPrecisionOption, {Number formats}
8071*/
8072
8073QString &QString::setNum(double n, char format, int precision)
8074{
8075 return *this = number(n, format, precision);
8076}
8077
8078/*!
8079 \fn QString &QString::setNum(float n, char format, int precision)
8080 \overload
8081
8082 Sets the string to the printed value of \a n, formatted according
8083 to the given \a format and \a precision, and returns a reference
8084 to the string.
8085
8086 The formatting always uses QLocale::C, i.e., English/UnitedStates.
8087 To get a localized string representation of a number, use
8088 QLocale::toString() with the appropriate locale.
8089
8090 \sa number()
8091*/
8092
8093
8094/*!
8095 \fn QString QString::number(long n, int base)
8096
8097 Returns a string equivalent of the number \a n according to the
8098 specified \a base.
8099
8100 The base is 10 by default and must be between 2
8101 and 36. For bases other than 10, \a n is treated as an
8102 unsigned integer.
8103
8104 The formatting always uses QLocale::C, i.e., English/UnitedStates.
8105 To get a localized string representation of a number, use
8106 QLocale::toString() with the appropriate locale.
8107
8108 \snippet qstring/main.cpp 35
8109
8110 \sa setNum()
8111*/
8112
8113QString QString::number(long n, int base)
8114{
8115 return number(qlonglong(n), base);
8116}
8117
8118/*!
8119 \fn QString QString::number(ulong n, int base)
8120
8121 \overload
8122*/
8123QString QString::number(ulong n, int base)
8124{
8125 return number(qulonglong(n), base);
8126}
8127
8128/*!
8129 \overload
8130*/
8131QString QString::number(int n, int base)
8132{
8133 return number(qlonglong(n), base);
8134}
8135
8136/*!
8137 \overload
8138*/
8139QString QString::number(uint n, int base)
8140{
8141 return number(qulonglong(n), base);
8142}
8143
8144/*!
8145 \overload
8146*/
8147QString QString::number(qlonglong n, int base)
8148{
8149#if defined(QT_CHECK_RANGE)
8150 if (base < 2 || base > 36) {
8151 qWarning("QString::setNum: Invalid base (%d)", base);
8152 base = 10;
8153 }
8154#endif
8155 bool negative = n < 0;
8156 /*
8157 Negating std::numeric_limits<qlonglong>::min() hits undefined behavior, so
8158 taking an absolute value has to take a slight detour.
8159 */
8160 return qulltoBasicLatin(negative ? 1u + qulonglong(-(n + 1)) : qulonglong(n), base, negative);
8161}
8162
8163/*!
8164 \overload
8165*/
8166QString QString::number(qulonglong n, int base)
8167{
8168#if defined(QT_CHECK_RANGE)
8169 if (base < 2 || base > 36) {
8170 qWarning("QString::setNum: Invalid base (%d)", base);
8171 base = 10;
8172 }
8173#endif
8174 return qulltoBasicLatin(n, base, false);
8175}
8176
8177
8178/*!
8179 Returns a string representing the floating-point number \a n.
8180
8181 Returns a string that represents \a n, formatted according to the specified
8182 \a format and \a precision.
8183
8184 For formats with an exponent, the exponent will show its sign and have at
8185 least two digits, left-padding the exponent with zero if needed.
8186
8187 \sa setNum(), QLocale::toString(), QLocale::FloatingPointPrecisionOption, {Number formats}
8188*/
8189QString QString::number(double n, char format, int precision)
8190{
8191 QLocaleData::DoubleForm form = QLocaleData::DFDecimal;
8192
8193 switch (QtMiscUtils::toAsciiLower(format)) {
8194 case 'f':
8195 form = QLocaleData::DFDecimal;
8196 break;
8197 case 'e':
8198 form = QLocaleData::DFExponent;
8199 break;
8200 case 'g':
8201 form = QLocaleData::DFSignificantDigits;
8202 break;
8203 default:
8204#if defined(QT_CHECK_RANGE)
8205 qWarning("QString::setNum: Invalid format char '%c'", format);
8206#endif
8207 break;
8208 }
8209
8210 return qdtoBasicLatin(n, form, precision, isAsciiUpper(format));
8211}
8212
8213namespace {
8214template<class ResultList, class StringSource>
8215static ResultList splitString(const StringSource &source, QStringView sep,
8216 Qt::SplitBehavior behavior, Qt::CaseSensitivity cs)
8217{
8218 ResultList list;
8219 typename StringSource::size_type start = 0;
8220 typename StringSource::size_type end;
8221 typename StringSource::size_type extra = 0;
8222 while ((end = QtPrivate::findString(QStringView(source.constData(), source.size()), start + extra, sep, cs)) != -1) {
8223 if (start != end || behavior == Qt::KeepEmptyParts)
8224 list.append(source.sliced(start, end - start));
8225 start = end + sep.size();
8226 extra = (sep.size() == 0 ? 1 : 0);
8227 }
8228 if (start != source.size() || behavior == Qt::KeepEmptyParts)
8229 list.append(source.sliced(start));
8230 return list;
8231}
8232
8233} // namespace
8234
8235/*!
8236 Splits the string into substrings wherever \a sep occurs, and
8237 returns the list of those strings. If \a sep does not match
8238 anywhere in the string, split() returns a single-element list
8239 containing this string.
8240
8241 \a cs specifies whether \a sep should be matched case
8242 sensitively or case insensitively.
8243
8244 If \a behavior is Qt::SkipEmptyParts, empty entries don't
8245 appear in the result. By default, empty entries are kept.
8246
8247 Example:
8248
8249 \snippet qstring/main.cpp 62
8250
8251 If \a sep is empty, split() returns an empty string, followed
8252 by each of the string's characters, followed by another empty string:
8253
8254 \snippet qstring/main.cpp 62-empty
8255
8256 To understand this behavior, recall that the empty string matches
8257 everywhere, so the above is qualitatively the same as:
8258
8259 \snippet qstring/main.cpp 62-slashes
8260
8261 \sa QStringList::join(), section()
8262
8263 \since 5.14
8264*/
8265QStringList QString::split(const QString &sep, Qt::SplitBehavior behavior, Qt::CaseSensitivity cs) const
8266{
8267 return splitString<QStringList>(*this, sep, behavior, cs);
8268}
8269
8270/*!
8271 \overload
8272 \since 5.14
8273*/
8274QStringList QString::split(QChar sep, Qt::SplitBehavior behavior, Qt::CaseSensitivity cs) const
8275{
8276 return splitString<QStringList>(*this, QStringView(&sep, 1), behavior, cs);
8277}
8278
8279/*!
8280 \fn QList<QStringView> QStringView::split(QChar sep, Qt::SplitBehavior behavior, Qt::CaseSensitivity cs) const
8281 \fn QList<QStringView> QStringView::split(QStringView sep, Qt::SplitBehavior behavior, Qt::CaseSensitivity cs) const
8282
8283
8284 Splits the view into substring views wherever \a sep occurs, and
8285 returns the list of those string views.
8286
8287 See QString::split() for how \a sep, \a behavior and \a cs interact to form
8288 the result.
8289
8290 \note All the returned views are valid as long as the data referenced by
8291 this string view is valid. Destroying the data will cause all views to
8292 become dangling.
8293
8294 \since 6.0
8295*/
8296QList<QStringView> QStringView::split(QStringView sep, Qt::SplitBehavior behavior, Qt::CaseSensitivity cs) const
8297{
8298 return splitString<QList<QStringView>>(QStringView(*this), sep, behavior, cs);
8299}
8300
8301QList<QStringView> QStringView::split(QChar sep, Qt::SplitBehavior behavior, Qt::CaseSensitivity cs) const
8302{
8303 return split(QStringView(&sep, 1), behavior, cs);
8304}
8305
8306#if QT_CONFIG(regularexpression)
8307namespace {
8308template<class ResultList, typename String, typename MatchingFunction>
8309static ResultList splitString(const String &source, const QRegularExpression &re,
8310 MatchingFunction matchingFunction,
8311 Qt::SplitBehavior behavior)
8312{
8313 ResultList list;
8314 if (!re.isValid()) {
8315 qtWarnAboutInvalidRegularExpression(re, "QString", "split");
8316 return list;
8317 }
8318
8319 qsizetype start = 0;
8320 qsizetype end = 0;
8321 QRegularExpressionMatchIterator iterator = (re.*matchingFunction)(source, 0, QRegularExpression::NormalMatch, QRegularExpression::NoMatchOption);
8322 while (iterator.hasNext()) {
8323 QRegularExpressionMatch match = iterator.next();
8324 end = match.capturedStart();
8325 if (start != end || behavior == Qt::KeepEmptyParts)
8326 list.append(source.sliced(start, end - start));
8327 start = match.capturedEnd();
8328 }
8329
8330 if (start != source.size() || behavior == Qt::KeepEmptyParts)
8331 list.append(source.sliced(start));
8332
8333 return list;
8334}
8335} // namespace
8336
8337/*!
8338 \overload
8339 \since 5.14
8340
8341 Splits the string into substrings wherever the regular expression
8342 \a re matches, and returns the list of those strings. If \a re
8343 does not match anywhere in the string, split() returns a
8344 single-element list containing this string.
8345
8346 Here is an example where we extract the words in a sentence
8347 using one or more whitespace characters as the separator:
8348
8349 \snippet qstring/main.cpp 90
8350
8351 Here is a similar example, but this time we use any sequence of
8352 non-word characters as the separator:
8353
8354 \snippet qstring/main.cpp 91
8355
8356 Here is a third example where we use a zero-length assertion,
8357 \b{\\b} (word boundary), to split the string into an
8358 alternating sequence of non-word and word tokens:
8359
8360 \snippet qstring/main.cpp 92
8361
8362 \sa QStringList::join(), section()
8363*/
8364QStringList QString::split(const QRegularExpression &re, Qt::SplitBehavior behavior) const
8365{
8366#if QT_VERSION < QT_VERSION_CHECK(7, 0, 0)
8367 const auto matchingFunction = qOverload<const QString &, qsizetype, QRegularExpression::MatchType, QRegularExpression::MatchOptions>(&QRegularExpression::globalMatch);
8368#else
8369 const auto matchingFunction = &QRegularExpression::globalMatch;
8370#endif
8371 return splitString<QStringList>(*this,
8372 re,
8373 matchingFunction,
8374 behavior);
8375}
8376
8377/*!
8378 \overload
8379 \since 6.0
8380
8381 Splits the string into substring views wherever the regular expression \a re
8382 matches, and returns the list of those strings. If \a re does not match
8383 anywhere in the string, split() returns a single-element list containing
8384 this string as view.
8385
8386 \note The views in the returned list are sub-views of this view; as such,
8387 they reference the same data as it and only remain valid for as long as that
8388 data remains live.
8389*/
8390QList<QStringView> QStringView::split(const QRegularExpression &re, Qt::SplitBehavior behavior) const
8391{
8392 return splitString<QList<QStringView>>(*this, re, &QRegularExpression::globalMatchView, behavior);
8393}
8394
8395#endif // QT_CONFIG(regularexpression)
8396
8397/*!
8398 \enum QString::NormalizationForm
8399
8400 This enum describes the various normalized forms of Unicode text.
8401
8402 \value NormalizationForm_D Canonical Decomposition
8403 \value NormalizationForm_C Canonical Decomposition followed by Canonical Composition
8404 \value NormalizationForm_KD Compatibility Decomposition
8405 \value NormalizationForm_KC Compatibility Decomposition followed by Canonical Composition
8406
8407 \sa normalized(),
8408 {https://www.unicode.org/reports/tr15/}{Unicode Standard Annex #15}
8409*/
8410
8411/*!
8412 \since 4.5
8413
8414 Returns a copy of this string repeated the specified number of \a times.
8415
8416 If \a times is less than 1, an empty string is returned.
8417
8418 Example:
8419
8420 \snippet code/src_corelib_text_qstring.cpp 8
8421*/
8422QString QString::repeated(qsizetype times) const
8423{
8424 if (d.size == 0)
8425 return *this;
8426
8427 if (times <= 1) {
8428 if (times == 1)
8429 return *this;
8430 return QString();
8431 }
8432
8433 const qsizetype resultSize = times * d.size;
8434
8435 QString result;
8436 result.reserve(resultSize);
8437 if (result.capacity() != resultSize)
8438 return QString(); // not enough memory
8439
8440 memcpy(result.d.data(), d.data(), d.size * sizeof(QChar));
8441
8442 qsizetype sizeSoFar = d.size;
8443 char16_t *end = result.d.data() + sizeSoFar;
8444
8445 const qsizetype halfResultSize = resultSize >> 1;
8446 while (sizeSoFar <= halfResultSize) {
8447 memcpy(end, result.d.data(), sizeSoFar * sizeof(QChar));
8448 end += sizeSoFar;
8449 sizeSoFar <<= 1;
8450 }
8451 memcpy(end, result.d.data(), (resultSize - sizeSoFar) * sizeof(QChar));
8452 result.d.data()[resultSize] = '\0';
8453 result.d.size = resultSize;
8454 return result;
8455}
8456
8457void qt_string_normalize(QString *data, QString::NormalizationForm mode, QChar::UnicodeVersion version, qsizetype from)
8458{
8459 {
8460 // check if it's fully ASCII first, because then we have no work
8461 auto start = reinterpret_cast<const char16_t *>(data->constData());
8462 const char16_t *p = start + from;
8463 if (isAscii_helper(p, p + data->size() - from))
8464 return;
8465 if (p > start + from)
8466 from = p - start - 1; // need one before the non-ASCII to perform NFC
8467 }
8468
8469 if (version == QChar::Unicode_Unassigned) {
8470 version = QChar::currentUnicodeVersion();
8471 } else if (int(version) <= NormalizationCorrectionsVersionMax) {
8472 const QString &s = *data;
8473 QChar *d = nullptr;
8475 if (n.version > version) {
8476 qsizetype pos = from;
8477 if (QChar::requiresSurrogates(n.ucs4)) {
8478 char16_t ucs4High = QChar::highSurrogate(n.ucs4);
8479 char16_t ucs4Low = QChar::lowSurrogate(n.ucs4);
8480
8481 // scan for this codepoint
8482 for ( ; pos < s.size() - 1; ++pos) {
8483 if (s.at(pos).unicode() == ucs4High && s.at(pos + 1).unicode() == ucs4Low)
8484 break;
8485 }
8486 if (pos == s.size())
8487 continue; // no correction necessary
8488
8489 // detach if necessary
8490 if (!d)
8491 d = data->data();
8492 if (QChar::requiresSurrogates(n.old_mapping)) {
8493 // no shrinking
8494 char16_t oldHigh = QChar::highSurrogate(n.old_mapping);
8495 char16_t oldLow = QChar::lowSurrogate(n.old_mapping);
8496 while (pos < s.size() - 1) {
8497 if (s.at(pos).unicode() == ucs4High && s.at(pos + 1).unicode() == ucs4Low) {
8498 d[pos] = QChar(oldHigh);
8499 d[++pos] = QChar(oldLow);
8500 }
8501 ++pos;
8502 }
8503 } else {
8504 // shrinking, so a little harder
8505 char16_t old = char16_t(n.old_mapping);
8506 qsizetype outpos = pos;
8507 for ( ; pos < s.size(); ++outpos, ++pos) {
8508 if (pos < s.size() - 1 && s.at(pos).unicode() == ucs4High
8509 && s.at(pos + 1).unicode() == ucs4Low) {
8510 d[outpos] = QChar(old);
8511 ++pos;
8512 }
8513 }
8514 data->truncate(outpos);
8515 d = nullptr;
8516 }
8517 } else {
8518 Q_ASSERT(!QChar::requiresSurrogates(n.old_mapping)); // BMP maps to BMP
8519 while (pos < s.size()) {
8520 if (s.at(pos).unicode() == n.ucs4) {
8521 if (!d)
8522 d = data->data();
8523 d[pos] = QChar(n.old_mapping);
8524 }
8525 ++pos;
8526 }
8527 }
8528 }
8529 }
8530 }
8531
8532 if (normalizationQuickCheckHelper(data, mode, from, &from))
8533 return;
8534
8535 decomposeHelper(data, mode < QString::NormalizationForm_KD, version, from);
8536
8537 canonicalOrderHelper(data, version, from);
8538
8539 if (mode == QString::NormalizationForm_D || mode == QString::NormalizationForm_KD)
8540 return;
8541
8542 composeHelper(data, version, from);
8543}
8544
8545/*!
8546 Returns the string in the given Unicode normalization \a mode,
8547 according to the given \a version of the Unicode standard.
8548*/
8549QString QString::normalized(QString::NormalizationForm mode, QChar::UnicodeVersion version) const
8550{
8551 QString copy = *this;
8552 qt_string_normalize(&copy, mode, version, 0);
8553 return copy;
8554}
8555
8556#if QT_VERSION < QT_VERSION_CHECK(7, 0, 0) && !defined(QT_BOOTSTRAPPED)
8557static void checkArgEscape(QStringView s)
8558{
8559 // If we're in here, it means that qArgDigitValue has accepted the
8560 // digit. We can skip the check in case we already know it will
8561 // succeed.
8562 if (!supportUnicodeDigitValuesInArg())
8563 return;
8564
8565 const auto isNonAsciiDigit = [](QChar c) {
8566 return c.unicode() < u'0' || c.unicode() > u'9';
8567 };
8568
8569 if (std::any_of(s.begin(), s.end(), isNonAsciiDigit)) {
8570 const auto accumulateDigit = [](int partial, QChar digit) {
8571 return partial * 10 + digit.digitValue();
8572 };
8573 const int parsedNumber = std::accumulate(s.begin(), s.end(), 0, accumulateDigit);
8574
8575 qWarning("QString::arg(): the replacement \"%%%ls\" contains non-ASCII digits;\n"
8576 " it is currently being interpreted as the %d-th substitution.\n"
8577 " This is deprecated; support for non-ASCII digits will be dropped\n"
8578 " in a future version of Qt.",
8579 qUtf16Printable(s.toString()),
8580 parsedNumber);
8581 }
8582}
8583#endif
8584
8586{
8587 int min_escape; // lowest escape sequence number
8588 qsizetype occurrences; // number of occurrences of the lowest escape sequence number
8589 qsizetype locale_occurrences; // number of occurrences of the lowest escape sequence number that
8590 // contain 'L'
8591 qsizetype escape_len; // total length of escape sequences which will be replaced
8592};
8593
8594static ArgEscapeData findArgEscapes(QStringView s)
8595{
8596 const QChar *uc_begin = s.begin();
8597 const QChar *uc_end = s.end();
8598
8599 ArgEscapeData d;
8600
8601 d.min_escape = INT_MAX;
8602 d.occurrences = 0;
8603 d.escape_len = 0;
8604 d.locale_occurrences = 0;
8605
8606 const QChar *c = uc_begin;
8607 while (c != uc_end) {
8608 while (c != uc_end && c->unicode() != '%')
8609 ++c;
8610
8611 if (c == uc_end)
8612 break;
8613 const QChar *escape_start = c;
8614 if (++c == uc_end)
8615 break;
8616
8617 bool locale_arg = false;
8618 if (c->unicode() == 'L') {
8619 locale_arg = true;
8620 if (++c == uc_end)
8621 break;
8622 }
8623
8624 int escape = qArgDigitValue(*c);
8625 if (escape == -1)
8626 continue;
8627
8628 // ### Qt 7: do not allow anything but ASCII digits
8629 // in arg()'s replacements.
8630#if QT_VERSION <= QT_VERSION_CHECK(7, 0, 0) && !defined(QT_BOOTSTRAPPED)
8631 const QChar *escapeBegin = c;
8632 const QChar *escapeEnd = escapeBegin + 1;
8633#endif
8634
8635 ++c;
8636
8637 if (c != uc_end) {
8638 const int next_escape = qArgDigitValue(*c);
8639 if (next_escape != -1) {
8640 escape = (10 * escape) + next_escape;
8641 ++c;
8642#if QT_VERSION <= QT_VERSION_CHECK(7, 0, 0) && !defined(QT_BOOTSTRAPPED)
8643 ++escapeEnd;
8644#endif
8645 }
8646 }
8647
8648#if QT_VERSION <= QT_VERSION_CHECK(7, 0, 0) && !defined(QT_BOOTSTRAPPED)
8649 checkArgEscape(QStringView(escapeBegin, escapeEnd));
8650#endif
8651
8652 if (escape > d.min_escape)
8653 continue;
8654
8655 if (escape < d.min_escape) {
8656 d.min_escape = escape;
8657 d.occurrences = 0;
8658 d.escape_len = 0;
8659 d.locale_occurrences = 0;
8660 }
8661
8662 ++d.occurrences;
8663 if (locale_arg)
8664 ++d.locale_occurrences;
8665 d.escape_len += c - escape_start;
8666 }
8667 return d;
8668}
8669
8670static QString replaceArgEscapes(QStringView s, const ArgEscapeData &d, qsizetype field_width,
8671 QStringView arg, QStringView larg, QChar fillChar)
8672{
8673 // Negative field-width for right-padding, positive for left-padding:
8674 const qsizetype abs_field_width = qAbs(field_width);
8675 const qsizetype result_len =
8676 s.size() - d.escape_len
8677 + (d.occurrences - d.locale_occurrences) * qMax(abs_field_width, arg.size())
8678 + d.locale_occurrences * qMax(abs_field_width, larg.size());
8679
8680 QString result(result_len, Qt::Uninitialized);
8681 QChar *rc = const_cast<QChar *>(result.unicode());
8682 QChar *const result_end = rc + result_len;
8683 qsizetype repl_cnt = 0;
8684
8685 const QChar *c = s.begin();
8686 const QChar *const uc_end = s.end();
8687 while (c != uc_end) {
8688 Q_ASSERT(d.occurrences > repl_cnt);
8689 /* We don't have to check increments of c against uc_end because, as
8690 long as d.occurrences > repl_cnt, we KNOW there are valid escape
8691 sequences remaining. */
8692
8693 const QChar *text_start = c;
8694 while (c->unicode() != '%')
8695 ++c;
8696
8697 const QChar *escape_start = c++;
8698 const bool localize = c->unicode() == 'L';
8699 if (localize)
8700 ++c;
8701
8702 int escape = qArgDigitValue(*c);
8703 if (escape != -1 && c + 1 != uc_end) {
8704 const int digit = qArgDigitValue(c[1]);
8705 if (digit != -1) {
8706 ++c;
8707 escape = 10 * escape + digit;
8708 }
8709 }
8710
8711 if (escape != d.min_escape) {
8712 memcpy(rc, text_start, (c - text_start) * sizeof(QChar));
8713 rc += c - text_start;
8714 } else {
8715 ++c;
8716
8717 memcpy(rc, text_start, (escape_start - text_start) * sizeof(QChar));
8718 rc += escape_start - text_start;
8719
8720 const QStringView use = localize ? larg : arg;
8721 const qsizetype pad_chars = abs_field_width - use.size();
8722 // (If negative, relevant loops are no-ops: no need to check.)
8723
8724 if (field_width > 0) { // left padded
8725 rc = std::fill_n(rc, pad_chars, fillChar);
8726 }
8727
8728 if (use.size())
8729 memcpy(rc, use.data(), use.size() * sizeof(QChar));
8730 rc += use.size();
8731
8732 if (field_width < 0) { // right padded
8733 rc = std::fill_n(rc, pad_chars, fillChar);
8734 }
8735
8736 if (++repl_cnt == d.occurrences) {
8737 memcpy(rc, c, (uc_end - c) * sizeof(QChar));
8738 rc += uc_end - c;
8739 Q_ASSERT(rc == result_end);
8740 c = uc_end;
8741 }
8742 }
8743 }
8744 Q_ASSERT(rc == result_end);
8745
8746 return result;
8747}
8748
8749/*!
8750 \fn template <typename T, QString::if_string_like<T> = true> QString QString::arg(const T &a, int fieldWidth, QChar fillChar) const
8751
8752 Returns a copy of this string with the lowest-numbered place-marker
8753 replaced by string \a a, i.e., \c %1, \c %2, ..., \c %99.
8754
8755 \a fieldWidth specifies the minimum amount of space that \a a
8756 shall occupy. If \a a requires less space than \a fieldWidth, it
8757 is padded to \a fieldWidth with character \a fillChar. A positive
8758 \a fieldWidth produces right-aligned text. A negative \a fieldWidth
8759 produces left-aligned text.
8760
8761 This example shows how we might create a \c status string for
8762 reporting progress while processing a list of files:
8763
8764 \snippet qstring/main.cpp 11-qstringview
8765
8766 First, \c arg(i) replaces \c %1. Then \c arg(total) replaces \c
8767 %2. Finally, \c arg(fileName) replaces \c %3.
8768
8769 One advantage of using arg() over asprintf() is that the order of the
8770 numbered place markers can change, if the application's strings are
8771 translated into other languages, but each arg() will still replace
8772 the lowest-numbered unreplaced place-marker, no matter where it
8773 appears. Also, if place-marker \c %i appears more than once in the
8774 string, arg() replaces all of them.
8775
8776 If there is no unreplaced place-marker remaining, a warning message
8777 is printed and the result is undefined. Place-marker numbers must be
8778 in the range 1 to 99.
8779
8780 \note In Qt versions prior to 6.9, this function was overloaded on
8781 \c{char}, QChar, QString, QStringView, and QLatin1StringView and in some
8782 cases, \c{wchar_t} and \c{char16_t} arguments would resolve to the integer
8783 overloads. In Qt versions prior to 5.10, this function lacked the
8784 QStringView and QLatin1StringView overloads.
8785*/
8786QString QString::arg_impl(QAnyStringView a, int fieldWidth, QChar fillChar) const
8787{
8788 ArgEscapeData d = findArgEscapes(*this);
8789
8790 if (Q_UNLIKELY(d.occurrences == 0)) {
8791 qWarning("QString::arg: Argument missing: \"%ls\", \"%ls\"", qUtf16Printable(*this),
8792 qUtf16Printable(a.toString()));
8793 return *this;
8794 }
8795 struct {
8796 QVarLengthArray<char16_t> out;
8797 QStringView operator()(QStringView in) noexcept { return in; }
8798 QStringView operator()(QLatin1StringView in)
8799 {
8800 out.resize(in.size());
8801 qt_from_latin1(out.data(), in.data(), size_t(in.size()));
8802 return out;
8803 }
8804 QStringView operator()(QUtf8StringView in)
8805 {
8806 out.resize(in.size());
8807 return QStringView{out.data(), QUtf8::convertToUnicode(out.data(), in)};
8808 }
8809 } convert;
8810
8811 QStringView sv = a.visit(std::ref(convert));
8812 return replaceArgEscapes(*this, d, fieldWidth, sv, sv, fillChar);
8813}
8814
8815/*!
8816 \fn template <typename T, QString::if_integral_non_char<T> = true> QString QString::arg(T a, int fieldWidth, int base, QChar fillChar) const
8817 \overload arg()
8818
8819 The \a a argument is expressed in base \a base, which is 10 by
8820 default and must be between 2 and 36. For bases other than 10, \a a
8821 is treated as an unsigned integer.
8822
8823 \a fieldWidth specifies the minimum amount of space that \a a is
8824 padded to and filled with the character \a fillChar. A positive
8825 value produces right-aligned text; a negative value produces
8826 left-aligned text.
8827
8828 The '%' can be followed by an 'L', in which case the sequence is
8829 replaced with a localized representation of \a a. The conversion
8830 uses the default locale, set by QLocale::setDefault(). If no default
8831 locale was specified, the system locale is used. The 'L' flag is
8832 ignored if \a base is not 10.
8833
8834 \snippet qstring/main.cpp 12
8835 \snippet qstring/main.cpp 14
8836
8837 \note In Qt versions prior to 6.10.1, this function accepted arguments of
8838 types that implicitly convert to integral types. This is no longer supported,
8839 except for (unscoped) enums, because it also accepted types convertible to
8840 floating-point types, losing precision when those were printed as integers. A
8841 backwards-compatible fix is to cast such types to a C++ type whose displayed
8842 form matches your intent (\c int, \c float, ...).
8843
8844 \note In Qt versions prior to 6.9, this function was overloaded on various
8845 integral types and sometimes incorrectly accepted \c char and \c char16_t
8846 arguments.
8847
8848 \sa {Number formats}
8849*/
8850QString QString::arg_impl(qlonglong a, int fieldWidth, int base, QChar fillChar) const
8851{
8852 ArgEscapeData d = findArgEscapes(*this);
8853
8854 if (d.occurrences == 0) {
8855 qWarning("QString::arg: Argument missing: \"%ls\", %llu", qUtf16Printable(*this), a);
8856 return *this;
8857 }
8858
8859 unsigned flags = QLocaleData::NoFlags;
8860 // ZeroPadded sorts out left-padding when the fill is zero, to the right of sign:
8861 if (fillChar == u'0')
8862 flags = QLocaleData::ZeroPadded;
8863
8864 QString arg;
8865 if (d.occurrences > d.locale_occurrences) {
8866 arg = QLocaleData::c()->longLongToString(a, -1, base, fieldWidth, flags);
8867 Q_ASSERT(fillChar != u'0' || fieldWidth <= arg.size());
8868 }
8869
8870 QString localeArg;
8871 if (d.locale_occurrences > 0) {
8872 QLocale locale;
8873 if (!(locale.numberOptions() & QLocale::OmitGroupSeparator))
8874 flags |= QLocaleData::GroupDigits;
8875 localeArg = locale.d->m_data->longLongToString(a, -1, base, fieldWidth, flags);
8876 Q_ASSERT(fillChar != u'0' || fieldWidth <= localeArg.size());
8877 }
8878
8879 return replaceArgEscapes(*this, d, fieldWidth, arg, localeArg, fillChar);
8880}
8881
8882QString QString::arg_impl(qulonglong a, int fieldWidth, int base, QChar fillChar) const
8883{
8884 ArgEscapeData d = findArgEscapes(*this);
8885
8886 if (d.occurrences == 0) {
8887 qWarning("QString::arg: Argument missing: \"%ls\", %lld", qUtf16Printable(*this), a);
8888 return *this;
8889 }
8890
8891 unsigned flags = QLocaleData::NoFlags;
8892 // ZeroPadded sorts out left-padding when the fill is zero, to the right of sign:
8893 if (fillChar == u'0')
8894 flags = QLocaleData::ZeroPadded;
8895
8896 QString arg;
8897 if (d.occurrences > d.locale_occurrences) {
8898 arg = QLocaleData::c()->unsLongLongToString(a, -1, base, fieldWidth, flags);
8899 Q_ASSERT(fillChar != u'0' || fieldWidth <= arg.size());
8900 }
8901
8902 QString localeArg;
8903 if (d.locale_occurrences > 0) {
8904 QLocale locale;
8905 if (!(locale.numberOptions() & QLocale::OmitGroupSeparator))
8906 flags |= QLocaleData::GroupDigits;
8907 localeArg = locale.d->m_data->unsLongLongToString(a, -1, base, fieldWidth, flags);
8908 Q_ASSERT(fillChar != u'0' || fieldWidth <= localeArg.size());
8909 }
8910
8911 return replaceArgEscapes(*this, d, fieldWidth, arg, localeArg, fillChar);
8912}
8913
8914/*!
8915 \fn template <typename T, QString::if_floating_point<T> = true> QString QString::arg(T a, int fieldWidth, char format, int precision, QChar fillChar) const
8916 \overload arg()
8917
8918 Argument \a a is formatted according to the specified \a format and
8919 \a precision. See \l{Floating-point Formats} for details.
8920
8921 \a fieldWidth specifies the minimum amount of space that \a a is
8922 padded to and filled with the character \a fillChar. A positive
8923 value produces right-aligned text; a negative value produces
8924 left-aligned text.
8925
8926 \snippet code/src_corelib_text_qstring.cpp 2
8927
8928 \note In Qt versions prior to 6.9, this function was a regular function
8929 taking \c double. As a consequence of being a template function now, it no
8930 longer accepts arguments that merely implicitly convert to floating-point
8931 types. A backwards-compatible fix is to cast such types to one of the C++
8932 floating-point types.
8933
8934 \sa QLocale::toString(), QLocale::FloatingPointPrecisionOption, {Number formats}
8935*/
8936QString QString::arg_impl(double a, int fieldWidth, char format, int precision, QChar fillChar) const
8937{
8938 ArgEscapeData d = findArgEscapes(*this);
8939
8940 if (d.occurrences == 0) {
8941 qWarning("QString::arg: Argument missing: \"%ls\", %g", qUtf16Printable(*this), a);
8942 return *this;
8943 }
8944
8945 unsigned flags = QLocaleData::NoFlags;
8946 // ZeroPadded sorts out left-padding when the fill is zero, to the right of sign:
8947 if (fillChar == u'0')
8948 flags |= QLocaleData::ZeroPadded;
8949
8950 if (isAsciiUpper(format))
8951 flags |= QLocaleData::CapitalEorX;
8952
8953 QLocaleData::DoubleForm form = QLocaleData::DFDecimal;
8954 switch (QtMiscUtils::toAsciiLower(format)) {
8955 case 'f':
8956 form = QLocaleData::DFDecimal;
8957 break;
8958 case 'e':
8959 form = QLocaleData::DFExponent;
8960 break;
8961 case 'g':
8962 form = QLocaleData::DFSignificantDigits;
8963 break;
8964 default:
8965#if defined(QT_CHECK_RANGE)
8966 qWarning("QString::arg: Invalid format char '%c'", format);
8967#endif
8968 break;
8969 }
8970
8971 QString arg;
8972 if (d.occurrences > d.locale_occurrences) {
8973 arg = QLocaleData::c()->doubleToString(a, precision, form, fieldWidth,
8974 flags | QLocaleData::ZeroPadExponent);
8975 Q_ASSERT(fillChar != u'0' || !qt_is_finite(a)
8976 || fieldWidth <= arg.size());
8977 }
8978
8979 QString localeArg;
8980 if (d.locale_occurrences > 0) {
8981 QLocale locale;
8982
8983 const QLocale::NumberOptions numberOptions = locale.numberOptions();
8984 if (!(numberOptions & QLocale::OmitGroupSeparator))
8985 flags |= QLocaleData::GroupDigits;
8986 if (!(numberOptions & QLocale::OmitLeadingZeroInExponent))
8987 flags |= QLocaleData::ZeroPadExponent;
8988 if (numberOptions & QLocale::IncludeTrailingZeroesAfterDot)
8989 flags |= QLocaleData::AddTrailingZeroes;
8990 localeArg = locale.d->m_data->doubleToString(a, precision, form, fieldWidth, flags);
8991 Q_ASSERT(fillChar != u'0' || !qt_is_finite(a)
8992 || fieldWidth <= localeArg.size());
8993 }
8994
8995 return replaceArgEscapes(*this, d, fieldWidth, arg, localeArg, fillChar);
8996}
8997
8998static inline char16_t to_unicode(const QChar c) { return c.unicode(); }
8999static inline char16_t to_unicode(const char c) { return QLatin1Char{c}.unicode(); }
9000
9001template <typename Char>
9002static int getEscape(const Char *uc, qsizetype *pos, qsizetype len)
9003{
9004 qsizetype i = *pos;
9005 ++i;
9006 if (i < len && uc[i] == u'L')
9007 ++i;
9008 if (i < len) {
9009 int escape = to_unicode(uc[i]) - '0';
9010 if (uint(escape) >= 10U)
9011 return -1;
9012 ++i;
9013 if (i < len) {
9014 // there's a second digit
9015 int digit = to_unicode(uc[i]) - '0';
9016 if (uint(digit) < 10U) {
9017 escape = (escape * 10) + digit;
9018 ++i;
9019 }
9020 }
9021 *pos = i;
9022 return escape;
9023 }
9024 return -1;
9025}
9026
9027/*
9028 Algorithm for multiArg:
9029
9030 1. Parse the string as a sequence of verbatim text and placeholders (%L?\d{,3}).
9031 The L is parsed and accepted for compatibility with non-multi-arg, but since
9032 multiArg only accepts strings as replacements, the localization request can
9033 be safely ignored.
9034 2. The result of step (1) is a list of (string-ref,int)-tuples. The string-ref
9035 either points at text to be copied verbatim (in which case the int is -1),
9036 or, initially, at the textual representation of the placeholder. In that case,
9037 the int contains the numerical number as parsed from the placeholder.
9038 3. Next, collect all the non-negative ints found, sort them in ascending order and
9039 remove duplicates.
9040 3a. If the result has more entries than multiArg() was given replacement strings,
9041 we have found placeholders we can't satisfy with replacement strings. That is
9042 fine (there could be another .arg() call coming after this one), so just
9043 truncate the result to the number of actual multiArg() replacement strings.
9044 3b. If the result has less entries than multiArg() was given replacement strings,
9045 the string is missing placeholders. This is an error that the user should be
9046 warned about.
9047 4. The result of step (3) is a mapping from the index of any replacement string to
9048 placeholder number. This is the wrong way around, but since placeholder
9049 numbers could get as large as 999, while we typically don't have more than 9
9050 replacement strings, we trade 4K of sparsely-used memory for doing a reverse lookup
9051 each time we need to map a placeholder number to a replacement string index
9052 (that's a linear search; but still *much* faster than using an associative container).
9053 5. Next, for each of the tuples found in step (1), do the following:
9054 5a. If the int is negative, do nothing.
9055 5b. Otherwise, if the int is found in the result of step (3) at index I, replace
9056 the string-ref with a string-ref for the (complete) I'th replacement string.
9057 5c. Otherwise, do nothing.
9058 6. Concatenate all string refs into a single result string.
9059*/
9060
9061namespace {
9062struct Part
9063{
9064 Part() = default; // for QVarLengthArray; do not use
9065 constexpr Part(QAnyStringView s, int num = -1)
9066 : string{s}, number{num} {}
9067
9068 void reset(QAnyStringView s) noexcept { *this = {s, number}; }
9069
9070 QAnyStringView string;
9071 int number;
9072};
9073} // unnamed namespace
9074
9076
9077namespace {
9078
9079enum { ExpectedParts = 32 };
9080
9081typedef QVarLengthArray<Part, ExpectedParts> ParseResult;
9082typedef QVarLengthArray<int, ExpectedParts/2> ArgIndexToPlaceholderMap;
9083
9084template <typename StringView>
9085static ParseResult parseMultiArgFormatString_impl(StringView s)
9086{
9087 ParseResult result;
9088
9089 const auto uc = s.data();
9090 const auto len = s.size();
9091 const auto end = len - 1;
9092 qsizetype i = 0;
9093 qsizetype last = 0;
9094
9095 while (i < end) {
9096 if (uc[i] == u'%') {
9097 qsizetype percent = i;
9098 int number = getEscape(uc, &i, len);
9099 if (number != -1) {
9100 if (last != percent)
9101 result.push_back(Part{s.sliced(last, percent - last)}); // literal text (incl. failed placeholders)
9102 result.push_back(Part{s.sliced(percent, i - percent), number}); // parsed placeholder
9103 last = i;
9104 continue;
9105 }
9106 }
9107 ++i;
9108 }
9109
9110 if (last < len)
9111 result.push_back(Part{s.sliced(last, len - last)}); // trailing literal text
9112
9113 return result;
9114}
9115
9116static ParseResult parseMultiArgFormatString(QAnyStringView s)
9117{
9118 return s.visit([] (auto s) { return parseMultiArgFormatString_impl(s); });
9119}
9120
9121static ArgIndexToPlaceholderMap makeArgIndexToPlaceholderMap(const ParseResult &parts)
9122{
9123 ArgIndexToPlaceholderMap result;
9124
9125 for (const Part &part : parts) {
9126 if (part.number >= 0)
9127 result.push_back(part.number);
9128 }
9129
9130 std::sort(result.begin(), result.end());
9131 result.erase(std::unique(result.begin(), result.end()),
9132 result.end());
9133
9134 return result;
9135}
9136
9137static qsizetype resolveStringRefsAndReturnTotalSize(ParseResult &parts, const ArgIndexToPlaceholderMap &argIndexToPlaceholderMap, const QtPrivate::ArgBase *args[])
9138{
9139 using namespace QtPrivate;
9140 qsizetype totalSize = 0;
9141 for (Part &part : parts) {
9142 if (part.number != -1) {
9143 const auto it = std::find(argIndexToPlaceholderMap.begin(), argIndexToPlaceholderMap.end(), part.number);
9144 if (it != argIndexToPlaceholderMap.end()) {
9145 const auto &arg = *args[it - argIndexToPlaceholderMap.begin()];
9146 switch (arg.tag) {
9147 case ArgBase::L1:
9148 part.reset(static_cast<const QLatin1StringArg&>(arg).string);
9149 break;
9150 case ArgBase::Any:
9151 part.reset(static_cast<const QAnyStringArg&>(arg).string);
9152 break;
9153 case ArgBase::U16:
9154 part.reset(static_cast<const QStringViewArg&>(arg).string);
9155 break;
9156 }
9157 }
9158 }
9159 totalSize += part.string.size();
9160 }
9161 return totalSize;
9162}
9163
9164} // unnamed namespace
9165
9166QString QtPrivate::argToQString(QAnyStringView pattern, size_t numArgs, const ArgBase **args)
9167{
9168 // Step 1-2 above
9169 ParseResult parts = parseMultiArgFormatString(pattern);
9170
9171 // 3-4
9172 ArgIndexToPlaceholderMap argIndexToPlaceholderMap = makeArgIndexToPlaceholderMap(parts);
9173
9174 if (static_cast<size_t>(argIndexToPlaceholderMap.size()) > numArgs) // 3a
9175 argIndexToPlaceholderMap.resize(qsizetype(numArgs));
9176 else if (Q_UNLIKELY(static_cast<size_t>(argIndexToPlaceholderMap.size()) < numArgs)) // 3b
9177 qWarning("QString::arg: %d argument(s) missing in %ls",
9178 int(numArgs - argIndexToPlaceholderMap.size()), qUtf16Printable(pattern.toString()));
9179
9180 // 5
9181 const qsizetype totalSize = resolveStringRefsAndReturnTotalSize(parts, argIndexToPlaceholderMap, args);
9182
9183 // 6:
9184 QString result(totalSize, Qt::Uninitialized);
9185 auto out = const_cast<QChar*>(result.constData());
9186
9187 struct Concatenate {
9188 QChar *out;
9189 QChar *operator()(QLatin1String part) noexcept
9190 {
9191 if (part.size()) {
9192 qt_from_latin1(reinterpret_cast<char16_t*>(out),
9193 part.data(), part.size());
9194 }
9195 return out + part.size();
9196 }
9197 QChar *operator()(QUtf8StringView part) noexcept
9198 {
9199 return QUtf8::convertToUnicode(out, part);
9200 }
9201 QChar *operator()(QStringView part) noexcept
9202 {
9203 if (part.size())
9204 memcpy(out, part.data(), part.size() * sizeof(QChar));
9205 return out + part.size();
9206 }
9207 };
9208
9209 for (const Part &part : parts)
9210 out = part.string.visit(Concatenate{out});
9211
9212 // UTF-8 decoding may have caused an overestimate of totalSize - correct it:
9213 result.truncate(out - result.cbegin());
9214
9215 return result;
9216}
9217
9218/*! \fn bool QString::isRightToLeft() const
9219
9220 Returns \c true if the string is read right to left.
9221
9222 \sa QStringView::isRightToLeft()
9223*/
9224bool QString::isRightToLeft() const
9225{
9226 return QtPrivate::isRightToLeft(QStringView(*this));
9227}
9228
9229/*!
9230 \fn bool QString::isValidUtf16() const noexcept
9231 \since 5.15
9232
9233 Returns \c true if the string contains valid UTF-16 encoded data,
9234 or \c false otherwise.
9235
9236 Note that this function does not perform any special validation of the
9237 data; it merely checks if it can be successfully decoded from UTF-16.
9238 The data is assumed to be in host byte order; the presence of a BOM
9239 is meaningless.
9240
9241 \sa QStringView::isValidUtf16()
9242*/
9243
9244/*! \fn QChar *QString::data()
9245
9246 Returns a pointer to the data stored in the QString. The pointer
9247 can be used to access and modify the characters that compose the
9248 string.
9249
9250 Unlike constData() and unicode(), the returned data is always
9251 '\\0'-terminated.
9252
9253 Example:
9254
9255 \snippet qstring/main.cpp 19
9256
9257 Note that the pointer remains valid only as long as the string is
9258 not modified by other means. For read-only access, constData() is
9259 faster because it never causes a \l{deep copy} to occur.
9260
9261 \sa constData(), operator[]()
9262*/
9263
9264/*! \fn const QChar *QString::data() const
9265
9266 \overload
9267
9268 \note The returned string may not be '\\0'-terminated.
9269 Use size() to determine the length of the array.
9270
9271 \sa fromRawData()
9272*/
9273
9274/*! \fn const QChar *QString::constData() const
9275
9276 Returns a pointer to the data stored in the QString. The pointer
9277 can be used to access the characters that compose the string.
9278
9279 Note that the pointer remains valid only as long as the string is
9280 not modified.
9281
9282 \note The returned string may not be '\\0'-terminated.
9283 Use size() to determine the length of the array.
9284
9285 \sa data(), operator[](), fromRawData()
9286*/
9287
9288/*! \fn void QString::push_front(const QString &other)
9289
9290 This function is provided for STL compatibility, prepending the
9291 given \a other string to the beginning of this string. It is
9292 equivalent to \c prepend(other).
9293
9294 \sa prepend()
9295*/
9296
9297/*! \fn void QString::push_front(QChar ch)
9298
9299 \overload
9300
9301 Prepends the given \a ch character to the beginning of this string.
9302*/
9303
9304/*! \fn void QString::push_back(const QString &other)
9305
9306 This function is provided for STL compatibility, appending the
9307 given \a other string onto the end of this string. It is
9308 equivalent to \c append(other).
9309
9310 \sa append()
9311*/
9312
9313/*! \fn void QString::push_back(QChar ch)
9314
9315 \overload
9316
9317 Appends the given \a ch character onto the end of this string.
9318*/
9319
9320/*!
9321 \since 6.1
9322
9323 Removes from the string the characters in the half-open range
9324 [ \a first , \a last ). Returns an iterator to the character
9325 immediately after the last erased character (i.e. the character
9326 referred to by \a last before the erase).
9327*/
9328QString::iterator QString::erase(QString::const_iterator first, QString::const_iterator last)
9329{
9330 const auto start = std::distance(cbegin(), first);
9331 const auto len = std::distance(first, last);
9332 remove(start, len);
9333 return begin() + start;
9334}
9335
9336/*!
9337 \fn QString::iterator QString::erase(QString::const_iterator it)
9338
9339 \overload
9340 \since 6.5
9341
9342 Removes the character denoted by \c it from the string.
9343 Returns an iterator to the character immediately after the
9344 erased character.
9345
9346 \code
9347 QString c = "abcdefg";
9348 auto it = c.erase(c.cbegin()); // c is now "bcdefg"; "it" points to "b"
9349 \endcode
9350*/
9351
9352/*! \fn void QString::shrink_to_fit()
9353 \since 5.10
9354
9355 This function is provided for STL compatibility. It is
9356 equivalent to squeeze().
9357
9358 \sa squeeze()
9359*/
9360
9361/*!
9362 \fn std::string QString::toStdString() const
9363
9364 Returns a std::string object with the data contained in this
9365 QString. The Unicode data is converted into 8-bit characters using
9366 the toUtf8() function.
9367
9368 This method is mostly useful to pass a QString to a function
9369 that accepts a std::string object.
9370
9371 \sa toLatin1(), toUtf8(), toLocal8Bit(), QByteArray::toStdString()
9372*/
9373std::string QString::toStdString() const
9374{
9375 std::string result;
9376 if (isEmpty())
9377 return result;
9378
9379 auto writeToBuffer = [this](char *out, size_t) {
9380 char *last = QUtf8::convertFromUnicode(out, *this);
9381 return last - out;
9382 };
9383 size_t maxSize = size() * 3; // worst case for UTF-8
9384#ifdef __cpp_lib_string_resize_and_overwrite
9385 // C++23
9386 result.resize_and_overwrite(maxSize, writeToBuffer);
9387#else
9388 result.resize(maxSize);
9389 result.resize(writeToBuffer(result.data(), result.size()));
9390#endif
9391 return result;
9392}
9393
9394/*!
9395 \fn QString QString::fromRawData(const char16_t *unicode, qsizetype size)
9396 \since 6.10
9397
9398 Constructs a QString that uses the first \a size Unicode characters
9399 in the array \a unicode. The data in \a unicode is \e not
9400 copied. The caller must be able to guarantee that \a unicode will
9401 not be deleted or modified as long as the QString (or an
9402 unmodified copy of it) exists.
9403
9404 Any attempts to modify the QString or copies of it will cause it
9405 to create a deep copy of the data, ensuring that the raw data
9406 isn't modified.
9407
9408 Here is an example of how we can use a QRegularExpression on raw data in
9409 memory without requiring to copy the data into a QString:
9410
9411 \snippet qstring/main.cpp 22
9412 \snippet qstring/main.cpp 23
9413
9414 \warning A string created with fromRawData() is \e not
9415 '\\0'-terminated, unless the raw data contains a '\\0' character
9416 at position \a size. This means unicode() will \e not return a
9417 '\\0'-terminated string (although utf16() does, at the cost of
9418 copying the raw data).
9419
9420 \sa fromUtf16(), setRawData(), data(), constData(),
9421 nullTerminate(), nullTerminated()
9422*/
9423
9424/*!
9425 \fn QString QString::fromRawData(const QChar *unicode, qsizetype size)
9426 \overload
9427*/
9428
9429/*!
9430 \since 4.7
9431
9432 Resets the QString to use the first \a size Unicode characters
9433 in the array \a unicode. The data in \a unicode is \e not
9434 copied. The caller must be able to guarantee that \a unicode will
9435 not be deleted or modified as long as the QString (or an
9436 unmodified copy of it) exists.
9437
9438 This function can be used instead of fromRawData() to re-use
9439 existings QString objects to save memory re-allocations.
9440
9441 \sa fromRawData(), nullTerminate(), nullTerminated()
9442*/
9443QString &QString::setRawData(const QChar *unicode, qsizetype size)
9444{
9445 if (!unicode || !size) {
9446 clear();
9447 }
9448 *this = fromRawData(unicode, size);
9449 return *this;
9450}
9451
9452/*! \fn QString QString::fromStdU16String(const std::u16string &str)
9453 \since 5.5
9454
9455 \include qstring.cpp {from-std-string} {UTF-16} {fromUtf16()}
9456
9457 \sa fromUtf16(), fromStdWString(), fromStdU32String()
9458*/
9459
9460/*!
9461 \fn std::u16string QString::toStdU16String() const
9462 \since 5.5
9463
9464 Returns a std::u16string object with the data contained in this
9465 QString. The Unicode data is the same as returned by the utf16()
9466 method.
9467
9468 \sa utf16(), toStdWString(), toStdU32String()
9469*/
9470
9471/*! \fn QString QString::fromStdU32String(const std::u32string &str)
9472 \since 5.5
9473
9474 \include qstring.cpp {from-std-string} {UTF-32} {fromUcs4()}
9475
9476 \sa fromUcs4(), fromStdWString(), fromStdU16String()
9477*/
9478
9479/*!
9480 \fn std::u32string QString::toStdU32String() const
9481 \since 5.5
9482
9483 Returns a std::u32string object with the data contained in this
9484 QString. The Unicode data is the same as returned by the toUcs4()
9485 method.
9486
9487 \sa toUcs4(), toStdWString(), toStdU16String()
9488*/
9489
9490#if !defined(QT_NO_DATASTREAM)
9491/*!
9492 \fn QDataStream &operator<<(QDataStream &stream, const QString &string)
9493 \relates QString
9494
9495 Writes the given \a string to the specified \a stream.
9496
9497 \sa {Serializing Qt Data Types}
9498*/
9499
9500QDataStream &operator<<(QDataStream &out, const QString &str)
9501{
9502 if (out.version() == 1) {
9503 out << str.toLatin1();
9504 } else {
9505 if (!str.isNull() || out.version() < 3) {
9506 if ((out.byteOrder() == QDataStream::BigEndian) == (QSysInfo::ByteOrder == QSysInfo::BigEndian)) {
9507 out.writeBytes(reinterpret_cast<const char *>(str.unicode()),
9508 static_cast<qsizetype>(sizeof(QChar) * str.size()));
9509 } else {
9510 QVarLengthArray<char16_t> buffer(str.size());
9511 qbswap<sizeof(char16_t)>(str.constData(), str.size(), buffer.data());
9512 out.writeBytes(reinterpret_cast<const char *>(buffer.data()),
9513 static_cast<qsizetype>(sizeof(char16_t) * buffer.size()));
9514 }
9515 } else {
9516 QDataStream::writeQSizeType(out, -1); // write null marker
9517 }
9518 }
9519 return out;
9520}
9521
9522/*!
9523 \fn QDataStream &operator>>(QDataStream &stream, QString &string)
9524 \relates QString
9525
9526 Reads a string from the specified \a stream into the given \a string.
9527
9528 \sa {Serializing Qt Data Types}
9529*/
9530
9531QDataStream &operator>>(QDataStream &in, QString &str)
9532{
9533 if (in.version() == 1) {
9534 QByteArray l;
9535 in >> l;
9536 str = QString::fromLatin1(l);
9537 } else {
9538 qint64 size = QDataStream::readQSizeType(in);
9539 qsizetype bytes = size;
9540 if (size != bytes || size < -1) {
9541 str.clear();
9542 in.setStatus(QDataStream::SizeLimitExceeded);
9543 return in;
9544 }
9545 if (bytes == -1) { // null string
9546 str = QString();
9547 } else if (bytes > 0) {
9548 if (bytes & 0x1) {
9549 str.clear();
9550 in.setStatus(QDataStream::ReadCorruptData);
9551 return in;
9552 }
9553
9554 const qsizetype Step = 1024 * 1024;
9555 qsizetype len = bytes / 2;
9556 qsizetype allocated = 0;
9557
9558 while (allocated < len) {
9559 int blockSize = qMin(Step, len - allocated);
9560 str.resize(allocated + blockSize);
9561 if (in.readRawData(reinterpret_cast<char *>(str.data()) + allocated * 2,
9562 blockSize * 2) != blockSize * 2) {
9563 str.clear();
9564 in.setStatus(QDataStream::ReadPastEnd);
9565 return in;
9566 }
9567 allocated += blockSize;
9568 }
9569
9570 if ((in.byteOrder() == QDataStream::BigEndian)
9571 != (QSysInfo::ByteOrder == QSysInfo::BigEndian)) {
9572 char16_t *data = reinterpret_cast<char16_t *>(str.data());
9573 qbswap<sizeof(*data)>(data, len, data);
9574 }
9575 } else {
9576 str = QString(QLatin1StringView(""));
9577 }
9578 }
9579 return in;
9580}
9581#endif // QT_NO_DATASTREAM
9582
9583/*!
9584 \typedef QString::Data
9585 \internal
9586*/
9587
9588/*!
9589 \typedef QString::DataPtr
9590 \internal
9591*/
9592
9593/*!
9594 \fn DataPtr & QString::data_ptr()
9595 \internal
9596*/
9597
9598/*!
9599 \since 5.11
9600 \internal
9601 \relates QStringView
9602
9603 Returns \c true if the string is read right to left.
9604
9605 \sa QString::isRightToLeft()
9606*/
9607bool QtPrivate::isRightToLeft(QStringView string) noexcept
9608{
9609 int isolateLevel = 0;
9610
9611 for (QStringIterator i(string); i.hasNext();) {
9612 const char32_t c = i.next();
9613
9614 switch (QChar::direction(c)) {
9615 case QChar::DirRLI:
9616 case QChar::DirLRI:
9617 case QChar::DirFSI:
9618 ++isolateLevel;
9619 break;
9620 case QChar::DirPDI:
9621 if (isolateLevel)
9622 --isolateLevel;
9623 break;
9624 case QChar::DirL:
9625 if (isolateLevel)
9626 break;
9627 return false;
9628 case QChar::DirR:
9629 case QChar::DirAL:
9630 if (isolateLevel)
9631 break;
9632 return true;
9633 case QChar::DirEN:
9634 case QChar::DirES:
9635 case QChar::DirET:
9636 case QChar::DirAN:
9637 case QChar::DirCS:
9638 case QChar::DirB:
9639 case QChar::DirS:
9640 case QChar::DirWS:
9641 case QChar::DirON:
9642 case QChar::DirLRE:
9643 case QChar::DirLRO:
9644 case QChar::DirRLE:
9645 case QChar::DirRLO:
9646 case QChar::DirPDF:
9647 case QChar::DirNSM:
9648 case QChar::DirBN:
9649 break;
9650 }
9651 }
9652 return false;
9653}
9654
9655qsizetype QtPrivate::count(QStringView haystack, QStringView needle, Qt::CaseSensitivity cs) noexcept
9656{
9657 qsizetype num = 0;
9658 qsizetype i = -1;
9659 if (haystack.size() > 500 && needle.size() > 5) {
9660 QStringMatcher matcher(needle, cs);
9661 while ((i = matcher.indexIn(haystack, i + 1)) != -1)
9662 ++num;
9663 } else {
9664 while ((i = QtPrivate::findString(haystack, i + 1, needle, cs)) != -1)
9665 ++num;
9666 }
9667 return num;
9668}
9669
9670qsizetype QtPrivate::count(QStringView haystack, QChar needle, Qt::CaseSensitivity cs) noexcept
9671{
9672 if (cs == Qt::CaseSensitive)
9673 return std::count(haystack.cbegin(), haystack.cend(), needle);
9674
9675 needle = foldCase(needle);
9676 return std::count_if(haystack.cbegin(), haystack.cend(),
9677 [needle](const QChar c) { return foldAndCompare(c, needle); });
9678}
9679
9680qsizetype QtPrivate::count(QLatin1StringView haystack, QLatin1StringView needle, Qt::CaseSensitivity cs)
9681{
9682 qsizetype num = 0;
9683 qsizetype i = -1;
9684
9685 QLatin1StringMatcher matcher(needle, cs);
9686 while ((i = matcher.indexIn(haystack, i + 1)) != -1)
9687 ++num;
9688
9689 return num;
9690}
9691
9692qsizetype QtPrivate::count(QLatin1StringView haystack, QStringView needle, Qt::CaseSensitivity cs)
9693{
9694 if (haystack.size() < needle.size())
9695 return 0;
9696
9697 if (!QtPrivate::isLatin1(needle)) // won't find non-L1 UTF-16 needles in a L1 haystack!
9698 return 0;
9699
9700 qsizetype num = 0;
9701 qsizetype i = -1;
9702
9703 QVarLengthArray<uchar> s(needle.size());
9704 qt_to_latin1_unchecked(s.data(), needle.utf16(), needle.size());
9705
9706 QLatin1StringMatcher matcher(QLatin1StringView(reinterpret_cast<char *>(s.data()), s.size()),
9707 cs);
9708 while ((i = matcher.indexIn(haystack, i + 1)) != -1)
9709 ++num;
9710
9711 return num;
9712}
9713
9714qsizetype QtPrivate::count(QStringView haystack, QLatin1StringView needle, Qt::CaseSensitivity cs)
9715{
9716 if (haystack.size() < needle.size())
9717 return -1;
9718
9719 QVarLengthArray<char16_t> s = qt_from_latin1_to_qvla(needle);
9720 return QtPrivate::count(haystack, QStringView(s.data(), s.size()), cs);
9721}
9722
9723qsizetype QtPrivate::count(QLatin1StringView haystack, QChar needle, Qt::CaseSensitivity cs) noexcept
9724{
9725 // non-L1 needles cannot possibly match in L1-only haystacks
9726 if (needle.unicode() > 0xff)
9727 return 0;
9728
9729 if (cs == Qt::CaseSensitive) {
9730 return std::count(haystack.cbegin(), haystack.cend(), needle.toLatin1());
9731 } else {
9732 return std::count_if(haystack.cbegin(), haystack.cend(),
9733 CaseInsensitiveL1::matcher(needle.toLatin1()));
9734 }
9735}
9736
9737/*!
9738 \fn bool QtPrivate::startsWith(QStringView haystack, QStringView needle, Qt::CaseSensitivity cs)
9739 \since 5.10
9740 \fn bool QtPrivate::startsWith(QStringView haystack, QLatin1StringView needle, Qt::CaseSensitivity cs)
9741 \since 5.10
9742 \fn bool QtPrivate::startsWith(QLatin1StringView haystack, QStringView needle, Qt::CaseSensitivity cs)
9743 \since 5.10
9744 \fn bool QtPrivate::startsWith(QLatin1StringView haystack, QLatin1StringView needle, Qt::CaseSensitivity cs)
9745 \since 5.10
9746 \internal
9747 \relates QStringView
9748
9749 Returns \c true if \a haystack starts with \a needle,
9750 otherwise returns \c false.
9751
9752 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
9753
9754 \sa QtPrivate::endsWith(), QString::endsWith(), QStringView::endsWith(), QLatin1StringView::endsWith()
9755*/
9756
9757bool QtPrivate::startsWith(QStringView haystack, QStringView needle, Qt::CaseSensitivity cs) noexcept
9758{
9759 return qt_starts_with_impl(haystack, needle, cs);
9760}
9761
9762bool QtPrivate::startsWith(QStringView haystack, QLatin1StringView needle, Qt::CaseSensitivity cs) noexcept
9763{
9764 return qt_starts_with_impl(haystack, needle, cs);
9765}
9766
9767bool QtPrivate::startsWith(QLatin1StringView haystack, QStringView needle, Qt::CaseSensitivity cs) noexcept
9768{
9769 return qt_starts_with_impl(haystack, needle, cs);
9770}
9771
9772bool QtPrivate::startsWith(QLatin1StringView haystack, QLatin1StringView needle, Qt::CaseSensitivity cs) noexcept
9773{
9774 return qt_starts_with_impl(haystack, needle, cs);
9775}
9776
9777/*!
9778 \fn bool QtPrivate::endsWith(QStringView haystack, QStringView needle, Qt::CaseSensitivity cs)
9779 \since 5.10
9780 \fn bool QtPrivate::endsWith(QStringView haystack, QLatin1StringView needle, Qt::CaseSensitivity cs)
9781 \since 5.10
9782 \fn bool QtPrivate::endsWith(QLatin1StringView haystack, QStringView needle, Qt::CaseSensitivity cs)
9783 \since 5.10
9784 \fn bool QtPrivate::endsWith(QLatin1StringView haystack, QLatin1StringView needle, Qt::CaseSensitivity cs)
9785 \since 5.10
9786 \internal
9787 \relates QStringView
9788
9789 Returns \c true if \a haystack ends with \a needle,
9790 otherwise returns \c false.
9791
9792 \include qstring.qdocinc {search-comparison-case-sensitivity} {search}
9793
9794 \sa QtPrivate::startsWith(), QString::endsWith(), QStringView::endsWith(), QLatin1StringView::endsWith()
9795*/
9796
9797bool QtPrivate::endsWith(QStringView haystack, QStringView needle, Qt::CaseSensitivity cs) noexcept
9798{
9799 return qt_ends_with_impl(haystack, needle, cs);
9800}
9801
9802bool QtPrivate::endsWith(QStringView haystack, QLatin1StringView needle, Qt::CaseSensitivity cs) noexcept
9803{
9804 return qt_ends_with_impl(haystack, needle, cs);
9805}
9806
9807bool QtPrivate::endsWith(QLatin1StringView haystack, QStringView needle, Qt::CaseSensitivity cs) noexcept
9808{
9809 return qt_ends_with_impl(haystack, needle, cs);
9810}
9811
9812bool QtPrivate::endsWith(QLatin1StringView haystack, QLatin1StringView needle, Qt::CaseSensitivity cs) noexcept
9813{
9814 return qt_ends_with_impl(haystack, needle, cs);
9815}
9816
9817qsizetype QtPrivate::findString(QStringView haystack0, qsizetype from, QStringView needle0, Qt::CaseSensitivity cs) noexcept
9818{
9819 const qsizetype l = haystack0.size();
9820 const qsizetype sl = needle0.size();
9821 if (sl == 1)
9822 return findString(haystack0, from, needle0[0], cs);
9823 if (from < 0)
9824 from += l;
9825 if (std::size_t(sl + from) > std::size_t(l))
9826 return -1;
9827 if (!sl)
9828 return from;
9829 if (!l)
9830 return -1;
9831
9832 /*
9833 We use the Boyer-Moore algorithm in cases where the overhead
9834 for the skip table should pay off, otherwise we use a simple
9835 hash function.
9836 */
9837 if (l > 500 && sl > 5)
9838 return qFindStringBoyerMoore(haystack0, from, needle0, cs);
9839
9840 auto sv = [sl](const char16_t *v) { return QStringView(v, sl); };
9841 /*
9842 We use some hashing for efficiency's sake. Instead of
9843 comparing strings, we compare the hash value of str with that
9844 of a part of this QString. Only if that matches, we call
9845 qt_string_compare().
9846 */
9847 const char16_t *needle = needle0.utf16();
9848 const char16_t *haystack = haystack0.utf16() + from;
9849 const char16_t *end = haystack0.utf16() + (l - sl);
9850 const qregisteruint sl_minus_1 = sl - 1;
9851 qregisteruint hashNeedle = 0, hashHaystack = 0;
9852 qsizetype idx;
9853
9854 if (cs == Qt::CaseSensitive) {
9855 for (idx = 0; idx < sl; ++idx) {
9856 hashNeedle = ((hashNeedle<<1) + needle[idx]);
9857 hashHaystack = ((hashHaystack<<1) + haystack[idx]);
9858 }
9859 hashHaystack -= haystack[sl_minus_1];
9860
9861 while (haystack <= end) {
9862 hashHaystack += haystack[sl_minus_1];
9863 if (hashHaystack == hashNeedle
9864 && QtPrivate::compareStrings(needle0, sv(haystack), Qt::CaseSensitive) == 0)
9865 return haystack - haystack0.utf16();
9866
9867 REHASH(*haystack);
9868 ++haystack;
9869 }
9870 } else {
9871 const char16_t *haystack_start = haystack0.utf16();
9872 for (idx = 0; idx < sl; ++idx) {
9873 hashNeedle = (hashNeedle<<1) + foldCase(needle + idx, needle);
9874 hashHaystack = (hashHaystack<<1) + foldCase(haystack + idx, haystack_start);
9875 }
9876 hashHaystack -= foldCase(haystack + sl_minus_1, haystack_start);
9877
9878 while (haystack <= end) {
9879 hashHaystack += foldCase(haystack + sl_minus_1, haystack_start);
9880 if (hashHaystack == hashNeedle
9881 && QtPrivate::compareStrings(needle0, sv(haystack), Qt::CaseInsensitive) == 0)
9882 return haystack - haystack0.utf16();
9883
9884 REHASH(foldCase(haystack, haystack_start));
9885 ++haystack;
9886 }
9887 }
9888 return -1;
9889}
9890
9891qsizetype QtPrivate::findString(QStringView haystack, qsizetype from, QLatin1StringView needle, Qt::CaseSensitivity cs) noexcept
9892{
9893 if (haystack.size() < needle.size())
9894 return -1;
9895
9896 QVarLengthArray<char16_t> s = qt_from_latin1_to_qvla(needle);
9897 return QtPrivate::findString(haystack, from, QStringView(reinterpret_cast<const QChar*>(s.constData()), s.size()), cs);
9898}
9899
9900qsizetype QtPrivate::findString(QLatin1StringView haystack, qsizetype from, QStringView needle, Qt::CaseSensitivity cs) noexcept
9901{
9902 if (haystack.size() < needle.size())
9903 return -1;
9904
9905 if (!QtPrivate::isLatin1(needle)) // won't find non-L1 UTF-16 needles in a L1 haystack!
9906 return -1;
9907
9908 if (needle.size() == 1) {
9909 const char n = needle.front().toLatin1();
9910 return QtPrivate::findString(haystack, from, QLatin1StringView(&n, 1), cs);
9911 }
9912
9913 QVarLengthArray<char> s(needle.size());
9914 qt_to_latin1_unchecked(reinterpret_cast<uchar *>(s.data()), needle.utf16(), needle.size());
9915 return QtPrivate::findString(haystack, from, QLatin1StringView(s.data(), s.size()), cs);
9916}
9917
9918qsizetype QtPrivate::findString(QLatin1StringView haystack, qsizetype from, QLatin1StringView needle, Qt::CaseSensitivity cs) noexcept
9919{
9920 if (from < 0)
9921 from += haystack.size();
9922 if (from < 0)
9923 return -1;
9924 qsizetype adjustedSize = haystack.size() - from;
9925 if (adjustedSize < needle.size())
9926 return -1;
9927 if (needle.size() == 0)
9928 return from;
9929
9930 if (cs == Qt::CaseSensitive) {
9931
9932 if (needle.size() == 1) {
9933 Q_ASSERT(haystack.data() != nullptr); // see size check above
9934 if (auto it = memchr(haystack.data() + from, needle.front().toLatin1(), adjustedSize))
9935 return static_cast<const char *>(it) - haystack.data();
9936 return -1;
9937 }
9938
9939 const QLatin1StringMatcher matcher(needle, Qt::CaseSensitivity::CaseSensitive);
9940 return matcher.indexIn(haystack, from);
9941 }
9942
9943 // If the needle is sufficiently small we simply iteratively search through
9944 // the haystack. When the needle is too long we use a boyer-moore searcher
9945 // from the standard library, if available. If it is not available then the
9946 // QLatin1Strings are converted to QString and compared as such. Though
9947 // initialization is slower the boyer-moore search it employs still makes up
9948 // for it when haystack and needle are sufficiently long.
9949 // The needle size was chosen by testing various lengths using the
9950 // qstringtokenizer benchmark with the
9951 // "tokenize_qlatin1string_qlatin1string" test.
9952#ifdef Q_CC_MSVC
9953 const qsizetype threshold = 1;
9954#else
9955 const qsizetype threshold = 13;
9956#endif
9957 if (needle.size() <= threshold) {
9958 const auto begin = haystack.begin();
9959 const auto end = haystack.end() - needle.size() + 1;
9960 auto ciMatch = CaseInsensitiveL1::matcher(needle[0].toLatin1());
9961 const qsizetype nlen1 = needle.size() - 1;
9962 for (auto it = std::find_if(begin + from, end, ciMatch); it != end;
9963 it = std::find_if(it + 1, end, ciMatch)) {
9964 // In this comparison we skip the first character because we know it's a match
9965 if (!nlen1 || QLatin1StringView(it + 1, nlen1).compare(needle.sliced(1), cs) == 0)
9966 return std::distance(begin, it);
9967 }
9968 return -1;
9969 }
9970
9971 QLatin1StringMatcher matcher(needle, Qt::CaseSensitivity::CaseInsensitive);
9972 return matcher.indexIn(haystack, from);
9973}
9974
9975qsizetype QtPrivate::lastIndexOf(QStringView haystack, qsizetype from, char16_t needle, Qt::CaseSensitivity cs) noexcept
9976{
9977 return qLastIndexOf(haystack, QChar(needle), from, cs);
9978}
9979
9980qsizetype QtPrivate::lastIndexOf(QStringView haystack, qsizetype from, QStringView needle, Qt::CaseSensitivity cs) noexcept
9981{
9982 return qLastIndexOf(haystack, from, needle, cs);
9983}
9984
9985qsizetype QtPrivate::lastIndexOf(QStringView haystack, qsizetype from, QLatin1StringView needle, Qt::CaseSensitivity cs) noexcept
9986{
9987 return qLastIndexOf(haystack, from, needle, cs);
9988}
9989
9990qsizetype QtPrivate::lastIndexOf(QLatin1StringView haystack, qsizetype from, QStringView needle, Qt::CaseSensitivity cs) noexcept
9991{
9992 return qLastIndexOf(haystack, from, needle, cs);
9993}
9994
9995qsizetype QtPrivate::lastIndexOf(QLatin1StringView haystack, qsizetype from, QLatin1StringView needle, Qt::CaseSensitivity cs) noexcept
9996{
9997 return qLastIndexOf(haystack, from, needle, cs);
9998}
9999
10000#if QT_CONFIG(regularexpression)
10001qsizetype QtPrivate::indexOf(QStringView viewHaystack, const QString *stringHaystack, const QRegularExpression &re, qsizetype from, QRegularExpressionMatch *rmatch)
10002{
10003 if (!re.isValid()) {
10004 qtWarnAboutInvalidRegularExpression(re, "QString(View)", "indexOf");
10005 return -1;
10006 }
10007
10008 QRegularExpressionMatch match = stringHaystack
10009 ? re.match(*stringHaystack, from)
10010 : re.matchView(viewHaystack, from);
10011 if (match.hasMatch()) {
10012 const qsizetype ret = match.capturedStart();
10013 if (rmatch)
10014 *rmatch = std::move(match);
10015 return ret;
10016 }
10017
10018 return -1;
10019}
10020
10021qsizetype QtPrivate::indexOf(QStringView haystack, const QRegularExpression &re, qsizetype from, QRegularExpressionMatch *rmatch)
10022{
10023 return indexOf(haystack, nullptr, re, from, rmatch);
10024}
10025
10026qsizetype QtPrivate::lastIndexOf(QStringView viewHaystack, const QString *stringHaystack, const QRegularExpression &re, qsizetype from, QRegularExpressionMatch *rmatch)
10027{
10028 if (!re.isValid()) {
10029 qtWarnAboutInvalidRegularExpression(re, "QString(View)", "lastIndexOf");
10030 return -1;
10031 }
10032
10033 qsizetype endpos = (from < 0) ? (viewHaystack.size() + from + 1) : (from + 1);
10034 QRegularExpressionMatchIterator iterator = stringHaystack
10035 ? re.globalMatch(*stringHaystack)
10036 : re.globalMatchView(viewHaystack);
10037 qsizetype lastIndex = -1;
10038 while (iterator.hasNext()) {
10039 QRegularExpressionMatch match = iterator.next();
10040 qsizetype start = match.capturedStart();
10041 if (start < endpos) {
10042 lastIndex = start;
10043 if (rmatch)
10044 *rmatch = std::move(match);
10045 } else {
10046 break;
10047 }
10048 }
10049
10050 return lastIndex;
10051}
10052
10053qsizetype QtPrivate::lastIndexOf(QStringView haystack, const QRegularExpression &re, qsizetype from, QRegularExpressionMatch *rmatch)
10054{
10055 return lastIndexOf(haystack, nullptr, re, from, rmatch);
10056}
10057
10058bool QtPrivate::contains(QStringView viewHaystack, const QString *stringHaystack, const QRegularExpression &re, QRegularExpressionMatch *rmatch)
10059{
10060 if (!re.isValid()) {
10061 qtWarnAboutInvalidRegularExpression(re, "QString(View)", "contains");
10062 return false;
10063 }
10064 QRegularExpressionMatch m = stringHaystack
10065 ? re.match(*stringHaystack)
10066 : re.matchView(viewHaystack);
10067 bool hasMatch = m.hasMatch();
10068 if (hasMatch && rmatch)
10069 *rmatch = std::move(m);
10070 return hasMatch;
10071}
10072
10073bool QtPrivate::contains(QStringView haystack, const QRegularExpression &re, QRegularExpressionMatch *rmatch)
10074{
10075 return contains(haystack, nullptr, re, rmatch);
10076}
10077
10078qsizetype QtPrivate::count(QStringView haystack, const QRegularExpression &re)
10079{
10080 if (!re.isValid()) {
10081 qtWarnAboutInvalidRegularExpression(re, "QString(View)", "count");
10082 return 0;
10083 }
10084 qsizetype count = 0;
10085 qsizetype index = -1;
10086 qsizetype len = haystack.size();
10087 while (index <= len - 1) {
10088 QRegularExpressionMatch match = re.matchView(haystack, index + 1);
10089 if (!match.hasMatch())
10090 break;
10091 count++;
10092
10093 // Search again, from the next character after the beginning of this
10094 // capture. If the capture starts with a surrogate pair, both together
10095 // count as "one character".
10096 index = match.capturedStart();
10097 if (index < len && haystack[index].isHighSurrogate())
10098 ++index;
10099 }
10100 return count;
10101}
10102
10103#endif // QT_CONFIG(regularexpression)
10104
10105/*!
10106 \since 5.0
10107
10108 Converts a plain text string to an HTML string with
10109 HTML metacharacters \c{<}, \c{>}, \c{&}, and \c{"} replaced by HTML
10110 entities.
10111
10112 Example:
10113
10114 \snippet code/src_corelib_text_qstring.cpp 7
10115*/
10116QString QString::toHtmlEscaped() const
10117{
10118 const auto pos = std::u16string_view(*this).find_first_of(u"<>&\"");
10119 if (pos == std::u16string_view::npos)
10120 return *this;
10121 QString rich;
10122 const qsizetype len = size();
10123 rich.reserve(qsizetype(len * 1.1));
10124 rich += qToStringViewIgnoringNull(*this).first(pos);
10125 for (auto ch : qToStringViewIgnoringNull(*this).sliced(pos)) {
10126 if (ch == u'<')
10127 rich += "&lt;"_L1;
10128 else if (ch == u'>')
10129 rich += "&gt;"_L1;
10130 else if (ch == u'&')
10131 rich += "&amp;"_L1;
10132 else if (ch == u'"')
10133 rich += "&quot;"_L1;
10134 else
10135 rich += ch;
10136 }
10137 rich.squeeze();
10138 return rich;
10139}
10140
10141/*!
10142 \macro QStringLiteral(str)
10143 \relates QString
10144
10145 The macro generates the data for a QString out of the string literal \a str
10146 at compile time. Creating a QString from it is free in this case, and the
10147 generated string data is stored in the read-only segment of the compiled
10148 object file.
10149
10150 If you have code that looks like this:
10151
10152 \snippet code/src_corelib_text_qstring.cpp 9
10153
10154 then a temporary QString will be created to be passed as the \c{hasAttribute}
10155 function parameter. This can be quite expensive, as it involves a memory
10156 allocation and the copy/conversion of the data into QString's internal
10157 encoding.
10158
10159 This cost can be avoided by using QStringLiteral instead:
10160
10161 \snippet code/src_corelib_text_qstring.cpp 10
10162
10163 In this case, QString's internal data will be generated at compile time; no
10164 conversion or allocation will occur at runtime.
10165
10166 Using QStringLiteral instead of a double quoted plain C++ string literal can
10167 significantly speed up creation of QString instances from data known at
10168 compile time.
10169
10170 \note QLatin1StringView can still be more efficient than QStringLiteral
10171 when the string is passed to a function that has an overload taking
10172 QLatin1StringView and this overload avoids conversion to QString. For
10173 instance, QString::operator==() can compare to a QLatin1StringView
10174 directly:
10175
10176 \snippet code/src_corelib_text_qstring.cpp 11
10177
10178 \note Some compilers have bugs encoding strings containing characters outside
10179 the US-ASCII character set. Make sure you prefix your string with \c{u} in
10180 those cases. It is optional otherwise.
10181
10182 \note QStringLiteral is interchangeable with \l operator""_s. The latter saves
10183 typing when many string literals are present in the code.
10184
10185 \sa QByteArrayLiteral
10186*/
10187
10188#if QT_DEPRECATED_SINCE(6, 8)
10189/*!
10190 \fn QtLiterals::operator""_qs(const char16_t *str, size_t size)
10191
10192 \relates QString
10193 \since 6.2
10194 \deprecated [6.8] Use \c _s from Qt::StringLiterals namespace instead.
10195
10196 Literal operator that creates a QString out of the first \a size characters in
10197 the char16_t string literal \a str.
10198
10199 The QString is created at compile time, and the generated string data is stored
10200 in the read-only segment of the compiled object file. Duplicate literals may
10201 share the same read-only memory. This functionality is interchangeable with
10202 QStringLiteral, but saves typing when many string literals are present in the
10203 code.
10204
10205 The following code creates a QString:
10206 \code
10207 auto str = u"hello"_qs;
10208 \endcode
10209
10210 \sa QStringLiteral, QtLiterals::operator""_qba(const char *str, size_t size)
10211*/
10212#endif // QT_DEPRECATED_SINCE(6, 8)
10213
10214/*!
10215 \fn Qt::Literals::StringLiterals::operator""_s(const char16_t *str, size_t size)
10216
10217 \relates QString
10218 \since 6.4
10219
10220 Literal operator that creates a QString out of the first \a size characters in
10221 the char16_t string literal \a str.
10222
10223 The QString is created at compile time, and the generated string data is stored
10224 in the read-only segment of the compiled object file. Duplicate literals may
10225 share the same read-only memory. This functionality is interchangeable with
10226 QStringLiteral, but saves typing when many string literals are present in the
10227 code.
10228
10229 The following code creates a QString:
10230 \code
10231 using namespace Qt::StringLiterals;
10232
10233 auto str = u"hello"_s;
10234 \endcode
10235
10236 \sa Qt::Literals::StringLiterals
10237*/
10238
10239/*!
10240 \internal
10241 */
10242void QAbstractConcatenable::appendLatin1To(QLatin1StringView in, QChar *out) noexcept
10243{
10244 qt_from_latin1(reinterpret_cast<char16_t *>(out), in.data(), size_t(in.size()));
10245}
10246
10247/*!
10248 \fn template <typename T> qsizetype erase(QString &s, const T &t)
10249 \relates QString
10250 \since 6.1
10251
10252 Removes all elements that compare equal to \a t from the
10253 string \a s. Returns the number of elements removed, if any.
10254
10255 \sa erase_if
10256*/
10257
10258/*!
10259 \fn template <typename Predicate> qsizetype erase_if(QString &s, Predicate pred)
10260 \relates QString
10261 \since 6.1
10262
10263 Removes all elements for which the predicate \a pred returns true
10264 from the string \a s. Returns the number of elements removed, if
10265 any.
10266
10267 \sa erase
10268*/
10269
10270/*!
10271 \macro const char *qPrintable(const QString &str)
10272 \relates QString
10273
10274 Returns \a str as a \c{const char *}. This is equivalent to
10275 \a{str}.toLocal8Bit().\l{QByteArray::}{constData()}.
10276
10277 The char pointer will be invalid after the statement in which
10278 qPrintable() is used. This is because the array returned by
10279 QString::toLocal8Bit() will fall out of scope.
10280
10281 \note qDebug(), qInfo(), qWarning(), qCritical(), qFatal() expect
10282 %s arguments to be UTF-8 encoded, while qPrintable() converts to
10283 local 8-bit encoding. Therefore qUtf8Printable() should be used
10284 for logging strings instead of qPrintable().
10285
10286 \sa qUtf8Printable()
10287*/
10288
10289/*!
10290 \macro const char *qUtf8Printable(const QString &str)
10291 \relates QString
10292 \since 5.4
10293
10294 Returns \a str as a \c{const char *}. This is equivalent to
10295 \a{str}.toUtf8().\l{QByteArray::}{constData()}.
10296
10297 The char pointer will be invalid after the statement in which
10298 qUtf8Printable() is used. This is because the array returned by
10299 QString::toUtf8() will fall out of scope.
10300
10301 Example:
10302
10303 \snippet code/src_corelib_text_qstring.cpp qUtf8Printable
10304
10305 \sa qPrintable(), qDebug(), qInfo(), qWarning(), qCritical(), qFatal()
10306*/
10307
10308/*!
10309 \macro const wchar_t *qUtf16Printable(const QString &str)
10310 \relates QString
10311 \since 5.7
10312
10313 Returns \a str as a \c{const ushort *}, but cast to a \c{const wchar_t *}
10314 to avoid warnings. This is equivalent to \a{str}.utf16() plus some casting.
10315
10316 The only useful thing you can do with the return value of this macro is to
10317 pass it to QString::asprintf() for use in a \c{%ls} conversion. In particular,
10318 the return value is \e{not} a valid \c{const wchar_t*}!
10319
10320 In general, the pointer will be invalid after the statement in which
10321 qUtf16Printable() is used. This is because the pointer may have been
10322 obtained from a temporary expression, which will fall out of scope.
10323
10324 Example:
10325
10326 \snippet code/src_corelib_text_qstring.cpp qUtf16Printable
10327
10328 \sa qPrintable(), qDebug(), qInfo(), qWarning(), qCritical(), qFatal()
10329*/
10330
10331QT_END_NAMESPACE
10332
10333#undef REHASH
QString convertToQString(QAnyStringView string)
Definition qstring.cpp:5566
Definition qlist.h:81
char32_t next(char32_t invalidAs=QChar::ReplacementCharacter)
bool hasNext() const
\inmodule QtCore
QList< uint > convertToUcs4(QStringView string)
Definition qstring.cpp:5822
QByteArray convertToUtf8(QStringView string)
Definition qstring.cpp:5767
QByteArray convertToLocal8Bit(QStringView string)
Definition qstring.cpp:5724
QByteArray convertToLatin1(QStringView string)
Definition qstring.cpp:5583
Combined button and popup list for selecting options.
static QString convertCase(T &str, QUnicodeTables::Case which)
Definition qstring.cpp:7196
static constexpr NormalizationCorrection uc_normalization_corrections[]
Q_CORE_EXPORT Q_DECL_PURE_FUNCTION bool startsWith(QStringView haystack, QStringView needle, Qt::CaseSensitivity cs=Qt::CaseSensitive) noexcept
Definition qstring.cpp:9757
Q_CORE_EXPORT Q_DECL_PURE_FUNCTION bool endsWith(QStringView haystack, QStringView needle, Qt::CaseSensitivity cs=Qt::CaseSensitive) noexcept
Definition qstring.cpp:9797
Q_CORE_EXPORT Q_DECL_PURE_FUNCTION bool isLower(QStringView s) noexcept
Definition qstring.cpp:5503
const QString & asString(const QString &s)
Definition qstring.h:1700
Q_CORE_EXPORT Q_DECL_PURE_FUNCTION bool isValidUtf16(QStringView s) noexcept
Definition qstring.cpp:906
Q_CORE_EXPORT Q_DECL_PURE_FUNCTION bool equalStrings(QStringView lhs, QStringView rhs) noexcept
Definition qstring.cpp:1374
qsizetype findString(QStringView str, qsizetype from, QChar needle, Qt::CaseSensitivity cs=Qt::CaseSensitive) noexcept
Q_CORE_EXPORT Q_DECL_PURE_FUNCTION bool isRightToLeft(QStringView string) noexcept
Q_CORE_EXPORT Q_DECL_PURE_FUNCTION int compareStrings(QStringView lhs, QStringView rhs, Qt::CaseSensitivity cs=Qt::CaseSensitive) noexcept
Q_CORE_EXPORT Q_DECL_PURE_FUNCTION bool isAscii(QLatin1StringView s) noexcept
Definition qstring.cpp:851
constexpr bool isLatin1(QLatin1StringView s) noexcept
Definition qstring.h:77
Q_CORE_EXPORT Q_DECL_PURE_FUNCTION const char16_t * qustrcasechr(QStringView str, char16_t ch) noexcept
Definition qstring.cpp:776
Q_CORE_EXPORT Q_DECL_PURE_FUNCTION bool isUpper(QStringView s) noexcept
Definition qstring.cpp:5508
Q_CORE_EXPORT Q_DECL_PURE_FUNCTION const char16_t * qustrchr(QStringView str, char16_t ch) noexcept
Definition qstring.cpp:688
void qt_to_latin1_unchecked(uchar *dst, const char16_t *uc, qsizetype len)
Definition qstring.cpp:1189
static char16_t foldCase(char16_t ch) noexcept
Definition qchar.cpp:1696
#define __has_feature(x)
uint QT_FASTCALL fetch1Pixel< QPixelLayout::BPP1LSB >(const uchar *src, int index)
bool comparesEqual(const QFileInfo &lhs, const QFileInfo &rhs)
static bool isAscii_helper(const char16_t *&ptr, const char16_t *end)
Definition qstring.cpp:859
static Int toIntegral(QStringView string, bool *ok, int base)
Definition qstring.cpp:7685
void qt_to_latin1(uchar *dst, const char16_t *src, qsizetype length)
Definition qstring.cpp:1184
Qt::strong_ordering compareThreeWay(const QByteArray &lhs, const QChar &rhs) noexcept
Definition qstring.cpp:6740
static void append_utf8(QString &qs, const char *cs, qsizetype len)
Definition qstring.cpp:7319
#define ATTRIBUTE_NO_SANITIZE
Definition qstring.cpp:367
bool qt_is_ascii(const char *&ptr, const char *end) noexcept
Definition qstring.cpp:787
static bool checkCase(QStringView s, QUnicodeTables::Case c) noexcept
Definition qstring.cpp:5492
static void replace_helper(QString &str, QSpan< qsizetype > indices, qsizetype blen, QStringView after)
Definition qstring.cpp:3693
Q_CORE_EXPORT void qt_from_latin1(char16_t *dst, const char *str, size_t size) noexcept
Definition qstring.cpp:921
static int ucstrcmp(const char16_t *a, size_t alen, const Char2 *b, size_t blen)
Definition qstring.cpp:1347
bool comparesEqual(const QByteArray &lhs, char16_t rhs) noexcept
Definition qstring.cpp:6746
Q_DECLARE_TYPEINFO(Part, Q_PRIMITIVE_TYPE)
static void removeStringImpl(QString &s, const T &needle, Qt::CaseSensitivity cs)
Definition qstring.cpp:3502
static bool needsReallocate(const QString &str, qsizetype newSize)
Definition qstring.cpp:2638
static int qArgDigitValue(QChar ch) noexcept
Definition qstring.cpp:1614
bool comparesEqual(const QByteArray &lhs, const QChar &rhs) noexcept
Definition qstring.cpp:6735
#define REHASH(a)
Definition qstring.cpp:66
bool comparesEqual(const QByteArrayView &lhs, char16_t rhs) noexcept
Definition qstring.cpp:6724
static int ucstrncmp(const char16_t *a, const char16_t *b, size_t l)
Definition qstring.cpp:1265
static Q_NEVER_INLINE int ucstricmp(qsizetype alen, const char16_t *a, qsizetype blen, const char *b)
Definition qstring.cpp:1220
static QByteArray qt_convert_to_latin1(QStringView string)
Definition qstring.cpp:5589
static bool ucstreq(const char16_t *a, size_t alen, const Char2 *b)
Definition qstring.cpp:1340
static QList< uint > qt_convert_to_ucs4(QStringView string)
Definition qstring.cpp:5794
qsizetype qFindStringBoyerMoore(QStringView haystack, qsizetype from, QStringView needle, Qt::CaseSensitivity cs)
static QByteArray qt_convert_to_local_8bit(QStringView string)
Definition qstring.cpp:5701
static LengthMod parse_length_modifier(const char *&c) noexcept
Definition qstring.cpp:7375
static ArgEscapeData findArgEscapes(QStringView s)
Definition qstring.cpp:8594
static QByteArray qt_convert_to_utf8(QStringView str)
Definition qstring.cpp:5747
static void qt_to_latin1_internal(uchar *dst, const char16_t *src, qsizetype length)
Definition qstring.cpp:1005
QtPrivate::QCaseInsensitiveLatin1Hash CaseInsensitiveL1
Definition qstring.cpp:1354
LengthMod
Definition qstring.cpp:7364
@ lm_z
Definition qstring.cpp:7364
@ lm_none
Definition qstring.cpp:7364
@ lm_t
Definition qstring.cpp:7364
@ lm_l
Definition qstring.cpp:7364
@ lm_ll
Definition qstring.cpp:7364
@ lm_hh
Definition qstring.cpp:7364
@ lm_L
Definition qstring.cpp:7364
@ lm_h
Definition qstring.cpp:7364
@ lm_j
Definition qstring.cpp:7364
static void insert_helper(QString &str, qsizetype i, const T &toInsert)
Definition qstring.cpp:2977
static int latin1nicmp(const char *lhsChar, qsizetype lSize, const char *rhsChar, qsizetype rSize)
Definition qstring.cpp:1356
Qt::strong_ordering compareThreeWay(const QByteArrayView &lhs, const QChar &rhs) noexcept
Definition qstring.cpp:6718
static char16_t to_unicode(const char c)
Definition qstring.cpp:8999
Qt::strong_ordering compareThreeWay(const QByteArray &lhs, char16_t rhs) noexcept
Definition qstring.cpp:6751
static QString replaceArgEscapes(QStringView s, const ArgEscapeData &d, qsizetype field_width, QStringView arg, QStringView larg, QChar fillChar)
Definition qstring.cpp:8670
static QVarLengthArray< char16_t > qt_from_latin1_to_qvla(QLatin1StringView str)
Definition qstring.cpp:996
static Q_NEVER_INLINE int ucstricmp8(const char *utf8, const char *utf8end, const QChar *utf16, const QChar *utf16end)
Definition qstring.cpp:1238
void qt_string_normalize(QString *data, QString::NormalizationForm mode, QChar::UnicodeVersion version, qsizetype from)
Definition qstring.cpp:8457
static uint parse_flag_characters(const char *&c) noexcept
Definition qstring.cpp:7327
static Q_NEVER_INLINE int ucstricmp(qsizetype alen, const char16_t *a, qsizetype blen, const char16_t *b)
Definition qstring.cpp:1195
static char16_t to_unicode(const QChar c)
Definition qstring.cpp:8998
QDataStream & operator>>(QDataStream &in, QString &str)
Definition qstring.cpp:9531
static int getEscape(const Char *uc, qsizetype *pos, qsizetype len)
Definition qstring.cpp:9002
static int ucstrncmp(const char16_t *a, const char *b, size_t l)
Definition qstring.cpp:1318
static bool can_consume(const char *&c, char ch) noexcept
Definition qstring.cpp:7366
static int parse_field_width(const char *&c, qsizetype size)
Definition qstring.cpp:7347
Qt::strong_ordering compareThreeWay(const QByteArrayView &lhs, char16_t rhs) noexcept
Definition qstring.cpp:6729
#define qUtf16Printable(string)
Definition qstring.h:1717
qsizetype occurrences
Definition qstring.cpp:8588
qsizetype escape_len
Definition qstring.cpp:8591
qsizetype locale_occurrences
Definition qstring.cpp:8589
\inmodule QtCore \reentrant
Definition qchar.h:18
constexpr char16_t unicode() const noexcept
Converts a Latin-1 character to an 16-bit-encoded Unicode representation of the character.
Definition qchar.h:22
constexpr QLatin1Char(char c) noexcept
Constructs a Latin-1 character for c.
Definition qchar.h:20
@ BlankBeforePositive
Definition qlocale_p.h:270
@ AddTrailingZeroes
Definition qlocale_p.h:267
static int difference(char lhs, char rhs)