OpenTTD Source 15.0-beta2
string.cpp
Go to the documentation of this file.
1/*
2 * This file is part of OpenTTD.
3 * OpenTTD is free software; you can redistribute it and/or modify it under the terms of the GNU General Public License as published by the Free Software Foundation, version 2.
4 * OpenTTD is distributed in the hope that it will be useful, but WITHOUT ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.
5 * See the GNU General Public License for more details. You should have received a copy of the GNU General Public License along with OpenTTD. If not, see <http://www.gnu.org/licenses/>.
6 */
7
10#include "stdafx.h"
11#include "debug.h"
12#include "core/math_func.hpp"
13#include "error_func.h"
14#include "string_func.h"
15#include "string_base.h"
16#include "core/utf8.hpp"
17
18#include "table/control_codes.h"
19
20#ifdef _MSC_VER
21# define strncasecmp strnicmp
22#endif
23
24#ifdef _WIN32
25# include "os/windows/win32.h"
26#endif
27
28#ifdef WITH_UNISCRIBE
30#endif
31
32#ifdef WITH_ICU_I18N
33/* Required by StrNaturalCompare. */
34# include <unicode/ustring.h>
35# include "language.h"
36# include "gfx_func.h"
37#endif /* WITH_ICU_I18N */
38
39#if defined(WITH_COCOA)
40# include "os/macosx/string_osx.h"
41#endif
42
43#include "safeguards.h"
44
45
57void strecpy(std::span<char> dst, std::string_view src)
58{
59 /* Ensure source string fits with NUL terminator; dst must be at least 1 character longer than src. */
60 if (std::empty(dst) || std::size(src) >= std::size(dst) - 1U) {
61#if defined(STRGEN) || defined(SETTINGSGEN)
62 FatalError("String too long for destination buffer");
63#else /* STRGEN || SETTINGSGEN */
64 Debug(misc, 0, "String too long for destination buffer");
65 src = src.substr(0, std::size(dst) - 1U);
66#endif /* STRGEN || SETTINGSGEN */
67 }
68
69 auto it = std::copy(std::begin(src), std::end(src), std::begin(dst));
70 *it = '\0';
71}
72
78std::string FormatArrayAsHex(std::span<const uint8_t> data)
79{
80 std::string str;
81 str.reserve(data.size() * 2 + 1);
82
83 for (auto b : data) {
84 fmt::format_to(std::back_inserter(str), "{:02X}", b);
85 }
86
87 return str;
88}
89
95static bool IsSccEncodedCode(char32_t c)
96{
97 switch (c) {
98 case SCC_RECORD_SEPARATOR:
99 case SCC_ENCODED:
103 return true;
104
105 default:
106 return false;
107 }
108}
109
122template <class T>
123static void StrMakeValid(T &dst, const char *str, const char *last, StringValidationSettings settings)
124{
125 /* Assume the ABSOLUTE WORST to be in str as it comes from the outside. */
126
127 while (str <= last && *str != '\0') {
128 size_t len = Utf8EncodedCharLen(*str);
129 char32_t c;
130 /* If the first byte does not look like the first byte of an encoded
131 * character, i.e. encoded length is 0, then this byte is definitely bad
132 * and it should be skipped.
133 * When the first byte looks like the first byte of an encoded character,
134 * then the remaining bytes in the string are checked whether the whole
135 * encoded character can be there. If that is not the case, this byte is
136 * skipped.
137 * Finally we attempt to decode the encoded character, which does certain
138 * extra validations to see whether the correct number of bytes were used
139 * to encode the character. If that is not the case, the byte is probably
140 * invalid and it is skipped. We could emit a question mark, but then the
141 * logic below cannot just copy bytes, it would need to re-encode the
142 * decoded characters as the length in bytes may have changed.
143 *
144 * The goals here is to get as much valid Utf8 encoded characters from the
145 * source string to the destination string.
146 *
147 * Note: a multi-byte encoded termination ('\0') will trigger the encoded
148 * char length and the decoded length to differ, so it will be ignored as
149 * invalid character data. If it were to reach the termination, then we
150 * would also reach the "last" byte of the string and a normal '\0'
151 * termination will be placed after it.
152 */
153 if (len == 0 || str + len > last + 1 || len != Utf8Decode(&c, str)) {
154 /* Maybe the next byte is still a valid character? */
155 str++;
156 continue;
157 }
158
159 if ((IsPrintable(c) && (c < SCC_SPRITE_START || c > SCC_SPRITE_END)) || (settings.Test(StringValidationSetting::AllowControlCode) && IsSccEncodedCode(c))) {
160 /* Copy the character back. Even if dst is current the same as str
161 * (i.e. no characters have been changed) this is quicker than
162 * moving the pointers ahead by len */
163 do {
164 *dst++ = *str++;
165 } while (--len != 0);
166 } else if (settings.Test(StringValidationSetting::AllowNewline) && c == '\n') {
167 *dst++ = *str++;
168 } else {
169 if (settings.Test(StringValidationSetting::AllowNewline) && c == '\r' && str[1] == '\n') {
170 str += len;
171 continue;
172 }
173 str += len;
174 if (settings.Test(StringValidationSetting::ReplaceTabCrNlWithSpace) && (c == '\r' || c == '\n' || c == '\t')) {
175 /* Replace the tab, carriage return or newline with a space. */
176 *dst++ = ' ';
178 /* Replace the undesirable character with a question mark */
179 *dst++ = '?';
180 }
181 }
182 }
183
184 /* String termination, if needed, is left to the caller of this function. */
185}
186
195{
196 char *dst = str;
197 StrMakeValid(dst, str, str + strlen(str), settings);
198 *dst = '\0';
199}
200
209{
210 if (str.empty()) return;
211
212 char *buf = str.data();
213 char *last = buf + str.size() - 1;
214 char *dst = buf;
215 StrMakeValid(dst, buf, last, settings);
216 str.erase(dst - buf, std::string::npos);
217}
218
226std::string StrMakeValid(std::string_view str, StringValidationSettings settings)
227{
228 if (str.empty()) return {};
229
230 auto buf = str.data();
231 auto last = buf + str.size() - 1;
232
233 std::ostringstream dst;
234 std::ostreambuf_iterator<char> dst_iter(dst);
235 StrMakeValid(dst_iter, buf, last, settings);
236
237 return dst.str();
238}
239
248bool StrValid(std::span<const char> str)
249{
250 /* Assume the ABSOLUTE WORST to be in str as it comes from the outside. */
251 auto it = std::begin(str);
252 auto last = std::prev(std::end(str));
253
254 while (it <= last && *it != '\0') {
255 size_t len = Utf8EncodedCharLen(*it);
256 /* Encoded length is 0 if the character isn't known.
257 * The length check is needed to prevent Utf8Decode to read
258 * over the terminating '\0' if that happens to be placed
259 * within the encoding of an UTF8 character. */
260 if (len == 0 || it + len > last) return false;
261
262 char32_t c;
263 len = Utf8Decode(&c, &*it);
264 if (!IsPrintable(c) || (c >= SCC_SPRITE_START && c <= SCC_SPRITE_END)) {
265 return false;
266 }
267
268 it += len;
269 }
270
271 return *it == '\0';
272}
273
281void StrTrimInPlace(std::string &str)
282{
283 str = StrTrimView(str);
284}
285
286std::string_view StrTrimView(std::string_view str)
287{
288 size_t first_pos = str.find_first_not_of(' ');
289 if (first_pos == std::string::npos) {
290 return std::string_view{};
291 }
292 size_t last_pos = str.find_last_not_of(' ');
293 return str.substr(first_pos, last_pos - first_pos + 1);
294}
295
302bool StrStartsWithIgnoreCase(std::string_view str, const std::string_view prefix)
303{
304 if (str.size() < prefix.size()) return false;
305 return StrEqualsIgnoreCase(str.substr(0, prefix.size()), prefix);
306}
307
309struct CaseInsensitiveCharTraits : public std::char_traits<char> {
310 static bool eq(char c1, char c2) { return toupper(c1) == toupper(c2); }
311 static bool ne(char c1, char c2) { return toupper(c1) != toupper(c2); }
312 static bool lt(char c1, char c2) { return toupper(c1) < toupper(c2); }
313
314 static int compare(const char *s1, const char *s2, size_t n)
315 {
316 while (n-- != 0) {
317 if (toupper(*s1) < toupper(*s2)) return -1;
318 if (toupper(*s1) > toupper(*s2)) return 1;
319 ++s1; ++s2;
320 }
321 return 0;
322 }
323
324 static const char *find(const char *s, size_t n, char a)
325 {
326 for (; n > 0; --n, ++s) {
327 if (toupper(*s) == toupper(a)) return s;
328 }
329 return nullptr;
330 }
331};
332
334typedef std::basic_string_view<char, CaseInsensitiveCharTraits> CaseInsensitiveStringView;
335
342bool StrEndsWithIgnoreCase(std::string_view str, const std::string_view suffix)
343{
344 if (str.size() < suffix.size()) return false;
345 return StrEqualsIgnoreCase(str.substr(str.size() - suffix.size()), suffix);
346}
347
355int StrCompareIgnoreCase(const std::string_view str1, const std::string_view str2)
356{
357 CaseInsensitiveStringView ci_str1{ str1.data(), str1.size() };
358 CaseInsensitiveStringView ci_str2{ str2.data(), str2.size() };
359 return ci_str1.compare(ci_str2);
360}
361
368bool StrEqualsIgnoreCase(const std::string_view str1, const std::string_view str2)
369{
370 if (str1.size() != str2.size()) return false;
371 return StrCompareIgnoreCase(str1, str2) == 0;
372}
373
381bool StrContainsIgnoreCase(const std::string_view str, const std::string_view value)
382{
383 CaseInsensitiveStringView ci_str{ str.data(), str.size() };
384 CaseInsensitiveStringView ci_value{ value.data(), value.size() };
385 return ci_str.find(ci_value) != ci_str.npos;
386}
387
394size_t Utf8StringLength(std::string_view str)
395{
396 Utf8View view(str);
397 return std::distance(view.begin(), view.end());
398}
399
400bool strtolower(std::string &str, std::string::size_type offs)
401{
402 bool changed = false;
403 for (auto ch = str.begin() + offs; ch != str.end(); ++ch) {
404 auto new_ch = static_cast<char>(tolower(static_cast<unsigned char>(*ch)));
405 changed |= new_ch != *ch;
406 *ch = new_ch;
407 }
408 return changed;
409}
410
418bool IsValidChar(char32_t key, CharSetFilter afilter)
419{
420 switch (afilter) {
421 case CS_ALPHANUMERAL: return IsPrintable(key);
422 case CS_NUMERAL: return (key >= '0' && key <= '9');
423 case CS_NUMERAL_SPACE: return (key >= '0' && key <= '9') || key == ' ';
424 case CS_NUMERAL_SIGNED: return (key >= '0' && key <= '9') || key == '-';
425 case CS_ALPHA: return IsPrintable(key) && !(key >= '0' && key <= '9');
426 case CS_HEXADECIMAL: return (key >= '0' && key <= '9') || (key >= 'a' && key <= 'f') || (key >= 'A' && key <= 'F');
427 default: NOT_REACHED();
428 }
429}
430
431
432/* UTF-8 handling routines */
433
434
441size_t Utf8Decode(char32_t *c, const char *s)
442{
443 assert(c != nullptr);
444
445 if (!HasBit(s[0], 7)) {
446 /* Single byte character: 0xxxxxxx */
447 *c = s[0];
448 return 1;
449 } else if (GB(s[0], 5, 3) == 6) {
450 if (IsUtf8Part(s[1])) {
451 /* Double byte character: 110xxxxx 10xxxxxx */
452 *c = GB(s[0], 0, 5) << 6 | GB(s[1], 0, 6);
453 if (*c >= 0x80) return 2;
454 }
455 } else if (GB(s[0], 4, 4) == 14) {
456 if (IsUtf8Part(s[1]) && IsUtf8Part(s[2])) {
457 /* Triple byte character: 1110xxxx 10xxxxxx 10xxxxxx */
458 *c = GB(s[0], 0, 4) << 12 | GB(s[1], 0, 6) << 6 | GB(s[2], 0, 6);
459 if (*c >= 0x800) return 3;
460 }
461 } else if (GB(s[0], 3, 5) == 30) {
462 if (IsUtf8Part(s[1]) && IsUtf8Part(s[2]) && IsUtf8Part(s[3])) {
463 /* 4 byte character: 11110xxx 10xxxxxx 10xxxxxx 10xxxxxx */
464 *c = GB(s[0], 0, 3) << 18 | GB(s[1], 0, 6) << 12 | GB(s[2], 0, 6) << 6 | GB(s[3], 0, 6);
465 if (*c >= 0x10000 && *c <= 0x10FFFF) return 4;
466 }
467 }
468
469 *c = '?';
470 return 1;
471}
472
478static bool IsGarbageCharacter(char32_t c)
479{
480 if (c >= '0' && c <= '9') return false;
481 if (c >= 'A' && c <= 'Z') return false;
482 if (c >= 'a' && c <= 'z') return false;
483 if (c >= SCC_CONTROL_START && c <= SCC_CONTROL_END) return true;
484 if (c >= 0xC0 && c <= 0x10FFFF) return false;
485
486 return true;
487}
488
497static std::string_view SkipGarbage(std::string_view str)
498{
499 Utf8View view(str);
500 auto it = view.begin();
501 const auto end = view.end();
502 while (it != end && IsGarbageCharacter(*it)) ++it;
503 return str.substr(it.GetByteOffset());
504}
505
514int StrNaturalCompare(std::string_view s1, std::string_view s2, bool ignore_garbage_at_front)
515{
516 if (ignore_garbage_at_front) {
517 s1 = SkipGarbage(s1);
518 s2 = SkipGarbage(s2);
519 }
520
521#ifdef WITH_ICU_I18N
522 if (_current_collator) {
523 UErrorCode status = U_ZERO_ERROR;
524 int result = _current_collator->compareUTF8(icu::StringPiece(s1.data(), s1.size()), icu::StringPiece(s2.data(), s2.size()), status);
525 if (U_SUCCESS(status)) return result;
526 }
527#endif /* WITH_ICU_I18N */
528
529#if defined(_WIN32) && !defined(STRGEN) && !defined(SETTINGSGEN)
530 int res = OTTDStringCompare(s1, s2);
531 if (res != 0) return res - 2; // Convert to normal C return values.
532#endif
533
534#if defined(WITH_COCOA) && !defined(STRGEN) && !defined(SETTINGSGEN)
535 int res = MacOSStringCompare(s1, s2);
536 if (res != 0) return res - 2; // Convert to normal C return values.
537#endif
538
539 /* Do a normal comparison if ICU is missing or if we cannot create a collator. */
540 return StrCompareIgnoreCase(s1, s2);
541}
542
543#ifdef WITH_ICU_I18N
544
545#include <unicode/stsearch.h>
546
555static int ICUStringContains(const std::string_view str, const std::string_view value, bool case_insensitive)
556{
557 if (_current_collator) {
558 std::unique_ptr<icu::RuleBasedCollator> coll(dynamic_cast<icu::RuleBasedCollator *>(_current_collator->clone()));
559 if (coll) {
560 UErrorCode status = U_ZERO_ERROR;
561 coll->setStrength(case_insensitive ? icu::Collator::SECONDARY : icu::Collator::TERTIARY);
562 coll->setAttribute(UCOL_NUMERIC_COLLATION, UCOL_OFF, status);
563
564 auto u_str = icu::UnicodeString::fromUTF8(icu::StringPiece(str.data(), str.size()));
565 auto u_value = icu::UnicodeString::fromUTF8(icu::StringPiece(value.data(), value.size()));
566 icu::StringSearch u_searcher(u_value, u_str, coll.get(), nullptr, status);
567 if (U_SUCCESS(status)) {
568 auto pos = u_searcher.first(status);
569 if (U_SUCCESS(status)) return pos != USEARCH_DONE ? 1 : 0;
570 }
571 }
572 }
573
574 return -1;
575}
576#endif /* WITH_ICU_I18N */
577
585[[nodiscard]] bool StrNaturalContains(const std::string_view str, const std::string_view value)
586{
587#ifdef WITH_ICU_I18N
588 int res_u = ICUStringContains(str, value, false);
589 if (res_u >= 0) return res_u > 0;
590#endif /* WITH_ICU_I18N */
591
592#if defined(_WIN32) && !defined(STRGEN) && !defined(SETTINGSGEN)
593 int res = Win32StringContains(str, value, false);
594 if (res >= 0) return res > 0;
595#endif
596
597#if defined(WITH_COCOA) && !defined(STRGEN) && !defined(SETTINGSGEN)
598 int res = MacOSStringContains(str, value, false);
599 if (res >= 0) return res > 0;
600#endif
601
602 return str.find(value) != std::string_view::npos;
603}
604
612[[nodiscard]] bool StrNaturalContainsIgnoreCase(const std::string_view str, const std::string_view value)
613{
614#ifdef WITH_ICU_I18N
615 int res_u = ICUStringContains(str, value, true);
616 if (res_u >= 0) return res_u > 0;
617#endif /* WITH_ICU_I18N */
618
619#if defined(_WIN32) && !defined(STRGEN) && !defined(SETTINGSGEN)
620 int res = Win32StringContains(str, value, true);
621 if (res >= 0) return res > 0;
622#endif
623
624#if defined(WITH_COCOA) && !defined(STRGEN) && !defined(SETTINGSGEN)
625 int res = MacOSStringContains(str, value, true);
626 if (res >= 0) return res > 0;
627#endif
628
629 CaseInsensitiveStringView ci_str{ str.data(), str.size() };
630 CaseInsensitiveStringView ci_value{ value.data(), value.size() };
631 return ci_str.find(ci_value) != CaseInsensitiveStringView::npos;
632}
633
640static int ConvertHexNibbleToByte(char c)
641{
642 if (c >= '0' && c <= '9') return c - '0';
643 if (c >= 'A' && c <= 'F') return c + 10 - 'A';
644 if (c >= 'a' && c <= 'f') return c + 10 - 'a';
645 return -1;
646}
647
659bool ConvertHexToBytes(std::string_view hex, std::span<uint8_t> bytes)
660{
661 if (bytes.size() != hex.size() / 2) {
662 return false;
663 }
664
665 /* Hex-string lengths are always divisible by 2. */
666 if (hex.size() % 2 != 0) {
667 return false;
668 }
669
670 for (size_t i = 0; i < hex.size() / 2; i++) {
671 auto hi = ConvertHexNibbleToByte(hex[i * 2]);
672 auto lo = ConvertHexNibbleToByte(hex[i * 2 + 1]);
673
674 if (hi < 0 || lo < 0) {
675 return false;
676 }
677
678 bytes[i] = (hi << 4) | lo;
679 }
680
681 return true;
682}
683
684#ifdef WITH_UNISCRIBE
685
686/* static */ std::unique_ptr<StringIterator> StringIterator::Create()
687{
688 return std::make_unique<UniscribeStringIterator>();
689}
690
691#elif defined(WITH_ICU_I18N)
692
693#include <unicode/utext.h>
694#include <unicode/brkiter.h>
695
698{
699 icu::BreakIterator *char_itr;
700 icu::BreakIterator *word_itr;
701
702 std::vector<UChar> utf16_str;
703 std::vector<size_t> utf16_to_utf8;
704
705public:
706 IcuStringIterator() : char_itr(nullptr), word_itr(nullptr)
707 {
708 UErrorCode status = U_ZERO_ERROR;
709 this->char_itr = icu::BreakIterator::createCharacterInstance(icu::Locale(_current_language != nullptr ? _current_language->isocode : "en"), status);
710 this->word_itr = icu::BreakIterator::createWordInstance(icu::Locale(_current_language != nullptr ? _current_language->isocode : "en"), status);
711
712 this->utf16_str.push_back('\0');
713 this->utf16_to_utf8.push_back(0);
714 }
715
716 ~IcuStringIterator() override
717 {
718 delete this->char_itr;
719 delete this->word_itr;
720 }
721
722 void SetString(std::string_view s) override
723 {
724 /* Unfortunately current ICU versions only provide rudimentary support
725 * for word break iterators (especially for CJK languages) in combination
726 * with UTF-8 input. As a work around we have to convert the input to
727 * UTF-16 and create a mapping back to UTF-8 character indices. */
728 this->utf16_str.clear();
729 this->utf16_to_utf8.clear();
730
731 Utf8View view(s);
732 for (auto it = view.begin(), end = view.end(); it != end; ++it) {
733 size_t idx = it.GetByteOffset();
734 char32_t c = *it;
735 if (c < 0x10000) {
736 this->utf16_str.push_back((UChar)c);
737 } else {
738 /* Make a surrogate pair. */
739 this->utf16_str.push_back((UChar)(0xD800 + ((c - 0x10000) >> 10)));
740 this->utf16_str.push_back((UChar)(0xDC00 + ((c - 0x10000) & 0x3FF)));
741 this->utf16_to_utf8.push_back(idx);
742 }
743 this->utf16_to_utf8.push_back(idx);
744 }
745 this->utf16_str.push_back('\0');
746 this->utf16_to_utf8.push_back(s.size());
747
748 UText text = UTEXT_INITIALIZER;
749 UErrorCode status = U_ZERO_ERROR;
750 utext_openUChars(&text, this->utf16_str.data(), this->utf16_str.size() - 1, &status);
751 this->char_itr->setText(&text, status);
752 this->word_itr->setText(&text, status);
753 this->char_itr->first();
754 this->word_itr->first();
755 }
756
757 size_t SetCurPosition(size_t pos) override
758 {
759 /* Convert incoming position to an UTF-16 string index. */
760 uint utf16_pos = 0;
761 for (uint i = 0; i < this->utf16_to_utf8.size(); i++) {
762 if (this->utf16_to_utf8[i] == pos) {
763 utf16_pos = i;
764 break;
765 }
766 }
767
768 /* isBoundary has the documented side-effect of setting the current
769 * position to the first valid boundary equal to or greater than
770 * the passed value. */
771 this->char_itr->isBoundary(utf16_pos);
772 return this->utf16_to_utf8[this->char_itr->current()];
773 }
774
775 size_t Next(IterType what) override
776 {
777 int32_t pos;
778 switch (what) {
779 case ITER_CHARACTER:
780 pos = this->char_itr->next();
781 break;
782
783 case ITER_WORD:
784 pos = this->word_itr->following(this->char_itr->current());
785 /* The ICU word iterator considers both the start and the end of a word a valid
786 * break point, but we only want word starts. Move to the next location in
787 * case the new position points to whitespace. */
788 while (pos != icu::BreakIterator::DONE &&
789 IsWhitespace(Utf16DecodeChar((const uint16_t *)&this->utf16_str[pos]))) {
790 int32_t new_pos = this->word_itr->next();
791 /* Don't set it to DONE if it was valid before. Otherwise we'll return END
792 * even though the iterator wasn't at the end of the string before. */
793 if (new_pos == icu::BreakIterator::DONE) break;
794 pos = new_pos;
795 }
796
797 this->char_itr->isBoundary(pos);
798 break;
799
800 default:
801 NOT_REACHED();
802 }
803
804 return pos == icu::BreakIterator::DONE ? END : this->utf16_to_utf8[pos];
805 }
806
807 size_t Prev(IterType what) override
808 {
809 int32_t pos;
810 switch (what) {
811 case ITER_CHARACTER:
812 pos = this->char_itr->previous();
813 break;
814
815 case ITER_WORD:
816 pos = this->word_itr->preceding(this->char_itr->current());
817 /* The ICU word iterator considers both the start and the end of a word a valid
818 * break point, but we only want word starts. Move to the previous location in
819 * case the new position points to whitespace. */
820 while (pos != icu::BreakIterator::DONE &&
821 IsWhitespace(Utf16DecodeChar((const uint16_t *)&this->utf16_str[pos]))) {
822 int32_t new_pos = this->word_itr->previous();
823 /* Don't set it to DONE if it was valid before. Otherwise we'll return END
824 * even though the iterator wasn't at the start of the string before. */
825 if (new_pos == icu::BreakIterator::DONE) break;
826 pos = new_pos;
827 }
828
829 this->char_itr->isBoundary(pos);
830 break;
831
832 default:
833 NOT_REACHED();
834 }
835
836 return pos == icu::BreakIterator::DONE ? END : this->utf16_to_utf8[pos];
837 }
838};
839
840/* static */ std::unique_ptr<StringIterator> StringIterator::Create()
841{
842 return std::make_unique<IcuStringIterator>();
843}
844
845#else
846
848class DefaultStringIterator : public StringIterator
849{
850 Utf8View string;
851 Utf8View::iterator cur_pos; //< Current iteration position.
852
853public:
854 void SetString(std::string_view s) override
855 {
856 this->string = s;
857 this->cur_pos = this->string.begin();
858 }
859
860 size_t SetCurPosition(size_t pos) override
861 {
862 this->cur_pos = this->string.GetIterAtByte(pos);
863 return this->cur_pos.GetByteOffset();
864 }
865
866 size_t Next(IterType what) override
867 {
868 const auto end = this->string.end();
869 /* Already at the end? */
870 if (this->cur_pos >= end) return END;
871
872 switch (what) {
873 case ITER_CHARACTER:
874 ++this->cur_pos;
875 return this->cur_pos.GetByteOffset();
876
877 case ITER_WORD:
878 /* Consume current word. */
879 while (this->cur_pos != end && !IsWhitespace(*this->cur_pos)) {
880 ++this->cur_pos;
881 }
882 /* Consume whitespace to the next word. */
883 while (this->cur_pos != end && IsWhitespace(*this->cur_pos)) {
884 ++this->cur_pos;
885 }
886 return this->cur_pos.GetByteOffset();
887
888 default:
889 NOT_REACHED();
890 }
891
892 return END;
893 }
894
895 size_t Prev(IterType what) override
896 {
897 const auto begin = this->string.begin();
898 /* Already at the beginning? */
899 if (this->cur_pos == begin) return END;
900
901 switch (what) {
902 case ITER_CHARACTER:
903 --this->cur_pos;
904 return this->cur_pos.GetByteOffset();
905
906 case ITER_WORD:
907 /* Consume preceding whitespace. */
908 do {
909 --this->cur_pos;
910 } while (this->cur_pos != begin && IsWhitespace(*this->cur_pos));
911 /* Consume preceding word. */
912 while (this->cur_pos != begin && !IsWhitespace(*this->cur_pos)) {
913 --this->cur_pos;
914 }
915 /* Move caret back to the beginning of the word. */
916 if (IsWhitespace(*this->cur_pos)) ++this->cur_pos;
917 return this->cur_pos.GetByteOffset();
918
919 default:
920 NOT_REACHED();
921 }
922
923 return END;
924 }
925};
926
927#if defined(WITH_COCOA) && !defined(STRGEN) && !defined(SETTINGSGEN)
928/* static */ std::unique_ptr<StringIterator> StringIterator::Create()
929{
930 std::unique_ptr<StringIterator> i = OSXStringIterator::Create();
931 if (i != nullptr) return i;
932
933 return std::make_unique<DefaultStringIterator>();
934}
935#else
936/* static */ std::unique_ptr<StringIterator> StringIterator::Create()
937{
938 return std::make_unique<DefaultStringIterator>();
939}
940#endif /* defined(WITH_COCOA) && !defined(STRGEN) && !defined(SETTINGSGEN) */
941
942#endif
debug_inline constexpr bool HasBit(const T x, const uint8_t y)
Checks if a bit in a value is set.
debug_inline static constexpr uint GB(const T x, const uint8_t s, const uint8_t n)
Fetch n bits from x, started at bit s.
Enum-as-bit-set wrapper.
String iterator using ICU as a backend.
Definition string.cpp:698
size_t Prev(IterType what) override
Move the cursor back by one iteration unit.
Definition string.cpp:807
size_t Next(IterType what) override
Advance the cursor by one iteration unit.
Definition string.cpp:775
std::vector< size_t > utf16_to_utf8
Mapping from UTF-16 code point position to index in the UTF-8 source string.
Definition string.cpp:703
void SetString(std::string_view s) override
Set a new iteration string.
Definition string.cpp:722
size_t SetCurPosition(size_t pos) override
Change the current string cursor.
Definition string.cpp:757
std::vector< UChar > utf16_str
UTF-16 copy of the string.
Definition string.cpp:702
icu::BreakIterator * char_itr
ICU iterator for characters.
Definition string.cpp:699
icu::BreakIterator * word_itr
ICU iterator for words.
Definition string.cpp:700
Class for iterating over different kind of parts of a string.
Definition string_base.h:14
static const size_t END
Sentinel to indicate end-of-iteration.
Definition string_base.h:23
virtual size_t Prev(IterType what=ITER_CHARACTER)=0
Move the cursor back by one iteration unit.
virtual size_t SetCurPosition(size_t pos)=0
Change the current string cursor.
virtual size_t Next(IterType what=ITER_CHARACTER)=0
Advance the cursor by one iteration unit.
static std::unique_ptr< StringIterator > Create()
Create a new iterator instance.
Definition string.cpp:840
IterType
Type of the iterator.
Definition string_base.h:17
@ ITER_WORD
Iterate over words.
Definition string_base.h:19
@ ITER_CHARACTER
Iterate over characters (or more exactly grapheme clusters).
Definition string_base.h:18
virtual void SetString(std::string_view s)=0
Set a new iteration string.
Bidirectional input iterator over codepoints.
Definition utf8.hpp:37
Constant span of UTF-8 encoded data.
Definition utf8.hpp:24
Control codes that are embedded in the translation strings.
@ SCC_ENCODED
Encoded string marker and sub-string parameter.
@ SCC_ENCODED_NUMERIC
Encoded numeric parameter.
@ SCC_ENCODED_STRING
Encoded string parameter.
@ SCC_ENCODED_INTERNAL
Encoded text from OpenTTD.
Functions related to debugging.
#define Debug(category, level, format_string,...)
Output a line of debugging information.
Definition debug.h:37
Error reporting related functions.
fluid_settings_t * settings
FluidSynth settings handle.
Functions related to the gfx engine.
Information about languages and their files.
const LanguageMetadata * _current_language
The currently loaded language.
Definition strings.cpp:55
std::unique_ptr< icu::Collator > _current_collator
Collator for the language currently in use.
Definition strings.cpp:60
Integer math functions.
A number of safeguards to prevent using unsafe methods.
Definition of base types and functions in a cross-platform compatible way.
static void StrMakeValid(T &dst, const char *str, const char *last, StringValidationSettings settings)
Copies the valid (UTF-8) characters from str up to last to the dst.
Definition string.cpp:123
bool ConvertHexToBytes(std::string_view hex, std::span< uint8_t > bytes)
Convert a hex-string to a byte-array, while validating it was actually hex.
Definition string.cpp:659
size_t Utf8StringLength(std::string_view str)
Get the length of an UTF-8 encoded string in number of characters and thus not the number of bytes th...
Definition string.cpp:394
bool IsValidChar(char32_t key, CharSetFilter afilter)
Only allow certain keys.
Definition string.cpp:418
bool StrContainsIgnoreCase(const std::string_view str, const std::string_view value)
Checks if a string is contained in another string, while ignoring the case of the characters.
Definition string.cpp:381
bool StrEqualsIgnoreCase(const std::string_view str1, const std::string_view str2)
Compares two string( view)s for equality, while ignoring the case of the characters.
Definition string.cpp:368
void StrMakeValidInPlace(char *str, StringValidationSettings settings)
Scans the string for invalid characters and replaces them with a question mark '?' (if not ignored).
Definition string.cpp:194
bool StrNaturalContains(const std::string_view str, const std::string_view value)
Checks if a string is contained in another string with a locale-aware comparison that is case sensiti...
Definition string.cpp:585
void strecpy(std::span< char > dst, std::string_view src)
Copies characters from one buffer to another.
Definition string.cpp:57
std::string FormatArrayAsHex(std::span< const uint8_t > data)
Format a byte array into a continuous hex string.
Definition string.cpp:78
bool StrStartsWithIgnoreCase(std::string_view str, const std::string_view prefix)
Check whether the given string starts with the given prefix, ignoring case.
Definition string.cpp:302
bool StrValid(std::span< const char > str)
Checks whether the given string is valid, i.e.
Definition string.cpp:248
static int ConvertHexNibbleToByte(char c)
Convert a single hex-nibble to a byte.
Definition string.cpp:640
static int ICUStringContains(const std::string_view str, const std::string_view value, bool case_insensitive)
Search if a string is contained in another string using the current locale.
Definition string.cpp:555
static std::string_view SkipGarbage(std::string_view str)
Skip some of the 'garbage' in the string that we don't want to use to sort on.
Definition string.cpp:497
static bool IsSccEncodedCode(char32_t c)
Test if a character is (only) part of an encoded string.
Definition string.cpp:95
size_t Utf8Decode(char32_t *c, const char *s)
Decode and consume the next UTF-8 encoded character.
Definition string.cpp:441
int StrNaturalCompare(std::string_view s1, std::string_view s2, bool ignore_garbage_at_front)
Compares two strings using case insensitive natural sort.
Definition string.cpp:514
bool StrNaturalContainsIgnoreCase(const std::string_view str, const std::string_view value)
Checks if a string is contained in another string with a locale-aware comparison that is case insensi...
Definition string.cpp:612
std::basic_string_view< char, CaseInsensitiveCharTraits > CaseInsensitiveStringView
Case insensitive string view.
Definition string.cpp:334
int StrCompareIgnoreCase(const std::string_view str1, const std::string_view str2)
Compares two string( view)s, while ignoring the case of the characters.
Definition string.cpp:355
bool StrEndsWithIgnoreCase(std::string_view str, const std::string_view suffix)
Check whether the given string ends with the given suffix, ignoring case.
Definition string.cpp:342
void StrTrimInPlace(std::string &str)
Trim the spaces from given string in place, i.e.
Definition string.cpp:281
static bool IsGarbageCharacter(char32_t c)
Test if a unicode character is considered garbage to be skipped.
Definition string.cpp:478
Functions related to low-level strings.
char32_t Utf16DecodeChar(const uint16_t *c)
Decode an UTF-16 character.
bool IsWhitespace(char32_t c)
Check whether UNICODE character is whitespace or not, i.e.
int8_t Utf8EncodedCharLen(char c)
Return the length of an UTF-8 encoded value based on a single char.
int MacOSStringCompare(std::string_view s1, std::string_view s2)
Compares two strings using case insensitive natural sort.
int MacOSStringContains(const std::string_view str, const std::string_view value, bool case_insensitive)
Search if a string is contained in another string using the current locale.
Functions related to localized text support on OSX.
@ ReplaceWithQuestionMark
Replace the unknown/bad bits with question marks.
@ AllowControlCode
Allow the special control codes.
@ AllowNewline
Allow newlines; replaces '\r ' with ' ' during processing.
@ ReplaceTabCrNlWithSpace
Replace tabs ('\t'), carriage returns ('\r') and newlines (' ') with spaces.
CharSetFilter
Valid filter types for IsValidChar.
Definition string_type.h:24
@ CS_NUMERAL_SPACE
Only numbers and spaces.
Definition string_type.h:27
@ CS_HEXADECIMAL
Only hexadecimal characters.
Definition string_type.h:30
@ CS_NUMERAL
Only numeric ones.
Definition string_type.h:26
@ CS_NUMERAL_SIGNED
Only numbers and '-' for negative values.
Definition string_type.h:28
@ CS_ALPHA
Only alphabetic values.
Definition string_type.h:29
@ CS_ALPHANUMERAL
Both numeric and alphabetic and spaces and stuff.
Definition string_type.h:25
Functions related to laying out text on Win32.
Case insensitive implementation of the standard character type traits.
Definition string.cpp:309
char isocode[16]
the ISO code for the language (not country code)
Definition language.h:31
Handling of UTF-8 encoded data.
int Win32StringContains(const std::string_view str, const std::string_view value, bool case_insensitive)
Search if a string is contained in another string using the current locale.
Definition win32.cpp:479
declarations of functions for MS windows systems