OpenTTD Source 15.0-beta2
strgen_base.cpp
Go to the documentation of this file.
1/*
2 * This file is part of OpenTTD.
3 * OpenTTD is free software; you can redistribute it and/or modify it under the terms of the GNU General Public License as published by the Free Software Foundation, version 2.
4 * OpenTTD is distributed in the hope that it will be useful, but WITHOUT ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.
5 * See the GNU General Public License for more details. You should have received a copy of the GNU General Public License along with OpenTTD. If not, see <http://www.gnu.org/licenses/>.
6 */
7
10#include "../stdafx.h"
11#include "../core/endian_func.hpp"
12#include "../core/mem_func.hpp"
13#include "../error_func.h"
14#include "../string_func.h"
15#include "../core/string_builder.hpp"
16#include "../table/control_codes.h"
17
18#include "strgen.h"
19
20#include "../table/strgen_tables.h"
21
22#include "../safeguards.h"
23
24StrgenState _strgen;
25static bool _translated;
26static const char *_cur_ident;
27static ParsedCommandStruct _cur_pcs;
28static size_t _cur_argidx;
29
31 const CmdStruct *cmd = nullptr;
32 std::string param;
33 std::optional<size_t> argno;
34 std::optional<uint8_t> casei;
35};
36static ParsedCommandString ParseCommandString(const char **str);
37static size_t TranslateArgumentIdx(size_t arg, size_t offset = 0);
38
44Case::Case(uint8_t caseidx, const std::string &string) :
45 caseidx(caseidx), string(string)
46{
47}
48
56LangString::LangString(const std::string &name, const std::string &english, size_t index, size_t line) :
57 name(name), english(english), index(index), line(line)
58{
59}
60
63{
64 this->translated.clear();
65 this->translated_cases.clear();
66}
67
72StringData::StringData(size_t tabs) : tabs(tabs), max_strings(tabs * TAB_SIZE)
73{
74 this->strings.resize(max_strings);
75 this->next_string_id = 0;
76}
77
80{
81 for (size_t i = 0; i < this->max_strings; i++) {
82 LangString *ls = this->strings[i].get();
83 if (ls != nullptr) ls->FreeTranslation();
84 }
85}
86
92void StringData::Add(std::shared_ptr<LangString> ls)
93{
94 this->name_to_string[ls->name] = ls;
95 this->strings[ls->index] = std::move(ls);
96}
97
103LangString *StringData::Find(const std::string &s)
104{
105 auto it = this->name_to_string.find(s);
106 if (it == this->name_to_string.end()) return nullptr;
107
108 return it->second.get();
109}
110
117static uint32_t VersionHashStr(uint32_t hash, std::string_view s)
118{
119 for (auto c : s) {
120 hash = std::rotl(hash, 3) ^ c;
121 hash = (hash & 1 ? hash >> 1 ^ 0xDEADBEEF : hash >> 1);
122 }
123 return hash;
124}
125
130uint32_t StringData::Version() const
131{
132 uint32_t hash = 0;
133
134 for (size_t i = 0; i < this->max_strings; i++) {
135 const LangString *ls = this->strings[i].get();
136
137 if (ls != nullptr) {
138 hash ^= i * 0x717239;
139 hash = (hash & 1 ? hash >> 1 ^ 0xDEADBEEF : hash >> 1);
140 hash = VersionHashStr(hash, ls->name);
141
142 const char *s = ls->english.c_str();
144 while ((cs = ParseCommandString(&s)).cmd != nullptr) {
145 if (cs.cmd->flags.Test(CmdFlag::DontCount)) continue;
146
147 hash ^= (cs.cmd - _cmd_structs) * 0x1234567;
148 hash = (hash & 1 ? hash >> 1 ^ 0xF00BAA4 : hash >> 1);
149 }
150 }
151 }
152
153 return hash;
154}
155
160size_t StringData::CountInUse(size_t tab) const
161{
162 size_t count = TAB_SIZE;
163 while (count > 0 && this->strings[(tab * TAB_SIZE) + count - 1] == nullptr) --count;
164 return count;
165}
166
167static size_t Utf8Validate(const char *s)
168{
169 char32_t c;
170
171 if (!HasBit(s[0], 7)) {
172 /* 1 byte */
173 return 1;
174 } else if (GB(s[0], 5, 3) == 6 && IsUtf8Part(s[1])) {
175 /* 2 bytes */
176 c = GB(s[0], 0, 5) << 6 | GB(s[1], 0, 6);
177 if (c >= 0x80) return 2;
178 } else if (GB(s[0], 4, 4) == 14 && IsUtf8Part(s[1]) && IsUtf8Part(s[2])) {
179 /* 3 bytes */
180 c = GB(s[0], 0, 4) << 12 | GB(s[1], 0, 6) << 6 | GB(s[2], 0, 6);
181 if (c >= 0x800) return 3;
182 } else if (GB(s[0], 3, 5) == 30 && IsUtf8Part(s[1]) && IsUtf8Part(s[2]) && IsUtf8Part(s[3])) {
183 /* 4 bytes */
184 c = GB(s[0], 0, 3) << 18 | GB(s[1], 0, 6) << 12 | GB(s[2], 0, 6) << 6 | GB(s[3], 0, 6);
185 if (c >= 0x10000 && c <= 0x10FFFF) return 4;
186 }
187
188 return 0;
189}
190
191void EmitSingleChar(StringBuilder &builder, const char *buf, char32_t value)
192{
193 if (*buf != '\0') StrgenWarning("Ignoring trailing letters in command");
194 builder.PutUtf8(value);
195}
196
197/* The plural specifier looks like
198 * {NUM} {PLURAL <ARG#> passenger passengers} then it picks either passenger/passengers depending on the count in NUM */
199static std::pair<std::optional<size_t>, std::optional<size_t>> ParseRelNum(const char **buf)
200{
201 const char *s = *buf;
202 char *end;
203
204 while (*s == ' ' || *s == '\t') s++;
205 size_t v = std::strtoul(s, &end, 0);
206 if (end == s) return {};
207 std::optional<size_t> offset;
208 if (*end == ':') {
209 /* Take the Nth within */
210 s = end + 1;
211 offset = std::strtoul(s, &end, 0);
212 if (end == s) return {};
213 }
214 *buf = end;
215 return {v, offset};
216}
217
218/* Parse out the next word, or nullptr */
219std::optional<std::string_view> ParseWord(const char **buf)
220{
221 const char *s = *buf;
222
223 while (*s == ' ' || *s == '\t') s++;
224 if (*s == '\0') return {};
225
226 if (*s == '"') {
227 const char *begin = ++s;
228 /* parse until next " or NUL */
229 for (;;) {
230 if (*s == '\0') StrgenFatal("Unterminated quotes");
231 if (*s == '"') {
232 *buf = s + 1;
233 return std::string_view(begin, s - begin);
234 }
235 s++;
236 }
237 } else {
238 /* proceed until whitespace or NUL */
239 const char *begin = s;
240 for (;;) {
241 if (*s == '\0' || *s == ' ' || *s == '\t') {
242 *buf = s;
243 return std::string_view(begin, s - begin);
244 }
245 s++;
246 }
247 }
248}
249
250/* This is encoded like
251 * CommandByte <ARG#> <NUM> {Length of each string} {each string} */
252static void EmitWordList(StringBuilder &builder, const std::vector<std::string> &words)
253{
254 builder.PutUint8(static_cast<uint8_t>(words.size()));
255 for (size_t i = 0; i < words.size(); i++) {
256 size_t len = words[i].size();
257 if (len > UINT8_MAX) StrgenFatal("WordList {}/{} string '{}' too long, max bytes {}", i + 1, words.size(), words[i], UINT8_MAX);
258 builder.PutUint8(static_cast<uint8_t>(len));
259 }
260 for (size_t i = 0; i < words.size(); i++) {
261 builder.Put(words[i]);
262 }
263}
264
265void EmitPlural(StringBuilder &builder, const char *buf, char32_t)
266{
267 /* Parse out the number, if one exists. Otherwise default to prev arg. */
268 auto [argidx, offset] = ParseRelNum(&buf);
269 if (!argidx.has_value()) {
270 if (_cur_argidx == 0) StrgenFatal("Plural choice needs positional reference");
271 argidx = _cur_argidx - 1;
272 }
273
274 const CmdStruct *cmd = _cur_pcs.consuming_commands[*argidx];
275 if (!offset.has_value()) {
276 /* Use default offset */
277 if (cmd == nullptr || !cmd->default_plural_offset.has_value()) {
278 StrgenFatal("Command '{}' has no (default) plural position", cmd == nullptr ? "<empty>" : cmd->cmd);
279 }
280 offset = cmd->default_plural_offset;
281 }
282
283 /* Parse each string */
284 std::vector<std::string> words;
285 for (;;) {
286 auto word = ParseWord(&buf);
287 if (!word.has_value()) break;
288 words.emplace_back(*word);
289 }
290
291 if (words.empty()) {
292 StrgenFatal("{}: No plural words", _cur_ident);
293 }
294
295 size_t expected = _plural_forms[_strgen.lang.plural_form].plural_count;
296 if (expected != words.size()) {
297 if (_translated) {
298 StrgenFatal("{}: Invalid number of plural forms. Expecting {}, found {}.", _cur_ident,
299 expected, words.size());
300 } else {
301 if (_strgen.show_warnings) StrgenWarning("'{}' is untranslated. Tweaking english string to allow compilation for plural forms", _cur_ident);
302 if (words.size() > expected) {
303 words.resize(expected);
304 } else {
305 while (words.size() < expected) {
306 words.push_back(words.back());
307 }
308 }
309 }
310 }
311
312 builder.PutUtf8(SCC_PLURAL_LIST);
313 builder.PutUint8(_strgen.lang.plural_form);
314 builder.PutUint8(static_cast<uint8_t>(TranslateArgumentIdx(*argidx, *offset)));
315 EmitWordList(builder, words);
316}
317
318void EmitGender(StringBuilder &builder, const char *buf, char32_t)
319{
320 if (buf[0] == '=') {
321 buf++;
322
323 /* This is a {G=DER} command */
324 auto nw = _strgen.lang.GetGenderIndex(buf);
325 if (nw >= MAX_NUM_GENDERS) StrgenFatal("G argument '{}' invalid", buf);
326
327 /* now nw contains the gender index */
328 builder.PutUtf8(SCC_GENDER_INDEX);
329 builder.PutUint8(nw);
330 } else {
331 /* This is a {G 0 foo bar two} command.
332 * If no relative number exists, default to +0 */
333 auto [argidx, offset] = ParseRelNum(&buf);
334 if (!argidx.has_value()) argidx = _cur_argidx;
335 if (!offset.has_value()) offset = 0;
336
337 const CmdStruct *cmd = _cur_pcs.consuming_commands[*argidx];
338 if (cmd == nullptr || !cmd->flags.Test(CmdFlag::Gender)) {
339 StrgenFatal("Command '{}' can't have a gender", cmd == nullptr ? "<empty>" : cmd->cmd);
340 }
341
342 std::vector<std::string> words;
343 for (;;) {
344 auto word = ParseWord(&buf);
345 if (!word.has_value()) break;
346 words.emplace_back(*word);
347 }
348 if (words.size() != _strgen.lang.num_genders) StrgenFatal("Bad # of arguments for gender command");
349
350 assert(IsInsideBS(cmd->value, SCC_CONTROL_START, UINT8_MAX));
351 builder.PutUtf8(SCC_GENDER_LIST);
352 builder.PutUint8(static_cast<uint8_t>(TranslateArgumentIdx(*argidx, *offset)));
353 EmitWordList(builder, words);
354 }
355}
356
357static const CmdStruct *FindCmd(std::string_view s)
358{
359 for (const auto &cs : _cmd_structs) {
360 if (cs.cmd == s) return &cs;
361 }
362 return nullptr;
363}
364
365static uint8_t ResolveCaseName(std::string_view str)
366{
367 uint8_t case_idx = _strgen.lang.GetCaseIndex(str);
368 if (case_idx >= MAX_NUM_CASES) StrgenFatal("Invalid case-name '{}'", str);
369 return case_idx + 1;
370}
371
372/* returns cmd == nullptr on eof */
373static ParsedCommandString ParseCommandString(const char **str)
374{
375 ParsedCommandString result;
376 const char *s = *str;
377
378 /* Scan to the next command, exit if there's no next command. */
379 for (; *s != '{'; s++) {
380 if (*s == '\0') return {};
381 }
382 s++; // Skip past the {
383
384 if (*s >= '0' && *s <= '9') {
385 char *end;
386
387 result.argno = std::strtoul(s, &end, 0);
388 if (*end != ':') StrgenFatal("missing arg #");
389 s = end + 1;
390 }
391
392 /* parse command name */
393 const char *start = s;
394 char c;
395 do {
396 c = *s++;
397 } while (c != '}' && c != ' ' && c != '=' && c != '.' && c != 0);
398
399 std::string_view command(start, s - start - 1);
400 result.cmd = FindCmd(command);
401 if (result.cmd == nullptr) {
402 StrgenError("Undefined command '{}'", command);
403 return {};
404 }
405
406 if (c == '.') {
407 const char *casep = s;
408
409 if (!result.cmd->flags.Test(CmdFlag::Case)) {
410 StrgenFatal("Command '{}' can't have a case", result.cmd->cmd);
411 }
412
413 do {
414 c = *s++;
415 } while (c != '}' && c != ' ' && c != '\0');
416 result.casei = ResolveCaseName(std::string_view(casep, s - casep - 1));
417 }
418
419 if (c == '\0') {
420 StrgenError("Missing }} from command '{}'", start);
421 return {};
422 }
423
424 if (c != '}') {
425 if (c == '=') s--;
426 /* copy params */
427 start = s;
428 for (;;) {
429 c = *s++;
430 if (c == '}') break;
431 if (c == '\0') {
432 StrgenError("Missing }} from command '{}'", start);
433 return {};
434 }
435 result.param += c;
436 }
437 }
438
439 *str = s;
440
441 return result;
442}
443
451StringReader::StringReader(StringData &data, const std::string &file, bool master, bool translation) :
452 data(data), file(file), master(master), translation(translation)
453{
454}
455
456ParsedCommandStruct ExtractCommandString(const char *s, bool)
457{
459
460 size_t argidx = 0;
461 for (;;) {
462 /* read until next command from a. */
463 auto cs = ParseCommandString(&s);
464
465 if (cs.cmd == nullptr) break;
466
467 /* Sanity checking */
468 if (cs.argno.has_value() && cs.cmd->consumes == 0) StrgenFatal("Non consumer param can't have a paramindex");
469
470 if (cs.cmd->consumes > 0) {
471 if (cs.argno.has_value()) argidx = *cs.argno;
472 if (argidx >= p.consuming_commands.max_size()) StrgenFatal("invalid param idx {}", argidx);
473 if (p.consuming_commands[argidx] != nullptr && p.consuming_commands[argidx] != cs.cmd) StrgenFatal("duplicate param idx {}", argidx);
474
475 p.consuming_commands[argidx++] = cs.cmd;
476 } else if (!cs.cmd->flags.Test(CmdFlag::DontCount)) { // Ignore some of them
477 p.non_consuming_commands.emplace_back(cs.cmd, std::move(cs.param));
478 }
479 }
480
481 return p;
482}
483
484const CmdStruct *TranslateCmdForCompare(const CmdStruct *a)
485{
486 if (a == nullptr) return nullptr;
487
488 if (a->cmd == "STRING1" ||
489 a->cmd == "STRING2" ||
490 a->cmd == "STRING3" ||
491 a->cmd == "STRING4" ||
492 a->cmd == "STRING5" ||
493 a->cmd == "STRING6" ||
494 a->cmd == "STRING7" ||
495 a->cmd == "RAW_STRING") {
496 return FindCmd("STRING");
497 }
498
499 return a;
500}
501
502static bool CheckCommandsMatch(const char *a, const char *b, const char *name)
503{
504 /* If we're not translating, i.e. we're compiling the base language,
505 * it is pointless to do all these checks as it'll always be correct.
506 * After all, all checks are based on the base language.
507 */
508 if (!_strgen.translation) return true;
509
510 bool result = true;
511
512 ParsedCommandStruct templ = ExtractCommandString(b, true);
513 ParsedCommandStruct lang = ExtractCommandString(a, true);
514
515 /* For each string in templ, see if we find it in lang */
516 if (templ.non_consuming_commands.max_size() != lang.non_consuming_commands.max_size()) {
517 StrgenWarning("{}: template string and language string have a different # of commands", name);
518 result = false;
519 }
520
521 for (auto &templ_nc : templ.non_consuming_commands) {
522 /* see if we find it in lang, and zero it out */
523 bool found = false;
524 for (auto &lang_nc : lang.non_consuming_commands) {
525 if (templ_nc.cmd == lang_nc.cmd && templ_nc.param == lang_nc.param) {
526 /* it was found in both. zero it out from lang so we don't find it again */
527 lang_nc.cmd = nullptr;
528 found = true;
529 break;
530 }
531 }
532
533 if (!found) {
534 StrgenWarning("{}: command '{}' exists in template file but not in language file", name, templ_nc.cmd->cmd);
535 result = false;
536 }
537 }
538
539 /* if we reach here, all non consumer commands match up.
540 * Check if the non consumer commands match up also. */
541 for (size_t i = 0; i < templ.consuming_commands.max_size(); i++) {
542 if (TranslateCmdForCompare(templ.consuming_commands[i]) != lang.consuming_commands[i]) {
543 StrgenWarning("{}: Param idx #{} '{}' doesn't match with template command '{}'", name, i,
544 lang.consuming_commands[i] == nullptr ? "<empty>" : TranslateCmdForCompare(lang.consuming_commands[i])->cmd,
545 templ.consuming_commands[i] == nullptr ? "<empty>" : templ.consuming_commands[i]->cmd);
546 result = false;
547 }
548 }
549
550 return result;
551}
552
553void StringReader::HandleString(char *str)
554{
555 if (*str == '#') {
556 if (str[1] == '#' && str[2] != '#') this->HandlePragma(str + 2, _strgen.lang);
557 return;
558 }
559
560 /* Ignore comments & blank lines */
561 if (*str == ';' || *str == ' ' || *str == '\0') return;
562
563 char *s = strchr(str, ':');
564 if (s == nullptr) {
565 StrgenError("Line has no ':' delimiter");
566 return;
567 }
568
569 char *t;
570 /* Trim spaces.
571 * After this str points to the command name, and s points to the command contents */
572 for (t = s; t > str && (t[-1] == ' ' || t[-1] == '\t'); t--) {}
573 *t = 0;
574 s++;
575
576 /* Check string is valid UTF-8 */
577 const char *tmp;
578 for (tmp = s; *tmp != '\0';) {
579 size_t len = Utf8Validate(tmp);
580 if (len == 0) StrgenFatal("Invalid UTF-8 sequence in '{}'", s);
581
582 char32_t c;
583 Utf8Decode(&c, tmp);
584 if (c <= 0x001F || // ASCII control character range
585 c == 0x200B || // Zero width space
586 (c >= 0xE000 && c <= 0xF8FF) || // Private range
587 (c >= 0xFFF0 && c <= 0xFFFF)) { // Specials range
588 StrgenFatal("Unwanted UTF-8 character U+{:04X} in sequence '{}'", static_cast<uint32_t>(c), s);
589 }
590
591 tmp += len;
592 }
593
594 /* Check if the string has a case..
595 * The syntax for cases is IDENTNAME.case */
596 char *casep = strchr(str, '.');
597 if (casep != nullptr) *casep++ = '\0';
598
599 /* Check if this string already exists.. */
600 LangString *ent = this->data.Find(str);
601
602 if (this->master) {
603 if (casep != nullptr) {
604 StrgenError("Cases in the base translation are not supported.");
605 return;
606 }
607
608 if (ent != nullptr) {
609 StrgenError("String name '{}' is used multiple times", str);
610 return;
611 }
612
613 if (this->data.strings[this->data.next_string_id] != nullptr) {
614 StrgenError("String ID 0x{:X} for '{}' already in use by '{}'", this->data.next_string_id, str, this->data.strings[this->data.next_string_id]->name);
615 return;
616 }
617
618 /* Allocate a new LangString */
619 this->data.Add(std::make_unique<LangString>(str, s, this->data.next_string_id++, _strgen.cur_line));
620 } else {
621 if (ent == nullptr) {
622 StrgenWarning("String name '{}' does not exist in master file", str);
623 return;
624 }
625
626 if (!ent->translated.empty() && casep == nullptr) {
627 StrgenError("String name '{}' is used multiple times", str);
628 return;
629 }
630
631 /* make sure that the commands match */
632 if (!CheckCommandsMatch(s, ent->english.c_str(), str)) return;
633
634 if (casep != nullptr) {
635 ent->translated_cases.emplace_back(ResolveCaseName(casep), s);
636 } else {
637 ent->translated = s;
638 /* If the string was translated, use the line from the
639 * translated language so errors in the translated file
640 * are properly referenced to. */
641 ent->line = _strgen.cur_line;
642 }
643 }
644}
645
647{
648 if (!memcmp(str, "plural ", 7)) {
649 lang.plural_form = atoi(str + 7);
650 if (lang.plural_form >= lengthof(_plural_forms)) {
651 StrgenFatal("Invalid pluralform {}", lang.plural_form);
652 }
653 } else {
654 StrgenFatal("unknown pragma '{}'", str);
655 }
656}
657
658static void StripTrailingWhitespace(std::string &str)
659{
660 str.erase(str.find_last_not_of("\r\n ") + 1);
661}
662
664{
665 _strgen.warnings = _strgen.errors = 0;
666
667 _strgen.translation = this->translation;
668 _strgen.file = this->file;
669
670 /* For each new file we parse, reset the genders, and language codes. */
671 MemSetT(&_strgen.lang, 0);
672 strecpy(_strgen.lang.digit_group_separator, ",");
675
676 _strgen.cur_line = 1;
677 while (this->data.next_string_id < this->data.max_strings) {
678 std::optional<std::string> line = this->ReadLine();
679 if (!line.has_value()) return;
680
681 StripTrailingWhitespace(line.value());
682 this->HandleString(line.value().data());
683 _strgen.cur_line++;
684 }
685
686 if (this->data.next_string_id == this->data.max_strings) {
687 StrgenError("Too many strings, maximum allowed is {}", this->data.max_strings);
688 }
689}
690
696{
697 size_t last = 0;
698 for (size_t i = 0; i < data.max_strings; i++) {
699 if (data.strings[i] != nullptr) {
700 this->WriteStringID(data.strings[i]->name, i);
701 last = i;
702 }
703 }
704
705 this->WriteStringID("STR_LAST_STRINGID", last);
706}
707
708static size_t TranslateArgumentIdx(size_t argidx, size_t offset)
709{
710 if (argidx >= _cur_pcs.consuming_commands.max_size()) {
711 StrgenFatal("invalid argidx {}", argidx);
712 }
713 const CmdStruct *cs = _cur_pcs.consuming_commands[argidx];
714 if (cs != nullptr && cs->consumes <= offset) {
715 StrgenFatal("invalid argidx offset {}:{}", argidx, offset);
716 }
717
718 if (_cur_pcs.consuming_commands[argidx] == nullptr) {
719 StrgenFatal("no command for this argidx {}", argidx);
720 }
721
722 size_t sum = 0;
723 for (size_t i = 0; i < argidx; i++) {
724 cs = _cur_pcs.consuming_commands[i];
725
726 sum += (cs != nullptr) ? cs->consumes : 1;
727 }
728
729 return sum + offset;
730}
731
732static void PutArgidxCommand(StringBuilder &builder)
733{
734 builder.PutUtf8(SCC_ARG_INDEX);
735 builder.PutUint8(static_cast<uint8_t>(TranslateArgumentIdx(_cur_argidx)));
736}
737
738static std::string PutCommandString(const char *str)
739{
740 std::string result;
741 StringBuilder builder(result);
742 _cur_argidx = 0;
743
744 while (*str != '\0') {
745 /* Process characters as they are until we encounter a { */
746 if (*str != '{') {
747 builder.PutChar(*str++);
748 continue;
749 }
750
751 auto cs = ParseCommandString(&str);
752 auto *cmd = cs.cmd;
753 if (cmd == nullptr) break;
754
755 if (cs.casei.has_value()) {
756 builder.PutUtf8(SCC_SET_CASE); // {SET_CASE}
757 builder.PutUint8(*cs.casei);
758 }
759
760 /* For params that consume values, we need to handle the argindex properly */
761 if (cmd->consumes > 0) {
762 /* Check if we need to output a move-param command */
763 if (cs.argno.has_value() && *cs.argno != _cur_argidx) {
764 _cur_argidx = *cs.argno;
765 PutArgidxCommand(builder);
766 }
767
768 /* Output the one from the master string... it's always accurate. */
769 cmd = _cur_pcs.consuming_commands[_cur_argidx++];
770 if (cmd == nullptr) {
771 StrgenFatal("{}: No argument exists at position {}", _cur_ident, _cur_argidx - 1);
772 }
773 }
774
775 cmd->proc(builder, cs.param.c_str(), cmd->value);
776 }
777 return result;
778}
779
785{
786 char buffer[2];
787 size_t offs = 0;
788 if (length >= 0x4000) {
789 StrgenFatal("string too long");
790 }
791
792 if (length >= 0xC0) {
793 buffer[offs++] = static_cast<char>(static_cast<uint8_t>((length >> 8) | 0xC0));
794 }
795 buffer[offs++] = static_cast<char>(static_cast<uint8_t>(length & 0xFF));
796 this->Write(buffer, offs);
797}
798
804{
805 std::vector<size_t> in_use;
806 for (size_t tab = 0; tab < data.tabs; tab++) {
807 size_t n = data.CountInUse(tab);
808
809 in_use.push_back(n);
810 _strgen.lang.offsets[tab] = TO_LE16(static_cast<uint16_t>(n));
811
812 for (size_t j = 0; j != in_use[tab]; j++) {
813 const LangString *ls = data.strings[(tab * TAB_SIZE) + j].get();
814 if (ls != nullptr && ls->translated.empty()) _strgen.lang.missing++;
815 }
816 }
817
818 _strgen.lang.ident = TO_LE32(LanguagePackHeader::IDENT);
819 _strgen.lang.version = TO_LE32(data.Version());
820 _strgen.lang.missing = TO_LE16(_strgen.lang.missing);
821 _strgen.lang.winlangid = TO_LE16(_strgen.lang.winlangid);
822
823 this->WriteHeader(&_strgen.lang);
824
825 for (size_t tab = 0; tab < data.tabs; tab++) {
826 for (size_t j = 0; j != in_use[tab]; j++) {
827 const LangString *ls = data.strings[(tab * TAB_SIZE) + j].get();
828
829 /* For undefined strings, just set that it's an empty string */
830 if (ls == nullptr) {
831 this->WriteLength(0);
832 continue;
833 }
834
835 std::string output;
836 StringBuilder builder(output);
837 _cur_ident = ls->name.c_str();
838 _strgen.cur_line = ls->line;
839
840 /* Produce a message if a string doesn't have a translation. */
841 if (ls->translated.empty()) {
842 if (_strgen.show_warnings) {
843 StrgenWarning("'{}' is untranslated", ls->name);
844 }
845 if (_strgen.annotate_todos) {
846 builder.Put("<TODO> ");
847 }
848 }
849
850 /* Extract the strings and stuff from the english command string */
851 _cur_pcs = ExtractCommandString(ls->english.c_str(), false);
852
853 _translated = !ls->translated_cases.empty() || !ls->translated.empty();
854 const std::string &cmdp = _translated ? ls->translated : ls->english;
855
856 if (!ls->translated_cases.empty()) {
857 /* Need to output a case-switch.
858 * It has this format
859 * <0x9E> <NUM CASES> <CASE1> <LEN1> <STRING1> <CASE2> <LEN2> <STRING2> <CASE3> <LEN3> <STRING3> <LENDEFAULT> <STRINGDEFAULT>
860 * Each LEN is printed using 2 bytes in little endian order. */
861 builder.PutUtf8(SCC_SWITCH_CASE);
862 builder.PutUint8(static_cast<uint8_t>(ls->translated_cases.size()));
863
864 /* Write each case */
865 for (const Case &c : ls->translated_cases) {
866 auto case_str = PutCommandString(c.string.c_str());
867 builder.PutUint8(c.caseidx);
868 builder.PutUint16LE(static_cast<uint16_t>(case_str.size()));
869 builder.Put(case_str);
870 }
871 }
872
873 std::string def_str;
874 if (!cmdp.empty()) def_str = PutCommandString(cmdp.c_str());
875 if (!ls->translated_cases.empty()) {
876 builder.PutUint16LE(static_cast<uint16_t>(def_str.size()));
877 }
878 builder.Put(def_str);
879
880 this->WriteLength(output.size());
881 this->Write(output.data(), output.size());
882 }
883 }
884}
debug_inline constexpr bool HasBit(const T x, const uint8_t y)
Checks if a bit in a value is set.
debug_inline static constexpr uint GB(const T x, const uint8_t s, const uint8_t n)
Fetch n bits from x, started at bit s.
constexpr bool Test(Tvalue_type value) const
Test if the value-th bit is set.
void PutUtf8(char32_t c)
Append UTF.8 char.
void PutUint16LE(uint16_t value)
Append binary uint16 using little endian.
void Put(std::string_view str)
Append string.
void PutChar(char c)
Append 8-bit char.
void PutUint8(uint8_t value)
Append binary uint8.
Compose data into a growing std::string.
static const uint8_t MAX_NUM_GENDERS
Maximum number of supported genders.
Definition language.h:20
static const uint8_t MAX_NUM_CASES
Maximum number of supported cases.
Definition language.h:21
constexpr bool IsInsideBS(const T x, const size_t base, const size_t size)
Checks if a value is between a window started at some base point.
void MemSetT(T *ptr, uint8_t value, size_t num=1)
Type-safe version of memset().
Definition mem_func.hpp:36
#define lengthof(array)
Return the length of an fixed size array.
Definition stdafx.h:271
Structures related to strgen.
static bool _translated
Whether the current language is not the master language.
static uint32_t VersionHashStr(uint32_t hash, std::string_view s)
Create a compound hash.
static const PluralForm _plural_forms[]
All plural forms used.
@ Gender
These commands support genders.
@ Case
These commands support cases.
@ DontCount
These commands aren't counted for comparison.
void strecpy(std::span< char > dst, std::string_view src)
Copies characters from one buffer to another.
Definition string.cpp:57
size_t Utf8Decode(char32_t *c, const char *s)
Decode and consume the next UTF-8 encoded character.
Definition string.cpp:441
static const uint TAB_SIZE
Number of strings per StringTab.
Container for the different cases of a string.
Definition strgen.h:20
Case(uint8_t caseidx, const std::string &string)
Create a new case.
virtual void WriteStringID(const std::string &name, size_t stringid)=0
Write the string ID.
void WriteHeader(const StringData &data)
Write the header information.
Information about a single string.
Definition strgen.h:28
size_t line
Line of string in source-file.
Definition strgen.h:33
std::string english
English text.
Definition strgen.h:30
std::vector< Case > translated_cases
Cases of the translation.
Definition strgen.h:34
std::string translated
Translated text.
Definition strgen.h:31
void FreeTranslation()
Free all data related to the translation.
std::string name
Name of the string.
Definition strgen.h:29
LangString(const std::string &name, const std::string &english, size_t index, size_t line)
Create a new string.
Header of a language file.
Definition language.h:24
uint8_t GetCaseIndex(std::string_view case_str) const
Get the index for the given case.
Definition language.h:81
uint8_t plural_form
plural form index
Definition language.h:41
uint32_t version
32-bits of auto generated version info which is basically a hash of strings.h
Definition language.h:28
uint16_t offsets[TEXT_TAB_END]
the offsets
Definition language.h:32
uint16_t winlangid
Windows language ID: Windows cannot and will not convert isocodes to something it can use to determin...
Definition language.h:51
uint8_t num_genders
the number of genders of this language
Definition language.h:53
uint16_t missing
number of missing strings.
Definition language.h:40
char digit_group_separator[8]
Thousand separator used for anything not currencies.
Definition language.h:35
uint32_t ident
32-bits identifier
Definition language.h:27
char digit_decimal_separator[8]
Decimal separator.
Definition language.h:39
char digit_group_separator_currency[8]
Thousand separator used for currencies.
Definition language.h:37
static const uint32_t IDENT
Identifier for OpenTTD language files, big endian for "LANG".
Definition language.h:25
uint8_t GetGenderIndex(std::string_view gender_str) const
Get the index for the given gender.
Definition language.h:68
virtual void WriteHeader(const LanguagePackHeader *header)=0
Write the header metadata.
virtual void WriteLength(size_t length)
Write the length as a simple gamma.
virtual void Write(const char *buffer, size_t length)=0
Write a number of bytes.
virtual void WriteLang(const StringData &data)
Actually write the language.
size_t plural_count
The number of plural forms.
Global state shared between strgen.cpp, game_text.cpp and strgen_base.cpp.
Definition strgen.h:158
std::string file
The filename of the input, so we can refer to it in errors/warnings.
Definition strgen.h:159
bool translation
Is the current file actually a translation or not.
Definition strgen.h:165
LanguagePackHeader lang
Header information about a language.
Definition strgen.h:166
size_t cur_line
The current line we're parsing in the input file.
Definition strgen.h:160
Information about the currently known strings.
Definition strgen.h:41
size_t tabs
The number of 'tabs' of strings.
Definition strgen.h:44
std::unordered_map< std::string, std::shared_ptr< LangString > > name_to_string
Lookup table for the strings.
Definition strgen.h:43
void Add(std::shared_ptr< LangString > ls)
Add a newly created LangString.
size_t max_strings
The maximum number of strings.
Definition strgen.h:45
size_t next_string_id
The next string ID to allocate.
Definition strgen.h:46
void FreeTranslation()
Free all data related to the translation.
StringData(size_t tabs)
Create a new string data container.
LangString * Find(const std::string &s)
Find a LangString based on the string name.
std::vector< std::shared_ptr< LangString > > strings
List of all known strings.
Definition strgen.h:42
uint32_t Version() const
Make a hash of the file to get a unique "version number".
size_t CountInUse(size_t tab) const
Count the number of tab elements that are in use.
const std::string file
The file we are reading.
Definition strgen.h:59
StringReader(StringData &data, const std::string &file, bool master, bool translation)
Prepare reading.
StringData & data
The data to fill during reading.
Definition strgen.h:58
virtual void HandlePragma(char *str, LanguagePackHeader &lang)
Handle the pragma of the file.
virtual void ParseFile()
Start parsing the file.
bool translation
Are we reading a translation, implies !master. However, the base translation will have this false.
Definition strgen.h:61
virtual std::optional< std::string > ReadLine()=0
Read a single line from the source of strings.
bool master
Are we reading the master file?
Definition strgen.h:60