-
Notifications
You must be signed in to change notification settings - Fork 8
/
rc.cc
3871 lines (3474 loc) · 129 KB
/
rc.cc
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
/*
clang++ -std=c++14 -o rc rc.cc -Wall -Wno-c++11-narrowing
or
cl rc.cc /EHsc /wd4838 /nologo shlwapi.lib
./rc < foo.rc
A sketch of a reimplemenation of rc.exe, for research purposes.
Doesn't do any preprocessing for now.
For files that this successfully processes, the goal is that the output is
bit-for-bit equal to what Microsoft rc.exe produces. It's ok if this program
rejects some inputs that Microsoft rc.exe accepts.
Missing for chromium:
- real string literal parser (L"\0")
- preprocessor (but see pptest next to this; `pptest file | rc` kinda works)
Also missing, but not yet for chromium:
- #pragma code_page()
- FONT
- MENUEX (including int expression parse/eval)
- MESSAGETABLE
- inline block data for DIALOG controls
- rc.exe probably supports int exprs in more places (see all the IntValue calls)
- mem attrs (PRELOAD LOADONCALL FIXED MOVEABLE DISCARDABLE PURE IMPURE SHARED
NONSHARED) on all resources. all no-ops nowadays, but sometimes in rc files.
https://msdn.microsoft.com/en-us/library/windows/desktop/aa380908(v=vs.85).aspx
- CHARACTERISTICS LANGUAGE VERSION for ACCELERATORS, DIALOG(EX), MENU(EX),
RCDATA, or STRINGTABLE (and custom elts?)
- MUI, https://msdn.microsoft.com/en-us/library/windows/desktop/ee264325(v=vs.85).aspx
Unicode handling:
MS rc.exe allows either UTF-16LE input (FIXME: test non-BMP files) or codepage'd
inputs. This program either accepts UTF-16 or UTF-8 input. UTF-16 is converted
to UTF-8 at the start. If the input wasn't UTF-16, this program will error
out on bytes > 127 in string literals so that real codepage'd inputs are
rejected rather than miscompiled.
Warning ideas:
- duplicate IDs in a dialog
- duplicate names for a given resource type with the same LANGUAGE
- VERSIONINFO with ID != 1
- non-numeric resource names
*/
#if defined(__linux__)
// Work around broken older libstdc++s, http://llvm.org/PR31562
extern char *gets (char *__s) __attribute__ ((__deprecated__));
#endif
#include <algorithm>
#include <assert.h>
#include <iostream>
#include <list>
#include <locale>
#include <limits.h>
#include <map>
#include <memory>
#include <stdint.h>
#include <string.h>
#include <string>
#include <unordered_map>
#include <unordered_set>
#include <vector>
#define GCC_VERSION (__GNUC__ * 10000 \
+ __GNUC_MINOR__ * 100 \
+ __GNUC_PATCHLEVEL__)
#if !(defined(__linux__) && GCC_VERSION < 50100)
// No codecvt before libstdc++5.1, need to roll our own utf8<->utf16 routines.
#include <codecvt>
#endif
#if defined(_MSC_VER)
#include <direct.h>
#include <io.h>
#include <fcntl.h>
#define NOMINMAX
#define WIN32_LEAN_AND_MEAN
#include <windows.h> // GetFullPathName
#include <shlwapi.h> // PathIsRelative
#else
#include <unistd.h>
#endif
#if __cplusplus >= 201703L
#include <optional>
#else
namespace std {
inline namespace fundamentals_v1 {
template<class T>
class optional {
bool has_val = false;
T val;
public:
optional() = default;
optional(const T& t) { val = t; has_val = true; }
void operator=(const T& t) { val = t; has_val = true; }
operator bool() const { return has_val; }
T* operator->() { return &val; }
const T* operator->() const { return &val; }
T& operator*() { return val; }
const T& operator*() const { return val; }
};
} // fundamentals_v1
} // std
#endif
#if !defined(_MSC_VER) && !defined(__linux__)
#include <string_view>
#else
namespace std {
inline namespace fundamentals_v1 {
class string_view {
const char* str_;
size_t size_;
public:
string_view() : size_(0) {}
string_view(const char* str) : str_(str), size_(strlen(str)) {}
string_view(const char* str, size_t size) : str_(str), size_(size) {}
string_view(const std::string& s) : str_(s.data()), size_(s.size()) {}
size_t size() const { return size_; }
char operator[](size_t i) const { return str_[i]; }
bool operator==(const string_view& rhs) const {
return size_ == rhs.size_ && strncmp(str_, rhs.str_, size_) == 0;
}
bool operator!=(const string_view& rhs) const { return !(*this == rhs); }
string_view substr(size_t start, size_t len = -1) const {
return string_view(str_ + start, std::min(len, size() - start));
}
const char* data() const { return str_; }
bool empty() const { return size_ == 0; }
const char* begin() const { return data(); }
const char* end() const { return begin() + size(); }
size_t find(char c) {
const char* found = (const char*)memchr(str_, c, size_);
return found ? found - str_ : -1;
}
static constexpr size_t npos = -1;
};
} // fundamentals_v1
template<> struct hash<string_view> {
size_t operator()(const string_view& x) const {
size_t hash = 5381; // djb2
for (char c : x)
hash = 33 * hash + c;
return hash;
}
};
// https://gcc.gnu.org/develop.html#timeline for __GLIBCXX__.
#if defined(__linux__) && GCC_VERSION <= 40800 && __GLIBCXX__ <= 20150623
template<typename T, typename... Args>
std::unique_ptr<T> make_unique(Args&&... args) {
return std::unique_ptr<T>(new T(std::forward<Args>(args)...));
}
#endif
} // std
#endif
std::string AsString(std::string_view v) {
return std::string(v.begin(), v.end());
}
// Like toupper(), but locale-independent.
int ascii_toupper(int c) {
if (c >= 'a' && c <= 'z')
return c - 'a' + 'A';
return c;
}
bool IsEqualAsciiUppercaseChar(char a, char b) {
return ascii_toupper(a) == ascii_toupper(b);
}
bool IsEqualAsciiUppercase(std::string_view a, std::string_view b) {
return a.size() == b.size() &&
std::equal(a.begin(), a.end(), b.begin(), IsEqualAsciiUppercaseChar);
}
size_t CaseInsensitiveHash(std::string_view x) {
size_t hash = 5381; // djb2
for (char c : x)
hash = 33 * hash + ascii_toupper(c);
return hash;
}
template <class T>
using CaseInsensitiveStringMap = std::unordered_map<
std::string_view, T,
size_t(*)(std::string_view),
bool(*)(std::string_view, std::string_view)>;
using CaseInsensitiveStringSet = std::unordered_set<
std::string_view,
size_t(*)(std::string_view),
bool(*)(std::string_view, std::string_view)>;
bool hexchar(int c, int* nibble) {
if (c >= '0' && c <= '9') {
*nibble = c - '0';
return true;
}
if (c >= 'a' && c <= 'f') {
*nibble = c - 'a' + 10;
return true;
}
if (c >= 'A' && c <= 'F') {
*nibble = c - 'A' + 10;
return true;
}
return false;
}
typedef uint8_t UTF8;
#if !defined(_MSC_VER) && !(defined(__linux__) && GCC_VERSION < 50100)
bool isLegalUTF8String(const UTF8 *source, const UTF8 *sourceEnd) {
// Validate that the input if valid utf-8.
std::mbstate_t mbstate;
return std::codecvt_utf8<char32_t>().length(
mbstate, (char*)source, (char*)sourceEnd,
std::numeric_limits<size_t>::max())
== sourceEnd - source;
}
#else
// codecvt_utf8::length() is broken with MSVC, and doesn't exist for gcc
// older than 5.1.. Use code from the Unicode consortium there. License
// below applies to contents of this #else block only.
/*
* Copyright 2001-2004 Unicode, Inc.
*
* Disclaimer
*
* This source code is provided as is by Unicode, Inc. No claims are
* made as to fitness for any particular purpose. No warranties of any
* kind are expressed or implied. The recipient agrees to determine
* applicability of information provided. If this file has been
* purchased on magnetic or optical media from Unicode, Inc., the
* sole remedy for any claim will be exchange of defective media
* within 90 days of receipt.
*
* Limitations on Rights to Redistribute This Code
*
* Unicode, Inc. hereby grants the right to freely use the information
* supplied in this file in the creation of products supporting the
* Unicode Standard, and to make copies of this file in any form
* for internal or external distribution as long as this notice
* remains attached.
*/
static const char trailingBytesForUTF8[256] = {
0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0, 0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,
0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0, 0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,
0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0, 0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,
0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0, 0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,
0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0, 0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,
0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0, 0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,
1,1,1,1,1,1,1,1,1,1,1,1,1,1,1,1, 1,1,1,1,1,1,1,1,1,1,1,1,1,1,1,1,
2,2,2,2,2,2,2,2,2,2,2,2,2,2,2,2, 3,3,3,3,3,3,3,3,4,4,4,4,5,5,5,5
};
static bool isLegalUTF8(const UTF8 *source, int length) {
UTF8 a;
const UTF8 *srcptr = source+length;
switch (length) {
default: return false;
/* Everything else falls through when "true"... */
case 4: if ((a = (*--srcptr)) < 0x80 || a > 0xBF) return false;
case 3: if ((a = (*--srcptr)) < 0x80 || a > 0xBF) return false;
case 2: if ((a = (*--srcptr)) < 0x80 || a > 0xBF) return false;
switch (*source) {
/* no fall-through in this inner switch */
case 0xE0: if (a < 0xA0) return false; break;
case 0xED: if (a > 0x9F) return false; break;
case 0xF0: if (a < 0x90) return false; break;
case 0xF4: if (a > 0x8F) return false; break;
default: if (a < 0x80) return false;
}
case 1: if (*source >= 0x80 && *source < 0xC2) return false;
}
if (*source > 0xF4) return false;
return true;
}
bool isLegalUTF8String(const UTF8 *source, const UTF8 *sourceEnd) {
while (source != sourceEnd) {
int length = trailingBytesForUTF8[*source] + 1;
if (length > sourceEnd - source || !isLegalUTF8(source, length))
return false;
source += length;
}
return true;
}
#endif
#if defined(__linux__) && GCC_VERSION < 50100
// No codecvt before libstdc++5.1, need to roll our own utf8<->utf16 routines.
// Use code from the Unicode consortium there. License below applies to
// contents of this #if block only.
/*
* Copyright 2001-2004 Unicode, Inc.
*
* Disclaimer
*
* This source code is provided as is by Unicode, Inc. No claims are
* made as to fitness for any particular purpose. No warranties of any
* kind are expressed or implied. The recipient agrees to determine
* applicability of information provided. If this file has been
* purchased on magnetic or optical media from Unicode, Inc., the
* sole remedy for any claim will be exchange of defective media
* within 90 days of receipt.
*
* Limitations on Rights to Redistribute This Code
*
* Unicode, Inc. hereby grants the right to freely use the information
* supplied in this file in the creation of products supporting the
* Unicode Standard, and to make copies of this file in any form
* for internal or external distribution as long as this notice
* remains attached.
*/
typedef uint16_t UTF16;
typedef uint32_t UTF32;
#define UNI_REPLACEMENT_CHAR (UTF32)0x0000FFFD
#define UNI_MAX_BMP (UTF32)0x0000FFFF
#define UNI_MAX_UTF16 (UTF32)0x0010FFFF
typedef enum {
conversionOK, /* conversion successful */
sourceExhausted, /* partial character in source, but hit end */
targetExhausted, /* insuff. room in target for conversion */
sourceIllegal /* source sequence is illegal/malformed */
} ConversionResult;
typedef enum {
strictConversion = 0,
lenientConversion
} ConversionFlags;
ConversionResult ConvertUTF8toUTF16 (
const UTF8** sourceStart, const UTF8* sourceEnd,
UTF16** targetStart, UTF16* targetEnd, ConversionFlags flags);
ConversionResult ConvertUTF16toUTF8 (
const UTF16** sourceStart, const UTF16* sourceEnd,
UTF8** targetStart, UTF8* targetEnd, ConversionFlags flags);
static const int halfShift = 10; /* used for shifting by 10 bits */
static const UTF32 halfBase = 0x0010000UL;
static const UTF32 halfMask = 0x3FFUL;
#define UNI_SUR_HIGH_START (UTF32)0xD800
#define UNI_SUR_HIGH_END (UTF32)0xDBFF
#define UNI_SUR_LOW_START (UTF32)0xDC00
#define UNI_SUR_LOW_END (UTF32)0xDFFF
/* --------------------------------------------------------------------- */
/*
* Magic values subtracted from a buffer value during UTF8 conversion.
* This table contains as many values as there might be trailing bytes
* in a UTF-8 sequence.
*/
static const UTF32 offsetsFromUTF8[6] = { 0x00000000UL, 0x00003080UL, 0x000E2080UL,
0x03C82080UL, 0xFA082080UL, 0x82082080UL };
/*
* Once the bits are split out into bytes of UTF-8, this is a mask OR-ed
* into the first byte, depending on how many bytes follow. There are
* as many entries in this table as there are UTF-8 sequence types.
* (I.e., one byte sequence, two byte... etc.). Remember that sequencs
* for *legal* UTF-8 will be 4 or fewer bytes total.
*/
static const UTF8 firstByteMark[7] = { 0x00, 0x00, 0xC0, 0xE0, 0xF0, 0xF8, 0xFC };
/* The interface converts a whole buffer to avoid function-call overhead.
* Constants have been gathered. Loops & conditionals have been removed as
* much as possible for efficiency, in favor of drop-through switches.
* (See "Note A" at the bottom of the file for equivalent code.)
* If your compiler supports it, the "isLegalUTF8" call can be turned
* into an inline function.
*/
/* --------------------------------------------------------------------- */
ConversionResult ConvertUTF16toUTF8 (
const UTF16** sourceStart, const UTF16* sourceEnd,
UTF8** targetStart, UTF8* targetEnd, ConversionFlags flags) {
ConversionResult result = conversionOK;
const UTF16* source = *sourceStart;
UTF8* target = *targetStart;
while (source < sourceEnd) {
UTF32 ch;
unsigned short bytesToWrite = 0;
const UTF32 byteMask = 0xBF;
const UTF32 byteMark = 0x80;
const UTF16* oldSource = source; /* In case we have to back up because of target overflow. */
ch = *source++;
/* If we have a surrogate pair, convert to UTF32 first. */
if (ch >= UNI_SUR_HIGH_START && ch <= UNI_SUR_HIGH_END) {
/* If the 16 bits following the high surrogate are in the source buffer... */
if (source < sourceEnd) {
UTF32 ch2 = *source;
/* If it's a low surrogate, convert to UTF32. */
if (ch2 >= UNI_SUR_LOW_START && ch2 <= UNI_SUR_LOW_END) {
ch = ((ch - UNI_SUR_HIGH_START) << halfShift)
+ (ch2 - UNI_SUR_LOW_START) + halfBase;
++source;
} else if (flags == strictConversion) { /* it's an unpaired high surrogate */
--source; /* return to the illegal value itself */
result = sourceIllegal;
break;
}
} else { /* We don't have the 16 bits following the high surrogate. */
--source; /* return to the high surrogate */
result = sourceExhausted;
break;
}
} else if (flags == strictConversion) {
/* UTF-16 surrogate values are illegal in UTF-32 */
if (ch >= UNI_SUR_LOW_START && ch <= UNI_SUR_LOW_END) {
--source; /* return to the illegal value itself */
result = sourceIllegal;
break;
}
}
/* Figure out how many bytes the result will require */
if (ch < (UTF32)0x80) { bytesToWrite = 1;
} else if (ch < (UTF32)0x800) { bytesToWrite = 2;
} else if (ch < (UTF32)0x10000) { bytesToWrite = 3;
} else if (ch < (UTF32)0x110000) { bytesToWrite = 4;
} else { bytesToWrite = 3;
ch = UNI_REPLACEMENT_CHAR;
}
target += bytesToWrite;
if (target > targetEnd) {
source = oldSource; /* Back up source pointer! */
target -= bytesToWrite; result = targetExhausted; break;
}
switch (bytesToWrite) { /* note: everything falls through. */
case 4: *--target = (UTF8)((ch | byteMark) & byteMask); ch >>= 6;
case 3: *--target = (UTF8)((ch | byteMark) & byteMask); ch >>= 6;
case 2: *--target = (UTF8)((ch | byteMark) & byteMask); ch >>= 6;
case 1: *--target = (UTF8)(ch | firstByteMark[bytesToWrite]);
}
target += bytesToWrite;
}
*sourceStart = source;
*targetStart = target;
return result;
}
/* --------------------------------------------------------------------- */
ConversionResult ConvertUTF8toUTF16 (
const UTF8** sourceStart, const UTF8* sourceEnd,
UTF16** targetStart, UTF16* targetEnd, ConversionFlags flags) {
ConversionResult result = conversionOK;
const UTF8* source = *sourceStart;
UTF16* target = *targetStart;
while (source < sourceEnd) {
UTF32 ch = 0;
unsigned short extraBytesToRead = trailingBytesForUTF8[*source];
if (source + extraBytesToRead >= sourceEnd) {
result = sourceExhausted; break;
}
/* Do this check whether lenient or strict */
if (! isLegalUTF8(source, extraBytesToRead+1)) {
result = sourceIllegal;
break;
}
/*
* The cases all fall through. See "Note A" below.
*/
switch (extraBytesToRead) {
case 5: ch += *source++; ch <<= 6; /* remember, illegal UTF-8 */
case 4: ch += *source++; ch <<= 6; /* remember, illegal UTF-8 */
case 3: ch += *source++; ch <<= 6;
case 2: ch += *source++; ch <<= 6;
case 1: ch += *source++; ch <<= 6;
case 0: ch += *source++;
}
ch -= offsetsFromUTF8[extraBytesToRead];
if (target >= targetEnd) {
source -= (extraBytesToRead+1); /* Back up source pointer! */
result = targetExhausted; break;
}
if (ch <= UNI_MAX_BMP) { /* Target is a character <= 0xFFFF */
/* UTF-16 surrogate values are illegal in UTF-32 */
if (ch >= UNI_SUR_HIGH_START && ch <= UNI_SUR_LOW_END) {
if (flags == strictConversion) {
source -= (extraBytesToRead+1); /* return to the illegal value itself */
result = sourceIllegal;
break;
} else {
*target++ = UNI_REPLACEMENT_CHAR;
}
} else {
*target++ = (UTF16)ch; /* normal case */
}
} else if (ch > UNI_MAX_UTF16) {
if (flags == strictConversion) {
result = sourceIllegal;
source -= (extraBytesToRead+1); /* return to the start */
break; /* Bail out; shouldn't continue */
} else {
*target++ = UNI_REPLACEMENT_CHAR;
}
} else {
/* target is a character in range 0xFFFF - 0x10FFFF. */
if (target + 1 >= targetEnd) {
source -= (extraBytesToRead+1); /* Back up source pointer! */
result = targetExhausted; break;
}
ch -= halfBase;
*target++ = (UTF16)((ch >> halfShift) + UNI_SUR_HIGH_START);
*target++ = (UTF16)((ch & halfMask) + UNI_SUR_LOW_START);
}
}
*sourceStart = source;
*targetStart = target;
return result;
}
#else
#endif
#if _MSC_VER == 1900 || _MSC_VER == 1911
// lol msvc: http://stackoverflow.com/questions/32055357/visual-studio-c-2015-stdcodecvt-with-char16-t-or-char32-t
using Char16 = int16_t;
#else
using Char16 = char16_t;
#endif
typedef std::basic_string<Char16> C16string;
//////////////////////////////////////////////////////////////////////////////
// Lexer
struct Location {
std::string Error(std::string message) const {
std::string str = AsString(file_);
if (line_)
str += ":" + std::to_string(line_);
if (column_)
str += ":" + std::to_string(column_);
return str + ": error: " + message;
}
std::string_view file_;
uint32_t line_;
uint32_t column_;
};
struct Token {
public:
enum Type {
kInvalid,
kInt, // 123
kString, // "foo"
kIdentifier, // foo
kComma, // ,
kBeginBlock, // { or BEGIN (rc.exe accepts `{ .. END`)
kEndBlock, // } or END
kPlus, // +
kMinus, // -
kPipe, // |
kAmp, // &
kTilde, // ~
kLeftParen, // (
kRightParen, // )
kDirective, // #foo
kLineComment, // //foo
kStarComment, // /* foo */
};
Token(Type type, std::string_view value, Location location)
: type_(type), value_(value), location_(location) {}
Type type() const { return type_; }
// Only valid if type() == kInt.
uint32_t IntValue(bool* is_32 = nullptr) const;
Type type_;
std::string_view value_;
Location location_;
};
uint32_t Token::IntValue(bool* is_32) const {
// C++11 has std::stoll, C++17 will likey have string_view, but there's
// no std::stoll overload taking a string_view. C has atoi / strtol, but
// both assume \0-termination. Also no std::stoul.
size_t idx;
int64_t val;
if (value_.size() >= 2 && value_[0] == '0' &&
ascii_toupper(value_[1]) == 'X') {
val = std::stoll(AsString(value_.substr(2)), &idx, 16);
idx += 2;
} else if (value_.size() >= 2 && value_[0] == '0' &&
ascii_toupper(value_[1]) == 'O') {
val = std::stoll(AsString(value_.substr(2)), &idx, 8);
idx += 2;
} else
val = std::stoll(AsString(value_), &idx, 10);
if (is_32 && idx < value_.size() && ascii_toupper(value_[idx]) == 'L')
*is_32 = true;
return (uint32_t)(uint64_t)val;;
}
class Tokenizer {
public:
static std::vector<Token> Tokenize(std::string_view file,
std::string_view source,
std::string* err);
static bool IsDigit(char c);
static bool IsDigitContinuingChar(char c);
static bool IsIdentifierFirstChar(char c);
static bool IsIdentifierContinuingChar(char c);
static bool IsWhitespace(char c);
private:
Tokenizer(std::string_view file, std::string_view input)
: file_(file), line_(1), column_(1), input_(input), cur_(0) {}
std::vector<Token> Run(std::string* err);
void Advance() { // Must only be called if not at_end().
if (IsCurrentNewline()) {
++line_;
column_ = 1;
} else {
++column_;
}
++cur_;
}
void AdvanceToNextToken();
Token::Type ClassifyCurrent() const;
void AdvanceToEndOfToken(Token::Type type);
bool FindStringTerminator(char quote_char);
bool IsCurrentNewline() const;
bool done() const { return at_end() || has_error(); }
bool at_end() const { return cur_ == input_.size(); }
char cur_char() const { return input_[cur_]; }
void SetError(std::string message) {
err_ = CurrentLocation().Error(message);
}
bool has_error() const { return !err_.empty(); }
Location CurrentLocation() const {
return Location{ file_, line_, column_ };
}
std::vector<Token> tokens_;
std::string_view file_;
uint32_t line_;
uint32_t column_;
const std::string_view input_;
std::string err_;
size_t cur_; // Byte offset into input buffer.
};
// static
std::vector<Token> Tokenizer::Tokenize(std::string_view file,
std::string_view source,
std::string* err) {
Tokenizer t(file, source);
return t.Run(err);
}
std::vector<Token> Tokenizer::Run(std::string* err) {
// Skip optional UTF-8 BOM.
if (input_.size() >= 3 && uint8_t(input_[0]) == 0xef &&
uint8_t(input_[1]) == 0xbb && uint8_t(input_[2]) == 0xbf) {
cur_ = 3;
}
while (!done()) {
AdvanceToNextToken();
if (done())
break;
Token::Type type = ClassifyCurrent();
if (type == Token::kInvalid) {
SetError("invalid token around " + AsString(input_.substr(cur_, 20)));
break;
}
size_t token_begin = cur_;
AdvanceToEndOfToken(type);
if (has_error())
break;
size_t token_end = cur_;
std::string_view token_value(&input_.data()[token_begin],
token_end - token_begin);
if (type == Token::kIdentifier) {
if (IsEqualAsciiUppercase(token_value, "BEGIN"))
type = Token::kBeginBlock;
else if (IsEqualAsciiUppercase(token_value, "END"))
type = Token::kEndBlock;
}
// For now, skip all preprocessor directives. Note that rc expects to be
// fed preprocessed input (e.g. via `clang-cl /P`, so only pragma lines
// and linemarkers should remain in the input.
// FIXME: Support linemarker directives for better diagnostics?
// FIXME: Also, probably do something with #pragma code_page(n).
if (type != Token::kLineComment && type != Token::kStarComment &&
type != Token::kDirective)
tokens_.push_back(Token(type, token_value, CurrentLocation()));
}
if (has_error()) {
tokens_.clear();
*err = err_;
}
return tokens_;
}
// static
bool Tokenizer::IsDigit(char c) {
return c >= '0' && c <= '9';
}
// static
bool Tokenizer::IsDigitContinuingChar(char c) {
// FIXME: it feels like rc.exe just allows most non-whitespace things here
// (but not + - ~).
return (c >= '0' && c <= '9') || (c >= 'a' && c <= 'f') ||
(c >= 'A' && c <= 'F') || c == 'l' || c == 'L' || c == 'x' ||
c == 'X' || c == 'o' || c == 'O';
}
// static
bool Tokenizer::IsIdentifierFirstChar(char c) {
return (c >= 'A' && c <= 'Z') || (c >= 'a' && c <= 'z') || c == '_';
}
// static
bool Tokenizer::IsIdentifierContinuingChar(char c) {
// FIXME: for '/', need to strip comments from middle of identifier
return !IsWhitespace(c) && c != ',';
}
// static
bool Tokenizer::IsWhitespace(char c) {
// Note that tab (0x09), vertical tab (0x0B), and formfeed (0x0C) are illegal.
return c == '\t' || c == '\n' || c == '\r' || c == ' ';
}
bool Tokenizer::FindStringTerminator(char quote_char) {
if (cur_char() != quote_char)
return false;
// Check for escaping. "" is not a string terminator, but """ is. Count
// the number of preceeding quotes. \" is a string terminator (but \"" isn't).
int num_inner_quotes = 0;
while (cur_ + 1 < input_.size() && input_[cur_ + 1] == quote_char) {
Advance();
num_inner_quotes++;
}
// Even quotes mean that they were escaping each other and don't count
// as terminating this string.
return (num_inner_quotes % 2) == 0;
}
bool Tokenizer::IsCurrentNewline() const {
return cur_char() == '\n';
}
void Tokenizer::AdvanceToNextToken() {
while (!at_end() && IsWhitespace(cur_char()))
Advance();
}
Token::Type Tokenizer::ClassifyCurrent() const {
char next_char = cur_char();
if (IsDigit(next_char))
return Token::kInt;
if (next_char == '"' ||
(ascii_toupper(next_char) == 'L' && cur_ + 1 < input_.size() &&
input_[cur_ + 1] == '"'))
return Token::kString;
if (IsIdentifierFirstChar(next_char))
return Token::kIdentifier;
if (next_char == ',')
return Token::kComma;
if (next_char == '{')
return Token::kBeginBlock;
if (next_char == '}')
return Token::kEndBlock;
if (next_char == '+')
return Token::kPlus;
if (next_char == '-')
return Token::kMinus;
if (next_char == '|')
return Token::kPipe;
if (next_char == '&')
return Token::kAmp;
if (next_char == '~')
return Token::kTilde;
if (next_char == '(')
return Token::kLeftParen;
if (next_char == ')')
return Token::kRightParen;
if (next_char == '#')
return Token::kDirective;
if (next_char == '/' && cur_ + 1 < input_.size()) {
// FIXME: probably better to not make tokens for those at all
if (input_[cur_ + 1] == '/')
return Token::kLineComment;
if (input_[cur_ + 1] == '*')
return Token::kStarComment;
}
return Token::kInvalid;
}
void Tokenizer::AdvanceToEndOfToken(Token::Type type) {
switch (type) {
case Token::kInt:
do {
Advance();
} while (!at_end() && IsDigitContinuingChar(cur_char()));
break;
case Token::kString: {
char initial = cur_char();
if (ascii_toupper(initial) == 'L') {
Advance();
initial = cur_char();
}
Advance(); // Advance past initial "
for (;;) {
if (at_end()) {
SetError("Unterminated string literal.");
break;
}
if (FindStringTerminator(initial)) {
Advance(); // Skip past last "
break;
} else if (IsCurrentNewline()) {
SetError("Newline in string constant.");
}
Advance();
}
break;
}
case Token::kIdentifier:
while (!at_end() && IsIdentifierContinuingChar(cur_char()))
Advance();
break;
case Token::kComma:
case Token::kBeginBlock:
case Token::kEndBlock:
case Token::kPlus:
case Token::kMinus:
case Token::kPipe:
case Token::kAmp:
case Token::kTilde:
case Token::kLeftParen:
case Token::kRightParen:
Advance(); // All are one char.
break;
case Token::kDirective: {
// Skip whitespaces.
do {
Advance();
} while (!at_end() && IsWhitespace(cur_char()) && !IsCurrentNewline());
if (!IsDigit(cur_char())) {
// Skip rest of the string if this is not a line marker.
while (!at_end() && !IsCurrentNewline())
Advance();
break;
}
// Read line marker.
size_t token_begin = cur_;
AdvanceToEndOfToken(Token::kInt);
if (has_error())
break;
size_t token_end = cur_;
std::string_view token_value(&input_.data()[token_begin],
token_end - token_begin);
uint32_t line = Token(
Token::kInt, token_value, CurrentLocation()).IntValue();
// Skip whitespaces.
do {
Advance();
} while (!at_end() && IsWhitespace(cur_char()) && !IsCurrentNewline());
// Read file name.
std::string_view file = file_;
if (cur_char() == '"') {
token_begin = cur_;
AdvanceToEndOfToken(Token::kString);
if (has_error())
break;
token_end = cur_;
token_value = std::string_view(&input_.data()[token_begin],
token_end - token_begin);
file = token_value.substr(1, token_value.size() - 2);
}
// Skip rest of the line.
while (!at_end() && !IsCurrentNewline())
Advance();
// Set new file/line.
file_ = file;
line_ = line-1;
column_ = 0;
break;
}
case Token::kLineComment:
// Eat to EOL.
while (!at_end() && !IsCurrentNewline())
Advance();
break;
case Token::kStarComment:
// Eat to */.
while (!at_end()) {
bool is_star = cur_char() == '*';
Advance();
if (is_star && !at_end() && cur_char() == '/') {
Advance();
break;
}
}
break;
case Token::kInvalid:
SetError("Everything is all messed up");
return;
}
}
//////////////////////////////////////////////////////////////////////////////
// AST
class LanguageResource;
class CursorResource;
class BitmapResource;
class IconResource;
class MenuResource;
class DialogResource;
class StringtableResource;
class AcceleratorsResource;
class RcdataResource;
class VersioninfoResource;
class DlgincludeResource;
class HtmlResource;
class UserDefinedResource;
class Visitor {
public:
virtual bool VisitLanguageResource(const LanguageResource* r) = 0;
virtual bool VisitCursorResource(const CursorResource* r) = 0;
virtual bool VisitBitmapResource(const BitmapResource* r) = 0;
virtual bool VisitIconResource(const IconResource* r) = 0;
virtual bool VisitMenuResource(const MenuResource* r) = 0;
virtual bool VisitDialogResource(const DialogResource* r) = 0;
virtual bool VisitStringtableResource(const StringtableResource* r) = 0;
virtual bool VisitAcceleratorsResource(const AcceleratorsResource* r) = 0;
virtual bool VisitRcdataResource(const RcdataResource* r) = 0;
virtual bool VisitVersioninfoResource(const VersioninfoResource* r) = 0;
virtual bool VisitDlgincludeResource(const DlgincludeResource* r) = 0;
virtual bool VisitHtmlResource(const HtmlResource* r) = 0;
virtual bool VisitUserDefinedResource(const UserDefinedResource* r) = 0;
protected:
~Visitor() {}
};
// Every resource's type or name can be a uint16_t or an utf-16 string.
// The uint16_t looks like (0xffff value) when serialized, the utf-16 string
// is just a \0-terminated utf-16le string. This class represents that concept.
// (Also used in a few other places, e.g. DIALOG's CLASS, MENU, etc).
class IntOrStringName {
public:
static IntOrStringName MakeInt(uint16_t val) {
IntOrStringName r{0xffff, val};
return r;
}
// This takes an UTF16-encoded string.
static IntOrStringName MakeStringUTF16(const C16string& val) {
IntOrStringName r;
for (size_t j = 0; j < val.size(); ++j)