Skip to content

Commit 02d716f

Browse files
haxarshad-yaseen
andauthored
perf(lexer): vectorize comment and string scanning (#196)
Co-authored-by: arshad <arshadpyaseen@gmail.com>
1 parent 07d295f commit 02d716f

2 files changed

Lines changed: 140 additions & 54 deletions

File tree

‎src/parser/lexer.zig‎

Lines changed: 137 additions & 54 deletions
Original file line numberDiff line numberDiff line change
@@ -46,6 +46,39 @@ pub const LexerMode = enum {
4646
jsx_tag,
4747
};
4848

49+
/// First offset >= `from` at which `src` holds one of the comptime
50+
/// `chars`, searched 16 bytes at a time, `src.len` on a miss.
51+
fn findAnyPos(comptime chars: []const u8, src: []const u8, from: u32) u32 {
52+
const Vec = @Vector(16, u8);
53+
54+
var i: usize = from;
55+
while (i + 16 <= src.len) : (i += 16) {
56+
const v: Vec = src[i..][0..16].*;
57+
var hit: @Vector(16, bool) = @splat(false);
58+
inline for (chars) |ch| {
59+
hit = hit | (v == @as(Vec, @splat(ch)));
60+
}
61+
const mask: u16 = @bitCast(hit);
62+
if (mask != 0) return @intCast(i + @ctz(mask));
63+
}
64+
if (i + 8 <= src.len) {
65+
const v: @Vector(8, u8) = src[i..][0..8].*;
66+
var hit: @Vector(8, bool) = @splat(false);
67+
inline for (chars) |ch| {
68+
hit = hit | (v == @as(@Vector(8, u8), @splat(ch)));
69+
}
70+
const mask: u8 = @bitCast(hit);
71+
if (mask != 0) return @intCast(i + @ctz(mask));
72+
i += 8;
73+
}
74+
while (i < src.len) : (i += 1) {
75+
inline for (chars) |ch| {
76+
if (src[i] == ch) return @intCast(i);
77+
}
78+
}
79+
return @intCast(src.len);
80+
}
81+
4982
pub const LexerState = struct {
5083
/// Flags attached to the next emitted token.
5184
token_flags: u8 = 0,
@@ -391,23 +424,35 @@ pub const Lexer = struct {
391424

392425
self.cursor += 1;
393426

394-
while (self.cursor < self.source.len) {
395-
const c = self.source[self.cursor];
396-
if (c == '\\') {
397-
try self.consumeEscape(.template);
398-
continue;
399-
}
400-
if (c == '`') {
401-
self.cursor += 1;
402-
return self.createToken(.template_tail, start, self.cursor);
403-
}
404-
if (c == '$' and self.peek(1) == '{') {
405-
self.cursor += 2;
406-
return self.createToken(.template_middle, start, self.cursor);
427+
const src = self.source;
428+
var pos = self.cursor;
429+
while (true) {
430+
pos = findAnyPos("`\\$\r", src, pos);
431+
if (pos >= src.len) break;
432+
switch (src[pos]) {
433+
'`' => {
434+
self.cursor = pos + 1;
435+
return self.createToken(.template_tail, start, self.cursor);
436+
},
437+
'\\' => {
438+
self.cursor = pos;
439+
try self.consumeEscape(.template);
440+
pos = self.cursor;
441+
},
442+
'$' => {
443+
if (pos + 1 < src.len and src[pos + 1] == '{') {
444+
self.cursor = pos + 2;
445+
return self.createToken(.template_middle, start, self.cursor);
446+
}
447+
pos += 1;
448+
},
449+
// a raw CR must be normalized in the cooked value, so it counts as escaped
450+
'\r' => {
451+
self.setTokenFlag(.escaped);
452+
pos += 1;
453+
},
454+
else => unreachable,
407455
}
408-
// a raw CR must be normalized in the cooked value, so it counts as escaped
409-
if (c == '\r') self.setTokenFlag(.escaped);
410-
self.cursor += 1;
411456
}
412457
return error.NonTerminatedTemplateLiteral;
413458
}
@@ -569,27 +614,28 @@ pub const Lexer = struct {
569614

570615
if (self.mode == .normal) {
571616
while (pos < src.len) {
572-
const c = src[pos];
573-
574-
if (c == quote) {
575-
pos += 1;
617+
const hit = if (quote == '"')
618+
findAnyPos("\"\\\n\r", src, pos)
619+
else
620+
findAnyPos("'\\\n\r", src, pos);
621+
if (hit >= src.len) {
622+
pos = hit;
623+
break;
624+
}
625+
if (src[hit] == quote) {
626+
pos = hit + 1;
576627
self.cursor = pos;
577628
return self.createToken(.string_literal, start, pos);
578629
}
579-
580-
if (c == '\\') {
581-
self.cursor = pos;
630+
if (src[hit] == '\\') {
631+
self.cursor = hit;
582632
try self.consumeEscape(.string);
583633
pos = self.cursor;
584634
continue;
585635
}
586-
587-
if (c == '\n' or c == '\r') {
588-
self.cursor = pos;
589-
return error.UnterminatedString;
590-
}
591-
592-
pos += 1;
636+
// '\n' or '\r'
637+
self.cursor = hit;
638+
return error.UnterminatedString;
593639
}
594640
} else {
595641
// jsx attribute values have no escapes and may span lines
@@ -612,32 +658,39 @@ pub const Lexer = struct {
612658
std.debug.assert(self.source[self.cursor] == '`');
613659

614660
const start = self.cursor;
615-
self.cursor += 1;
616-
617-
while (self.cursor < self.source.len) {
618-
const c = self.source[self.cursor];
619-
620-
if (c == '\\') {
621-
try self.consumeEscape(.template);
622-
continue;
623-
}
624-
625-
if (c == '`') {
626-
self.cursor += 1;
627-
return self.createToken(.no_substitution_template, start, self.cursor);
628-
}
661+
const src = self.source;
629662

630-
if (c == '$' and self.peek(1) == '{') {
631-
self.cursor += 2;
632-
return self.createToken(.template_head, start, self.cursor);
663+
var pos = start + 1;
664+
while (true) {
665+
pos = findAnyPos("`\\$\r", src, pos);
666+
if (pos >= src.len) break;
667+
switch (src[pos]) {
668+
'`' => {
669+
self.cursor = pos + 1;
670+
return self.createToken(.no_substitution_template, start, self.cursor);
671+
},
672+
'\\' => {
673+
self.cursor = pos;
674+
try self.consumeEscape(.template);
675+
pos = self.cursor;
676+
},
677+
'$' => {
678+
if (pos + 1 < src.len and src[pos + 1] == '{') {
679+
self.cursor = pos + 2;
680+
return self.createToken(.template_head, start, self.cursor);
681+
}
682+
pos += 1;
683+
},
684+
// a raw CR must be normalized in the cooked value, so it counts as escaped
685+
'\r' => {
686+
self.setTokenFlag(.escaped);
687+
pos += 1;
688+
},
689+
else => unreachable,
633690
}
634-
635-
// a raw CR must be normalized in the cooked value, so it counts as escaped
636-
if (c == '\r') self.setTokenFlag(.escaped);
637-
638-
self.cursor += 1;
639691
}
640692

693+
self.cursor = @intCast(src.len);
641694
return error.NonTerminatedTemplateLiteral;
642695
}
643696

@@ -1359,7 +1412,9 @@ pub const Lexer = struct {
13591412
const start = self.cursor;
13601413
const src = self.source;
13611414
var pos = start + 2;
1362-
while (pos < src.len) {
1415+
while (true) {
1416+
pos = findAnyPos("\r\n\xe2", src, pos);
1417+
if (pos >= src.len) break;
13631418
const c = src[pos];
13641419
if (c == '\n' or c == '\r') break;
13651420
if (c == 0xE2 and util.Utf.unicodeSeparatorLen(src, pos) > 0) break;
@@ -1391,18 +1446,46 @@ pub const Lexer = struct {
13911446
'\n', '\r' => {
13921447
self.setTokenFlag(.line_terminator_before);
13931448
pos += 1;
1449+
break;
13941450
},
13951451
0x80...0xFF => {
13961452
const lt_len = util.Utf.unicodeSeparatorLen(src, pos);
13971453
if (lt_len > 0) {
13981454
self.setTokenFlag(.line_terminator_before);
13991455
pos += lt_len;
1456+
break;
14001457
} else pos += 1;
14011458
},
14021459
else => pos += 1,
14031460
}
14041461
}
1405-
self.cursor = pos;
1462+
// multi-line body: vectorized search for the two-byte '*/'
1463+
// sequence (star and slash masks combined per lane). line leads
1464+
// are " * ", star without a slash, so they never restart the
1465+
// scan. windows overlap by one byte to catch a straddling '*/'.
1466+
var w = pos - 1;
1467+
while (w + 16 <= src.len) {
1468+
const v: @Vector(16, u8) = src[w..][0..16].*;
1469+
var stars: @Vector(16, bool) = @splat(false);
1470+
var slashes: @Vector(16, bool) = @splat(false);
1471+
stars = stars | (v == @as(@Vector(16, u8), @splat('*')));
1472+
slashes = slashes | (v == @as(@Vector(16, u8), @splat('/')));
1473+
const ends: u16 = @as(u16, @bitCast(stars)) & (@as(u16, @bitCast(slashes)) >> 1);
1474+
if (ends != 0) {
1475+
self.cursor = @intCast(w + @ctz(ends) + 2);
1476+
try self.recordComment(.block, start, self.cursor);
1477+
return;
1478+
}
1479+
w += 15;
1480+
}
1481+
while (w + 1 < src.len) : (w += 1) {
1482+
if (src[w] == '*' and src[w + 1] == '/') {
1483+
self.cursor = @intCast(w + 2);
1484+
try self.recordComment(.block, start, self.cursor);
1485+
return;
1486+
}
1487+
}
1488+
self.cursor = @intCast(src.len);
14061489
return error.UnterminatedMultiLineComment;
14071490
}
14081491

‎src/parser/parser.zig‎

Lines changed: 3 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -432,6 +432,9 @@ pub const Parser = struct {
432432
comments_len: usize,
433433

434434
pub inline fn next(self: *Peek) Token {
435+
// the inline fast path covers idents and simple punctuation,
436+
// which is most of what multi-token lookahead fetches
437+
if (self.parser.lexer.tryNextToken()) |token| return token;
435438
return self.parser.lexer.nextToken() catch
436439
Token.invalid(self.parser.lexer.cursor);
437440
}

0 commit comments

Comments
 (0)