tree-sitter-sh 0.3.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/scanner.c CHANGED
@@ -276,15 +276,18 @@ clear_document_array(struct HereDocument **documents, size_t *count) {
276
276
  *count = 0;
277
277
  }
278
278
 
279
- static void
280
- discard_uncommitted_here_document_delimiter(struct Scanner *scanner) {
281
- clear_document_array(&scanner->captured_documents, &scanner->captured_count);
279
+ // A here-document operator whose delimiter word has not begun cannot span a
280
+ // newline, so the pre-scan flags reset there. Captured delimiters stay until
281
+ // their commit: the word after HERE_END_BEGIN can itself contain newline
282
+ // tokens, inside double-quotes or inside a nested here-document's body.
283
+ static void reset_here_document_delimiter_scan(struct Scanner *scanner) {
282
284
  scanner->expecting_delimiter = false;
283
285
  scanner->delimiter_strips_tabs = false;
284
286
  }
285
287
 
286
288
  static void clear_scanner(struct Scanner *scanner) {
287
- discard_uncommitted_here_document_delimiter(scanner);
289
+ reset_here_document_delimiter_scan(scanner);
290
+ clear_document_array(&scanner->captured_documents, &scanner->captured_count);
288
291
  clear_document_array(&scanner->pending_documents, &scanner->pending_count);
289
292
  clear_document_array(&scanner->active_documents, &scanner->active_count);
290
293
  for (size_t index = 0; index < scanner->suspended_frame_count; index += 1) {
@@ -748,6 +751,14 @@ static bool is_special_parameter_character(int32_t character) {
748
751
  );
749
752
  }
750
753
 
754
+ static bool is_parameter_start_character(int32_t character) {
755
+ return (
756
+ is_name_start_character(character) ||
757
+ is_decimal_digit(character) ||
758
+ is_special_parameter_character(character)
759
+ );
760
+ }
761
+
751
762
  static bool is_lowercase_letter(int32_t character) {
752
763
  return character >= 'a' && character <= 'z';
753
764
  }
@@ -1067,24 +1078,148 @@ struct DelimiterGroupBuffer {
1067
1078
  size_t capacity;
1068
1079
  };
1069
1080
 
1070
- enum DelimiterCaseState {
1071
- DELIMITER_CASE_EXPECT_WORD,
1072
- DELIMITER_CASE_EXPECT_IN,
1073
- DELIMITER_CASE_EXPECT_PATTERN,
1074
- DELIMITER_CASE_BODY,
1081
+ // Case tracking is shared between the here-document delimiter scan and the
1082
+ // embedded construct skip: both must know where an unquoted esac or a
1083
+ // pattern parenthesis can end a case command nested in substitution source.
1084
+ // EXPECT_PATTERN is the first token position of a pattern, where an
1085
+ // unquoted esac terminates the case; IN_PATTERN covers the rest of the
1086
+ // pattern list, where esac is an ordinary word.
1087
+ enum CaseTrackerState {
1088
+ CASE_TRACKER_EXPECT_WORD,
1089
+ CASE_TRACKER_EXPECT_IN,
1090
+ CASE_TRACKER_EXPECT_PATTERN,
1091
+ CASE_TRACKER_IN_PATTERN,
1092
+ CASE_TRACKER_BODY,
1075
1093
  };
1076
1094
 
1077
- struct DelimiterCaseFrame {
1078
- size_t group_depth;
1079
- enum DelimiterCaseState state;
1095
+ struct CaseTracker {
1096
+ size_t depth;
1097
+ uint8_t state;
1080
1098
  };
1081
1099
 
1082
- struct DelimiterCaseBuffer {
1083
- struct DelimiterCaseFrame *data;
1100
+ static bool case_tracker_in_pattern(uint8_t state) {
1101
+ return (
1102
+ state == CASE_TRACKER_EXPECT_PATTERN || state == CASE_TRACKER_IN_PATTERN
1103
+ );
1104
+ }
1105
+
1106
+ struct CaseTrackerBuffer {
1107
+ struct CaseTracker *data;
1084
1108
  size_t length;
1085
1109
  size_t capacity;
1086
1110
  };
1087
1111
 
1112
+ static struct CaseTracker *
1113
+ active_case_tracker(const struct CaseTrackerBuffer *cases, size_t depth) {
1114
+ if (cases->length == 0) {
1115
+ return NULL;
1116
+ }
1117
+ struct CaseTracker *tracker = &cases->data[cases->length - 1];
1118
+ return tracker->depth == depth ? tracker : NULL;
1119
+ }
1120
+
1121
+ static void
1122
+ pop_case_trackers_at_depth(struct CaseTrackerBuffer *cases, size_t depth) {
1123
+ while (cases->length > 0 && cases->data[cases->length - 1].depth >= depth) {
1124
+ cases->length -= 1;
1125
+ }
1126
+ }
1127
+
1128
+ enum CaseWordKind {
1129
+ CASE_WORD_GENERIC,
1130
+ CASE_WORD_IN,
1131
+ CASE_WORD_ESAC,
1132
+ CASE_WORD_CASE,
1133
+ // Reserved prefixes whose following word is still at command start.
1134
+ CASE_WORD_COMMAND_PREFIX,
1135
+ };
1136
+
1137
+ static enum CaseWordKind classify_case_word(const char *word) {
1138
+ if (strcmp(word, "in") == 0) {
1139
+ return CASE_WORD_IN;
1140
+ }
1141
+ if (strcmp(word, "esac") == 0) {
1142
+ return CASE_WORD_ESAC;
1143
+ }
1144
+ if (strcmp(word, "case") == 0) {
1145
+ return CASE_WORD_CASE;
1146
+ }
1147
+
1148
+ static const char *const COMMAND_PREFIXES[] = {
1149
+ "!",
1150
+ "{",
1151
+ "if",
1152
+ "then",
1153
+ "elif",
1154
+ "else",
1155
+ "while",
1156
+ "until",
1157
+ "do",
1158
+ };
1159
+ for (
1160
+ size_t index = 0;
1161
+ index < sizeof(COMMAND_PREFIXES) / sizeof(COMMAND_PREFIXES[0]);
1162
+ index += 1
1163
+ ) {
1164
+ if (strcmp(word, COMMAND_PREFIXES[index]) == 0) {
1165
+ return CASE_WORD_COMMAND_PREFIX;
1166
+ }
1167
+ }
1168
+ return CASE_WORD_GENERIC;
1169
+ }
1170
+
1171
+ enum CaseTrackerNote {
1172
+ CASE_TRACKER_NOTE_WORD,
1173
+ CASE_TRACKER_NOTE_COMMAND_PREFIX,
1174
+ CASE_TRACKER_NOTE_END,
1175
+ CASE_TRACKER_NOTE_BEGIN,
1176
+ };
1177
+
1178
+ // Advances the tracked case across one completed command word and reports
1179
+ // how the caller must react: END pops the active tracker, BEGIN pushes a
1180
+ // nested one, and COMMAND_PREFIX leaves the command-start position open.
1181
+ static enum CaseTrackerNote case_tracker_note_word(
1182
+ struct CaseTracker *tracker,
1183
+ enum CaseWordKind kind,
1184
+ bool at_command_start
1185
+ ) {
1186
+ if (tracker != NULL && tracker->state != CASE_TRACKER_BODY) {
1187
+ switch (tracker->state) {
1188
+ case CASE_TRACKER_EXPECT_WORD:
1189
+ tracker->state = CASE_TRACKER_EXPECT_IN;
1190
+ break;
1191
+ case CASE_TRACKER_EXPECT_IN:
1192
+ if (kind == CASE_WORD_IN) {
1193
+ tracker->state = CASE_TRACKER_EXPECT_PATTERN;
1194
+ }
1195
+ break;
1196
+ case CASE_TRACKER_EXPECT_PATTERN:
1197
+ if (kind == CASE_WORD_ESAC) {
1198
+ return CASE_TRACKER_NOTE_END;
1199
+ }
1200
+ tracker->state = CASE_TRACKER_IN_PATTERN;
1201
+ break;
1202
+ default:
1203
+ break;
1204
+ }
1205
+ return CASE_TRACKER_NOTE_WORD;
1206
+ }
1207
+
1208
+ if (!at_command_start) {
1209
+ return CASE_TRACKER_NOTE_WORD;
1210
+ }
1211
+ if (kind == CASE_WORD_ESAC && tracker != NULL) {
1212
+ return CASE_TRACKER_NOTE_END;
1213
+ }
1214
+ if (kind == CASE_WORD_COMMAND_PREFIX) {
1215
+ return CASE_TRACKER_NOTE_COMMAND_PREFIX;
1216
+ }
1217
+ if (kind == CASE_WORD_CASE) {
1218
+ return CASE_TRACKER_NOTE_BEGIN;
1219
+ }
1220
+ return CASE_TRACKER_NOTE_WORD;
1221
+ }
1222
+
1088
1223
  static bool grow_element_buffer(
1089
1224
  void **data,
1090
1225
  size_t *capacity,
@@ -1110,26 +1245,35 @@ static bool grow_element_buffer(
1110
1245
  return true;
1111
1246
  }
1112
1247
 
1113
- static bool
1114
- append_delimiter_case(struct DelimiterCaseBuffer *cases, size_t group_depth) {
1248
+ static bool append_case_tracker(struct CaseTrackerBuffer *cases, size_t depth) {
1115
1249
  if (!grow_element_buffer(
1116
1250
  (void **)&cases->data,
1117
1251
  &cases->capacity,
1118
1252
  cases->length,
1119
- sizeof(struct DelimiterCaseFrame),
1253
+ sizeof(struct CaseTracker),
1120
1254
  8
1121
1255
  )) {
1122
1256
  return false;
1123
1257
  }
1124
1258
 
1125
- cases->data[cases->length] = (struct DelimiterCaseFrame){
1126
- .group_depth = group_depth,
1127
- .state = DELIMITER_CASE_EXPECT_WORD,
1259
+ cases->data[cases->length] = (struct CaseTracker){
1260
+ .depth = depth,
1261
+ .state = CASE_TRACKER_EXPECT_WORD,
1128
1262
  };
1129
1263
  cases->length += 1;
1130
1264
  return true;
1131
1265
  }
1132
1266
 
1267
+ static bool delimiter_group_holds_commands(enum DelimiterGroupKind kind) {
1268
+ return (
1269
+ kind ==
1270
+ DELIMITER_GROUP_COMMAND ||
1271
+ kind ==
1272
+ DELIMITER_GROUP_SUBSHELL ||
1273
+ kind == DELIMITER_GROUP_BACKQUOTE
1274
+ );
1275
+ }
1276
+
1133
1277
  static bool push_delimiter_group(
1134
1278
  struct DelimiterGroupBuffer *groups,
1135
1279
  int32_t closing,
@@ -1155,11 +1299,7 @@ static bool push_delimiter_group(
1155
1299
  .kind = kind,
1156
1300
  .parent_quote = parent_quote,
1157
1301
  .word = {.candidate = true},
1158
- .command_start = kind ==
1159
- DELIMITER_GROUP_COMMAND ||
1160
- kind ==
1161
- DELIMITER_GROUP_SUBSHELL ||
1162
- kind == DELIMITER_GROUP_BACKQUOTE,
1302
+ .command_start = delimiter_group_holds_commands(kind),
1163
1303
  };
1164
1304
  groups->length += 1;
1165
1305
  return true;
@@ -1180,19 +1320,6 @@ static bool is_delimiter_word_character(int32_t character) {
1180
1320
  );
1181
1321
  }
1182
1322
 
1183
- static bool delimiter_word_equals(
1184
- const struct DelimiterWordTracker *word,
1185
- const char *text
1186
- ) {
1187
- size_t length = strlen(text);
1188
- return (
1189
- word->candidate &&
1190
- word->length ==
1191
- length &&
1192
- memcmp(word->text, text, length) == 0
1193
- );
1194
- }
1195
-
1196
1323
  static void reset_delimiter_word(struct DelimiterWordTracker *word) {
1197
1324
  word->length = 0;
1198
1325
  word->active = false;
@@ -1200,7 +1327,7 @@ static void reset_delimiter_word(struct DelimiterWordTracker *word) {
1200
1327
  }
1201
1328
 
1202
1329
  static bool finish_delimiter_word(
1203
- struct DelimiterCaseBuffer *cases,
1330
+ struct CaseTrackerBuffer *cases,
1204
1331
  struct DelimiterGroupBuffer *groups,
1205
1332
  size_t group_depth
1206
1333
  ) {
@@ -1211,60 +1338,28 @@ static bool finish_delimiter_word(
1211
1338
  return true;
1212
1339
  }
1213
1340
 
1214
- bool at_command_start = group->command_start;
1215
- struct DelimiterCaseFrame *active_case =
1216
- cases->length == 0 ? NULL : &cases->data[cases->length - 1];
1217
- if (active_case != NULL && active_case->group_depth == group_depth) {
1218
- if (active_case->state == DELIMITER_CASE_EXPECT_WORD) {
1219
- active_case->state = DELIMITER_CASE_EXPECT_IN;
1220
- group->command_start = false;
1221
- reset_delimiter_word(word);
1222
- return true;
1223
- }
1224
-
1225
- if (active_case->state == DELIMITER_CASE_EXPECT_IN) {
1226
- if (delimiter_word_equals(word, "in")) {
1227
- active_case->state = DELIMITER_CASE_EXPECT_PATTERN;
1228
- }
1229
- group->command_start = false;
1230
- reset_delimiter_word(word);
1231
- return true;
1232
- }
1233
-
1234
- if (
1235
- (active_case->state ==
1236
- DELIMITER_CASE_EXPECT_PATTERN ||
1237
- (active_case->state == DELIMITER_CASE_BODY && at_command_start)) &&
1238
- delimiter_word_equals(word, "esac")
1239
- ) {
1240
- cases->length -= 1;
1241
- group->command_start = false;
1242
- reset_delimiter_word(word);
1243
- return true;
1244
- }
1341
+ char text[sizeof(word->text) + 1] = {0};
1342
+ if (word->candidate) {
1343
+ memcpy(text, word->text, word->length);
1245
1344
  }
1246
-
1247
- if (
1248
- at_command_start &&
1249
- (delimiter_word_equals(word, "!") ||
1250
- delimiter_word_equals(word, "{") ||
1251
- delimiter_word_equals(word, "if") ||
1252
- delimiter_word_equals(word, "then") ||
1253
- delimiter_word_equals(word, "elif") ||
1254
- delimiter_word_equals(word, "else") ||
1255
- delimiter_word_equals(word, "while") ||
1256
- delimiter_word_equals(word, "until") ||
1257
- delimiter_word_equals(word, "do"))
1258
- ) {
1259
- group->command_start = true;
1345
+ switch (case_tracker_note_word(
1346
+ active_case_tracker(cases, group_depth),
1347
+ word->candidate ? classify_case_word(text) : CASE_WORD_GENERIC,
1348
+ group->command_start
1349
+ )) {
1350
+ case CASE_TRACKER_NOTE_COMMAND_PREFIX:
1260
1351
  reset_delimiter_word(word);
1261
1352
  return true;
1262
- }
1263
-
1264
- if (at_command_start && delimiter_word_equals(word, "case")) {
1265
- if (!append_delimiter_case(cases, group_depth)) {
1353
+ case CASE_TRACKER_NOTE_END:
1354
+ cases->length -= 1;
1355
+ break;
1356
+ case CASE_TRACKER_NOTE_BEGIN:
1357
+ if (!append_case_tracker(cases, group_depth)) {
1266
1358
  return false;
1267
1359
  }
1360
+ break;
1361
+ default:
1362
+ break;
1268
1363
  }
1269
1364
  group->command_start = false;
1270
1365
  reset_delimiter_word(word);
@@ -1273,18 +1368,14 @@ static bool finish_delimiter_word(
1273
1368
 
1274
1369
  static size_t
1275
1370
  delimiter_command_group_depth(const struct DelimiterGroupBuffer *groups) {
1276
- if (groups->length == 0) {
1371
+ if (
1372
+ groups->length ==
1373
+ 0 ||
1374
+ !delimiter_group_holds_commands(groups->data[groups->length - 1].kind)
1375
+ ) {
1277
1376
  return 0;
1278
1377
  }
1279
-
1280
- enum DelimiterGroupKind kind = groups->data[groups->length - 1].kind;
1281
- return (kind ==
1282
- DELIMITER_GROUP_COMMAND ||
1283
- kind ==
1284
- DELIMITER_GROUP_SUBSHELL ||
1285
- kind == DELIMITER_GROUP_BACKQUOTE)
1286
- ? groups->length
1287
- : 0;
1378
+ return groups->length;
1288
1379
  }
1289
1380
 
1290
1381
  static bool
@@ -1308,18 +1399,12 @@ append_repeated_byte(struct ByteBuffer *buffer, uint8_t byte, size_t count) {
1308
1399
 
1309
1400
  static void pop_delimiter_group(
1310
1401
  struct DelimiterGroupBuffer *groups,
1311
- struct DelimiterCaseBuffer *cases,
1402
+ struct CaseTrackerBuffer *cases,
1312
1403
  enum DelimiterQuote *quote
1313
1404
  ) {
1314
1405
  size_t group_depth = groups->length;
1315
1406
  *quote = groups->data[group_depth - 1].parent_quote;
1316
- while (
1317
- cases->length >
1318
- 0 &&
1319
- cases->data[cases->length - 1].group_depth >= group_depth
1320
- ) {
1321
- cases->length -= 1;
1322
- }
1407
+ pop_case_trackers_at_depth(cases, group_depth);
1323
1408
  groups->length -= 1;
1324
1409
  }
1325
1410
 
@@ -1357,7 +1442,7 @@ static enum DelimiterBackslashResult scan_delimiter_backslash_run(
1357
1442
  TSLexer *lexer,
1358
1443
  struct ByteBuffer *delimiter,
1359
1444
  struct DelimiterGroupBuffer *groups,
1360
- struct DelimiterCaseBuffer *cases,
1445
+ struct CaseTrackerBuffer *cases,
1361
1446
  enum DelimiterQuote *quote,
1362
1447
  size_t *backquote_depth,
1363
1448
  bool at_delimiter_source_start,
@@ -1925,6 +2010,124 @@ static bool push_dollar_delimiter_group(
1925
2010
  return true;
1926
2011
  }
1927
2012
 
2013
+ // Consumes single-quoted delimiter source through the closing quote; the
2014
+ // quoting never nests, so the segment reads to completion or input end.
2015
+ static bool scan_delimiter_single_quoted_segment(
2016
+ TSLexer *lexer,
2017
+ struct ByteBuffer *delimiter,
2018
+ bool dollar
2019
+ ) {
2020
+ while (!lexer_at_eof(lexer)) {
2021
+ int32_t character = lexer->lookahead;
2022
+ if (character == '\'') {
2023
+ lexer->advance(lexer, false);
2024
+ return true;
2025
+ }
2026
+ if (dollar && character == '\\') {
2027
+ lexer->advance(lexer, false);
2028
+ if (!scan_dollar_single_quote_escape(lexer, delimiter)) {
2029
+ return false;
2030
+ }
2031
+ continue;
2032
+ }
2033
+ if (!append_codepoint(delimiter, character)) {
2034
+ return false;
2035
+ }
2036
+ lexer->advance(lexer, false);
2037
+ }
2038
+ return false;
2039
+ }
2040
+
2041
+ // Handles one double-quoted delimiter character; a substitution start hands
2042
+ // control back to the group machinery with *quote reset for its interior.
2043
+ static bool scan_delimiter_double_quoted_character(
2044
+ TSLexer *lexer,
2045
+ struct ByteBuffer *delimiter,
2046
+ struct DelimiterGroupBuffer *groups,
2047
+ enum DelimiterQuote *quote,
2048
+ size_t *backquote_depth
2049
+ ) {
2050
+ int32_t character = lexer->lookahead;
2051
+ if (lexer_at_eof(lexer)) {
2052
+ return false;
2053
+ }
2054
+
2055
+ if (character == '"') {
2056
+ *quote = DELIMITER_UNQUOTED;
2057
+ lexer->advance(lexer, false);
2058
+ return true;
2059
+ }
2060
+
2061
+ if (character == '$') {
2062
+ lexer->advance(lexer, false);
2063
+ if (!append_byte(delimiter, '$')) {
2064
+ return false;
2065
+ }
2066
+ if (lexer->lookahead == '(' || lexer->lookahead == '{') {
2067
+ if (!push_dollar_delimiter_group(
2068
+ lexer,
2069
+ delimiter,
2070
+ groups,
2071
+ DELIMITER_DOUBLE_QUOTED
2072
+ )) {
2073
+ return false;
2074
+ }
2075
+ *quote = DELIMITER_UNQUOTED;
2076
+ }
2077
+ return true;
2078
+ }
2079
+
2080
+ if (character == '`') {
2081
+ if (
2082
+ *backquote_depth ==
2083
+ SIZE_MAX ||
2084
+ !append_byte(delimiter, '`') ||
2085
+ !push_delimiter_group(
2086
+ groups,
2087
+ '`',
2088
+ DELIMITER_GROUP_BACKQUOTE,
2089
+ DELIMITER_DOUBLE_QUOTED
2090
+ )
2091
+ ) {
2092
+ return false;
2093
+ }
2094
+ *backquote_depth += 1;
2095
+ *quote = DELIMITER_UNQUOTED;
2096
+ lexer->advance(lexer, false);
2097
+ return true;
2098
+ }
2099
+
2100
+ if (character == '\\') {
2101
+ lexer->advance(lexer, false);
2102
+ if (lexer->lookahead == '\n') {
2103
+ lexer->advance(lexer, false);
2104
+ return true;
2105
+ }
2106
+ if (
2107
+ lexer->lookahead ==
2108
+ '$' ||
2109
+ lexer->lookahead ==
2110
+ '`' ||
2111
+ lexer->lookahead ==
2112
+ '"' ||
2113
+ lexer->lookahead == '\\'
2114
+ ) {
2115
+ if (!append_codepoint(delimiter, lexer->lookahead)) {
2116
+ return false;
2117
+ }
2118
+ lexer->advance(lexer, false);
2119
+ return true;
2120
+ }
2121
+ return append_byte(delimiter, '\\');
2122
+ }
2123
+
2124
+ if (!append_codepoint(delimiter, character)) {
2125
+ return false;
2126
+ }
2127
+ lexer->advance(lexer, false);
2128
+ return true;
2129
+ }
2130
+
1928
2131
  static bool scan_here_document_delimiter(
1929
2132
  struct Scanner *scanner,
1930
2133
  TSLexer *lexer,
@@ -1934,7 +2137,7 @@ static bool scan_here_document_delimiter(
1934
2137
  enum DelimiterQuote quote = DELIMITER_UNQUOTED;
1935
2138
  struct ByteBuffer delimiter = {.limit = SCANNER_STATE_CAPACITY};
1936
2139
  struct DelimiterGroupBuffer groups = {0};
1937
- struct DelimiterCaseBuffer cases = {0};
2140
+ struct CaseTrackerBuffer cases = {0};
1938
2141
  struct HereDocument *nested_documents = NULL;
1939
2142
  size_t nested_document_count = 0;
1940
2143
  size_t nested_delimiter_start = 0;
@@ -1951,92 +2154,30 @@ static bool scan_here_document_delimiter(
1951
2154
  while (valid) {
1952
2155
  int32_t character = lexer->lookahead;
1953
2156
 
1954
- if (quote == DELIMITER_SINGLE_QUOTED) {
1955
- if (lexer_at_eof(lexer)) {
1956
- valid = false;
1957
- } else if (character == '\'') {
1958
- quote = DELIMITER_UNQUOTED;
1959
- lexer->advance(lexer, false);
1960
- } else {
1961
- valid = append_codepoint(&delimiter, character);
1962
- lexer->advance(lexer, false);
1963
- }
1964
- continue;
1965
- }
1966
-
1967
- if (quote == DELIMITER_DOLLAR_SINGLE_QUOTED) {
1968
- if (lexer_at_eof(lexer)) {
1969
- valid = false;
1970
- } else if (character == '\'') {
2157
+ if (
2158
+ quote ==
2159
+ DELIMITER_SINGLE_QUOTED ||
2160
+ quote == DELIMITER_DOLLAR_SINGLE_QUOTED
2161
+ ) {
2162
+ valid = scan_delimiter_single_quoted_segment(
2163
+ lexer,
2164
+ &delimiter,
2165
+ quote == DELIMITER_DOLLAR_SINGLE_QUOTED
2166
+ );
2167
+ if (valid) {
1971
2168
  quote = DELIMITER_UNQUOTED;
1972
- lexer->advance(lexer, false);
1973
- } else if (character == '\\') {
1974
- lexer->advance(lexer, false);
1975
- valid = scan_dollar_single_quote_escape(lexer, &delimiter);
1976
- } else {
1977
- valid = append_codepoint(&delimiter, character);
1978
- lexer->advance(lexer, false);
1979
2169
  }
1980
2170
  continue;
1981
2171
  }
1982
2172
 
1983
2173
  if (quote == DELIMITER_DOUBLE_QUOTED) {
1984
- if (lexer_at_eof(lexer)) {
1985
- valid = false;
1986
- } else if (character == '"') {
1987
- quote = DELIMITER_UNQUOTED;
1988
- lexer->advance(lexer, false);
1989
- } else if (character == '$') {
1990
- lexer->advance(lexer, false);
1991
- valid = append_byte(&delimiter, '$');
1992
- if (valid && (lexer->lookahead == '(' || lexer->lookahead == '{')) {
1993
- valid = push_dollar_delimiter_group(
1994
- lexer,
1995
- &delimiter,
1996
- &groups,
1997
- DELIMITER_DOUBLE_QUOTED
1998
- );
1999
- if (valid) {
2000
- quote = DELIMITER_UNQUOTED;
2001
- }
2002
- }
2003
- } else if (character == '`') {
2004
- valid = delimiter_backquote_depth <
2005
- SIZE_MAX &&
2006
- append_byte(&delimiter, '`') &&
2007
- push_delimiter_group(
2008
- &groups,
2009
- '`',
2010
- DELIMITER_GROUP_BACKQUOTE,
2011
- DELIMITER_DOUBLE_QUOTED
2012
- );
2013
- if (valid) {
2014
- delimiter_backquote_depth += 1;
2015
- quote = DELIMITER_UNQUOTED;
2016
- lexer->advance(lexer, false);
2017
- }
2018
- } else if (character == '\\') {
2019
- lexer->advance(lexer, false);
2020
- if (lexer->lookahead == '\n') {
2021
- lexer->advance(lexer, false);
2022
- } else if (
2023
- lexer->lookahead ==
2024
- '$' ||
2025
- lexer->lookahead ==
2026
- '`' ||
2027
- lexer->lookahead ==
2028
- '"' ||
2029
- lexer->lookahead == '\\'
2030
- ) {
2031
- valid = append_codepoint(&delimiter, lexer->lookahead);
2032
- lexer->advance(lexer, false);
2033
- } else {
2034
- valid = append_byte(&delimiter, '\\');
2035
- }
2036
- } else {
2037
- valid = append_codepoint(&delimiter, character);
2038
- lexer->advance(lexer, false);
2039
- }
2174
+ valid = scan_delimiter_double_quoted_character(
2175
+ lexer,
2176
+ &delimiter,
2177
+ &groups,
2178
+ &quote,
2179
+ &delimiter_backquote_depth
2180
+ );
2040
2181
  continue;
2041
2182
  }
2042
2183
 
@@ -2085,13 +2226,10 @@ static bool scan_here_document_delimiter(
2085
2226
  }
2086
2227
  }
2087
2228
 
2088
- struct DelimiterCaseFrame *operator_case =
2089
- cases.length == 0 ? NULL : &cases.data[cases.length - 1];
2090
- bool in_case_pattern = operator_case !=
2091
- NULL &&
2092
- operator_case->group_depth ==
2093
- command_group_depth &&
2094
- operator_case->state == DELIMITER_CASE_EXPECT_PATTERN;
2229
+ struct CaseTracker *operator_case =
2230
+ active_case_tracker(&cases, command_group_depth);
2231
+ bool in_case_pattern =
2232
+ operator_case != NULL && case_tracker_in_pattern(operator_case->state);
2095
2233
  if (
2096
2234
  command_group_depth >
2097
2235
  0 &&
@@ -2367,19 +2505,17 @@ static bool scan_here_document_delimiter(
2367
2505
  if (
2368
2506
  groups.length > 0 && character == groups.data[groups.length - 1].closing
2369
2507
  ) {
2370
- struct DelimiterCaseFrame *active_case =
2371
- cases.length == 0 ? NULL : &cases.data[cases.length - 1];
2508
+ struct CaseTracker *active_case =
2509
+ active_case_tracker(&cases, groups.length);
2372
2510
  if (
2373
2511
  character ==
2374
2512
  ')' &&
2375
2513
  active_case !=
2376
2514
  NULL &&
2377
- active_case->group_depth ==
2378
- groups.length &&
2379
- active_case->state == DELIMITER_CASE_EXPECT_PATTERN
2515
+ case_tracker_in_pattern(active_case->state)
2380
2516
  ) {
2381
2517
  valid = append_codepoint(&delimiter, character);
2382
- active_case->state = DELIMITER_CASE_BODY;
2518
+ active_case->state = CASE_TRACKER_BODY;
2383
2519
  groups.data[groups.length - 1].command_start = true;
2384
2520
  lexer->advance(lexer, false);
2385
2521
  continue;
@@ -2398,17 +2534,12 @@ static bool scan_here_document_delimiter(
2398
2534
  ')' &&
2399
2535
  character == '('
2400
2536
  ) {
2401
- struct DelimiterCaseFrame *active_case =
2402
- cases.length == 0 ? NULL : &cases.data[cases.length - 1];
2403
- if (
2404
- active_case !=
2405
- NULL &&
2406
- active_case->group_depth ==
2407
- groups.length &&
2408
- active_case->state == DELIMITER_CASE_EXPECT_PATTERN
2409
- ) {
2537
+ struct CaseTracker *active_case =
2538
+ active_case_tracker(&cases, groups.length);
2539
+ if (active_case != NULL && case_tracker_in_pattern(active_case->state)) {
2410
2540
  has_word_content = true;
2411
2541
  valid = append_codepoint(&delimiter, character);
2542
+ active_case->state = CASE_TRACKER_IN_PATTERN;
2412
2543
  lexer->advance(lexer, false);
2413
2544
  continue;
2414
2545
  }
@@ -2437,20 +2568,18 @@ static bool scan_here_document_delimiter(
2437
2568
 
2438
2569
  size_t active_command_depth = delimiter_command_group_depth(&groups);
2439
2570
  if (active_command_depth > 0) {
2440
- struct DelimiterCaseFrame *active_case =
2441
- cases.length == 0 ? NULL : &cases.data[cases.length - 1];
2571
+ struct CaseTracker *active_case =
2572
+ active_case_tracker(&cases, active_command_depth);
2442
2573
  if (
2443
2574
  character ==
2444
2575
  ';' &&
2445
2576
  active_case !=
2446
2577
  NULL &&
2447
- active_case->group_depth ==
2448
- active_command_depth &&
2449
2578
  active_case->state ==
2450
- DELIMITER_CASE_BODY &&
2579
+ CASE_TRACKER_BODY &&
2451
2580
  (lexer->lookahead == ';' || lexer->lookahead == '&')
2452
2581
  ) {
2453
- active_case->state = DELIMITER_CASE_EXPECT_PATTERN;
2582
+ active_case->state = CASE_TRACKER_EXPECT_PATTERN;
2454
2583
  }
2455
2584
 
2456
2585
  if (
@@ -2462,11 +2591,7 @@ static bool scan_here_document_delimiter(
2462
2591
  '&' ||
2463
2592
  (character ==
2464
2593
  '|' &&
2465
- !(active_case !=
2466
- NULL &&
2467
- active_case->group_depth ==
2468
- active_command_depth &&
2469
- active_case->state == DELIMITER_CASE_EXPECT_PATTERN))
2594
+ !(active_case != NULL && case_tracker_in_pattern(active_case->state)))
2470
2595
  ) {
2471
2596
  groups.data[active_command_depth - 1].command_start = true;
2472
2597
  }
@@ -2540,15 +2665,12 @@ static bool scan_here_end_commit(struct Scanner *scanner, TSLexer *lexer) {
2540
2665
  return true;
2541
2666
  }
2542
2667
 
2543
- static bool read_reserved_character(
2668
+ // The caller has verified that the lookahead is the token's character.
2669
+ static bool scan_delimited_character_token(
2544
2670
  const struct Scanner *scanner,
2545
2671
  TSLexer *lexer,
2546
- int32_t character
2672
+ enum TokenType symbol
2547
2673
  ) {
2548
- if (lexer->lookahead != character) {
2549
- return false;
2550
- }
2551
-
2552
2674
  lexer->advance(lexer, false);
2553
2675
  lexer->mark_end(lexer);
2554
2676
 
@@ -2560,19 +2682,6 @@ static bool read_reserved_character(
2560
2682
  return false;
2561
2683
  }
2562
2684
 
2563
- return true;
2564
- }
2565
-
2566
- static bool scan_reserved_character(
2567
- const struct Scanner *scanner,
2568
- TSLexer *lexer,
2569
- int32_t character,
2570
- enum TokenType symbol
2571
- ) {
2572
- if (!read_reserved_character(scanner, lexer, character)) {
2573
- return false;
2574
- }
2575
-
2576
2685
  lexer->result_symbol = (TSSymbol)symbol;
2577
2686
  return true;
2578
2687
  }
@@ -2644,6 +2753,15 @@ static bool classify_case_item_ns_end(
2644
2753
  return true;
2645
2754
  }
2646
2755
 
2756
+ static bool classify_reserved_word_or_case_end(
2757
+ const char *word,
2758
+ const bool *valid_symbols,
2759
+ TSSymbol *symbol
2760
+ ) {
2761
+ return classify_reserved_word(word, valid_symbols, symbol) ||
2762
+ classify_case_item_ns_end(word, valid_symbols, symbol);
2763
+ }
2764
+
2647
2765
  static bool scan_lowercase_dispatch(
2648
2766
  const struct Scanner *scanner,
2649
2767
  TSLexer *lexer,
@@ -2660,10 +2778,7 @@ static bool scan_lowercase_dispatch(
2660
2778
  }
2661
2779
 
2662
2780
  TSSymbol symbol;
2663
- if (
2664
- classify_reserved_word(word, valid_symbols, &symbol) ||
2665
- classify_case_item_ns_end(word, valid_symbols, &symbol)
2666
- ) {
2781
+ if (classify_reserved_word_or_case_end(word, valid_symbols, &symbol)) {
2667
2782
  lexer->result_symbol = symbol;
2668
2783
  return true;
2669
2784
  }
@@ -2758,20 +2873,21 @@ static bool scan_name_equals_begin_or_reserved_word(
2758
2873
  }
2759
2874
 
2760
2875
  TSSymbol symbol;
2761
- if (classify_reserved_word(word, valid_symbols, &symbol)) {
2762
- lexer->result_symbol = symbol;
2763
- return true;
2764
- }
2765
- if (classify_case_item_ns_end(word, valid_symbols, &symbol)) {
2876
+ if (classify_reserved_word_or_case_end(word, valid_symbols, &symbol)) {
2766
2877
  lexer->result_symbol = symbol;
2767
2878
  return true;
2768
2879
  }
2769
2880
  return false;
2770
2881
  }
2771
2882
 
2772
- static bool scan_case_item_terminator(TSLexer *lexer);
2883
+ static bool scan_case_item_terminator(TSLexer *lexer) {
2884
+ if (lexer->lookahead != ';') {
2885
+ return false;
2886
+ }
2887
+ lexer->advance(lexer, false);
2773
2888
 
2774
- static bool scan_horizontal_layout(TSLexer *lexer);
2889
+ return lexer->lookahead == ';' || lexer->lookahead == '&';
2890
+ }
2775
2891
 
2776
2892
  static bool
2777
2893
  classify_comment_boundary(TSLexer *lexer, const bool *valid_symbols);
@@ -3099,9 +3215,7 @@ static bool scan_function_body_boundary(
3099
3215
 
3100
3216
  bool has_valid_layout = true;
3101
3217
  while (true) {
3102
- while (is_horizontal_blank(lexer->lookahead)) {
3103
- lexer->advance(lexer, false);
3104
- }
3218
+ scan_horizontal_blanks(lexer);
3105
3219
 
3106
3220
  if (lexer->lookahead == '\\') {
3107
3221
  if (!skip_line_continuations(lexer)) {
@@ -3405,8 +3519,7 @@ static bool scan_element_boundary_core(
3405
3519
  }
3406
3520
 
3407
3521
  if (character == ';') {
3408
- lexer->advance(lexer, false);
3409
- if (lexer->lookahead == ';' || lexer->lookahead == '&') {
3522
+ if (scan_case_item_terminator(lexer)) {
3410
3523
  if (valid_symbols[CASE_ITEM_END]) {
3411
3524
  lexer->result_symbol = CASE_ITEM_END;
3412
3525
  return true;
@@ -3809,27 +3922,6 @@ static void record_comment_line_lookahead(
3809
3922
  );
3810
3923
  }
3811
3924
 
3812
- static bool
3813
- scan_pipeline_negation(const struct Scanner *scanner, TSLexer *lexer) {
3814
- if (lexer->lookahead != '!') {
3815
- return false;
3816
- }
3817
-
3818
- lexer->advance(lexer, false);
3819
- lexer->mark_end(lexer);
3820
-
3821
- if (!skip_line_continuations(lexer)) {
3822
- return false;
3823
- }
3824
-
3825
- if (!is_token_delimiter(scanner, lexer)) {
3826
- return false;
3827
- }
3828
-
3829
- lexer->result_symbol = PIPELINE_NEGATION;
3830
- return true;
3831
- }
3832
-
3833
3925
  static bool scan_file_descriptor(TSLexer *lexer) {
3834
3926
  if (!is_decimal_digit(lexer->lookahead)) {
3835
3927
  return false;
@@ -3906,15 +3998,6 @@ static bool scan_here_document_operator_commit(
3906
3998
  return true;
3907
3999
  }
3908
4000
 
3909
- static bool scan_case_item_terminator(TSLexer *lexer) {
3910
- if (lexer->lookahead != ';') {
3911
- return false;
3912
- }
3913
- lexer->advance(lexer, false);
3914
-
3915
- return lexer->lookahead == ';' || lexer->lookahead == '&';
3916
- }
3917
-
3918
4001
  static bool classify_shell_boundary(
3919
4002
  const struct Scanner *scanner,
3920
4003
  TSLexer *lexer,
@@ -4034,7 +4117,7 @@ scan_here_document_line_end(struct Scanner *scanner, TSLexer *lexer) {
4034
4117
  if (!activate_startable_pending_documents(scanner)) {
4035
4118
  return false;
4036
4119
  }
4037
- discard_uncommitted_here_document_delimiter(scanner);
4120
+ reset_here_document_delimiter_scan(scanner);
4038
4121
  scanner->at_here_document_line_start = true;
4039
4122
  lexer->mark_end(lexer);
4040
4123
  lexer->result_symbol = HERE_DOCUMENT_LINE_END;
@@ -4090,11 +4173,6 @@ classify_comment_boundary(TSLexer *lexer, const bool *valid_symbols) {
4090
4173
  );
4091
4174
  }
4092
4175
 
4093
- static enum ArithmeticOperatorCategory
4094
- classify_arithmetic_operator(int32_t first, int32_t second, int32_t third);
4095
- static bool is_arithmetic_operator_start(int32_t character);
4096
- static bool is_arithmetic_operand_start(int32_t character);
4097
-
4098
4176
  static bool arithmetic_operand_boundary_is_valid(const bool *valid_symbols) {
4099
4177
  return (
4100
4178
  valid_symbols[ARITHMETIC_PLUS_OPERAND_BOUNDARY] ||
@@ -4103,26 +4181,18 @@ static bool arithmetic_operand_boundary_is_valid(const bool *valid_symbols) {
4103
4181
  );
4104
4182
  }
4105
4183
 
4106
- static bool finish_line_continuation(TSLexer *lexer, enum TokenType symbol) {
4107
- if (lexer->lookahead != '\n') {
4108
- return false;
4109
- }
4110
-
4111
- lexer->advance(lexer, false);
4112
- lexer->mark_end(lexer);
4113
- lexer->result_symbol = symbol;
4114
- return true;
4115
- }
4116
-
4117
4184
  static bool scan_line_continuation_after_backslash(
4118
4185
  TSLexer *lexer,
4119
4186
  const bool *valid_symbols
4120
4187
  ) {
4121
- if (!valid_symbols[LINE_CONTINUATION]) {
4188
+ if (!valid_symbols[LINE_CONTINUATION] || lexer->lookahead != '\n') {
4122
4189
  return false;
4123
4190
  }
4124
4191
 
4125
- return finish_line_continuation(lexer, LINE_CONTINUATION);
4192
+ lexer->advance(lexer, false);
4193
+ lexer->mark_end(lexer);
4194
+ lexer->result_symbol = LINE_CONTINUATION;
4195
+ return true;
4126
4196
  }
4127
4197
 
4128
4198
  static bool scan_line_continuation(TSLexer *lexer, const bool *valid_symbols) {
@@ -4223,6 +4293,26 @@ static bool is_arithmetic_operator_start(int32_t character) {
4223
4293
  }
4224
4294
  }
4225
4295
 
4296
+ static bool is_arithmetic_operand_start(int32_t character) {
4297
+ return (
4298
+ is_name_start_character(character) ||
4299
+ is_decimal_digit(character) ||
4300
+ character ==
4301
+ '(' ||
4302
+ character ==
4303
+ '$' ||
4304
+ character ==
4305
+ '`' ||
4306
+ character ==
4307
+ '+' ||
4308
+ character ==
4309
+ '-' ||
4310
+ character ==
4311
+ '!' ||
4312
+ character == '~'
4313
+ );
4314
+ }
4315
+
4226
4316
  static bool scan_arithmetic_layout(TSLexer *lexer) {
4227
4317
  while (true) {
4228
4318
  while (
@@ -4291,26 +4381,6 @@ scan_arithmetic_boundary(TSLexer *lexer, const bool *valid_symbols) {
4291
4381
  return false;
4292
4382
  }
4293
4383
 
4294
- static bool is_arithmetic_operand_start(int32_t character) {
4295
- return (
4296
- is_name_start_character(character) ||
4297
- is_decimal_digit(character) ||
4298
- character ==
4299
- '(' ||
4300
- character ==
4301
- '$' ||
4302
- character ==
4303
- '`' ||
4304
- character ==
4305
- '+' ||
4306
- character ==
4307
- '-' ||
4308
- character ==
4309
- '!' ||
4310
- character == '~'
4311
- );
4312
- }
4313
-
4314
4384
  // Resolve the arithmetic readings before parsing; racing them lets an edited
4315
4385
  // tree reuse a flat subtree where a fresh parse selects the structured one.
4316
4386
 
@@ -4375,25 +4445,11 @@ struct EmbeddedFrame {
4375
4445
  bool command_context;
4376
4446
  };
4377
4447
 
4378
- enum EmbeddedCaseState {
4379
- EMBEDDED_CASE_EXPECT_WORD,
4380
- EMBEDDED_CASE_EXPECT_IN,
4381
- EMBEDDED_CASE_PATTERN,
4382
- EMBEDDED_CASE_BODY,
4383
- };
4384
-
4385
- struct EmbeddedCaseFrame {
4386
- size_t frame_count;
4387
- uint8_t state;
4388
- };
4389
-
4390
4448
  struct EmbeddedSkip {
4391
4449
  struct EmbeddedFrame *frames;
4392
4450
  size_t frame_count;
4393
4451
  size_t frame_capacity;
4394
- struct EmbeddedCaseFrame *cases;
4395
- size_t case_count;
4396
- size_t case_capacity;
4452
+ struct CaseTrackerBuffer cases;
4397
4453
  struct HereDocument *pending;
4398
4454
  size_t pending_count;
4399
4455
  size_t pending_capacity;
@@ -4401,7 +4457,7 @@ struct EmbeddedSkip {
4401
4457
 
4402
4458
  static void clear_embedded_skip(struct EmbeddedSkip *skip) {
4403
4459
  ts_free(skip->frames);
4404
- ts_free(skip->cases);
4460
+ ts_free(skip->cases.data);
4405
4461
  for (size_t index = 0; index < skip->pending_count; index += 1) {
4406
4462
  clear_document(&skip->pending[index]);
4407
4463
  }
@@ -4431,41 +4487,49 @@ static bool embedded_push_frame(
4431
4487
  return true;
4432
4488
  }
4433
4489
 
4434
- static bool embedded_push_case(struct EmbeddedSkip *skip) {
4435
- if (!grow_element_buffer(
4436
- (void **)&skip->cases,
4437
- &skip->case_capacity,
4438
- skip->case_count,
4439
- sizeof(struct EmbeddedCaseFrame),
4440
- 8
4441
- )) {
4442
- return false;
4490
+ static void embedded_note_word(struct EmbeddedSkip *skip, bool in_command) {
4491
+ if (!in_command) {
4492
+ return;
4443
4493
  }
4444
- skip->cases[skip->case_count] = (struct EmbeddedCaseFrame){
4445
- .frame_count = skip->frame_count,
4446
- .state = EMBEDDED_CASE_EXPECT_WORD,
4447
- };
4448
- skip->case_count += 1;
4449
- return true;
4494
+ (void)case_tracker_note_word(
4495
+ active_case_tracker(&skip->cases, skip->frame_count),
4496
+ CASE_WORD_GENERIC,
4497
+ false
4498
+ );
4450
4499
  }
4451
4500
 
4452
- static struct EmbeddedCaseFrame *
4453
- embedded_active_case(struct EmbeddedSkip *skip) {
4454
- if (skip->case_count == 0) {
4455
- return NULL;
4456
- }
4457
- struct EmbeddedCaseFrame *frame = &skip->cases[skip->case_count - 1];
4458
- return frame->frame_count == skip->frame_count ? frame : NULL;
4501
+ static bool embedded_word_is_delimited(const TSLexer *lexer) {
4502
+ return (
4503
+ lexer_at_eof(lexer) ||
4504
+ is_horizontal_blank(lexer->lookahead) ||
4505
+ lexer->lookahead ==
4506
+ '\n' ||
4507
+ is_control_operator_start(lexer->lookahead)
4508
+ );
4459
4509
  }
4460
4510
 
4461
- static void embedded_note_word(struct EmbeddedSkip *skip, bool in_command) {
4462
- if (!in_command) {
4463
- return;
4511
+ // Pushes the frames for a "$(", "${", or "$((" introducer whose "(" or "{"
4512
+ // is at the lookahead, mirroring push_dollar_delimiter_group.
4513
+ static bool
4514
+ embedded_push_dollar_group(struct EmbeddedSkip *skip, TSLexer *lexer) {
4515
+ bool command_context = lexer->lookahead == '(';
4516
+ if (!embedded_push_frame(
4517
+ skip,
4518
+ command_context ? ')' : '}',
4519
+ command_context
4520
+ )) {
4521
+ return false;
4464
4522
  }
4465
- struct EmbeddedCaseFrame *active_case = embedded_active_case(skip);
4466
- if (active_case != NULL && active_case->state == EMBEDDED_CASE_EXPECT_WORD) {
4467
- active_case->state = EMBEDDED_CASE_EXPECT_IN;
4523
+ lexer->advance(lexer, false);
4524
+
4525
+ if (command_context && lexer->lookahead == '(') {
4526
+ skip->frames[skip->frame_count - 1].command_context = false;
4527
+ if (!embedded_push_frame(skip, ')', false)) {
4528
+ return false;
4529
+ }
4530
+ lexer->advance(lexer, false);
4468
4531
  }
4532
+ return true;
4469
4533
  }
4470
4534
 
4471
4535
  static bool embedded_append_pending(
@@ -4691,8 +4755,8 @@ skip_embedded_construct(TSLexer *lexer, char initial_closer) {
4691
4755
 
4692
4756
  struct EmbeddedFrame *frame = &skip.frames[skip.frame_count - 1];
4693
4757
  bool in_command = frame->closer == ')' && frame->command_context;
4694
- struct EmbeddedCaseFrame *active_case =
4695
- in_command ? embedded_active_case(&skip) : NULL;
4758
+ struct CaseTracker *active_case =
4759
+ in_command ? active_case_tracker(&skip.cases, skip.frame_count) : NULL;
4696
4760
  int32_t character = lexer->lookahead;
4697
4761
 
4698
4762
  if (frame->closer == '`') {
@@ -4735,24 +4799,10 @@ skip_embedded_construct(TSLexer *lexer, char initial_closer) {
4735
4799
  if (character == '$') {
4736
4800
  lexer->advance(lexer, false);
4737
4801
  if (lexer->lookahead == '(' || lexer->lookahead == '{') {
4738
- bool command_context = lexer->lookahead == '(';
4739
- if (!embedded_push_frame(
4740
- &skip,
4741
- command_context ? ')' : '}',
4742
- command_context
4743
- )) {
4802
+ if (!embedded_push_dollar_group(&skip, lexer)) {
4744
4803
  result = ARITHMETIC_VALIDATION_INVALID;
4745
4804
  break;
4746
4805
  }
4747
- lexer->advance(lexer, false);
4748
- if (command_context && lexer->lookahead == '(') {
4749
- skip.frames[skip.frame_count - 1].command_context = false;
4750
- if (!embedded_push_frame(&skip, ')', false)) {
4751
- result = ARITHMETIC_VALIDATION_INVALID;
4752
- break;
4753
- }
4754
- lexer->advance(lexer, false);
4755
- }
4756
4806
  at_command_position = true;
4757
4807
  }
4758
4808
  continue;
@@ -4812,24 +4862,10 @@ skip_embedded_construct(TSLexer *lexer, char initial_closer) {
4812
4862
  at_word = true;
4813
4863
  at_command_position = false;
4814
4864
  if (lexer->lookahead == '(' || lexer->lookahead == '{') {
4815
- bool command_context = lexer->lookahead == '(';
4816
- if (!embedded_push_frame(
4817
- &skip,
4818
- command_context ? ')' : '}',
4819
- command_context
4820
- )) {
4865
+ if (!embedded_push_dollar_group(&skip, lexer)) {
4821
4866
  result = ARITHMETIC_VALIDATION_INVALID;
4822
4867
  break;
4823
4868
  }
4824
- lexer->advance(lexer, false);
4825
- if (command_context && lexer->lookahead == '(') {
4826
- skip.frames[skip.frame_count - 1].command_context = false;
4827
- if (!embedded_push_frame(&skip, ')', false)) {
4828
- result = ARITHMETIC_VALIDATION_INVALID;
4829
- break;
4830
- }
4831
- lexer->advance(lexer, false);
4832
- }
4833
4869
  at_command_position = true;
4834
4870
  continue;
4835
4871
  }
@@ -4868,7 +4904,7 @@ skip_embedded_construct(TSLexer *lexer, char initial_closer) {
4868
4904
  in_command &&
4869
4905
  character ==
4870
4906
  '<' &&
4871
- (active_case == NULL || active_case->state == EMBEDDED_CASE_BODY)
4907
+ (active_case == NULL || active_case->state == CASE_TRACKER_BODY)
4872
4908
  ) {
4873
4909
  lexer->advance(lexer, false);
4874
4910
  if (lexer->lookahead != '<') {
@@ -4906,7 +4942,8 @@ skip_embedded_construct(TSLexer *lexer, char initial_closer) {
4906
4942
 
4907
4943
  if (character == '(' && frame->closer == ')') {
4908
4944
  lexer->advance(lexer, false);
4909
- if (active_case != NULL && active_case->state == EMBEDDED_CASE_PATTERN) {
4945
+ if (active_case != NULL && case_tracker_in_pattern(active_case->state)) {
4946
+ active_case->state = CASE_TRACKER_IN_PATTERN;
4910
4947
  at_word = false;
4911
4948
  continue;
4912
4949
  }
@@ -4922,8 +4959,8 @@ skip_embedded_construct(TSLexer *lexer, char initial_closer) {
4922
4959
  if (character == ')' && frame->closer == ')') {
4923
4960
  if (active_case != NULL) {
4924
4961
  lexer->advance(lexer, false);
4925
- if (active_case->state == EMBEDDED_CASE_PATTERN) {
4926
- active_case->state = EMBEDDED_CASE_BODY;
4962
+ if (case_tracker_in_pattern(active_case->state)) {
4963
+ active_case->state = CASE_TRACKER_BODY;
4927
4964
  at_word = false;
4928
4965
  at_command_position = true;
4929
4966
  continue;
@@ -4934,13 +4971,7 @@ skip_embedded_construct(TSLexer *lexer, char initial_closer) {
4934
4971
  }
4935
4972
  lexer->advance(lexer, false);
4936
4973
  skip.frame_count -= 1;
4937
- while (
4938
- skip.case_count >
4939
- 0 &&
4940
- skip.cases[skip.case_count - 1].frame_count > skip.frame_count
4941
- ) {
4942
- skip.case_count -= 1;
4943
- }
4974
+ pop_case_trackers_at_depth(&skip.cases, skip.frame_count + 1);
4944
4975
  at_word = true;
4945
4976
  at_command_position = false;
4946
4977
  continue;
@@ -4959,19 +4990,33 @@ skip_embedded_construct(TSLexer *lexer, char initial_closer) {
4959
4990
  active_case !=
4960
4991
  NULL &&
4961
4992
  active_case->state ==
4962
- EMBEDDED_CASE_BODY &&
4993
+ CASE_TRACKER_BODY &&
4963
4994
  character == ';'
4964
4995
  ) {
4965
4996
  lexer->advance(lexer, false);
4966
4997
  if (lexer->lookahead == ';' || lexer->lookahead == '&') {
4967
4998
  lexer->advance(lexer, false);
4968
- active_case->state = EMBEDDED_CASE_PATTERN;
4999
+ active_case->state = CASE_TRACKER_EXPECT_PATTERN;
4969
5000
  }
4970
5001
  at_word = false;
4971
5002
  at_command_position = true;
4972
5003
  continue;
4973
5004
  }
4974
5005
 
5006
+ if (!at_word && (character == '{' || character == '!')) {
5007
+ lexer->advance(lexer, false);
5008
+ if (
5009
+ in_command && at_command_position && embedded_word_is_delimited(lexer)
5010
+ ) {
5011
+ at_word = true;
5012
+ continue;
5013
+ }
5014
+ embedded_note_word(&skip, in_command);
5015
+ at_word = true;
5016
+ at_command_position = false;
5017
+ continue;
5018
+ }
5019
+
4975
5020
  if (!at_word && is_lowercase_letter(character)) {
4976
5021
  char word[6];
4977
5022
  size_t length = 0;
@@ -4987,61 +5032,29 @@ skip_embedded_construct(TSLexer *lexer, char initial_closer) {
4987
5032
  lexer->advance(lexer, false);
4988
5033
  }
4989
5034
  word[length] = '\0';
4990
- int32_t next = lexer->lookahead;
4991
- bool delimited = lexer_at_eof(lexer) ||
4992
- is_horizontal_blank(next) ||
4993
- next ==
4994
- '\n' ||
4995
- is_control_operator_start(next) ||
4996
- next ==
4997
- '<' ||
4998
- next ==
4999
- '>' ||
5000
- next ==
5001
- '(' ||
5002
- next == ')';
5035
+ candidate = candidate && embedded_word_is_delimited(lexer);
5003
5036
  if (in_command) {
5004
- if (active_case != NULL) {
5005
- if (active_case->state == EMBEDDED_CASE_EXPECT_WORD) {
5006
- active_case->state = EMBEDDED_CASE_EXPECT_IN;
5007
- } else if (
5008
- active_case->state ==
5009
- EMBEDDED_CASE_EXPECT_IN &&
5010
- candidate &&
5011
- delimited &&
5012
- strcmp(word, "in") == 0
5013
- ) {
5014
- active_case->state = EMBEDDED_CASE_PATTERN;
5015
- } else if (
5016
- candidate &&
5017
- delimited &&
5018
- strcmp(word, "esac") ==
5019
- 0 &&
5020
- (active_case->state ==
5021
- EMBEDDED_CASE_PATTERN ||
5022
- (active_case->state == EMBEDDED_CASE_BODY && position))
5023
- ) {
5024
- skip.case_count -= 1;
5025
- } else if (
5026
- candidate &&
5027
- delimited &&
5028
- strcmp(word, "case") ==
5029
- 0 &&
5030
- position &&
5031
- active_case->state == EMBEDDED_CASE_BODY
5032
- ) {
5033
- if (!embedded_push_case(&skip)) {
5034
- result = ARITHMETIC_VALIDATION_INVALID;
5035
- break;
5036
- }
5037
- }
5038
- } else if (
5039
- candidate && delimited && position && strcmp(word, "case") == 0
5040
- ) {
5041
- if (!embedded_push_case(&skip)) {
5037
+ switch (case_tracker_note_word(
5038
+ active_case,
5039
+ candidate ? classify_case_word(word) : CASE_WORD_GENERIC,
5040
+ position
5041
+ )) {
5042
+ case CASE_TRACKER_NOTE_COMMAND_PREFIX:
5043
+ at_word = true;
5044
+ continue;
5045
+ case CASE_TRACKER_NOTE_END:
5046
+ skip.cases.length -= 1;
5047
+ break;
5048
+ case CASE_TRACKER_NOTE_BEGIN:
5049
+ if (!append_case_tracker(&skip.cases, skip.frame_count)) {
5042
5050
  result = ARITHMETIC_VALIDATION_INVALID;
5043
- break;
5044
5051
  }
5052
+ break;
5053
+ default:
5054
+ break;
5055
+ }
5056
+ if (result != ARITHMETIC_VALIDATION_VALID) {
5057
+ break;
5045
5058
  }
5046
5059
  }
5047
5060
  at_word = true;
@@ -5066,6 +5079,7 @@ skip_embedded_construct(TSLexer *lexer, char initial_closer) {
5066
5079
  } else if (character == '<' || character == '>') {
5067
5080
  at_word = false;
5068
5081
  } else {
5082
+ embedded_note_word(&skip, in_command);
5069
5083
  at_word = true;
5070
5084
  at_command_position = false;
5071
5085
  }
@@ -5789,11 +5803,7 @@ static bool scan_unbraced_parameter_start(TSLexer *lexer) {
5789
5803
 
5790
5804
  lexer->mark_end(lexer);
5791
5805
  lexer->advance(lexer, false);
5792
- if (
5793
- (is_name_start_character(lexer->lookahead) ||
5794
- is_decimal_digit(lexer->lookahead) ||
5795
- is_special_parameter_character(lexer->lookahead))
5796
- ) {
5806
+ if (is_parameter_start_character(lexer->lookahead)) {
5797
5807
  lexer->result_symbol = UNBRACED_PARAMETER_START;
5798
5808
  return true;
5799
5809
  }
@@ -5823,13 +5833,8 @@ scan_braced_numeric_parameter_start(TSLexer *lexer, const bool *valid_symbols) {
5823
5833
  lexer->advance(lexer, false);
5824
5834
  }
5825
5835
 
5826
- while (lexer->lookahead == '\\') {
5827
- lexer->advance(lexer, false);
5828
- if (lexer->lookahead != '\n') {
5829
- break;
5830
- }
5831
- lexer->advance(lexer, false);
5832
- }
5836
+ // A backslash pair that is not a continuation just ends the digit run.
5837
+ skip_line_continuations(lexer);
5833
5838
 
5834
5839
  if (is_decimal_digit(lexer->lookahead)) {
5835
5840
  return false;
@@ -5966,9 +5971,7 @@ static bool scan_backquote_prefix_after_first_backslash(
5966
5971
  '{' &&
5967
5972
  lexer->lookahead !=
5968
5973
  '(' &&
5969
- !is_name_start_character(lexer->lookahead) &&
5970
- !is_decimal_digit(lexer->lookahead) &&
5971
- !is_special_parameter_character(lexer->lookahead)
5974
+ !is_parameter_start_character(lexer->lookahead)
5972
5975
  ) {
5973
5976
  return false;
5974
5977
  }
@@ -6235,7 +6238,7 @@ static bool scan_active_here_document(
6235
6238
 
6236
6239
  if (valid_symbols[NEWLINE] && lexer->lookahead == '\n') {
6237
6240
  lexer->advance(lexer, false);
6238
- discard_uncommitted_here_document_delimiter(scanner);
6241
+ reset_here_document_delimiter_scan(scanner);
6239
6242
  scanner->at_here_document_line_start = true;
6240
6243
  lexer->mark_end(lexer);
6241
6244
  lexer->result_symbol = NEWLINE;
@@ -6921,7 +6924,7 @@ static bool scan_dispatch(
6921
6924
  if (!has_startable_pending_document(scanner)) {
6922
6925
  record_comment_line_lookahead(scanner, lexer, valid_symbols);
6923
6926
  }
6924
- discard_uncommitted_here_document_delimiter(scanner);
6927
+ reset_here_document_delimiter_scan(scanner);
6925
6928
  if (scanner->active_count > 0) {
6926
6929
  scanner->at_here_document_line_start = true;
6927
6930
  }
@@ -7037,9 +7040,7 @@ static bool scan_dispatch(
7037
7040
  if (
7038
7041
  valid_symbols[PRE_NEWLINE_BLANK] && is_horizontal_blank(lexer->lookahead)
7039
7042
  ) {
7040
- while (is_horizontal_blank(lexer->lookahead)) {
7041
- lexer->advance(lexer, false);
7042
- }
7043
+ scan_horizontal_blanks(lexer);
7043
7044
  lexer->mark_end(lexer);
7044
7045
  while (true) {
7045
7046
  if (lexer->lookahead == '\n') {
@@ -7053,12 +7054,11 @@ static bool scan_dispatch(
7053
7054
  return classify_comment_boundary(lexer, valid_symbols);
7054
7055
  }
7055
7056
  if (lexer->lookahead == ';' && valid_symbols[CASE_ITEM_END]) {
7056
- lexer->advance(lexer, false);
7057
- if (lexer->lookahead == ';' || lexer->lookahead == '&') {
7058
- lexer->result_symbol = CASE_ITEM_END;
7059
- return true;
7057
+ if (!scan_case_item_terminator(lexer)) {
7058
+ return false;
7060
7059
  }
7061
- return false;
7060
+ lexer->result_symbol = CASE_ITEM_END;
7061
+ return true;
7062
7062
  }
7063
7063
  if (lexer->lookahead != '\\') {
7064
7064
  return false;
@@ -7068,9 +7068,7 @@ static bool scan_dispatch(
7068
7068
  return false;
7069
7069
  }
7070
7070
  lexer->advance(lexer, false);
7071
- while (is_horizontal_blank(lexer->lookahead)) {
7072
- lexer->advance(lexer, false);
7073
- }
7071
+ scan_horizontal_blanks(lexer);
7074
7072
  }
7075
7073
  }
7076
7074
 
@@ -7152,7 +7150,7 @@ static bool scan_dispatch(
7152
7150
  if (valid_symbols[NEWLINE] && lexer->lookahead == '\n') {
7153
7151
  lexer->advance(lexer, false);
7154
7152
  lexer->mark_end(lexer);
7155
- discard_uncommitted_here_document_delimiter(scanner);
7153
+ reset_here_document_delimiter_scan(scanner);
7156
7154
  lexer->result_symbol = NEWLINE;
7157
7155
  return true;
7158
7156
  }
@@ -7202,10 +7200,8 @@ static bool scan_dispatch(
7202
7200
  }
7203
7201
 
7204
7202
  if (lexer->lookahead == '}') {
7205
- return (
7206
- valid_symbols[RIGHT_BRACE] &&
7207
- scan_reserved_character(scanner, lexer, '}', RIGHT_BRACE)
7208
- );
7203
+ return valid_symbols[RIGHT_BRACE] &&
7204
+ scan_delimited_character_token(scanner, lexer, RIGHT_BRACE);
7209
7205
  }
7210
7206
 
7211
7207
  if (
@@ -7243,9 +7239,8 @@ static bool scan_dispatch(
7243
7239
  }
7244
7240
 
7245
7241
  if (lexer->lookahead == '!') {
7246
- return (
7247
- valid_symbols[PIPELINE_NEGATION] && scan_pipeline_negation(scanner, lexer)
7248
- );
7242
+ return valid_symbols[PIPELINE_NEGATION] &&
7243
+ scan_delimited_character_token(scanner, lexer, PIPELINE_NEGATION);
7249
7244
  }
7250
7245
 
7251
7246
  if (is_lowercase_letter(lexer->lookahead)) {