summaryrefslogtreecommitdiff
path: root/py/lexer.c
diff options
context:
space:
mode:
Diffstat (limited to 'py/lexer.c')
-rw-r--r--py/lexer.c166
1 files changed, 162 insertions, 4 deletions
diff --git a/py/lexer.c b/py/lexer.c
index 755fa625b..00cd59bca 100644
--- a/py/lexer.c
+++ b/py/lexer.c
@@ -64,6 +64,12 @@ STATIC bool is_char_or3(mp_lexer_t *lex, byte c1, byte c2, byte c3) {
return lex->chr0 == c1 || lex->chr0 == c2 || lex->chr0 == c3;
}
+#if MICROPY_COMP_FSTRING_LITERAL
+STATIC bool is_char_or4(mp_lexer_t *lex, byte c1, byte c2, byte c3, byte c4) {
+ return lex->chr0 == c1 || lex->chr0 == c2 || lex->chr0 == c3 || lex->chr0 == c4;
+}
+#endif
+
STATIC bool is_char_following(mp_lexer_t *lex, byte c) {
return lex->chr1 == c;
}
@@ -107,7 +113,13 @@ STATIC bool is_following_odigit(mp_lexer_t *lex) {
STATIC bool is_string_or_bytes(mp_lexer_t *lex) {
return is_char_or(lex, '\'', '\"')
+#if MICROPY_COMP_FSTRING_LITERAL
+ || (is_char_or4(lex, 'r', 'u', 'b', 'f') && is_char_following_or(lex, '\'', '\"'))
+ || ((is_char_and(lex, 'r', 'f') || is_char_and(lex, 'f', 'r'))
+ && is_char_following_following_or(lex, '\'', '\"'))
+#else
|| (is_char_or3(lex, 'r', 'u', 'b') && is_char_following_or(lex, '\'', '\"'))
+#endif
|| ((is_char_and(lex, 'r', 'b') || is_char_and(lex, 'b', 'r'))
&& is_char_following_following_or(lex, '\'', '\"'));
}
@@ -121,6 +133,31 @@ STATIC bool is_tail_of_identifier(mp_lexer_t *lex) {
return is_head_of_identifier(lex) || is_digit(lex);
}
+#if MICROPY_COMP_FSTRING_LITERAL
+STATIC void swap_char_banks(mp_lexer_t *lex) {
+ if (lex->vstr_postfix_processing) {
+ lex->chr3 = lex->chr0;
+ lex->chr4 = lex->chr1;
+ lex->chr5 = lex->chr2;
+ lex->chr0 = lex->vstr_postfix.buf[0];
+ lex->chr1 = lex->vstr_postfix.buf[1];
+ lex->chr2 = lex->vstr_postfix.buf[2];
+
+ lex->vstr_postfix_idx = 3;
+ } else {
+ // blindly reset to the "backup" bank when done postfix processing
+ // this restores control to the mp_reader
+ lex->chr0 = lex->chr3;
+ lex->chr1 = lex->chr4;
+ lex->chr2 = lex->chr5;
+ // willfully ignoring setting chr3-5 here - WARNING consider those garbage data now
+
+ vstr_reset(&lex->vstr_postfix);
+ lex->vstr_postfix_idx = 0;
+ }
+}
+#endif
+
STATIC void next_char(mp_lexer_t *lex) {
if (lex->chr0 == '\n') {
// a new line
@@ -136,7 +173,19 @@ STATIC void next_char(mp_lexer_t *lex) {
lex->chr0 = lex->chr1;
lex->chr1 = lex->chr2;
- lex->chr2 = lex->reader.readbyte(lex->reader.data);
+
+#if MICROPY_COMP_FSTRING_LITERAL
+ if (lex->vstr_postfix_processing) {
+ if (lex->vstr_postfix_idx == lex->vstr_postfix.len) {
+ lex->chr2 = '\0';
+ } else {
+ lex->chr2 = lex->vstr_postfix.buf[lex->vstr_postfix_idx++];
+ }
+ } else
+#endif
+ {
+ lex->chr2 = lex->reader.readbyte(lex->reader.data);
+ }
if (lex->chr1 == '\r') {
// CR is a new line, converted to LF
@@ -151,6 +200,13 @@ STATIC void next_char(mp_lexer_t *lex) {
if (lex->chr2 == MP_LEXER_EOF && lex->chr1 != MP_LEXER_EOF && lex->chr1 != '\n') {
lex->chr2 = '\n';
}
+
+#if MICROPY_COMP_FSTRING_LITERAL
+ if (lex->vstr_postfix_processing && lex->chr0 == '\0') {
+ lex->vstr_postfix_processing = false;
+ swap_char_banks(lex);
+ }
+#endif
}
STATIC void indent_push(mp_lexer_t *lex, size_t indent) {
@@ -270,7 +326,7 @@ STATIC bool get_hex(mp_lexer_t *lex, size_t num_digits, mp_uint_t *result) {
return true;
}
-STATIC void parse_string_literal(mp_lexer_t *lex, bool is_raw) {
+STATIC void parse_string_literal(mp_lexer_t *lex, bool is_raw, bool is_fstring) {
// get first quoting character
char quote_char = '\'';
if (is_char(lex, '\"')) {
@@ -291,15 +347,71 @@ STATIC void parse_string_literal(mp_lexer_t *lex, bool is_raw) {
}
size_t n_closing = 0;
+#if MICROPY_COMP_FSTRING_LITERAL
+ bool in_expression = false;
+ bool expression_eat = true;
+#endif
+
while (!is_end(lex) && (num_quotes > 1 || !is_char(lex, '\n')) && n_closing < num_quotes) {
if (is_char(lex, quote_char)) {
n_closing += 1;
vstr_add_char(&lex->vstr, CUR_CHAR(lex));
} else {
n_closing = 0;
+#if MICROPY_COMP_FSTRING_LITERAL
+ if (is_fstring && is_char(lex, '{')) {
+ vstr_add_char(&lex->vstr, CUR_CHAR(lex));
+ in_expression = !in_expression;
+ expression_eat = in_expression;
+
+ if (lex->vstr_postfix.len == 0) {
+ vstr_add_str(&lex->vstr_postfix, ".format(");
+ }
+
+ next_char(lex);
+ continue;
+ }
+
+ if (is_fstring && is_char(lex, '}')) {
+ vstr_add_char(&lex->vstr, CUR_CHAR(lex));
+
+ if (in_expression) {
+ in_expression = false;
+ vstr_add_char(&lex->vstr_postfix, ',');
+ }
+
+ next_char(lex);
+ continue;
+ }
+
+ if (in_expression) {
+ // throw errors for illegal chars inside f-string expressions
+ if (is_char(lex, '#')) {
+ lex->tok_kind = MP_TOKEN_FSTRING_COMMENT;
+ return;
+ } else if (is_char(lex, '\\')) {
+ lex->tok_kind = MP_TOKEN_FSTRING_BACKSLASH;
+ return;
+ } else if (is_char(lex, ':')) {
+ expression_eat = false;
+ }
+
+ unichar c = CUR_CHAR(lex);
+ if (expression_eat) {
+ vstr_add_char(&lex->vstr_postfix, c);
+ } else {
+ vstr_add_char(&lex->vstr, c);
+ }
+
+ next_char(lex);
+ continue;
+ }
+#endif
+
if (is_char(lex, '\\')) {
next_char(lex);
unichar c = CUR_CHAR(lex);
+
if (is_raw) {
// raw strings allow escaping of quotes, but the backslash is also emitted
vstr_add_char(&lex->vstr, '\\');
@@ -430,6 +542,15 @@ STATIC bool skip_whitespace(mp_lexer_t *lex, bool stop_at_newline) {
}
void mp_lexer_to_next(mp_lexer_t *lex) {
+#if MICROPY_COMP_FSTRING_LITERAL
+ if (lex->vstr_postfix.len && !lex->vstr_postfix_processing) {
+ // end format call injection
+ vstr_add_char(&lex->vstr_postfix, ')');
+ lex->vstr_postfix_processing = true;
+ swap_char_banks(lex);
+ }
+#endif
+
// start new token text
vstr_reset(&lex->vstr);
@@ -481,10 +602,19 @@ void mp_lexer_to_next(mp_lexer_t *lex) {
// MP_TOKEN_END is used to indicate that this is the first string token
lex->tok_kind = MP_TOKEN_END;
+#if MICROPY_COMP_FSTRING_LITERAL
+ bool saw_normal = false, saw_fstring = false;
+#endif
+
// Loop to accumulate string/bytes literals
do {
// parse type codes
bool is_raw = false;
+#if MICROPY_COMP_FSTRING_LITERAL
+ bool is_fstring = false;
+#else
+ const bool is_fstring = false;
+#endif
mp_token_kind_t kind = MP_TOKEN_STRING;
int n_char = 0;
if (is_char(lex, 'u')) {
@@ -503,7 +633,33 @@ void mp_lexer_to_next(mp_lexer_t *lex) {
kind = MP_TOKEN_BYTES;
n_char = 2;
}
+#if MICROPY_COMP_FSTRING_LITERAL
+ if (is_char_following(lex, 'f')) {
+ lex->tok_kind = MP_TOKEN_FSTRING_RAW;
+ break;
+ }
+ } else if (is_char(lex, 'f')) {
+ if (is_char_following(lex, 'r')) {
+ lex->tok_kind = MP_TOKEN_FSTRING_RAW;
+ break;
+ }
+ n_char = 1;
+ is_fstring = true;
+#endif
+ }
+
+#if MICROPY_COMP_FSTRING_LITERAL
+ if (is_fstring) {
+ saw_fstring = true;
+ } else {
+ saw_normal = true;
+ }
+
+ if (saw_fstring && saw_normal) {
+ // Can't concatenate f-string with normal string
+ break;
}
+#endif
// Set or check token kind
if (lex->tok_kind == MP_TOKEN_END) {
@@ -522,13 +678,12 @@ void mp_lexer_to_next(mp_lexer_t *lex) {
}
// Parse the literal
- parse_string_literal(lex, is_raw);
+ parse_string_literal(lex, is_raw, is_fstring);
// Skip whitespace so we can check if there's another string following
skip_whitespace(lex, true);
} while (is_string_or_bytes(lex));
-
} else if (is_head_of_identifier(lex)) {
lex->tok_kind = MP_TOKEN_NAME;
@@ -682,6 +837,9 @@ mp_lexer_t *mp_lexer_new(qstr src_name, mp_reader_t reader) {
lex->num_indent_level = 1;
lex->indent_level = m_new(uint16_t, lex->alloc_indent_level);
vstr_init(&lex->vstr, 32);
+#if MICROPY_COMP_FSTRING_LITERAL
+ vstr_init(&lex->vstr_postfix, 0);
+#endif
// store sentinel for first indentation level
lex->indent_level[0] = 0;