diff --git a/Parser/lexer/lexer.c b/Parser/lexer/lexer.c index a4fdb9484abd8de..83071e9e0b8ddf0 100644 --- a/Parser/lexer/lexer.c +++ b/Parser/lexer/lexer.c @@ -13,12 +13,6 @@ tokenizing. */ static const char* type_comment_prefix = "# type: "; -static inline int -contains_null_bytes(const char* str, size_t size) -{ - return memchr(str, 0, size) != NULL; -} - int _PyLexer_refill(struct tok_state *tok) { @@ -29,8 +23,9 @@ _PyLexer_refill(struct tok_state *tok) #if defined(Py_DEBUG) if (tok->debug) { fprintf(stderr, "line[%d] = ", tok->lineno); - _PyTokenizer_print_escape(stderr, _PyLexer_BufferPointer(tok, tok->cur), - tok->inp - tok->cur); + _PyTokenizer_print_escape( + stderr, _PyTok_SourcePointer(&tok->source, tok->cur), + tok->inp - tok->cur); fprintf(stderr, " tok->done = %d\n", tok->done); } #endif @@ -39,8 +34,8 @@ _PyLexer_refill(struct tok_state *tok) return 0; } tok->line_start = tok->cur; - if (contains_null_bytes(_PyLexer_BufferPointer(tok, tok->line_start), - tok->inp - tok->line_start)) { + _PyTok_Span line = {tok->line_start, tok->inp}; + if (_PyTok_SourceFindByte(&tok->source, line, '\0') >= 0) { _PyTokenizer_syntaxerror(tok, "source code cannot contain null bytes"); tok->cur = tok->inp; return 0; @@ -56,7 +51,7 @@ _PyLexer_backup(struct tok_state *tok, int c) if (--tok->cur < tok->buf_offset) { Py_FatalError("tokenizer beginning of buffer"); } - if ((int)(unsigned char)*_PyLexer_BufferPointer(tok, tok->cur) != Py_CHARMASK(c)) { + if (_PyTok_SourceByte(&tok->source, tok->cur) != Py_CHARMASK(c)) { Py_FatalError("tok_backup: wrong character"); } } @@ -74,7 +69,8 @@ verify_identifier(struct tok_state *tok) PyObject *s; if (tok_failed(tok)) return 0; - s = PyUnicode_DecodeUTF8(_PyLexer_BufferPointer(tok, tok->start), tok->cur - tok->start, NULL); + s = PyUnicode_DecodeUTF8(_PyTok_SourcePointer(&tok->source, tok->start), + tok->cur - tok->start, NULL); if (s == NULL) { if (PyErr_ExceptionMatches(PyExc_UnicodeDecodeError)) { tok->done = E_DECODE; @@ -105,13 +101,13 @@ verify_identifier(struct tok_state *tok) Py_DECREF(s); if (Py_UNICODE_ISPRINTABLE(ch)) { _PyTokenizer_syntaxerror_at( - tok, _PyLexer_BufferPointer(tok, tok->line_start), + tok, _PyTok_SourcePointer(&tok->source, tok->line_start), error_cursor - tok->line_start, tok->lineno, -1, -1, "invalid character '%c' (U+%04X)", ch, ch); } else { _PyTokenizer_syntaxerror_at( - tok, _PyLexer_BufferPointer(tok, tok->line_start), + tok, _PyTok_SourcePointer(&tok->source, tok->line_start), error_cursor - tok->line_start, tok->lineno, -1, -1, "invalid non-printable character U+%04X", ch); } @@ -172,10 +168,6 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token /* Skip comment, unless it's a type comment */ if (c == '#') { - const char* p = NULL; - const char *prefix, *type_start; - int current_starting_col_offset; - while (c != EOF && c != '\n' && c != '\r') { c = tok_nextc(tok); } @@ -192,23 +184,20 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token } } - if (tok->tok_extra_tokens) { - p = _PyLexer_BufferPointer(tok, tok->start); - } - + _PyTok_Off comment = tok->start; if (tok->type_comments) { - p = _PyLexer_BufferPointer(tok, tok->start); - current_starting_col_offset = tok->start_loc.byte_col; - prefix = type_comment_prefix; - while (*prefix && p < _PyLexer_BufferPointer(tok, tok->cur)) { + const char *prefix = type_comment_prefix; + while (*prefix && comment < tok->cur) { if (*prefix == ' ') { - while (*p == ' ' || *p == '\t') { - p++; - current_starting_col_offset++; + while (comment < tok->cur) { + int ch = _PyTok_SourceByte(&tok->source, comment); + if (ch != ' ' && ch != '\t') { + break; + } + comment++; } - } else if (*prefix == *p) { - p++; - current_starting_col_offset++; + } else if (*prefix == _PyTok_SourceByte(&tok->source, comment)) { + comment++; } else { break; } @@ -218,35 +207,26 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token /* This is a type comment if we matched all of type_comment_prefix. */ if (!*prefix) { - int is_type_ignore = 1; - // +6 in order to skip the word 'ignore' - const char *ignore_end = p + 6; - const int ignore_end_col_offset = current_starting_col_offset + 6; tok_backup(tok, c); /* don't eat the newline or EOF */ - - type_start = p; - /* A TYPE_IGNORE is "type: ignore" followed by the end of the token * or anything ASCII and non-alphanumeric. */ - is_type_ignore = ( - _PyLexer_BufferPointer(tok, tok->cur) >= ignore_end && memcmp(p, "ignore", 6) == 0 - && !(_PyLexer_BufferPointer(tok, tok->cur) > ignore_end - && ((unsigned char)ignore_end[0] >= 128 || Py_ISALNUM(ignore_end[0])))); + int is_type_ignore = tok->cur - comment >= 6 && + memcmp(_PyTok_SourcePointer(&tok->source, comment), + "ignore", 6) == 0; + if (is_type_ignore && comment + 6 < tok->cur) { + int ch = _PyTok_SourceByte(&tok->source, comment + 6); + is_type_ignore = ch < 128 && !Py_ISALNUM(ch); + } int type = is_type_ignore ? TYPE_IGNORE : TYPE_COMMENT; - int start_col_offset = is_type_ignore - ? ignore_end_col_offset : current_starting_col_offset; + p_start = comment + (is_type_ignore ? 6 : 0); p_end = tok->cur; - if (is_type_ignore) { - p_start = _PyLexer_BufferOffset(tok, ignore_end); - - /* If this type ignore is the only thing on the line, consume the newline also. */ - if (blankline) { - tok_nextc(tok); - tok->layout.at_bol = 1; - } - } else { - p_start = _PyLexer_BufferOffset(tok, type_start); + int start_col_offset = tok->start_loc.byte_col + + (int)(p_start - tok->start); + /* Consume the newline after a standalone type ignore. */ + if (is_type_ignore && blankline) { + tok_nextc(tok); + tok->layout.at_bol = 1; } _PyLexer_token_setup(tok, token, type, p_start, p_end); token->start_loc = (_PyTok_Loc){tok->lineno, start_col_offset}; @@ -257,7 +237,7 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token } if (tok->tok_extra_tokens) { tok_backup(tok, c); /* don't eat the newline or EOF */ - p_start = _PyLexer_BufferOffset(tok, p); + p_start = comment; p_end = tok->cur; tok->layout.comment_newline = blankline; return MAKE_TOKEN(COMMENT); diff --git a/Parser/lexer/lexer_internal.h b/Parser/lexer/lexer_internal.h index 652512a3e88c22d..922a23bd97ef620 100644 --- a/Parser/lexer/lexer_internal.h +++ b/Parser/lexer/lexer_internal.h @@ -40,14 +40,11 @@ tok_nextc(struct tok_state *tok) return EOF; } } - assert(tok->cur >= tok->source.base_offset); - assert(tok->cur - tok->source.base_offset < tok->source.len); if (tok->cur - tok->line_start >= INT_MAX) { tok->done = E_COLUMNOVERFLOW; return EOF; } - return Py_CHARMASK( - tok->source.bytes[tok->cur++ - tok->source.base_offset]); + return _PyTok_SourceByte(&tok->source, tok->cur++); } /* Return -1 on error, otherwise whether the line is blank. */ diff --git a/Parser/lexer/state.c b/Parser/lexer/state.c index 75ff26f16d47ba7..71f1c86461c4bfb 100644 --- a/Parser/lexer/state.c +++ b/Parser/lexer/state.c @@ -58,9 +58,6 @@ _PyLexer_PopFTString(struct tok_state *tok) void _PyTokenizer_Free(struct tok_state *tok) { - if (tok->encoding != NULL) { - PyMem_Free(tok->encoding); - } Py_XDECREF(tok->filename); Py_XDECREF(tok->module); _PyTok_ReaderFree(tok); diff --git a/Parser/lexer/state.h b/Parser/lexer/state.h index 3b5d47e12a6c38f..51ad278b13c40db 100644 --- a/Parser/lexer/state.h +++ b/Parser/lexer/state.h @@ -80,7 +80,6 @@ struct tok_state { _PyTok_SourceText source; int done; /* E_OK normally, E_EOF at EOF, otherwise error code */ /* NB If done != E_OK, cur must be == inp!!! */ - FILE *fp; /* Rest of input; NULL if tokenizing a string */ lexer_layout_state layout; int lineno; /* Current line number */ _PyTok_Loc start_loc; @@ -92,9 +91,6 @@ struct tok_state { int parencolstack[MAXLEVEL]; PyObject *filename; PyObject *module; - /* Stuff for PEP 0263 */ - char *encoding; /* Source encoding. */ - struct _PyTok_Reader *reader; int type_comments; /* Whether to look for type comments */ @@ -135,33 +131,6 @@ _PyLexer_FTStringBracketDepth(const struct tok_state *tok, return tok->level - state->paren_level; } -static inline _PyTok_Off -_PyLexer_BufferOffset(const struct tok_state *tok, const char *position) -{ - const char *base = _PyTok_SourceData(&tok->source); - assert(position >= base && position <= base + tok->source.len); - return tok->source.base_offset + (position - base); -} - -static inline const char * -_PyLexer_BufferPointer(const struct tok_state *tok, _PyTok_Off offset) -{ - assert(offset >= tok->source.base_offset); - assert(offset - tok->source.base_offset <= tok->source.len); - return _PyTok_SourceData(&tok->source) + (offset - tok->source.base_offset); -} - -static inline const char * -_PyLexer_BufferSpanView(const struct tok_state *tok, _PyTok_Span span, - Py_ssize_t *length) -{ - assert(length != NULL); - assert(_PyTok_SpanIsValid(span)); - *length = span.end - span.start; - (void)_PyLexer_BufferPointer(tok, span.end); - return _PyLexer_BufferPointer(tok, span.start); -} - static inline int _PyLexer_ByteColumn(const struct tok_state *tok) { diff --git a/Parser/lexer/string.c b/Parser/lexer/string.c index 99e45578b6ca16e..d0f568dfa857580 100644 --- a/Parser/lexer/string.c +++ b/Parser/lexer/string.c @@ -76,13 +76,11 @@ finish_ftstring_expr(struct tok_state *tok, ftstring_state *state, if (!(state->debug_expr || tstring_interpolation) || token->metadata) { return 0; } - Py_ssize_t expr_len; - const char *expr = _PyLexer_BufferSpanView( - tok, state->expr_span, &expr_len); tokenizer_comments *comments = state->comments; PyObject *res; if (comments != NULL && comments->count > 0) { - Py_ssize_t stripped_size = expr_len; + Py_ssize_t stripped_size = + state->expr_span.end - state->expr_span.start; Py_ssize_t comment_count = 0; for (Py_ssize_t i = 0; i < comments->count; i++) { _PyTok_Span comment = comments->spans[i]; @@ -101,26 +99,28 @@ finish_ftstring_expr(struct tok_state *tok, ftstring_state *state, PyErr_NoMemory(); return -1; } - _PyTok_Off copied_to = state->expr_span.start; + _PyTok_Span kept = {state->expr_span.start, state->expr_span.start}; Py_ssize_t stripped_len = 0; for (Py_ssize_t i = 0; i < comment_count; i++) { - _PyTok_Span comment = comments->spans[i]; - Py_ssize_t length = comment.start - copied_to; - memcpy(stripped + stripped_len, - expr + copied_to - state->expr_span.start, - (size_t)length); + kept.end = comments->spans[i].start; + Py_ssize_t length; + const char *text = _PyTok_SourceSpanView( + &tok->source, kept, &length); + memcpy(stripped + stripped_len, text, (size_t)length); stripped_len += length; - copied_to = comment.end; + kept.start = comments->spans[i].end; } - Py_ssize_t length = state->expr_span.end - copied_to; - memcpy(stripped + stripped_len, - expr + copied_to - state->expr_span.start, - (size_t)length); - stripped_len += length; - res = PyUnicode_DecodeUTF8(stripped, stripped_len, NULL); + kept.end = state->expr_span.end; + Py_ssize_t length; + const char *text = _PyTok_SourceSpanView(&tok->source, kept, &length); + memcpy(stripped + stripped_len, text, (size_t)length); + res = PyUnicode_DecodeUTF8(stripped, stripped_size, NULL); PyMem_Free(stripped); } else { + Py_ssize_t expr_len; + const char *expr = _PyTok_SourceSpanView( + &tok->source, state->expr_span, &expr_len); res = PyUnicode_DecodeUTF8(expr, expr_len, NULL); } @@ -339,7 +339,8 @@ _PyLexer_scan_string(struct tok_state *tok, struct token *token, int c) } int end_lineno = tok->lineno; _PyTok_Loc location = tok->start_loc; - const char *line = _PyLexer_BufferPointer(tok, tok->start) - location.byte_col; + const char *line = _PyTok_SourcePointer( + &tok->source, tok->start - location.byte_col); Py_ssize_t cursor_offset = (Py_ssize_t)location.byte_col + 1; const ftstring_state *state = _PyLexer_CurrentFTString(tok); @@ -460,7 +461,8 @@ _PyLexer_get_ftstring(struct tok_state *tok, ftstring_state *current, struct tok int end_lineno = tok->lineno; _PyTok_Loc location = current->start_loc; - const char *line = _PyLexer_BufferPointer(tok, current->start) - location.byte_col; + const char *line = _PyTok_SourcePointer( + &tok->source, current->start - location.byte_col); Py_ssize_t cursor_offset = (Py_ssize_t)location.byte_col + 1; if (quote_size == 3) { diff --git a/Parser/tokenizer/api.c b/Parser/tokenizer/api.c index 3f5efa8dc8f26e6..d955c6b335e1db6 100644 --- a/Parser/tokenizer/api.c +++ b/Parser/tokenizer/api.c @@ -4,6 +4,7 @@ #include "tokenizer.h" #include "reader.h" +#include "reader_internal.h" #include "../lexer/state.h" _PyTokenizer_Info @@ -21,10 +22,10 @@ _PyTokenizer_GetInfo(const struct tok_state *tok) .delimiter_loc = {-1, -1}, .in_formatted_string = tok->ftstring_depth != 0, .is_interactive = _PyTok_ReaderIsInteractive(tok), - .is_file = tok->fp != NULL && tok->fp != stdin, + .is_file = tok->reader->fp != NULL && tok->reader->fp != stdin, .filename = tok->filename, .module = tok->module, - .encoding = tok->encoding, + .encoding = tok->reader->encoding, }; if (tok->level > 0) { int level = tok->level - 1; @@ -45,14 +46,14 @@ _PyToken_TextView(const struct tok_state *tok, const struct token *token, *length = 0; return ""; } - return _PyLexer_BufferSpanView(tok, token->span, length); + return _PyTok_SourceSpanView(&tok->source, token->span, length); } const char * _PyTokenizer_SpanView(const struct tok_state *tok, _PyTok_Span span, Py_ssize_t *length) { - return _PyLexer_BufferSpanView(tok, span, length); + return _PyTok_SourceSpanView(&tok->source, span, length); } void @@ -60,19 +61,13 @@ _PyToken_GetView(const struct tok_state *tok, const struct token *token, _PyToken_View *view) { assert(view != NULL); - assert((token->span.start == -1 && token->span.end == -1) || - _PyTok_SpanIsValid(token->span)); - if (token->span.start >= 0) { - (void)_PyLexer_BufferPointer(tok, token->span.end); - } - view->text = token->span.start < 0 - ? NULL : _PyLexer_BufferPointer(tok, token->span.start); - view->length = token->span.end - token->span.start; - view->end_line = _PyLexer_BufferPointer(tok, tok->line_start); - view->line = ISSTRINGLIT(token->type) - ? view->text - token->start_loc.byte_col : view->end_line; - view->line_length = tok->inp - tok->line_start + - (view->end_line - view->line); + view->text = _PyToken_TextView(tok, token, &view->length); + view->end_line_start = tok->line_start; + view->line_span = (_PyTok_Span){ + ISSTRINGLIT(token->type) + ? token->span.start - token->start_loc.byte_col : tok->line_start, + tok->inp, + }; view->implicit_newline = tok->implicit_newline; view->at_eof = tok->done == E_EOF; } @@ -111,22 +106,28 @@ _PyTokenizer_ImplyDedents(struct tok_state *tok) int _PyTokenizer_HasTrailingStatement(const struct tok_state *tok) { - const char *cur = _PyLexer_BufferPointer(tok, tok->cur); - char c = *cur; - for (;;) { - while (c == ' ' || c == '\t' || c == '\n' || c == '\014') { - c = *++cur; - } - if (!c) { + _PyTok_Off cur = tok->cur; + _PyTok_Off end = tok->source.base_offset + tok->source.len; + while (cur < end) { + int c = _PyTok_SourceByte(&tok->source, cur++); + if (c == '\0') { return 0; } + if (c == ' ' || c == '\t' || c == '\n' || c == '\014') { + continue; + } if (c != '#') { return 1; } - while (c && c != '\n') { - c = *++cur; + while (cur < end) { + c = _PyTok_SourceByte(&tok->source, cur); + if (c == '\0' || c == '\n') { + break; + } + cur++; } } + return 0; } int diff --git a/Parser/tokenizer/decoder.c b/Parser/tokenizer/decoder.c index 220fed65ad4e450..48f906845651aab 100644 --- a/Parser/tokenizer/decoder.c +++ b/Parser/tokenizer/decoder.c @@ -117,8 +117,8 @@ _PyTok_SetEncoding(struct tok_state *tok, const char *encoding) tok->done = E_NOMEM; return -1; } - PyMem_Free(tok->encoding); - tok->encoding = copy; + PyMem_Free(tok->reader->encoding); + tok->reader->encoding = copy; return 0; } @@ -246,8 +246,8 @@ _PyTok_DetectEncoding(struct tok_state *tok, const _PyTok_Chunk *first, PyMem_Free(cookie); return _PYTOK_ENCODING_ERROR; } - PyMem_Free(tok->encoding); - tok->encoding = cookie; + PyMem_Free(tok->reader->encoding); + tok->reader->encoding = cookie; return _PYTOK_ENCODING_DONE; } @@ -393,9 +393,10 @@ _PyTok_PrepareString(struct tok_state *tok, const char *input, int utf8_only, .len = raw_len, .ownership = _PYTOK_CHUNK_BORROWED, }; - if (tok->encoding != NULL && strcmp(tok->encoding, "utf-8") != 0) { + const char *encoding = tok->reader->encoding; + if (encoding != NULL && strcmp(encoding, "utf-8") != 0) { if (_PyTok_DecodeOnce( - tok, &decoded, tok->encoding, NULL) < 0) { + tok, &decoded, encoding, NULL) < 0) { return -1; } } @@ -407,7 +408,7 @@ _PyTok_PrepareString(struct tok_state *tok, const char *input, int utf8_only, return -1; } if (!utf8_only && - (tok->encoding == NULL || strcmp(tok->encoding, "utf-8") == 0) && + (encoding == NULL || strcmp(encoding, "utf-8") == 0) && !_PyTokenizer_ensure_utf8(_PyTok_SourceData(&tok->source), tok, 1)) { return -1; } @@ -418,22 +419,18 @@ int _PyTok_StartDecoder(struct tok_state *tok, const char *errors) { _PyTok_Reader *reader = tok->reader; - if (tok->encoding == NULL || reader->decoder != NULL) { + if (reader->encoding == NULL || reader->decoder != NULL) { return 0; } if (reader->kind == _PYTOK_READER_FILE && - strcmp(tok->encoding, "utf-8") == 0) { + strcmp(reader->encoding, "utf-8") == 0) { return 0; } - PyObject *codec = _PyCodec_LookupTextEncoding(tok->encoding, NULL); + PyObject *codec = _PyCodec_LookupTextEncoding(reader->encoding, NULL); if (codec != NULL) { - PyObject *factory = PyObject_GetAttrString(codec, "incrementaldecoder"); + reader->decoder = _PyCodecInfo_GetIncrementalDecoder(codec, errors); Py_DECREF(codec); - if (factory != NULL) { - reader->decoder = PyObject_CallFunction(factory, "s", errors); - Py_DECREF(factory); - } } if (reader->decoder == NULL) { tok->done = PyErr_ExceptionMatches(PyExc_MemoryError) diff --git a/Parser/tokenizer/helpers.c b/Parser/tokenizer/helpers.c index 0d3ea85109ec49e..c39e6756476b97c 100644 --- a/Parser/tokenizer/helpers.c +++ b/Parser/tokenizer/helpers.c @@ -91,9 +91,9 @@ _PyTokenizer_syntaxerror(struct tok_state *tok, const char *format, ...) // These errors are cleaned on startup. Todo: Fix it. va_list vargs; va_start(vargs, format); - int ret = _syntaxerror_range(tok, _PyLexer_BufferPointer(tok, tok->line_start), - tok->cur - tok->line_start, tok->lineno, - format, -1, -1, vargs); + int ret = _syntaxerror_range( + tok, _PyTok_SourcePointer(&tok->source, tok->line_start), + tok->cur - tok->line_start, tok->lineno, format, -1, -1, vargs); va_end(vargs); return ret; } @@ -105,9 +105,10 @@ _PyTokenizer_syntaxerror_known_range(struct tok_state *tok, { va_list vargs; va_start(vargs, format); - int ret = _syntaxerror_range(tok, _PyLexer_BufferPointer(tok, tok->line_start), - tok->cur - tok->line_start, tok->lineno, - format, col_offset, end_col_offset, vargs); + int ret = _syntaxerror_range( + tok, _PyTok_SourcePointer(&tok->source, tok->line_start), + tok->cur - tok->line_start, tok->lineno, + format, col_offset, end_col_offset, vargs); va_end(vargs); return ret; } diff --git a/Parser/tokenizer/reader.c b/Parser/tokenizer/reader.c index 36f910e3199756d..ce2bb1cf5c4bc30 100644 --- a/Parser/tokenizer/reader.c +++ b/Parser/tokenizer/reader.c @@ -34,6 +34,7 @@ _PyTok_ReaderFree(struct tok_state *tok) i < (int)Py_ARRAY_LENGTH(reader->prefetched_lines); i++) { _PyTok_ChunkClear(&reader->prefetched_lines[i]); } + PyMem_Free(reader->encoding); PyMem_Free(reader->file_buffer); PyMem_Free(reader->decoded); PyMem_Free(reader); @@ -173,15 +174,13 @@ next_prepared(struct tok_state *tok, _PyTok_Chunk *chunk) if (tok->lineno >= tok->source.nlines) { return _PYTOK_READ_EOF; } - const char *start = _PyLexer_BufferPointer(tok, tok->inp); - const char *newline = memchr( - start, '\n', tok->source.bytes + tok->source.len - start); - _PyTok_Off end = newline != NULL - ? newline - tok->source.bytes + 1 : tok->source.len; - chunk->data = (char *)start; - chunk->len = tok->source.bytes + end - start; + _PyTok_Span tail = {tok->inp, tok->source.base_offset + tok->source.len}; + _PyTok_Off newline = _PyTok_SourceFindByte(&tok->source, tail, '\n'); + _PyTok_Span line = {tail.start, newline >= 0 ? newline + 1 : tail.end}; + chunk->data = (char *)_PyTok_SourceSpanView( + &tok->source, line, &chunk->len); chunk->ownership = _PYTOK_CHUNK_BORROWED; - chunk->implicit_newline = end == tok->source.len && + chunk->implicit_newline = line.end == tail.end && tok->reader->prepared_final_newline_is_implicit; return _PYTOK_READ_LINE; } @@ -203,7 +202,7 @@ read_file_line(struct tok_state *tok, _PyTok_Chunk *chunk) int available = (int)Py_MIN(reader->file_buffer_cap - len, INT_MAX); size_t read = 0; char *result = _Py_UniversalNewlineFgetsWithSize( - reader->file_buffer + len, available, tok->fp, NULL, &read); + reader->file_buffer + len, available, reader->fp, NULL, &read); if (result == NULL) { if (len == 0) { return _PYTOK_READ_EOF; @@ -228,7 +227,7 @@ initialize_file(struct tok_state *tok) { _PyTok_Reader *reader = tok->reader; reader->file_initialized = 1; - if (tok->encoding != NULL) { + if (reader->encoding != NULL) { return _PyTok_StartDecoder(tok, "strict"); } @@ -411,7 +410,7 @@ next_readline(struct tok_state *tok, _PyTok_Chunk *chunk) } _PyTok_Chunk input = {0}; - if (tok->encoding != NULL) { + if (reader->encoding != NULL) { if (!PyBytes_Check(raw)) { PyErr_SetString(PyExc_TypeError, "readline() returned a non-bytes object"); @@ -434,7 +433,7 @@ next_readline(struct tok_state *tok, _PyTok_Chunk *chunk) input.ownership = _PYTOK_CHUNK_PYOBJECT; int decoded; if (reader->decoder == NULL && - strcmp(tok->encoding, "utf-8") == 0 && + strcmp(reader->encoding, "utf-8") == 0 && chunk_is_line(&input)) { decoded = _PyTok_DecodeOnce( tok, &input, "utf-8", "replace"); @@ -510,7 +509,7 @@ next_interactive(struct tok_state *tok, _PyTok_Chunk *chunk) return _PYTOK_READ_STOPPED; } char *input = PyOS_Readline( - tok->fp != NULL ? tok->fp : stdin, stdout, reader->prompt); + reader->fp != NULL ? reader->fp : stdin, stdout, reader->prompt); if (reader->nextprompt != NULL) { reader->prompt = reader->nextprompt; } @@ -527,9 +526,9 @@ next_interactive(struct tok_state *tok, _PyTok_Chunk *chunk) .len = len, .ownership = _PYTOK_CHUNK_PYMEM, }; - if (tok->encoding != NULL && + if (reader->encoding != NULL && _PyTok_DecodeOnce( - tok, &decoded, tok->encoding, NULL) < 0) { + tok, &decoded, reader->encoding, NULL) < 0) { _PyTok_ChunkClear(&decoded); return _PYTOK_READ_ERROR; } @@ -605,10 +604,11 @@ int _PyTok_ReaderUnderflow(struct tok_state *tok) { assert(tok->cur >= tok->buf_offset && tok->cur <= tok->inp); - _PyTok_ReaderKind kind = tok->reader->kind; + _PyTok_Reader *reader = tok->reader; + _PyTok_ReaderKind kind = reader->kind; int prepared = kind == _PYTOK_READER_PREPARED; int streaming = reader_is_streaming(kind); - int reset_buffer = !prepared && tok->start < 0 && + int reset_buffer = tok->start < 0 && _PyLexer_CurrentFTString(tok) == NULL; _PyTok_Chunk chunk; @@ -661,23 +661,24 @@ _PyTok_ReaderUnderflow(struct tok_state *tok) tok->cur = source_start; tok->buf_offset = source_start; tok->line_start = tok->buf_offset; - tok->start = -1; } tok->inp = source_start + chunk.len; } - if (prepared) { - if (tok->start < 0 && _PyLexer_CurrentFTString(tok) == NULL) { - tok->buf_offset = tok->source.base_offset + - (chunk.data - tok->source.bytes); + else { + if (reset_buffer) { + tok->buf_offset = tok->inp; } - tok->inp = _PyLexer_BufferOffset(tok, chunk.data) + chunk.len; + tok->inp += chunk.len; } tok->implicit_newline = chunk.implicit_newline; tok->lineno++; if (kind == _PYTOK_READER_FILE && - (tok->encoding == NULL || strcmp(tok->encoding, "utf-8") == 0) && - !_PyTokenizer_ensure_utf8(_PyLexer_BufferPointer(tok, tok->cur), tok, tok->lineno)) { + (reader->encoding == NULL || + strcmp(reader->encoding, "utf-8") == 0) && + !_PyTokenizer_ensure_utf8( + _PyTok_SourcePointer(&tok->source, tok->cur), + tok, tok->lineno)) { _PyTok_ChunkClear(&chunk); return 0; } @@ -779,7 +780,7 @@ _PyTokenizer_FromFile(FILE *fp, const char *encoding, _PyTokenizer_Free(tok); return NULL; } - tok->fp = fp; + tok->reader->fp = fp; tok->reader->prompt = ps1; tok->reader->nextprompt = ps2; return tok; @@ -840,8 +841,8 @@ _PyTokenizer_FindEncodingFilename(int fd, PyObject *filename) tok->filename = Py_NewRef(filename != NULL ? filename : &_Py_STR(anon_string)); char *encoding = NULL; if (initialize_file(tok) == 0) { - encoding = tok->encoding; - tok->encoding = NULL; + encoding = tok->reader->encoding; + tok->reader->encoding = NULL; } fclose(fp); _PyTokenizer_Free(tok); diff --git a/Parser/tokenizer/reader_internal.h b/Parser/tokenizer/reader_internal.h index f74063de8d653a4..a7a68269fffaa4a 100644 --- a/Parser/tokenizer/reader_internal.h +++ b/Parser/tokenizer/reader_internal.h @@ -41,6 +41,8 @@ typedef struct { } _PyTok_Chunk; typedef struct _PyTok_Reader { + FILE *fp; // Borrowed input stream; NULL for string and readline input. + char *encoding; // Owned source encoding. PyObject *readline; PyObject *decoder; const char *prompt; diff --git a/Parser/tokenizer/source.h b/Parser/tokenizer/source.h index 66980a41fb7c481..0fd9fe1877bed2b 100644 --- a/Parser/tokenizer/source.h +++ b/Parser/tokenizer/source.h @@ -19,6 +19,46 @@ _PyTok_SourceData(const _PyTok_SourceText *source) return source->bytes != NULL ? source->bytes : ""; } +/* Convert positions within the retained source window. Pointers and views + are borrowed; append, discard, and clear invalidate them. */ +static inline const char * +_PyTok_SourcePointer(const _PyTok_SourceText *source, _PyTok_Off offset) +{ + assert(offset >= source->base_offset); + assert(offset - source->base_offset <= source->len); + return _PyTok_SourceData(source) + (offset - source->base_offset); +} + +static inline unsigned char +_PyTok_SourceByte(const _PyTok_SourceText *source, _PyTok_Off offset) +{ + assert(offset >= source->base_offset); + assert(offset - source->base_offset < source->len); + return (unsigned char)source->bytes[offset - source->base_offset]; +} + +static inline const char * +_PyTok_SourceSpanView(const _PyTok_SourceText *source, _PyTok_Span span, + Py_ssize_t *length) +{ + assert(length != NULL); + assert(_PyTok_SpanIsValid(span)); + assert(span.end - source->base_offset <= source->len); + *length = span.end - span.start; + return _PyTok_SourcePointer(source, span.start); +} + +/* Return the first matching offset within span, or -1 if absent. */ +static inline _PyTok_Off +_PyTok_SourceFindByte(const _PyTok_SourceText *source, _PyTok_Span span, + int byte) +{ + Py_ssize_t length; + const char *data = _PyTok_SourceSpanView(source, span, &length); + const char *found = memchr(data, byte, length); + return found != NULL ? span.start + (found - data) : -1; +} + PyAPI_FUNC(void) _PyTok_SourceInit(_PyTok_SourceText *); /* Clear invalidates all spans and views for the source. */ PyAPI_FUNC(void) _PyTok_SourceClear(_PyTok_SourceText *); diff --git a/Parser/tokenizer/tokenizer.h b/Parser/tokenizer/tokenizer.h index 2ef87add3646c3c..bc4009700fdc967 100644 --- a/Parser/tokenizer/tokenizer.h +++ b/Parser/tokenizer/tokenizer.h @@ -21,9 +21,8 @@ struct token { typedef struct { const char *text; Py_ssize_t length; - const char *line; - Py_ssize_t line_length; - const char *end_line; + _PyTok_Span line_span; + _PyTok_Off end_line_start; int implicit_newline; int at_eof; } _PyToken_View; @@ -73,8 +72,9 @@ _PyTokenizer_Info _PyTokenizer_GetInfo(const struct tok_state *); /* An absent token span has a nonnull empty text view. */ const char *_PyToken_TextView( const struct tok_state *, const struct token *, Py_ssize_t *); -/* Use the token from the most recent Get. text is NULL for an absent span; - line includes the token's complete physical line range. */ +/* Use the token from the most recent Get. An absent span has nonnull empty text. + line_span covers the token's complete physical line range. end_line_start + is the logical offset of its final physical line. */ void _PyToken_GetView( const struct tok_state *tok, const struct token *token, _PyToken_View *view); diff --git a/Python/Python-tokenize.c b/Python/Python-tokenize.c index e91141bd1b9e610..731444e13a02750 100644 --- a/Python/Python-tokenize.c +++ b/Python/Python-tokenize.c @@ -158,12 +158,16 @@ _tokenizer_error(tokenizeriterobject *it) static PyObject * _get_current_line(tokenizeriterobject *it, int current_lineno, - const char *line_start, Py_ssize_t size, int *line_changed) + const _PyToken_View *view, int *line_changed) { _Py_CRITICAL_SECTION_ASSERT_OBJECT_LOCKED(it); if (current_lineno != it->last_lineno) { - // Line has changed since last token, so we fetch the new line and cache it - // in the iter object. + Py_ssize_t size; + const char *line_start = _PyTokenizer_SpanView( + it->tok, view->line_span, &size); + if (size > 0 && view->implicit_newline) { + size--; + } Py_XDECREF(it->last_line); it->last_line = PyUnicode_DecodeUTF8(line_start, size, "replace"); it->byte_col_offset_diff = 0; @@ -180,14 +184,13 @@ _get_col_offsets(tokenizeriterobject *it, const struct token *token, Py_ssize_t *col_offset, Py_ssize_t *end_col_offset) { _Py_CRITICAL_SECTION_ASSERT_OBJECT_LOCKED(it); - const char *token_start = view->text; - const char *token_end = token_start == NULL - ? NULL : token_start + view->length; + _PyTok_Off token_start = token->span.start; + _PyTok_Off token_end = token->span.end; Py_ssize_t lineno = token->start_loc.lineno; Py_ssize_t end_lineno = token->end_loc.lineno; Py_ssize_t byte_offset = -1; - if (token_start != NULL && token_start >= view->line) { - byte_offset = token_start - view->line; + if (token_start >= view->line_span.start) { + byte_offset = token_start - view->line_span.start; if (line_changed) { *col_offset = _PyPegen_byte_offset_to_character_offset_line(line, 0, byte_offset); if (*col_offset < 0) { @@ -200,8 +203,8 @@ _get_col_offsets(tokenizeriterobject *it, const struct token *token, } } - if (token_end != NULL && token_end >= view->end_line) { - Py_ssize_t end_byte_offset = token_end - view->end_line; + if (token_end >= view->end_line_start) { + Py_ssize_t end_byte_offset = token_end - view->end_line_start; if (lineno == end_lineno) { // Avoid rescanning the prefix of a very long line. Py_ssize_t token_col_offset = _PyPegen_byte_offset_to_character_offset_line(line, byte_offset, end_byte_offset); @@ -213,7 +216,8 @@ _get_col_offsets(tokenizeriterobject *it, const struct token *token, } else { *end_col_offset = _PyPegen_byte_offset_to_character_offset_line( - line, view->end_line - view->line, token_end - view->line); + line, view->end_line_start - view->line_span.start, + token_end - view->line_span.start); if (*end_col_offset < 0) { return -1; } @@ -251,15 +255,7 @@ tokenizeriter_next(PyObject *op) } _PyToken_View view; _PyToken_GetView(it->tok, &token, &view); - const char *token_start = view.text; - PyObject *str; - if (token.span.start < 0) { - assert(token.span.start == -1 && token.span.end == -1); - str = Py_GetConstant(Py_CONSTANT_EMPTY_STR); - } - else { - str = PyUnicode_FromStringAndSize(token_start, view.length); - } + PyObject *str = PyUnicode_FromStringAndSize(view.text, view.length); if (str == NULL) { goto exit; } @@ -271,13 +267,8 @@ tokenizeriter_next(PyObject *op) if (it->extra_tokens && is_trailing_token) { line = Py_GetConstant(Py_CONSTANT_EMPTY_STR); } else { - Py_ssize_t size = view.line_length; - if (size >= 1 && view.implicit_newline) { - size -= 1; - } - line = _get_current_line( - it, token.end_loc.lineno, view.line, size, &line_changed); + it, token.end_loc.lineno, &view, &line_changed); } if (line == NULL) { Py_DECREF(str); @@ -307,8 +298,7 @@ tokenizeriter_next(PyObject *op) else if (type == NEWLINE) { if (!view.implicit_newline) { Py_DECREF(str); - assert(token_start != NULL); - if (token_start[0] == '\r') { + if (view.text[0] == '\r') { str = PyUnicode_FromString("\r\n"); } else { str = PyUnicode_FromString("\n");