From 8f7e822f6b2e1bb9ecff82253776988d63f5e6c4 Mon Sep 17 00:00:00 2001 From: Pablo Galindo Salgado Date: Mon, 5 Oct 2026 16:45:16 +0100 Subject: [PATCH 1/6] gh-153569: Move tokenizer source access into the source API --- Parser/lexer/lexer.c | 34 ++++++++++++++++++-------------- Parser/lexer/lexer_internal.h | 5 +---- Parser/lexer/state.h | 27 ------------------------- Parser/lexer/string.c | 10 ++++++---- Parser/tokenizer/api.c | 12 ++++++------ Parser/tokenizer/helpers.c | 13 ++++++------ Parser/tokenizer/reader.c | 24 +++++++++++------------ Parser/tokenizer/source.h | 37 +++++++++++++++++++++++++++++++++++ 8 files changed, 88 insertions(+), 74 deletions(-) diff --git a/Parser/lexer/lexer.c b/Parser/lexer/lexer.c index a4fdb9484abd8de..687059605516030 100644 --- a/Parser/lexer/lexer.c +++ b/Parser/lexer/lexer.c @@ -29,8 +29,9 @@ _PyLexer_refill(struct tok_state *tok) #if defined(Py_DEBUG) if (tok->debug) { fprintf(stderr, "line[%d] = ", tok->lineno); - _PyTokenizer_print_escape(stderr, _PyLexer_BufferPointer(tok, tok->cur), - tok->inp - tok->cur); + _PyTokenizer_print_escape( + stderr, _PyTok_SourcePointer(&tok->source, tok->cur), + tok->inp - tok->cur); fprintf(stderr, " tok->done = %d\n", tok->done); } #endif @@ -39,7 +40,7 @@ _PyLexer_refill(struct tok_state *tok) return 0; } tok->line_start = tok->cur; - if (contains_null_bytes(_PyLexer_BufferPointer(tok, tok->line_start), + if (contains_null_bytes(_PyTok_SourcePointer(&tok->source, tok->line_start), tok->inp - tok->line_start)) { _PyTokenizer_syntaxerror(tok, "source code cannot contain null bytes"); tok->cur = tok->inp; @@ -56,7 +57,8 @@ _PyLexer_backup(struct tok_state *tok, int c) if (--tok->cur < tok->buf_offset) { Py_FatalError("tokenizer beginning of buffer"); } - if ((int)(unsigned char)*_PyLexer_BufferPointer(tok, tok->cur) != Py_CHARMASK(c)) { + const char *cur = _PyTok_SourcePointer(&tok->source, tok->cur); + if ((int)(unsigned char)*cur != Py_CHARMASK(c)) { Py_FatalError("tok_backup: wrong character"); } } @@ -74,7 +76,8 @@ verify_identifier(struct tok_state *tok) PyObject *s; if (tok_failed(tok)) return 0; - s = PyUnicode_DecodeUTF8(_PyLexer_BufferPointer(tok, tok->start), tok->cur - tok->start, NULL); + s = PyUnicode_DecodeUTF8(_PyTok_SourcePointer(&tok->source, tok->start), + tok->cur - tok->start, NULL); if (s == NULL) { if (PyErr_ExceptionMatches(PyExc_UnicodeDecodeError)) { tok->done = E_DECODE; @@ -105,13 +108,13 @@ verify_identifier(struct tok_state *tok) Py_DECREF(s); if (Py_UNICODE_ISPRINTABLE(ch)) { _PyTokenizer_syntaxerror_at( - tok, _PyLexer_BufferPointer(tok, tok->line_start), + tok, _PyTok_SourcePointer(&tok->source, tok->line_start), error_cursor - tok->line_start, tok->lineno, -1, -1, "invalid character '%c' (U+%04X)", ch, ch); } else { _PyTokenizer_syntaxerror_at( - tok, _PyLexer_BufferPointer(tok, tok->line_start), + tok, _PyTok_SourcePointer(&tok->source, tok->line_start), error_cursor - tok->line_start, tok->lineno, -1, -1, "invalid non-printable character U+%04X", ch); } @@ -193,14 +196,14 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token } if (tok->tok_extra_tokens) { - p = _PyLexer_BufferPointer(tok, tok->start); + p = _PyTok_SourcePointer(&tok->source, tok->start); } if (tok->type_comments) { - p = _PyLexer_BufferPointer(tok, tok->start); + p = _PyTok_SourcePointer(&tok->source, tok->start); current_starting_col_offset = tok->start_loc.byte_col; prefix = type_comment_prefix; - while (*prefix && p < _PyLexer_BufferPointer(tok, tok->cur)) { + while (*prefix && p < _PyTok_SourcePointer(&tok->source, tok->cur)) { if (*prefix == ' ') { while (*p == ' ' || *p == '\t') { p++; @@ -229,8 +232,9 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token /* A TYPE_IGNORE is "type: ignore" followed by the end of the token * or anything ASCII and non-alphanumeric. */ is_type_ignore = ( - _PyLexer_BufferPointer(tok, tok->cur) >= ignore_end && memcmp(p, "ignore", 6) == 0 - && !(_PyLexer_BufferPointer(tok, tok->cur) > ignore_end + _PyTok_SourcePointer(&tok->source, tok->cur) >= ignore_end + && memcmp(p, "ignore", 6) == 0 + && !(_PyTok_SourcePointer(&tok->source, tok->cur) > ignore_end && ((unsigned char)ignore_end[0] >= 128 || Py_ISALNUM(ignore_end[0])))); int type = is_type_ignore ? TYPE_IGNORE : TYPE_COMMENT; @@ -238,7 +242,7 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token ? ignore_end_col_offset : current_starting_col_offset; p_end = tok->cur; if (is_type_ignore) { - p_start = _PyLexer_BufferOffset(tok, ignore_end); + p_start = _PyTok_SourceOffset(&tok->source, ignore_end); /* If this type ignore is the only thing on the line, consume the newline also. */ if (blankline) { @@ -246,7 +250,7 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token tok->layout.at_bol = 1; } } else { - p_start = _PyLexer_BufferOffset(tok, type_start); + p_start = _PyTok_SourceOffset(&tok->source, type_start); } _PyLexer_token_setup(tok, token, type, p_start, p_end); token->start_loc = (_PyTok_Loc){tok->lineno, start_col_offset}; @@ -257,7 +261,7 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token } if (tok->tok_extra_tokens) { tok_backup(tok, c); /* don't eat the newline or EOF */ - p_start = _PyLexer_BufferOffset(tok, p); + p_start = _PyTok_SourceOffset(&tok->source, p); p_end = tok->cur; tok->layout.comment_newline = blankline; return MAKE_TOKEN(COMMENT); diff --git a/Parser/lexer/lexer_internal.h b/Parser/lexer/lexer_internal.h index 652512a3e88c22d..922a23bd97ef620 100644 --- a/Parser/lexer/lexer_internal.h +++ b/Parser/lexer/lexer_internal.h @@ -40,14 +40,11 @@ tok_nextc(struct tok_state *tok) return EOF; } } - assert(tok->cur >= tok->source.base_offset); - assert(tok->cur - tok->source.base_offset < tok->source.len); if (tok->cur - tok->line_start >= INT_MAX) { tok->done = E_COLUMNOVERFLOW; return EOF; } - return Py_CHARMASK( - tok->source.bytes[tok->cur++ - tok->source.base_offset]); + return _PyTok_SourceByte(&tok->source, tok->cur++); } /* Return -1 on error, otherwise whether the line is blank. */ diff --git a/Parser/lexer/state.h b/Parser/lexer/state.h index 3b5d47e12a6c38f..32c6fdeb568c8a2 100644 --- a/Parser/lexer/state.h +++ b/Parser/lexer/state.h @@ -135,33 +135,6 @@ _PyLexer_FTStringBracketDepth(const struct tok_state *tok, return tok->level - state->paren_level; } -static inline _PyTok_Off -_PyLexer_BufferOffset(const struct tok_state *tok, const char *position) -{ - const char *base = _PyTok_SourceData(&tok->source); - assert(position >= base && position <= base + tok->source.len); - return tok->source.base_offset + (position - base); -} - -static inline const char * -_PyLexer_BufferPointer(const struct tok_state *tok, _PyTok_Off offset) -{ - assert(offset >= tok->source.base_offset); - assert(offset - tok->source.base_offset <= tok->source.len); - return _PyTok_SourceData(&tok->source) + (offset - tok->source.base_offset); -} - -static inline const char * -_PyLexer_BufferSpanView(const struct tok_state *tok, _PyTok_Span span, - Py_ssize_t *length) -{ - assert(length != NULL); - assert(_PyTok_SpanIsValid(span)); - *length = span.end - span.start; - (void)_PyLexer_BufferPointer(tok, span.end); - return _PyLexer_BufferPointer(tok, span.start); -} - static inline int _PyLexer_ByteColumn(const struct tok_state *tok) { diff --git a/Parser/lexer/string.c b/Parser/lexer/string.c index 99e45578b6ca16e..fdfe4cd27b7ffff 100644 --- a/Parser/lexer/string.c +++ b/Parser/lexer/string.c @@ -77,8 +77,8 @@ finish_ftstring_expr(struct tok_state *tok, ftstring_state *state, return 0; } Py_ssize_t expr_len; - const char *expr = _PyLexer_BufferSpanView( - tok, state->expr_span, &expr_len); + const char *expr = _PyTok_SourceSpanView( + &tok->source, state->expr_span, &expr_len); tokenizer_comments *comments = state->comments; PyObject *res; if (comments != NULL && comments->count > 0) { @@ -339,7 +339,8 @@ _PyLexer_scan_string(struct tok_state *tok, struct token *token, int c) } int end_lineno = tok->lineno; _PyTok_Loc location = tok->start_loc; - const char *line = _PyLexer_BufferPointer(tok, tok->start) - location.byte_col; + const char *line = _PyTok_SourcePointer( + &tok->source, tok->start - location.byte_col); Py_ssize_t cursor_offset = (Py_ssize_t)location.byte_col + 1; const ftstring_state *state = _PyLexer_CurrentFTString(tok); @@ -460,7 +461,8 @@ _PyLexer_get_ftstring(struct tok_state *tok, ftstring_state *current, struct tok int end_lineno = tok->lineno; _PyTok_Loc location = current->start_loc; - const char *line = _PyLexer_BufferPointer(tok, current->start) - location.byte_col; + const char *line = _PyTok_SourcePointer( + &tok->source, current->start - location.byte_col); Py_ssize_t cursor_offset = (Py_ssize_t)location.byte_col + 1; if (quote_size == 3) { diff --git a/Parser/tokenizer/api.c b/Parser/tokenizer/api.c index 3f5efa8dc8f26e6..11f9870eb187055 100644 --- a/Parser/tokenizer/api.c +++ b/Parser/tokenizer/api.c @@ -45,14 +45,14 @@ _PyToken_TextView(const struct tok_state *tok, const struct token *token, *length = 0; return ""; } - return _PyLexer_BufferSpanView(tok, token->span, length); + return _PyTok_SourceSpanView(&tok->source, token->span, length); } const char * _PyTokenizer_SpanView(const struct tok_state *tok, _PyTok_Span span, Py_ssize_t *length) { - return _PyLexer_BufferSpanView(tok, span, length); + return _PyTok_SourceSpanView(&tok->source, span, length); } void @@ -63,12 +63,12 @@ _PyToken_GetView(const struct tok_state *tok, const struct token *token, assert((token->span.start == -1 && token->span.end == -1) || _PyTok_SpanIsValid(token->span)); if (token->span.start >= 0) { - (void)_PyLexer_BufferPointer(tok, token->span.end); + (void)_PyTok_SourcePointer(&tok->source, token->span.end); } view->text = token->span.start < 0 - ? NULL : _PyLexer_BufferPointer(tok, token->span.start); + ? NULL : _PyTok_SourcePointer(&tok->source, token->span.start); view->length = token->span.end - token->span.start; - view->end_line = _PyLexer_BufferPointer(tok, tok->line_start); + view->end_line = _PyTok_SourcePointer(&tok->source, tok->line_start); view->line = ISSTRINGLIT(token->type) ? view->text - token->start_loc.byte_col : view->end_line; view->line_length = tok->inp - tok->line_start + @@ -111,7 +111,7 @@ _PyTokenizer_ImplyDedents(struct tok_state *tok) int _PyTokenizer_HasTrailingStatement(const struct tok_state *tok) { - const char *cur = _PyLexer_BufferPointer(tok, tok->cur); + const char *cur = _PyTok_SourcePointer(&tok->source, tok->cur); char c = *cur; for (;;) { while (c == ' ' || c == '\t' || c == '\n' || c == '\014') { diff --git a/Parser/tokenizer/helpers.c b/Parser/tokenizer/helpers.c index 0d3ea85109ec49e..c39e6756476b97c 100644 --- a/Parser/tokenizer/helpers.c +++ b/Parser/tokenizer/helpers.c @@ -91,9 +91,9 @@ _PyTokenizer_syntaxerror(struct tok_state *tok, const char *format, ...) // These errors are cleaned on startup. Todo: Fix it. va_list vargs; va_start(vargs, format); - int ret = _syntaxerror_range(tok, _PyLexer_BufferPointer(tok, tok->line_start), - tok->cur - tok->line_start, tok->lineno, - format, -1, -1, vargs); + int ret = _syntaxerror_range( + tok, _PyTok_SourcePointer(&tok->source, tok->line_start), + tok->cur - tok->line_start, tok->lineno, format, -1, -1, vargs); va_end(vargs); return ret; } @@ -105,9 +105,10 @@ _PyTokenizer_syntaxerror_known_range(struct tok_state *tok, { va_list vargs; va_start(vargs, format); - int ret = _syntaxerror_range(tok, _PyLexer_BufferPointer(tok, tok->line_start), - tok->cur - tok->line_start, tok->lineno, - format, col_offset, end_col_offset, vargs); + int ret = _syntaxerror_range( + tok, _PyTok_SourcePointer(&tok->source, tok->line_start), + tok->cur - tok->line_start, tok->lineno, + format, col_offset, end_col_offset, vargs); va_end(vargs); return ret; } diff --git a/Parser/tokenizer/reader.c b/Parser/tokenizer/reader.c index 36f910e3199756d..bb766a2c65f5003 100644 --- a/Parser/tokenizer/reader.c +++ b/Parser/tokenizer/reader.c @@ -173,15 +173,14 @@ next_prepared(struct tok_state *tok, _PyTok_Chunk *chunk) if (tok->lineno >= tok->source.nlines) { return _PYTOK_READ_EOF; } - const char *start = _PyLexer_BufferPointer(tok, tok->inp); - const char *newline = memchr( - start, '\n', tok->source.bytes + tok->source.len - start); - _PyTok_Off end = newline != NULL - ? newline - tok->source.bytes + 1 : tok->source.len; + _PyTok_Span tail = {tok->inp, tok->source.base_offset + tok->source.len}; + Py_ssize_t remaining; + const char *start = _PyTok_SourceSpanView(&tok->source, tail, &remaining); + const char *newline = memchr(start, '\n', remaining); chunk->data = (char *)start; - chunk->len = tok->source.bytes + end - start; + chunk->len = newline != NULL ? newline - start + 1 : remaining; chunk->ownership = _PYTOK_CHUNK_BORROWED; - chunk->implicit_newline = end == tok->source.len && + chunk->implicit_newline = chunk->len == remaining && tok->reader->prepared_final_newline_is_implicit; return _PYTOK_READ_LINE; } @@ -665,19 +664,20 @@ _PyTok_ReaderUnderflow(struct tok_state *tok) } tok->inp = source_start + chunk.len; } - if (prepared) { + else { + _PyTok_Off source_start = _PyTok_SourceOffset(&tok->source, chunk.data); if (tok->start < 0 && _PyLexer_CurrentFTString(tok) == NULL) { - tok->buf_offset = tok->source.base_offset + - (chunk.data - tok->source.bytes); + tok->buf_offset = source_start; } - tok->inp = _PyLexer_BufferOffset(tok, chunk.data) + chunk.len; + tok->inp = source_start + chunk.len; } tok->implicit_newline = chunk.implicit_newline; tok->lineno++; if (kind == _PYTOK_READER_FILE && (tok->encoding == NULL || strcmp(tok->encoding, "utf-8") == 0) && - !_PyTokenizer_ensure_utf8(_PyLexer_BufferPointer(tok, tok->cur), tok, tok->lineno)) { + !_PyTokenizer_ensure_utf8( + _PyTok_SourcePointer(&tok->source, tok->cur), tok, tok->lineno)) { _PyTok_ChunkClear(&chunk); return 0; } diff --git a/Parser/tokenizer/source.h b/Parser/tokenizer/source.h index 66980a41fb7c481..e1aee720152b7f7 100644 --- a/Parser/tokenizer/source.h +++ b/Parser/tokenizer/source.h @@ -19,6 +19,43 @@ _PyTok_SourceData(const _PyTok_SourceText *source) return source->bytes != NULL ? source->bytes : ""; } +/* Convert positions within the retained source window. Pointers and views + are borrowed; append, discard, and clear invalidate them. */ +static inline _PyTok_Off +_PyTok_SourceOffset(const _PyTok_SourceText *source, const char *position) +{ + const char *base = _PyTok_SourceData(source); + assert(position >= base && position <= base + source->len); + return source->base_offset + (position - base); +} + +static inline const char * +_PyTok_SourcePointer(const _PyTok_SourceText *source, _PyTok_Off offset) +{ + assert(offset >= source->base_offset); + assert(offset - source->base_offset <= source->len); + return _PyTok_SourceData(source) + (offset - source->base_offset); +} + +static inline unsigned char +_PyTok_SourceByte(const _PyTok_SourceText *source, _PyTok_Off offset) +{ + assert(offset >= source->base_offset); + assert(offset - source->base_offset < source->len); + return (unsigned char)source->bytes[offset - source->base_offset]; +} + +static inline const char * +_PyTok_SourceSpanView(const _PyTok_SourceText *source, _PyTok_Span span, + Py_ssize_t *length) +{ + assert(length != NULL); + assert(_PyTok_SpanIsValid(span)); + *length = span.end - span.start; + (void)_PyTok_SourcePointer(source, span.end); + return _PyTok_SourcePointer(source, span.start); +} + PyAPI_FUNC(void) _PyTok_SourceInit(_PyTok_SourceText *); /* Clear invalidates all spans and views for the source. */ PyAPI_FUNC(void) _PyTok_SourceClear(_PyTok_SourceText *); From 2b46f42c865d1e812022672d25310307b72e54e3 Mon Sep 17 00:00:00 2001 From: Pablo Galindo Salgado Date: Mon, 5 Oct 2026 16:45:16 +0100 Subject: [PATCH 2/6] gh-153569: Use logical spans in tokenizer views and column calculations --- Parser/tokenizer/api.c | 21 ++++++++------------- Parser/tokenizer/tokenizer.h | 9 +++++---- Python/Python-tokenize.c | 31 +++++++++++-------------------- 3 files changed, 24 insertions(+), 37 deletions(-) diff --git a/Parser/tokenizer/api.c b/Parser/tokenizer/api.c index 11f9870eb187055..62319c5e0b2dab0 100644 --- a/Parser/tokenizer/api.c +++ b/Parser/tokenizer/api.c @@ -60,19 +60,14 @@ _PyToken_GetView(const struct tok_state *tok, const struct token *token, _PyToken_View *view) { assert(view != NULL); - assert((token->span.start == -1 && token->span.end == -1) || - _PyTok_SpanIsValid(token->span)); - if (token->span.start >= 0) { - (void)_PyTok_SourcePointer(&tok->source, token->span.end); - } - view->text = token->span.start < 0 - ? NULL : _PyTok_SourcePointer(&tok->source, token->span.start); - view->length = token->span.end - token->span.start; - view->end_line = _PyTok_SourcePointer(&tok->source, tok->line_start); - view->line = ISSTRINGLIT(token->type) - ? view->text - token->start_loc.byte_col : view->end_line; - view->line_length = tok->inp - tok->line_start + - (view->end_line - view->line); + view->text = _PyToken_TextView(tok, token, &view->length); + view->end_line_start = tok->line_start; + view->line_span = (_PyTok_Span){ + ISSTRINGLIT(token->type) + ? token->span.start - token->start_loc.byte_col : tok->line_start, + tok->inp, + }; + view->line = _PyTok_SourcePointer(&tok->source, view->line_span.start); view->implicit_newline = tok->implicit_newline; view->at_eof = tok->done == E_EOF; } diff --git a/Parser/tokenizer/tokenizer.h b/Parser/tokenizer/tokenizer.h index 2ef87add3646c3c..39d68fa4010571d 100644 --- a/Parser/tokenizer/tokenizer.h +++ b/Parser/tokenizer/tokenizer.h @@ -22,8 +22,8 @@ typedef struct { const char *text; Py_ssize_t length; const char *line; - Py_ssize_t line_length; - const char *end_line; + _PyTok_Span line_span; + _PyTok_Off end_line_start; int implicit_newline; int at_eof; } _PyToken_View; @@ -73,8 +73,9 @@ _PyTokenizer_Info _PyTokenizer_GetInfo(const struct tok_state *); /* An absent token span has a nonnull empty text view. */ const char *_PyToken_TextView( const struct tok_state *, const struct token *, Py_ssize_t *); -/* Use the token from the most recent Get. text is NULL for an absent span; - line includes the token's complete physical line range. */ +/* Use the token from the most recent Get. An absent span has nonnull empty text. + line contains the bytes of line_span, the token's complete physical line + range. end_line_start is the logical offset of its final physical line. */ void _PyToken_GetView( const struct tok_state *tok, const struct token *token, _PyToken_View *view); diff --git a/Python/Python-tokenize.c b/Python/Python-tokenize.c index e91141bd1b9e610..ad70ef514d6a25b 100644 --- a/Python/Python-tokenize.c +++ b/Python/Python-tokenize.c @@ -180,14 +180,13 @@ _get_col_offsets(tokenizeriterobject *it, const struct token *token, Py_ssize_t *col_offset, Py_ssize_t *end_col_offset) { _Py_CRITICAL_SECTION_ASSERT_OBJECT_LOCKED(it); - const char *token_start = view->text; - const char *token_end = token_start == NULL - ? NULL : token_start + view->length; + _PyTok_Off token_start = token->span.start; + _PyTok_Off token_end = token->span.end; Py_ssize_t lineno = token->start_loc.lineno; Py_ssize_t end_lineno = token->end_loc.lineno; Py_ssize_t byte_offset = -1; - if (token_start != NULL && token_start >= view->line) { - byte_offset = token_start - view->line; + if (token_start >= view->line_span.start) { + byte_offset = token_start - view->line_span.start; if (line_changed) { *col_offset = _PyPegen_byte_offset_to_character_offset_line(line, 0, byte_offset); if (*col_offset < 0) { @@ -200,8 +199,8 @@ _get_col_offsets(tokenizeriterobject *it, const struct token *token, } } - if (token_end != NULL && token_end >= view->end_line) { - Py_ssize_t end_byte_offset = token_end - view->end_line; + if (token_end >= view->end_line_start) { + Py_ssize_t end_byte_offset = token_end - view->end_line_start; if (lineno == end_lineno) { // Avoid rescanning the prefix of a very long line. Py_ssize_t token_col_offset = _PyPegen_byte_offset_to_character_offset_line(line, byte_offset, end_byte_offset); @@ -213,7 +212,8 @@ _get_col_offsets(tokenizeriterobject *it, const struct token *token, } else { *end_col_offset = _PyPegen_byte_offset_to_character_offset_line( - line, view->end_line - view->line, token_end - view->line); + line, view->end_line_start - view->line_span.start, + token_end - view->line_span.start); if (*end_col_offset < 0) { return -1; } @@ -251,15 +251,7 @@ tokenizeriter_next(PyObject *op) } _PyToken_View view; _PyToken_GetView(it->tok, &token, &view); - const char *token_start = view.text; - PyObject *str; - if (token.span.start < 0) { - assert(token.span.start == -1 && token.span.end == -1); - str = Py_GetConstant(Py_CONSTANT_EMPTY_STR); - } - else { - str = PyUnicode_FromStringAndSize(token_start, view.length); - } + PyObject *str = PyUnicode_FromStringAndSize(view.text, view.length); if (str == NULL) { goto exit; } @@ -271,7 +263,7 @@ tokenizeriter_next(PyObject *op) if (it->extra_tokens && is_trailing_token) { line = Py_GetConstant(Py_CONSTANT_EMPTY_STR); } else { - Py_ssize_t size = view.line_length; + Py_ssize_t size = view.line_span.end - view.line_span.start; if (size >= 1 && view.implicit_newline) { size -= 1; } @@ -307,8 +299,7 @@ tokenizeriter_next(PyObject *op) else if (type == NEWLINE) { if (!view.implicit_newline) { Py_DECREF(str); - assert(token_start != NULL); - if (token_start[0] == '\r') { + if (view.text[0] == '\r') { str = PyUnicode_FromString("\r\n"); } else { str = PyUnicode_FromString("\n"); From 1cfbc932fbf4364d364e5300f0d194ed767188bc Mon Sep 17 00:00:00 2001 From: Pablo Galindo Salgado Date: Mon, 5 Oct 2026 16:45:17 +0100 Subject: [PATCH 3/6] gh-153569: Move tokenizer input state and encoding ownership into the reader --- Parser/lexer/state.c | 3 --- Parser/lexer/state.h | 4 ---- Parser/tokenizer/api.c | 5 +++-- Parser/tokenizer/decoder.c | 21 +++++++++++---------- Parser/tokenizer/reader.c | 26 ++++++++++++++------------ Parser/tokenizer/reader_internal.h | 2 ++ 6 files changed, 30 insertions(+), 31 deletions(-) diff --git a/Parser/lexer/state.c b/Parser/lexer/state.c index 75ff26f16d47ba7..71f1c86461c4bfb 100644 --- a/Parser/lexer/state.c +++ b/Parser/lexer/state.c @@ -58,9 +58,6 @@ _PyLexer_PopFTString(struct tok_state *tok) void _PyTokenizer_Free(struct tok_state *tok) { - if (tok->encoding != NULL) { - PyMem_Free(tok->encoding); - } Py_XDECREF(tok->filename); Py_XDECREF(tok->module); _PyTok_ReaderFree(tok); diff --git a/Parser/lexer/state.h b/Parser/lexer/state.h index 32c6fdeb568c8a2..51ad278b13c40db 100644 --- a/Parser/lexer/state.h +++ b/Parser/lexer/state.h @@ -80,7 +80,6 @@ struct tok_state { _PyTok_SourceText source; int done; /* E_OK normally, E_EOF at EOF, otherwise error code */ /* NB If done != E_OK, cur must be == inp!!! */ - FILE *fp; /* Rest of input; NULL if tokenizing a string */ lexer_layout_state layout; int lineno; /* Current line number */ _PyTok_Loc start_loc; @@ -92,9 +91,6 @@ struct tok_state { int parencolstack[MAXLEVEL]; PyObject *filename; PyObject *module; - /* Stuff for PEP 0263 */ - char *encoding; /* Source encoding. */ - struct _PyTok_Reader *reader; int type_comments; /* Whether to look for type comments */ diff --git a/Parser/tokenizer/api.c b/Parser/tokenizer/api.c index 62319c5e0b2dab0..5e55d14f67317fd 100644 --- a/Parser/tokenizer/api.c +++ b/Parser/tokenizer/api.c @@ -4,6 +4,7 @@ #include "tokenizer.h" #include "reader.h" +#include "reader_internal.h" #include "../lexer/state.h" _PyTokenizer_Info @@ -21,10 +22,10 @@ _PyTokenizer_GetInfo(const struct tok_state *tok) .delimiter_loc = {-1, -1}, .in_formatted_string = tok->ftstring_depth != 0, .is_interactive = _PyTok_ReaderIsInteractive(tok), - .is_file = tok->fp != NULL && tok->fp != stdin, + .is_file = tok->reader->fp != NULL && tok->reader->fp != stdin, .filename = tok->filename, .module = tok->module, - .encoding = tok->encoding, + .encoding = tok->reader->encoding, }; if (tok->level > 0) { int level = tok->level - 1; diff --git a/Parser/tokenizer/decoder.c b/Parser/tokenizer/decoder.c index 220fed65ad4e450..8790f4b91a403bd 100644 --- a/Parser/tokenizer/decoder.c +++ b/Parser/tokenizer/decoder.c @@ -117,8 +117,8 @@ _PyTok_SetEncoding(struct tok_state *tok, const char *encoding) tok->done = E_NOMEM; return -1; } - PyMem_Free(tok->encoding); - tok->encoding = copy; + PyMem_Free(tok->reader->encoding); + tok->reader->encoding = copy; return 0; } @@ -246,8 +246,8 @@ _PyTok_DetectEncoding(struct tok_state *tok, const _PyTok_Chunk *first, PyMem_Free(cookie); return _PYTOK_ENCODING_ERROR; } - PyMem_Free(tok->encoding); - tok->encoding = cookie; + PyMem_Free(tok->reader->encoding); + tok->reader->encoding = cookie; return _PYTOK_ENCODING_DONE; } @@ -393,9 +393,10 @@ _PyTok_PrepareString(struct tok_state *tok, const char *input, int utf8_only, .len = raw_len, .ownership = _PYTOK_CHUNK_BORROWED, }; - if (tok->encoding != NULL && strcmp(tok->encoding, "utf-8") != 0) { + const char *encoding = tok->reader->encoding; + if (encoding != NULL && strcmp(encoding, "utf-8") != 0) { if (_PyTok_DecodeOnce( - tok, &decoded, tok->encoding, NULL) < 0) { + tok, &decoded, encoding, NULL) < 0) { return -1; } } @@ -407,7 +408,7 @@ _PyTok_PrepareString(struct tok_state *tok, const char *input, int utf8_only, return -1; } if (!utf8_only && - (tok->encoding == NULL || strcmp(tok->encoding, "utf-8") == 0) && + (encoding == NULL || strcmp(encoding, "utf-8") == 0) && !_PyTokenizer_ensure_utf8(_PyTok_SourceData(&tok->source), tok, 1)) { return -1; } @@ -418,15 +419,15 @@ int _PyTok_StartDecoder(struct tok_state *tok, const char *errors) { _PyTok_Reader *reader = tok->reader; - if (tok->encoding == NULL || reader->decoder != NULL) { + if (reader->encoding == NULL || reader->decoder != NULL) { return 0; } if (reader->kind == _PYTOK_READER_FILE && - strcmp(tok->encoding, "utf-8") == 0) { + strcmp(reader->encoding, "utf-8") == 0) { return 0; } - PyObject *codec = _PyCodec_LookupTextEncoding(tok->encoding, NULL); + PyObject *codec = _PyCodec_LookupTextEncoding(reader->encoding, NULL); if (codec != NULL) { PyObject *factory = PyObject_GetAttrString(codec, "incrementaldecoder"); Py_DECREF(codec); diff --git a/Parser/tokenizer/reader.c b/Parser/tokenizer/reader.c index bb766a2c65f5003..2f426d7c26a60e5 100644 --- a/Parser/tokenizer/reader.c +++ b/Parser/tokenizer/reader.c @@ -34,6 +34,7 @@ _PyTok_ReaderFree(struct tok_state *tok) i < (int)Py_ARRAY_LENGTH(reader->prefetched_lines); i++) { _PyTok_ChunkClear(&reader->prefetched_lines[i]); } + PyMem_Free(reader->encoding); PyMem_Free(reader->file_buffer); PyMem_Free(reader->decoded); PyMem_Free(reader); @@ -202,7 +203,7 @@ read_file_line(struct tok_state *tok, _PyTok_Chunk *chunk) int available = (int)Py_MIN(reader->file_buffer_cap - len, INT_MAX); size_t read = 0; char *result = _Py_UniversalNewlineFgetsWithSize( - reader->file_buffer + len, available, tok->fp, NULL, &read); + reader->file_buffer + len, available, reader->fp, NULL, &read); if (result == NULL) { if (len == 0) { return _PYTOK_READ_EOF; @@ -227,7 +228,7 @@ initialize_file(struct tok_state *tok) { _PyTok_Reader *reader = tok->reader; reader->file_initialized = 1; - if (tok->encoding != NULL) { + if (reader->encoding != NULL) { return _PyTok_StartDecoder(tok, "strict"); } @@ -410,7 +411,7 @@ next_readline(struct tok_state *tok, _PyTok_Chunk *chunk) } _PyTok_Chunk input = {0}; - if (tok->encoding != NULL) { + if (reader->encoding != NULL) { if (!PyBytes_Check(raw)) { PyErr_SetString(PyExc_TypeError, "readline() returned a non-bytes object"); @@ -433,7 +434,7 @@ next_readline(struct tok_state *tok, _PyTok_Chunk *chunk) input.ownership = _PYTOK_CHUNK_PYOBJECT; int decoded; if (reader->decoder == NULL && - strcmp(tok->encoding, "utf-8") == 0 && + strcmp(reader->encoding, "utf-8") == 0 && chunk_is_line(&input)) { decoded = _PyTok_DecodeOnce( tok, &input, "utf-8", "replace"); @@ -509,7 +510,7 @@ next_interactive(struct tok_state *tok, _PyTok_Chunk *chunk) return _PYTOK_READ_STOPPED; } char *input = PyOS_Readline( - tok->fp != NULL ? tok->fp : stdin, stdout, reader->prompt); + reader->fp != NULL ? reader->fp : stdin, stdout, reader->prompt); if (reader->nextprompt != NULL) { reader->prompt = reader->nextprompt; } @@ -526,9 +527,9 @@ next_interactive(struct tok_state *tok, _PyTok_Chunk *chunk) .len = len, .ownership = _PYTOK_CHUNK_PYMEM, }; - if (tok->encoding != NULL && + if (reader->encoding != NULL && _PyTok_DecodeOnce( - tok, &decoded, tok->encoding, NULL) < 0) { + tok, &decoded, reader->encoding, NULL) < 0) { _PyTok_ChunkClear(&decoded); return _PYTOK_READ_ERROR; } @@ -604,7 +605,8 @@ int _PyTok_ReaderUnderflow(struct tok_state *tok) { assert(tok->cur >= tok->buf_offset && tok->cur <= tok->inp); - _PyTok_ReaderKind kind = tok->reader->kind; + _PyTok_Reader *reader = tok->reader; + _PyTok_ReaderKind kind = reader->kind; int prepared = kind == _PYTOK_READER_PREPARED; int streaming = reader_is_streaming(kind); int reset_buffer = !prepared && tok->start < 0 && @@ -675,7 +677,7 @@ _PyTok_ReaderUnderflow(struct tok_state *tok) tok->lineno++; if (kind == _PYTOK_READER_FILE && - (tok->encoding == NULL || strcmp(tok->encoding, "utf-8") == 0) && + (reader->encoding == NULL || strcmp(reader->encoding, "utf-8") == 0) && !_PyTokenizer_ensure_utf8( _PyTok_SourcePointer(&tok->source, tok->cur), tok, tok->lineno)) { _PyTok_ChunkClear(&chunk); @@ -779,7 +781,7 @@ _PyTokenizer_FromFile(FILE *fp, const char *encoding, _PyTokenizer_Free(tok); return NULL; } - tok->fp = fp; + tok->reader->fp = fp; tok->reader->prompt = ps1; tok->reader->nextprompt = ps2; return tok; @@ -840,8 +842,8 @@ _PyTokenizer_FindEncodingFilename(int fd, PyObject *filename) tok->filename = Py_NewRef(filename != NULL ? filename : &_Py_STR(anon_string)); char *encoding = NULL; if (initialize_file(tok) == 0) { - encoding = tok->encoding; - tok->encoding = NULL; + encoding = tok->reader->encoding; + tok->reader->encoding = NULL; } fclose(fp); _PyTokenizer_Free(tok); diff --git a/Parser/tokenizer/reader_internal.h b/Parser/tokenizer/reader_internal.h index f74063de8d653a4..a7a68269fffaa4a 100644 --- a/Parser/tokenizer/reader_internal.h +++ b/Parser/tokenizer/reader_internal.h @@ -41,6 +41,8 @@ typedef struct { } _PyTok_Chunk; typedef struct _PyTok_Reader { + FILE *fp; // Borrowed input stream; NULL for string and readline input. + char *encoding; // Owned source encoding. PyObject *readline; PyObject *decoder; const char *prompt; From e60d602e6beb7fcd379b1da6548189beab487897 Mon Sep 17 00:00:00 2001 From: Pablo Galindo Salgado Date: Mon, 5 Oct 2026 16:45:17 +0100 Subject: [PATCH 4/6] gh-153569: Keep tokenizer source scans and copies on offsets and spans --- Parser/lexer/lexer.c | 82 ++++++++++++++------------------------- Parser/lexer/string.c | 31 +++++++-------- Parser/tokenizer/api.c | 24 +++++++----- Parser/tokenizer/reader.c | 15 +++---- Parser/tokenizer/source.h | 18 +++++---- 5 files changed, 75 insertions(+), 95 deletions(-) diff --git a/Parser/lexer/lexer.c b/Parser/lexer/lexer.c index 687059605516030..83071e9e0b8ddf0 100644 --- a/Parser/lexer/lexer.c +++ b/Parser/lexer/lexer.c @@ -13,12 +13,6 @@ tokenizing. */ static const char* type_comment_prefix = "# type: "; -static inline int -contains_null_bytes(const char* str, size_t size) -{ - return memchr(str, 0, size) != NULL; -} - int _PyLexer_refill(struct tok_state *tok) { @@ -40,8 +34,8 @@ _PyLexer_refill(struct tok_state *tok) return 0; } tok->line_start = tok->cur; - if (contains_null_bytes(_PyTok_SourcePointer(&tok->source, tok->line_start), - tok->inp - tok->line_start)) { + _PyTok_Span line = {tok->line_start, tok->inp}; + if (_PyTok_SourceFindByte(&tok->source, line, '\0') >= 0) { _PyTokenizer_syntaxerror(tok, "source code cannot contain null bytes"); tok->cur = tok->inp; return 0; @@ -57,8 +51,7 @@ _PyLexer_backup(struct tok_state *tok, int c) if (--tok->cur < tok->buf_offset) { Py_FatalError("tokenizer beginning of buffer"); } - const char *cur = _PyTok_SourcePointer(&tok->source, tok->cur); - if ((int)(unsigned char)*cur != Py_CHARMASK(c)) { + if (_PyTok_SourceByte(&tok->source, tok->cur) != Py_CHARMASK(c)) { Py_FatalError("tok_backup: wrong character"); } } @@ -175,10 +168,6 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token /* Skip comment, unless it's a type comment */ if (c == '#') { - const char* p = NULL; - const char *prefix, *type_start; - int current_starting_col_offset; - while (c != EOF && c != '\n' && c != '\r') { c = tok_nextc(tok); } @@ -195,23 +184,20 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token } } - if (tok->tok_extra_tokens) { - p = _PyTok_SourcePointer(&tok->source, tok->start); - } - + _PyTok_Off comment = tok->start; if (tok->type_comments) { - p = _PyTok_SourcePointer(&tok->source, tok->start); - current_starting_col_offset = tok->start_loc.byte_col; - prefix = type_comment_prefix; - while (*prefix && p < _PyTok_SourcePointer(&tok->source, tok->cur)) { + const char *prefix = type_comment_prefix; + while (*prefix && comment < tok->cur) { if (*prefix == ' ') { - while (*p == ' ' || *p == '\t') { - p++; - current_starting_col_offset++; + while (comment < tok->cur) { + int ch = _PyTok_SourceByte(&tok->source, comment); + if (ch != ' ' && ch != '\t') { + break; + } + comment++; } - } else if (*prefix == *p) { - p++; - current_starting_col_offset++; + } else if (*prefix == _PyTok_SourceByte(&tok->source, comment)) { + comment++; } else { break; } @@ -221,36 +207,26 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token /* This is a type comment if we matched all of type_comment_prefix. */ if (!*prefix) { - int is_type_ignore = 1; - // +6 in order to skip the word 'ignore' - const char *ignore_end = p + 6; - const int ignore_end_col_offset = current_starting_col_offset + 6; tok_backup(tok, c); /* don't eat the newline or EOF */ - - type_start = p; - /* A TYPE_IGNORE is "type: ignore" followed by the end of the token * or anything ASCII and non-alphanumeric. */ - is_type_ignore = ( - _PyTok_SourcePointer(&tok->source, tok->cur) >= ignore_end - && memcmp(p, "ignore", 6) == 0 - && !(_PyTok_SourcePointer(&tok->source, tok->cur) > ignore_end - && ((unsigned char)ignore_end[0] >= 128 || Py_ISALNUM(ignore_end[0])))); + int is_type_ignore = tok->cur - comment >= 6 && + memcmp(_PyTok_SourcePointer(&tok->source, comment), + "ignore", 6) == 0; + if (is_type_ignore && comment + 6 < tok->cur) { + int ch = _PyTok_SourceByte(&tok->source, comment + 6); + is_type_ignore = ch < 128 && !Py_ISALNUM(ch); + } int type = is_type_ignore ? TYPE_IGNORE : TYPE_COMMENT; - int start_col_offset = is_type_ignore - ? ignore_end_col_offset : current_starting_col_offset; + p_start = comment + (is_type_ignore ? 6 : 0); p_end = tok->cur; - if (is_type_ignore) { - p_start = _PyTok_SourceOffset(&tok->source, ignore_end); - - /* If this type ignore is the only thing on the line, consume the newline also. */ - if (blankline) { - tok_nextc(tok); - tok->layout.at_bol = 1; - } - } else { - p_start = _PyTok_SourceOffset(&tok->source, type_start); + int start_col_offset = tok->start_loc.byte_col + + (int)(p_start - tok->start); + /* Consume the newline after a standalone type ignore. */ + if (is_type_ignore && blankline) { + tok_nextc(tok); + tok->layout.at_bol = 1; } _PyLexer_token_setup(tok, token, type, p_start, p_end); token->start_loc = (_PyTok_Loc){tok->lineno, start_col_offset}; @@ -261,7 +237,7 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token } if (tok->tok_extra_tokens) { tok_backup(tok, c); /* don't eat the newline or EOF */ - p_start = _PyTok_SourceOffset(&tok->source, p); + p_start = comment; p_end = tok->cur; tok->layout.comment_newline = blankline; return MAKE_TOKEN(COMMENT); diff --git a/Parser/lexer/string.c b/Parser/lexer/string.c index fdfe4cd27b7ffff..ea245042b2ee3dc 100644 --- a/Parser/lexer/string.c +++ b/Parser/lexer/string.c @@ -76,13 +76,10 @@ finish_ftstring_expr(struct tok_state *tok, ftstring_state *state, if (!(state->debug_expr || tstring_interpolation) || token->metadata) { return 0; } - Py_ssize_t expr_len; - const char *expr = _PyTok_SourceSpanView( - &tok->source, state->expr_span, &expr_len); tokenizer_comments *comments = state->comments; PyObject *res; if (comments != NULL && comments->count > 0) { - Py_ssize_t stripped_size = expr_len; + Py_ssize_t stripped_size = state->expr_span.end - state->expr_span.start; Py_ssize_t comment_count = 0; for (Py_ssize_t i = 0; i < comments->count; i++) { _PyTok_Span comment = comments->spans[i]; @@ -103,24 +100,26 @@ finish_ftstring_expr(struct tok_state *tok, ftstring_state *state, } _PyTok_Off copied_to = state->expr_span.start; Py_ssize_t stripped_len = 0; - for (Py_ssize_t i = 0; i < comment_count; i++) { - _PyTok_Span comment = comments->spans[i]; - Py_ssize_t length = comment.start - copied_to; - memcpy(stripped + stripped_len, - expr + copied_to - state->expr_span.start, - (size_t)length); + for (Py_ssize_t i = 0; i <= comment_count; i++) { + _PyTok_Span span = { + copied_to, + i < comment_count ? comments->spans[i].start : state->expr_span.end, + }; + Py_ssize_t length; + const char *text = _PyTok_SourceSpanView(&tok->source, span, &length); + memcpy(stripped + stripped_len, text, (size_t)length); stripped_len += length; - copied_to = comment.end; + if (i < comment_count) { + copied_to = comments->spans[i].end; + } } - Py_ssize_t length = state->expr_span.end - copied_to; - memcpy(stripped + stripped_len, - expr + copied_to - state->expr_span.start, - (size_t)length); - stripped_len += length; res = PyUnicode_DecodeUTF8(stripped, stripped_len, NULL); PyMem_Free(stripped); } else { + Py_ssize_t expr_len; + const char *expr = _PyTok_SourceSpanView( + &tok->source, state->expr_span, &expr_len); res = PyUnicode_DecodeUTF8(expr, expr_len, NULL); } diff --git a/Parser/tokenizer/api.c b/Parser/tokenizer/api.c index 5e55d14f67317fd..c774b96e8a1b2b6 100644 --- a/Parser/tokenizer/api.c +++ b/Parser/tokenizer/api.c @@ -107,22 +107,28 @@ _PyTokenizer_ImplyDedents(struct tok_state *tok) int _PyTokenizer_HasTrailingStatement(const struct tok_state *tok) { - const char *cur = _PyTok_SourcePointer(&tok->source, tok->cur); - char c = *cur; - for (;;) { - while (c == ' ' || c == '\t' || c == '\n' || c == '\014') { - c = *++cur; - } - if (!c) { + _PyTok_Off cur = tok->cur; + _PyTok_Off end = tok->source.base_offset + tok->source.len; + while (cur < end) { + int c = _PyTok_SourceByte(&tok->source, cur++); + if (c == '\0') { return 0; } + if (c == ' ' || c == '\t' || c == '\n' || c == '\014') { + continue; + } if (c != '#') { return 1; } - while (c && c != '\n') { - c = *++cur; + while (cur < end) { + c = _PyTok_SourceByte(&tok->source, cur); + if (c == '\0' || c == '\n') { + break; + } + cur++; } } + return 0; } int diff --git a/Parser/tokenizer/reader.c b/Parser/tokenizer/reader.c index 2f426d7c26a60e5..df3c7184e1ba33b 100644 --- a/Parser/tokenizer/reader.c +++ b/Parser/tokenizer/reader.c @@ -175,13 +175,11 @@ next_prepared(struct tok_state *tok, _PyTok_Chunk *chunk) return _PYTOK_READ_EOF; } _PyTok_Span tail = {tok->inp, tok->source.base_offset + tok->source.len}; - Py_ssize_t remaining; - const char *start = _PyTok_SourceSpanView(&tok->source, tail, &remaining); - const char *newline = memchr(start, '\n', remaining); - chunk->data = (char *)start; - chunk->len = newline != NULL ? newline - start + 1 : remaining; + _PyTok_Off newline = _PyTok_SourceFindByte(&tok->source, tail, '\n'); + _PyTok_Span line = {tail.start, newline >= 0 ? newline + 1 : tail.end}; + chunk->data = (char *)_PyTok_SourceSpanView(&tok->source, line, &chunk->len); chunk->ownership = _PYTOK_CHUNK_BORROWED; - chunk->implicit_newline = chunk->len == remaining && + chunk->implicit_newline = line.end == tail.end && tok->reader->prepared_final_newline_is_implicit; return _PYTOK_READ_LINE; } @@ -667,11 +665,10 @@ _PyTok_ReaderUnderflow(struct tok_state *tok) tok->inp = source_start + chunk.len; } else { - _PyTok_Off source_start = _PyTok_SourceOffset(&tok->source, chunk.data); if (tok->start < 0 && _PyLexer_CurrentFTString(tok) == NULL) { - tok->buf_offset = source_start; + tok->buf_offset = tok->inp; } - tok->inp = source_start + chunk.len; + tok->inp += chunk.len; } tok->implicit_newline = chunk.implicit_newline; diff --git a/Parser/tokenizer/source.h b/Parser/tokenizer/source.h index e1aee720152b7f7..6ee98077face8f6 100644 --- a/Parser/tokenizer/source.h +++ b/Parser/tokenizer/source.h @@ -21,14 +21,6 @@ _PyTok_SourceData(const _PyTok_SourceText *source) /* Convert positions within the retained source window. Pointers and views are borrowed; append, discard, and clear invalidate them. */ -static inline _PyTok_Off -_PyTok_SourceOffset(const _PyTok_SourceText *source, const char *position) -{ - const char *base = _PyTok_SourceData(source); - assert(position >= base && position <= base + source->len); - return source->base_offset + (position - base); -} - static inline const char * _PyTok_SourcePointer(const _PyTok_SourceText *source, _PyTok_Off offset) { @@ -56,6 +48,16 @@ _PyTok_SourceSpanView(const _PyTok_SourceText *source, _PyTok_Span span, return _PyTok_SourcePointer(source, span.start); } +/* Return the first matching offset within span, or -1 if absent. */ +static inline _PyTok_Off +_PyTok_SourceFindByte(const _PyTok_SourceText *source, _PyTok_Span span, int byte) +{ + Py_ssize_t length; + const char *data = _PyTok_SourceSpanView(source, span, &length); + const char *found = memchr(data, byte, length); + return found != NULL ? span.start + (found - data) : -1; +} + PyAPI_FUNC(void) _PyTok_SourceInit(_PyTok_SourceText *); /* Clear invalidates all spans and views for the source. */ PyAPI_FUNC(void) _PyTok_SourceClear(_PyTok_SourceText *); From ffb781d20a91aea10b7542fddff0475ddf0c8eef Mon Sep 17 00:00:00 2001 From: Pablo Galindo Salgado Date: Mon, 5 Oct 2026 16:51:11 +0100 Subject: [PATCH 5/6] gh-153569: Remove redundant tokenizer views and simplify reader branching --- Parser/lexer/string.c | 25 +++++++++++++------------ Parser/tokenizer/api.c | 1 - Parser/tokenizer/reader.c | 14 ++++++++------ Parser/tokenizer/source.h | 5 +++-- Parser/tokenizer/tokenizer.h | 5 ++--- Python/Python-tokenize.c | 17 ++++++++--------- 6 files changed, 34 insertions(+), 33 deletions(-) diff --git a/Parser/lexer/string.c b/Parser/lexer/string.c index ea245042b2ee3dc..d0f568dfa857580 100644 --- a/Parser/lexer/string.c +++ b/Parser/lexer/string.c @@ -79,7 +79,8 @@ finish_ftstring_expr(struct tok_state *tok, ftstring_state *state, tokenizer_comments *comments = state->comments; PyObject *res; if (comments != NULL && comments->count > 0) { - Py_ssize_t stripped_size = state->expr_span.end - state->expr_span.start; + Py_ssize_t stripped_size = + state->expr_span.end - state->expr_span.start; Py_ssize_t comment_count = 0; for (Py_ssize_t i = 0; i < comments->count; i++) { _PyTok_Span comment = comments->spans[i]; @@ -98,22 +99,22 @@ finish_ftstring_expr(struct tok_state *tok, ftstring_state *state, PyErr_NoMemory(); return -1; } - _PyTok_Off copied_to = state->expr_span.start; + _PyTok_Span kept = {state->expr_span.start, state->expr_span.start}; Py_ssize_t stripped_len = 0; - for (Py_ssize_t i = 0; i <= comment_count; i++) { - _PyTok_Span span = { - copied_to, - i < comment_count ? comments->spans[i].start : state->expr_span.end, - }; + for (Py_ssize_t i = 0; i < comment_count; i++) { + kept.end = comments->spans[i].start; Py_ssize_t length; - const char *text = _PyTok_SourceSpanView(&tok->source, span, &length); + const char *text = _PyTok_SourceSpanView( + &tok->source, kept, &length); memcpy(stripped + stripped_len, text, (size_t)length); stripped_len += length; - if (i < comment_count) { - copied_to = comments->spans[i].end; - } + kept.start = comments->spans[i].end; } - res = PyUnicode_DecodeUTF8(stripped, stripped_len, NULL); + kept.end = state->expr_span.end; + Py_ssize_t length; + const char *text = _PyTok_SourceSpanView(&tok->source, kept, &length); + memcpy(stripped + stripped_len, text, (size_t)length); + res = PyUnicode_DecodeUTF8(stripped, stripped_size, NULL); PyMem_Free(stripped); } else { diff --git a/Parser/tokenizer/api.c b/Parser/tokenizer/api.c index c774b96e8a1b2b6..d955c6b335e1db6 100644 --- a/Parser/tokenizer/api.c +++ b/Parser/tokenizer/api.c @@ -68,7 +68,6 @@ _PyToken_GetView(const struct tok_state *tok, const struct token *token, ? token->span.start - token->start_loc.byte_col : tok->line_start, tok->inp, }; - view->line = _PyTok_SourcePointer(&tok->source, view->line_span.start); view->implicit_newline = tok->implicit_newline; view->at_eof = tok->done == E_EOF; } diff --git a/Parser/tokenizer/reader.c b/Parser/tokenizer/reader.c index df3c7184e1ba33b..ce2bb1cf5c4bc30 100644 --- a/Parser/tokenizer/reader.c +++ b/Parser/tokenizer/reader.c @@ -177,7 +177,8 @@ next_prepared(struct tok_state *tok, _PyTok_Chunk *chunk) _PyTok_Span tail = {tok->inp, tok->source.base_offset + tok->source.len}; _PyTok_Off newline = _PyTok_SourceFindByte(&tok->source, tail, '\n'); _PyTok_Span line = {tail.start, newline >= 0 ? newline + 1 : tail.end}; - chunk->data = (char *)_PyTok_SourceSpanView(&tok->source, line, &chunk->len); + chunk->data = (char *)_PyTok_SourceSpanView( + &tok->source, line, &chunk->len); chunk->ownership = _PYTOK_CHUNK_BORROWED; chunk->implicit_newline = line.end == tail.end && tok->reader->prepared_final_newline_is_implicit; @@ -607,7 +608,7 @@ _PyTok_ReaderUnderflow(struct tok_state *tok) _PyTok_ReaderKind kind = reader->kind; int prepared = kind == _PYTOK_READER_PREPARED; int streaming = reader_is_streaming(kind); - int reset_buffer = !prepared && tok->start < 0 && + int reset_buffer = tok->start < 0 && _PyLexer_CurrentFTString(tok) == NULL; _PyTok_Chunk chunk; @@ -660,12 +661,11 @@ _PyTok_ReaderUnderflow(struct tok_state *tok) tok->cur = source_start; tok->buf_offset = source_start; tok->line_start = tok->buf_offset; - tok->start = -1; } tok->inp = source_start + chunk.len; } else { - if (tok->start < 0 && _PyLexer_CurrentFTString(tok) == NULL) { + if (reset_buffer) { tok->buf_offset = tok->inp; } tok->inp += chunk.len; @@ -674,9 +674,11 @@ _PyTok_ReaderUnderflow(struct tok_state *tok) tok->lineno++; if (kind == _PYTOK_READER_FILE && - (reader->encoding == NULL || strcmp(reader->encoding, "utf-8") == 0) && + (reader->encoding == NULL || + strcmp(reader->encoding, "utf-8") == 0) && !_PyTokenizer_ensure_utf8( - _PyTok_SourcePointer(&tok->source, tok->cur), tok, tok->lineno)) { + _PyTok_SourcePointer(&tok->source, tok->cur), + tok, tok->lineno)) { _PyTok_ChunkClear(&chunk); return 0; } diff --git a/Parser/tokenizer/source.h b/Parser/tokenizer/source.h index 6ee98077face8f6..0fd9fe1877bed2b 100644 --- a/Parser/tokenizer/source.h +++ b/Parser/tokenizer/source.h @@ -43,14 +43,15 @@ _PyTok_SourceSpanView(const _PyTok_SourceText *source, _PyTok_Span span, { assert(length != NULL); assert(_PyTok_SpanIsValid(span)); + assert(span.end - source->base_offset <= source->len); *length = span.end - span.start; - (void)_PyTok_SourcePointer(source, span.end); return _PyTok_SourcePointer(source, span.start); } /* Return the first matching offset within span, or -1 if absent. */ static inline _PyTok_Off -_PyTok_SourceFindByte(const _PyTok_SourceText *source, _PyTok_Span span, int byte) +_PyTok_SourceFindByte(const _PyTok_SourceText *source, _PyTok_Span span, + int byte) { Py_ssize_t length; const char *data = _PyTok_SourceSpanView(source, span, &length); diff --git a/Parser/tokenizer/tokenizer.h b/Parser/tokenizer/tokenizer.h index 39d68fa4010571d..bc4009700fdc967 100644 --- a/Parser/tokenizer/tokenizer.h +++ b/Parser/tokenizer/tokenizer.h @@ -21,7 +21,6 @@ struct token { typedef struct { const char *text; Py_ssize_t length; - const char *line; _PyTok_Span line_span; _PyTok_Off end_line_start; int implicit_newline; @@ -74,8 +73,8 @@ _PyTokenizer_Info _PyTokenizer_GetInfo(const struct tok_state *); const char *_PyToken_TextView( const struct tok_state *, const struct token *, Py_ssize_t *); /* Use the token from the most recent Get. An absent span has nonnull empty text. - line contains the bytes of line_span, the token's complete physical line - range. end_line_start is the logical offset of its final physical line. */ + line_span covers the token's complete physical line range. end_line_start + is the logical offset of its final physical line. */ void _PyToken_GetView( const struct tok_state *tok, const struct token *token, _PyToken_View *view); diff --git a/Python/Python-tokenize.c b/Python/Python-tokenize.c index ad70ef514d6a25b..731444e13a02750 100644 --- a/Python/Python-tokenize.c +++ b/Python/Python-tokenize.c @@ -158,12 +158,16 @@ _tokenizer_error(tokenizeriterobject *it) static PyObject * _get_current_line(tokenizeriterobject *it, int current_lineno, - const char *line_start, Py_ssize_t size, int *line_changed) + const _PyToken_View *view, int *line_changed) { _Py_CRITICAL_SECTION_ASSERT_OBJECT_LOCKED(it); if (current_lineno != it->last_lineno) { - // Line has changed since last token, so we fetch the new line and cache it - // in the iter object. + Py_ssize_t size; + const char *line_start = _PyTokenizer_SpanView( + it->tok, view->line_span, &size); + if (size > 0 && view->implicit_newline) { + size--; + } Py_XDECREF(it->last_line); it->last_line = PyUnicode_DecodeUTF8(line_start, size, "replace"); it->byte_col_offset_diff = 0; @@ -263,13 +267,8 @@ tokenizeriter_next(PyObject *op) if (it->extra_tokens && is_trailing_token) { line = Py_GetConstant(Py_CONSTANT_EMPTY_STR); } else { - Py_ssize_t size = view.line_span.end - view.line_span.start; - if (size >= 1 && view.implicit_newline) { - size -= 1; - } - line = _get_current_line( - it, token.end_loc.lineno, view.line, size, &line_changed); + it, token.end_loc.lineno, &view, &line_changed); } if (line == NULL) { Py_DECREF(str); From 33d219cabc8ad50f1a78d396b4f69d924caba304 Mon Sep 17 00:00:00 2001 From: Pablo Galindo Salgado Date: Mon, 5 Oct 2026 17:54:26 +0100 Subject: [PATCH 6/6] gh-153569: Reuse the incremental decoder factory helper --- Parser/tokenizer/decoder.c | 6 +----- 1 file changed, 1 insertion(+), 5 deletions(-) diff --git a/Parser/tokenizer/decoder.c b/Parser/tokenizer/decoder.c index 8790f4b91a403bd..48f906845651aab 100644 --- a/Parser/tokenizer/decoder.c +++ b/Parser/tokenizer/decoder.c @@ -429,12 +429,8 @@ _PyTok_StartDecoder(struct tok_state *tok, const char *errors) PyObject *codec = _PyCodec_LookupTextEncoding(reader->encoding, NULL); if (codec != NULL) { - PyObject *factory = PyObject_GetAttrString(codec, "incrementaldecoder"); + reader->decoder = _PyCodecInfo_GetIncrementalDecoder(codec, errors); Py_DECREF(codec); - if (factory != NULL) { - reader->decoder = PyObject_CallFunction(factory, "s", errors); - Py_DECREF(factory); - } } if (reader->decoder == NULL) { tok->done = PyErr_ExceptionMatches(PyExc_MemoryError)