Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
92 changes: 36 additions & 56 deletions Parser/lexer/lexer.c
Original file line number Diff line number Diff line change
Expand Up @@ -13,12 +13,6 @@
tokenizing. */
static const char* type_comment_prefix = "# type: ";

static inline int
contains_null_bytes(const char* str, size_t size)
{
return memchr(str, 0, size) != NULL;
}

int
_PyLexer_refill(struct tok_state *tok)
{
Expand All @@ -29,8 +23,9 @@ _PyLexer_refill(struct tok_state *tok)
#if defined(Py_DEBUG)
if (tok->debug) {
fprintf(stderr, "line[%d] = ", tok->lineno);
_PyTokenizer_print_escape(stderr, _PyLexer_BufferPointer(tok, tok->cur),
tok->inp - tok->cur);
_PyTokenizer_print_escape(
stderr, _PyTok_SourcePointer(&tok->source, tok->cur),
tok->inp - tok->cur);
fprintf(stderr, " tok->done = %d\n", tok->done);
}
#endif
Expand All @@ -39,8 +34,8 @@ _PyLexer_refill(struct tok_state *tok)
return 0;
}
tok->line_start = tok->cur;
if (contains_null_bytes(_PyLexer_BufferPointer(tok, tok->line_start),
tok->inp - tok->line_start)) {
_PyTok_Span line = {tok->line_start, tok->inp};
if (_PyTok_SourceFindByte(&tok->source, line, '\0') >= 0) {
_PyTokenizer_syntaxerror(tok, "source code cannot contain null bytes");
tok->cur = tok->inp;
return 0;
Expand All @@ -56,7 +51,7 @@ _PyLexer_backup(struct tok_state *tok, int c)
if (--tok->cur < tok->buf_offset) {
Py_FatalError("tokenizer beginning of buffer");
}
if ((int)(unsigned char)*_PyLexer_BufferPointer(tok, tok->cur) != Py_CHARMASK(c)) {
if (_PyTok_SourceByte(&tok->source, tok->cur) != Py_CHARMASK(c)) {
Py_FatalError("tok_backup: wrong character");
}
}
Expand All @@ -74,7 +69,8 @@ verify_identifier(struct tok_state *tok)
PyObject *s;
if (tok_failed(tok))
return 0;
s = PyUnicode_DecodeUTF8(_PyLexer_BufferPointer(tok, tok->start), tok->cur - tok->start, NULL);
s = PyUnicode_DecodeUTF8(_PyTok_SourcePointer(&tok->source, tok->start),
tok->cur - tok->start, NULL);
if (s == NULL) {
if (PyErr_ExceptionMatches(PyExc_UnicodeDecodeError)) {
tok->done = E_DECODE;
Expand Down Expand Up @@ -105,13 +101,13 @@ verify_identifier(struct tok_state *tok)
Py_DECREF(s);
if (Py_UNICODE_ISPRINTABLE(ch)) {
_PyTokenizer_syntaxerror_at(
tok, _PyLexer_BufferPointer(tok, tok->line_start),
tok, _PyTok_SourcePointer(&tok->source, tok->line_start),
error_cursor - tok->line_start, tok->lineno, -1, -1,
"invalid character '%c' (U+%04X)", ch, ch);
}
else {
_PyTokenizer_syntaxerror_at(
tok, _PyLexer_BufferPointer(tok, tok->line_start),
tok, _PyTok_SourcePointer(&tok->source, tok->line_start),
error_cursor - tok->line_start, tok->lineno, -1, -1,
"invalid non-printable character U+%04X", ch);
}
Expand Down Expand Up @@ -172,10 +168,6 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token
/* Skip comment, unless it's a type comment */
if (c == '#') {

const char* p = NULL;
const char *prefix, *type_start;
int current_starting_col_offset;

while (c != EOF && c != '\n' && c != '\r') {
c = tok_nextc(tok);
}
Expand All @@ -192,23 +184,20 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token
}
}

if (tok->tok_extra_tokens) {
p = _PyLexer_BufferPointer(tok, tok->start);
}

_PyTok_Off comment = tok->start;
if (tok->type_comments) {
p = _PyLexer_BufferPointer(tok, tok->start);
current_starting_col_offset = tok->start_loc.byte_col;
prefix = type_comment_prefix;
while (*prefix && p < _PyLexer_BufferPointer(tok, tok->cur)) {
const char *prefix = type_comment_prefix;
while (*prefix && comment < tok->cur) {
if (*prefix == ' ') {
while (*p == ' ' || *p == '\t') {
p++;
current_starting_col_offset++;
while (comment < tok->cur) {
int ch = _PyTok_SourceByte(&tok->source, comment);
if (ch != ' ' && ch != '\t') {
break;
}
comment++;
}
} else if (*prefix == *p) {
p++;
current_starting_col_offset++;
} else if (*prefix == _PyTok_SourceByte(&tok->source, comment)) {
comment++;
} else {
break;
}
Expand All @@ -218,35 +207,26 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token

/* This is a type comment if we matched all of type_comment_prefix. */
if (!*prefix) {
int is_type_ignore = 1;
// +6 in order to skip the word 'ignore'
const char *ignore_end = p + 6;
const int ignore_end_col_offset = current_starting_col_offset + 6;
tok_backup(tok, c); /* don't eat the newline or EOF */

type_start = p;

/* A TYPE_IGNORE is "type: ignore" followed by the end of the token
* or anything ASCII and non-alphanumeric. */
is_type_ignore = (
_PyLexer_BufferPointer(tok, tok->cur) >= ignore_end && memcmp(p, "ignore", 6) == 0
&& !(_PyLexer_BufferPointer(tok, tok->cur) > ignore_end
&& ((unsigned char)ignore_end[0] >= 128 || Py_ISALNUM(ignore_end[0]))));
int is_type_ignore = tok->cur - comment >= 6 &&
memcmp(_PyTok_SourcePointer(&tok->source, comment),
"ignore", 6) == 0;
if (is_type_ignore && comment + 6 < tok->cur) {
int ch = _PyTok_SourceByte(&tok->source, comment + 6);
is_type_ignore = ch < 128 && !Py_ISALNUM(ch);
}

int type = is_type_ignore ? TYPE_IGNORE : TYPE_COMMENT;
int start_col_offset = is_type_ignore
? ignore_end_col_offset : current_starting_col_offset;
p_start = comment + (is_type_ignore ? 6 : 0);
p_end = tok->cur;
if (is_type_ignore) {
p_start = _PyLexer_BufferOffset(tok, ignore_end);

/* If this type ignore is the only thing on the line, consume the newline also. */
if (blankline) {
tok_nextc(tok);
tok->layout.at_bol = 1;
}
} else {
p_start = _PyLexer_BufferOffset(tok, type_start);
int start_col_offset = tok->start_loc.byte_col +
(int)(p_start - tok->start);
/* Consume the newline after a standalone type ignore. */
if (is_type_ignore && blankline) {
tok_nextc(tok);
tok->layout.at_bol = 1;
}
_PyLexer_token_setup(tok, token, type, p_start, p_end);
token->start_loc = (_PyTok_Loc){tok->lineno, start_col_offset};
Expand All @@ -257,7 +237,7 @@ _PyLexer_get_normal(struct tok_state *tok, ftstring_state *current, struct token
}
if (tok->tok_extra_tokens) {
tok_backup(tok, c); /* don't eat the newline or EOF */
p_start = _PyLexer_BufferOffset(tok, p);
p_start = comment;
p_end = tok->cur;
tok->layout.comment_newline = blankline;
return MAKE_TOKEN(COMMENT);
Expand Down
5 changes: 1 addition & 4 deletions Parser/lexer/lexer_internal.h
Original file line number Diff line number Diff line change
Expand Up @@ -40,14 +40,11 @@ tok_nextc(struct tok_state *tok)
return EOF;
}
}
assert(tok->cur >= tok->source.base_offset);
assert(tok->cur - tok->source.base_offset < tok->source.len);
if (tok->cur - tok->line_start >= INT_MAX) {
tok->done = E_COLUMNOVERFLOW;
return EOF;
}
return Py_CHARMASK(
tok->source.bytes[tok->cur++ - tok->source.base_offset]);
return _PyTok_SourceByte(&tok->source, tok->cur++);
}

/* Return -1 on error, otherwise whether the line is blank. */
Expand Down
3 changes: 0 additions & 3 deletions Parser/lexer/state.c
Original file line number Diff line number Diff line change
Expand Up @@ -58,9 +58,6 @@ _PyLexer_PopFTString(struct tok_state *tok)
void
_PyTokenizer_Free(struct tok_state *tok)
{
if (tok->encoding != NULL) {
PyMem_Free(tok->encoding);
}
Py_XDECREF(tok->filename);
Py_XDECREF(tok->module);
_PyTok_ReaderFree(tok);
Expand Down
31 changes: 0 additions & 31 deletions Parser/lexer/state.h
Original file line number Diff line number Diff line change
Expand Up @@ -80,7 +80,6 @@ struct tok_state {
_PyTok_SourceText source;
int done; /* E_OK normally, E_EOF at EOF, otherwise error code */
/* NB If done != E_OK, cur must be == inp!!! */
FILE *fp; /* Rest of input; NULL if tokenizing a string */
lexer_layout_state layout;
int lineno; /* Current line number */
_PyTok_Loc start_loc;
Expand All @@ -92,9 +91,6 @@ struct tok_state {
int parencolstack[MAXLEVEL];
PyObject *filename;
PyObject *module;
/* Stuff for PEP 0263 */
char *encoding; /* Source encoding. */

struct _PyTok_Reader *reader;

int type_comments; /* Whether to look for type comments */
Expand Down Expand Up @@ -135,33 +131,6 @@ _PyLexer_FTStringBracketDepth(const struct tok_state *tok,
return tok->level - state->paren_level;
}

static inline _PyTok_Off
_PyLexer_BufferOffset(const struct tok_state *tok, const char *position)
{
const char *base = _PyTok_SourceData(&tok->source);
assert(position >= base && position <= base + tok->source.len);
return tok->source.base_offset + (position - base);
}

static inline const char *
_PyLexer_BufferPointer(const struct tok_state *tok, _PyTok_Off offset)
{
assert(offset >= tok->source.base_offset);
assert(offset - tok->source.base_offset <= tok->source.len);
return _PyTok_SourceData(&tok->source) + (offset - tok->source.base_offset);
}

static inline const char *
_PyLexer_BufferSpanView(const struct tok_state *tok, _PyTok_Span span,
Py_ssize_t *length)
{
assert(length != NULL);
assert(_PyTok_SpanIsValid(span));
*length = span.end - span.start;
(void)_PyLexer_BufferPointer(tok, span.end);
return _PyLexer_BufferPointer(tok, span.start);
}

static inline int
_PyLexer_ByteColumn(const struct tok_state *tok)
{
Expand Down
40 changes: 21 additions & 19 deletions Parser/lexer/string.c
Original file line number Diff line number Diff line change
Expand Up @@ -76,13 +76,11 @@ finish_ftstring_expr(struct tok_state *tok, ftstring_state *state,
if (!(state->debug_expr || tstring_interpolation) || token->metadata) {
return 0;
}
Py_ssize_t expr_len;
const char *expr = _PyLexer_BufferSpanView(
tok, state->expr_span, &expr_len);
tokenizer_comments *comments = state->comments;
PyObject *res;
if (comments != NULL && comments->count > 0) {
Py_ssize_t stripped_size = expr_len;
Py_ssize_t stripped_size =
state->expr_span.end - state->expr_span.start;
Py_ssize_t comment_count = 0;
for (Py_ssize_t i = 0; i < comments->count; i++) {
_PyTok_Span comment = comments->spans[i];
Expand All @@ -101,26 +99,28 @@ finish_ftstring_expr(struct tok_state *tok, ftstring_state *state,
PyErr_NoMemory();
return -1;
}
_PyTok_Off copied_to = state->expr_span.start;
_PyTok_Span kept = {state->expr_span.start, state->expr_span.start};
Py_ssize_t stripped_len = 0;
for (Py_ssize_t i = 0; i < comment_count; i++) {
_PyTok_Span comment = comments->spans[i];
Py_ssize_t length = comment.start - copied_to;
memcpy(stripped + stripped_len,
expr + copied_to - state->expr_span.start,
(size_t)length);
kept.end = comments->spans[i].start;
Py_ssize_t length;
const char *text = _PyTok_SourceSpanView(
&tok->source, kept, &length);
memcpy(stripped + stripped_len, text, (size_t)length);
stripped_len += length;
copied_to = comment.end;
kept.start = comments->spans[i].end;
}
Py_ssize_t length = state->expr_span.end - copied_to;
memcpy(stripped + stripped_len,
expr + copied_to - state->expr_span.start,
(size_t)length);
stripped_len += length;
res = PyUnicode_DecodeUTF8(stripped, stripped_len, NULL);
kept.end = state->expr_span.end;
Py_ssize_t length;
const char *text = _PyTok_SourceSpanView(&tok->source, kept, &length);
memcpy(stripped + stripped_len, text, (size_t)length);
res = PyUnicode_DecodeUTF8(stripped, stripped_size, NULL);
PyMem_Free(stripped);
}
else {
Py_ssize_t expr_len;
const char *expr = _PyTok_SourceSpanView(
&tok->source, state->expr_span, &expr_len);
res = PyUnicode_DecodeUTF8(expr, expr_len, NULL);
}

Expand Down Expand Up @@ -339,7 +339,8 @@ _PyLexer_scan_string(struct tok_state *tok, struct token *token, int c)
}
int end_lineno = tok->lineno;
_PyTok_Loc location = tok->start_loc;
const char *line = _PyLexer_BufferPointer(tok, tok->start) - location.byte_col;
const char *line = _PyTok_SourcePointer(
&tok->source, tok->start - location.byte_col);
Py_ssize_t cursor_offset = (Py_ssize_t)location.byte_col + 1;

const ftstring_state *state = _PyLexer_CurrentFTString(tok);
Expand Down Expand Up @@ -460,7 +461,8 @@ _PyLexer_get_ftstring(struct tok_state *tok, ftstring_state *current, struct tok

int end_lineno = tok->lineno;
_PyTok_Loc location = current->start_loc;
const char *line = _PyLexer_BufferPointer(tok, current->start) - location.byte_col;
const char *line = _PyTok_SourcePointer(
&tok->source, current->start - location.byte_col);
Py_ssize_t cursor_offset = (Py_ssize_t)location.byte_col + 1;

if (quote_size == 3) {
Expand Down
Loading
Loading