FazBrowse GitHub Viewer | Trending |
URL:
| Home
Tools: [Download Repo ZIP]   [Original HTTPS Page]

Match CPython's SyntaxError range for unparenthesized `except` types by zzarbttoo · Pull Request #8661 · RustPython/RustPython · GitHub

Repository navigation

Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension .py  (1) .rs  (1) All 2 file types selected
Viewed files
Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Unified
Split
Hide whitespace
Diff view
Unified
Split
Hide whitespace
92 changes: 92 additions & 0 deletions crates/vm/src/vm/vm_new.rs
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters. Learn more about bidirectional Unicode characters
Original file line number Diff line number Diff line change
Expand Up @@ -952,6 +952,26 @@ impl VirtualMachine {
}
_ => false,
};
// CPython reports the unparenthesized exception types error with a range stopping
// just before the `:` that closes the `except` clause, which covers the `as NAME`
// part, while the parser reports the exception types alone. The exclusive end that
// yields is the column of the `:`. See `invalid_except_stmt_end`.
let except_as_end = cfg_select! {
feature = "parser" => {
if msg == "multiple exception types must be parenthesized when using 'as'"
&& let crate::compiler::CompileError::Parse(rustpython_compiler::ParseError {
raw_location,
..
}) = error
&& let Some(source) = source
{
invalid_except_stmt_end(source, raw_location.end().to_usize())
} else {
None
}
}
_ => None,
};

let syntax_error = self.new_exception_msg(syntax_error_type, msg.into());

Expand Down Expand Up @@ -984,6 +1004,8 @@ impl VirtualMachine {
} else if narrow_caret {
let (l, o) = error.python_location();
(l, (o + 1) as isize)
} else if let Some((l, o)) = except_as_end {
(l, o as isize)
} else {
(end_lineno, end_offset as isize)
};
Comment thread
coderabbitai[bot] marked this conversation as resolved.
Expand Down Expand Up @@ -1263,3 +1285,73 @@ fn scan_quoted_string_for_incomplete(bytes: &[u8], quote_index: usize) -> Quoted
unescaped_newline: false,
}
}

/// Returns the exclusive end of the range CPython reports for its `invalid_except_stmt`
/// rule, as a 1-based `(line, column)` pair whose column counts characters, not bytes.
/// Being exclusive, that column is the one the `:` closing the `except` clause sits on:
/// the `:` itself is not part of the range.
///
/// CPython raises that error only once the whole clause has matched, and reports a range
/// starting at the first exception type and stopping just before the `:`, so the range
/// covers the `as NAME` part as well:
///
/// ```text
/// except A, B as e:
/// ^^^^^^^^^ offset = 8, end_offset = 17
/// ```
///
/// ```text
/// invalid_except_stmt:
/// | 'except' a=expression ',' expressions 'as' NAME ':' {
/// RAISE_SYNTAX_ERROR_STARTING_FROM(a, "multiple exception types must be parenthesized when using 'as'") }
/// ```
///
/// The parser reports the exception types alone, so the `:` is looked up here. Only
/// `as NAME` can follow the exception types, which is why the first `:` after them is the
/// one closing the clause. `types_end` is a byte offset into `source`.
///
/// Returns `None` when the clause has no `:`, in which case CPython reports a different
/// error and the range is left alone.
#[cfg(feature = "parser")]
fn invalid_except_stmt_end(source: &str, types_end: usize) -> Option<(usize, usize)> {
let bytes = source.as_bytes();
let mut index = types_end;

let colon = loop {
match *bytes.get(index)? {
b':' => break index,
// An explicit line join continues the clause on the next line.
b'\\' => {
index += 1;
if bytes.get(index) == Some(&b'\r') {
index += 1;
}
if bytes.get(index) != Some(&b'\n') {
return None;
}
index += 1;
}
b'\n' | b'\r' | b'#' => return None,
_ => index += 1,
}
};

// Only `as NAME` may appear between the exception types and the `:`. CPython reports a
// different error when the name is missing, so leave the range alone in that case.
// An explicit line join may sit anywhere in between, so drop the backslashes and let
// the newline they escape count as the ordinary whitespace around `as`.
let between = source.get(types_end..colon)?.replace('\\', " ");
let name = between.trim().strip_prefix("as")?;
if !name.starts_with(char::is_whitespace) || name.trim().is_empty() {
return None;
}

let before = source.get(..colon)?;
let line = before.bytes().filter(|&byte| byte == b'\n').count() + 1;
let line_start = before.rfind('\n').map_or(0, |index| index + 1);
// `line_start` and `colon` are byte offsets, but the column counts characters, so a
// non-ASCII exception type or `as NAME` would otherwise push the column too far right.
let column = source.get(line_start..colon)?.chars().count() + 1;

Some((line, column))
}
43 changes: 43 additions & 0 deletions extra_tests/snippets/syntax_invalid.py
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters. Learn more about bidirectional Unicode characters
Original file line number Diff line number Diff line change
Expand Up @@ -77,3 +77,46 @@ def valid_func():
from __future__ import print_function
"""
compile(src, 'test.py', 'exec')


# CPython reports the unparenthesized `except ... as` error with a range that starts at the
# first exception type and stops just before the `:` closing the clause. `end_offset` is an
# exclusive 1-based *character* column, so that end lands on the column the `:` sits on.
#
# The cases below pin the parts that are easy to get wrong: the column counts characters
# rather than UTF-8 bytes, and an explicit line join may sit anywhere between the exception
# types and the `:`. Values verified against CPython 3.14.
except_as_ranges = [
# (clause, (lineno, offset), (end_lineno, end_offset))
("except A, B as e:", (3, 8), (3, 17)),
("except A, B as e :", (3, 8), (3, 18)),
("except A, B as e :", (3, 8), (3, 20)),
("except A, B as e\t:", (3, 8), (3, 18)),
("except A, B, C as blech:", (3, 8), (3, 24)),
("except* A, B, C as e:", (3, 9), (3, 21)),
# A non-ASCII name must not push the column right by its extra UTF-8 bytes.
("except Ä, B as e:", (3, 8), (3, 17)),
("except 사과, B as e:", (3, 8), (3, 18)),
("except A, B as 오:", (3, 8), (3, 17)),
# An explicit line join may precede or follow `as`, or close the clause on its own line.
("except A, B as exc\\\n :", (3, 8), (4, 5)),
("except A, B \\\nas exc:", (3, 8), (4, 7)),
("except A, \\\nB as exc:", (3, 8), (4, 9)),
("except A, B as \\\nexc:", (3, 8), (4, 4)),
]

for clause, start, end in except_as_ranges:
src = "try:\n pass\n%s\n pass\n" % clause
with assert_raises(SyntaxError) as ae:
compile(src, "test.py", "exec")
exc = ae.exception
assert exc.msg == "multiple exception types must be parenthesized when using 'as'", (
clause,
exc.msg,
)
assert (exc.lineno, exc.offset) == start, (clause, (exc.lineno, exc.offset), start)
assert (exc.end_lineno, exc.end_offset) == end, (
clause,
(exc.end_lineno, exc.end_offset),
end,
)
Loading

Back | FazBrowse Home | New Git URL