diff --git a/CHANGELOG.md b/CHANGELOG.md index 35eb182c006a..84f3b32ff75d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -29,6 +29,9 @@ **Fixes**: +- Error labels after non-ASCII text now point at the right column, rather than + drifting right or crashing `prqlc` when a label ran past the end of the query. + (@prql-bot, #6378) - `date.to_text` errors now name the format specifier the target dialect rejected — `format specifier %P is not supported for Postgres` rather than `PRQL doesn't support this format specifier`, whose span covers the whole diff --git a/prqlc/prqlc-parser/src/lexer/mod.rs b/prqlc/prqlc-parser/src/lexer/mod.rs index 497d236318db..d59e950cc4fc 100644 --- a/prqlc/prqlc-parser/src/lexer/mod.rs +++ b/prqlc/prqlc-parser/src/lexer/mod.rs @@ -68,7 +68,7 @@ pub fn lex_source_recovery(source: &str, source_id: u16) -> (Option>, let result = lexer().parse(source).into_result(); match result { - Ok(tokens) => (Some(insert_start(tokens.to_vec())), vec![]), + Ok(tokens) => (Some(insert_start(to_char_spans(source, tokens))), vec![]), Err(errors) => { // Convert chumsky Simple errors to our Error type let errors = errors @@ -86,7 +86,7 @@ pub fn lex_source(source: &str) -> Result> { let result = lexer().parse(source).into_result(); match result { - Ok(tokens) => Ok(Tokens(insert_start(tokens.to_vec()))), + Ok(tokens) => Ok(Tokens(insert_start(to_char_spans(source, tokens)))), Err(errors) => { // Convert chumsky Simple errors to our Error type let errors = errors @@ -99,6 +99,27 @@ pub fn lex_source(source: &str) -> Result> { } } +/// Convert token spans from the byte offsets chumsky produces over `&str` to +/// the char offsets the rest of the compiler (and ariadne) expects. +fn to_char_spans(source: &str, mut tokens: Vec) -> Vec { + if source.is_ascii() { + return tokens; + } + + // Char offset of every byte offset that falls on a char boundary, built + // once so the conversion stays linear in the source length. + let mut char_offsets = vec![0; source.len() + 1]; + for (char_idx, (byte_idx, _)) in source.char_indices().enumerate() { + char_offsets[byte_idx] = char_idx; + } + char_offsets[source.len()] = source.chars().count(); + + for token in &mut tokens { + token.span = char_offsets[token.span.start]..char_offsets[token.span.end]; + } + tokens +} + /// Insert a start token so later stages can treat the start of a file like a newline fn insert_start(tokens: Vec) -> Vec { std::iter::once(Token { diff --git a/prqlc/prqlc-parser/src/lexer/test.rs b/prqlc/prqlc-parser/src/lexer/test.rs index c698d69915df..5d01acee5dc9 100644 --- a/prqlc/prqlc-parser/src/lexer/test.rs +++ b/prqlc/prqlc-parser/src/lexer/test.rs @@ -551,3 +551,23 @@ fn recovery_returns_no_tokens_on_error() { ] "#); } + +#[test] +fn test_lex_source_non_ascii_spans() { + // Spans count chars, not bytes, so a multi-byte char doesn't shift the + // spans of the tokens after it. + assert_debug_snapshot!(lex_source("# é\nx 'ü' y"), @r#" + Ok( + Tokens( + [ + 0..0: Start, + 0..3: Comment(" é"), + 3..4: NewLine, + 4..5: Ident("x"), + 6..9: Literal(String("ü")), + 10..11: Ident("y"), + ], + ), + ) + "#); +} diff --git a/prqlc/prqlc-parser/src/parser/interpolation.rs b/prqlc/prqlc-parser/src/parser/interpolation.rs index d6558f680fe7..6a8b692af41c 100644 --- a/prqlc/prqlc-parser/src/parser/interpolation.rs +++ b/prqlc/prqlc-parser/src/parser/interpolation.rs @@ -11,14 +11,18 @@ pub(crate) fn parse(string: String, span_base: Span) -> Result Result { let adjusted_expr = Box::new(Expr { span: expr.span.map(|s| Span { - start: span_base.start + s.start, - end: span_base.start + s.end, + start: span_base.start + to_char(s.start), + end: span_base.start + to_char(s.end), source_id: span_base.source_id, }), ..(*expr) diff --git a/prqlc/prqlc-parser/src/parser/mod.rs b/prqlc/prqlc-parser/src/parser/mod.rs index e3ea7f64ce2f..f0dcfa66fb4a 100644 --- a/prqlc/prqlc-parser/src/parser/mod.rs +++ b/prqlc/prqlc-parser/src/parser/mod.rs @@ -35,14 +35,14 @@ pub fn parse_lr_to_pr(source_id: u16, lr: Vec) -> (Option