From 128a0dd5b22d2883a9b84020e63fb2c1f5ccdb4b Mon Sep 17 00:00:00 2001 From: prql-bot <107324867+prql-bot@users.noreply.github.com> Date: Fri, 25 Sep 2026 07:21:20 +0000 Subject: [PATCH 1/4] fix: count token spans in chars so non-ASCII text doesn't shift or crash error labels chumsky 0.10 produces byte offsets over &str, while ariadne and the rest of the compiler read spans as char offsets. Convert lexer token spans and interpolated-string spans to char offsets. Closes #6377 --- prqlc/prqlc-parser/src/lexer/mod.rs | 25 +++++++++- prqlc/prqlc-parser/src/lexer/test.rs | 20 ++++++++ .../prqlc-parser/src/parser/interpolation.rs | 12 +++-- prqlc/prqlc-parser/src/test.rs | 6 +-- .../prqlc/tests/integration/error_messages.rs | 48 +++++++++++++++++++ 5 files changed, 102 insertions(+), 9 deletions(-) diff --git a/prqlc/prqlc-parser/src/lexer/mod.rs b/prqlc/prqlc-parser/src/lexer/mod.rs index 497d236318db..d59e950cc4fc 100644 --- a/prqlc/prqlc-parser/src/lexer/mod.rs +++ b/prqlc/prqlc-parser/src/lexer/mod.rs @@ -68,7 +68,7 @@ pub fn lex_source_recovery(source: &str, source_id: u16) -> (Option>, let result = lexer().parse(source).into_result(); match result { - Ok(tokens) => (Some(insert_start(tokens.to_vec())), vec![]), + Ok(tokens) => (Some(insert_start(to_char_spans(source, tokens))), vec![]), Err(errors) => { // Convert chumsky Simple errors to our Error type let errors = errors @@ -86,7 +86,7 @@ pub fn lex_source(source: &str) -> Result> { let result = lexer().parse(source).into_result(); match result { - Ok(tokens) => Ok(Tokens(insert_start(tokens.to_vec()))), + Ok(tokens) => Ok(Tokens(insert_start(to_char_spans(source, tokens)))), Err(errors) => { // Convert chumsky Simple errors to our Error type let errors = errors @@ -99,6 +99,27 @@ pub fn lex_source(source: &str) -> Result> { } } +/// Convert token spans from the byte offsets chumsky produces over `&str` to +/// the char offsets the rest of the compiler (and ariadne) expects. +fn to_char_spans(source: &str, mut tokens: Vec) -> Vec { + if source.is_ascii() { + return tokens; + } + + // Char offset of every byte offset that falls on a char boundary, built + // once so the conversion stays linear in the source length. + let mut char_offsets = vec![0; source.len() + 1]; + for (char_idx, (byte_idx, _)) in source.char_indices().enumerate() { + char_offsets[byte_idx] = char_idx; + } + char_offsets[source.len()] = source.chars().count(); + + for token in &mut tokens { + token.span = char_offsets[token.span.start]..char_offsets[token.span.end]; + } + tokens +} + /// Insert a start token so later stages can treat the start of a file like a newline fn insert_start(tokens: Vec) -> Vec { std::iter::once(Token { diff --git a/prqlc/prqlc-parser/src/lexer/test.rs b/prqlc/prqlc-parser/src/lexer/test.rs index c698d69915df..5d01acee5dc9 100644 --- a/prqlc/prqlc-parser/src/lexer/test.rs +++ b/prqlc/prqlc-parser/src/lexer/test.rs @@ -551,3 +551,23 @@ fn recovery_returns_no_tokens_on_error() { ] "#); } + +#[test] +fn test_lex_source_non_ascii_spans() { + // Spans count chars, not bytes, so a multi-byte char doesn't shift the + // spans of the tokens after it. + assert_debug_snapshot!(lex_source("# é\nx 'ü' y"), @r#" + Ok( + Tokens( + [ + 0..0: Start, + 0..3: Comment(" é"), + 3..4: NewLine, + 4..5: Ident("x"), + 6..9: Literal(String("ü")), + 10..11: Ident("y"), + ], + ), + ) + "#); +} diff --git a/prqlc/prqlc-parser/src/parser/interpolation.rs b/prqlc/prqlc-parser/src/parser/interpolation.rs index d6558f680fe7..6a8b692af41c 100644 --- a/prqlc/prqlc-parser/src/parser/interpolation.rs +++ b/prqlc/prqlc-parser/src/parser/interpolation.rs @@ -11,14 +11,18 @@ pub(crate) fn parse(string: String, span_base: Span) -> Result Result { let adjusted_expr = Box::new(Expr { span: expr.span.map(|s| Span { - start: span_base.start + s.start, - end: span_base.start + s.end, + start: span_base.start + to_char(s.start), + end: span_base.start + to_char(s.end), source_id: span_base.source_id, }), ..(*expr) diff --git a/prqlc/prqlc-parser/src/test.rs b/prqlc/prqlc-parser/src/test.rs index 1a0f064f61df..a9a8f47e6e2d 100644 --- a/prqlc/prqlc-parser/src/test.rs +++ b/prqlc/prqlc-parser/src/test.rs @@ -1624,9 +1624,9 @@ fn test_unicode() { args: - Ident: - tète - span: "0:5-10" - span: "0:0-10" - span: "0:0-10" + span: "0:5-9" + span: "0:0-9" + span: "0:0-9" "#); } diff --git a/prqlc/prqlc/tests/integration/error_messages.rs b/prqlc/prqlc/tests/integration/error_messages.rs index 4f609aa02de2..fae88d118405 100644 --- a/prqlc/prqlc/tests/integration/error_messages.rs +++ b/prqlc/prqlc/tests/integration/error_messages.rs @@ -806,3 +806,51 @@ fn unknown_named_arg() { ───╯ "); } + +#[test] +fn test_error_after_non_ascii() { + // Multi-byte chars before an error mustn't shift its label, or push the + // span past the end of the source. + assert_snapshot!(compile(r#" + from t + derive {x = "éééééééééé"} + select {x, foo.bar.baz} + "#).unwrap_err(), @" + Error: + ╭─[ :4:16 ] + │ + 4 │ select {x, foo.bar.baz} + │ ─────┬───── + │ ╰─────── Unknown name `foo.bar.baz` + │ + │ Help: available columns: x + ───╯ + "); + + assert_snapshot!(compile(r#" + # café + from t + select {foo.bar.baz} + "#).unwrap_err(), @" + Error: + ╭─[ :4:13 ] + │ + 4 │ select {foo.bar.baz} + │ ─────┬───── + │ ╰─────── Unknown name `foo.bar.baz` + ───╯ + "); + + assert_snapshot!(compile(r#" + from t + derive {x = f"éééé{foo.}"} + "#).unwrap_err(), @r#" + Error: + ╭─[ :3:28 ] + │ + 3 │ derive {x = f"éééé{foo.}"} + │ ┬ + │ ╰── expected interpolated string or interp:backticks, but found "}" + ───╯ + "#); +} From 6770e4c491e4036ad2a897e604216ea1eff0cd09 Mon Sep 17 00:00:00 2001 From: prql-bot <107324867+prql-bot@users.noreply.github.com> Date: Fri, 25 Sep 2026 07:21:48 +0000 Subject: [PATCH 2/4] docs: add changelog entry for #6378 --- CHANGELOG.md | 3 +++ 1 file changed, 3 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 86394beb73e8..7fa444695bf6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -29,6 +29,9 @@ **Fixes**: +- Error labels after non-ASCII text now point at the right column, rather than + drifting right or crashing `prqlc` when a label ran past the end of the + query. (@prql-bot, #6378) - Report a circular `import` as an error rather than crashing with a stack overflow. (@prql-bot, #6376) - `prqlc experimental doc --format=html` now escapes HTML in the page it From 57641a6636d6a6c87863216826ad423ffb334fdc Mon Sep 17 00:00:00 2001 From: prql-bot <107324867+prql-bot@users.noreply.github.com> Date: Fri, 25 Sep 2026 07:22:01 +0000 Subject: [PATCH 3/4] docs: rewrap changelog entry --- CHANGELOG.md | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 7fa444695bf6..50de4e7c97a9 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -30,8 +30,8 @@ **Fixes**: - Error labels after non-ASCII text now point at the right column, rather than - drifting right or crashing `prqlc` when a label ran past the end of the - query. (@prql-bot, #6378) + drifting right or crashing `prqlc` when a label ran past the end of the query. + (@prql-bot, #6378) - Report a circular `import` as an error rather than crashing with a stack overflow. (@prql-bot, #6376) - `prqlc experimental doc --format=html` now escapes HTML in the page it From 8f07a6579c9a9a57d044744656480cfd1d9138e3 Mon Sep 17 00:00:00 2001 From: prql-bot <107324867+prql-bot@users.noreply.github.com> Date: Fri, 25 Sep 2026 07:24:18 +0000 Subject: [PATCH 4/4] refactor: update parser span comments to say char offsets --- prqlc/prqlc-parser/src/parser/mod.rs | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/prqlc/prqlc-parser/src/parser/mod.rs b/prqlc/prqlc-parser/src/parser/mod.rs index e3ea7f64ce2f..f0dcfa66fb4a 100644 --- a/prqlc/prqlc-parser/src/parser/mod.rs +++ b/prqlc/prqlc-parser/src/parser/mod.rs @@ -35,14 +35,14 @@ pub fn parse_lr_to_pr(source_id: u16, lr: Vec) -> (Option