Skip to main content

sqlparser/
tokenizer.rs

1// Licensed to the Apache Software Foundation (ASF) under one
2// or more contributor license agreements.  See the NOTICE file
3// distributed with this work for additional information
4// regarding copyright ownership.  The ASF licenses this file
5// to you under the Apache License, Version 2.0 (the
6// "License"); you may not use this file except in compliance
7// with the License.  You may obtain a copy of the License at
8//
9//   http://www.apache.org/licenses/LICENSE-2.0
10//
11// Unless required by applicable law or agreed to in writing,
12// software distributed under the License is distributed on an
13// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
14// KIND, either express or implied.  See the License for the
15// specific language governing permissions and limitations
16// under the License.
17
18//! SQL Tokenizer
19//!
20//! The tokenizer (a.k.a. lexer) converts a string into a sequence of tokens.
21//!
22//! The tokens then form the input for the parser, which outputs an Abstract Syntax Tree (AST).
23
24#[cfg(not(feature = "std"))]
25use alloc::{
26    borrow::ToOwned,
27    format,
28    string::{String, ToString},
29    vec,
30    vec::Vec,
31};
32use core::num::NonZeroU8;
33use core::str::Chars;
34use core::{cmp, fmt};
35use core::{iter::Peekable, str};
36
37#[cfg(feature = "serde")]
38use serde::{Deserialize, Serialize};
39
40#[cfg(feature = "visitor")]
41use sqlparser_derive::{Visit, VisitMut};
42
43use crate::dialect::Dialect;
44use crate::dialect::{
45    BigQueryDialect, DuckDbDialect, GenericDialect, MySqlDialect, PostgreSqlDialect,
46    SnowflakeDialect,
47};
48use crate::keywords::{Keyword, ALL_KEYWORDS, ALL_KEYWORDS_INDEX};
49use crate::{
50    ast::{DollarQuotedString, QuoteDelimitedString},
51    dialect::HiveDialect,
52};
53
54/// SQL Token enumeration
55#[derive(Debug, Clone, PartialEq, PartialOrd, Eq, Ord, Hash)]
56#[cfg_attr(feature = "serde", derive(Serialize, Deserialize))]
57#[cfg_attr(feature = "visitor", derive(Visit, VisitMut))]
58pub enum Token {
59    /// An end-of-file marker, not a real token
60    EOF,
61    /// A keyword (like SELECT) or an optionally quoted SQL identifier
62    Word(Word),
63    /// An unsigned numeric literal
64    Number(String, bool),
65    /// A character that could not be tokenized
66    Char(char),
67    /// Single quoted string: i.e: 'string'
68    SingleQuotedString(String),
69    /// Double quoted string: i.e: "string"
70    DoubleQuotedString(String),
71    /// Triple single quoted strings: Example '''abc'''
72    /// [BigQuery](https://cloud.google.com/bigquery/docs/reference/standard-sql/lexical#quoted_literals)
73    TripleSingleQuotedString(String),
74    /// Triple double quoted strings: Example """abc"""
75    /// [BigQuery](https://cloud.google.com/bigquery/docs/reference/standard-sql/lexical#quoted_literals)
76    TripleDoubleQuotedString(String),
77    /// Dollar quoted string: i.e: $$string$$ or $tag_name$string$tag_name$
78    DollarQuotedString(DollarQuotedString),
79    /// Byte string literal: i.e: b'string' or B'string' (note that some backends, such as
80    /// PostgreSQL, may treat this syntax as a bit string literal instead, i.e: b'10010101')
81    SingleQuotedByteStringLiteral(String),
82    /// Byte string literal: i.e: b"string" or B"string"
83    DoubleQuotedByteStringLiteral(String),
84    /// Triple single quoted literal with byte string prefix. Example `B'''abc'''`
85    /// [BigQuery](https://cloud.google.com/bigquery/docs/reference/standard-sql/lexical#quoted_literals)
86    TripleSingleQuotedByteStringLiteral(String),
87    /// Triple double quoted literal with byte string prefix. Example `B"""abc"""`
88    /// [BigQuery](https://cloud.google.com/bigquery/docs/reference/standard-sql/lexical#quoted_literals)
89    TripleDoubleQuotedByteStringLiteral(String),
90    /// Single quoted literal with raw string prefix. Example `R'abc'`
91    /// [BigQuery](https://cloud.google.com/bigquery/docs/reference/standard-sql/lexical#quoted_literals)
92    SingleQuotedRawStringLiteral(String),
93    /// Double quoted literal with raw string prefix. Example `R"abc"`
94    /// [BigQuery](https://cloud.google.com/bigquery/docs/reference/standard-sql/lexical#quoted_literals)
95    DoubleQuotedRawStringLiteral(String),
96    /// Triple single quoted literal with raw string prefix. Example `R'''abc'''`
97    /// [BigQuery](https://cloud.google.com/bigquery/docs/reference/standard-sql/lexical#quoted_literals)
98    TripleSingleQuotedRawStringLiteral(String),
99    /// Triple double quoted literal with raw string prefix. Example `R"""abc"""`
100    /// [BigQuery](https://cloud.google.com/bigquery/docs/reference/standard-sql/lexical#quoted_literals)
101    TripleDoubleQuotedRawStringLiteral(String),
102    /// "National" string literal: i.e: N'string'
103    NationalStringLiteral(String),
104    /// Quote delimited literal. Examples `Q'{ab'c}'`, `Q'|ab'c|'`, `Q'|ab|c|'`
105    /// [Oracle](https://docs.oracle.com/en/database/oracle/oracle-database/21/sqlrf/Literals.html#GUID-1824CBAA-6E16-4921-B2A6-112FB02248DA)
106    QuoteDelimitedStringLiteral(QuoteDelimitedString),
107    /// "Nationa" quote delimited literal. Examples `NQ'{ab'c}'`, `NQ'|ab'c|'`, `NQ'|ab|c|'`
108    /// [Oracle](https://docs.oracle.com/en/database/oracle/oracle-database/21/sqlrf/Literals.html#GUID-1824CBAA-6E16-4921-B2A6-112FB02248DA)
109    NationalQuoteDelimitedStringLiteral(QuoteDelimitedString),
110    /// "escaped" string literal, which are an extension to the SQL standard: i.e: e'first \n second' or E 'first \n second'
111    EscapedStringLiteral(String),
112    /// Unicode string literal: i.e: U&'first \000A second'
113    UnicodeStringLiteral(String),
114    /// Hexadecimal string literal: i.e.: X'deadbeef'
115    HexStringLiteral(String),
116    /// Comma
117    Comma,
118    /// Whitespace (space, tab, etc)
119    Whitespace(Whitespace),
120    /// Double equals sign `==`
121    DoubleEq,
122    /// Equality operator `=`
123    Eq,
124    /// Not Equals operator `<>` (or `!=` in some dialects)
125    Neq,
126    /// Less Than operator `<`
127    Lt,
128    /// Greater Than operator `>`
129    Gt,
130    /// Less Than Or Equals operator `<=`
131    LtEq,
132    /// Greater Than Or Equals operator `>=`
133    GtEq,
134    /// Spaceship operator <=>
135    Spaceship,
136    /// Plus operator `+`
137    Plus,
138    /// Minus operator `-`
139    Minus,
140    /// Multiplication operator `*`
141    Mul,
142    /// Division operator `/`
143    Div,
144    /// Integer division operator `//` in DuckDB
145    DuckIntDiv,
146    /// Modulo Operator `%`
147    Mod,
148    /// String concatenation `||`
149    StringConcat,
150    /// Left parenthesis `(`
151    LParen,
152    /// Right parenthesis `)`
153    RParen,
154    /// Period (used for compound identifiers or projections into nested types)
155    Period,
156    /// Colon `:`
157    Colon,
158    /// DoubleColon `::` (used for casting in PostgreSQL)
159    DoubleColon,
160    /// Assignment `:=` (used for keyword argument in DuckDB macros and some functions, and for variable declarations in DuckDB and Snowflake)
161    Assignment,
162    /// SemiColon `;` used as separator for COPY and payload
163    SemiColon,
164    /// Backslash `\` used in terminating the COPY payload with `\.`
165    Backslash,
166    /// Left bracket `[`
167    LBracket,
168    /// Right bracket `]`
169    RBracket,
170    /// Ampersand `&`
171    Ampersand,
172    /// Pipe `|`
173    Pipe,
174    /// Caret `^`
175    Caret,
176    /// Left brace `{`
177    LBrace,
178    /// Right brace `}`
179    RBrace,
180    /// Right Arrow `=>`
181    RArrow,
182    /// Sharp `#` used for PostgreSQL Bitwise XOR operator, also PostgreSQL/Redshift geometrical unary/binary operator (Number of points in path or polygon/Intersection)
183    Sharp,
184    /// `##` PostgreSQL/Redshift geometrical binary operator (Point of closest proximity)
185    DoubleSharp,
186    /// Tilde `~` used for PostgreSQL Bitwise NOT operator or case sensitive match regular expression operator
187    Tilde,
188    /// `~*` , a case insensitive match regular expression operator in PostgreSQL
189    TildeAsterisk,
190    /// `!~` , a case sensitive not match regular expression operator in PostgreSQL
191    ExclamationMarkTilde,
192    /// `!~*` , a case insensitive not match regular expression operator in PostgreSQL
193    ExclamationMarkTildeAsterisk,
194    /// `~~`, a case sensitive match pattern operator in PostgreSQL
195    DoubleTilde,
196    /// `~~*`, a case insensitive match pattern operator in PostgreSQL
197    DoubleTildeAsterisk,
198    /// `!~~`, a case sensitive not match pattern operator in PostgreSQL
199    ExclamationMarkDoubleTilde,
200    /// `!~~*`, a case insensitive not match pattern operator in PostgreSQL
201    ExclamationMarkDoubleTildeAsterisk,
202    /// `<<`, a bitwise shift left operator in PostgreSQL
203    ShiftLeft,
204    /// `>>`, a bitwise shift right operator in PostgreSQL
205    ShiftRight,
206    /// `&&`, an overlap operator in PostgreSQL
207    Overlap,
208    /// Exclamation Mark `!` used for PostgreSQL factorial operator
209    ExclamationMark,
210    /// Double Exclamation Mark `!!` used for PostgreSQL prefix factorial operator
211    DoubleExclamationMark,
212    /// AtSign `@` used for PostgreSQL abs operator, also PostgreSQL/Redshift geometrical unary/binary operator (Center, Contained or on)
213    AtSign,
214    /// `^@`, a "starts with" string operator in PostgreSQL
215    CaretAt,
216    /// `|/`, a square root math operator in PostgreSQL
217    PGSquareRoot,
218    /// `||/`, a cube root math operator in PostgreSQL
219    PGCubeRoot,
220    /// `?` or `$` , a prepared statement arg placeholder
221    Placeholder(String),
222    /// `->`, used as a operator to extract json field in PostgreSQL
223    Arrow,
224    /// `->>`, used as a operator to extract json field as text in PostgreSQL
225    LongArrow,
226    /// `#>`, extracts JSON sub-object at the specified path
227    HashArrow,
228    /// `@-@` PostgreSQL/Redshift geometrical unary operator (Length or circumference)
229    AtDashAt,
230    /// `?-` PostgreSQL/Redshift geometrical unary/binary operator (Is horizontal?/Are horizontally aligned?)
231    QuestionMarkDash,
232    /// `&<` PostgreSQL/Redshift geometrical binary operator (Overlaps to left?)
233    AmpersandLeftAngleBracket,
234    /// `&>` PostgreSQL/Redshift geometrical binary operator (Overlaps to right?)`
235    AmpersandRightAngleBracket,
236    /// `&<|` PostgreSQL/Redshift geometrical binary operator (Does not extend above?)`
237    AmpersandLeftAngleBracketVerticalBar,
238    /// `|&>` PostgreSQL/Redshift geometrical binary operator (Does not extend below?)`
239    VerticalBarAmpersandRightAngleBracket,
240    /// `<->` PostgreSQL/Redshift geometrical binary operator (Distance between)
241    TwoWayArrow,
242    /// `<^` PostgreSQL/Redshift geometrical binary operator (Is below?)
243    LeftAngleBracketCaret,
244    /// `>^` PostgreSQL/Redshift geometrical binary operator (Is above?)
245    RightAngleBracketCaret,
246    /// `?#` PostgreSQL/Redshift geometrical binary operator (Intersects or overlaps)
247    QuestionMarkSharp,
248    /// `?-|` PostgreSQL/Redshift geometrical binary operator (Is perpendicular?)
249    QuestionMarkDashVerticalBar,
250    /// `?||` PostgreSQL/Redshift geometrical binary operator (Are parallel?)
251    QuestionMarkDoubleVerticalBar,
252    /// `~=` PostgreSQL/Redshift geometrical binary operator (Same as)
253    TildeEqual,
254    /// `<<| PostgreSQL/Redshift geometrical binary operator (Is strictly below?)
255    ShiftLeftVerticalBar,
256    /// `|>> PostgreSQL/Redshift geometrical binary operator (Is strictly above?)
257    VerticalBarShiftRight,
258    /// `|> BigQuery pipe operator
259    VerticalBarRightAngleBracket,
260    /// `#>>`, extracts JSON sub-object at the specified path as text
261    HashLongArrow,
262    /// jsonb @> jsonb -> boolean: Test whether left json contains the right json
263    AtArrow,
264    /// jsonb <@ jsonb -> boolean: Test whether right json contains the left json
265    ArrowAt,
266    /// jsonb #- text[] -> jsonb: Deletes the field or array element at the specified
267    /// path, where path elements can be either field keys or array indexes.
268    HashMinus,
269    /// jsonb @? jsonpath -> boolean: Does JSON path return any item for the specified
270    /// JSON value?
271    AtQuestion,
272    /// jsonb @@ jsonpath → boolean: Returns the result of a JSON path predicate check
273    /// for the specified JSON value. Only the first item of the result is taken into
274    /// account. If the result is not Boolean, then NULL is returned.
275    AtAt,
276    /// jsonb ? text -> boolean: Checks whether the string exists as a top-level key within the
277    /// jsonb object
278    Question,
279    /// jsonb ?& text[] -> boolean: Check whether all members of the text array exist as top-level
280    /// keys within the jsonb object
281    QuestionAnd,
282    /// jsonb ?| text[] -> boolean: Check whether any member of the text array exists as top-level
283    /// keys within the jsonb object
284    QuestionPipe,
285    /// Custom binary operator
286    /// This is used to represent any custom binary operator that is not part of the SQL standard.
287    /// PostgreSQL allows defining custom binary operators using CREATE OPERATOR.
288    CustomBinaryOperator(String),
289}
290
291impl fmt::Display for Token {
292    fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result {
293        match self {
294            Token::EOF => f.write_str("EOF"),
295            Token::Word(ref w) => write!(f, "{w}"),
296            Token::Number(ref n, l) => write!(f, "{}{long}", n, long = if *l { "L" } else { "" }),
297            Token::Char(ref c) => write!(f, "{c}"),
298            Token::SingleQuotedString(ref s) => write!(f, "'{s}'"),
299            Token::TripleSingleQuotedString(ref s) => write!(f, "'''{s}'''"),
300            Token::DoubleQuotedString(ref s) => write!(f, "\"{s}\""),
301            Token::TripleDoubleQuotedString(ref s) => write!(f, "\"\"\"{s}\"\"\""),
302            Token::DollarQuotedString(ref s) => write!(f, "{s}"),
303            Token::NationalStringLiteral(ref s) => write!(f, "N'{s}'"),
304            Token::QuoteDelimitedStringLiteral(ref s) => s.fmt(f),
305            Token::NationalQuoteDelimitedStringLiteral(ref s) => write!(f, "N{s}"),
306            Token::EscapedStringLiteral(ref s) => write!(f, "E'{s}'"),
307            Token::UnicodeStringLiteral(ref s) => write!(f, "U&'{s}'"),
308            Token::HexStringLiteral(ref s) => write!(f, "X'{s}'"),
309            Token::SingleQuotedByteStringLiteral(ref s) => write!(f, "B'{s}'"),
310            Token::TripleSingleQuotedByteStringLiteral(ref s) => write!(f, "B'''{s}'''"),
311            Token::DoubleQuotedByteStringLiteral(ref s) => write!(f, "B\"{s}\""),
312            Token::TripleDoubleQuotedByteStringLiteral(ref s) => write!(f, "B\"\"\"{s}\"\"\""),
313            Token::SingleQuotedRawStringLiteral(ref s) => write!(f, "R'{s}'"),
314            Token::DoubleQuotedRawStringLiteral(ref s) => write!(f, "R\"{s}\""),
315            Token::TripleSingleQuotedRawStringLiteral(ref s) => write!(f, "R'''{s}'''"),
316            Token::TripleDoubleQuotedRawStringLiteral(ref s) => write!(f, "R\"\"\"{s}\"\"\""),
317            Token::Comma => f.write_str(","),
318            Token::Whitespace(ws) => write!(f, "{ws}"),
319            Token::DoubleEq => f.write_str("=="),
320            Token::Spaceship => f.write_str("<=>"),
321            Token::Eq => f.write_str("="),
322            Token::Neq => f.write_str("<>"),
323            Token::Lt => f.write_str("<"),
324            Token::Gt => f.write_str(">"),
325            Token::LtEq => f.write_str("<="),
326            Token::GtEq => f.write_str(">="),
327            Token::Plus => f.write_str("+"),
328            Token::Minus => f.write_str("-"),
329            Token::Mul => f.write_str("*"),
330            Token::Div => f.write_str("/"),
331            Token::DuckIntDiv => f.write_str("//"),
332            Token::StringConcat => f.write_str("||"),
333            Token::Mod => f.write_str("%"),
334            Token::LParen => f.write_str("("),
335            Token::RParen => f.write_str(")"),
336            Token::Period => f.write_str("."),
337            Token::Colon => f.write_str(":"),
338            Token::DoubleColon => f.write_str("::"),
339            Token::Assignment => f.write_str(":="),
340            Token::SemiColon => f.write_str(";"),
341            Token::Backslash => f.write_str("\\"),
342            Token::LBracket => f.write_str("["),
343            Token::RBracket => f.write_str("]"),
344            Token::Ampersand => f.write_str("&"),
345            Token::Caret => f.write_str("^"),
346            Token::Pipe => f.write_str("|"),
347            Token::LBrace => f.write_str("{"),
348            Token::RBrace => f.write_str("}"),
349            Token::RArrow => f.write_str("=>"),
350            Token::Sharp => f.write_str("#"),
351            Token::DoubleSharp => f.write_str("##"),
352            Token::ExclamationMark => f.write_str("!"),
353            Token::DoubleExclamationMark => f.write_str("!!"),
354            Token::Tilde => f.write_str("~"),
355            Token::TildeAsterisk => f.write_str("~*"),
356            Token::ExclamationMarkTilde => f.write_str("!~"),
357            Token::ExclamationMarkTildeAsterisk => f.write_str("!~*"),
358            Token::DoubleTilde => f.write_str("~~"),
359            Token::DoubleTildeAsterisk => f.write_str("~~*"),
360            Token::ExclamationMarkDoubleTilde => f.write_str("!~~"),
361            Token::ExclamationMarkDoubleTildeAsterisk => f.write_str("!~~*"),
362            Token::AtSign => f.write_str("@"),
363            Token::CaretAt => f.write_str("^@"),
364            Token::ShiftLeft => f.write_str("<<"),
365            Token::ShiftRight => f.write_str(">>"),
366            Token::Overlap => f.write_str("&&"),
367            Token::PGSquareRoot => f.write_str("|/"),
368            Token::PGCubeRoot => f.write_str("||/"),
369            Token::AtDashAt => f.write_str("@-@"),
370            Token::QuestionMarkDash => f.write_str("?-"),
371            Token::AmpersandLeftAngleBracket => f.write_str("&<"),
372            Token::AmpersandRightAngleBracket => f.write_str("&>"),
373            Token::AmpersandLeftAngleBracketVerticalBar => f.write_str("&<|"),
374            Token::VerticalBarAmpersandRightAngleBracket => f.write_str("|&>"),
375            Token::VerticalBarRightAngleBracket => f.write_str("|>"),
376            Token::TwoWayArrow => f.write_str("<->"),
377            Token::LeftAngleBracketCaret => f.write_str("<^"),
378            Token::RightAngleBracketCaret => f.write_str(">^"),
379            Token::QuestionMarkSharp => f.write_str("?#"),
380            Token::QuestionMarkDashVerticalBar => f.write_str("?-|"),
381            Token::QuestionMarkDoubleVerticalBar => f.write_str("?||"),
382            Token::TildeEqual => f.write_str("~="),
383            Token::ShiftLeftVerticalBar => f.write_str("<<|"),
384            Token::VerticalBarShiftRight => f.write_str("|>>"),
385            Token::Placeholder(ref s) => write!(f, "{s}"),
386            Token::Arrow => write!(f, "->"),
387            Token::LongArrow => write!(f, "->>"),
388            Token::HashArrow => write!(f, "#>"),
389            Token::HashLongArrow => write!(f, "#>>"),
390            Token::AtArrow => write!(f, "@>"),
391            Token::ArrowAt => write!(f, "<@"),
392            Token::HashMinus => write!(f, "#-"),
393            Token::AtQuestion => write!(f, "@?"),
394            Token::AtAt => write!(f, "@@"),
395            Token::Question => write!(f, "?"),
396            Token::QuestionAnd => write!(f, "?&"),
397            Token::QuestionPipe => write!(f, "?|"),
398            Token::CustomBinaryOperator(s) => f.write_str(s),
399        }
400    }
401}
402
403impl Token {
404    /// Create a `Token::Word` from an unquoted `keyword`.
405    ///
406    /// The lookup is case-insensitive; unknown values become `Keyword::NoKeyword`.
407    pub fn make_keyword(keyword: &str) -> Self {
408        Token::make_word(keyword, None)
409    }
410
411    /// Create a `Token::Word` from `word` with an optional `quote_style`.
412    ///
413    /// When `quote_style` is `None`, the parser attempts a case-insensitive keyword
414    /// lookup and sets the `Word::keyword` accordingly.
415    pub fn make_word(word: &str, quote_style: Option<char>) -> Self {
416        Token::Word(Word {
417            keyword: keyword_lookup(word, quote_style),
418            value: word.to_string(),
419            quote_style,
420        })
421    }
422
423    /// Like [`Self::make_word`] but takes ownership of the word `String`,
424    /// avoiding an extra allocation when the caller already has an owned value.
425    fn make_word_owned(word: String, quote_style: Option<char>) -> Self {
426        Token::Word(Word {
427            keyword: keyword_lookup(&word, quote_style),
428            value: word,
429            quote_style,
430        })
431    }
432}
433
434/// Case-insensitive keyword lookup using binary search over [`ALL_KEYWORDS`].
435fn keyword_lookup(word: &str, quote_style: Option<char>) -> Keyword {
436    if quote_style.is_some() {
437        return Keyword::NoKeyword;
438    }
439    ALL_KEYWORDS
440        .binary_search_by(|probe| {
441            let probe = probe.as_bytes();
442            let word = word.as_bytes();
443            for (p, w) in probe.iter().zip(word.iter()) {
444                let cmp = p.cmp(&w.to_ascii_uppercase());
445                if cmp != core::cmp::Ordering::Equal {
446                    return cmp;
447                }
448            }
449            probe.len().cmp(&word.len())
450        })
451        .map_or(Keyword::NoKeyword, |x| ALL_KEYWORDS_INDEX[x])
452}
453
454/// A keyword (like SELECT) or an optionally quoted SQL identifier
455#[derive(Debug, Clone, PartialEq, PartialOrd, Eq, Ord, Hash)]
456#[cfg_attr(feature = "serde", derive(Serialize, Deserialize))]
457#[cfg_attr(feature = "visitor", derive(Visit, VisitMut))]
458pub struct Word {
459    /// The value of the token, without the enclosing quotes, and with the
460    /// escape sequences (if any) processed (TODO: escapes are not handled)
461    pub value: String,
462    /// An identifier can be "quoted" (&lt;delimited identifier> in ANSI parlance).
463    /// The standard and most implementations allow using double quotes for this,
464    /// but some implementations support other quoting styles as well (e.g. \[MS SQL])
465    pub quote_style: Option<char>,
466    /// If the word was not quoted and it matched one of the known keywords,
467    /// this will have one of the values from dialect::keywords, otherwise empty
468    pub keyword: Keyword,
469}
470
471impl fmt::Display for Word {
472    fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result {
473        match self.quote_style {
474            Some(s) if s == '"' || s == '[' || s == '`' => {
475                write!(f, "{}{}{}", s, self.value, Word::matching_end_quote(s))
476            }
477            None => f.write_str(&self.value),
478            _ => panic!("Unexpected quote_style!"),
479        }
480    }
481}
482
483impl Word {
484    fn matching_end_quote(ch: char) -> char {
485        match ch {
486            '"' => '"', // ANSI and most dialects
487            '[' => ']', // MS SQL
488            '`' => '`', // MySQL
489            _ => panic!("unexpected quoting style!"),
490        }
491    }
492}
493
494/// Represents whitespace in the input: spaces, newlines, tabs and comments.
495#[derive(Debug, Clone, PartialEq, PartialOrd, Eq, Ord, Hash)]
496#[cfg_attr(feature = "serde", derive(Serialize, Deserialize))]
497#[cfg_attr(feature = "visitor", derive(Visit, VisitMut))]
498pub enum Whitespace {
499    /// A single space character.
500    Space,
501    /// A newline character.
502    Newline,
503    /// A tab character.
504    Tab,
505    /// A single-line comment (e.g. `-- comment` or `# comment`).
506    /// The `comment` field contains the text, and `prefix` contains the comment prefix.
507    SingleLineComment {
508        /// The content of the comment (without the prefix).
509        comment: String,
510        /// The prefix used for the comment (for example `--` or `#`).
511        prefix: String,
512    },
513
514    /// A multi-line comment (without the `/* ... */` delimiters).
515    MultiLineComment(String),
516}
517
518impl fmt::Display for Whitespace {
519    fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result {
520        match self {
521            Whitespace::Space => f.write_str(" "),
522            Whitespace::Newline => f.write_str("\n"),
523            Whitespace::Tab => f.write_str("\t"),
524            Whitespace::SingleLineComment { prefix, comment } => writeln!(f, "{prefix}{comment}"),
525            Whitespace::MultiLineComment(s) => write!(f, "/*{s}*/"),
526        }
527    }
528}
529
530/// Location in input string
531///
532/// # Create an "empty" (unknown) `Location`
533/// ```
534/// # use sqlparser::tokenizer::Location;
535/// let location = Location::empty();
536/// ```
537///
538/// # Create a `Location` from a line and column
539/// ```
540/// # use sqlparser::tokenizer::Location;
541/// let location = Location::new(1, 1);
542/// ```
543///
544/// # Create a `Location` from a pair
545/// ```
546/// # use sqlparser::tokenizer::Location;
547/// let location = Location::from((1, 1));
548/// ```
549#[derive(Eq, PartialEq, Hash, Clone, Copy, Ord, PartialOrd)]
550#[cfg_attr(feature = "serde", derive(Serialize, Deserialize))]
551#[cfg_attr(feature = "visitor", derive(Visit, VisitMut))]
552pub struct Location {
553    /// Line number, starting from 1.
554    ///
555    /// Note: Line 0 is used for empty spans
556    pub line: u64,
557    /// Line column, starting from 1.
558    ///
559    /// Note: Column 0 is used for empty spans
560    pub column: u64,
561}
562
563impl fmt::Display for Location {
564    fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result {
565        if self.line == 0 {
566            return Ok(());
567        }
568        write!(f, " at Line: {}, Column: {}", self.line, self.column)
569    }
570}
571
572impl fmt::Debug for Location {
573    fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result {
574        write!(f, "Location({},{})", self.line, self.column)
575    }
576}
577
578impl Location {
579    /// Return an "empty" / unknown location
580    pub fn empty() -> Self {
581        Self { line: 0, column: 0 }
582    }
583
584    /// Create a new `Location` for a given line and column
585    pub fn new(line: u64, column: u64) -> Self {
586        Self { line, column }
587    }
588
589    /// Create a new location for a given line and column
590    ///
591    /// Alias for [`Self::new`]
592    // TODO: remove / deprecate in favor of` `new` for consistency?
593    pub fn of(line: u64, column: u64) -> Self {
594        Self::new(line, column)
595    }
596
597    /// Combine self and `end` into a new `Span`
598    pub fn span_to(self, end: Self) -> Span {
599        Span { start: self, end }
600    }
601}
602
603impl From<(u64, u64)> for Location {
604    fn from((line, column): (u64, u64)) -> Self {
605        Self { line, column }
606    }
607}
608
609/// A span represents a linear portion of the input string (start, end)
610///
611/// See [Spanned](crate::ast::Spanned) for more information.
612#[derive(Eq, PartialEq, Hash, Clone, PartialOrd, Ord, Copy)]
613#[cfg_attr(feature = "serde", derive(Serialize, Deserialize))]
614#[cfg_attr(feature = "visitor", derive(Visit, VisitMut))]
615pub struct Span {
616    /// Start `Location` (inclusive).
617    pub start: Location,
618    /// End `Location` (inclusive).
619    pub end: Location,
620}
621
622impl fmt::Debug for Span {
623    fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result {
624        write!(f, "Span({:?}..{:?})", self.start, self.end)
625    }
626}
627
628impl Span {
629    // An empty span (0, 0) -> (0, 0)
630    // We need a const instance for pattern matching
631    const EMPTY: Span = Self::empty();
632
633    /// Create a new span from a start and end [`Location`]
634    pub fn new(start: Location, end: Location) -> Span {
635        Span { start, end }
636    }
637
638    /// Returns an empty span `(0, 0) -> (0, 0)`
639    ///
640    /// Empty spans represent no knowledge of source location
641    /// See [Spanned](crate::ast::Spanned) for more information.
642    pub const fn empty() -> Span {
643        Span {
644            start: Location { line: 0, column: 0 },
645            end: Location { line: 0, column: 0 },
646        }
647    }
648
649    /// Returns the smallest Span that contains both `self` and `other`
650    /// If either span is [Span::empty], the other span is returned
651    ///
652    /// # Examples
653    /// ```
654    /// # use sqlparser::tokenizer::{Span, Location};
655    /// // line 1, column1 -> line 2, column 5
656    /// let span1 = Span::new(Location::new(1, 1), Location::new(2, 5));
657    /// // line 2, column 3 -> line 3, column 7
658    /// let span2 = Span::new(Location::new(2, 3), Location::new(3, 7));
659    /// // Union of the two is the min/max of the two spans
660    /// // line 1, column 1 -> line 3, column 7
661    /// let union = span1.union(&span2);
662    /// assert_eq!(union, Span::new(Location::new(1, 1), Location::new(3, 7)));
663    /// ```
664    pub fn union(&self, other: &Span) -> Span {
665        // If either span is empty, return the other
666        // this prevents propagating (0, 0) through the tree
667        match (self, other) {
668            (&Span::EMPTY, _) => *other,
669            (_, &Span::EMPTY) => *self,
670            _ => Span {
671                start: cmp::min(self.start, other.start),
672                end: cmp::max(self.end, other.end),
673            },
674        }
675    }
676
677    /// Same as [Span::union] for `Option<Span>`
678    ///
679    /// If `other` is `None`, `self` is returned
680    pub fn union_opt(&self, other: &Option<Span>) -> Span {
681        match other {
682            Some(other) => self.union(other),
683            None => *self,
684        }
685    }
686
687    /// Return the [Span::union] of all spans in the iterator
688    ///
689    /// If the iterator is empty, an empty span is returned
690    ///
691    /// # Example
692    /// ```
693    /// # use sqlparser::tokenizer::{Span, Location};
694    /// let spans = vec![
695    ///     Span::new(Location::new(1, 1), Location::new(2, 5)),
696    ///     Span::new(Location::new(2, 3), Location::new(3, 7)),
697    ///     Span::new(Location::new(3, 1), Location::new(4, 2)),
698    /// ];
699    /// // line 1, column 1 -> line 4, column 2
700    /// assert_eq!(
701    ///   Span::union_iter(spans),
702    ///   Span::new(Location::new(1, 1), Location::new(4, 2))
703    /// );
704    pub fn union_iter<I: IntoIterator<Item = Span>>(iter: I) -> Span {
705        iter.into_iter()
706            .reduce(|acc, item| acc.union(&item))
707            .unwrap_or(Span::empty())
708    }
709}
710
711/// Backwards compatibility struct for [`TokenWithSpan`]
712#[deprecated(since = "0.53.0", note = "please use `TokenWithSpan` instead")]
713pub type TokenWithLocation = TokenWithSpan;
714
715/// A [Token] with [Span] attached to it
716///
717/// This is used to track the location of a token in the input string
718///
719/// # Examples
720/// ```
721/// # use sqlparser::tokenizer::{Location, Span, Token, TokenWithSpan};
722/// // commas @ line 1, column 10
723/// let tok1 = TokenWithSpan::new(
724///   Token::Comma,
725///   Span::new(Location::new(1, 10), Location::new(1, 11)),
726/// );
727/// assert_eq!(tok1, Token::Comma); // can compare the token
728///
729/// // commas @ line 2, column 20
730/// let tok2 = TokenWithSpan::new(
731///   Token::Comma,
732///   Span::new(Location::new(2, 20), Location::new(2, 21)),
733/// );
734/// // same token but different locations are not equal
735/// assert_ne!(tok1, tok2);
736/// ```
737#[derive(Debug, Clone, Hash, Ord, PartialOrd, Eq, PartialEq)]
738#[cfg_attr(feature = "serde", derive(Serialize, Deserialize))]
739#[cfg_attr(feature = "visitor", derive(Visit, VisitMut))]
740/// A `Token` together with its `Span` (location in the source).
741pub struct TokenWithSpan {
742    /// The token value.
743    pub token: Token,
744    /// The span covering the token in the input.
745    pub span: Span,
746}
747
748impl TokenWithSpan {
749    /// Create a new [`TokenWithSpan`] from a [`Token`] and a [`Span`]
750    pub fn new(token: Token, span: Span) -> Self {
751        Self { token, span }
752    }
753
754    /// Wrap a token with an empty span
755    pub fn wrap(token: Token) -> Self {
756        Self::new(token, Span::empty())
757    }
758
759    /// Wrap a token with a location from `start` to `end`
760    pub fn at(token: Token, start: Location, end: Location) -> Self {
761        Self::new(token, Span::new(start, end))
762    }
763
764    /// Return an EOF token with no location
765    pub fn new_eof() -> Self {
766        Self::wrap(Token::EOF)
767    }
768}
769
770impl PartialEq<Token> for TokenWithSpan {
771    fn eq(&self, other: &Token) -> bool {
772        &self.token == other
773    }
774}
775
776impl PartialEq<TokenWithSpan> for Token {
777    fn eq(&self, other: &TokenWithSpan) -> bool {
778        self == &other.token
779    }
780}
781
782impl fmt::Display for TokenWithSpan {
783    fn fmt(&self, f: &mut fmt::Formatter) -> fmt::Result {
784        self.token.fmt(f)
785    }
786}
787
788/// An error reported by the tokenizer, with a human-readable `message` and a `location`.
789#[derive(Debug, PartialEq, Eq)]
790pub struct TokenizerError {
791    /// A descriptive error message.
792    pub message: String,
793    /// The `Location` where the error was detected.
794    pub location: Location,
795}
796
797impl fmt::Display for TokenizerError {
798    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
799        write!(f, "{}{}", self.message, self.location,)
800    }
801}
802
803impl core::error::Error for TokenizerError {}
804
805struct State<'a> {
806    peekable: Peekable<Chars<'a>>,
807    line: u64,
808    col: u64,
809}
810
811impl State<'_> {
812    /// return the next character and advance the stream
813    pub fn next(&mut self) -> Option<char> {
814        match self.peekable.next() {
815            None => None,
816            Some(s) => {
817                if s == '\n' {
818                    self.line += 1;
819                    self.col = 1;
820                } else {
821                    self.col += 1;
822                }
823                Some(s)
824            }
825        }
826    }
827
828    /// return the next character but do not advance the stream
829    pub fn peek(&mut self) -> Option<&char> {
830        self.peekable.peek()
831    }
832
833    /// Return the current `Location` (line and column)
834    pub fn location(&self) -> Location {
835        Location {
836            line: self.line,
837            column: self.col,
838        }
839    }
840}
841
842/// Represents how many quote characters enclose a string literal.
843#[derive(Copy, Clone)]
844enum NumStringQuoteChars {
845    /// e.g. `"abc"`, `'abc'`, `r'abc'`
846    One,
847    /// e.g. `"""abc"""`, `'''abc'''`, `r'''abc'''`
848    Many(NonZeroU8),
849}
850
851/// Settings for tokenizing a quoted string literal.
852struct TokenizeQuotedStringSettings {
853    /// The character used to quote the string.
854    quote_style: char,
855    /// Represents how many quotes characters enclose the string literal.
856    num_quote_chars: NumStringQuoteChars,
857    /// The number of opening quotes left to consume, before parsing
858    /// the remaining string literal.
859    /// For example: given initial string `"""abc"""`. If the caller has
860    /// already parsed the first quote for some reason, then this value
861    /// is set to 1, flagging to look to consume only 2 leading quotes.
862    num_opening_quotes_to_consume: u8,
863    /// True if the string uses backslash escaping of special characters
864    /// e.g `'abc\ndef\'ghi'
865    backslash_escape: bool,
866}
867
868/// SQL Tokenizer
869pub struct Tokenizer<'a> {
870    dialect: &'a dyn Dialect,
871    query: &'a str,
872    /// If true (the default), the tokenizer will un-escape literal
873    /// SQL strings See [`Tokenizer::with_unescape`] for more details.
874    unescape: bool,
875}
876
877impl<'a> Tokenizer<'a> {
878    /// Create a new SQL tokenizer for the specified SQL statement
879    ///
880    /// ```
881    /// # use sqlparser::tokenizer::{Token, Whitespace, Tokenizer};
882    /// # use sqlparser::dialect::GenericDialect;
883    /// # let dialect = GenericDialect{};
884    /// let query = r#"SELECT 'foo'"#;
885    ///
886    /// // Parsing the query
887    /// let tokens = Tokenizer::new(&dialect, &query).tokenize().unwrap();
888    ///
889    /// assert_eq!(tokens, vec![
890    ///   Token::make_word("SELECT", None),
891    ///   Token::Whitespace(Whitespace::Space),
892    ///   Token::SingleQuotedString("foo".to_string()),
893    /// ]);
894    pub fn new(dialect: &'a dyn Dialect, query: &'a str) -> Self {
895        Self {
896            dialect,
897            query,
898            unescape: true,
899        }
900    }
901
902    /// Set unescape mode
903    ///
904    /// When true (default) the tokenizer unescapes literal values
905    /// (for example, `""` in SQL is unescaped to the literal `"`).
906    ///
907    /// When false, the tokenizer provides the raw strings as provided
908    /// in the query.  This can be helpful for programs that wish to
909    /// recover the *exact* original query text without normalizing
910    /// the escaping
911    ///
912    /// # Example
913    ///
914    /// ```
915    /// # use sqlparser::tokenizer::{Token, Tokenizer};
916    /// # use sqlparser::dialect::GenericDialect;
917    /// # let dialect = GenericDialect{};
918    /// let query = r#""Foo "" Bar""#;
919    /// let unescaped = Token::make_word(r#"Foo " Bar"#, Some('"'));
920    /// let original  = Token::make_word(r#"Foo "" Bar"#, Some('"'));
921    ///
922    /// // Parsing with unescaping (default)
923    /// let tokens = Tokenizer::new(&dialect, &query).tokenize().unwrap();
924    /// assert_eq!(tokens, vec![unescaped]);
925    ///
926    /// // Parsing with unescape = false
927    /// let tokens = Tokenizer::new(&dialect, &query)
928    ///    .with_unescape(false)
929    ///    .tokenize().unwrap();
930    /// assert_eq!(tokens, vec![original]);
931    /// ```
932    pub fn with_unescape(mut self, unescape: bool) -> Self {
933        self.unescape = unescape;
934        self
935    }
936
937    /// Tokenize the statement and produce a vector of tokens
938    pub fn tokenize(&mut self) -> Result<Vec<Token>, TokenizerError> {
939        let twl = self.tokenize_with_location()?;
940        Ok(twl.into_iter().map(|t| t.token).collect())
941    }
942
943    /// Tokenize the statement and produce a vector of tokens with location information
944    pub fn tokenize_with_location(&mut self) -> Result<Vec<TokenWithSpan>, TokenizerError> {
945        let mut tokens: Vec<TokenWithSpan> = vec![];
946        self.tokenize_with_location_into_buf(&mut tokens)
947            .map(|_| tokens)
948    }
949
950    /// Tokenize the statement and append tokens with location information into the provided buffer.
951    /// If an error is thrown, the buffer will contain all tokens that were successfully parsed before the error.
952    pub fn tokenize_with_location_into_buf(
953        &mut self,
954        buf: &mut Vec<TokenWithSpan>,
955    ) -> Result<(), TokenizerError> {
956        self.tokenize_with_location_into_buf_with_mapper(buf, |token| token)
957    }
958
959    /// Tokenize the statement and produce a vector of tokens, mapping each token
960    /// with provided `mapper`
961    pub fn tokenize_with_location_into_buf_with_mapper(
962        &mut self,
963        buf: &mut Vec<TokenWithSpan>,
964        mut mapper: impl FnMut(TokenWithSpan) -> TokenWithSpan,
965    ) -> Result<(), TokenizerError> {
966        let mut state = State {
967            peekable: self.query.chars().peekable(),
968            line: 1,
969            col: 1,
970        };
971
972        let mut location = state.location();
973        while let Some(token) = self.next_token(&mut state, buf.last().map(|t| &t.token))? {
974            let span = location.span_to(state.location());
975
976            // Check if this is a multiline comment hint that should be expanded
977            match &token {
978                Token::Whitespace(Whitespace::MultiLineComment(comment))
979                    if self.dialect.supports_multiline_comment_hints()
980                        && comment.starts_with('!') =>
981                {
982                    // Re-tokenize the hints and add them to the buffer
983                    self.tokenize_comment_hints(comment, span, buf, &mut mapper)?;
984                }
985                _ => {
986                    buf.push(mapper(TokenWithSpan { token, span }));
987                }
988            }
989
990            location = state.location();
991        }
992        Ok(())
993    }
994
995    /// Re-tokenize optimizer hints from a multiline comment and add them to the buffer.
996    /// For example, `/*!50110 KEY_BLOCK_SIZE = 1024*/` becomes tokens for `KEY_BLOCK_SIZE = 1024`
997    fn tokenize_comment_hints(
998        &self,
999        comment: &str,
1000        span: Span,
1001        buf: &mut Vec<TokenWithSpan>,
1002        mut mapper: impl FnMut(TokenWithSpan) -> TokenWithSpan,
1003    ) -> Result<(), TokenizerError> {
1004        // Strip the leading '!' and any version digits (e.g., "50110")
1005        let hint_content = comment
1006            .strip_prefix('!')
1007            .unwrap_or(comment)
1008            .trim_start_matches(|c: char| c.is_ascii_digit());
1009
1010        // If there's no content after stripping, nothing to tokenize
1011        if hint_content.is_empty() {
1012            return Ok(());
1013        }
1014
1015        // Create a new tokenizer for the hint content
1016        let inner = Tokenizer::new(self.dialect, hint_content).with_unescape(self.unescape);
1017
1018        // Create a state for tracking position within the hint
1019        let mut state = State {
1020            peekable: hint_content.chars().peekable(),
1021            line: span.start.line,
1022            col: span.start.column,
1023        };
1024
1025        // Tokenize the hint content and add tokens to the buffer
1026        let mut location = state.location();
1027        while let Some(token) = inner.next_token(&mut state, buf.last().map(|t| &t.token))? {
1028            let token_span = location.span_to(state.location());
1029            buf.push(mapper(TokenWithSpan {
1030                token,
1031                span: token_span,
1032            }));
1033            location = state.location();
1034        }
1035
1036        Ok(())
1037    }
1038
1039    // Tokenize the identifier or keywords in `ch`
1040    fn tokenize_identifier_or_keyword(
1041        &self,
1042        ch: impl IntoIterator<Item = char>,
1043        chars: &mut State,
1044    ) -> Result<Option<Token>, TokenizerError> {
1045        chars.next(); // consume the first char
1046        let ch: String = ch.into_iter().collect();
1047        let word = self.tokenize_word(ch, chars);
1048
1049        // TODO: implement parsing of exponent here
1050        if word.chars().all(|x| x.is_ascii_digit() || x == '.') {
1051            let mut inner_state = State {
1052                peekable: word.chars().peekable(),
1053                line: 0,
1054                col: 0,
1055            };
1056            let mut s = peeking_take_while(&mut inner_state, |ch| matches!(ch, '0'..='9' | '.'));
1057            let s2 = peeking_take_while(chars, |ch| matches!(ch, '0'..='9' | '.'));
1058            s += s2.as_str();
1059            return Ok(Some(Token::Number(s, false)));
1060        }
1061
1062        Ok(Some(Token::make_word_owned(word, None)))
1063    }
1064
1065    /// Get the next token or return None
1066    fn next_token(
1067        &self,
1068        chars: &mut State,
1069        prev_token: Option<&Token>,
1070    ) -> Result<Option<Token>, TokenizerError> {
1071        match chars.peek() {
1072            Some(&ch) => match ch {
1073                ' ' => self.consume_and_return(chars, Token::Whitespace(Whitespace::Space)),
1074                '\t' => self.consume_and_return(chars, Token::Whitespace(Whitespace::Tab)),
1075                '\n' => self.consume_and_return(chars, Token::Whitespace(Whitespace::Newline)),
1076                '\r' => {
1077                    // Emit a single Whitespace::Newline token for \r and \r\n
1078                    chars.next();
1079                    if let Some('\n') = chars.peek() {
1080                        chars.next();
1081                    }
1082                    Ok(Some(Token::Whitespace(Whitespace::Newline)))
1083                }
1084                // BigQuery and MySQL use b or B for byte string literal, Postgres for bit strings
1085                b @ 'B' | b @ 'b' if dialect_of!(self is BigQueryDialect | PostgreSqlDialect | MySqlDialect | GenericDialect) =>
1086                {
1087                    chars.next(); // consume
1088                    match chars.peek() {
1089                        Some('\'') => {
1090                            if self.dialect.supports_triple_quoted_string() {
1091                                return self
1092                                    .tokenize_single_or_triple_quoted_string::<fn(String) -> Token>(
1093                                        chars,
1094                                        '\'',
1095                                        false,
1096                                        Token::SingleQuotedByteStringLiteral,
1097                                        Token::TripleSingleQuotedByteStringLiteral,
1098                                    );
1099                            }
1100                            let s = self.tokenize_single_quoted_string(chars, '\'', false)?;
1101                            Ok(Some(Token::SingleQuotedByteStringLiteral(s)))
1102                        }
1103                        Some('\"') => {
1104                            if self.dialect.supports_triple_quoted_string() {
1105                                return self
1106                                    .tokenize_single_or_triple_quoted_string::<fn(String) -> Token>(
1107                                        chars,
1108                                        '"',
1109                                        false,
1110                                        Token::DoubleQuotedByteStringLiteral,
1111                                        Token::TripleDoubleQuotedByteStringLiteral,
1112                                    );
1113                            }
1114                            let s = self.tokenize_single_quoted_string(chars, '\"', false)?;
1115                            Ok(Some(Token::DoubleQuotedByteStringLiteral(s)))
1116                        }
1117                        _ => {
1118                            // regular identifier starting with an "b" or "B"
1119                            let s = self.tokenize_word(b, chars);
1120                            Ok(Some(Token::make_word_owned(s, None)))
1121                        }
1122                    }
1123                }
1124                // BigQuery uses r or R for raw string literal
1125                b @ 'R' | b @ 'r' if dialect_of!(self is BigQueryDialect | GenericDialect) => {
1126                    chars.next(); // consume
1127                    match chars.peek() {
1128                        Some('\'') => self
1129                            .tokenize_single_or_triple_quoted_string::<fn(String) -> Token>(
1130                                chars,
1131                                '\'',
1132                                false,
1133                                Token::SingleQuotedRawStringLiteral,
1134                                Token::TripleSingleQuotedRawStringLiteral,
1135                            ),
1136                        Some('\"') => self
1137                            .tokenize_single_or_triple_quoted_string::<fn(String) -> Token>(
1138                                chars,
1139                                '"',
1140                                false,
1141                                Token::DoubleQuotedRawStringLiteral,
1142                                Token::TripleDoubleQuotedRawStringLiteral,
1143                            ),
1144                        _ => {
1145                            // regular identifier starting with an "r" or "R"
1146                            let s = self.tokenize_word(b, chars);
1147                            Ok(Some(Token::make_word_owned(s, None)))
1148                        }
1149                    }
1150                }
1151                // Redshift uses lower case n for national string literal
1152                n @ 'N' | n @ 'n' => {
1153                    chars.next(); // consume, to check the next char
1154                    match chars.peek() {
1155                        Some('\'') => {
1156                            // N'...' - a <national character string literal>
1157                            let backslash_escape =
1158                                self.dialect.supports_string_literal_backslash_escape();
1159                            let s =
1160                                self.tokenize_single_quoted_string(chars, '\'', backslash_escape)?;
1161                            Ok(Some(Token::NationalStringLiteral(s)))
1162                        }
1163                        Some(&q @ 'q') | Some(&q @ 'Q')
1164                            if self.dialect.supports_quote_delimited_string() =>
1165                        {
1166                            chars.next(); // consume and check the next char
1167                            if let Some('\'') = chars.peek() {
1168                                self.tokenize_quote_delimited_string(chars, &[n, q])
1169                                    .map(|s| Some(Token::NationalQuoteDelimitedStringLiteral(s)))
1170                            } else {
1171                                let s = self.tokenize_word(String::from_iter([n, q]), chars);
1172                                Ok(Some(Token::make_word_owned(s, None)))
1173                            }
1174                        }
1175                        _ => {
1176                            // regular identifier starting with an "N"
1177                            let s = self.tokenize_word(n, chars);
1178                            Ok(Some(Token::make_word_owned(s, None)))
1179                        }
1180                    }
1181                }
1182                q @ 'Q' | q @ 'q' if self.dialect.supports_quote_delimited_string() => {
1183                    chars.next(); // consume and check the next char
1184                    if let Some('\'') = chars.peek() {
1185                        self.tokenize_quote_delimited_string(chars, &[q])
1186                            .map(|s| Some(Token::QuoteDelimitedStringLiteral(s)))
1187                    } else {
1188                        let s = self.tokenize_word(q, chars);
1189                        Ok(Some(Token::make_word_owned(s, None)))
1190                    }
1191                }
1192                // PostgreSQL accepts "escape" string constants, which are an extension to the SQL standard.
1193                x @ 'e' | x @ 'E' if self.dialect.supports_string_escape_constant() => {
1194                    let starting_loc = chars.location();
1195                    chars.next(); // consume, to check the next char
1196                    match chars.peek() {
1197                        Some('\'') => {
1198                            let s =
1199                                self.tokenize_escaped_single_quoted_string(starting_loc, chars)?;
1200                            Ok(Some(Token::EscapedStringLiteral(s)))
1201                        }
1202                        _ => {
1203                            // regular identifier starting with an "E" or "e"
1204                            let s = self.tokenize_word(x, chars);
1205                            Ok(Some(Token::make_word_owned(s, None)))
1206                        }
1207                    }
1208                }
1209                // Unicode string literals like U&'first \000A second' are supported in some dialects, including PostgreSQL
1210                x @ 'u' | x @ 'U' if self.dialect.supports_unicode_string_literal() => {
1211                    chars.next(); // consume, to check the next char
1212                    if chars.peek() == Some(&'&') {
1213                        // we cannot advance the iterator here, as we need to consume the '&' later if the 'u' was an identifier
1214                        let mut chars_clone = chars.peekable.clone();
1215                        chars_clone.next(); // consume the '&' in the clone
1216                        if chars_clone.peek() == Some(&'\'') {
1217                            chars.next(); // consume the '&' in the original iterator
1218                            let s = unescape_unicode_single_quoted_string(chars)?;
1219                            return Ok(Some(Token::UnicodeStringLiteral(s)));
1220                        }
1221                    }
1222                    // regular identifier starting with an "U" or "u"
1223                    let s = self.tokenize_word(x, chars);
1224                    Ok(Some(Token::make_word_owned(s, None)))
1225                }
1226                // The spec only allows an uppercase 'X' to introduce a hex
1227                // string, but PostgreSQL, at least, allows a lowercase 'x' too.
1228                x @ 'x' | x @ 'X' => {
1229                    chars.next(); // consume, to check the next char
1230                    match chars.peek() {
1231                        Some('\'') => {
1232                            // X'...' - a <binary string literal>
1233                            let s = self.tokenize_single_quoted_string(chars, '\'', true)?;
1234                            Ok(Some(Token::HexStringLiteral(s)))
1235                        }
1236                        _ => {
1237                            // regular identifier starting with an "X"
1238                            let s = self.tokenize_word(x, chars);
1239                            Ok(Some(Token::make_word_owned(s, None)))
1240                        }
1241                    }
1242                }
1243                // single quoted string
1244                '\'' => {
1245                    if self.dialect.supports_triple_quoted_string() {
1246                        return self
1247                            .tokenize_single_or_triple_quoted_string::<fn(String) -> Token>(
1248                                chars,
1249                                '\'',
1250                                self.dialect.supports_string_literal_backslash_escape(),
1251                                Token::SingleQuotedString,
1252                                Token::TripleSingleQuotedString,
1253                            );
1254                    }
1255                    let s = self.tokenize_single_quoted_string(
1256                        chars,
1257                        '\'',
1258                        self.dialect.supports_string_literal_backslash_escape(),
1259                    )?;
1260
1261                    Ok(Some(Token::SingleQuotedString(s)))
1262                }
1263                // double quoted string
1264                '\"' if !self.dialect.is_delimited_identifier_start(ch)
1265                    && !self.dialect.is_identifier_start(ch) =>
1266                {
1267                    if self.dialect.supports_triple_quoted_string() {
1268                        return self
1269                            .tokenize_single_or_triple_quoted_string::<fn(String) -> Token>(
1270                                chars,
1271                                '"',
1272                                self.dialect.supports_string_literal_backslash_escape(),
1273                                Token::DoubleQuotedString,
1274                                Token::TripleDoubleQuotedString,
1275                            );
1276                    }
1277                    let s = self.tokenize_single_quoted_string(
1278                        chars,
1279                        '"',
1280                        self.dialect.supports_string_literal_backslash_escape(),
1281                    )?;
1282
1283                    Ok(Some(Token::DoubleQuotedString(s)))
1284                }
1285                // delimited (quoted) identifier
1286                quote_start if self.dialect.is_delimited_identifier_start(ch) => {
1287                    let word = self.tokenize_quoted_identifier(quote_start, chars)?;
1288                    Ok(Some(Token::make_word_owned(word, Some(quote_start))))
1289                }
1290                // Potentially nested delimited (quoted) identifier
1291                quote_start
1292                    if self
1293                        .dialect
1294                        .is_nested_delimited_identifier_start(quote_start)
1295                        && self
1296                            .dialect
1297                            .peek_nested_delimited_identifier_quotes(chars.peekable.clone())
1298                            .is_some() =>
1299                {
1300                    let Some((quote_start, nested_quote_start)) = self
1301                        .dialect
1302                        .peek_nested_delimited_identifier_quotes(chars.peekable.clone())
1303                    else {
1304                        return self.tokenizer_error(
1305                            chars.location(),
1306                            format!("Expected nested delimiter '{quote_start}' before EOF."),
1307                        );
1308                    };
1309
1310                    let Some(nested_quote_start) = nested_quote_start else {
1311                        let word = self.tokenize_quoted_identifier(quote_start, chars)?;
1312                        return Ok(Some(Token::make_word_owned(word, Some(quote_start))));
1313                    };
1314
1315                    let mut word = vec![];
1316                    let quote_end = Word::matching_end_quote(quote_start);
1317                    let nested_quote_end = Word::matching_end_quote(nested_quote_start);
1318                    let error_loc = chars.location();
1319
1320                    chars.next(); // skip the first delimiter
1321                    peeking_take_while(chars, |ch| ch.is_whitespace());
1322                    if chars.peek() != Some(&nested_quote_start) {
1323                        return self.tokenizer_error(
1324                            error_loc,
1325                            format!("Expected nested delimiter '{nested_quote_start}' before EOF."),
1326                        );
1327                    }
1328                    word.push(nested_quote_start.into());
1329                    word.push(self.tokenize_quoted_identifier(nested_quote_end, chars)?);
1330                    word.push(nested_quote_end.into());
1331                    peeking_take_while(chars, |ch| ch.is_whitespace());
1332                    if chars.peek() != Some(&quote_end) {
1333                        return self.tokenizer_error(
1334                            error_loc,
1335                            format!("Expected close delimiter '{quote_end}' before EOF."),
1336                        );
1337                    }
1338                    chars.next(); // skip close delimiter
1339
1340                    Ok(Some(Token::make_word_owned(
1341                        word.concat(),
1342                        Some(quote_start),
1343                    )))
1344                }
1345                // numbers and period
1346                '0'..='9' | '.' => {
1347                    // special case where if ._ is encountered after a word then that word
1348                    // is a table and the _ is the start of the col name.
1349                    // if the prev token is not a word, then this is not a valid sql
1350                    // word or number.
1351                    if ch == '.' && chars.peekable.clone().nth(1) == Some('_') {
1352                        if let Some(Token::Word(_)) = prev_token {
1353                            chars.next();
1354                            return Ok(Some(Token::Period));
1355                        }
1356
1357                        return self.tokenizer_error(
1358                            chars.location(),
1359                            "Unexpected character '_'".to_string(),
1360                        );
1361                    }
1362
1363                    let mut s = self.tokenize_number_part(chars, |ch| ch.is_ascii_digit())?;
1364
1365                    // match binary literal that starts with 0x
1366                    if s == "0" && chars.peek() == Some(&'x') {
1367                        chars.next();
1368                        let s2 = self.tokenize_number_part(chars, |ch| ch.is_ascii_hexdigit())?;
1369                        return Ok(Some(Token::HexStringLiteral(s2)));
1370                    }
1371
1372                    // match one period
1373                    if let Some('.') = chars.peek() {
1374                        s.push('.');
1375                        chars.next();
1376                    }
1377
1378                    // If the dialect supports identifiers that start with a numeric prefix
1379                    // and we have now consumed a dot, check if the previous token was a Word.
1380                    // If so, what follows is definitely not part of a decimal number and
1381                    // we should yield the dot as a dedicated token so compound identifiers
1382                    // starting with digits can be parsed correctly.
1383                    if s == "." && self.dialect.supports_numeric_prefix() {
1384                        if let Some(Token::Word(_)) = prev_token {
1385                            return Ok(Some(Token::Period));
1386                        }
1387                    }
1388
1389                    // Consume fractional digits.
1390                    s += &self.tokenize_number_part(chars, |ch| ch.is_ascii_digit())?;
1391
1392                    // No fraction -> Token::Period
1393                    if s == "." {
1394                        return Ok(Some(Token::Period));
1395                    }
1396
1397                    // Parse exponent as number
1398                    let mut exponent_part = String::new();
1399                    if chars.peek() == Some(&'e') || chars.peek() == Some(&'E') {
1400                        let mut char_clone = chars.peekable.clone();
1401                        exponent_part.push(char_clone.next().unwrap());
1402
1403                        // Optional sign
1404                        match char_clone.peek() {
1405                            Some(&c) if matches!(c, '+' | '-') => {
1406                                exponent_part.push(c);
1407                                char_clone.next();
1408                            }
1409                            _ => (),
1410                        }
1411
1412                        match char_clone.peek() {
1413                            // Definitely an exponent, get original iterator up to speed and use it
1414                            Some(&c)
1415                                if c.is_ascii_digit()
1416                                    || (c == '_'
1417                                        && self.dialect.supports_numeric_literal_underscores()) =>
1418                            {
1419                                for _ in 0..exponent_part.len() {
1420                                    chars.next();
1421                                }
1422                                exponent_part +=
1423                                    &self.tokenize_number_part(chars, |ch| ch.is_ascii_digit())?;
1424                                s += exponent_part.as_str();
1425                            }
1426                            // Not an exponent, discard the work done
1427                            _ => (),
1428                        }
1429                    }
1430
1431                    // If the dialect supports identifiers that start with a numeric prefix,
1432                    // we need to check if the value is in fact an identifier and must thus
1433                    // be tokenized as a word.
1434                    if self.dialect.supports_numeric_prefix() {
1435                        if exponent_part.is_empty() {
1436                            // If it is not a number with an exponent, it may be
1437                            // an identifier starting with digits.
1438                            let word =
1439                                peeking_take_while(chars, |ch| self.dialect.is_identifier_part(ch));
1440
1441                            if !word.is_empty() {
1442                                s += word.as_str();
1443                                return Ok(Some(Token::make_word_owned(s, None)));
1444                            }
1445                        } else if prev_token == Some(&Token::Period) {
1446                            // If the previous token was a period, thus not belonging to a number,
1447                            // the value we have is part of an identifier.
1448                            return Ok(Some(Token::make_word_owned(s, None)));
1449                        }
1450                    }
1451
1452                    let long = if chars.peek() == Some(&'L') {
1453                        chars.next();
1454                        true
1455                    } else {
1456                        false
1457                    };
1458                    Ok(Some(Token::Number(s, long)))
1459                }
1460                // punctuation
1461                '(' => self.consume_and_return(chars, Token::LParen),
1462                ')' => self.consume_and_return(chars, Token::RParen),
1463                ',' => self.consume_and_return(chars, Token::Comma),
1464                // operators
1465                '-' => {
1466                    chars.next(); // consume the '-'
1467
1468                    match chars.peek() {
1469                        Some('-') => {
1470                            let mut is_comment = true;
1471                            if self.dialect.requires_single_line_comment_whitespace() {
1472                                is_comment = chars
1473                                    .peekable
1474                                    .clone()
1475                                    .nth(1)
1476                                    .is_some_and(char::is_whitespace);
1477                            }
1478
1479                            if is_comment {
1480                                chars.next(); // consume second '-'
1481                                let comment = self.tokenize_single_line_comment(chars);
1482                                return Ok(Some(Token::Whitespace(
1483                                    Whitespace::SingleLineComment {
1484                                        prefix: "--".to_owned(),
1485                                        comment,
1486                                    },
1487                                )));
1488                            }
1489
1490                            self.start_binop(chars, "-", Token::Minus)
1491                        }
1492                        Some('>') => {
1493                            chars.next();
1494                            match chars.peek() {
1495                                Some('>') => self.consume_for_binop(chars, "->>", Token::LongArrow),
1496                                _ => self.start_binop(chars, "->", Token::Arrow),
1497                            }
1498                        }
1499                        // a regular '-' operator
1500                        _ => self.start_binop(chars, "-", Token::Minus),
1501                    }
1502                }
1503                '/' => {
1504                    chars.next(); // consume the '/'
1505                    match chars.peek() {
1506                        Some('*') => {
1507                            chars.next(); // consume the '*', starting a multi-line comment
1508                            self.tokenize_multiline_comment(chars)
1509                        }
1510                        Some('/') if dialect_of!(self is SnowflakeDialect) => {
1511                            chars.next(); // consume the second '/', starting a snowflake single-line comment
1512                            let comment = self.tokenize_single_line_comment(chars);
1513                            Ok(Some(Token::Whitespace(Whitespace::SingleLineComment {
1514                                prefix: "//".to_owned(),
1515                                comment,
1516                            })))
1517                        }
1518                        Some('/') if dialect_of!(self is DuckDbDialect | GenericDialect) => {
1519                            self.consume_and_return(chars, Token::DuckIntDiv)
1520                        }
1521                        // a regular '/' operator
1522                        _ => Ok(Some(Token::Div)),
1523                    }
1524                }
1525                '+' => self.consume_and_return(chars, Token::Plus),
1526                '*' => self.consume_and_return(chars, Token::Mul),
1527                '%' => {
1528                    chars.next(); // advance past '%'
1529                    match chars.peek() {
1530                        Some(s) if s.is_whitespace() => Ok(Some(Token::Mod)),
1531                        Some(sch) if self.dialect.is_identifier_start('%') => {
1532                            self.tokenize_identifier_or_keyword([ch, *sch], chars)
1533                        }
1534                        _ => self.start_binop(chars, "%", Token::Mod),
1535                    }
1536                }
1537                '|' => {
1538                    chars.next(); // consume the '|'
1539                    match chars.peek() {
1540                        Some('/') => self.consume_for_binop(chars, "|/", Token::PGSquareRoot),
1541                        Some('|') => {
1542                            chars.next(); // consume the second '|'
1543                            match chars.peek() {
1544                                Some('/') => {
1545                                    self.consume_for_binop(chars, "||/", Token::PGCubeRoot)
1546                                }
1547                                _ => self.start_binop(chars, "||", Token::StringConcat),
1548                            }
1549                        }
1550                        Some('&') if self.dialect.supports_geometric_types() => {
1551                            chars.next(); // consume
1552                            match chars.peek() {
1553                                Some('>') => self.consume_for_binop(
1554                                    chars,
1555                                    "|&>",
1556                                    Token::VerticalBarAmpersandRightAngleBracket,
1557                                ),
1558                                _ => self.start_binop_opt(chars, "|&", None),
1559                            }
1560                        }
1561                        Some('>') if self.dialect.supports_geometric_types() => {
1562                            chars.next(); // consume
1563                            match chars.peek() {
1564                                Some('>') => self.consume_for_binop(
1565                                    chars,
1566                                    "|>>",
1567                                    Token::VerticalBarShiftRight,
1568                                ),
1569                                _ => self.start_binop_opt(chars, "|>", None),
1570                            }
1571                        }
1572                        Some('>') if self.dialect.supports_pipe_operator() => {
1573                            self.consume_for_binop(chars, "|>", Token::VerticalBarRightAngleBracket)
1574                        }
1575                        // Bitshift '|' operator
1576                        _ => self.start_binop(chars, "|", Token::Pipe),
1577                    }
1578                }
1579                '=' => {
1580                    chars.next(); // consume
1581                    match chars.peek() {
1582                        Some('>') => self.consume_and_return(chars, Token::RArrow),
1583                        Some('=') => self.consume_and_return(chars, Token::DoubleEq),
1584                        _ => Ok(Some(Token::Eq)),
1585                    }
1586                }
1587                '!' => {
1588                    chars.next(); // consume
1589                    match chars.peek() {
1590                        Some('=') => self.consume_and_return(chars, Token::Neq),
1591                        Some('!') => self.consume_and_return(chars, Token::DoubleExclamationMark),
1592                        Some('~') => {
1593                            chars.next();
1594                            match chars.peek() {
1595                                Some('*') => self
1596                                    .consume_and_return(chars, Token::ExclamationMarkTildeAsterisk),
1597                                Some('~') => {
1598                                    chars.next();
1599                                    match chars.peek() {
1600                                        Some('*') => self.consume_and_return(
1601                                            chars,
1602                                            Token::ExclamationMarkDoubleTildeAsterisk,
1603                                        ),
1604                                        _ => Ok(Some(Token::ExclamationMarkDoubleTilde)),
1605                                    }
1606                                }
1607                                _ => Ok(Some(Token::ExclamationMarkTilde)),
1608                            }
1609                        }
1610                        _ => Ok(Some(Token::ExclamationMark)),
1611                    }
1612                }
1613                '<' => {
1614                    chars.next(); // consume
1615                    match chars.peek() {
1616                        Some('=') => {
1617                            chars.next();
1618                            match chars.peek() {
1619                                Some('>') => self.consume_for_binop(chars, "<=>", Token::Spaceship),
1620                                // `<=+` and `<=-` are not valid combined operators; treat `<=` as
1621                                // the operator and leave `+`/`-` to be tokenized separately.
1622                                Some('+') | Some('-') => Ok(Some(Token::LtEq)),
1623                                _ => self.start_binop(chars, "<=", Token::LtEq),
1624                            }
1625                        }
1626                        Some('|') if self.dialect.supports_geometric_types() => {
1627                            self.consume_for_binop(chars, "<<|", Token::ShiftLeftVerticalBar)
1628                        }
1629                        Some('>') => self.consume_for_binop(chars, "<>", Token::Neq),
1630                        Some('<') if self.dialect.supports_geometric_types() => {
1631                            chars.next(); // consume
1632                            match chars.peek() {
1633                                Some('|') => self.consume_for_binop(
1634                                    chars,
1635                                    "<<|",
1636                                    Token::ShiftLeftVerticalBar,
1637                                ),
1638                                _ => self.start_binop(chars, "<<", Token::ShiftLeft),
1639                            }
1640                        }
1641                        Some('<') => self.consume_for_binop(chars, "<<", Token::ShiftLeft),
1642                        // `<+` is not a valid combined operator; treat `<` as the operator
1643                        // and leave `+` to be tokenized separately.
1644                        Some('+') => Ok(Some(Token::Lt)),
1645                        Some('-') if self.dialect.supports_geometric_types() => {
1646                            if chars.peekable.clone().nth(1) == Some('>') {
1647                                chars.next(); // consume `-`
1648                                self.consume_for_binop(chars, "<->", Token::TwoWayArrow)
1649                            } else {
1650                                Ok(Some(Token::Lt))
1651                            }
1652                        }
1653                        Some('^') if self.dialect.supports_geometric_types() => {
1654                            self.consume_for_binop(chars, "<^", Token::LeftAngleBracketCaret)
1655                        }
1656                        Some('@') => self.consume_for_binop(chars, "<@", Token::ArrowAt),
1657                        _ => self.start_binop(chars, "<", Token::Lt),
1658                    }
1659                }
1660                '>' => {
1661                    chars.next(); // consume
1662                    match chars.peek() {
1663                        Some('=') => self.consume_for_binop(chars, ">=", Token::GtEq),
1664                        Some('>') => self.consume_for_binop(chars, ">>", Token::ShiftRight),
1665                        Some('^') if self.dialect.supports_geometric_types() => {
1666                            self.consume_for_binop(chars, ">^", Token::RightAngleBracketCaret)
1667                        }
1668                        _ => self.start_binop(chars, ">", Token::Gt),
1669                    }
1670                }
1671                ':' => {
1672                    chars.next();
1673                    match chars.peek() {
1674                        Some(':') => self.consume_and_return(chars, Token::DoubleColon),
1675                        Some('=') => self.consume_and_return(chars, Token::Assignment),
1676                        _ => Ok(Some(Token::Colon)),
1677                    }
1678                }
1679                ';' => self.consume_and_return(chars, Token::SemiColon),
1680                '\\' => self.consume_and_return(chars, Token::Backslash),
1681                '[' => self.consume_and_return(chars, Token::LBracket),
1682                ']' => self.consume_and_return(chars, Token::RBracket),
1683                '&' => {
1684                    chars.next(); // consume the '&'
1685                    match chars.peek() {
1686                        Some('>') if self.dialect.supports_geometric_types() => {
1687                            chars.next();
1688                            self.consume_and_return(chars, Token::AmpersandRightAngleBracket)
1689                        }
1690                        Some('<') if self.dialect.supports_geometric_types() => {
1691                            chars.next(); // consume
1692                            match chars.peek() {
1693                                Some('|') => self.consume_and_return(
1694                                    chars,
1695                                    Token::AmpersandLeftAngleBracketVerticalBar,
1696                                ),
1697                                _ => {
1698                                    self.start_binop(chars, "&<", Token::AmpersandLeftAngleBracket)
1699                                }
1700                            }
1701                        }
1702                        Some('&') => {
1703                            chars.next(); // consume the second '&'
1704                            self.start_binop(chars, "&&", Token::Overlap)
1705                        }
1706                        // Bitshift '&' operator
1707                        _ => self.start_binop(chars, "&", Token::Ampersand),
1708                    }
1709                }
1710                '^' => {
1711                    chars.next(); // consume the '^'
1712                    match chars.peek() {
1713                        Some('@') => self.consume_and_return(chars, Token::CaretAt),
1714                        _ => Ok(Some(Token::Caret)),
1715                    }
1716                }
1717                '{' => self.consume_and_return(chars, Token::LBrace),
1718                '}' => self.consume_and_return(chars, Token::RBrace),
1719                '#' if dialect_of!(self is SnowflakeDialect | BigQueryDialect | MySqlDialect | HiveDialect) =>
1720                {
1721                    chars.next(); // consume the '#', starting a snowflake single-line comment
1722                    let comment = self.tokenize_single_line_comment(chars);
1723                    Ok(Some(Token::Whitespace(Whitespace::SingleLineComment {
1724                        prefix: "#".to_owned(),
1725                        comment,
1726                    })))
1727                }
1728                '~' => {
1729                    chars.next(); // consume
1730                    match chars.peek() {
1731                        Some('*') => self.consume_for_binop(chars, "~*", Token::TildeAsterisk),
1732                        Some('=') if self.dialect.supports_geometric_types() => {
1733                            self.consume_for_binop(chars, "~=", Token::TildeEqual)
1734                        }
1735                        Some('~') => {
1736                            chars.next();
1737                            match chars.peek() {
1738                                Some('*') => {
1739                                    self.consume_for_binop(chars, "~~*", Token::DoubleTildeAsterisk)
1740                                }
1741                                _ => self.start_binop(chars, "~~", Token::DoubleTilde),
1742                            }
1743                        }
1744                        _ => self.start_binop(chars, "~", Token::Tilde),
1745                    }
1746                }
1747                '#' => {
1748                    chars.next();
1749                    match chars.peek() {
1750                        Some('-') => self.consume_for_binop(chars, "#-", Token::HashMinus),
1751                        Some('>') => {
1752                            chars.next();
1753                            match chars.peek() {
1754                                Some('>') => {
1755                                    self.consume_for_binop(chars, "#>>", Token::HashLongArrow)
1756                                }
1757                                _ => self.start_binop(chars, "#>", Token::HashArrow),
1758                            }
1759                        }
1760                        Some(' ') => Ok(Some(Token::Sharp)),
1761                        Some('#') if self.dialect.supports_geometric_types() => {
1762                            self.consume_for_binop(chars, "##", Token::DoubleSharp)
1763                        }
1764                        Some(sch) if self.dialect.is_identifier_start('#') => {
1765                            self.tokenize_identifier_or_keyword([ch, *sch], chars)
1766                        }
1767                        _ => self.start_binop(chars, "#", Token::Sharp),
1768                    }
1769                }
1770                '@' => {
1771                    chars.next();
1772                    match chars.peek() {
1773                        Some('@') if self.dialect.supports_geometric_types() => {
1774                            self.consume_and_return(chars, Token::AtAt)
1775                        }
1776                        Some('-') if self.dialect.supports_geometric_types() => {
1777                            chars.next();
1778                            match chars.peek() {
1779                                Some('@') => self.consume_and_return(chars, Token::AtDashAt),
1780                                _ => self.start_binop_opt(chars, "@-", None),
1781                            }
1782                        }
1783                        Some('>') => self.consume_and_return(chars, Token::AtArrow),
1784                        Some('?') => self.consume_and_return(chars, Token::AtQuestion),
1785                        Some('@') => {
1786                            chars.next();
1787                            match chars.peek() {
1788                                Some(' ') => Ok(Some(Token::AtAt)),
1789                                Some(tch) if self.dialect.is_identifier_start('@') => {
1790                                    self.tokenize_identifier_or_keyword([ch, '@', *tch], chars)
1791                                }
1792                                _ => Ok(Some(Token::AtAt)),
1793                            }
1794                        }
1795                        Some(' ') => Ok(Some(Token::AtSign)),
1796                        // We break on quotes here, because no dialect allows identifiers starting
1797                        // with @ and containing quotation marks (e.g. `@'foo'`) unless they are
1798                        // quoted, which is tokenized as a quoted string, not here (e.g.
1799                        // `"@'foo'"`). Further, at least two dialects parse `@` followed by a
1800                        // quoted string as two separate tokens, which this allows. For example,
1801                        // Postgres parses `@'1'` as the absolute value of '1' which is implicitly
1802                        // cast to a numeric type. And when parsing MySQL-style grantees (e.g.
1803                        // `GRANT ALL ON *.* to 'root'@'localhost'`), we also want separate tokens
1804                        // for the user, the `@`, and the host.
1805                        Some('\'') => Ok(Some(Token::AtSign)),
1806                        Some('\"') => Ok(Some(Token::AtSign)),
1807                        Some('`') => Ok(Some(Token::AtSign)),
1808                        Some(sch) if self.dialect.is_identifier_start('@') => {
1809                            self.tokenize_identifier_or_keyword([ch, *sch], chars)
1810                        }
1811                        _ => Ok(Some(Token::AtSign)),
1812                    }
1813                }
1814                // Postgres uses ? for jsonb operators, not prepared statements
1815                '?' if self.dialect.supports_geometric_types() => {
1816                    chars.next(); // consume
1817                    match chars.peek() {
1818                        Some('|') => {
1819                            chars.next();
1820                            match chars.peek() {
1821                                Some('|') => self.consume_and_return(
1822                                    chars,
1823                                    Token::QuestionMarkDoubleVerticalBar,
1824                                ),
1825                                _ => Ok(Some(Token::QuestionPipe)),
1826                            }
1827                        }
1828
1829                        Some('&') => self.consume_and_return(chars, Token::QuestionAnd),
1830                        Some('-') => {
1831                            chars.next(); // consume
1832                            match chars.peek() {
1833                                Some('|') => self
1834                                    .consume_and_return(chars, Token::QuestionMarkDashVerticalBar),
1835                                _ => Ok(Some(Token::QuestionMarkDash)),
1836                            }
1837                        }
1838                        Some('#') => self.consume_and_return(chars, Token::QuestionMarkSharp),
1839                        _ => Ok(Some(Token::Question)),
1840                    }
1841                }
1842                '?' => {
1843                    chars.next();
1844                    let s = peeking_take_while(chars, |ch| ch.is_numeric());
1845                    Ok(Some(Token::Placeholder(format!("?{s}"))))
1846                }
1847
1848                // identifier or keyword
1849                ch if self.dialect.is_identifier_start(ch) => {
1850                    self.tokenize_identifier_or_keyword([ch], chars)
1851                }
1852                '$' => Ok(Some(self.tokenize_dollar_preceded_value(chars)?)),
1853
1854                // whitespace check (including unicode chars) should be last as it covers some of the chars above
1855                ch if ch.is_whitespace() => {
1856                    self.consume_and_return(chars, Token::Whitespace(Whitespace::Space))
1857                }
1858                other => self.consume_and_return(chars, Token::Char(other)),
1859            },
1860            None => Ok(None),
1861        }
1862    }
1863
1864    /// Consume the next character, then parse a custom binary operator. The next character should be included in the prefix
1865    fn consume_for_binop(
1866        &self,
1867        chars: &mut State,
1868        prefix: &str,
1869        default: Token,
1870    ) -> Result<Option<Token>, TokenizerError> {
1871        chars.next(); // consume the first char
1872        self.start_binop_opt(chars, prefix, Some(default))
1873    }
1874
1875    /// parse a custom binary operator
1876    fn start_binop(
1877        &self,
1878        chars: &mut State,
1879        prefix: &str,
1880        default: Token,
1881    ) -> Result<Option<Token>, TokenizerError> {
1882        self.start_binop_opt(chars, prefix, Some(default))
1883    }
1884
1885    /// parse a custom binary operator
1886    fn start_binop_opt(
1887        &self,
1888        chars: &mut State,
1889        prefix: &str,
1890        default: Option<Token>,
1891    ) -> Result<Option<Token>, TokenizerError> {
1892        let mut custom = None;
1893        while let Some(&ch) = chars.peek() {
1894            if !self.dialect.is_custom_operator_part(ch) {
1895                break;
1896            }
1897
1898            custom.get_or_insert_with(|| prefix.to_string()).push(ch);
1899            chars.next();
1900        }
1901        match (custom, default) {
1902            (Some(custom), _) => Ok(Token::CustomBinaryOperator(custom).into()),
1903            (None, Some(tok)) => Ok(Some(tok)),
1904            (None, None) => self.tokenizer_error(
1905                chars.location(),
1906                format!("Expected a valid binary operator after '{prefix}'"),
1907            ),
1908        }
1909    }
1910
1911    /// Tokenize dollar preceded value (i.e: a string/placeholder)
1912    fn tokenize_dollar_preceded_value(&self, chars: &mut State) -> Result<Token, TokenizerError> {
1913        let mut s = String::new();
1914        let mut value = String::new();
1915
1916        chars.next();
1917
1918        // If the dialect does not support dollar-quoted strings, then `$$` is rather a placeholder.
1919        if matches!(chars.peek(), Some('$')) && !self.dialect.supports_dollar_placeholder() {
1920            chars.next();
1921
1922            let mut is_terminated = false;
1923            let mut prev: Option<char> = None;
1924
1925            while let Some(&ch) = chars.peek() {
1926                if prev == Some('$') {
1927                    if ch == '$' {
1928                        chars.next();
1929                        is_terminated = true;
1930                        break;
1931                    } else {
1932                        s.push('$');
1933                        s.push(ch);
1934                    }
1935                } else if ch != '$' {
1936                    s.push(ch);
1937                }
1938
1939                prev = Some(ch);
1940                chars.next();
1941            }
1942
1943            return if chars.peek().is_none() && !is_terminated {
1944                self.tokenizer_error(chars.location(), "Unterminated dollar-quoted string")
1945            } else {
1946                Ok(Token::DollarQuotedString(DollarQuotedString {
1947                    value: s,
1948                    tag: None,
1949                }))
1950            };
1951        } else {
1952            value.push_str(&peeking_take_while(chars, |ch| {
1953                ch.is_alphanumeric()
1954                    || ch == '_'
1955                    // Allow $ as a placeholder character if the dialect supports it
1956                    || matches!(ch, '$' if self.dialect.supports_dollar_placeholder())
1957            }));
1958
1959            // If the dialect supports a dollar sign as a money prefix (e.g. SQL Server),
1960            // and the value so far is all digits, check for a decimal part, e.g. `$123.45`
1961            if matches!(chars.peek(), Some('.'))
1962                && self.dialect.supports_dollar_as_money_prefix()
1963                && !value.is_empty()
1964                && value.chars().all(|c| c.is_ascii_digit())
1965            {
1966                value.push('.');
1967                chars.next();
1968                value.push_str(&peeking_take_while(chars, |ch| ch.is_ascii_digit()));
1969                return Ok(Token::Placeholder(format!("${value}")));
1970            }
1971
1972            // If the dialect does not support dollar-quoted strings, don't look for the end delimiter.
1973            if matches!(chars.peek(), Some('$')) && !self.dialect.supports_dollar_placeholder() {
1974                chars.next();
1975
1976                let mut temp = String::new();
1977                let end_delimiter = format!("${value}$");
1978
1979                loop {
1980                    match chars.next() {
1981                        Some(ch) => {
1982                            temp.push(ch);
1983
1984                            if temp.ends_with(&end_delimiter) {
1985                                if let Some(temp) = temp.strip_suffix(&end_delimiter) {
1986                                    s.push_str(temp);
1987                                }
1988                                break;
1989                            }
1990                        }
1991                        None => {
1992                            if temp.ends_with(&end_delimiter) {
1993                                if let Some(temp) = temp.strip_suffix(&end_delimiter) {
1994                                    s.push_str(temp);
1995                                }
1996                                break;
1997                            }
1998
1999                            return self.tokenizer_error(
2000                                chars.location(),
2001                                "Unterminated dollar-quoted, expected $",
2002                            );
2003                        }
2004                    }
2005                }
2006            } else {
2007                return Ok(Token::Placeholder(format!("${value}")));
2008            }
2009        }
2010
2011        Ok(Token::DollarQuotedString(DollarQuotedString {
2012            value: s,
2013            tag: if value.is_empty() { None } else { Some(value) },
2014        }))
2015    }
2016
2017    fn tokenizer_error<R>(
2018        &self,
2019        loc: Location,
2020        message: impl Into<String>,
2021    ) -> Result<R, TokenizerError> {
2022        Err(TokenizerError {
2023            message: message.into(),
2024            location: loc,
2025        })
2026    }
2027
2028    fn tokenize_number_part(
2029        &self,
2030        chars: &mut State,
2031        is_digit: impl Fn(char) -> bool,
2032    ) -> Result<String, TokenizerError> {
2033        let supports_separator = self.dialect.supports_numeric_literal_underscores();
2034        let mut s = String::new();
2035
2036        while let Some(&ch) = chars.peek() {
2037            if is_digit(ch) {
2038                chars.next();
2039                s.push(ch);
2040            } else if supports_separator && ch == '_' {
2041                let next_char = chars.peekable.clone().nth(1);
2042                if s.is_empty() || !next_char.is_some_and(&is_digit) {
2043                    return self.tokenizer_error(chars.location(), "Unexpected character '_'");
2044                }
2045                chars.next();
2046                s.push(ch);
2047            } else {
2048                break;
2049            }
2050        }
2051
2052        Ok(s)
2053    }
2054
2055    // Consume characters until newline
2056    fn tokenize_single_line_comment(&self, chars: &mut State) -> String {
2057        peeking_take_while(chars, |ch| match ch {
2058            '\n' => false,                                           // Always stop at \n
2059            '\r' if dialect_of!(self is PostgreSqlDialect) => false, // Stop at \r for Postgres
2060            _ => true, // Keep consuming for other characters
2061        })
2062    }
2063
2064    /// Tokenize an identifier or keyword, after the first char is already consumed.
2065    fn tokenize_word(&self, first_chars: impl Into<String>, chars: &mut State) -> String {
2066        let mut s = first_chars.into();
2067        s.push_str(&peeking_take_while(chars, |ch| {
2068            self.dialect.is_identifier_part(ch)
2069        }));
2070        s
2071    }
2072
2073    /// Read a quoted identifier
2074    fn tokenize_quoted_identifier(
2075        &self,
2076        quote_start: char,
2077        chars: &mut State,
2078    ) -> Result<String, TokenizerError> {
2079        let error_loc = chars.location();
2080        chars.next(); // consume the opening quote
2081        let quote_end = Word::matching_end_quote(quote_start);
2082        let (s, last_char) = self.parse_quoted_ident(chars, quote_end);
2083
2084        if last_char == Some(quote_end) {
2085            Ok(s)
2086        } else {
2087            self.tokenizer_error(
2088                error_loc,
2089                format!("Expected close delimiter '{quote_end}' before EOF."),
2090            )
2091        }
2092    }
2093
2094    /// Read a single quoted string, starting with the opening quote.
2095    fn tokenize_escaped_single_quoted_string(
2096        &self,
2097        starting_loc: Location,
2098        chars: &mut State,
2099    ) -> Result<String, TokenizerError> {
2100        if let Some(s) = unescape_single_quoted_string(chars) {
2101            return Ok(s);
2102        }
2103
2104        self.tokenizer_error(starting_loc, "Unterminated encoded string literal")
2105    }
2106
2107    /// Reads a string literal quoted by a single or triple quote characters.
2108    /// Examples: `'abc'`, `'''abc'''`, `"""abc"""`.
2109    fn tokenize_single_or_triple_quoted_string<F>(
2110        &self,
2111        chars: &mut State,
2112        quote_style: char,
2113        backslash_escape: bool,
2114        single_quote_token: F,
2115        triple_quote_token: F,
2116    ) -> Result<Option<Token>, TokenizerError>
2117    where
2118        F: Fn(String) -> Token,
2119    {
2120        let error_loc = chars.location();
2121
2122        let mut num_opening_quotes = 0u8;
2123        for _ in 0..3 {
2124            if Some(&quote_style) == chars.peek() {
2125                chars.next(); // Consume quote.
2126                num_opening_quotes += 1;
2127            } else {
2128                break;
2129            }
2130        }
2131
2132        let (token_fn, num_quote_chars) = match num_opening_quotes {
2133            1 => (single_quote_token, NumStringQuoteChars::One),
2134            2 => {
2135                // If we matched double quotes, then this is an empty string.
2136                return Ok(Some(single_quote_token("".into())));
2137            }
2138            3 => {
2139                let Some(num_quote_chars) = NonZeroU8::new(3) else {
2140                    return self.tokenizer_error(error_loc, "invalid number of opening quotes");
2141                };
2142                (
2143                    triple_quote_token,
2144                    NumStringQuoteChars::Many(num_quote_chars),
2145                )
2146            }
2147            _ => {
2148                return self.tokenizer_error(error_loc, "invalid string literal opening");
2149            }
2150        };
2151
2152        let settings = TokenizeQuotedStringSettings {
2153            quote_style,
2154            num_quote_chars,
2155            num_opening_quotes_to_consume: 0,
2156            backslash_escape,
2157        };
2158
2159        self.tokenize_quoted_string(chars, settings)
2160            .map(token_fn)
2161            .map(Some)
2162    }
2163
2164    /// Reads a string literal quoted by a single quote character.
2165    fn tokenize_single_quoted_string(
2166        &self,
2167        chars: &mut State,
2168        quote_style: char,
2169        backslash_escape: bool,
2170    ) -> Result<String, TokenizerError> {
2171        self.tokenize_quoted_string(
2172            chars,
2173            TokenizeQuotedStringSettings {
2174                quote_style,
2175                num_quote_chars: NumStringQuoteChars::One,
2176                num_opening_quotes_to_consume: 1,
2177                backslash_escape,
2178            },
2179        )
2180    }
2181
2182    /// Reads a quote delimited string expecting `chars.next()` to deliver a quote.
2183    ///
2184    /// See <https://docs.oracle.com/en/database/oracle/oracle-database/21/sqlrf/Literals.html#GUID-1824CBAA-6E16-4921-B2A6-112FB02248DA>
2185    fn tokenize_quote_delimited_string(
2186        &self,
2187        chars: &mut State,
2188        // the prefix that introduced the possible literal or word,
2189        // e.g. "Q" or "nq"
2190        literal_prefix: &[char],
2191    ) -> Result<QuoteDelimitedString, TokenizerError> {
2192        let literal_start_loc = chars.location();
2193        chars.next();
2194
2195        let start_quote_loc = chars.location();
2196        let (start_quote, end_quote) = match chars.next() {
2197            None | Some(' ') | Some('\t') | Some('\r') | Some('\n') => {
2198                return self.tokenizer_error(
2199                    start_quote_loc,
2200                    format!(
2201                        "Invalid space, tab, newline, or EOF after '{}''",
2202                        String::from_iter(literal_prefix)
2203                    ),
2204                );
2205            }
2206            Some(c) => (
2207                c,
2208                match c {
2209                    '[' => ']',
2210                    '{' => '}',
2211                    '<' => '>',
2212                    '(' => ')',
2213                    c => c,
2214                },
2215            ),
2216        };
2217
2218        // read the string literal until the "quote character" following a by literal quote
2219        let mut value = String::new();
2220        while let Some(ch) = chars.next() {
2221            if ch == end_quote {
2222                if let Some('\'') = chars.peek() {
2223                    chars.next(); // ~ consume the quote
2224                    return Ok(QuoteDelimitedString {
2225                        start_quote,
2226                        value,
2227                        end_quote,
2228                    });
2229                }
2230            }
2231            value.push(ch);
2232        }
2233
2234        self.tokenizer_error(literal_start_loc, "Unterminated string literal")
2235    }
2236
2237    /// Read a quoted string.
2238    fn tokenize_quoted_string(
2239        &self,
2240        chars: &mut State,
2241        settings: TokenizeQuotedStringSettings,
2242    ) -> Result<String, TokenizerError> {
2243        let mut s = String::new();
2244        let error_loc = chars.location();
2245
2246        // Consume any opening quotes.
2247        for _ in 0..settings.num_opening_quotes_to_consume {
2248            if Some(settings.quote_style) != chars.next() {
2249                return self.tokenizer_error(error_loc, "invalid string literal opening");
2250            }
2251        }
2252
2253        let mut num_consecutive_quotes = 0;
2254        while let Some(&ch) = chars.peek() {
2255            let pending_final_quote = match settings.num_quote_chars {
2256                NumStringQuoteChars::One => Some(NumStringQuoteChars::One),
2257                n @ NumStringQuoteChars::Many(count)
2258                    if num_consecutive_quotes + 1 == count.get() =>
2259                {
2260                    Some(n)
2261                }
2262                NumStringQuoteChars::Many(_) => None,
2263            };
2264
2265            match ch {
2266                char if char == settings.quote_style && pending_final_quote.is_some() => {
2267                    chars.next(); // consume
2268
2269                    if let Some(NumStringQuoteChars::Many(count)) = pending_final_quote {
2270                        // For an initial string like `"""abc"""`, at this point we have
2271                        // `abc""` in the buffer and have now matched the final `"`.
2272                        // However, the string to return is simply `abc`, so we strip off
2273                        // the trailing quotes before returning.
2274                        let mut buf = s.chars();
2275                        for _ in 1..count.get() {
2276                            buf.next_back();
2277                        }
2278                        return Ok(buf.as_str().to_string());
2279                    } else if chars
2280                        .peek()
2281                        .map(|c| *c == settings.quote_style)
2282                        .unwrap_or(false)
2283                    {
2284                        s.push(ch);
2285                        if !self.unescape {
2286                            // In no-escape mode, the given query has to be saved completely
2287                            s.push(ch);
2288                        }
2289                        chars.next();
2290                    } else {
2291                        return Ok(s);
2292                    }
2293                }
2294                '\\' if settings.backslash_escape => {
2295                    // consume backslash
2296                    chars.next();
2297
2298                    num_consecutive_quotes = 0;
2299
2300                    if let Some(next) = chars.peek() {
2301                        if !self.unescape
2302                            || (self.dialect.ignores_wildcard_escapes()
2303                                && (*next == '%' || *next == '_'))
2304                        {
2305                            // In no-escape mode, the given query has to be saved completely
2306                            // including backslashes. Similarly, with ignore_like_wildcard_escapes,
2307                            // the backslash is not stripped.
2308                            s.push(ch);
2309                            s.push(*next);
2310                            chars.next(); // consume next
2311                        } else {
2312                            let n = match next {
2313                                '0' => '\0',
2314                                'a' => '\u{7}',
2315                                'b' => '\u{8}',
2316                                'f' => '\u{c}',
2317                                'n' => '\n',
2318                                'r' => '\r',
2319                                't' => '\t',
2320                                'Z' => '\u{1a}',
2321                                _ => *next,
2322                            };
2323                            s.push(n);
2324                            chars.next(); // consume next
2325                        }
2326                    }
2327                }
2328                ch => {
2329                    chars.next(); // consume ch
2330
2331                    if ch == settings.quote_style {
2332                        num_consecutive_quotes += 1;
2333                    } else {
2334                        num_consecutive_quotes = 0;
2335                    }
2336
2337                    s.push(ch);
2338                }
2339            }
2340        }
2341        self.tokenizer_error(error_loc, "Unterminated string literal")
2342    }
2343
2344    fn tokenize_multiline_comment(
2345        &self,
2346        chars: &mut State,
2347    ) -> Result<Option<Token>, TokenizerError> {
2348        let mut s = String::new();
2349        let mut nested = 1;
2350        let supports_nested_comments = self.dialect.supports_nested_comments();
2351        loop {
2352            match chars.next() {
2353                Some('/') if matches!(chars.peek(), Some('*')) && supports_nested_comments => {
2354                    chars.next(); // consume the '*'
2355                    s.push('/');
2356                    s.push('*');
2357                    nested += 1;
2358                }
2359                Some('*') if matches!(chars.peek(), Some('/')) => {
2360                    chars.next(); // consume the '/'
2361                    nested -= 1;
2362                    if nested == 0 {
2363                        break Ok(Some(Token::Whitespace(Whitespace::MultiLineComment(s))));
2364                    }
2365                    s.push('*');
2366                    s.push('/');
2367                }
2368                Some(ch) => {
2369                    s.push(ch);
2370                }
2371                None => {
2372                    break self.tokenizer_error(
2373                        chars.location(),
2374                        "Unexpected EOF while in a multi-line comment",
2375                    );
2376                }
2377            }
2378        }
2379    }
2380
2381    fn parse_quoted_ident(&self, chars: &mut State, quote_end: char) -> (String, Option<char>) {
2382        let mut last_char = None;
2383        let mut s = String::new();
2384        while let Some(ch) = chars.next() {
2385            if ch == quote_end {
2386                if chars.peek() == Some(&quote_end) {
2387                    chars.next();
2388                    s.push(ch);
2389                    if !self.unescape {
2390                        // In no-escape mode, the given query has to be saved completely
2391                        s.push(ch);
2392                    }
2393                } else {
2394                    last_char = Some(quote_end);
2395                    break;
2396                }
2397            } else {
2398                s.push(ch);
2399            }
2400        }
2401        (s, last_char)
2402    }
2403
2404    #[allow(clippy::unnecessary_wraps)]
2405    fn consume_and_return(
2406        &self,
2407        chars: &mut State,
2408        t: Token,
2409    ) -> Result<Option<Token>, TokenizerError> {
2410        chars.next();
2411        Ok(Some(t))
2412    }
2413}
2414
2415/// Read from `chars` until `predicate` returns `false` or EOF is hit.
2416/// Return the characters read as String, and keep the first non-matching
2417/// char available as `chars.next()`.
2418fn peeking_take_while(chars: &mut State, mut predicate: impl FnMut(char) -> bool) -> String {
2419    let mut s = String::new();
2420    while let Some(&ch) = chars.peek() {
2421        if predicate(ch) {
2422            chars.next(); // consume
2423            s.push(ch);
2424        } else {
2425            break;
2426        }
2427    }
2428    s
2429}
2430
2431fn unescape_single_quoted_string(chars: &mut State<'_>) -> Option<String> {
2432    Unescape::new(chars).unescape()
2433}
2434
2435struct Unescape<'a: 'b, 'b> {
2436    chars: &'b mut State<'a>,
2437}
2438
2439impl<'a: 'b, 'b> Unescape<'a, 'b> {
2440    fn new(chars: &'b mut State<'a>) -> Self {
2441        Self { chars }
2442    }
2443    fn unescape(mut self) -> Option<String> {
2444        let mut unescaped = String::new();
2445
2446        self.chars.next();
2447
2448        while let Some(c) = self.chars.next() {
2449            if c == '\'' {
2450                // case: ''''
2451                if self.chars.peek().map(|c| *c == '\'').unwrap_or(false) {
2452                    self.chars.next();
2453                    unescaped.push('\'');
2454                    continue;
2455                }
2456                return Some(unescaped);
2457            }
2458
2459            if c != '\\' {
2460                unescaped.push(c);
2461                continue;
2462            }
2463
2464            let c = match self.chars.next()? {
2465                'b' => '\u{0008}',
2466                'f' => '\u{000C}',
2467                'n' => '\n',
2468                'r' => '\r',
2469                't' => '\t',
2470                'u' => self.unescape_unicode_16()?,
2471                'U' => self.unescape_unicode_32()?,
2472                'x' => self.unescape_hex()?,
2473                c if c.is_digit(8) => self.unescape_octal(c)?,
2474                c => c,
2475            };
2476
2477            unescaped.push(Self::check_null(c)?);
2478        }
2479
2480        None
2481    }
2482
2483    #[inline]
2484    fn check_null(c: char) -> Option<char> {
2485        if c == '\0' {
2486            None
2487        } else {
2488            Some(c)
2489        }
2490    }
2491
2492    #[inline]
2493    fn byte_to_char<const RADIX: u32>(s: &str) -> Option<char> {
2494        // u32 is used here because Pg has an overflow operation rather than throwing an exception directly.
2495        match u32::from_str_radix(s, RADIX) {
2496            Err(_) => None,
2497            Ok(n) => {
2498                let n = n & 0xFF;
2499                if n <= 127 {
2500                    char::from_u32(n)
2501                } else {
2502                    None
2503                }
2504            }
2505        }
2506    }
2507
2508    // Hexadecimal byte value. \xh, \xhh (h = 0–9, A–F)
2509    fn unescape_hex(&mut self) -> Option<char> {
2510        let mut s = String::new();
2511
2512        for _ in 0..2 {
2513            match self.next_hex_digit() {
2514                Some(c) => s.push(c),
2515                None => break,
2516            }
2517        }
2518
2519        if s.is_empty() {
2520            return Some('x');
2521        }
2522
2523        Self::byte_to_char::<16>(&s)
2524    }
2525
2526    #[inline]
2527    fn next_hex_digit(&mut self) -> Option<char> {
2528        match self.chars.peek() {
2529            Some(c) if c.is_ascii_hexdigit() => self.chars.next(),
2530            _ => None,
2531        }
2532    }
2533
2534    // Octal byte value. \o, \oo, \ooo (o = 0–7)
2535    fn unescape_octal(&mut self, c: char) -> Option<char> {
2536        let mut s = String::new();
2537
2538        s.push(c);
2539        for _ in 0..2 {
2540            match self.next_octal_digest() {
2541                Some(c) => s.push(c),
2542                None => break,
2543            }
2544        }
2545
2546        Self::byte_to_char::<8>(&s)
2547    }
2548
2549    #[inline]
2550    fn next_octal_digest(&mut self) -> Option<char> {
2551        match self.chars.peek() {
2552            Some(c) if c.is_digit(8) => self.chars.next(),
2553            _ => None,
2554        }
2555    }
2556
2557    // 16-bit hexadecimal Unicode character value. \uxxxx (x = 0–9, A–F)
2558    fn unescape_unicode_16(&mut self) -> Option<char> {
2559        self.unescape_unicode::<4>()
2560    }
2561
2562    // 32-bit hexadecimal Unicode character value. \Uxxxxxxxx (x = 0–9, A–F)
2563    fn unescape_unicode_32(&mut self) -> Option<char> {
2564        self.unescape_unicode::<8>()
2565    }
2566
2567    fn unescape_unicode<const NUM: usize>(&mut self) -> Option<char> {
2568        let mut s = String::new();
2569        for _ in 0..NUM {
2570            s.push(self.chars.next()?);
2571        }
2572        match u32::from_str_radix(&s, 16) {
2573            Err(_) => None,
2574            Ok(n) => char::from_u32(n),
2575        }
2576    }
2577}
2578
2579fn unescape_unicode_single_quoted_string(chars: &mut State<'_>) -> Result<String, TokenizerError> {
2580    let mut unescaped = String::new();
2581    chars.next(); // consume the opening quote
2582    while let Some(c) = chars.next() {
2583        match c {
2584            '\'' => {
2585                if chars.peek() == Some(&'\'') {
2586                    chars.next();
2587                    unescaped.push('\'');
2588                } else {
2589                    return Ok(unescaped);
2590                }
2591            }
2592            '\\' => match chars.peek() {
2593                Some('\\') => {
2594                    chars.next();
2595                    unescaped.push('\\');
2596                }
2597                Some('+') => {
2598                    chars.next();
2599                    unescaped.push(take_char_from_hex_digits(chars, 6)?);
2600                }
2601                _ => unescaped.push(take_char_from_hex_digits(chars, 4)?),
2602            },
2603            _ => {
2604                unescaped.push(c);
2605            }
2606        }
2607    }
2608    Err(TokenizerError {
2609        message: "Unterminated unicode encoded string literal".to_string(),
2610        location: chars.location(),
2611    })
2612}
2613
2614fn take_char_from_hex_digits(
2615    chars: &mut State<'_>,
2616    max_digits: usize,
2617) -> Result<char, TokenizerError> {
2618    let mut result = 0u32;
2619    for _ in 0..max_digits {
2620        let next_char = chars.next().ok_or_else(|| TokenizerError {
2621            message: "Unexpected EOF while parsing hex digit in escaped unicode string."
2622                .to_string(),
2623            location: chars.location(),
2624        })?;
2625        let digit = next_char.to_digit(16).ok_or_else(|| TokenizerError {
2626            message: format!("Invalid hex digit in escaped unicode string: {next_char}"),
2627            location: chars.location(),
2628        })?;
2629        result = result * 16 + digit;
2630    }
2631    char::from_u32(result).ok_or_else(|| TokenizerError {
2632        message: format!("Invalid unicode character: {result:x}"),
2633        location: chars.location(),
2634    })
2635}
2636
2637#[cfg(test)]
2638mod tests {
2639    use super::*;
2640    use crate::dialect::{
2641        BigQueryDialect, ClickHouseDialect, HiveDialect, MsSqlDialect, MySqlDialect,
2642        PostgreSqlDialect, SQLiteDialect,
2643    };
2644    use crate::test_utils::{all_dialects, all_dialects_except, all_dialects_where};
2645    use core::fmt::Debug;
2646
2647    #[test]
2648    fn tokenizer_error_impl() {
2649        let err = TokenizerError {
2650            message: "test".into(),
2651            location: Location { line: 1, column: 1 },
2652        };
2653        {
2654            use core::error::Error;
2655            assert!(err.source().is_none());
2656        }
2657        assert_eq!(err.to_string(), "test at Line: 1, Column: 1");
2658    }
2659
2660    #[test]
2661    fn tokenize_select_1() {
2662        let sql = String::from("SELECT 1");
2663        let dialect = GenericDialect {};
2664        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
2665
2666        let expected = vec![
2667            Token::make_keyword("SELECT"),
2668            Token::Whitespace(Whitespace::Space),
2669            Token::Number(String::from("1"), false),
2670        ];
2671
2672        compare(expected, tokens);
2673    }
2674
2675    #[test]
2676    fn tokenize_select_float() {
2677        let sql = String::from("SELECT .1");
2678        let dialect = GenericDialect {};
2679        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
2680
2681        let expected = vec![
2682            Token::make_keyword("SELECT"),
2683            Token::Whitespace(Whitespace::Space),
2684            Token::Number(String::from(".1"), false),
2685        ];
2686
2687        compare(expected, tokens);
2688    }
2689
2690    #[test]
2691    fn tokenize_with_mapper() {
2692        let sql = String::from("SELECT ?");
2693        let dialect = GenericDialect {};
2694        let mut param_num = 1;
2695
2696        let mut tokens = vec![];
2697        Tokenizer::new(&dialect, &sql)
2698            .tokenize_with_location_into_buf_with_mapper(&mut tokens, |mut token_span| {
2699                token_span.token = match token_span.token {
2700                    Token::Placeholder(n) => Token::Placeholder(if n == "?" {
2701                        let ret = format!("${}", param_num);
2702                        param_num += 1;
2703                        ret
2704                    } else {
2705                        n
2706                    }),
2707                    token => token,
2708                };
2709                token_span
2710            })
2711            .unwrap();
2712        let actual = tokens.into_iter().map(|t| t.token).collect();
2713        let expected = vec![
2714            Token::make_keyword("SELECT"),
2715            Token::Whitespace(Whitespace::Space),
2716            Token::Placeholder("$1".to_string()),
2717        ];
2718
2719        compare(expected, actual);
2720    }
2721
2722    #[test]
2723    fn tokenize_clickhouse_double_equal() {
2724        let sql = String::from("SELECT foo=='1'");
2725        let dialect = ClickHouseDialect {};
2726        let mut tokenizer = Tokenizer::new(&dialect, &sql);
2727        let tokens = tokenizer.tokenize().unwrap();
2728
2729        let expected = vec![
2730            Token::make_keyword("SELECT"),
2731            Token::Whitespace(Whitespace::Space),
2732            Token::Word(Word {
2733                value: "foo".to_string(),
2734                quote_style: None,
2735                keyword: Keyword::NoKeyword,
2736            }),
2737            Token::DoubleEq,
2738            Token::SingleQuotedString("1".to_string()),
2739        ];
2740
2741        compare(expected, tokens);
2742    }
2743
2744    #[test]
2745    fn tokenize_numeric_literal_underscore() {
2746        let dialect = GenericDialect {};
2747        let sql = String::from("SELECT 10_000");
2748        let mut tokenizer = Tokenizer::new(&dialect, &sql);
2749        let tokens = tokenizer.tokenize().unwrap();
2750        let expected = vec![
2751            Token::make_keyword("SELECT"),
2752            Token::Whitespace(Whitespace::Space),
2753            Token::Number("10".to_string(), false),
2754            Token::make_word("_000", None),
2755        ];
2756        compare(expected, tokens);
2757
2758        let numeric_underscore_dialects =
2759            all_dialects_where(|dialect| dialect.supports_numeric_literal_underscores());
2760
2761        numeric_underscore_dialects.tokenizes_to(
2762            "SELECT 10_000, _10_000, 1_000.123, 1_000.123_456, 1e1_0",
2763            vec![
2764                Token::make_keyword("SELECT"),
2765                Token::Whitespace(Whitespace::Space),
2766                Token::Number("10_000".to_string(), false),
2767                Token::Comma,
2768                Token::Whitespace(Whitespace::Space),
2769                Token::make_word("_10_000", None), // leading underscore tokenizes as a word (parsed as column identifier)
2770                Token::Comma,
2771                Token::Whitespace(Whitespace::Space),
2772                Token::Number("1_000.123".to_string(), false), // with decimal digits
2773                Token::Comma,
2774                Token::Whitespace(Whitespace::Space),
2775                Token::Number("1_000.123_456".to_string(), false), // with an underscore in the decimal digits
2776                Token::Comma,
2777                Token::Whitespace(Whitespace::Space),
2778                Token::Number("1e1_0".to_string(), false), // with an underscore in the exponent
2779            ],
2780        );
2781
2782        numeric_underscore_dialects.tokenizes_to(
2783            "0xFF_FF",
2784            vec![Token::HexStringLiteral("FF_FF".to_string())],
2785        );
2786
2787        for dialect in &numeric_underscore_dialects.dialects {
2788            for sql in [
2789                "SELECT 10_00_",
2790                "SELECT 10___0",
2791                "SELECT 1_000.123_",
2792                "SELECT 1._000",
2793                "SELECT 1_a",
2794                "SELECT 1e_1",
2795                "SELECT 1e1_",
2796                "SELECT 1e1__0",
2797                "SELECT 0x_1",
2798                "SELECT 0x1_",
2799            ] {
2800                let err = Tokenizer::new(&**dialect, sql).tokenize().unwrap_err();
2801                assert_eq!("Unexpected character '_'", err.message);
2802            }
2803        }
2804    }
2805
2806    #[test]
2807    fn tokenize_select_exponent() {
2808        let sql = String::from("SELECT 1e10, 1e-10, 1e+10, 1ea, 1e-10a, 1e-10-10");
2809        let dialect = GenericDialect {};
2810        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
2811
2812        let expected = vec![
2813            Token::make_keyword("SELECT"),
2814            Token::Whitespace(Whitespace::Space),
2815            Token::Number(String::from("1e10"), false),
2816            Token::Comma,
2817            Token::Whitespace(Whitespace::Space),
2818            Token::Number(String::from("1e-10"), false),
2819            Token::Comma,
2820            Token::Whitespace(Whitespace::Space),
2821            Token::Number(String::from("1e+10"), false),
2822            Token::Comma,
2823            Token::Whitespace(Whitespace::Space),
2824            Token::Number(String::from("1"), false),
2825            Token::make_word("ea", None),
2826            Token::Comma,
2827            Token::Whitespace(Whitespace::Space),
2828            Token::Number(String::from("1e-10"), false),
2829            Token::make_word("a", None),
2830            Token::Comma,
2831            Token::Whitespace(Whitespace::Space),
2832            Token::Number(String::from("1e-10"), false),
2833            Token::Minus,
2834            Token::Number(String::from("10"), false),
2835        ];
2836
2837        compare(expected, tokens);
2838    }
2839
2840    #[test]
2841    fn tokenize_scalar_function() {
2842        let sql = String::from("SELECT sqrt(1)");
2843        let dialect = GenericDialect {};
2844        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
2845
2846        let expected = vec![
2847            Token::make_keyword("SELECT"),
2848            Token::Whitespace(Whitespace::Space),
2849            Token::make_word("sqrt", None),
2850            Token::LParen,
2851            Token::Number(String::from("1"), false),
2852            Token::RParen,
2853        ];
2854
2855        compare(expected, tokens);
2856    }
2857
2858    #[test]
2859    fn tokenize_string_string_concat() {
2860        let sql = String::from("SELECT 'a' || 'b'");
2861        let dialect = GenericDialect {};
2862        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
2863
2864        let expected = vec![
2865            Token::make_keyword("SELECT"),
2866            Token::Whitespace(Whitespace::Space),
2867            Token::SingleQuotedString(String::from("a")),
2868            Token::Whitespace(Whitespace::Space),
2869            Token::StringConcat,
2870            Token::Whitespace(Whitespace::Space),
2871            Token::SingleQuotedString(String::from("b")),
2872        ];
2873
2874        compare(expected, tokens);
2875    }
2876    #[test]
2877    fn tokenize_bitwise_op() {
2878        let sql = String::from("SELECT one | two ^ three");
2879        let dialect = GenericDialect {};
2880        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
2881
2882        let expected = vec![
2883            Token::make_keyword("SELECT"),
2884            Token::Whitespace(Whitespace::Space),
2885            Token::make_word("one", None),
2886            Token::Whitespace(Whitespace::Space),
2887            Token::Pipe,
2888            Token::Whitespace(Whitespace::Space),
2889            Token::make_word("two", None),
2890            Token::Whitespace(Whitespace::Space),
2891            Token::Caret,
2892            Token::Whitespace(Whitespace::Space),
2893            Token::make_word("three", None),
2894        ];
2895        compare(expected, tokens);
2896    }
2897
2898    #[test]
2899    fn tokenize_logical_xor() {
2900        let sql =
2901            String::from("SELECT true XOR true, false XOR false, true XOR false, false XOR true");
2902        let dialect = GenericDialect {};
2903        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
2904
2905        let expected = vec![
2906            Token::make_keyword("SELECT"),
2907            Token::Whitespace(Whitespace::Space),
2908            Token::make_keyword("true"),
2909            Token::Whitespace(Whitespace::Space),
2910            Token::make_keyword("XOR"),
2911            Token::Whitespace(Whitespace::Space),
2912            Token::make_keyword("true"),
2913            Token::Comma,
2914            Token::Whitespace(Whitespace::Space),
2915            Token::make_keyword("false"),
2916            Token::Whitespace(Whitespace::Space),
2917            Token::make_keyword("XOR"),
2918            Token::Whitespace(Whitespace::Space),
2919            Token::make_keyword("false"),
2920            Token::Comma,
2921            Token::Whitespace(Whitespace::Space),
2922            Token::make_keyword("true"),
2923            Token::Whitespace(Whitespace::Space),
2924            Token::make_keyword("XOR"),
2925            Token::Whitespace(Whitespace::Space),
2926            Token::make_keyword("false"),
2927            Token::Comma,
2928            Token::Whitespace(Whitespace::Space),
2929            Token::make_keyword("false"),
2930            Token::Whitespace(Whitespace::Space),
2931            Token::make_keyword("XOR"),
2932            Token::Whitespace(Whitespace::Space),
2933            Token::make_keyword("true"),
2934        ];
2935        compare(expected, tokens);
2936    }
2937
2938    #[test]
2939    fn tokenize_simple_select() {
2940        let sql = String::from("SELECT * FROM customer WHERE id = 1 LIMIT 5");
2941        let dialect = GenericDialect {};
2942        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
2943
2944        let expected = vec![
2945            Token::make_keyword("SELECT"),
2946            Token::Whitespace(Whitespace::Space),
2947            Token::Mul,
2948            Token::Whitespace(Whitespace::Space),
2949            Token::make_keyword("FROM"),
2950            Token::Whitespace(Whitespace::Space),
2951            Token::make_word("customer", None),
2952            Token::Whitespace(Whitespace::Space),
2953            Token::make_keyword("WHERE"),
2954            Token::Whitespace(Whitespace::Space),
2955            Token::make_word("id", None),
2956            Token::Whitespace(Whitespace::Space),
2957            Token::Eq,
2958            Token::Whitespace(Whitespace::Space),
2959            Token::Number(String::from("1"), false),
2960            Token::Whitespace(Whitespace::Space),
2961            Token::make_keyword("LIMIT"),
2962            Token::Whitespace(Whitespace::Space),
2963            Token::Number(String::from("5"), false),
2964        ];
2965
2966        compare(expected, tokens);
2967    }
2968
2969    #[test]
2970    fn tokenize_explain_select() {
2971        let sql = String::from("EXPLAIN SELECT * FROM customer WHERE id = 1");
2972        let dialect = GenericDialect {};
2973        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
2974
2975        let expected = vec![
2976            Token::make_keyword("EXPLAIN"),
2977            Token::Whitespace(Whitespace::Space),
2978            Token::make_keyword("SELECT"),
2979            Token::Whitespace(Whitespace::Space),
2980            Token::Mul,
2981            Token::Whitespace(Whitespace::Space),
2982            Token::make_keyword("FROM"),
2983            Token::Whitespace(Whitespace::Space),
2984            Token::make_word("customer", None),
2985            Token::Whitespace(Whitespace::Space),
2986            Token::make_keyword("WHERE"),
2987            Token::Whitespace(Whitespace::Space),
2988            Token::make_word("id", None),
2989            Token::Whitespace(Whitespace::Space),
2990            Token::Eq,
2991            Token::Whitespace(Whitespace::Space),
2992            Token::Number(String::from("1"), false),
2993        ];
2994
2995        compare(expected, tokens);
2996    }
2997
2998    #[test]
2999    fn tokenize_explain_analyze_select() {
3000        let sql = String::from("EXPLAIN ANALYZE SELECT * FROM customer WHERE id = 1");
3001        let dialect = GenericDialect {};
3002        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
3003
3004        let expected = vec![
3005            Token::make_keyword("EXPLAIN"),
3006            Token::Whitespace(Whitespace::Space),
3007            Token::make_keyword("ANALYZE"),
3008            Token::Whitespace(Whitespace::Space),
3009            Token::make_keyword("SELECT"),
3010            Token::Whitespace(Whitespace::Space),
3011            Token::Mul,
3012            Token::Whitespace(Whitespace::Space),
3013            Token::make_keyword("FROM"),
3014            Token::Whitespace(Whitespace::Space),
3015            Token::make_word("customer", None),
3016            Token::Whitespace(Whitespace::Space),
3017            Token::make_keyword("WHERE"),
3018            Token::Whitespace(Whitespace::Space),
3019            Token::make_word("id", None),
3020            Token::Whitespace(Whitespace::Space),
3021            Token::Eq,
3022            Token::Whitespace(Whitespace::Space),
3023            Token::Number(String::from("1"), false),
3024        ];
3025
3026        compare(expected, tokens);
3027    }
3028
3029    #[test]
3030    fn tokenize_string_predicate() {
3031        let sql = String::from("SELECT * FROM customer WHERE salary != 'Not Provided'");
3032        let dialect = GenericDialect {};
3033        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
3034
3035        let expected = vec![
3036            Token::make_keyword("SELECT"),
3037            Token::Whitespace(Whitespace::Space),
3038            Token::Mul,
3039            Token::Whitespace(Whitespace::Space),
3040            Token::make_keyword("FROM"),
3041            Token::Whitespace(Whitespace::Space),
3042            Token::make_word("customer", None),
3043            Token::Whitespace(Whitespace::Space),
3044            Token::make_keyword("WHERE"),
3045            Token::Whitespace(Whitespace::Space),
3046            Token::make_word("salary", None),
3047            Token::Whitespace(Whitespace::Space),
3048            Token::Neq,
3049            Token::Whitespace(Whitespace::Space),
3050            Token::SingleQuotedString(String::from("Not Provided")),
3051        ];
3052
3053        compare(expected, tokens);
3054    }
3055
3056    #[test]
3057    fn tokenize_invalid_string() {
3058        let sql = String::from("\n💝مصطفىh");
3059
3060        let dialect = GenericDialect {};
3061        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
3062        // println!("tokens: {:#?}", tokens);
3063        let expected = vec![
3064            Token::Whitespace(Whitespace::Newline),
3065            Token::Char('💝'),
3066            Token::make_word("مصطفىh", None),
3067        ];
3068        compare(expected, tokens);
3069    }
3070
3071    #[test]
3072    fn tokenize_newline_in_string_literal() {
3073        let sql = String::from("'foo\r\nbar\nbaz'");
3074
3075        let dialect = GenericDialect {};
3076        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
3077        let expected = vec![Token::SingleQuotedString("foo\r\nbar\nbaz".to_string())];
3078        compare(expected, tokens);
3079    }
3080
3081    #[test]
3082    fn tokenize_unterminated_string_literal() {
3083        let sql = String::from("select 'foo");
3084
3085        let dialect = GenericDialect {};
3086        let mut tokenizer = Tokenizer::new(&dialect, &sql);
3087        assert_eq!(
3088            tokenizer.tokenize(),
3089            Err(TokenizerError {
3090                message: "Unterminated string literal".to_string(),
3091                location: Location { line: 1, column: 8 },
3092            })
3093        );
3094    }
3095
3096    #[test]
3097    fn tokenize_unterminated_string_literal_utf8() {
3098        let sql = String::from("SELECT \"なにか\" FROM Y WHERE \"なにか\" = 'test;");
3099
3100        let dialect = GenericDialect {};
3101        let mut tokenizer = Tokenizer::new(&dialect, &sql);
3102        assert_eq!(
3103            tokenizer.tokenize(),
3104            Err(TokenizerError {
3105                message: "Unterminated string literal".to_string(),
3106                location: Location {
3107                    line: 1,
3108                    column: 35
3109                }
3110            })
3111        );
3112    }
3113
3114    #[test]
3115    fn tokenize_invalid_string_cols() {
3116        let sql = String::from("\n\nSELECT * FROM table\t💝مصطفىh");
3117
3118        let dialect = GenericDialect {};
3119        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
3120        // println!("tokens: {:#?}", tokens);
3121        let expected = vec![
3122            Token::Whitespace(Whitespace::Newline),
3123            Token::Whitespace(Whitespace::Newline),
3124            Token::make_keyword("SELECT"),
3125            Token::Whitespace(Whitespace::Space),
3126            Token::Mul,
3127            Token::Whitespace(Whitespace::Space),
3128            Token::make_keyword("FROM"),
3129            Token::Whitespace(Whitespace::Space),
3130            Token::make_keyword("table"),
3131            Token::Whitespace(Whitespace::Tab),
3132            Token::Char('💝'),
3133            Token::make_word("مصطفىh", None),
3134        ];
3135        compare(expected, tokens);
3136    }
3137
3138    #[test]
3139    fn tokenize_dollar_quoted_string_tagged() {
3140        let test_cases = vec![
3141            (
3142                String::from("SELECT $tag$dollar '$' quoted strings have $tags like this$ or like this $$$tag$"),
3143                vec![
3144                    Token::make_keyword("SELECT"),
3145                    Token::Whitespace(Whitespace::Space),
3146                    Token::DollarQuotedString(DollarQuotedString {
3147                        value: "dollar '$' quoted strings have $tags like this$ or like this $$".into(),
3148                        tag: Some("tag".into()),
3149                    })
3150                ]
3151            ),
3152            (
3153                String::from("SELECT $abc$x$ab$abc$"),
3154                vec![
3155                    Token::make_keyword("SELECT"),
3156                    Token::Whitespace(Whitespace::Space),
3157                    Token::DollarQuotedString(DollarQuotedString {
3158                        value: "x$ab".into(),
3159                        tag: Some("abc".into()),
3160                    })
3161                ]
3162            ),
3163            (
3164                String::from("SELECT $abc$$abc$"),
3165                vec![
3166                    Token::make_keyword("SELECT"),
3167                    Token::Whitespace(Whitespace::Space),
3168                    Token::DollarQuotedString(DollarQuotedString {
3169                        value: "".into(),
3170                        tag: Some("abc".into()),
3171                    })
3172                ]
3173            ),
3174            (
3175                String::from("0$abc$$abc$1"),
3176                vec![
3177                    Token::Number("0".into(), false),
3178                    Token::DollarQuotedString(DollarQuotedString {
3179                        value: "".into(),
3180                        tag: Some("abc".into()),
3181                    }),
3182                    Token::Number("1".into(), false),
3183                ]
3184            ),
3185            (
3186                String::from("$function$abc$q$data$q$$function$"),
3187                vec![
3188                    Token::DollarQuotedString(DollarQuotedString {
3189                        value: "abc$q$data$q$".into(),
3190                        tag: Some("function".into()),
3191                    }),
3192                ]
3193            ),
3194        ];
3195
3196        let dialect = GenericDialect {};
3197        for (sql, expected) in test_cases {
3198            let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
3199            compare(expected, tokens);
3200        }
3201    }
3202
3203    #[test]
3204    fn tokenize_dollar_quoted_string_tagged_unterminated() {
3205        let sql = String::from("SELECT $tag$dollar '$' quoted strings have $tags like this$ or like this $$$different tag$");
3206        let dialect = GenericDialect {};
3207        assert_eq!(
3208            Tokenizer::new(&dialect, &sql).tokenize(),
3209            Err(TokenizerError {
3210                message: "Unterminated dollar-quoted, expected $".into(),
3211                location: Location {
3212                    line: 1,
3213                    column: 91
3214                }
3215            })
3216        );
3217    }
3218
3219    #[test]
3220    fn tokenize_dollar_quoted_string_tagged_unterminated_mirror() {
3221        let sql = String::from("SELECT $abc$abc$");
3222        let dialect = GenericDialect {};
3223        assert_eq!(
3224            Tokenizer::new(&dialect, &sql).tokenize(),
3225            Err(TokenizerError {
3226                message: "Unterminated dollar-quoted, expected $".into(),
3227                location: Location {
3228                    line: 1,
3229                    column: 17
3230                }
3231            })
3232        );
3233    }
3234
3235    #[test]
3236    fn tokenize_dollar_placeholder() {
3237        let sql = String::from("SELECT $$, $$ABC$$, $ABC$, $ABC");
3238        let dialect = SQLiteDialect {};
3239        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
3240        assert_eq!(
3241            tokens,
3242            vec![
3243                Token::make_keyword("SELECT"),
3244                Token::Whitespace(Whitespace::Space),
3245                Token::Placeholder("$$".into()),
3246                Token::Comma,
3247                Token::Whitespace(Whitespace::Space),
3248                Token::Placeholder("$$ABC$$".into()),
3249                Token::Comma,
3250                Token::Whitespace(Whitespace::Space),
3251                Token::Placeholder("$ABC$".into()),
3252                Token::Comma,
3253                Token::Whitespace(Whitespace::Space),
3254                Token::Placeholder("$ABC".into()),
3255            ]
3256        );
3257    }
3258
3259    #[test]
3260    fn tokenize_nested_dollar_quoted_strings() {
3261        let sql = String::from("SELECT $tag$dollar $nested$ string$tag$");
3262        let dialect = GenericDialect {};
3263        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
3264        let expected = vec![
3265            Token::make_keyword("SELECT"),
3266            Token::Whitespace(Whitespace::Space),
3267            Token::DollarQuotedString(DollarQuotedString {
3268                value: "dollar $nested$ string".into(),
3269                tag: Some("tag".into()),
3270            }),
3271        ];
3272        compare(expected, tokens);
3273    }
3274
3275    #[test]
3276    fn tokenize_dollar_quoted_string_untagged_empty() {
3277        let sql = String::from("SELECT $$$$");
3278        let dialect = GenericDialect {};
3279        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
3280        let expected = vec![
3281            Token::make_keyword("SELECT"),
3282            Token::Whitespace(Whitespace::Space),
3283            Token::DollarQuotedString(DollarQuotedString {
3284                value: "".into(),
3285                tag: None,
3286            }),
3287        ];
3288        compare(expected, tokens);
3289    }
3290
3291    #[test]
3292    fn tokenize_dollar_quoted_string_untagged() {
3293        let sql =
3294            String::from("SELECT $$within dollar '$' quoted strings have $tags like this$ $$");
3295        let dialect = GenericDialect {};
3296        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
3297        let expected = vec![
3298            Token::make_keyword("SELECT"),
3299            Token::Whitespace(Whitespace::Space),
3300            Token::DollarQuotedString(DollarQuotedString {
3301                value: "within dollar '$' quoted strings have $tags like this$ ".into(),
3302                tag: None,
3303            }),
3304        ];
3305        compare(expected, tokens);
3306    }
3307
3308    #[test]
3309    fn tokenize_dollar_quoted_string_untagged_unterminated() {
3310        let sql = String::from(
3311            "SELECT $$dollar '$' quoted strings have $tags like this$ or like this $different tag$",
3312        );
3313        let dialect = GenericDialect {};
3314        assert_eq!(
3315            Tokenizer::new(&dialect, &sql).tokenize(),
3316            Err(TokenizerError {
3317                message: "Unterminated dollar-quoted string".into(),
3318                location: Location {
3319                    line: 1,
3320                    column: 86
3321                }
3322            })
3323        );
3324    }
3325
3326    #[test]
3327    fn tokenize_right_arrow() {
3328        let sql = String::from("FUNCTION(key=>value)");
3329        let dialect = GenericDialect {};
3330        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
3331        let expected = vec![
3332            Token::make_word("FUNCTION", None),
3333            Token::LParen,
3334            Token::make_word("key", None),
3335            Token::RArrow,
3336            Token::make_word("value", None),
3337            Token::RParen,
3338        ];
3339        compare(expected, tokens);
3340    }
3341
3342    #[test]
3343    fn tokenize_is_null() {
3344        let sql = String::from("a IS NULL");
3345        let dialect = GenericDialect {};
3346        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
3347
3348        let expected = vec![
3349            Token::make_word("a", None),
3350            Token::Whitespace(Whitespace::Space),
3351            Token::make_keyword("IS"),
3352            Token::Whitespace(Whitespace::Space),
3353            Token::make_keyword("NULL"),
3354        ];
3355
3356        compare(expected, tokens);
3357    }
3358
3359    #[test]
3360    fn tokenize_comment() {
3361        let test_cases = vec![
3362            (
3363                String::from("0--this is a comment\n1"),
3364                vec![
3365                    Token::Number("0".to_string(), false),
3366                    Token::Whitespace(Whitespace::SingleLineComment {
3367                        prefix: "--".to_string(),
3368                        comment: "this is a comment".to_string(),
3369                    }),
3370                    Token::Whitespace(Whitespace::Newline),
3371                    Token::Number("1".to_string(), false),
3372                ],
3373            ),
3374            (
3375                String::from("0--this is a comment\r1"),
3376                vec![
3377                    Token::Number("0".to_string(), false),
3378                    Token::Whitespace(Whitespace::SingleLineComment {
3379                        prefix: "--".to_string(),
3380                        comment: "this is a comment\r1".to_string(),
3381                    }),
3382                ],
3383            ),
3384            (
3385                String::from("0--this is a comment\r\n1"),
3386                vec![
3387                    Token::Number("0".to_string(), false),
3388                    Token::Whitespace(Whitespace::SingleLineComment {
3389                        prefix: "--".to_string(),
3390                        comment: "this is a comment\r".to_string(),
3391                    }),
3392                    Token::Whitespace(Whitespace::Newline),
3393                    Token::Number("1".to_string(), false),
3394                ],
3395            ),
3396        ];
3397
3398        let dialect = GenericDialect {};
3399
3400        for (sql, expected) in test_cases {
3401            let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
3402            compare(expected, tokens);
3403        }
3404    }
3405
3406    #[test]
3407    fn tokenize_comment_postgres() {
3408        let sql = String::from("1--\r0");
3409
3410        let dialect = PostgreSqlDialect {};
3411        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
3412        let expected = vec![
3413            Token::Number("1".to_string(), false),
3414            Token::Whitespace(Whitespace::SingleLineComment {
3415                prefix: "--".to_string(),
3416                comment: "".to_string(),
3417            }),
3418            Token::Whitespace(Whitespace::Newline), // Postgres treats \r as newline in single-line comments
3419            Token::Number("0".to_string(), false),
3420        ];
3421        compare(expected, tokens);
3422    }
3423
3424    #[test]
3425    fn tokenize_comment_at_eof() {
3426        let sql = String::from("--this is a comment");
3427
3428        let dialect = GenericDialect {};
3429        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
3430        let expected = vec![Token::Whitespace(Whitespace::SingleLineComment {
3431            prefix: "--".to_string(),
3432            comment: "this is a comment".to_string(),
3433        })];
3434        compare(expected, tokens);
3435    }
3436
3437    #[test]
3438    fn tokenize_multiline_comment() {
3439        let sql = String::from("0/*multi-line\n* /comment*/1");
3440
3441        let dialect = GenericDialect {};
3442        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
3443        let expected = vec![
3444            Token::Number("0".to_string(), false),
3445            Token::Whitespace(Whitespace::MultiLineComment(
3446                "multi-line\n* /comment".to_string(),
3447            )),
3448            Token::Number("1".to_string(), false),
3449        ];
3450        compare(expected, tokens);
3451    }
3452
3453    #[test]
3454    fn tokenize_nested_multiline_comment() {
3455        all_dialects_where(|d| d.supports_nested_comments()).tokenizes_to(
3456            "0/*multi-line\n* \n/* comment \n /*comment*/*/ */ /comment*/1",
3457            vec![
3458                Token::Number("0".to_string(), false),
3459                Token::Whitespace(Whitespace::MultiLineComment(
3460                    "multi-line\n* \n/* comment \n /*comment*/*/ ".into(),
3461                )),
3462                Token::Whitespace(Whitespace::Space),
3463                Token::Div,
3464                Token::Word(Word {
3465                    value: "comment".to_string(),
3466                    quote_style: None,
3467                    keyword: Keyword::COMMENT,
3468                }),
3469                Token::Mul,
3470                Token::Div,
3471                Token::Number("1".to_string(), false),
3472            ],
3473        );
3474
3475        all_dialects_where(|d| d.supports_nested_comments()).tokenizes_to(
3476            "0/*multi-line\n* \n/* comment \n /*comment/**/ */ /comment*/*/1",
3477            vec![
3478                Token::Number("0".to_string(), false),
3479                Token::Whitespace(Whitespace::MultiLineComment(
3480                    "multi-line\n* \n/* comment \n /*comment/**/ */ /comment*/".into(),
3481                )),
3482                Token::Number("1".to_string(), false),
3483            ],
3484        );
3485
3486        all_dialects_where(|d| d.supports_nested_comments()).tokenizes_to(
3487            "SELECT 1/* a /* b */ c */0",
3488            vec![
3489                Token::make_keyword("SELECT"),
3490                Token::Whitespace(Whitespace::Space),
3491                Token::Number("1".to_string(), false),
3492                Token::Whitespace(Whitespace::MultiLineComment(" a /* b */ c ".to_string())),
3493                Token::Number("0".to_string(), false),
3494            ],
3495        );
3496    }
3497
3498    #[test]
3499    fn tokenize_nested_multiline_comment_empty() {
3500        all_dialects_where(|d| d.supports_nested_comments()).tokenizes_to(
3501            "select 1/*/**/*/0",
3502            vec![
3503                Token::make_keyword("select"),
3504                Token::Whitespace(Whitespace::Space),
3505                Token::Number("1".to_string(), false),
3506                Token::Whitespace(Whitespace::MultiLineComment("/**/".to_string())),
3507                Token::Number("0".to_string(), false),
3508            ],
3509        );
3510    }
3511
3512    #[test]
3513    fn tokenize_nested_comments_if_not_supported() {
3514        all_dialects_except(|d| d.supports_nested_comments()).tokenizes_to(
3515            "SELECT 1/*/* nested comment */*/0",
3516            vec![
3517                Token::make_keyword("SELECT"),
3518                Token::Whitespace(Whitespace::Space),
3519                Token::Number("1".to_string(), false),
3520                Token::Whitespace(Whitespace::MultiLineComment(
3521                    "/* nested comment ".to_string(),
3522                )),
3523                Token::Mul,
3524                Token::Div,
3525                Token::Number("0".to_string(), false),
3526            ],
3527        );
3528    }
3529
3530    #[test]
3531    fn tokenize_multiline_comment_with_even_asterisks() {
3532        let sql = String::from("\n/** Comment **/\n");
3533
3534        let dialect = GenericDialect {};
3535        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
3536        let expected = vec![
3537            Token::Whitespace(Whitespace::Newline),
3538            Token::Whitespace(Whitespace::MultiLineComment("* Comment *".to_string())),
3539            Token::Whitespace(Whitespace::Newline),
3540        ];
3541        compare(expected, tokens);
3542    }
3543
3544    #[test]
3545    fn tokenize_unicode_whitespace() {
3546        let sql = String::from(" \u{2003}\n");
3547
3548        let dialect = GenericDialect {};
3549        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
3550        let expected = vec![
3551            Token::Whitespace(Whitespace::Space),
3552            Token::Whitespace(Whitespace::Space),
3553            Token::Whitespace(Whitespace::Newline),
3554        ];
3555        compare(expected, tokens);
3556    }
3557
3558    #[test]
3559    fn tokenize_mismatched_quotes() {
3560        let sql = String::from("\"foo");
3561
3562        let dialect = GenericDialect {};
3563        let mut tokenizer = Tokenizer::new(&dialect, &sql);
3564        assert_eq!(
3565            tokenizer.tokenize(),
3566            Err(TokenizerError {
3567                message: "Expected close delimiter '\"' before EOF.".to_string(),
3568                location: Location { line: 1, column: 1 },
3569            })
3570        );
3571    }
3572
3573    #[test]
3574    fn tokenize_newlines() {
3575        let sql = String::from("line1\nline2\rline3\r\nline4\r");
3576
3577        let dialect = GenericDialect {};
3578        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
3579        let expected = vec![
3580            Token::make_word("line1", None),
3581            Token::Whitespace(Whitespace::Newline),
3582            Token::make_word("line2", None),
3583            Token::Whitespace(Whitespace::Newline),
3584            Token::make_word("line3", None),
3585            Token::Whitespace(Whitespace::Newline),
3586            Token::make_word("line4", None),
3587            Token::Whitespace(Whitespace::Newline),
3588        ];
3589        compare(expected, tokens);
3590    }
3591
3592    #[test]
3593    fn tokenize_mssql_top() {
3594        let sql = "SELECT TOP 5 [bar] FROM foo";
3595        let dialect = MsSqlDialect {};
3596        let tokens = Tokenizer::new(&dialect, sql).tokenize().unwrap();
3597        let expected = vec![
3598            Token::make_keyword("SELECT"),
3599            Token::Whitespace(Whitespace::Space),
3600            Token::make_keyword("TOP"),
3601            Token::Whitespace(Whitespace::Space),
3602            Token::Number(String::from("5"), false),
3603            Token::Whitespace(Whitespace::Space),
3604            Token::make_word("bar", Some('[')),
3605            Token::Whitespace(Whitespace::Space),
3606            Token::make_keyword("FROM"),
3607            Token::Whitespace(Whitespace::Space),
3608            Token::make_word("foo", None),
3609        ];
3610        compare(expected, tokens);
3611    }
3612
3613    #[test]
3614    fn tokenize_pg_regex_match() {
3615        let sql = "SELECT col ~ '^a', col ~* '^a', col !~ '^a', col !~* '^a'";
3616        let dialect = GenericDialect {};
3617        let tokens = Tokenizer::new(&dialect, sql).tokenize().unwrap();
3618        let expected = vec![
3619            Token::make_keyword("SELECT"),
3620            Token::Whitespace(Whitespace::Space),
3621            Token::make_word("col", None),
3622            Token::Whitespace(Whitespace::Space),
3623            Token::Tilde,
3624            Token::Whitespace(Whitespace::Space),
3625            Token::SingleQuotedString("^a".into()),
3626            Token::Comma,
3627            Token::Whitespace(Whitespace::Space),
3628            Token::make_word("col", None),
3629            Token::Whitespace(Whitespace::Space),
3630            Token::TildeAsterisk,
3631            Token::Whitespace(Whitespace::Space),
3632            Token::SingleQuotedString("^a".into()),
3633            Token::Comma,
3634            Token::Whitespace(Whitespace::Space),
3635            Token::make_word("col", None),
3636            Token::Whitespace(Whitespace::Space),
3637            Token::ExclamationMarkTilde,
3638            Token::Whitespace(Whitespace::Space),
3639            Token::SingleQuotedString("^a".into()),
3640            Token::Comma,
3641            Token::Whitespace(Whitespace::Space),
3642            Token::make_word("col", None),
3643            Token::Whitespace(Whitespace::Space),
3644            Token::ExclamationMarkTildeAsterisk,
3645            Token::Whitespace(Whitespace::Space),
3646            Token::SingleQuotedString("^a".into()),
3647        ];
3648        compare(expected, tokens);
3649    }
3650
3651    #[test]
3652    fn tokenize_pg_like_match() {
3653        let sql = "SELECT col ~~ '_a%', col ~~* '_a%', col !~~ '_a%', col !~~* '_a%'";
3654        let dialect = GenericDialect {};
3655        let tokens = Tokenizer::new(&dialect, sql).tokenize().unwrap();
3656        let expected = vec![
3657            Token::make_keyword("SELECT"),
3658            Token::Whitespace(Whitespace::Space),
3659            Token::make_word("col", None),
3660            Token::Whitespace(Whitespace::Space),
3661            Token::DoubleTilde,
3662            Token::Whitespace(Whitespace::Space),
3663            Token::SingleQuotedString("_a%".into()),
3664            Token::Comma,
3665            Token::Whitespace(Whitespace::Space),
3666            Token::make_word("col", None),
3667            Token::Whitespace(Whitespace::Space),
3668            Token::DoubleTildeAsterisk,
3669            Token::Whitespace(Whitespace::Space),
3670            Token::SingleQuotedString("_a%".into()),
3671            Token::Comma,
3672            Token::Whitespace(Whitespace::Space),
3673            Token::make_word("col", None),
3674            Token::Whitespace(Whitespace::Space),
3675            Token::ExclamationMarkDoubleTilde,
3676            Token::Whitespace(Whitespace::Space),
3677            Token::SingleQuotedString("_a%".into()),
3678            Token::Comma,
3679            Token::Whitespace(Whitespace::Space),
3680            Token::make_word("col", None),
3681            Token::Whitespace(Whitespace::Space),
3682            Token::ExclamationMarkDoubleTildeAsterisk,
3683            Token::Whitespace(Whitespace::Space),
3684            Token::SingleQuotedString("_a%".into()),
3685        ];
3686        compare(expected, tokens);
3687    }
3688
3689    #[test]
3690    fn tokenize_quoted_identifier() {
3691        let sql = r#" "a "" b" "a """ "c """"" "#;
3692        let dialect = GenericDialect {};
3693        let tokens = Tokenizer::new(&dialect, sql).tokenize().unwrap();
3694        let expected = vec![
3695            Token::Whitespace(Whitespace::Space),
3696            Token::make_word(r#"a " b"#, Some('"')),
3697            Token::Whitespace(Whitespace::Space),
3698            Token::make_word(r#"a ""#, Some('"')),
3699            Token::Whitespace(Whitespace::Space),
3700            Token::make_word(r#"c """#, Some('"')),
3701            Token::Whitespace(Whitespace::Space),
3702        ];
3703        compare(expected, tokens);
3704    }
3705
3706    #[test]
3707    fn tokenize_snowflake_div() {
3708        let sql = r#"field/1000"#;
3709        let dialect = SnowflakeDialect {};
3710        let tokens = Tokenizer::new(&dialect, sql).tokenize().unwrap();
3711        let expected = vec![
3712            Token::make_word(r#"field"#, None),
3713            Token::Div,
3714            Token::Number("1000".to_string(), false),
3715        ];
3716        compare(expected, tokens);
3717    }
3718
3719    #[test]
3720    fn tokenize_quoted_identifier_with_no_escape() {
3721        let sql = r#" "a "" b" "a """ "c """"" "#;
3722        let dialect = GenericDialect {};
3723        let tokens = Tokenizer::new(&dialect, sql)
3724            .with_unescape(false)
3725            .tokenize()
3726            .unwrap();
3727        let expected = vec![
3728            Token::Whitespace(Whitespace::Space),
3729            Token::make_word(r#"a "" b"#, Some('"')),
3730            Token::Whitespace(Whitespace::Space),
3731            Token::make_word(r#"a """#, Some('"')),
3732            Token::Whitespace(Whitespace::Space),
3733            Token::make_word(r#"c """""#, Some('"')),
3734            Token::Whitespace(Whitespace::Space),
3735        ];
3736        compare(expected, tokens);
3737    }
3738
3739    #[test]
3740    fn tokenize_with_location() {
3741        let sql = "SELECT a,\n b";
3742        let dialect = GenericDialect {};
3743        let tokens = Tokenizer::new(&dialect, sql)
3744            .tokenize_with_location()
3745            .unwrap();
3746        let expected = vec![
3747            TokenWithSpan::at(Token::make_keyword("SELECT"), (1, 1).into(), (1, 7).into()),
3748            TokenWithSpan::at(
3749                Token::Whitespace(Whitespace::Space),
3750                (1, 7).into(),
3751                (1, 8).into(),
3752            ),
3753            TokenWithSpan::at(Token::make_word("a", None), (1, 8).into(), (1, 9).into()),
3754            TokenWithSpan::at(Token::Comma, (1, 9).into(), (1, 10).into()),
3755            TokenWithSpan::at(
3756                Token::Whitespace(Whitespace::Newline),
3757                (1, 10).into(),
3758                (2, 1).into(),
3759            ),
3760            TokenWithSpan::at(
3761                Token::Whitespace(Whitespace::Space),
3762                (2, 1).into(),
3763                (2, 2).into(),
3764            ),
3765            TokenWithSpan::at(Token::make_word("b", None), (2, 2).into(), (2, 3).into()),
3766        ];
3767        compare(expected, tokens);
3768    }
3769
3770    fn compare<T: PartialEq + fmt::Debug>(expected: Vec<T>, actual: Vec<T>) {
3771        //println!("------------------------------");
3772        //println!("tokens   = {:?}", actual);
3773        //println!("expected = {:?}", expected);
3774        //println!("------------------------------");
3775        assert_eq!(expected, actual);
3776    }
3777
3778    fn check_unescape(s: &str, expected: Option<&str>) {
3779        let s = format!("'{s}'");
3780        let mut state = State {
3781            peekable: s.chars().peekable(),
3782            line: 0,
3783            col: 0,
3784        };
3785
3786        assert_eq!(
3787            unescape_single_quoted_string(&mut state),
3788            expected.map(|s| s.to_string())
3789        );
3790    }
3791
3792    #[test]
3793    fn test_unescape() {
3794        check_unescape(r"\b", Some("\u{0008}"));
3795        check_unescape(r"\f", Some("\u{000C}"));
3796        check_unescape(r"\t", Some("\t"));
3797        check_unescape(r"\r\n", Some("\r\n"));
3798        check_unescape(r"\/", Some("/"));
3799        check_unescape(r"/", Some("/"));
3800        check_unescape(r"\\", Some("\\"));
3801
3802        // 16 and 32-bit hexadecimal Unicode character value
3803        check_unescape(r"\u0001", Some("\u{0001}"));
3804        check_unescape(r"\u4c91", Some("\u{4c91}"));
3805        check_unescape(r"\u4c916", Some("\u{4c91}6"));
3806        check_unescape(r"\u4c", None);
3807        check_unescape(r"\u0000", None);
3808        check_unescape(r"\U0010FFFF", Some("\u{10FFFF}"));
3809        check_unescape(r"\U00110000", None);
3810        check_unescape(r"\U00000000", None);
3811        check_unescape(r"\u", None);
3812        check_unescape(r"\U", None);
3813        check_unescape(r"\U1010FFFF", None);
3814
3815        // hexadecimal byte value
3816        check_unescape(r"\x4B", Some("\u{004b}"));
3817        check_unescape(r"\x4", Some("\u{0004}"));
3818        check_unescape(r"\x4L", Some("\u{0004}L"));
3819        check_unescape(r"\x", Some("x"));
3820        check_unescape(r"\xP", Some("xP"));
3821        check_unescape(r"\x0", None);
3822        check_unescape(r"\xCAD", None);
3823        check_unescape(r"\xA9", None);
3824
3825        // octal byte value
3826        check_unescape(r"\1", Some("\u{0001}"));
3827        check_unescape(r"\12", Some("\u{000a}"));
3828        check_unescape(r"\123", Some("\u{0053}"));
3829        check_unescape(r"\1232", Some("\u{0053}2"));
3830        check_unescape(r"\4", Some("\u{0004}"));
3831        check_unescape(r"\45", Some("\u{0025}"));
3832        check_unescape(r"\450", Some("\u{0028}"));
3833        check_unescape(r"\603", None);
3834        check_unescape(r"\0", None);
3835        check_unescape(r"\080", None);
3836
3837        // others
3838        check_unescape(r"\9", Some("9"));
3839        check_unescape(r"''", Some("'"));
3840        check_unescape(
3841            r"Hello\r\nRust/\u4c91 SQL Parser\U0010ABCD\1232",
3842            Some("Hello\r\nRust/\u{4c91} SQL Parser\u{10abcd}\u{0053}2"),
3843        );
3844        check_unescape(r"Hello\0", None);
3845        check_unescape(r"Hello\xCADRust", None);
3846    }
3847
3848    #[test]
3849    fn tokenize_numeric_prefix_trait() {
3850        #[derive(Debug)]
3851        struct NumericPrefixDialect;
3852
3853        impl Dialect for NumericPrefixDialect {
3854            fn is_identifier_start(&self, ch: char) -> bool {
3855                ch.is_ascii_lowercase()
3856                    || ch.is_ascii_uppercase()
3857                    || ch.is_ascii_digit()
3858                    || ch == '$'
3859            }
3860
3861            fn is_identifier_part(&self, ch: char) -> bool {
3862                ch.is_ascii_lowercase()
3863                    || ch.is_ascii_uppercase()
3864                    || ch.is_ascii_digit()
3865                    || ch == '_'
3866                    || ch == '$'
3867                    || ch == '{'
3868                    || ch == '}'
3869            }
3870
3871            fn supports_numeric_prefix(&self) -> bool {
3872                true
3873            }
3874        }
3875
3876        tokenize_numeric_prefix_inner(&NumericPrefixDialect {});
3877        tokenize_numeric_prefix_inner(&HiveDialect {});
3878        tokenize_numeric_prefix_inner(&MySqlDialect {});
3879    }
3880
3881    fn tokenize_numeric_prefix_inner(dialect: &dyn Dialect) {
3882        let sql = r#"SELECT * FROM 1"#;
3883        let tokens = Tokenizer::new(dialect, sql).tokenize().unwrap();
3884        let expected = vec![
3885            Token::make_keyword("SELECT"),
3886            Token::Whitespace(Whitespace::Space),
3887            Token::Mul,
3888            Token::Whitespace(Whitespace::Space),
3889            Token::make_keyword("FROM"),
3890            Token::Whitespace(Whitespace::Space),
3891            Token::Number(String::from("1"), false),
3892        ];
3893        compare(expected, tokens);
3894    }
3895
3896    #[test]
3897    fn tokenize_quoted_string_escape() {
3898        let dialect = SnowflakeDialect {};
3899        for (sql, expected, expected_unescaped) in [
3900            (r#"'%a\'%b'"#, r#"%a\'%b"#, r#"%a'%b"#),
3901            (r#"'a\'\'b\'c\'d'"#, r#"a\'\'b\'c\'d"#, r#"a''b'c'd"#),
3902            (r#"'\\'"#, r#"\\"#, r#"\"#),
3903            (
3904                r#"'\0\a\b\f\n\r\t\Z'"#,
3905                r#"\0\a\b\f\n\r\t\Z"#,
3906                "\0\u{7}\u{8}\u{c}\n\r\t\u{1a}",
3907            ),
3908            (r#"'\"'"#, r#"\""#, "\""),
3909            (r#"'\\a\\b\'c'"#, r#"\\a\\b\'c"#, r#"\a\b'c"#),
3910            (r#"'\'abcd'"#, r#"\'abcd"#, r#"'abcd"#),
3911            (r#"'''a''b'"#, r#"''a''b"#, r#"'a'b"#),
3912            (r#"'\q'"#, r#"\q"#, r#"q"#),
3913            (r#"'\%\_'"#, r#"\%\_"#, r#"%_"#),
3914            (r#"'\\%\\_'"#, r#"\\%\\_"#, r#"\%\_"#),
3915        ] {
3916            let tokens = Tokenizer::new(&dialect, sql)
3917                .with_unescape(false)
3918                .tokenize()
3919                .unwrap();
3920            let expected = vec![Token::SingleQuotedString(expected.to_string())];
3921            compare(expected, tokens);
3922
3923            let tokens = Tokenizer::new(&dialect, sql)
3924                .with_unescape(true)
3925                .tokenize()
3926                .unwrap();
3927            let expected = vec![Token::SingleQuotedString(expected_unescaped.to_string())];
3928            compare(expected, tokens);
3929        }
3930
3931        for sql in [r#"'\'"#, r#"'ab\'"#] {
3932            let mut tokenizer = Tokenizer::new(&dialect, sql);
3933            assert_eq!(
3934                "Unterminated string literal",
3935                tokenizer.tokenize().unwrap_err().message.as_str(),
3936            );
3937        }
3938
3939        // Non-escape dialect
3940        for (sql, expected) in [(r#"'\'"#, r#"\"#), (r#"'ab\'"#, r#"ab\"#)] {
3941            let dialect = GenericDialect {};
3942            let tokens = Tokenizer::new(&dialect, sql).tokenize().unwrap();
3943
3944            let expected = vec![Token::SingleQuotedString(expected.to_string())];
3945
3946            compare(expected, tokens);
3947        }
3948
3949        // MySQL special case for LIKE escapes
3950        for (sql, expected) in [(r#"'\%'"#, r#"\%"#), (r#"'\_'"#, r#"\_"#)] {
3951            let dialect = MySqlDialect {};
3952            let tokens = Tokenizer::new(&dialect, sql).tokenize().unwrap();
3953
3954            let expected = vec![Token::SingleQuotedString(expected.to_string())];
3955
3956            compare(expected, tokens);
3957        }
3958    }
3959
3960    #[test]
3961    fn tokenize_triple_quoted_string() {
3962        fn check<F>(
3963            q: char, // The quote character to test
3964            r: char, // An alternate quote character.
3965            quote_token: F,
3966        ) where
3967            F: Fn(String) -> Token,
3968        {
3969            let dialect = BigQueryDialect {};
3970
3971            for (sql, expected, expected_unescaped) in [
3972                // Empty string
3973                (format!(r#"{q}{q}{q}{q}{q}{q}"#), "".into(), "".into()),
3974                // Should not count escaped quote as end of string.
3975                (
3976                    format!(r#"{q}{q}{q}ab{q}{q}\{q}{q}cd{q}{q}{q}"#),
3977                    format!(r#"ab{q}{q}\{q}{q}cd"#),
3978                    format!(r#"ab{q}{q}{q}{q}cd"#),
3979                ),
3980                // Simple string
3981                (
3982                    format!(r#"{q}{q}{q}abc{q}{q}{q}"#),
3983                    "abc".into(),
3984                    "abc".into(),
3985                ),
3986                // Mix single-double quotes unescaped.
3987                (
3988                    format!(r#"{q}{q}{q}ab{r}{r}{r}c{r}def{r}{r}{r}{q}{q}{q}"#),
3989                    format!("ab{r}{r}{r}c{r}def{r}{r}{r}"),
3990                    format!("ab{r}{r}{r}c{r}def{r}{r}{r}"),
3991                ),
3992                // Escaped quote.
3993                (
3994                    format!(r#"{q}{q}{q}ab{q}{q}c{q}{q}\{q}de{q}{q}f{q}{q}{q}"#),
3995                    format!(r#"ab{q}{q}c{q}{q}\{q}de{q}{q}f"#),
3996                    format!(r#"ab{q}{q}c{q}{q}{q}de{q}{q}f"#),
3997                ),
3998                // backslash-escaped quote characters.
3999                (
4000                    format!(r#"{q}{q}{q}a\'\'b\'c\'d{q}{q}{q}"#),
4001                    r#"a\'\'b\'c\'d"#.into(),
4002                    r#"a''b'c'd"#.into(),
4003                ),
4004                // backslash-escaped characters
4005                (
4006                    format!(r#"{q}{q}{q}abc\0\n\rdef{q}{q}{q}"#),
4007                    r#"abc\0\n\rdef"#.into(),
4008                    "abc\0\n\rdef".into(),
4009                ),
4010            ] {
4011                let tokens = Tokenizer::new(&dialect, sql.as_str())
4012                    .with_unescape(false)
4013                    .tokenize()
4014                    .unwrap();
4015                let expected = vec![quote_token(expected.to_string())];
4016                compare(expected, tokens);
4017
4018                let tokens = Tokenizer::new(&dialect, sql.as_str())
4019                    .with_unescape(true)
4020                    .tokenize()
4021                    .unwrap();
4022                let expected = vec![quote_token(expected_unescaped.to_string())];
4023                compare(expected, tokens);
4024            }
4025
4026            for sql in [
4027                format!(r#"{q}{q}{q}{q}{q}\{q}"#),
4028                format!(r#"{q}{q}{q}abc{q}{q}\{q}"#),
4029                format!(r#"{q}{q}{q}{q}"#),
4030                format!(r#"{q}{q}{q}{r}{r}"#),
4031                format!(r#"{q}{q}{q}abc{q}"#),
4032                format!(r#"{q}{q}{q}abc{q}{q}"#),
4033                format!(r#"{q}{q}{q}abc"#),
4034            ] {
4035                let dialect = BigQueryDialect {};
4036                let mut tokenizer = Tokenizer::new(&dialect, sql.as_str());
4037                assert_eq!(
4038                    "Unterminated string literal",
4039                    tokenizer.tokenize().unwrap_err().message.as_str(),
4040                );
4041            }
4042        }
4043
4044        check('"', '\'', Token::TripleDoubleQuotedString);
4045
4046        check('\'', '"', Token::TripleSingleQuotedString);
4047
4048        let dialect = BigQueryDialect {};
4049
4050        let sql = r#"""''"#;
4051        let tokens = Tokenizer::new(&dialect, sql)
4052            .with_unescape(true)
4053            .tokenize()
4054            .unwrap();
4055        let expected = vec![
4056            Token::DoubleQuotedString("".to_string()),
4057            Token::SingleQuotedString("".to_string()),
4058        ];
4059        compare(expected, tokens);
4060
4061        let sql = r#"''"""#;
4062        let tokens = Tokenizer::new(&dialect, sql)
4063            .with_unescape(true)
4064            .tokenize()
4065            .unwrap();
4066        let expected = vec![
4067            Token::SingleQuotedString("".to_string()),
4068            Token::DoubleQuotedString("".to_string()),
4069        ];
4070        compare(expected, tokens);
4071
4072        // Non-triple quoted string dialect
4073        let dialect = SnowflakeDialect {};
4074        let sql = r#"''''''"#;
4075        let tokens = Tokenizer::new(&dialect, sql).tokenize().unwrap();
4076        let expected = vec![Token::SingleQuotedString("''".to_string())];
4077        compare(expected, tokens);
4078    }
4079
4080    #[test]
4081    fn test_mysql_users_grantees() {
4082        let dialect = MySqlDialect {};
4083
4084        let sql = "CREATE USER `root`@`%`";
4085        let tokens = Tokenizer::new(&dialect, sql).tokenize().unwrap();
4086        let expected = vec![
4087            Token::make_keyword("CREATE"),
4088            Token::Whitespace(Whitespace::Space),
4089            Token::make_keyword("USER"),
4090            Token::Whitespace(Whitespace::Space),
4091            Token::make_word("root", Some('`')),
4092            Token::AtSign,
4093            Token::make_word("%", Some('`')),
4094        ];
4095        compare(expected, tokens);
4096    }
4097
4098    #[test]
4099    fn test_postgres_abs_without_space_and_string_literal() {
4100        let dialect = MySqlDialect {};
4101
4102        let sql = "SELECT @'1'";
4103        let tokens = Tokenizer::new(&dialect, sql).tokenize().unwrap();
4104        let expected = vec![
4105            Token::make_keyword("SELECT"),
4106            Token::Whitespace(Whitespace::Space),
4107            Token::AtSign,
4108            Token::SingleQuotedString("1".to_string()),
4109        ];
4110        compare(expected, tokens);
4111    }
4112
4113    #[test]
4114    fn test_postgres_abs_without_space_and_quoted_column() {
4115        let dialect = MySqlDialect {};
4116
4117        let sql = r#"SELECT @"bar" FROM foo"#;
4118        let tokens = Tokenizer::new(&dialect, sql).tokenize().unwrap();
4119        let expected = vec![
4120            Token::make_keyword("SELECT"),
4121            Token::Whitespace(Whitespace::Space),
4122            Token::AtSign,
4123            Token::DoubleQuotedString("bar".to_string()),
4124            Token::Whitespace(Whitespace::Space),
4125            Token::make_keyword("FROM"),
4126            Token::Whitespace(Whitespace::Space),
4127            Token::make_word("foo", None),
4128        ];
4129        compare(expected, tokens);
4130    }
4131
4132    #[test]
4133    fn test_national_strings_backslash_escape_not_supported() {
4134        all_dialects_where(|dialect| !dialect.supports_string_literal_backslash_escape())
4135            .tokenizes_to(
4136                "select n'''''\\'",
4137                vec![
4138                    Token::make_keyword("select"),
4139                    Token::Whitespace(Whitespace::Space),
4140                    Token::NationalStringLiteral("''\\".to_string()),
4141                ],
4142            );
4143    }
4144
4145    #[test]
4146    fn test_national_strings_backslash_escape_supported() {
4147        all_dialects_where(|dialect| dialect.supports_string_literal_backslash_escape())
4148            .tokenizes_to(
4149                "select n'''''\\''",
4150                vec![
4151                    Token::make_keyword("select"),
4152                    Token::Whitespace(Whitespace::Space),
4153                    Token::NationalStringLiteral("'''".to_string()),
4154                ],
4155            );
4156    }
4157
4158    #[test]
4159    fn test_string_escape_constant_not_supported() {
4160        all_dialects_where(|dialect| !dialect.supports_string_escape_constant()).tokenizes_to(
4161            "select e'...'",
4162            vec![
4163                Token::make_keyword("select"),
4164                Token::Whitespace(Whitespace::Space),
4165                Token::make_word("e", None),
4166                Token::SingleQuotedString("...".to_string()),
4167            ],
4168        );
4169
4170        all_dialects_where(|dialect| !dialect.supports_string_escape_constant()).tokenizes_to(
4171            "select E'...'",
4172            vec![
4173                Token::make_keyword("select"),
4174                Token::Whitespace(Whitespace::Space),
4175                Token::make_word("E", None),
4176                Token::SingleQuotedString("...".to_string()),
4177            ],
4178        );
4179    }
4180
4181    #[test]
4182    fn test_string_escape_constant_supported() {
4183        all_dialects_where(|dialect| dialect.supports_string_escape_constant()).tokenizes_to(
4184            "select e'\\''",
4185            vec![
4186                Token::make_keyword("select"),
4187                Token::Whitespace(Whitespace::Space),
4188                Token::EscapedStringLiteral("'".to_string()),
4189            ],
4190        );
4191
4192        all_dialects_where(|dialect| dialect.supports_string_escape_constant()).tokenizes_to(
4193            "select E'\\''",
4194            vec![
4195                Token::make_keyword("select"),
4196                Token::Whitespace(Whitespace::Space),
4197                Token::EscapedStringLiteral("'".to_string()),
4198            ],
4199        );
4200    }
4201
4202    #[test]
4203    fn test_whitespace_required_after_single_line_comment() {
4204        all_dialects_where(|dialect| dialect.requires_single_line_comment_whitespace())
4205            .tokenizes_to(
4206                "SELECT --'abc'",
4207                vec![
4208                    Token::make_keyword("SELECT"),
4209                    Token::Whitespace(Whitespace::Space),
4210                    Token::Minus,
4211                    Token::Minus,
4212                    Token::SingleQuotedString("abc".to_string()),
4213                ],
4214            );
4215
4216        all_dialects_where(|dialect| dialect.requires_single_line_comment_whitespace())
4217            .tokenizes_to(
4218                "SELECT -- 'abc'",
4219                vec![
4220                    Token::make_keyword("SELECT"),
4221                    Token::Whitespace(Whitespace::Space),
4222                    Token::Whitespace(Whitespace::SingleLineComment {
4223                        prefix: "--".to_string(),
4224                        comment: " 'abc'".to_string(),
4225                    }),
4226                ],
4227            );
4228
4229        all_dialects_where(|dialect| dialect.requires_single_line_comment_whitespace())
4230            .tokenizes_to(
4231                "SELECT --",
4232                vec![
4233                    Token::make_keyword("SELECT"),
4234                    Token::Whitespace(Whitespace::Space),
4235                    Token::Minus,
4236                    Token::Minus,
4237                ],
4238            );
4239
4240        all_dialects_where(|d| d.requires_single_line_comment_whitespace()).tokenizes_to(
4241            "--\n-- Table structure for table...\n--\n",
4242            vec![
4243                Token::Whitespace(Whitespace::SingleLineComment {
4244                    prefix: "--".to_string(),
4245                    comment: "".to_string(),
4246                }),
4247                Token::Whitespace(Whitespace::Newline),
4248                Token::Whitespace(Whitespace::SingleLineComment {
4249                    prefix: "--".to_string(),
4250                    comment: " Table structure for table...".to_string(),
4251                }),
4252                Token::Whitespace(Whitespace::Newline),
4253                Token::Whitespace(Whitespace::SingleLineComment {
4254                    prefix: "--".to_string(),
4255                    comment: "".to_string(),
4256                }),
4257                Token::Whitespace(Whitespace::Newline),
4258            ],
4259        );
4260    }
4261
4262    #[test]
4263    fn test_whitespace_not_required_after_single_line_comment() {
4264        all_dialects_where(|dialect| !dialect.requires_single_line_comment_whitespace())
4265            .tokenizes_to(
4266                "SELECT --'abc'",
4267                vec![
4268                    Token::make_keyword("SELECT"),
4269                    Token::Whitespace(Whitespace::Space),
4270                    Token::Whitespace(Whitespace::SingleLineComment {
4271                        prefix: "--".to_string(),
4272                        comment: "'abc'".to_string(),
4273                    }),
4274                ],
4275            );
4276
4277        all_dialects_where(|dialect| !dialect.requires_single_line_comment_whitespace())
4278            .tokenizes_to(
4279                "SELECT -- 'abc'",
4280                vec![
4281                    Token::make_keyword("SELECT"),
4282                    Token::Whitespace(Whitespace::Space),
4283                    Token::Whitespace(Whitespace::SingleLineComment {
4284                        prefix: "--".to_string(),
4285                        comment: " 'abc'".to_string(),
4286                    }),
4287                ],
4288            );
4289
4290        all_dialects_where(|dialect| !dialect.requires_single_line_comment_whitespace())
4291            .tokenizes_to(
4292                "SELECT --",
4293                vec![
4294                    Token::make_keyword("SELECT"),
4295                    Token::Whitespace(Whitespace::Space),
4296                    Token::Whitespace(Whitespace::SingleLineComment {
4297                        prefix: "--".to_string(),
4298                        comment: "".to_string(),
4299                    }),
4300                ],
4301            );
4302    }
4303
4304    #[test]
4305    fn test_tokenize_identifiers_numeric_prefix() {
4306        all_dialects_where(|dialect| dialect.supports_numeric_prefix())
4307            .tokenizes_to("123abc", vec![Token::make_word("123abc", None)]);
4308
4309        all_dialects_where(|dialect| dialect.supports_numeric_prefix())
4310            .tokenizes_to("12e34", vec![Token::Number("12e34".to_string(), false)]);
4311
4312        all_dialects_where(|dialect| dialect.supports_numeric_prefix()).tokenizes_to(
4313            "t.12e34",
4314            vec![
4315                Token::make_word("t", None),
4316                Token::Period,
4317                Token::make_word("12e34", None),
4318            ],
4319        );
4320
4321        all_dialects_where(|dialect| dialect.supports_numeric_prefix()).tokenizes_to(
4322            "t.1two3",
4323            vec![
4324                Token::make_word("t", None),
4325                Token::Period,
4326                Token::make_word("1two3", None),
4327            ],
4328        );
4329    }
4330
4331    #[test]
4332    fn tokenize_period_underscore() {
4333        let sql = String::from("SELECT table._col");
4334        // a dialect that supports underscores in numeric literals
4335        let dialect = PostgreSqlDialect {};
4336        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
4337
4338        let expected = vec![
4339            Token::make_keyword("SELECT"),
4340            Token::Whitespace(Whitespace::Space),
4341            Token::Word(Word {
4342                value: "table".to_string(),
4343                quote_style: None,
4344                keyword: Keyword::TABLE,
4345            }),
4346            Token::Period,
4347            Token::Word(Word {
4348                value: "_col".to_string(),
4349                quote_style: None,
4350                keyword: Keyword::NoKeyword,
4351            }),
4352        ];
4353
4354        compare(expected, tokens);
4355
4356        let sql = String::from("SELECT ._123");
4357        if let Ok(tokens) = Tokenizer::new(&dialect, &sql).tokenize() {
4358            panic!("Tokenizer should have failed on {sql}, but it succeeded with {tokens:?}");
4359        }
4360
4361        let sql = String::from("SELECT ._abc");
4362        if let Ok(tokens) = Tokenizer::new(&dialect, &sql).tokenize() {
4363            panic!("Tokenizer should have failed on {sql}, but it succeeded with {tokens:?}");
4364        }
4365    }
4366
4367    #[test]
4368    fn tokenize_question_mark() {
4369        let dialect = PostgreSqlDialect {};
4370        let sql = "SELECT x ? y";
4371        let tokens = Tokenizer::new(&dialect, sql).tokenize().unwrap();
4372        compare(
4373            tokens,
4374            vec![
4375                Token::make_keyword("SELECT"),
4376                Token::Whitespace(Whitespace::Space),
4377                Token::make_word("x", None),
4378                Token::Whitespace(Whitespace::Space),
4379                Token::Question,
4380                Token::Whitespace(Whitespace::Space),
4381                Token::make_word("y", None),
4382            ],
4383        );
4384    }
4385
4386    #[test]
4387    fn tokenize_multiline_comment_with_comment_hint() {
4388        let sql = String::from("0/*! word */1");
4389
4390        let dialect = MySqlDialect {};
4391        let tokens = Tokenizer::new(&dialect, &sql).tokenize().unwrap();
4392        let expected = vec![
4393            Token::Number("0".to_string(), false),
4394            Token::Whitespace(Whitespace::Space),
4395            Token::Word(Word {
4396                value: "word".to_string(),
4397                quote_style: None,
4398                keyword: Keyword::NoKeyword,
4399            }),
4400            Token::Whitespace(Whitespace::Space),
4401            Token::Number("1".to_string(), false),
4402        ];
4403        compare(expected, tokens);
4404    }
4405
4406    #[test]
4407    fn tokenize_multiline_comment_with_comment_hint_and_version() {
4408        let sql_multi = String::from("0 /*!50110 KEY_BLOCK_SIZE = 1024*/ 1");
4409        let dialect = MySqlDialect {};
4410        let tokens = Tokenizer::new(&dialect, &sql_multi).tokenize().unwrap();
4411        let expected = vec![
4412            Token::Number("0".to_string(), false),
4413            Token::Whitespace(Whitespace::Space),
4414            Token::Whitespace(Whitespace::Space),
4415            Token::Word(Word {
4416                value: "KEY_BLOCK_SIZE".to_string(),
4417                quote_style: None,
4418                keyword: Keyword::KEY_BLOCK_SIZE,
4419            }),
4420            Token::Whitespace(Whitespace::Space),
4421            Token::Eq,
4422            Token::Whitespace(Whitespace::Space),
4423            Token::Number("1024".to_string(), false),
4424            Token::Whitespace(Whitespace::Space),
4425            Token::Number("1".to_string(), false),
4426        ];
4427        compare(expected, tokens);
4428
4429        let tokens = Tokenizer::new(&dialect, "0 /*!50110 */ 1")
4430            .tokenize()
4431            .unwrap();
4432        compare(
4433            vec![
4434                Token::Number("0".to_string(), false),
4435                Token::Whitespace(Whitespace::Space),
4436                Token::Whitespace(Whitespace::Space),
4437                Token::Whitespace(Whitespace::Space),
4438                Token::Number("1".to_string(), false),
4439            ],
4440            tokens,
4441        );
4442
4443        let tokens = Tokenizer::new(&dialect, "0 /*!*/ 1").tokenize().unwrap();
4444        compare(
4445            vec![
4446                Token::Number("0".to_string(), false),
4447                Token::Whitespace(Whitespace::Space),
4448                Token::Whitespace(Whitespace::Space),
4449                Token::Number("1".to_string(), false),
4450            ],
4451            tokens,
4452        );
4453        let tokens = Tokenizer::new(&dialect, "0 /*!   */ 1").tokenize().unwrap();
4454        compare(
4455            vec![
4456                Token::Number("0".to_string(), false),
4457                Token::Whitespace(Whitespace::Space),
4458                Token::Whitespace(Whitespace::Space),
4459                Token::Whitespace(Whitespace::Space),
4460                Token::Whitespace(Whitespace::Space),
4461                Token::Whitespace(Whitespace::Space),
4462                Token::Number("1".to_string(), false),
4463            ],
4464            tokens,
4465        );
4466    }
4467
4468    #[test]
4469    fn tokenize_lt() {
4470        all_dialects().tokenizes_to(
4471            "select a <-50",
4472            vec![
4473                Token::make_keyword("select"),
4474                Token::Whitespace(Whitespace::Space),
4475                Token::make_word("a", None),
4476                Token::Whitespace(Whitespace::Space),
4477                Token::Lt,
4478                Token::Minus,
4479                Token::Number("50".to_string(), false),
4480            ],
4481        );
4482        all_dialects().tokenizes_to(
4483            "select a <+50",
4484            vec![
4485                Token::make_keyword("select"),
4486                Token::Whitespace(Whitespace::Space),
4487                Token::make_word("a", None),
4488                Token::Whitespace(Whitespace::Space),
4489                Token::Lt,
4490                Token::Plus,
4491                Token::Number("50".to_string(), false),
4492            ],
4493        );
4494        all_dialects().tokenizes_to(
4495            "select a <=-50",
4496            vec![
4497                Token::make_keyword("select"),
4498                Token::Whitespace(Whitespace::Space),
4499                Token::make_word("a", None),
4500                Token::Whitespace(Whitespace::Space),
4501                Token::LtEq,
4502                Token::Minus,
4503                Token::Number("50".to_string(), false),
4504            ],
4505        );
4506        all_dialects().tokenizes_to(
4507            "select a <=+50",
4508            vec![
4509                Token::make_keyword("select"),
4510                Token::Whitespace(Whitespace::Space),
4511                Token::make_word("a", None),
4512                Token::Whitespace(Whitespace::Space),
4513                Token::LtEq,
4514                Token::Plus,
4515                Token::Number("50".to_string(), false),
4516            ],
4517        );
4518        all_dialects_where(|d| d.supports_geometric_types()).tokenizes_to(
4519            "select a <->b",
4520            vec![
4521                Token::make_keyword("select"),
4522                Token::Whitespace(Whitespace::Space),
4523                Token::make_word("a", None),
4524                Token::Whitespace(Whitespace::Space),
4525                Token::TwoWayArrow,
4526                Token::make_word("b", None),
4527            ],
4528        );
4529
4530        all_dialects().tokenizes_to(
4531            "select a <-b",
4532            vec![
4533                Token::make_keyword("select"),
4534                Token::Whitespace(Whitespace::Space),
4535                Token::make_word("a", None),
4536                Token::Whitespace(Whitespace::Space),
4537                Token::Lt,
4538                Token::Minus,
4539                Token::make_word("b", None),
4540            ],
4541        );
4542        all_dialects().tokenizes_to(
4543            "select a <+b",
4544            vec![
4545                Token::make_keyword("select"),
4546                Token::Whitespace(Whitespace::Space),
4547                Token::make_word("a", None),
4548                Token::Whitespace(Whitespace::Space),
4549                Token::Lt,
4550                Token::Plus,
4551                Token::make_word("b", None),
4552            ],
4553        );
4554    }
4555}