Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions .github/workflows/labeler/label_dialects.js
Original file line number Diff line number Diff line change
Expand Up @@ -19,6 +19,7 @@ const DIALECTS = [
{ label: "BigQuery", stems: ["bigquery"], pattern: /\bbig\s?query\b/i },
{ label: "ClickHouse", stems: ["clickhouse"], pattern: /\bclick\s?house\b/i },
{ label: "Databricks", stems: ["databricks"], pattern: /\bdatabricks\b/i },
{ label: "Doris", stems: ["doris"], pattern: /\bdoris\b/i },
{ label: "DuckDB", stems: ["duckdb"], pattern: /\bduck\s?db\b/i },
{ label: "Hive", stems: ["hive"], pattern: /\bhive\b/i },
{ label: "MySQL", stems: ["mysql"], pattern: /\b(mysql|maria\s?db)\b/i },
Expand Down
1 change: 1 addition & 0 deletions examples/cli.rs
Original file line number Diff line number Diff line change
Expand Up @@ -52,6 +52,7 @@ $ cargo run --example cli - [--dialectname]
"--postgres" => Box::new(PostgreSqlDialect {}),
"--ms" => Box::new(MsSqlDialect {}),
"--mysql" => Box::new(MySqlDialect {}),
"--doris" => Box::new(DorisDialect {}),
"--snowflake" => Box::new(SnowflakeDialect {}),
"--hive" => Box::new(HiveDialect {}),
"--redshift" => Box::new(RedshiftSqlDialect {}),
Expand Down
10 changes: 6 additions & 4 deletions fuzz/fuzz_targets/fuzz_parse_roundtrip.rs
Original file line number Diff line number Diff line change
Expand Up @@ -19,18 +19,20 @@

use libfuzzer_sys::fuzz_target;
use sqlparser::dialect::{
AnsiDialect, BigQueryDialect, ClickHouseDialect, DatabricksDialect, Dialect, DuckDbDialect,
GenericDialect, HiveDialect, MsSqlDialect, MySqlDialect, OracleDialect, PostgreSqlDialect,
RedshiftSqlDialect, SQLiteDialect, SnowflakeDialect, SparkSqlDialect, TeradataDialect,
AnsiDialect, BigQueryDialect, ClickHouseDialect, DatabricksDialect, Dialect, DorisDialect,
DuckDbDialect, GenericDialect, HiveDialect, MsSqlDialect, MySqlDialect, OracleDialect,
PostgreSqlDialect, RedshiftSqlDialect, SQLiteDialect, SnowflakeDialect, SparkSqlDialect,
TeradataDialect,
};
use sqlparser::parser::Parser;

fuzz_target!(|sql: &str| {
let dialects: [(&str, &dyn Dialect); 16] = [
let dialects: [(&str, &dyn Dialect); 17] = [
("ansi", &AnsiDialect {}),
("bigquery", &BigQueryDialect {}),
("clickhouse", &ClickHouseDialect {}),
("databricks", &DatabricksDialect {}),
("doris", &DorisDialect {}),
("duckdb", &DuckDbDialect {}),
("generic", &GenericDialect {}),
("hive", &HiveDialect {}),
Expand Down
10 changes: 6 additions & 4 deletions fuzz/fuzz_targets/fuzz_parse_sql.rs
Original file line number Diff line number Diff line change
Expand Up @@ -19,17 +19,19 @@

use libfuzzer_sys::fuzz_target;
use sqlparser::dialect::{
AnsiDialect, BigQueryDialect, ClickHouseDialect, DatabricksDialect, Dialect, DuckDbDialect,
GenericDialect, HiveDialect, MsSqlDialect, MySqlDialect, OracleDialect, PostgreSqlDialect,
RedshiftSqlDialect, SQLiteDialect, SnowflakeDialect, SparkSqlDialect, TeradataDialect,
AnsiDialect, BigQueryDialect, ClickHouseDialect, DatabricksDialect, Dialect, DorisDialect,
DuckDbDialect, GenericDialect, HiveDialect, MsSqlDialect, MySqlDialect, OracleDialect,
PostgreSqlDialect, RedshiftSqlDialect, SQLiteDialect, SnowflakeDialect, SparkSqlDialect,
TeradataDialect,
};
use sqlparser::parser::Parser;
fuzz_target!(|sql: &str| {
let dialects: [&dyn Dialect; 16] = [
let dialects: [&dyn Dialect; 17] = [
&AnsiDialect {},
&BigQueryDialect {},
&ClickHouseDialect {},
&DatabricksDialect {},
&DorisDialect {},
&DuckDbDialect {},
&GenericDialect {},
&HiveDialect {},
Expand Down
78 changes: 78 additions & 0 deletions src/dialect/doris.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,78 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.

use crate::{
ast::Expr,
dialect::{Dialect, MySqlDialect},
parser::{Parser, ParserError},
};

/// A [`Dialect`] for [Apache Doris](https://doris.apache.org/).
#[derive(Debug, Default, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
#[cfg_attr(feature = "serde", derive(serde::Serialize, serde::Deserialize))]
pub struct DorisDialect {}

impl Dialect for DorisDialect {

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I believe you are still currently missing:

  • supports_limit_comma
  • parse_infix (the DIV)
  • supports_group_by_with_modifier

I am not familiar enough with Doris to say anything about the other methods, but I suggest you audit them against the engine.

Comment thread
finchxxia marked this conversation as resolved.

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

You should also add DorisDialect to the dialect arrays in fuzz/fuzz_targets/fuzz_parse_sql.rs and fuzz/fuzz_targets/fuzz_parse_roundtrip.rs, and bump the [&dyn Dialect; 16] length. Without that, Doris is the only dialect that is never fuzzed.

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

You should add a { label: "Doris", stems: ["doris"], pattern: /\bdoris\b/i } entry to DIALECTS in .github/workflows/labeler/label_dialects.js. Otherwise the follow-up PRs titled Doris: ... get no dialect label and the bot asks their author for one.

fn supports_nested_comments(&self) -> bool {
true
}

fn is_delimited_identifier_start(&self, ch: char) -> bool {
MySqlDialect {}.is_delimited_identifier_start(ch)
}

fn identifier_quote_style(&self, identifier: &str) -> Option<char> {
MySqlDialect {}.identifier_quote_style(identifier)
}

fn is_identifier_start(&self, ch: char) -> bool {
MySqlDialect {}.is_identifier_start(ch)
}

fn is_identifier_part(&self, ch: char) -> bool {
MySqlDialect {}.is_identifier_part(ch)
}

fn supports_string_literal_backslash_escape(&self) -> bool {
MySqlDialect {}.supports_string_literal_backslash_escape()
}

fn ignores_wildcard_escapes(&self) -> bool {
MySqlDialect {}.ignores_wildcard_escapes()
}

fn supports_numeric_prefix(&self) -> bool {
MySqlDialect {}.supports_numeric_prefix()
}

fn supports_limit_comma(&self) -> bool {
MySqlDialect {}.supports_limit_comma()
}

fn parse_infix(
&self,
parser: &mut Parser,
expr: &Expr,
precedence: u8,
) -> Option<Result<Expr, ParserError>> {
MySqlDialect {}.parse_infix(parser, expr, precedence)
}

fn supports_group_by_with_modifier(&self) -> bool {
MySqlDialect {}.supports_group_by_with_modifier()
}
}
5 changes: 5 additions & 0 deletions src/dialect/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -19,6 +19,7 @@ mod ansi;
mod bigquery;
mod clickhouse;
mod databricks;
mod doris;
mod duckdb;
mod generic;
mod hive;
Expand All @@ -43,6 +44,7 @@ pub use self::ansi::AnsiDialect;
pub use self::bigquery::BigQueryDialect;
pub use self::clickhouse::ClickHouseDialect;
pub use self::databricks::DatabricksDialect;
pub use self::doris::DorisDialect;
pub use self::duckdb::DuckDbDialect;
pub use self::generic::GenericDialect;
pub use self::hive::HiveDialect;
Expand Down Expand Up @@ -2036,6 +2038,7 @@ pub fn dialect_from_str(dialect_name: impl AsRef<str>) -> Option<Box<dyn Dialect
"bigquery" => Some(Box::new(BigQueryDialect)),
"ansi" => Some(Box::new(AnsiDialect {})),
"duckdb" => Some(Box::new(DuckDbDialect {})),
"doris" => Some(Box::new(DorisDialect {})),
"databricks" => Some(Box::new(DatabricksDialect {})),
"spark" | "sparksql" => Some(Box::new(SparkSqlDialect {})),
"oracle" => Some(Box::new(OracleDialect {})),
Expand Down Expand Up @@ -2091,6 +2094,8 @@ mod tests {
assert!(parse_dialect("ANSI").is::<AnsiDialect>());
assert!(parse_dialect("duckdb").is::<DuckDbDialect>());
assert!(parse_dialect("DuckDb").is::<DuckDbDialect>());
assert!(parse_dialect("doris").is::<DorisDialect>());
assert!(parse_dialect("Doris").is::<DorisDialect>());
assert!(parse_dialect("DataBricks").is::<DatabricksDialect>());
assert!(parse_dialect("databricks").is::<DatabricksDialect>());
assert!(parse_dialect("teradata").is::<TeradataDialect>());
Expand Down
1 change: 1 addition & 0 deletions src/test_utils.rs
Original file line number Diff line number Diff line change
Expand Up @@ -286,6 +286,7 @@ pub fn all_dialects() -> TestedDialects {
Box::new(HiveDialect {}),
Box::new(RedshiftSqlDialect {}),
Box::new(MySqlDialect {}),
Box::new(DorisDialect {}),
Box::new(BigQueryDialect {}),
Box::new(SQLiteDialect {}),
Box::new(DuckDbDialect {}),
Expand Down
56 changes: 56 additions & 0 deletions tests/sqlparser_doris.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,56 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.

#![warn(clippy::all)]
//! Test SQL syntax specific to Apache Doris.

#[macro_use]
mod test_utils;

use sqlparser::dialect::DorisDialect;
use test_utils::*;

fn doris() -> TestedDialects {
TestedDialects::new(vec![Box::new(DorisDialect {})])
}

#[test]
fn parse_doris_strings_and_identifiers() {
doris().verified_only_select(
r#"SELECT "double quoted string", 'single quoted string', `select` FROM `db`.`table`"#,
);
}

#[test]
fn parse_doris_limit_comma() {
doris().verified_only_select("SELECT * FROM t LIMIT 5, 10");
}

#[test]
fn parse_doris_div_infix() {
doris().verified_only_select("SELECT 5 DIV 2");
}

#[test]
fn parse_doris_group_by_with_rollup() {
doris().verified_only_select("SELECT * FROM t GROUP BY col1, col2 WITH ROLLUP");
}
Comment thread
finchxxia marked this conversation as resolved.

#[test]
fn parse_doris_nested_comments() {
doris().one_statement_parses_to("SELECT 1 /* a /* b */ c */, 2", "SELECT 1, 2");
}
Loading