Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
@@ -0,0 +1,19 @@
/*
* Copyright 2025 SOFTNETWORK
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/

package app.softnetwork.elastic.client

class JestClientBiDialectExecutionSpec extends BiDialectExecutionSpec
Original file line number Diff line number Diff line change
@@ -0,0 +1,19 @@
/*
* Copyright 2025 SOFTNETWORK
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/

package app.softnetwork.elastic.client

class RestHighLevelClientBiDialectExecutionSpec extends BiDialectExecutionSpec
Original file line number Diff line number Diff line change
@@ -0,0 +1,19 @@
/*
* Copyright 2025 SOFTNETWORK
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/

package app.softnetwork.elastic.client

class RestHighLevelClientBiDialectExecutionSpec extends BiDialectExecutionSpec
Original file line number Diff line number Diff line change
@@ -0,0 +1,19 @@
/*
* Copyright 2025 SOFTNETWORK
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/

package app.softnetwork.elastic.client

class JavaClientBiDialectExecutionSpec extends BiDialectExecutionSpec
Original file line number Diff line number Diff line change
@@ -0,0 +1,19 @@
/*
* Copyright 2025 SOFTNETWORK
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/

package app.softnetwork.elastic.client

class JavaClientBiDialectExecutionSpec extends BiDialectExecutionSpec
Original file line number Diff line number Diff line change
Expand Up @@ -157,6 +157,7 @@ import app.softnetwork.elastic.sql.query.{
OrderBy,
RightJoin,
Select,
Top,
Unnest,
Where
}
Expand Down Expand Up @@ -190,6 +191,10 @@ object SQLKeywords {
/** Clause, join, operator and CASE syntax keywords (word-bearing TokenRegex objects). */
val clauseTokens: List[TokenRegex] = List(
Select,
// T-SQL's row bound, accepted as a spelling of LIMIT. Listed here because the `Expr` scan in
// `SQLKeywordsSpec` requires every word-bearing TokenRegex object to be registered; it is NOT
// reserved, and `SELECT top FROM t` still reads `top` as a column.
Top,
Distinct,
From,
Where,
Expand Down Expand Up @@ -439,6 +444,7 @@ object SQLKeywords {
"PARAMS",
"PARQUET",
"PARTITION",
"PERCENT",
"PIPELINE",
"PIPELINES",
"POLICIES",
Expand Down Expand Up @@ -466,6 +472,7 @@ object SQLKeywords {
"TABLE",
"TABLES",
"TEMPORARY",
"TIES",
"TO",
"TRUE",
"TRUNCATE",
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -72,8 +72,17 @@ package object string {
case object LeftOp extends Expr("LEFT") with StringOp
case object RightOp extends Expr("RIGHT") with StringOp
case object For extends Expr("FOR") with TokenRegex

/** `CHAR_LENGTH`/`CHARACTER_LENGTH` are the SQL-92 spellings a BI tool emits for its `LEN()`
* calculated field. They are aliases, so the render normalises to `LENGTH` and re-parses.
*
* Order is NOT load-bearing here, and an earlier draft of this comment claimed it was:
* `TokenRegex.regex` appends `\b`, so `LEN` cannot match the prefix of `LENGTH` whatever the
* order, and neither `CHAR_LENGTH` nor `CHARACTER_LENGTH` is a prefix of the other. Longest
* first is kept as house style only — the real rule lives on `TokenRegex.regex`.
*/
case object Length extends Expr("LENGTH") with StringOp {
override lazy val words: List[String] = List(sql, "LEN")
override lazy val words: List[String] = List(sql, "CHARACTER_LENGTH", "CHAR_LENGTH", "LEN")
}
case object Replace extends Expr("REPLACE") with StringOp {
override lazy val words: List[String] = List(sql, "STR_REPLACE")
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -451,7 +451,7 @@ package object time {

case object DateDiff extends Expr("DATE_DIFF") with TokenRegex with PainlessScript {
override def painless(context: Option[PainlessContext]): String = ".between"
override lazy val words: List[String] = List(sql, "DATEDIFF")
override lazy val words: List[String] = List(sql, "TIMESTAMPDIFF", "DATEDIFF")
}

case class DateDiff(
Expand Down Expand Up @@ -809,8 +809,13 @@ package object time {
}
}

/** `TIMESTAMPADD` is the ODBC/JDBC spelling (`{fn TIMESTAMPADD(SQL_TSI_DAY, -89, …)}`), which BI
* tools emit directly. It is an ALIAS, not a second mechanism: the `(unit, count, base)`
* argument order it uses is the `transactSql` form this function has always parsed, so the
* render normalises to `DATETIME_ADD` and re-parses.
*/
case object DateTimeAdd extends Expr("DATETIME_ADD") with TokenRegex {
override lazy val words: List[String] = List(sql, "DATETIMEADD")
override lazy val words: List[String] = List(sql, "DATETIMEADD", "TIMESTAMPADD")
}

case class DateTimeAdd(
Expand Down
33 changes: 28 additions & 5 deletions sql/src/main/scala/app/softnetwork/elastic/sql/parser/Parser.scala
Original file line number Diff line number Diff line change
Expand Up @@ -87,10 +87,19 @@ object Parser
with OrderByParser
with LimitParser {

/** 🔴 `TOP n` and `LIMIT n` are two spellings of ONE row bound, so carrying both would need a
* precedence rule nobody could guess from the SQL. Combining them is refused by name instead;
* `t.orElse(l)` then never has to choose.
*/
lazy val single: PackratParser[SingleSearch] = {
select ~ from ~ where.? ~ groupBy.? ~ having.? ~ orderBy.? ~ limit.? ~ onConflict.? ^^ {
case s ~ f ~ w ~ g ~ h ~ o ~ l ~ oc =>
SingleSearch(s, f, w, g, h, o, l, onConflict = oc).update()
select ~ from ~ where.? ~ groupBy.? ~ having.? ~ orderBy.? ~ limit.? ~ onConflict.? >> {
case (s, t) ~ f ~ w ~ g ~ h ~ o ~ l ~ oc =>
if (t.isDefined && l.isDefined)
err(
"TOP and LIMIT both bound the number of rows -- use one of them, not both. TOP " +
"carries no OFFSET, so paging needs the LIMIT n OFFSET m spelling"
)
else success(SingleSearch(s, f, w, g, h, o, t.orElse(l), onConflict = oc).update())
}
}

Expand Down Expand Up @@ -178,7 +187,14 @@ object Parser
* schemaProbeSql rewrites `SELECT 1` into.
*/
lazy val fromlessSelect: PackratParser[FromlessSelect] =
select ~ limit.? ^^ { case s ~ l => FromlessSelect(s, l) }
select ~ limit.? >> { case (s, t) ~ l =>
if (t.isDefined && l.isDefined)
err(
"TOP and LIMIT both bound the number of rows -- use one of them, not both. TOP " +
"carries no OFFSET, so paging needs the LIMIT n OFFSET m spelling"
)
else success(FromlessSelect(s, t.orElse(l)))
}

lazy val row: PackratParser[List[Value[_]]] =
lparen ~> repsep(array_of_struct | struct | value, comma) <~ rparen
Expand Down Expand Up @@ -957,8 +973,15 @@ object Parser
lazy val neverWatcherCondition: PackratParser[NeverWatcherCondition.type] =
keyword("NEVER") ^^ { _ => NeverWatcherCondition }

/** 🔴 Ordered LONGEST spelling first, and it is load-bearing. These are string literals, not
* anchored tokens: `gt` (`>`) matches the first character of `>=` and succeeds, leaving `=` for
* the value production, which then fails with *"A value or a date/datetime function must be
* provided for comparison"*. `WHEN x >= 0` and `WHEN x <= 0` were rejected for exactly that
* reason, while `>`, `<`, `=` and `<>` parsed. `WhereParser.comparisonOp` has always had the
* right order; this production had not.
*/
private lazy val comparison_operator: PackratParser[ComparisonOperator] =
eq | ne | diff | gt | ge | lt | le
eq | ne | diff | ge | gt | le | lt

private lazy val dateMathScript
: PackratParser[DateTimeFunction with FunctionWithIdentifier with DateMathScript] =
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -16,7 +16,7 @@

package app.softnetwork.elastic.sql.parser

import app.softnetwork.elastic.sql.query.{Except, Field, Select}
import app.softnetwork.elastic.sql.query.{Except, Field, Limit, Select, Top}

trait SelectParser {
self: Parser with WhereParser =>
Expand All @@ -40,12 +40,46 @@ trait SelectParser {
Except(e)
}

lazy val select: PackratParser[Select] =
Select.regex ~ rep1sep(
/** `TOP n` — T-SQL's row bound, which Tableau emits in its SQL-92 dialect. It is returned
* ALONGSIDE the `Select` rather than stored on it, so the statement keeps exactly one owner of
* its row bound (`Parser.single` folds it into `limit`) and the render normalises to `LIMIT n`.
*
* 🔴 `TOP` is NOT reserved, and must not become so: `SELECT top FROM t` selects a column named
* `top` today. `top.?` is safe because `Top.regex ~> long` FAILS (it does not error) when no
* number follows, and `opt` backtracks to the original position, where `field` reads `top` as
* the identifier it is. Both readings are pinned in `ParserSpec`.
*/
lazy val top: PackratParser[Limit] =
Top.regex ~> (start ~> long <~ end | long) >> { l =>
if (l.value < 0 || l.value > Int.MaxValue) {
// 🔴 `failure`, NEVER `err`, and the distinction is the whole correctness of `top.?`.
// `err` yields an `Error`, and `Parsers.|` does not try another alternative after an
// Error — so `opt` could not backtrack and `SELECT top -1 AS x FROM t`, which reads
// `top - 1` and parses on main, became a hard rejection. At this position `TOP` is
// genuinely ambiguous between the clause and a column called `top`; a `Failure` lets
// `field` settle it, which is the only reading that cannot regress.
failure(s"TOP takes a row count between 0 and ${Int.MaxValue}")
} else {
// `TOP n PERCENT` and `TOP n WITH TIES` are real T-SQL that this engine does not
// implement. They are refused BY NAME because the alternative is far worse: with `top`
// having consumed `TOP 5`, `field` reads `PERCENT a` as the column `PERCENT` aliased to
// `a`, so `SELECT TOP 5 PERCENT a FROM t` returned rows for a column the user never
// named — a loud rejection turned into a silent wrong answer (#205/#253 family).
// An `err` is safe HERE, unlike above: no spelling of `TOP <n> PERCENT` parsed before.
(keyword("PERCENT") | (keyword("WITH") ~ keyword("TIES") ^^ (_ => "WITH TIES"))) >> { w =>
err(
s"SELECT TOP n $w is not supported -- TOP takes a row COUNT. Use TOP n, or LIMIT n"
)
} | success(Limit(l.value.toInt, None))
}
}

lazy val select: PackratParser[(Select, Option[Limit])] =
Select.regex ~ top.? ~ rep1sep(
field,
separator
) ~ except.? ^^ { case _ ~ fields ~ e =>
Select(fields, e)
) ~ except.? ^^ { case _ ~ t ~ fields ~ e =>
(Select(fields, e), t)
}

}
Original file line number Diff line number Diff line change
Expand Up @@ -151,6 +151,14 @@ case class Field(

case object Except extends Expr("except") with TokenRegex

/** `SELECT TOP n` (T-SQL; Tableau emits it in its SQL-92 dialect, e.g. `SELECT TOP 1 *`).
*
* It is a SPELLING of the row bound, not a second bound: `SelectParser.select` hands it to
* `Parser.single`, which stores it in the statement's `limit` and renders `LIMIT n`. So there is
* exactly ONE owner of the row bound in the AST, and the render re-parses.
*/
case object Top extends Expr("TOP") with TokenRegex

case class Except(fields: Seq[Field]) extends Updateable {
override def sql: String = s" $Except(${fields.mkString(",")})"
def update(request: SingleSearch): Except =
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -99,7 +99,19 @@ package object time {
}

sealed trait TimeUnit extends PainlessScript with DateMathScript with DateMathRounding {
lazy val regex: Regex = s"\\b(?i)$sql(s)?\\b".r

/** Accepted spellings of this unit, canonical first. The ODBC/JDBC interval names
* (`SQL_TSI_DAY`, …) are what a BI tool passes to `TIMESTAMPADD`/`TIMESTAMPDIFF`, so they are
* aliases of the units we already have rather than units of their own — the AST carries one
* spelling and the render normalises to `sql`.
*
* 🔴 `SQL_TSI_DAY` could never have matched the previous `\b(?i)DAY(s)?\b`: `_` is a word
* character, so there is no word boundary before `DAY` inside it. Adding the alias therefore
* cannot change what any existing statement parses to — it can only accept more.
*/
def words: List[String] = List(sql)

lazy val regex: Regex = s"\\b(?i)(${words.mkString("|")})(s)?\\b".r

def timeUnit: String = sql.toUpperCase() + "S"

Expand Down Expand Up @@ -129,31 +141,39 @@ package object time {
}

case object YEARS extends Expr("YEAR") with CalendarUnit {
override def words: List[String] = List(sql, "SQL_TSI_YEAR")
override def script: Option[String] = Some("y")
}
case object MONTHS extends Expr("MONTH") with CalendarUnit {
override def words: List[String] = List(sql, "SQL_TSI_MONTH")
override def script: Option[String] = Some("M")
}
case object QUARTERS extends Expr("QUARTER") with CalendarUnit {
override def words: List[String] = List(sql, "SQL_TSI_QUARTER")
override def script: Option[String] = throw new IllegalArgumentException(
"Quarter must be converted to months (value * 3) before creating date-math"
)
}
case object WEEKS extends Expr("WEEK") with CalendarUnit {
override def words: List[String] = List(sql, "SQL_TSI_WEEK")
override def script: Option[String] = Some("w")
}

case object DAYS extends Expr("DAY") with CalendarUnit with FixedUnit {
override def words: List[String] = List(sql, "SQL_TSI_DAY")
override def script: Option[String] = Some("d")
}

case object HOURS extends Expr("HOUR") with FixedUnit {
override def words: List[String] = List(sql, "SQL_TSI_HOUR")
override def script: Option[String] = Some("H")
}
case object MINUTES extends Expr("MINUTE") with FixedUnit {
override def words: List[String] = List(sql, "SQL_TSI_MINUTE")
override def script: Option[String] = Some("m")
}
case object SECONDS extends Expr("SECOND") with FixedUnit {
override def words: List[String] = List(sql, "SQL_TSI_SECOND")
override def script: Option[String] = Some("s")
}

Expand Down
Loading
Loading