Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -757,13 +757,9 @@ object ElasticAggregation {

val currentNestedPath = nested.map(_.nestedPath).getOrElse("")

// No filtering
val fullScript = MetricSelectorScript
.metricSelector(criteria)
.replaceAll("1 == 1 &&", "")
.replaceAll("&& 1 == 1", "")
.replaceAll("1 == 1", "")
.trim
// No filtering at this level is `None`, never a placeholder to strip out of a script: a
// rendered comparison can hold the placeholder's text (`params.max_c1 == 1`).
val fullScript = MetricSelectorScript.selectorScript(criteria).map(_.trim).getOrElse("")

// println(s"[DEBUG] currentNestedPath = $currentNestedPath")
// println(s"[DEBUG] fullScript (complete) = $fullScript")
Expand Down
101 changes: 101 additions & 0 deletions bridge/src/test/scala/app/softnetwork/elastic/sql/BoolQueryModel.scala
Original file line number Diff line number Diff line change
@@ -0,0 +1,101 @@
package app.softnetwork.elastic.sql

import com.fasterxml.jackson.databind.JsonNode

import scala.jdk.CollectionConverters._

/** How Elasticsearch evaluates the `bool` queries a WHERE emits, over documents whose fields `c1 …
* cN` are 0 or 1 (`bits`: field `ci` is bit `i - 1`).
*
* The one rule that matters here, from the Elasticsearch reference: a `bool` with `should` and no
* explicit `minimum_should_match` requires ONE `should` clause only when it holds no `filter` and
* no `must` clause -- next to either, the `should` clauses are optional. Elasticsearch 6 adds one
* exception: a `bool` evaluated in a FILTER context requires one `should` clause anyway. `es6 =
* true` applies that exception. `nested` is read with a ONE-child document (its query is evaluated
* on the same bits); the evaluation context carries through it.
*/
object BoolQueryModel {

final case class Unmodelled(msg: String) extends Exception(msg)

def matches(q: JsonNode, bits: Int, es6: Boolean, filterContext: Boolean = false): Boolean = {
val kinds = q.fieldNames().asScala.toList
if (kinds.size != 1) throw Unmodelled(s"query node with keys $kinds")
val body = q.get(kinds.head)
kinds.head match {
case "bool" =>
val known = Set(
"filter",
"must",
"must_not",
"should",
"minimum_should_match",
"boost",
"adjust_pure_negative"
)
body
.fieldNames()
.asScala
.find(k => !known.contains(k))
.foreach(k => throw Unmodelled(s"bool key $k"))
def clauses(name: String): List[JsonNode] = Option(body.get(name)) match {
case Some(n) if n.isArray => n.elements().asScala.toList
case Some(n) => List(n)
case None => Nil
}
val (filters, musts, nots, shoulds) =
(clauses("filter"), clauses("must"), clauses("must_not"), clauses("should"))
val required = Option(body.get("minimum_should_match")).map(_.asText.toInt).getOrElse {
if (shoulds.isEmpty) 0
else if (es6 && filterContext) 1
else if (filters.isEmpty && musts.isEmpty) 1
else 0
}
filters.forall(matches(_, bits, es6, filterContext = true)) &&
musts.forall(matches(_, bits, es6, filterContext)) &&
nots.forall(n => !matches(n, bits, es6, filterContext = true)) &&
shoulds.count(matches(_, bits, es6, filterContext)) >= required
case "nested" => matches(body.get("query"), bits, es6, filterContext)
case "term" =>
val field = body.fieldNames().asScala.toList.head
val v = Option(body.get(field)).map(n => if (n.isObject) n.get("value") else n).get
val i = """c(\d+)""".r
.findFirstMatchIn(field)
.map(_.group(1).toInt)
.getOrElse(throw Unmodelled(s"term on $field"))
val bit = ((bits >> (i - 1)) & 1) == 1
(if (bit) 1.0 else 0.0) == v.asDouble
case "match" =>
// `MATCH (ci) AGAINST ('1')`: read like `ci = 1` -- one term, the field's value
val field = body.fieldNames().asScala.toList.head
val v = Option(body.get(field)).map(n => if (n.isObject) n.get("query") else n).get
val i = """c(\d+)""".r
.findFirstMatchIn(field)
.map(_.group(1).toInt)
.getOrElse(throw Unmodelled(s"match on $field"))
val bit = ((bits >> (i - 1)) & 1) == 1
bit == (v.asText == "1")
case "match_all" => true
case other => throw Unmodelled(s"query kind $other")
}
}

/** Every `bool` that holds `should` clauses next to `filter` or `must` clauses without an
* explicit `minimum_should_match` -- where its `should` clauses are optional.
*/
def optionalShoulds(q: JsonNode): Int = {
var count = 0
def walk(n: JsonNode): Unit =
if (n.isObject) {
Option(n.get("bool")).filter(_.isObject).foreach { b =>
val should = Option(b.get("should")).exists(_.size > 0)
val required =
Option(b.get("filter")).exists(_.size > 0) || Option(b.get("must")).exists(_.size > 0)
if (should && required && b.get("minimum_should_match") == null) count += 1
}
n.elements().asScala.foreach(walk)
} else if (n.isArray) n.elements().asScala.foreach(walk)
walk(q)
count
}
}
Original file line number Diff line number Diff line change
@@ -0,0 +1,128 @@
package app.softnetwork.elastic.sql

import app.softnetwork.elastic.sql.bridge._
import app.softnetwork.elastic.sql.parser.ConditionPopulation._
import app.softnetwork.elastic.sql.parser.Parser
import app.softnetwork.elastic.sql.query.SingleSearch
import com.fasterxml.jackson.databind.{JsonNode, ObjectMapper}
import org.scalatest.flatspec.AnyFlatSpec
import org.scalatest.matchers.should.Matchers

/** The Elasticsearch query a WHERE emits evaluates SQL's reading of the condition.
*
* 🔴 A correct condition TREE is not enough: an unparenthesised sub-condition used to be written
* into its parent's `bool`, where an OR under an AND put its `should` clauses next to `filter`
* clauses -- optional for Elasticsearch. The query is evaluated here with Elasticsearch's own
* `bool` rules (`BoolQueryModel`), for Elasticsearch 7+ and for the Elasticsearch 6 filter-context
* rule, against the oracle of the generated structure.
*/
class ConditionBoolEmissionSpec extends AnyFlatSpec with Matchers {

implicit val timestamp: Long = 0L

private val mapper = new ObjectMapper()

/** P2: the WHERE statements of P1 whose leaves are all `ci = 1` (every class but the kinds). */
private lazy val wherePopulation: List[Gen] =
population().filter(g => g.clause == WhereC && g.cls != "K")

private def queryOf(search: SingleSearch): JsonNode = {
val request: ElasticSearchRequest = search
mapper.readTree(request.query).get("query")
}

private def searchOf(sql: String): SingleSearch = Parser(sql) match {
case Right(s: SingleSearch) => s
case other => fail(s"[$sql] $other")
}

"a WHERE that combines AND and OR" should "emit a query that evaluates SQL's reading" in {
val wrong = wherePopulation.flatMap { g =>
val q = queryOf(searchOf(g.sql))
val bits = 0 until (1 << g.n)
val es7 = bits.map(BoolQueryModel.matches(q, _, es6 = false)).toVector
val es6 = bits.map(BoolQueryModel.matches(q, _, es6 = true)).toVector
if (es7 == g.ansiTable && es6 == g.ansiTable) None
else Some(s"[${g.sql}] es7=${es7 == g.ansiTable} es6=${es6 == g.ansiTable}: $q")
}
withClue(
s"${wrong.size} of ${wherePopulation.size} wrong, first 10:\n${wrong.take(10).mkString("\n")}\n"
) {
wrong shouldBe empty
}
}

it should "never leave should clauses beside filter or must clauses" in {
val offending = wherePopulation
.map(g => g.sql -> BoolQueryModel.optionalShoulds(queryOf(searchOf(g.sql))))
.filter(_._2 > 0)
withClue(s"first 10: ${offending.take(10).mkString("\n")}\n") { offending shouldBe empty }
}

"a WHERE with ONE operator" should "stay one flat bool" in {
queryOf(
searchOf("SELECT id FROM t WHERE c1 = 1 AND c2 = 1 AND c3 = 1 AND c4 = 1")
).toString shouldBe
"""{"bool":{"filter":[{"term":{"c1":{"value":1}}},{"term":{"c2":{"value":1}}},{"term":{"c3":{"value":1}}},{"term":{"c4":{"value":1}}}]}}"""
queryOf(searchOf("SELECT id FROM t WHERE c1 = 1 OR c2 = 1 OR c3 = 1")).toString shouldBe
"""{"bool":{"should":[{"term":{"c1":{"value":1}}},{"term":{"c2":{"value":1}}},{"term":{"c3":{"value":1}}}]}}"""
}

"DELETE and UPDATE" should "send the query the same WHERE sends in a SELECT" in {
val wrong = wherePopulation.filter(_.mixedLevel).flatMap { g =>
val select = searchOf(g.sql)
val expected = queryOf(select)
Seq(
"DELETE" -> select.copy(deleteByQuery = true),
"UPDATE" -> select.copy(updateByQuery = true)
)
.collect { case (kind, s) if queryOf(s) != expected => s"$kind [${g.sql}]: ${queryOf(s)}" }
}
wrong shouldBe empty
}

"an OR group holding a MATCH, under an AND" should "stay ONE condition of the AND" in {
// Spread into the root bool beside its `filter` clauses, the group's `should` clauses turned
// optional: `c1 = 1 AND (MATCH ... OR c3 = 1)` selected every document with c1 = 1. A MATCH
// over several columns is such a group too. The group that IS the whole condition is still
// spread, which is exact (the flat `should` of a lone MATCH).
Seq(
"c1 = 1 AND (MATCH (c2) AGAINST ('1') OR c3 = 1)" -> (3, A(L(1), O(L(2), L(3)))),
"(MATCH (c1) AGAINST ('1') OR c2 = 1) AND c3 = 1" -> (3, A(O(L(1), L(2)), L(3))),
"c1 = 1 AND MATCH (c2, c3) AGAINST ('1')" -> (3, A(L(1), O(L(2), L(3)))),
"(c1 = 1 OR MATCH (c2) AGAINST ('1')) AND (c3 = 1 OR MATCH (c4) AGAINST ('1'))" ->
(4, A(O(L(1), L(2)), O(L(3), L(4)))),
"MATCH (c1, c2) AGAINST ('1')" -> (2, O(L(1), L(2))),
"MATCH (c1) AGAINST ('1') OR c2 = 1 AND c3 = 1" -> (3, O(L(1), A(L(2), L(3))))
).foreach { case (cond, (n, expected)) =>
val q = queryOf(searchOf(s"SELECT id FROM t WHERE $cond"))
val bits = 0 until (1 << n)
withClue(s"[$cond] $q ") {
bits.map(BoolQueryModel.matches(q, _, es6 = false)).toVector shouldBe table(expected, n)
bits.map(BoolQueryModel.matches(q, _, es6 = true)).toVector shouldBe table(expected, n)
BoolQueryModel.optionalShoulds(q) shouldBe 0
}
}
}

"a WHERE over an UNNEST column" should "emit the nested query SQL's reading needs" in {
// one child per document: a nested query is its own query
Seq(
"inner_items.c1 = 1 OR inner_items.c2 = 1 AND inner_items.c3 = 1" -> O(L(1), A(L(2), L(3))),
"c1 = 1 OR inner_items.c2 = 1 AND inner_items.c3 = 1" -> O(L(1), A(L(2), L(3))),
"inner_items.c1 = 1 AND inner_items.c2 = 1 OR c3 = 1" -> O(A(L(1), L(2)), L(3))
).foreach { case (cond, expected) =>
val q = queryOf(searchOf(s"SELECT id FROM t JOIN UNNEST(t.items) AS inner_items WHERE $cond"))
withClue(s"[$cond] $q ") {
(0 until 8).map(BoolQueryModel.matches(q, _, es6 = false)).toVector shouldBe table(
expected,
3
)
(0 until 8).map(BoolQueryModel.matches(q, _, es6 = true)).toVector shouldBe table(
expected,
3
)
}
}
}
}
Original file line number Diff line number Diff line change
Expand Up @@ -351,6 +351,26 @@ class HavingFunctionEmissionSpec extends AnyFlatSpec with Matchers {
)
}

"a comparison of an aggregate with 1 or 10" should "read that aggregate's own parameter" in {
// 🔴 The emission stripped every `1 == 1` out of the script -- the text the selector answers
// when there is nothing to filter -- and `params.max_c1 == 1` holds that text. MEASURED on the
// base: `= 1` read `params.max_c`, `= 10` read `params.max_c0`, and `IN (1, 2)` lost its first
// member; on Elasticsearch 8.18.3 all three searches failed (`Cannot invoke
// "Object.getClass()" because "value" is null`).
Seq(
"MAX(c1) = 1" -> "(params.max_c1 == null ? false : (params.max_c1 == 1))",
"MAX(c1) = 10" -> "(params.max_c1 == null ? false : (params.max_c1 == 10))",
"MAX(c1) IN (1, 2)" -> "(params.max_c1 == null ? false : (params.max_c1 == 1 || params.max_c1 == 2))"
).foreach { case (condition, script) =>
withClue(s"[$condition] ") {
queryOf(s"SELECT g, COUNT(*) AS cnt FROM t GROUP BY g HAVING $condition") should include(
""""having_filter":{"bucket_selector":{"buckets_path":{"max_c1":"max_c1"},""" +
s""""script":{"source":"$script"}}}"""
)
}
}
}

"a wrapped aggregate beside a bucket-key predicate" should "honour BOTH mechanisms" in {
// The key predicate is a `terms` exclude, the aggregate one a `bucket_selector`. Neither may
// cost the other.
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -5002,8 +5002,8 @@ class SQLQuerySpec extends AnyFlatSpec with Matchers {
}

it should "emit no bucket_selector for a HAVING with no aggregate over an aggregate-free GROUP BY" in {
// `metricSelectorForBucket` strips "1 == 1" to the empty string, so the HAVING becomes a terms
// exclude and no bucket_selector is produced.
// `metricSelectorForBucket` finds nothing to filter at this level (`selectorScript` is `None`),
// so the HAVING becomes a terms exclude and no bucket_selector is produced.
val select: ElasticSearchRequest =
SelectStatement("SELECT category FROM Table GROUP BY category HAVING category <> 'x'")
val query = select.query
Expand Down
21 changes: 21 additions & 0 deletions documentation/sql/known_limitations.md
Original file line number Diff line number Diff line change
Expand Up @@ -295,6 +295,27 @@ within the elements an UNNEST projection returns per parent. Two inner columns w
coincide (`o.id` and `items.id`) are refused: alias one of them (measured on Elasticsearch 8.18 and
6.8). See [JOIN UNNEST](dql_statements.md#join-unnest).

## `NOT` applies to one condition

`NOT` negates the one condition written after it, and only some spellings of it are accepted:

| Written | Verdict |
|---------|---------|
| `NOT price > 100`, `NOT status = 'x'` (before a comparison) | Accepted |
| `status NOT IN (...)`, `name NOT LIKE 'x%'`, `price NOT BETWEEN 1 AND 2`, `manager IS NOT NULL` | Accepted |
| `NOT (a = 1 OR b = 2)`, `NOT (price > 100)` (before a parenthesised group) | Parse error |
| `NOT is_active` (before a bare boolean column) | Parse error |
| `NOT status IN (...)`, `NOT name LIKE 'x%'`, `NOT manager IS NULL`, `NOT MATCH (...) AGAINST (...)`, `NOT ISNULL(x)` at the start of a condition | Parse error |
| `a = 1 AND NOT name LIKE 'x%'` (the same spellings right after `AND` / `OR`) | Accepted in some positions only: `a = 1 AND b = 2 AND NOT name LIKE 'x%'` is a parse error |
| `a = 1 OR NOT ISNULL(x) AND b = 2` (also with `ISNOTNULL(x)`) | Accepted, read as `a = 1 OR (ISNOTNULL(x) AND b = 2)` |
| `a = 1 OR NOT MATCH (t) AGAINST ('x') AND b = 2` | Refused: *"NOT ... cannot start conditions joined by AND after an OR"* |

Write the condition De Morgan's laws give instead of a `NOT` over a group (`NOT (a = 1 OR b = 2)` is
`NOT a = 1 AND NOT b = 2`), compare a boolean column (`is_active = false`), put the `NOT` of `IN`,
`LIKE`, `RLIKE`, `BETWEEN` and `IS NULL` after the column, write `ISNOTNULL(x)` for `NOT ISNULL(x)`
(and `ISNULL(x)` for `NOT ISNOTNULL(x)`), and move a negated `MATCH` to the end of its `AND` group
(`a = 1 OR (b = 2 AND NOT MATCH (t) AGAINST ('x'))`).

## Coming in the upcoming release (Quarter 1 2027)

- **Heterogeneous federation**: JOIN or correlate Elasticsearch with PostgreSQL, MySQL, ClickHouse, Snowflake, and more — plus cross-cluster subqueries (e.g. correlate one cluster's data against another's).
Expand Down
Loading
Loading