From 59dbd88a99f8db6e15c3814ecb06f47a24578bac Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?St=C3=A9phane=20Manciot?= Date: Sat, 3 Oct 2026 00:37:46 +0200 Subject: [PATCH 1/4] fix(sql): DATEDIFF / DATE_DIFF / TIMESTAMPDIFF follow their vendors; DATE_TRUNC(..., WEEK) is the ISO Monday Direction by argument layout: the dates-first forms (DATEDIFF(a, b[, unit]), DATE_DIFF(a, b[, unit])) return first - second, as MySQL's DATEDIFF and BigQuery's DATE_DIFF(end_date, start_date, granularity) do; the unit-first forms return last - middle, as SQL Server, Snowflake, Redshift, DuckDB, Trino, Elasticsearch SQL and MySQL's TIMESTAMPDIFF do. Counting by name: DATEDIFF and DATE_DIFF count the calendar boundaries crossed; TIMESTAMPDIFF counts the whole units elapsed between instants, with MySQL's month algorithm. Weeks are ISO 8601 (Monday), boundaries UTC. DATE_TRUNC(x, WEEK) returns the ISO Monday in every venue. One rendering serves row level, WHERE, per group, the HAVING alias route, the view calculation channel and the ingest processor. The sign spec no longer rests on a wrong reading of BigQuery's DATE_DIFF. Docs and help state the definition, rewrite the age / tenure examples and add the 0.24.0 change and migration notes. Co-Authored-By: Claude Opus 5.5 (1M context) --- .../elastic/sql/SQLQuerySpec.scala | 8 +- .../help/commands/ddl/create_table.json | 2 +- .../help/functions/date/date_diff.json | 42 +- .../help/functions/date/datediff.json | 23 +- documentation/sql/ddl_statements.md | 10 +- documentation/sql/dql_statements.md | 40 +- documentation/sql/functions_date_time.md | 142 +++- documentation/sql/functions_math.md | 2 +- .../elastic/sql/SQLQuerySpec.scala | 8 +- .../elastic/sql/function/time/package.scala | 315 +++++++-- .../sql/parser/function/time/package.scala | 54 +- .../elastic/sql/parser/DateDiffSignSpec.scala | 167 +++-- .../elastic/sql/parser/ParserSpec.scala | 37 +- .../sql/parser/TableauDialectSpec.scala | 17 +- .../sql/query/HavingAliasResolutionSpec.scala | 2 +- .../query/MaterializedViewHavingSpec.scala | 12 +- .../sql/schema/IngestTemporalSpec.scala | 35 +- .../client/BiDialectExecutionSpec.scala | 72 +- .../client/DateFunctionExecutionSpec.scala | 657 ++++++++++++++++-- .../client/GatewayApiIntegrationSpec.scala | 33 +- .../client/GroupByCompletenessSpec.scala | 7 +- .../repl/ReplGatewayIntegrationSpec.scala | 6 +- 22 files changed, 1284 insertions(+), 407 deletions(-) diff --git a/bridge/src/test/scala/app/softnetwork/elastic/sql/SQLQuerySpec.scala b/bridge/src/test/scala/app/softnetwork/elastic/sql/SQLQuerySpec.scala index 785d3234d..b0f1f2888 100644 --- a/bridge/src/test/scala/app/softnetwork/elastic/sql/SQLQuerySpec.scala +++ b/bridge/src/test/scala/app/softnetwork/elastic/sql/SQLQuerySpec.scala @@ -1421,13 +1421,13 @@ class SQLQuerySpec extends AnyFlatSpec with Matchers { | "w": { | "script": { | "lang": "painless", - | "source": "def param1 = (doc['lastUpdated'].size() == 0 ? null : doc['lastUpdated'].value.toInstant().atZone(ZoneId.of('Z')).with(DayOfWeek.SUNDAY).truncatedTo(ChronoUnit.DAYS)); def param2 = DateTimeFormatter.ofPattern(\"yyyy-MM-dd\"); (param1 == null) ? null : param2.format(param1)" + | "source": "def param1 = (doc['lastUpdated'].size() == 0 ? null : doc['lastUpdated'].value.toInstant().atZone(ZoneId.of('Z')).with(DayOfWeek.MONDAY).truncatedTo(ChronoUnit.DAYS)); def param2 = DateTimeFormatter.ofPattern(\"yyyy-MM-dd\"); (param1 == null) ? null : param2.format(param1)" | } | }, | "w2": { | "script": { | "lang": "painless", - | "source": "def param1 = (doc['lastUpdated'].size() == 0 ? null : doc['lastUpdated'].value.toInstant().atZone(ZoneId.of('Z')).with(DayOfWeek.SUNDAY).truncatedTo(ChronoUnit.DAYS)); def param2 = DateTimeFormatter.ofPattern(\"yyyy-MM-dd\"); (param1 == null) ? null : param2.format(param1)" + | "source": "def param1 = (doc['lastUpdated'].size() == 0 ? null : doc['lastUpdated'].value.toInstant().atZone(ZoneId.of('Z')).with(DayOfWeek.MONDAY).truncatedTo(ChronoUnit.DAYS)); def param2 = DateTimeFormatter.ofPattern(\"yyyy-MM-dd\"); (param1 == null) ? null : param2.format(param1)" | } | }, | "d": { @@ -1655,7 +1655,7 @@ class SQLQuerySpec extends AnyFlatSpec with Matchers { | "diff": { | "script": { | "lang": "painless", - | "source": "def param1 = (doc['createdAt'].size() == 0 ? null : doc['createdAt'].value.toInstant().atZone(ZoneId.of('Z'))); def param2 = (doc['updatedAt'].size() == 0 ? null : doc['updatedAt'].value.toInstant().atZone(ZoneId.of('Z'))); (param1 == null || param2 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param1.toLocalDate(), param2.toLocalDate()))" + | "source": "def param1 = (doc['updatedAt'].size() == 0 ? null : doc['updatedAt'].value.toInstant().atZone(ZoneId.of('Z'))); def param2 = (doc['createdAt'].size() == 0 ? null : doc['createdAt'].value.toInstant().atZone(ZoneId.of('Z'))); (param1 == null || param2 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param1.toLocalDate(), param2.toLocalDate()))" | } | } | }, @@ -1712,7 +1712,7 @@ class SQLQuerySpec extends AnyFlatSpec with Matchers { | "max": { | "script": { | "lang": "painless", - | "source": "def param1 = (doc['createdAt'].size() == 0 ? null : doc['createdAt'].value); def param2 = (doc['updatedAt'].size() == 0 ? null : doc['updatedAt'].value.toInstant().atZone(ZoneId.of('Z'))); def param3 = (param1 == null) ? null : ZonedDateTime.parse(param1, new DateTimeFormatterBuilder().appendPattern(\"yyyy-MM-dd HH:mm:ss\").appendFraction(ChronoField.NANO_OF_SECOND, 0, 9, true).toFormatter().withZone(ZoneId.of('Z'))); (param1 == null || param2 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param3.toLocalDate(), param2.toLocalDate()))" + | "source": "def param1 = (doc['updatedAt'].size() == 0 ? null : doc['updatedAt'].value.toInstant().atZone(ZoneId.of('Z'))); def param2 = (doc['createdAt'].size() == 0 ? null : doc['createdAt'].value); def param3 = (param2 == null) ? null : ZonedDateTime.parse(param2, new DateTimeFormatterBuilder().appendPattern(\"yyyy-MM-dd HH:mm:ss\").appendFraction(ChronoField.NANO_OF_SECOND, 0, 9, true).toFormatter().withZone(ZoneId.of('Z'))); (param1 == null || param2 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param1.toLocalDate(), param3.toLocalDate()))" | } | } | } diff --git a/core/src/main/resources/help/commands/ddl/create_table.json b/core/src/main/resources/help/commands/ddl/create_table.json index ef23b06ef..3af497039 100644 --- a/core/src/main/resources/help/commands/ddl/create_table.json +++ b/core/src/main/resources/help/commands/ddl/create_table.json @@ -126,7 +126,7 @@ { "title": "Table with computed column", "description": "Create table with script-based computed field", - "sql": "CREATE TABLE employees (\n id INT NOT NULL,\n birthdate DATE,\n age INT SCRIPT AS (DATEDIFF(birthdate, CURRENT_DATE, YEAR)),\n hire_date DATE,\n tenure INT SCRIPT AS (DATEDIFF(hire_date, CURRENT_DATE, DAY)),\n PRIMARY KEY (id)\n)" + "sql": "CREATE TABLE employees (\n id INT NOT NULL,\n birthdate DATE,\n age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)),\n hire_date DATE,\n tenure INT SCRIPT AS (DATEDIFF(CURRENT_DATE, hire_date, DAY)),\n PRIMARY KEY (id)\n)" }, { "title": "Partitioned table", diff --git a/core/src/main/resources/help/functions/date/date_diff.json b/core/src/main/resources/help/functions/date/date_diff.json index aee75ba0e..aeec1fa14 100644 --- a/core/src/main/resources/help/functions/date/date_diff.json +++ b/core/src/main/resources/help/functions/date/date_diff.json @@ -7,26 +7,26 @@ "DATE_DIFF(unit, date1, date2)", "TIMESTAMPDIFF(unit, date1, date2)" ], - "description": "Returns the difference between two dates in the specified unit, as date2 - date1: the first argument is the START and the second is the END.", + "description": "Returns the difference between two dates in the specified unit. With the dates first it is date1 - date2, as in BigQuery's DATE_DIFF(end_date, start_date, unit); with the unit first it is date2 - date1. DATE_DIFF counts the calendar boundaries crossed; TIMESTAMPDIFF counts the whole units elapsed, as MySQL does.", "parameters": [ { "name": "date1", "type": "DATE/TIMESTAMP", - "description": "First date (start)", + "description": "First date: the end when the dates come first, the start when the unit comes first", "optional": false, "defaultValue": null }, { "name": "date2", "type": "DATE/TIMESTAMP", - "description": "Second date (end)", + "description": "Second date: the start when the dates come first, the end when the unit comes first", "optional": false, "defaultValue": null }, { "name": "unit", "type": "KEYWORD", - "description": "Unit for result (YEAR, MONTH, WEEK, DAY, HOUR, MINUTE, SECOND)", + "description": "Unit for result (YEAR, QUARTER, MONTH, WEEK, DAY, HOUR, MINUTE, SECOND)", "optional": true, "defaultValue": "DAY" } @@ -34,33 +34,39 @@ "returnType": "BIGINT", "examples": [ { - "title": "Calculate age", - "description": "Years between birthdate and now", - "sql": "SELECT DATE_DIFF(birthdate, CURRENT_DATE, YEAR) as age FROM users" + "title": "BigQuery's own documented example", + "description": "The end date first: returns 559", + "sql": "SELECT DATE_DIFF('2010-07-07'::DATE, '2008-12-25'::DATE, DAY) as days FROM orders" }, { "title": "Days since event", - "description": "Calculate days elapsed", - "sql": "SELECT DATE_DIFF(event_date, CURRENT_DATE, DAY) as days_ago FROM events" + "description": "Calendar days from the event to today, positive for a past event", + "sql": "SELECT DATE_DIFF(CURRENT_DATE, event_date, DAY) as days_ago FROM events" }, { "title": "Hours between timestamps", - "description": "Calculate duration", - "sql": "SELECT DATE_DIFF(start_time, end_time, HOUR) as duration_hours FROM sessions" + "description": "The hour boundaries crossed from start_time to end_time", + "sql": "SELECT DATE_DIFF(HOUR, start_time, end_time) as duration_hours FROM sessions" + }, + { + "title": "Calculate age", + "description": "The whole years elapsed since birthdate, which TIMESTAMPDIFF counts", + "sql": "SELECT TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE) as age FROM users" } ], "notes": [ - "Returns a negative value if date2 is before date1", - "Truncates to whole units (no fractions)", - "DATEDIFF is NOT an alias of this function. It is MySQL's, and subtracts the other way round (expr1 - expr2). See HELP DATEDIFF.", - "TIMESTAMPDIFF is the ODBC/JDBC spelling and takes the unit first; MySQL defines it as dt2 - dt1, which is what this function already computes." + "Every vendor computes end - start, and the layout decides where the end sits: the dates first is date1 - date2, the unit first is date2 - date1.", + "DATE_DIFF and DATEDIFF count the calendar boundaries crossed: one second across a year end is 1 YEAR, and 1992-09-15 to 1992-11-14 is 2 MONTHs.", + "TIMESTAMPDIFF counts the whole units elapsed, truncated toward zero: 1992-09-15 to 1992-11-14 is 1 MONTH. A DATE is the datetime at 00:00:00.", + "Boundaries are in UTC. A WEEK starts on Monday (ISO 8601); a QUARTER on January 1, April 1, July 1 and October 1.", + "DATEDIFF follows the same rules, and also takes MySQL's two-argument DATEDIFF(date1, date2), in days. See HELP DATEDIFF.", + "TIMESTAMPDIFF is not an alias of DATE_DIFF: it takes the same layouts but counts the whole units elapsed, where DATE_DIFF counts the boundaries crossed. TIMESTAMPDIFF(YEAR, '2025-12-31 23:59:59', '2026-01-01 00:00:00') is 0; DATE_DIFF(YEAR, '2025-12-31 23:59:59', '2026-01-01 00:00:00') is 1.", + "Changed in 0.24.0: the dates-first forms returned date2 - date1, and the units from WEEK up counted the whole units elapsed between two calendar dates. HOUR, MINUTE and SECOND count the boundaries crossed too: DATE_DIFF(HOUR, '2025-01-10 10:59:00', '2025-01-10 11:00:00') is 1." ], "seeAlso": [ "DATEDIFF", "DATE_ADD", "DATE_SUB" ], - "aliases": [ - "TIMESTAMPDIFF" - ] + "aliases": [] } diff --git a/core/src/main/resources/help/functions/date/datediff.json b/core/src/main/resources/help/functions/date/datediff.json index 965348d52..60bcad8c0 100644 --- a/core/src/main/resources/help/functions/date/datediff.json +++ b/core/src/main/resources/help/functions/date/datediff.json @@ -1,32 +1,32 @@ { "name": "DATEDIFF", "category": "Date/Time", - "shortDescription": "MySQL day difference: expr1 - expr2", + "shortDescription": "Difference between dates: date1 - date2 with the dates first", "syntax": [ "DATEDIFF(date1, date2)", "DATEDIFF(date1, date2, unit)", "DATEDIFF(unit, date1, date2)" ], - "description": "MySQL's day-difference function. DATEDIFF(date1, date2) returns date1 - date2 in DAYS, which is the opposite subtraction from DATE_DIFF. They are different functions with similar names.", + "description": "DATEDIFF(date1, date2) is MySQL's: date1 - date2 in DAYS. With a unit, the dates first is still date1 - date2, and the unit first - the spelling of SQL Server, Snowflake, Redshift, DuckDB and Elasticsearch SQL - is date2 - date1. It counts the calendar boundaries crossed, as DATE_DIFF does.", "parameters": [ { "name": "date1", "type": "DATE/TIMESTAMP", - "description": "Date subtracted FROM", + "description": "The date subtracted FROM when the dates come first; the start when the unit comes first", "optional": false, "defaultValue": null }, { "name": "date2", "type": "DATE/TIMESTAMP", - "description": "Date subtracted", + "description": "The date subtracted when the dates come first; the end when the unit comes first", "optional": false, "defaultValue": null }, { "name": "unit", "type": "KEYWORD", - "description": "Unit for result. NOT MySQL - this engine's own extension, and it reverses the subtraction to date2 - date1", + "description": "Unit for result (YEAR, QUARTER, MONTH, WEEK, DAY, HOUR, MINUTE, SECOND)", "optional": true, "defaultValue": "DAY" } @@ -44,16 +44,15 @@ "sql": "SELECT DATEDIFF(due_date, CURRENT_DATE) as days_left FROM tasks" }, { - "title": "The three-argument form reverses it", - "description": "This form is not MySQL and follows DATE_DIFF instead", - "sql": "SELECT DATEDIFF(start_date, end_date, MONTH) as months FROM projects" + "title": "The unit first", + "description": "The month boundaries crossed from start_date to end_date, as SQL Server counts them", + "sql": "SELECT DATEDIFF(MONTH, start_date, end_date) as months FROM projects" } ], "notes": [ - "WARNING: the two-argument and three-argument forms subtract in OPPOSITE directions. DATEDIFF(a, b) is a - b (MySQL, days only); DATEDIFF(a, b, unit) is b - a (this engine's own extension). Adding a unit reverses the sign.", - "If you want a unit, prefer DATE_DIFF(start, end, unit) or TIMESTAMPDIFF(unit, start, end), where one rule holds at every arity.", - "DATEDIFF(unit, date1, date2) - the unit FIRST - is T-SQL's and DuckDB's spelling and means date2 - date1, like TIMESTAMPDIFF. Only the two-argument form is MySQL's.", - "Before engine 0.24.0 this name was an alias of DATE_DIFF and returned the opposite sign. Statements written against the old behaviour need their arguments swapped." + "The dates first is date1 - date2, with or without a unit; the unit first is date2 - date1.", + "DATEDIFF counts the calendar boundaries crossed, in UTC, a WEEK starting on Monday: DATEDIFF(YEAR, '2025-12-31 23:59:59', '2026-01-01 00:00:00') is 1. For the whole units elapsed, use TIMESTAMPDIFF(unit, date1, date2).", + "Changed in 0.24.0: this name was an alias of DATE_DIFF; every dates-first form returned date2 - date1, and from WEEK up it counted the whole units elapsed between the two calendar dates. Swapping the dates of a statement written against the old behaviour restores its sign, not its count." ], "seeAlso": [ "DATE_DIFF", diff --git a/documentation/sql/ddl_statements.md b/documentation/sql/ddl_statements.md index 852c189c5..3c032b3eb 100644 --- a/documentation/sql/ddl_statements.md +++ b/documentation/sql/ddl_statements.md @@ -176,7 +176,7 @@ CREATE TABLE users ( zip VARCHAR ), join_date DATE, - seniority INT SCRIPT AS (DATEDIFF(profile.join_date, CURRENT_DATE, DAY)) + seniority INT SCRIPT AS (DATEDIFF(CURRENT_DATE, profile.join_date, DAY)) ) ) ``` @@ -304,7 +304,7 @@ CREATE TABLE users ( id INT, name VARCHAR DEFAULT 'anonymous', birthdate DATE, - age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR)), + age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)), PRIMARY KEY (id) ); ``` @@ -399,7 +399,7 @@ value derived for it. CREATE TABLE users ( id INT, birthdate DATE, - age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR)) + age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)) ); ``` @@ -410,7 +410,7 @@ computing it here: CREATE TABLE users_view ( id INT, birthdate DATE, - age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR)) STORED + age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)) STORED ); ``` @@ -705,7 +705,7 @@ WITH PROCESSORS ( value = "anonymous" ), SCRIPT ( - description = "age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR))", + description = "age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE))", lang = "painless", source = "...", ignore_failure = true diff --git a/documentation/sql/dql_statements.md b/documentation/sql/dql_statements.md index 85fe3cd07..1e5db4dc3 100644 --- a/documentation/sql/dql_statements.md +++ b/documentation/sql/dql_statements.md @@ -1297,7 +1297,7 @@ FROM dql_users; | `DATE_SUB(date, INTERVAL n unit)` | Subtract interval | | `DATETIME_ADD(ts, INTERVAL n unit)` | Add interval to timestamp | | `DATETIME_SUB(ts, INTERVAL n unit)` | Subtract interval from timestamp | -| `DATE_DIFF(date1, date2, unit)` | Difference in units: elapsed `HOUR` / `MINUTE` / `SECOND`, calendar dates from `DAY` up (UTC) | +| `DATE_DIFF(date1, date2, unit)` | `date1 - date2` in units: the calendar boundaries crossed (UTC, weeks start on Monday) | | `DATE_TRUNC(date, unit)` | Truncate to unit | ##### **Formatting & parsing:** @@ -1327,7 +1327,7 @@ SELECT id, MONTH(CURRENT_DATE) AS current_month, DAY(CURRENT_DATE) AS current_day, YEAR(birthdate) AS year_b, - DATE_DIFF(birthdate, CURRENT_DATE, YEAR) AS diff_years, + DATE_DIFF(CURRENT_DATE, birthdate, YEAR) AS diff_years, DATE_TRUNC(birthdate, MONTH) AS trunc_month, DATETIME_FORMAT(birthdate, '%Y-%m-%d') AS birth_str FROM dql_users; @@ -1561,13 +1561,13 @@ CREATE TABLE IF NOT EXISTS users ( id INT NOT NULL COMMENT 'user identifier', name VARCHAR FIELDS(raw Keyword COMMENT 'sortable') DEFAULT 'anonymous' OPTIONS (analyzer = 'french', search_analyzer = 'french'), birthdate DATE, - age INT SCRIPT AS (DATEDIFF(birthdate, CURRENT_DATE, YEAR)), + age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)), ingested_at TIMESTAMP DEFAULT _ingest.timestamp, profile STRUCT FIELDS( bio VARCHAR, followers INT, join_date DATE, - seniority INT SCRIPT AS (DATEDIFF(profile.join_date, CURRENT_DATE, DAY)) + seniority INT SCRIPT AS (DATEDIFF(CURRENT_DATE, profile.join_date, DAY)) ) COMMENT 'user profile', PRIMARY KEY (id) ) PARTITION BY birthdate (MONTH), OPTIONS (mappings = (dynamic = false)); @@ -1579,14 +1579,14 @@ SHOW TABLE users; | Field | Type | Null | Key | Default | Comment | Script | Extra | |-------------------|-----------|------|-----|-------------------|-----------------|-------------------------------------------------|---------------------------------------------------| -| age | INT | yes | | NULL | | DATE_DIFF(birthdate, CURRENT_DATE, YEAR) | () | +| age | INT | yes | | NULL | | TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE) | () | | birthdate | DATE | yes | | NULL | | | () | | id | INT | no | PRI | NULL | user identifier | | () | | ingested_at | TIMESTAMP | yes | | _ingest.timestamp | | | () | | name | VARCHAR | yes | | anonymous | | | (analyzer = "french", search_analyzer = "french") | | name.raw | KEYWORD | yes | | NULL | sortable | | () | | profile | STRUCT | yes | | NULL | user profile | | () | -| profile.seniority | INT | yes | | NULL | | DATE_DIFF(profile.join_date, CURRENT_DATE, DAY) | () | +| profile.seniority | INT | yes | | NULL | | DATE_DIFF(CURRENT_DATE, profile.join_date, DAY) | () | | profile.join_date | DATE | yes | | NULL | | | () | | profile.followers | INT | yes | | NULL | | | () | | profile.bio | VARCHAR | yes | | NULL | | | () | @@ -1606,7 +1606,7 @@ _meta: (primary_key = ('id'), partition_by = (column = 'birthdate', granularity 📝 DDL: ```sql CREATE OR REPLACE TABLE users ( - age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR)), + age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)), birthdate DATE, id INT NOT NULL COMMENT 'user identifier', ingested_at TIMESTAMP DEFAULT _ingest.timestamp, @@ -1614,7 +1614,7 @@ CREATE OR REPLACE TABLE users ( raw KEYWORD COMMENT 'sortable' ) DEFAULT 'anonymous' OPTIONS (analyzer = "french", search_analyzer = "french"), profile STRUCT FIELDS ( - seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY)), + seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY)), join_date DATE, followers INT, bio VARCHAR @@ -1640,7 +1640,7 @@ SHOW CREATE TABLE users; ```sql CREATE OR REPLACE TABLE users ( - age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR)), + age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)), birthdate DATE, id INT NOT NULL COMMENT 'user identifier', ingested_at TIMESTAMP DEFAULT _ingest.timestamp, @@ -1648,7 +1648,7 @@ CREATE OR REPLACE TABLE users ( raw KEYWORD COMMENT 'sortable' ) DEFAULT 'anonymous' OPTIONS (analyzer = "french", search_analyzer = "french"), profile STRUCT FIELDS ( - seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY)), + seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY)), join_date DATE, followers INT, bio VARCHAR @@ -1682,14 +1682,14 @@ Returns the **normalized SQL schema**, including : | Field | Type | Null | Key | Default | Comment | Script | Extra | |-------------------|-----------|------|-----|-------------------|-----------------|-------------------------------------------------|---------------------------------------------------| -| age | INT | yes | | NULL | | DATE_DIFF(birthdate, CURRENT_DATE, YEAR) | () | +| age | INT | yes | | NULL | | TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE) | () | | birthdate | DATE | yes | | NULL | | | () | | id | INT | no | PRI | NULL | user identifier | | () | | ingested_at | TIMESTAMP | yes | | _ingest.timestamp | | | () | | name | VARCHAR | yes | | anonymous | | | (analyzer = "french", search_analyzer = "french") | | name.raw | KEYWORD | yes | | NULL | sortable | | () | | profile | STRUCT | yes | | NULL | user profile | | () | -| profile.seniority | INT | yes | | NULL | | DATE_DIFF(profile.join_date, CURRENT_DATE, DAY) | () | +| profile.seniority | INT | yes | | NULL | | DATE_DIFF(CURRENT_DATE, profile.join_date, DAY) | () | | profile.join_date | DATE | yes | | NULL | | | () | | profile.followers | INT | yes | | NULL | | | () | | profile.bio | VARCHAR | yes | | NULL | | | () | @@ -1787,9 +1787,9 @@ Processors: (6) | processor_type | description | field | ignore_failure | options | |-----------------|-----------------------------------------------------------------------------------|-------------------|----------------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| | set | DEFAULT 'anonymous' | name | yes | (value = "anonymous", if = "ctx.name == null") | -| script | age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR)) | age | yes | (lang = "painless", source = "def param1 = ctx.birthdate; def param2 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(ctx['_ingest']['timestamp']), ZoneId.of('Z')).toLocalDate(); ctx.age = (param1 == null) ? null : Long.valueOf(ChronoUnit.YEARS.between(param1, param2))") | +| script | age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)) | age | yes | (lang = "painless", source = "def param1 = ctx.birthdate; def param2 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(ctx['_ingest']['timestamp']), ZoneId.of('Z')).toLocalDate(); ctx.age = (param1 == null) ? null : Long.valueOf(ChronoUnit.YEARS.between(param1, param2))") | | set | DEFAULT _ingest.timestamp | ingested_at | yes | (value = "_ingest.timestamp", if = "ctx.ingested_at == null") | -| script | profile.seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY)) | profile.seniority | yes | (lang = "painless", source = "def param1 = ctx.profile?.join_date; def param2 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(ctx['_ingest']['timestamp']), ZoneId.of('Z')).toLocalDate(); ctx.profile.seniority = (param1 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param1, param2))") | +| script | profile.seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY)) | profile.seniority | yes | (lang = "painless", source = "def param1 = ctx.profile?.join_date; def param2 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(ctx['_ingest']['timestamp']), ZoneId.of('Z')).toLocalDate(); ctx.profile.seniority = (param1 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param1, param2))") | | date_index_name | PARTITION BY birthdate (MONTH) | birthdate | yes | (date_rounding = "M", date_formats = ["yyyy-MM"], index_name_prefix = "users-") | | set | PRIMARY KEY (id) | _id | no | (value = "{{id}}", ignore_empty_value = false) | @@ -1804,7 +1804,7 @@ CREATE OR REPLACE PIPELINE user_pipeline WITH PROCESSORS ( if = "ctx.name == null" ), SCRIPT( - description = "age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR))", + description = "age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE))", lang = "painless", source = "...", ignore_failure = true @@ -1817,7 +1817,7 @@ CREATE OR REPLACE PIPELINE user_pipeline WITH PROCESSORS ( if = "ctx.ingested_at == null" ), SCRIPT( - description = "profile.seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY))", + description = "profile.seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY))", lang = "painless", source = "...", ignore_failure = true @@ -1868,7 +1868,7 @@ CREATE OR REPLACE PIPELINE user_pipeline WITH PROCESSORS ( if = "ctx.name == null" ), SCRIPT( - description = "age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR))", + description = "age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE))", lang = "painless", source = "...", ignore_failure = true @@ -1881,7 +1881,7 @@ CREATE OR REPLACE PIPELINE user_pipeline WITH PROCESSORS ( if = "ctx.ingested_at == null" ), SCRIPT( - description = "profile.seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY))", + description = "profile.seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY))", lang = "painless", source = "...", ignore_failure = true @@ -1928,9 +1928,9 @@ DESCRIBE PIPELINE user_pipeline; | processor_type | description | field | ignore_failure | options | |-----------------|-----------------------------------------------------------------------------------|-------------------|----------------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| | set | DEFAULT 'anonymous' | name | yes | (value = "anonymous", if = "ctx.name == null") | -| script | age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR)) | age | yes | (lang = "painless", source = "def param1 = ctx.birthdate; def param2 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(ctx['_ingest']['timestamp']), ZoneId.of('Z')).toLocalDate(); ctx.age = (param1 == null) ? null : Long.valueOf(ChronoUnit.YEARS.between(param1, param2))") | +| script | age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)) | age | yes | (lang = "painless", source = "def param1 = ctx.birthdate; def param2 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(ctx['_ingest']['timestamp']), ZoneId.of('Z')).toLocalDate(); ctx.age = (param1 == null) ? null : Long.valueOf(ChronoUnit.YEARS.between(param1, param2))") | | set | DEFAULT _ingest.timestamp | ingested_at | yes | (value = "_ingest.timestamp", if = "ctx.ingested_at == null") | -| script | profile.seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY)) | profile.seniority | yes | (lang = "painless", source = "def param1 = ctx.profile?.join_date; def param2 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(ctx['_ingest']['timestamp']), ZoneId.of('Z')).toLocalDate(); ctx.profile.seniority = (param1 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param1, param2))") | +| script | profile.seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY)) | profile.seniority | yes | (lang = "painless", source = "def param1 = ctx.profile?.join_date; def param2 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(ctx['_ingest']['timestamp']), ZoneId.of('Z')).toLocalDate(); ctx.profile.seniority = (param1 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param1, param2))") | | date_index_name | PARTITION BY birthdate (MONTH) | birthdate | yes | (date_rounding = "M", date_formats = ["yyyy-MM"], index_name_prefix = "users-") | | set | PRIMARY KEY (id) | _id | no | (value = "{{id}}", ignore_empty_value = false) | 📊 6 row(s) (1ms) diff --git a/documentation/sql/functions_date_time.md b/documentation/sql/functions_date_time.md index b8e312834..baecc694f 100644 --- a/documentation/sql/functions_date_time.md +++ b/documentation/sql/functions_date_time.md @@ -316,70 +316,158 @@ SELECT DATETIME_SUB('2025-01-10T12:00:00Z'::TIMESTAMP, INTERVAL 1 MONTH) AS last ### Date/Time Difference Functions -#### DATEDIFF / DATE_DIFF +#### DATEDIFF / DATE_DIFF / TIMESTAMPDIFF -Difference between 2 dates (date2 - date1) in the specified time unit: `date1` is the start and `date2` the end. -MySQL's two-argument `DATEDIFF(a, b)` gives `a - b`. +Difference between 2 dates in a time unit. These functions follow the definitions of the databases they come from: the **layout** of the arguments decides which date is subtracted from which, and the **name** decides what is counted. **Syntax:** ```sql -DATEDIFF(date1, date2) +-- the dates first: date1 - date2 +DATEDIFF(date1, date2) -- MySQL: in days DATEDIFF(date1, date2, unit) -DATE_DIFF(date1, date2) +DATE_DIFF(date1, date2) -- BigQuery: DATE_DIFF(end_date, start_date, unit) DATE_DIFF(date1, date2, unit) + +-- the unit first: date2 - date1 +DATEDIFF(unit, date1, date2) -- SQL Server, Snowflake, Redshift, DuckDB, Elasticsearch SQL +DATE_DIFF(unit, date1, date2) +TIMESTAMPDIFF(unit, date1, date2) -- MySQL ``` **Inputs:** -- `date1` - `DATE` or `DATETIME` -- `date2` - `DATE` or `DATETIME` -- `unit` (optional) - One of: `YEAR`, `QUARTER`, `MONTH`, `WEEK`, `DAY`, `HOUR`, `MINUTE`, `SECOND` - - Default: `DAY` +- `date1` - `DATE`, `DATETIME` or `TIMESTAMP` +- `date2` - `DATE`, `DATETIME` or `TIMESTAMP` +- `unit` - One of: `YEAR`, `QUARTER`, `MONTH`, `WEEK`, `DAY`, `HOUR`, `MINUTE`, `SECOND` (or their ODBC names, `SQL_TSI_YEAR` …) + - Default, when the dates come first: `DAY` **Output:** - `BIGINT` -**Units:** -- `HOUR`, `MINUTE`, `SECOND` count the elapsed whole units between the two instants, in UTC, truncated toward zero. A `DATE` operand counts from the start of its day (00:00 UTC). -- `DAY`, `WEEK`, `MONTH`, `QUARTER`, `YEAR` compare the two calendar dates, in UTC, whatever the time of day: `2025-01-10T23:30:00Z` and `2025-01-11T00:30:00Z` are 1 day apart, as MySQL's `DATEDIFF` counts them. A week is 7 whole days, a month counts once its day of month is reached, a quarter is 3 whole months and a year 12; every count is truncated toward zero. +**Direction, by layout:** every one of these databases computes *end − start*; the layout decides where the end sits. +- The dates first: `date1 - date2`. `DATEDIFF('2025-01-10', '2025-01-01')` is `9`, as in MySQL; `DATE_DIFF('2010-07-07', '2008-12-25', DAY)` is `559`, as in BigQuery. +- The unit first: `date2 - date1`. `DATEDIFF(DAY, '2025-01-01', '2025-01-10')` is `9`. + +**Counting, by name:** +- `DATEDIFF` and `DATE_DIFF` count the calendar **boundaries crossed**, as SQL Server, Snowflake, Redshift, BigQuery and DuckDB do: one second across a year end is 1 `YEAR`, `1992-09-15` to `1992-11-14` is 2 `MONTH`s, `10:59` to `11:00` is 1 `HOUR`. `DAY` counts calendar days: `2025-01-10T23:30:00Z` and `2025-01-11T00:30:00Z` are 1 day apart. +- `TIMESTAMPDIFF` counts the **whole units elapsed**, truncated toward zero, as MySQL does: a month counts once the same day of month and time of day are reached, so `2003-02-01` to `2003-05-01` is 3 `MONTH`s and `1992-09-15` to `1992-11-14` is 1. A `DATE` operand is the datetime at `00:00:00`. + +**Calendar:** +- Every boundary is in UTC. +- A `WEEK` starts on **Monday** (ISO 8601): the `WEEK` boundaries are the Mondays crossed, so a Saturday and the next Sunday are 0 weeks apart. A `QUARTER` starts on January 1, April 1, July 1 and October 1. **Literals:** - A string literal is read as the temporal it spells: `'2025-01-10'` (or `'2025/01/10'`) is a `DATE`; a literal with a time of day (`'2025-01-10 14:00:00'`, `'2025-01-10T14:00:00Z'`) is a `TIMESTAMP`, in UTC unless it names a zone. -- This holds in every clause, for each row and for each group: `DATEDIFF(MAX(created_at), '2025-01-10 08:00:00', HOUR)`. +- This holds in every clause, for each row and for each group, and in a computed column (`SCRIPT AS`): `DATEDIFF(MAX(created_at), '2025-01-10 08:00:00', HOUR)`. - A `NULL` operand, or an aggregate over a group that has no value, gives `NULL`. +> 🔴 **Changed in 0.24.0 — these functions follow their vendors' definitions.** Statements written +> against the old behaviour return other values, without an error: +> - **The dates first subtract the second date from the first.** `DATEDIFF(a, b)`, +> `DATEDIFF(a, b, unit)`, `DATE_DIFF(a, b)` and `DATE_DIFF(a, b, unit)` returned `b - a`; they +> return `a - b`, as MySQL's `DATEDIFF` and BigQuery's `DATE_DIFF` do. The unit-first forms +> (`DATEDIFF(unit, a, b)`, `DATE_DIFF(unit, a, b)`) keep `b - a`. +> - **`DATEDIFF` and `DATE_DIFF` count the boundaries crossed.** From `WEEK` up they counted the +> whole units elapsed between the two calendar dates (a week was 7 days, a month counted once its +> day of month was reached): `DATE_DIFF(MONTH, '1992-09-15', '1992-11-14')` was `1` and is `2`, and +> a `WEEK` boundary is now a Monday. `HOUR`, `MINUTE` and `SECOND` count the boundaries crossed too: +> `DATEDIFF(HOUR, '2025-01-10 10:59:00', '2025-01-10 11:00:00')` is `1`. `DAY` still counts +> calendar days. +> - **A computed column answers what a query answers.** In a `SCRIPT AS` column these functions +> counted the whole units elapsed between two instants whatever the unit (a `DAY` was 24 hours), and +> a string literal operand made the `CREATE TABLE` fail. +> +> New in 0.24.0, where they failed or were refused before: `TIMESTAMPDIFF`, the `QUARTER` unit, +> `HOUR` / `MINUTE` / `SECOND` over a column or an aggregate, and a string literal operand in a query. +> +> **What was deployed before 0.24.0 keeps its 0.23 results until it is re-created.** A computed +> column (`SCRIPT AS`), an ingest pipeline and a materialized view keep the script they were +> deployed with: adding a column with `ALTER TABLE`, `SHOW CREATE` and re-running the pipeline text +> that `SHOW CREATE PIPELINE` returns leave it as it was. +> - Re-creating it from the same SQL adopts the new definition, and these shapes then return other +> values: `DATEDIFF(a, b[, unit])` and `DATE_DIFF(a, b[, unit])` (the sign, and the count), +> `DATE_DIFF(unit, a, b)` (the count), `DATE_TRUNC(x, WEEK)` (the Monday), and `DAY` in a computed +> column (calendar days). +> - `CREATE OR REPLACE MATERIALIZED VIEW` with an unchanged text keeps the old results: to adopt the +> new definition, `DROP` the view and `CREATE` it. +> - `SHOW CREATE` shows the stored text, and that text now means the new definition. +> - To keep a 0.23 value, swap the dates: `DATE_DIFF(a, b, unit)` becomes `DATE_DIFF(b, a, unit)`, +> which restores the sign. Where it counted whole units elapsed, use `TIMESTAMPDIFF`: +> `TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)` is the age that +> `DATE_DIFF(birthdate, CURRENT_DATE, YEAR)` returned. + **Examples:** ```sql --- Difference in days (default), MySQL's two-argument form: date1 - date2 +-- MySQL's two-argument form, in days: date1 - date2 SELECT DATEDIFF('2025-01-10'::DATE, '2025-01-01'::DATE) AS diff; -- Result: 9 --- Difference in days (explicit) -SELECT DATEDIFF('2025-01-01'::DATE, '2025-01-10'::DATE, DAY) AS diff_days; +-- The dates first, with a unit: still date1 - date2 +SELECT DATEDIFF('2025-01-10'::DATE, '2025-01-01'::DATE, DAY) AS diff_days; -- Result: 9 --- Difference in weeks -SELECT DATE_DIFF('2025-01-01'::DATE, '2025-01-31'::DATE, WEEK) AS diff_weeks; +-- BigQuery's own example: the end date first +SELECT DATE_DIFF('2010-07-07'::DATE, '2008-12-25'::DATE, DAY) AS diff_days; +-- Result: 559 + +-- The unit first: date2 - date1 +SELECT DATEDIFF(DAY, '2025-01-01'::DATE, '2025-01-10'::DATE) AS diff_days; +-- Result: 9 + +-- Difference in weeks: the Mondays crossed (January 6, 13, 20 and 27) +SELECT DATE_DIFF('2025-01-31'::DATE, '2025-01-01'::DATE, WEEK) AS diff_weeks; -- Result: 4 +-- A week starts on Monday: Saturday to Sunday crosses none, Sunday to Monday one +SELECT DATEDIFF(WEEK, '2025-01-11'::DATE, '2025-01-12'::DATE) AS sat_to_sun, + DATEDIFF(WEEK, '2025-01-12'::DATE, '2025-01-13'::DATE) AS sun_to_mon; +-- Result: 0, 1 + -- Difference in months -SELECT DATEDIFF('2025-01-01'::DATE, '2025-06-01'::DATE, MONTH) AS diff_months; +SELECT DATEDIFF('2025-06-01'::DATE, '2025-01-01'::DATE, MONTH) AS diff_months; -- Result: 5 +-- Two month boundaries are crossed, one whole month elapses +SELECT DATE_DIFF(MONTH, '1992-09-15'::DATE, '1992-11-14'::DATE) AS boundaries, + TIMESTAMPDIFF(MONTH, '1992-09-15'::DATE, '1992-11-14'::DATE) AS elapsed; +-- Result: 2, 1 + +-- MySQL's own example +SELECT TIMESTAMPDIFF(MONTH, '2003-02-01', '2003-05-01') AS diff_months; +-- Result: 3 + -- Difference in years -SELECT DATEDIFF('2025-01-01'::DATE, '2027-01-01'::DATE, YEAR) AS diff_years; +SELECT DATEDIFF('2027-01-01'::DATE, '2025-01-01'::DATE, YEAR) AS diff_years; -- Result: 2 +-- One second across a year end +SELECT DATEDIFF(YEAR, '2025-12-31 23:59:59', '2026-01-01 00:00:00') AS boundaries, + TIMESTAMPDIFF(YEAR, '2025-12-31 23:59:59', '2026-01-01 00:00:00') AS elapsed; +-- Result: 1, 0 + -- Difference in hours (with timestamps) -SELECT DATEDIFF('2025-01-10T12:00:00Z'::TIMESTAMP, '2025-01-10T14:00:00Z'::TIMESTAMP, HOUR) AS diff_hours; +SELECT DATEDIFF('2025-01-10T14:00:00Z'::TIMESTAMP, '2025-01-10T12:00:00Z'::TIMESTAMP, HOUR) AS diff_hours; -- Result: 2 +-- One minute across an hour +SELECT DATEDIFF(HOUR, '2025-01-10 10:59:00', '2025-01-10 11:00:00') AS boundaries, + TIMESTAMPDIFF(HOUR, '2025-01-10 10:59:00', '2025-01-10 11:00:00') AS elapsed; +-- Result: 1, 0 + +-- One hour across midnight +SELECT DATEDIFF(DAY, '2025-01-10 23:30:00', '2025-01-11 00:30:00') AS boundaries, + TIMESTAMPDIFF(DAY, '2025-01-10 23:30:00', '2025-01-11 00:30:00') AS elapsed; +-- Result: 1, 0 + -- Difference in minutes -SELECT DATEDIFF('2025-01-10T12:00:00Z'::TIMESTAMP, '2025-01-10T12:30:00Z'::TIMESTAMP, MINUTE) AS diff_minutes; +SELECT DATEDIFF('2025-01-10T12:30:00Z'::TIMESTAMP, '2025-01-10T12:00:00Z'::TIMESTAMP, MINUTE) AS diff_minutes; -- Result: 30 -- Difference in seconds -SELECT DATEDIFF('2025-01-10T12:00:00Z'::TIMESTAMP, '2025-01-10T12:00:45Z'::TIMESTAMP, SECOND) AS diff_seconds; +SELECT DATEDIFF('2025-01-10T12:00:45Z'::TIMESTAMP, '2025-01-10T12:00:00Z'::TIMESTAMP, SECOND) AS diff_seconds; -- Result: 45 + +-- An age in whole years: the years elapsed, not the year boundaries crossed +SELECT TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE) AS age FROM users; ``` --- @@ -618,6 +706,14 @@ DATE_TRUNC(date_or_datetime_expr, unit) **Output:** - `DATE` or `DATETIME` (same type as input) +**Week:** +- `WEEK` truncates to the **Monday** that starts the ISO-8601 week, at 00:00 UTC, in every clause: a `GROUP BY DATE_TRUNC(…, WEEK)` buckets on that same Monday. + +> 🔴 **Changed in 0.24.0 — `DATE_TRUNC(…, WEEK)` is the Monday that starts the week.** Before 0.24.0, a +> `SELECT` item, a `WHERE` condition or a computed column (`SCRIPT AS`) moved a date FORWARD to the +> Sunday that ends its week (`2025-01-15` gave `2025-01-19`), while a `GROUP BY` of the same +> expression bucketed on the Monday. + **Examples:** ```sql -- Truncate to start of month diff --git a/documentation/sql/functions_math.md b/documentation/sql/functions_math.md index 2964cc482..0181f9eb4 100644 --- a/documentation/sql/functions_math.md +++ b/documentation/sql/functions_math.md @@ -206,7 +206,7 @@ SELECT FLOOR(123.999) AS f; SELECT user_id, name, - FLOOR(DATEDIFF(birth_date, CURRENT_DATE, DAY) / 365.25) AS age + FLOOR(DATEDIFF(CURRENT_DATE, birth_date, DAY) / 365.25) AS age FROM users; -- Bucket values diff --git a/es6/bridge/src/test/scala/app/softnetwork/elastic/sql/SQLQuerySpec.scala b/es6/bridge/src/test/scala/app/softnetwork/elastic/sql/SQLQuerySpec.scala index ab09f06a3..6f31f8661 100644 --- a/es6/bridge/src/test/scala/app/softnetwork/elastic/sql/SQLQuerySpec.scala +++ b/es6/bridge/src/test/scala/app/softnetwork/elastic/sql/SQLQuerySpec.scala @@ -1421,13 +1421,13 @@ class SQLQuerySpec extends AnyFlatSpec with Matchers { | "w": { | "script": { | "lang": "painless", - | "source": "def param1 = (doc['lastUpdated'].size() == 0 ? null : doc['lastUpdated'].value.toInstant().atZone(ZoneId.of('Z')).with(DayOfWeek.SUNDAY).truncatedTo(ChronoUnit.DAYS)); def param2 = DateTimeFormatter.ofPattern(\"yyyy-MM-dd\"); (param1 == null) ? null : param2.format(param1)" + | "source": "def param1 = (doc['lastUpdated'].size() == 0 ? null : doc['lastUpdated'].value.toInstant().atZone(ZoneId.of('Z')).with(DayOfWeek.MONDAY).truncatedTo(ChronoUnit.DAYS)); def param2 = DateTimeFormatter.ofPattern(\"yyyy-MM-dd\"); (param1 == null) ? null : param2.format(param1)" | } | }, | "w2": { | "script": { | "lang": "painless", - | "source": "def param1 = (doc['lastUpdated'].size() == 0 ? null : doc['lastUpdated'].value.toInstant().atZone(ZoneId.of('Z')).with(DayOfWeek.SUNDAY).truncatedTo(ChronoUnit.DAYS)); def param2 = DateTimeFormatter.ofPattern(\"yyyy-MM-dd\"); (param1 == null) ? null : param2.format(param1)" + | "source": "def param1 = (doc['lastUpdated'].size() == 0 ? null : doc['lastUpdated'].value.toInstant().atZone(ZoneId.of('Z')).with(DayOfWeek.MONDAY).truncatedTo(ChronoUnit.DAYS)); def param2 = DateTimeFormatter.ofPattern(\"yyyy-MM-dd\"); (param1 == null) ? null : param2.format(param1)" | } | }, | "d": { @@ -1655,7 +1655,7 @@ class SQLQuerySpec extends AnyFlatSpec with Matchers { | "diff": { | "script": { | "lang": "painless", - | "source": "def param1 = (doc['createdAt'].size() == 0 ? null : doc['createdAt'].value.toInstant().atZone(ZoneId.of('Z'))); def param2 = (doc['updatedAt'].size() == 0 ? null : doc['updatedAt'].value.toInstant().atZone(ZoneId.of('Z'))); (param1 == null || param2 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param1.toLocalDate(), param2.toLocalDate()))" + | "source": "def param1 = (doc['updatedAt'].size() == 0 ? null : doc['updatedAt'].value.toInstant().atZone(ZoneId.of('Z'))); def param2 = (doc['createdAt'].size() == 0 ? null : doc['createdAt'].value.toInstant().atZone(ZoneId.of('Z'))); (param1 == null || param2 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param1.toLocalDate(), param2.toLocalDate()))" | } | } | }, @@ -1712,7 +1712,7 @@ class SQLQuerySpec extends AnyFlatSpec with Matchers { | "max": { | "script": { | "lang": "painless", - | "source": "def param1 = (doc['createdAt'].size() == 0 ? null : doc['createdAt'].value); def param2 = (doc['updatedAt'].size() == 0 ? null : doc['updatedAt'].value.toInstant().atZone(ZoneId.of('Z'))); def param3 = (param1 == null) ? null : ZonedDateTime.parse(param1, new DateTimeFormatterBuilder().appendPattern(\"yyyy-MM-dd HH:mm:ss\").appendFraction(ChronoField.NANO_OF_SECOND, 0, 9, true).toFormatter().withZone(ZoneId.of('Z'))); (param1 == null || param2 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param3.toLocalDate(), param2.toLocalDate()))" + | "source": "def param1 = (doc['updatedAt'].size() == 0 ? null : doc['updatedAt'].value.toInstant().atZone(ZoneId.of('Z'))); def param2 = (doc['createdAt'].size() == 0 ? null : doc['createdAt'].value); def param3 = (param2 == null) ? null : ZonedDateTime.parse(param2, new DateTimeFormatterBuilder().appendPattern(\"yyyy-MM-dd HH:mm:ss\").appendFraction(ChronoField.NANO_OF_SECOND, 0, 9, true).toFormatter().withZone(ZoneId.of('Z'))); (param1 == null || param2 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param1.toLocalDate(), param3.toLocalDate()))" | } | } | } diff --git a/sql/src/main/scala/app/softnetwork/elastic/sql/function/time/package.scala b/sql/src/main/scala/app/softnetwork/elastic/sql/function/time/package.scala index cfc9e5687..c5a61641d 100644 --- a/sql/src/main/scala/app/softnetwork/elastic/sql/function/time/package.scala +++ b/sql/src/main/scala/app/softnetwork/elastic/sql/function/time/package.scala @@ -266,6 +266,18 @@ package object time { else Today.sql } + /** The start of the ISO-8601 week a temporal falls in: its MONDAY. It is what `DATE_TRUNC(…, + * WEEK)` truncates to and where `DATEDIFF` / `DATE_DIFF` count a WEEK boundary, and it is what + * Elasticsearch's calendar week (`date_histogram`, `/w` date math) rounds to, so every venue + * that truncates to a week agrees. + * + * 🔴 `with(DayOfWeek.SUNDAY)`, rendered before, is NOT a Sunday-start week: `DayOfWeek` adjusts + * within the Monday-to-Sunday week, so it moved a Wednesday FORWARD to the Sunday that ENDS its + * week (2025-01-08 -> 2025-01-12, JDK-verified), while a `GROUP BY` of the same expression went + * back to the Monday. + */ + private[sql] val isoWeekStart: String = ".with(DayOfWeek.MONDAY)" + case object DateTrunc extends Expr("DATE_TRUNC") with TokenRegex with PainlessScript { override def painless(context: Option[PainlessContext]): String = ".truncatedTo" override lazy val words: List[String] = List(sql, "DATETRUNC") @@ -341,7 +353,7 @@ package object time { unit match { case TimeUnit.YEARS => s".withDayOfYear(1)$truncateTime" case TimeUnit.MONTHS => s".withDayOfMonth(1)$truncateTime" - case TimeUnit.WEEKS => s".with(DayOfWeek.SUNDAY)$truncateTime" + case TimeUnit.WEEKS => s"$isoWeekStart$truncateTime" case TimeUnit.QUARTERS => context match { case Some(ctx) => @@ -546,6 +558,12 @@ package object time { override def painless(context: Option[PainlessContext]): String = ".between" override lazy val words: List[String] = List(sql, "TIMESTAMPDIFF") + /** MySQL's `TIMESTAMPDIFF`, this token's second word. It takes the layouts `DATE_DIFF` takes + * but COUNTS differently (whole units elapsed, see [[DateDiffSpelling]]), so the parser asks + * which word it read and the render writes this one back. + */ + lazy val timestampDiff: String = words(1) + /** A calendar date alone, in the two separators the DATE parse accepts (`SQLTypeUtils.coerce`). */ private val CalendarDate = """\d{4}[-/]\d{2}[-/]\d{2}""".r @@ -566,44 +584,124 @@ package object time { } case _ => None } + + /** The scale of [[monthKey]]: more milliseconds than any month holds (31 days: 2678400000). */ + private[time] val MonthKeyScale: Long = 4294967296L + + /** A temporal as one number whose difference with another, divided by [[MonthKeyScale]] and + * truncated toward zero, is MySQL's whole months elapsed between them (`TIMESTAMPDIFF`): the + * months between the two, less one when the later one's day of month and time of day come + * before the earlier one's. + * + * 🔴 Not `ChronoUnit.MONTHS.between`, which differs on one edge. It first moves an end whose + * time of day comes before the start's back by one DAY — on the 1st of a month, into the month + * before — and then counts that month short as well: `2024-12-31 23:59:59` to `2025-07-01 + * 00:00:00` is 5 months there and 6 in MySQL (found by the testkit's population). + */ + private[time] def monthKey(rendered: String): String = { + val t = if (postfixable(rendered)) rendered else s"($rendered)" + s"(($t.getYear() * 12L + $t.getMonthValue()) * ${MonthKeyScale}L + ($t.getDayOfMonth() - 1) * 86400000L + $t.getLong(ChronoField.MILLI_OF_DAY))" + } + + /** A temporal as the instant it denotes, in UTC, decided by its RUNTIME type: a `LocalDate` is + * the start of its day, a `LocalDateTime` that wall-clock time in UTC, anything zoned the same + * instant. `TIMESTAMPDIFF` compares instants, and an operand's declared type does not say + * which of the three it holds: `COALESCE(ts, CURRENT_DATE)` is either, row by row. + * + * `ref` is read up to four times, so it is a name or an expression cast to `def` (the methods + * below are resolved at runtime). NULL stays NULL: the MONTH, QUARTER and YEAR calls bind this + * value in the script's prologue, which runs before the null guard. + */ + private[time] def utcInstant(ref: String): String = + s"($ref == null ? null : $ref instanceof LocalDate ? $ref.atStartOfDay(ZoneId.of('Z')) : $ref instanceof LocalDateTime ? $ref.atZone(ZoneId.of('Z')) : $ref.withZoneSameInstant(ZoneId.of('Z')))" + + /** Whether a method can be appended to a rendered operand as it stands: outside parentheses, + * brackets and string literals it holds nothing but names, digits and dots. Anything else — a + * ternary (`p4 ? p3 : p1`, which is how a CASE operand renders), a cast (`(long) x`) — is + * parenthesised first, or the method would bind to its last term alone. + */ + private[time] def postfixable(rendered: String): Boolean = { + var depth = 0 + var quote: Char = 0 + var i = 0 + while (i < rendered.length) { + val c = rendered.charAt(i) + if (quote != 0) { + if (c == '\\') i += 1 + else if (c == quote) quote = 0 + } else if (c == '"' || c == '\'') quote = c + else if (c == '(' || c == '[') depth += 1 + else if (c == ')' || c == ']') depth -= 1 + else if (depth == 0 && !(c.isLetterOrDigit || c == '_' || c == '.')) return false + i += 1 + } + true + } } - /** MySQL's `DATEDIFF`, which is a DIFFERENT function from the one above and needs its own token - * so the parser can tell them apart (issue #363). - * - * MySQL 8.4 defines `DATEDIFF(expr1, expr2)` as `expr1 - expr2`; `DATE_DIFF(start, end, unit)` - * is BigQuery's and is `end - start`. While `DATEDIFF` was merely a WORD of the token above it - * inherited BigQuery's order, so it returned the opposite sign from the function it is named - * after — MySQL's own documented `DATEDIFF('2007-12-31','2007-12-30') -> 1` answered `-1` here. + /** `DATEDIFF`, the name MySQL, SQL Server, Snowflake, Redshift, DuckDB and Elasticsearch SQL give + * the function. It needs its own token because its two-argument form is MySQL's, which the + * `DATE_DIFF` productions do not parse (issue #363). * * 🔴 Neither spelling is a prefix of the other (`DATE_DIFF` has an underscore where `DATEDIFF` * has a `D`), so the two regexes cannot shadow one another whatever order they are tried in. */ case object MySqlDateDiff extends Expr("DATEDIFF") with TokenRegex - /** Which spelling a `DateDiff` was written as. It decides the RENDER and nothing else — the node - * itself always means `end - start`, so every consumer (`args`, `left`/`right`, the Painless - * emission, validation, `update`) has exactly ONE encoding of the decision to read. + /** How a `DateDiff` was written, which decides its RENDER and what it COUNTS. The node itself + * always means `end - start`: the parser stores the operands so, and every other consumer + * (`args`, `left`/`right`, the Painless emission, validation, `update`) reads that one encoding. * - * A `Boolean` cannot carry three forms, and silently widening one is how the old `transactSql` - * flag would have rotted; being sealed, the compiler now forces every render arm. + * The vendors' definitions, fetched from their references (2026-10-02), and the lead's ruling to + * follow them on both axes: + * + * - DIRECTION, by layout. Every vendor computes `end - start`; the layout decides where the + * end sits. Dates first is `first - second`: MySQL's `DATEDIFF(expr1, expr2)`, BigQuery's + * `DATE_DIFF(end_date, start_date, part)` (`DATE_DIFF('2010-07-07', '2008-12-25', DAY)` is + * 559). Unit first is `last - middle`: `DATEDIFF(unit, start, end)` in SQL Server, + * Snowflake, Redshift, DuckDB and Elasticsearch SQL, and MySQL's `TIMESTAMPDIFF(unit, dt1, + * dt2)`. + * - COUNTING, by name. `DATEDIFF` and `DATE_DIFF` count the calendar BOUNDARIES crossed, as + * SQL Server, Snowflake, Redshift, BigQuery and DuckDB's `date_diff` do: one second across a + * year end is one YEAR. `TIMESTAMPDIFF` counts the whole units ELAPSED, truncated toward + * zero, as MySQL does: a month counts once its day and time of day are reached. + * + * 🔴 Issue #363 read BigQuery's `DATE_DIFF` as `(start, end, unit)`, so until 0.24.0 the + * dates-first forms with a unit subtracted the other way round, and every unit from WEEK up + * counted elapsed units. Sealed, so the compiler forces every arm of the render. */ - sealed trait DateDiffSpelling + sealed trait DateDiffSpelling { + + /** `TIMESTAMPDIFF` counts whole units ELAPSED; every other spelling counts BOUNDARIES. */ + def elapsed: Boolean = false + } object DateDiffSpelling { - /** `DATE_DIFF(start, end, unit)` — BigQuery's order, this engine's canonical render. */ + /** `DATE_DIFF(end, start[, unit])` (BigQuery's order) and `DATEDIFF(end, start, unit)`: dates + * first, so the parser stores the SECOND operand as the start. + */ case object DateFirst extends DateDiffSpelling - /** `DATE_DIFF(unit, start, end)` / `TIMESTAMPDIFF(unit, start, end)` — the ODBC/T-SQL and MySQL - * `TIMESTAMPDIFF` order. MySQL defines that one as `dt2 - dt1`, which is what this engine - * already computed, so it needed no change. - */ + /** `DATE_DIFF(unit, start, end)` / `DATEDIFF(unit, start, end)`: unit first. */ case object UnitFirst extends DateDiffSpelling - /** `DATEDIFF(expr1, expr2)` — MySQL's, `expr1 - expr2`, days only. The parser stores it with - * `start`/`end` SWAPPED, so the node still means `end - start`; the render swaps them back. + /** `DATEDIFF(end, start)`: MySQL's, in days. It means what `DATE_DIFF(end, start, DAY)` means; + * the spelling is kept so that `SHOW CREATE …` hands back what was written. */ case object MySql extends DateDiffSpelling + + /** `TIMESTAMPDIFF(unit, start, end)`: MySQL's, whole units ELAPSED. */ + case object TimestampDiff extends DateDiffSpelling { + override val elapsed: Boolean = true + } + + /** `TIMESTAMPDIFF(end, start[, unit])`. No vendor writes it; it parses because `TIMESTAMPDIFF` + * is a word of the `DATE_DIFF` token, and the two rules above give it its meaning: dates + * first, whole units elapsed. + */ + case object TimestampDiffDateFirst extends DateDiffSpelling { + override val elapsed: Boolean = true + } } case class DateDiff( @@ -628,22 +726,24 @@ package object time { * `MaterializedViewExtension` persists this text and re-runs `client.run(alter.sql)`, `SHOW * CREATE MATERIALIZED VIEW` echoes it, and `SCRIPT AS` stores it beside the Painless. * - * 🔴 Keeping the `DATEDIFF` spelling is a READABILITY choice, not a correctness one, and the - * distinction is worth stating because the opposite claim is easy to reach for. Because the - * parser already stored MySQL's operands swapped, rendering this node as `DATE_DIFF(start, - * end, unit)` would ALSO re-parse to the same node and mean the same thing — a mutation that - * does exactly that reddens one assertion here, and it is the spelling one. What the swap - * protects against is the OTHER design, the one where the node keeps the operands as written - * and reverses them at emission: there the canonical render really does flip the sign of a - * stored statement on its next round trip (issue #363). + * 🔴 The dates-first forms are stored with their operands SWAPPED (the second is the start), + * and the render swaps them back. What the swap protects against is the other design, the one + * where the node keeps the operands as written and reverses them at emission: there the + * canonical render flips the sign of a stored statement on its next round trip (issue #363). * - * The spelling is preserved so that `SHOW CREATE …` hands back what was written, rather than - * the same statement with its two arguments visibly exchanged. + * 🔴 `TIMESTAMPDIFF` keeps its name, and that is correctness, not readability: it counts + * elapsed units where `DATE_DIFF` counts boundaries, so a render as `DATE_DIFF` would change + * what a stored statement answers. `DATEDIFF(a, b)` keeps its name for readability only (it + * means `DATE_DIFF(a, b, DAY)`), so that `SHOW CREATE …` hands back what was written. */ override def toSQL(base: String): String = spelling match { case DateDiffSpelling.UnitFirst => s"$sql(${unit.sql}, ${start.sql}, ${end.sql})" case DateDiffSpelling.MySql => s"${MySqlDateDiff.sql}(${end.sql}, ${start.sql})" - case DateDiffSpelling.DateFirst => s"$sql(${start.sql}, ${end.sql}, ${unit.sql})" + case DateDiffSpelling.DateFirst => s"$sql(${end.sql}, ${start.sql}, ${unit.sql})" + case DateDiffSpelling.TimestampDiff => + s"${DateDiff.timestampDiff}(${unit.sql}, ${start.sql}, ${end.sql})" + case DateDiffSpelling.TimestampDiffDateFirst => + s"${DateDiff.timestampDiff}(${end.sql}, ${start.sql}, ${unit.sql})" } /** The type a COLUMN operand is read as, whatever the unit: the instant it denotes, in UTC @@ -659,20 +759,39 @@ package object time { */ override def in: SQLType = SQLTypes.Timestamp - /** The type the two operands are COMPARED in, and the UNIT decides it. + /** The type the two operands are COMPARED in, in UTC; the counting and the unit decide it. * - * - `HOUR`, `MINUTE` and `SECOND` count the ELAPSED whole units between two instants (UTC), - * truncated toward zero; a DATE operand is the start of its day. They used to be compared - * as a `LocalDate`, which has no time of day, so every such call failed with `Unsupported - * unit: Hours`. - * - `DAY` and every larger unit compare the two CALENDAR dates (UTC): 23:30 and 00:30 the - * next day are one day apart, as MySQL's `DATEDIFF` answers. That is what they computed. + * - `TIMESTAMPDIFF` counts the whole units ELAPSED between two instants, whatever the unit: + * a DATE operand is the start of its day. + * - `DATEDIFF` / `DATE_DIFF` count BOUNDARIES: between two calendar dates from `DAY` up + * (23:30 and 00:30 the next day are one day apart), between two instants for `HOUR`, + * `MINUTE` and `SECOND`, which a `LocalDate` cannot hold (`Unsupported unit: Hours`). */ private def comparedIn: SQLType = unit match { + case _ if spelling.elapsed => SQLTypes.Timestamp case TimeUnit.HOURS | TimeUnit.MINUTES | TimeUnit.SECONDS => SQLTypes.Timestamp case _ => SQLTypes.Date } + /** What each operand is moved to before the units between them are counted: for a BOUNDARY + * count, the start of the unit it falls in (UTC), so that the whole units elapsed between the + * two starts are the boundaries crossed between the operands — 2005-12-31 23:59:59 and + * 2006-01-01 00:00:00 are one YEAR apart, 1992-09-15 and 1992-11-14 two MONTHs, 10:59 and + * 11:00 one HOUR. A WEEK starts on the ISO Monday ([[isoWeekStart]]), a QUARTER on January, + * April, July or October 1st. An ELAPSED count moves nothing. + */ + private def unitStart: String = + if (spelling.elapsed) "" + else + unit match { + case TimeUnit.YEARS => ".withDayOfYear(1)" + case TimeUnit.QUARTERS => ".with(java.time.temporal.IsoFields.DAY_OF_QUARTER, 1)" + case TimeUnit.MONTHS => ".withDayOfMonth(1)" + case TimeUnit.WEEKS => isoWeekStart + case TimeUnit.DAYS => "" + case subDay => s".truncatedTo(${subDay.painless(None)})" + } + /** A context-free rendering over an aggregate is the PER-GROUP calculation: the `bucket_script` * of a SELECT item (`DATEDIFF(MAX(d), '2024-01-01') AS x`), which a HAVING over `x` reads by * name and a materialized view's pivot computes too (`SingleSearch.transformBucketScripts`). @@ -688,32 +807,38 @@ package object time { case _ => super[BinaryFunction].painless(context) } - /** One operand, brought to [[comparedIn]] through the coercion arms a CAST uses. + /** One operand, brought to [[comparedIn]] through the coercion arms a CAST uses, then to the + * start of its unit ([[unitStart]]). * * - A string LITERAL is first read as the temporal it spells ([[DateDiff.literalType]]: UTC - * unless it names a zone). Row level used to hand Painless the bare string - * (`between("2024-01-01", param1)`, which it cannot call), and per group parsed it as a - * DATE whatever it held, so a time of day failed the search. + * unless it names a zone), in every venue. Row level used to hand Painless the bare string + * (`between("2024-01-01", param1)`, which it cannot call), and an ingest processor still + * did. * - Per group, an aggregate is the metric Elasticsearch computed, which a `bucket_script` * receives as a `java.lang.Double` holding EPOCH MILLIS (`(long)` is required: Painless * refuses to cast a `def` double to `long` implicitly). Any other operand is converted * from its own type. + * - In an INGEST processor any other operand is a UTC temporal the processor made itself: a + * column parsed at runtime by `SQLTypeUtils.processorTemporal`, the clock, a `::DATE` + * literal. The first two are `ZonedDateTime`s whatever their type, the last a `LocalDate`, + * and `LocalDate.from` reads the calendar date of either. A DATE under a time-of-day + * comparison is the start of its day, as at row level — `CURRENT_DATE` included, which a + * processor holds as the current instant. * - At row level a column operand holds its UTC instant ([[in]]), so its calendar date is * `toLocalDate()`. An operand with no column of its own (`'2025-01-10'::DATE`, - * `CURRENT_DATE`, `NOW()`) renders as it did, except a DATE under a sub-day unit, which is - * the start of its day: a `LocalDate` has no hours. - * - * 🔴 An INGEST processor renders as it did: there a column is the raw document value, parsed - * at runtime by `SQLTypeUtils.processorTemporal`. + * `CURRENT_DATE`, `NOW()`) is read by `LocalDate.from` for a calendar date, which takes + * the `LocalDate` and the UTC `ZonedDateTime` alike; for a time-of-day comparison a DATE + * is the start of its day (a `LocalDate` has no hours) and anything else renders as it + * did. + * - `TIMESTAMPDIFF` takes [[elapsedOperand]] instead, in every venue. */ private def operand( arg: PainlessScript, rendered: String, context: Option[PainlessContext], perGroup: Boolean - ): String = - DateDiff.literalType(arg) match { - case _ if context.exists(_.isProcessor) => rendered + ): String = { + val compared = DateDiff.literalType(arg) match { case Some(literal) => val value = SQLTypeUtils.coerce(rendered, SQLTypes.Varchar, literal, nullable = false, None) @@ -724,15 +849,22 @@ package object time { val epochMillis = SQLTypeUtils .coerce(rendered, SQLTypes.Double, SQLTypes.BigInt, nullable = false, None) // The instant in UTC already (`Instant.ofEpochMilli(...).atZone(ZoneId.of('Z'))`), so - // DAY and above read its calendar date off it: ONE conversion per aggregate. The - // TIMESTAMP -> DATE arm would normalise it to UTC a second time - // (`.toInstant().atZone(ZoneId.of('Z'))`), as it must an operand of unknown zone. + // a calendar date is read off it: ONE conversion per aggregate. The TIMESTAMP -> DATE + // arm would normalise it to UTC a second time (`.toInstant().atZone(ZoneId.of('Z'))`), + // as it must an operand of unknown zone. val utc = SQLTypeUtils .coerce(epochMillis, SQLTypes.BigInt, SQLTypes.Timestamp, nullable = false, None) if (comparedIn == SQLTypes.Date) s"$utc.toLocalDate()" else utc + case _ if spelling.elapsed => elapsedOperand(arg, rendered, context) case other => SQLTypeUtils.coerce(rendered, other.baseType, comparedIn, nullable = false, None) } + case None if spelling.elapsed => elapsedOperand(arg, rendered, context) + case None if context.exists(_.isProcessor) => + if (comparedIn == SQLTypes.Date) s"LocalDate.from($rendered)" + else if (arg.baseType == SQLTypes.Date) + s"LocalDate.from($rendered).atStartOfDay(ZoneId.of('Z'))" + else rendered case None if context.isDefined => arg match { case column: Identifier if column.name.trim.nonEmpty => @@ -740,20 +872,76 @@ package object time { case value: Identifier if comparedIn == SQLTypes.Timestamp && value.baseType == SQLTypes.Date => SQLTypeUtils.coerce(rendered, SQLTypes.Date, comparedIn, nullable = false, None) - case _ => rendered + // a `ZonedDateTime` here (`'…'::TIMESTAMP`, `NOW()`, a CASE) would keep its time of day + // through `unitStart`, and a calendar unit would count it + case _ if comparedIn == SQLTypes.Date => s"LocalDate.from($rendered)" + case _ => rendered } case None => rendered } + if (unitStart.isEmpty) compared + else if (DateDiff.postfixable(compared)) s"$compared$unitStart" + else s"($compared)$unitStart" + } + + /** A `TIMESTAMPDIFF` operand, as the instant it denotes in UTC. The function compares INSTANTS: + * `ChronoUnit.between` reads its end as the type of its start, so a `ZonedDateTime` start + * refused a `LocalDate` or a `LocalDateTime` end, and the month count ([[DateDiff.monthKey]]) + * reads a time of day a `LocalDate` does not have. Either way the statement failed. + * + * - An ingest processor reads a DATE-typed operand as the start of its calendar day + * (`LocalDate.from`): it holds `CURRENT_DATE` as the current INSTANT, so there the + * declared type, not the runtime one, is what says "a date". + * - A column is an instant already, in a script that reads it: row level reads it as its UTC + * instant ([[in]]), an ingest processor parses it into one + * (`SQLTypeUtils.processorTemporal`). A function over it need not be one: `ts::DATE` is a + * `LocalDate`. + * - Any other operand is converted by its RUNTIME type ([[DateDiff.utcInstant]]), which its + * declared type does not give: `COALESCE(ts, CURRENT_DATE)`, or a CASE over `CURRENT_DATE` + * and a DATE, is a `LocalDate` on one row and a `ZonedDateTime` on another, a `::DATETIME` + * a `LocalDateTime`. It is evaluated once, bound to a name where the script can bind one; + * a `bucket_script` cannot, and per group such an operand is a constant of the request. + */ + private def elapsedOperand( + arg: PainlessScript, + rendered: String, + context: Option[PainlessContext] + ): String = arg match { + case _ if context.exists(_.isProcessor) && arg.baseType == SQLTypes.Date => + s"LocalDate.from($rendered).atStartOfDay(ZoneId.of('Z'))" + case column: Identifier + if context.isDefined && column.name.trim.nonEmpty && column.functions.isEmpty => + rendered + case _ => + val ref = context match { + case Some(_) if FunctionN.isName(rendered) => rendered + case Some(ctx) => ctx.addParam(LiteralParam(rendered)).getOrElse(s"((def) ($rendered))") + case None => s"((def) ($rendered))" + } + DateDiff.utcInstant(ref) + } /** `ChronoUnit` has no `QUARTERS`, so a `QUARTER` call failed to compile in Elasticsearch. The - * ISO quarter-year unit counts the whole quarters between two calendar dates: the whole months - * between them, divided by 3. + * ISO quarter-year unit counts the whole quarters between two calendar dates, which between + * two quarter starts ([[unitStart]]) are the quarter boundaries crossed. */ private def unitPainless(context: Option[PainlessContext]): String = unit match { case TimeUnit.QUARTERS => "java.time.temporal.IsoFields.QUARTER_YEARS" case other => other.painless(context) } + /** `TIMESTAMPDIFF`'s MONTH, QUARTER and YEAR are whole months elapsed, divided by 1, 3 or 12. + */ + private def monthsPerUnit: Option[Int] = + if (!spelling.elapsed) None + else + unit match { + case TimeUnit.MONTHS => Some(1) + case TimeUnit.QUARTERS => Some(3) + case TimeUnit.YEARS => Some(12) + case _ => None + } + /** The function's ONE rendering: row level reaches it from `FunctionN.painless` with its * operands rendered, per group from [[painless]]; [[operand]] converts each one. */ @@ -765,8 +953,17 @@ package object time { val operands = args.zip(callArgs).map { case (arg, rendered) => operand(arg, rendered, context, perGroup) } - val ret = - s"Long.valueOf(${unitPainless(context)}${DateDiff.painless(context)}(${operands.mkString(", ")}))" + val ret = monthsPerUnit match { + case Some(perUnit) => + // each operand is read four times by `monthKey`: bound once where a script can bind it + val bound = + operands.map(op => context.flatMap(_.addParam(LiteralParam(op))).getOrElse(op)) + val months = + s"(${DateDiff.monthKey(bound(1))} - ${DateDiff.monthKey(bound.head)}) / ${DateDiff.MonthKeyScale}L" + s"Long.valueOf(${if (perUnit == 1) months else s"$months / $perUnit"})" + case None => + s"Long.valueOf(${unitPainless(context)}${DateDiff.painless(context)}(${operands.mkString(", ")}))" + } context match { case Some(ctx) if ctx.isProcessor => // to fix bug in painless script processor context with elasticsearch v6 diff --git a/sql/src/main/scala/app/softnetwork/elastic/sql/parser/function/time/package.scala b/sql/src/main/scala/app/softnetwork/elastic/sql/parser/function/time/package.scala index 909a0a62f..0d8233b89 100644 --- a/sql/src/main/scala/app/softnetwork/elastic/sql/parser/function/time/package.scala +++ b/sql/src/main/scala/app/softnetwork/elastic/sql/parser/function/time/package.scala @@ -204,51 +204,61 @@ package object time { // (`Parser.operandIdentifier`), as `ISNULL`'s is: an aggregate in them is the aggregate every // other position builds, not the bare `MAX` token -- which a view's transform did not // recognise, and whose rendered name (`MAX(d)`) did not match a qualified SELECT item's. + // + // The node means `end - start`, and the LAYOUT decides which operand is the end (see + // `DateDiffSpelling`): dates first is `first - second`, so the SECOND operand is stored as the + // start; unit first is `last - middle`. The NAME decides the counting: `TIMESTAMPDIFF` is the + // second word of `DateDiff.regex`, so the word read is kept. lazy val date_diff: PackratParser[BinaryFunction[_, _, _]] = DateDiff.regex ~ start ~ operandIdentifier ~ separator ~ operandIdentifier ~ (separator ~ time_unit).? ~ end ^^ { - case _ ~ _ ~ d1 ~ _ ~ d2 ~ u ~ _ => + case name ~ _ ~ d1 ~ _ ~ d2 ~ u ~ _ => DateDiff( - d1, d2, + d1, u match { case Some(_ ~ unit) => unit case None => TimeUnit.DAYS - } + }, + if (name.equalsIgnoreCase(DateDiff.timestampDiff)) + DateDiffSpelling.TimestampDiffDateFirst + else DateDiffSpelling.DateFirst ) } - /** 🔴 BOTH names. `DATEDIFF(unit, start, end)` is T-SQL's and DuckDB's spelling — what a BI - * tool set to a SQL Server or DuckDB dialect emits — and it is `end - start` in both, which is - * exactly what this production binds. Issue #363 moved `DATEDIFF` onto its own token to give - * the TWO-argument MySQL form its own meaning, and keying this production on `DateDiff.regex` - * alone silently dropped the unit-first spelling with it. + /** 🔴 BOTH names. `DATEDIFF(unit, start, end)` is the spelling of SQL Server, Snowflake, + * Redshift, DuckDB and Elasticsearch SQL — what a BI tool set to one of those dialects emits — + * and it is `end - start` in all of them, which is exactly what this production binds. Issue + * #363 moved `DATEDIFF` onto its own token to give the TWO-argument MySQL form its own + * meaning, and keying this production on `DateDiff.regex` alone silently dropped the + * unit-first spelling with it. * * Ordering is safe: this production is tried BEFORE `mysql_date_diff`, and `time_unit` fails * on a plain column, so `DATEDIFF(a, b)` still falls through to the MySQL form. */ lazy val date_diff_transact_sql: PackratParser[BinaryFunction[_, _, _]] = - (DateDiff.regex | MySqlDateDiff.regex) ~ start ~> time_unit ~ separator ~ operandIdentifier ~ separator ~ operandIdentifier <~ end ^^ { - case u ~ _ ~ d1 ~ _ ~ d2 => - DateDiff(d1, d2, u, DateDiffSpelling.UnitFirst) + (DateDiff.regex | MySqlDateDiff.regex) ~ start ~ time_unit ~ separator ~ operandIdentifier ~ separator ~ operandIdentifier ~ end ^^ { + case name ~ _ ~ u ~ _ ~ d1 ~ _ ~ d2 ~ _ => + DateDiff( + d1, + d2, + u, + if (name.equalsIgnoreCase(DateDiff.timestampDiff)) DateDiffSpelling.TimestampDiff + else DateDiffSpelling.UnitFirst + ) } - /** MySQL's `DATEDIFF` (issue #363), which is NOT the function `date_diff` above parses. - * - * Two arguments — the shape MySQL actually defines — means `expr1 - expr2`, so the operands - * are stored SWAPPED and the node keeps its single meaning of `end - start`. The dialect lives - * here and in `toSQL`, nowhere else. + /** `DATEDIFF` with the dates first, which `date_diff` above does not parse (issue #363). * - * 🔴 The THREE-argument `DATEDIFF(a, b, unit)` is NOT MySQL — MySQL's is days-only — it is - * this engine's own extension, and a lead ruling keeps it exactly as it behaved before: `end - - * start`, rendered as `DATE_DIFF`. So on this ONE spelling the arity changes the sign, and - * `DateDiffSignSpec` pins both readings side by side so that stays deliberate and visible - * rather than discovered. + * Two arguments is MySQL's `DATEDIFF(expr1, expr2)`, `expr1 - expr2` in days; with a unit it + * is the same layout, so the same direction: `first - second` either way, and the operands are + * stored SWAPPED so that the node keeps its single meaning of `end - start`. Before 0.24.0 the + * three-argument form subtracted the other way round (see `DateDiffSpelling`). */ lazy val mysql_date_diff: PackratParser[BinaryFunction[_, _, _]] = MySqlDateDiff.regex ~ start ~ operandIdentifier ~ separator ~ operandIdentifier ~ (separator ~ time_unit).? ~ end ^^ { case _ ~ _ ~ d1 ~ _ ~ d2 ~ u ~ _ => u match { - case Some(_ ~ unit) => DateDiff(d1, d2, unit, DateDiffSpelling.DateFirst) + case Some(_ ~ unit) => DateDiff(d2, d1, unit, DateDiffSpelling.DateFirst) case None => DateDiff(d2, d1, TimeUnit.DAYS, DateDiffSpelling.MySql) } } diff --git a/sql/src/test/scala/app/softnetwork/elastic/sql/parser/DateDiffSignSpec.scala b/sql/src/test/scala/app/softnetwork/elastic/sql/parser/DateDiffSignSpec.scala index 9da2b432d..d28ca73cc 100644 --- a/sql/src/test/scala/app/softnetwork/elastic/sql/parser/DateDiffSignSpec.scala +++ b/sql/src/test/scala/app/softnetwork/elastic/sql/parser/DateDiffSignSpec.scala @@ -20,15 +20,23 @@ import app.softnetwork.elastic.sql.function.time.{DateDiff, DateDiffSpelling} import org.scalatest.flatspec.AnyFlatSpec import org.scalatest.matchers.should.Matchers -/** Issue #363 — `DATEDIFF` is MySQL's function and must return MySQL's sign. +/** `DATEDIFF`, `DATE_DIFF` and `TIMESTAMPDIFF` subtract and count as the vendors define them. * - * MySQL 8.4 defines `DATEDIFF(expr1, expr2)` as `expr1 - expr2`; BigQuery's `DATE_DIFF(start, end, - * unit)` is `end - start`. While `DATEDIFF` was only a WORD of the `DATE_DIFF` token it inherited - * BigQuery's order and answered the opposite sign. + * Every vendor computes `end - start`, and the LAYOUT decides where the end sits. With the dates + * first it is the FIRST operand: MySQL's `DATEDIFF(expr1, expr2)`, BigQuery's `DATE_DIFF(end_date, + * start_date, part)` (`DATE_DIFF('2010-07-07', '2008-12-25', DAY)` is 559). With the unit first it + * is the LAST: `DATEDIFF(unit, start, end)` in SQL Server, Snowflake, Redshift, DuckDB and + * Elasticsearch SQL, and MySQL's `TIMESTAMPDIFF(unit, dt1, dt2)`. The NAME decides the counting: + * `TIMESTAMPDIFF` counts whole units elapsed, the other two calendar boundaries. + * + * 🔴 Issue #363 rested on a wrong premise, that BigQuery's `DATE_DIFF` takes `(start, end, unit)`. + * Until 0.24.0 the dates-first forms with a unit (`DATE_DIFF(a, b, unit)`, `DATEDIFF(a, b, unit)`) + * therefore answered `b - a`, and adding a unit flipped `DATEDIFF`'s sign. * * 🔴 The assertions here are on the AST's `start`/`end`, never on "it parsed". The node always - * means `end - start`, so which operand landed in which field IS the semantics — and a test that - * only checked `isRight`, or only the rendered text, would pass against the defect. + * means `end - start`, so which operand landed in which field IS the direction, and a test that + * only checked `isRight`, or only the rendered text, would pass against the defect. The answers + * themselves are RUN in `DateFunctionExecutionSpec`. */ class DateDiffSignSpec extends AnyFlatSpec with Matchers { @@ -50,51 +58,68 @@ class DateDiffSignSpec extends AnyFlatSpec with Matchers { private def renderOf(sql: String): String = Parser(s"SELECT $sql AS v FROM t").map(_.sql).getOrElse(fail(s"[$sql] rejected")) - // ─────────────────────────────── the sign itself ──────────────────────────────── + // ──────────────────────────── the direction, by layout ───────────────────────────── - /** MySQL's own two documented examples. Before this fix they answered -1 and 31. */ - "DATEDIFF with two arguments" should "compute expr1 - expr2, as MySQL does" in { - val d = dateDiffOf("DATEDIFF(a, b)") - // `end - start` is what the node means, so MySQL's `a - b` requires start = b, end = a. - d.start.sql shouldBe "b" - d.end.sql shouldBe "a" - d.spelling shouldBe DateDiffSpelling.MySql + "the dates first" should "subtract the second operand from the first, in both names and at every arity" in { + Seq( + "DATEDIFF(a, b)", + "DATEDIFF(a, b, DAY)", + "DATEDIFF(a, b, MONTH)", + "DATE_DIFF(a, b)", + "DATE_DIFF(a, b, YEAR)" + ).foreach { sql => + withClue(s"[$sql] ") { + val d = dateDiffOf(sql) + // `end - start` is what the node means, so `a - b` requires start = b, end = a. + (d.start.sql, d.end.sql) shouldBe (("b", "a")) + } + } } - "DATE_DIFF" should "keep BigQuery's end - start, untouched by this change" in { - val d = dateDiffOf("DATE_DIFF(a, b)") - d.start.sql shouldBe "a" - d.end.sql shouldBe "b" + it should "read BigQuery's own documented example with its end first" in { + // DATE_DIFF('2010-07-07', '2008-12-25', DAY) is 559, a positive number: the end comes first. + val d = dateDiffOf("DATE_DIFF('2010-07-07', '2008-12-25', DAY)") + (d.start.sql, d.end.sql) shouldBe (("'2008-12-25'", "'2010-07-07'")) d.spelling shouldBe DateDiffSpelling.DateFirst } - "TIMESTAMPDIFF" should "keep MySQL's dt2 - dt1, which it already matched" in { - val d = dateDiffOf("TIMESTAMPDIFF(DAY, a, b)") - d.start.sql shouldBe "a" - d.end.sql shouldBe "b" - d.spelling shouldBe DateDiffSpelling.UnitFirst + it should "no longer change DATEDIFF's sign when a unit is added" in { + // The pair #363 pinned as deliberately different: one layout, one direction. + val two = dateDiffOf("DATEDIFF(a, b)") + val three = dateDiffOf("DATEDIFF(a, b, DAY)") + (two.start.sql, two.end.sql) shouldBe ((three.start.sql, three.end.sql)) + two.spelling shouldBe DateDiffSpelling.MySql + three.spelling shouldBe DateDiffSpelling.DateFirst } - /** 🔴 The regression an independent review caught, and the reason it is pinned here. + "the unit first" should "subtract the middle operand from the last, in all three names" in { + Seq( + "DATEDIFF(DAY, a, b)", + "DATEDIFF(MONTH, a, b)", + "DATE_DIFF(YEAR, a, b)", + "TIMESTAMPDIFF(DAY, a, b)" + ).foreach { sql => + withClue(s"[$sql] ") { + val d = dateDiffOf(sql) + (d.start.sql, d.end.sql) shouldBe (("a", "b")) + } + } + } + + /** 🔴 The regression an independent review caught on #363, and the reason it is pinned here. * - * `DATEDIFF(unit, start, end)` is T-SQL's and DuckDB's spelling — what a BI tool set to a SQL - * Server or DuckDB dialect emits — and both define it as `end - start`, which is what this - * engine already computed. Moving `DATEDIFF` onto its own token for the MySQL form silently took + * `DATEDIFF(unit, start, end)` is what a BI tool set to a SQL Server, Snowflake, Redshift or + * DuckDB dialect emits. Moving `DATEDIFF` onto its own token for the MySQL form silently took * the unit-first spelling with it, because `date_diff_transact_sql` keyed on `DateDiff.regex` - * alone. It parsed on `main` and stopped parsing here. + * alone. * - * 🔴 `GrammarDiffProbe` did NOT catch it: its corpus contains no `DATEDIFF(unit, a, b)` input. - * That is the second time that probe has returned a clean differential over a real narrowing, so - * it is necessary and not sufficient — a spelling removed from a `words` list needs its own - * assertion, not a corpus sweep. + * 🔴 `GrammarDiffProbe` did NOT catch it: its corpus contains no `DATEDIFF(unit, a, b)` input. A + * spelling removed from a `words` list needs its own assertion, not a corpus sweep. */ - "DATEDIFF with the unit first" should "keep T-SQL's and DuckDB's spelling, and their order" in { + "DATEDIFF with the unit first" should "keep its spelling" in { Seq("DAY", "MONTH", "YEAR").foreach { unit => withClue(s"[DATEDIFF($unit, a, b)] ") { - val d = dateDiffOf(s"DATEDIFF($unit, a, b)") - d.start.sql shouldBe "a" - d.end.sql shouldBe "b" - d.spelling shouldBe DateDiffSpelling.UnitFirst + dateDiffOf(s"DATEDIFF($unit, a, b)").spelling shouldBe DateDiffSpelling.UnitFirst } } } @@ -102,28 +127,35 @@ class DateDiffSignSpec extends AnyFlatSpec with Matchers { it should "still let the two-argument MySQL form through, which follows it in the alternation" in { // `time_unit` fails on a plain column, so `DATEDIFF(a, b)` falls past the unit-first // production to the MySQL one. If that ordering ever breaks, this goes red rather than the - // sign silently reverting. + // direction silently changing. val d = dateDiffOf("DATEDIFF(a, b)") d.spelling shouldBe DateDiffSpelling.MySql (d.start.sql, d.end.sql) shouldBe (("b", "a")) } - /** 🔴 The lead ruling, pinned BOTH WAYS so it stays deliberate: on this one spelling, adding a - * unit changes the sign, because the 3-argument form is this engine's own extension and is not - * MySQL. If either half of this ever moves, it should move on purpose. - */ - "DATEDIFF with three arguments" should "keep this engine's end - start, unlike the two-arg form" in { - val two = dateDiffOf("DATEDIFF(a, b)") - val three = dateDiffOf("DATEDIFF(a, b, DAY)") - withClue("the 2-arg MySQL form swaps: ") { - (two.start.sql, two.end.sql) shouldBe ("b", "a") - } - withClue("the 3-arg extension does not: ") { - (three.start.sql, three.end.sql) shouldBe ("a", "b") + // ──────────────────────────── the counting, by name ───────────────────────────── + + "TIMESTAMPDIFF" should "count whole units elapsed, where DATEDIFF and DATE_DIFF count boundaries" in { + dateDiffOf("TIMESTAMPDIFF(MONTH, a, b)").spelling shouldBe DateDiffSpelling.TimestampDiff + dateDiffOf("TIMESTAMPDIFF(MONTH, a, b)").spelling.elapsed shouldBe true + Seq( + "DATEDIFF(a, b)", + "DATEDIFF(a, b, MONTH)", + "DATEDIFF(MONTH, a, b)", + "DATE_DIFF(a, b, MONTH)", + "DATE_DIFF(MONTH, a, b)" + ).foreach { sql => + withClue(s"[$sql] ")(dateDiffOf(sql).spelling.elapsed shouldBe false) } - three.spelling shouldBe DateDiffSpelling.DateFirst - // Stated as a difference, so the test fails if they are ever quietly unified. - (two.start.sql, two.end.sql) should not be (three.start.sql, three.end.sql) + } + + it should "take the dates first too, as no vendor writes it: first - second, elapsed" in { + // It parses because TIMESTAMPDIFF is a word of the DATE_DIFF token, and the two rules give it + // its meaning. + val d = dateDiffOf("TIMESTAMPDIFF(a, b, DAY)") + (d.start.sql, d.end.sql) shouldBe (("b", "a")) + d.spelling shouldBe DateDiffSpelling.TimestampDiffDateFirst + d.spelling.elapsed shouldBe true } // ────────────────────── the render must re-parse to the SAME node ────────────────────── @@ -131,27 +163,27 @@ class DateDiffSignSpec extends AnyFlatSpec with Matchers { /** 🔴 THIS is the correctness property, and it is the one the naive fix would have broken: with * the operands kept as written and reversed at emission, the canonical render flips the sign * back on a round trip. It matters because `MaterializedViewExtension` PERSISTS the render and - * re-runs it. - * - * Note what this does NOT say: it does not say the `DATEDIFF` spelling must survive the render. - * Because the parser stores MySQL's operands swapped, `DATE_DIFF(start, end, unit)` would - * re-parse to this same node too. That is a separate, weaker assertion below. + * re-runs it. The counting must survive it too, which is why `TIMESTAMPDIFF` keeps its name. */ - "the render" should "re-parse to a node with the same start and end, for every spelling" in { + "the render" should "re-parse to a node with the same start, end and counting, for every spelling" in { Seq( "DATEDIFF(a, b)", "DATEDIFF(a, b, DAY)", + "DATEDIFF(a, b, MONTH)", "DATE_DIFF(a, b)", "DATE_DIFF(a, b, MONTH)", + "DATEDIFF(WEEK, a, b)", + "DATE_DIFF(DAY, a, b)", "TIMESTAMPDIFF(DAY, a, b)", - "DATE_DIFF(DAY, a, b)" + "TIMESTAMPDIFF(a, b, QUARTER)" ).foreach { sql => withClue(s"[$sql] ") { val once = dateDiffOf(sql) val rendered = renderOf(sql) val twice = dateDiffOf(rendered.substring(rendered.indexOf("SELECT ") + 7).split(" AS ").head) - (twice.start.sql, twice.end.sql) shouldBe ((once.start.sql, once.end.sql)) + (twice.start.sql, twice.end.sql, twice.unit, twice.spelling.elapsed) shouldBe + ((once.start.sql, once.end.sql, once.unit, once.spelling.elapsed)) withClue(s"render [$rendered] must be a fixed point - ") { renderOf( rendered.substring(rendered.indexOf("SELECT ") + 7).split(" AS ").head @@ -162,16 +194,21 @@ class DateDiffSignSpec extends AnyFlatSpec with Matchers { } /** A READABILITY choice, pinned so it is not changed by accident — not a correctness one. - * Normalising to `DATE_DIFF(b, a, DAY)` would be equally correct; it would just hand the user - * back their own statement with the two arguments visibly exchanged. + * `DATE_DIFF(a, b, DAY)` means the same; it would just hand the user back another function. */ it should "keep MySQL's spelling, so SHOW CREATE hands back what was written" in { renderOf("DATEDIFF(a, b)") should include("DATEDIFF(a, b)") renderOf("DATEDIFF(a, b)") should not include "DATE_DIFF" } - it should "render the two ODBC/BigQuery spellings as DATE_DIFF, as before" in { + it should "keep TIMESTAMPDIFF's name, which decides what it counts" in { + renderOf("TIMESTAMPDIFF(DAY, a, b)") should include("TIMESTAMPDIFF(DAY, a, b)") + renderOf("TIMESTAMPDIFF(a, b, DAY)") should include("TIMESTAMPDIFF(a, b, DAY)") + } + + it should "render every other spelling as DATE_DIFF, its operands in the order written" in { renderOf("DATE_DIFF(a, b)") should include("DATE_DIFF(a, b, DAY)") - renderOf("TIMESTAMPDIFF(DAY, a, b)") should include("DATE_DIFF(DAY, a, b)") + renderOf("DATEDIFF(a, b, MONTH)") should include("DATE_DIFF(a, b, MONTH)") + renderOf("DATEDIFF(DAY, a, b)") should include("DATE_DIFF(DAY, a, b)") } } diff --git a/sql/src/test/scala/app/softnetwork/elastic/sql/parser/ParserSpec.scala b/sql/src/test/scala/app/softnetwork/elastic/sql/parser/ParserSpec.scala index 177e0d3ea..c092bdb25 100644 --- a/sql/src/test/scala/app/softnetwork/elastic/sql/parser/ParserSpec.scala +++ b/sql/src/test/scala/app/softnetwork/elastic/sql/parser/ParserSpec.scala @@ -1441,13 +1441,13 @@ class ParserSpec extends AnyFlatSpec with Matchers { | id INT NOT NULL COMMENT 'user identifier', | name VARCHAR FIELDS(raw Keyword COMMENT 'sortable') DEFAULT 'anonymous' OPTIONS (analyzer = 'french', search_analyzer = 'french'), | birthdate DATE, - | age INT SCRIPT AS (DATEDIFF(birthdate, CURRENT_DATE, YEAR)), + | age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)), | ingested_at TIMESTAMP DEFAULT _ingest.timestamp, | profile STRUCT FIELDS( | bio VARCHAR, | followers INT, | join_date DATE, - | seniority INT SCRIPT AS (DATEDIFF(profile.join_date, CURRENT_DATE, DAY)) + | seniority INT SCRIPT AS (DATEDIFF(CURRENT_DATE, profile.join_date, DAY)) | ) COMMENT 'user profile', | PRIMARY KEY (id) |) PARTITION BY birthdate (MONTH), OPTIONS (mappings = (dynamic = false))""".stripMargin @@ -1490,13 +1490,15 @@ class ParserSpec extends AnyFlatSpec with Matchers { // to a `LocalDate` in a processor, so both sides of `between` are the same type. """def param1 = ctx.birthdate; |def param2 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); - |def param3 = Long.valueOf(ChronoUnit.YEARS.between(param1, param2)); - |ctx.age = (param1 == null) ? null : param3""".stripMargin.replaceAll("\n", " ") + |def param3 = LocalDate.from(param2).atStartOfDay(ZoneId.of('Z')); + |def param4 = Long.valueOf((((param3.getYear() * 12L + param3.getMonthValue()) * 4294967296L + (param3.getDayOfMonth() - 1) * 86400000L + param3.getLong(ChronoField.MILLI_OF_DAY)) - ((param1.getYear() * 12L + param1.getMonthValue()) * 4294967296L + (param1.getDayOfMonth() - 1) * 86400000L + param1.getLong(ChronoField.MILLI_OF_DAY))) / 4294967296L / 12); + |ctx.age = (param1 == null) ? null : param4""".stripMargin.replaceAll("\n", " ") ) // The RESOLVED processor, which is what `CREATE TABLE` deploys. The operand is parsed // first and its runtime shape decided by `instanceof`, because `ctx.birthdate` is the raw - // JSON value of the document being indexed. This exact pipeline was run on all four - // majors: {"birthdate":"1990-05-20"} and {"birthdate":643161600000} both store `age: 36`. + // JSON value of the document being indexed. The column is the whole years elapsed from + // `birthdate` to the start of `CURRENT_DATE` (UTC), a positive age; `IngestTemporalSpec` + // pins the same processor, which was run as a real ingest pipeline. ct.schema.columns .find(_.name == "age") .flatMap(_.script) @@ -1505,8 +1507,9 @@ class ParserSpec extends AnyFlatSpec with Matchers { """def param1 = ctx.birthdate; |def param2 = (param1 instanceof String ? LocalDate.parse((param1).replace("/", "-"), DateTimeFormatter.ofPattern("yyyy-MM-dd")).atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); |def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); - |def param4 = Long.valueOf(ChronoUnit.YEARS.between(param2, param3)); - |ctx.age = (param1 == null) ? null : param4""".stripMargin.replaceAll("\n", " ") + |def param4 = LocalDate.from(param3).atStartOfDay(ZoneId.of('Z')); + |def param5 = Long.valueOf((((param4.getYear() * 12L + param4.getMonthValue()) * 4294967296L + (param4.getDayOfMonth() - 1) * 86400000L + param4.getLong(ChronoField.MILLI_OF_DAY)) - ((param2.getYear() * 12L + param2.getMonthValue()) * 4294967296L + (param2.getDayOfMonth() - 1) * 86400000L + param2.getLong(ChronoField.MILLI_OF_DAY))) / 4294967296L / 12); + |ctx.age = (param1 == null) ? null : param5""".stripMargin.replaceAll("\n", " ") ) cols.find(_.name == "ingested_at").get.defaultValue.map(_.value) shouldBe Some( "_ingest.timestamp" @@ -1518,16 +1521,16 @@ class ParserSpec extends AnyFlatSpec with Matchers { println(schema.defaultPipeline.ddl) val json = schema.defaultPipeline.json println(json) - json shouldBe """{"description":"CREATE OR REPLACE PIPELINE users_ddl_default_pipeline WITH PROCESSORS (name SET DEFAULT 'anonymous', age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR)), ingested_at SET DEFAULT _ingest.timestamp, profile.seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY)), PARTITION BY birthdate (MONTH), PRIMARY KEY (id))","processors":[{"set":{"description":"name SET DEFAULT 'anonymous'","field":"name","ignore_failure":true,"value":"anonymous","if":"ctx.name == null"}},{"script":{"description":"age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR))","lang":"painless","source":"def param1 = ctx.birthdate; def param2 = (param1 instanceof String ? LocalDate.parse((param1).replace(\"/\", \"-\"), DateTimeFormatter.ofPattern(\"yyyy-MM-dd\")).atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); def param4 = Long.valueOf(ChronoUnit.YEARS.between(param2, param3)); ctx.age = (param1 == null) ? null : param4","ignore_failure":true}},{"set":{"description":"ingested_at SET DEFAULT _ingest.timestamp","field":"ingested_at","ignore_failure":true,"value":"{{_ingest.timestamp}}","if":"ctx.ingested_at == null"}},{"script":{"description":"profile.seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY))","lang":"painless","source":"def param1 = ctx.profile?.join_date; def param2 = (param1 instanceof String ? LocalDate.parse((param1).replace(\"/\", \"-\"), DateTimeFormatter.ofPattern(\"yyyy-MM-dd\")).atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); def param4 = Long.valueOf(ChronoUnit.DAYS.between(param2, param3)); ctx.profile.seniority = (param1 == null) ? null : param4","ignore_failure":true}},{"date_index_name":{"description":"PARTITION BY birthdate (MONTH)","field":"birthdate","date_rounding":"M","date_formats":["yyyy-MM"],"index_name_prefix":"users-","ignore_failure":true}},{"set":{"description":"PRIMARY KEY (id)","field":"_id","value":"{{id}}","ignore_failure":false,"ignore_empty_value":false}}]}""" + json shouldBe """{"description":"CREATE OR REPLACE PIPELINE users_ddl_default_pipeline WITH PROCESSORS (name SET DEFAULT 'anonymous', age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)), ingested_at SET DEFAULT _ingest.timestamp, profile.seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY)), PARTITION BY birthdate (MONTH), PRIMARY KEY (id))","processors":[{"set":{"description":"name SET DEFAULT 'anonymous'","field":"name","ignore_failure":true,"value":"anonymous","if":"ctx.name == null"}},{"script":{"description":"age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE))","lang":"painless","source":"def param1 = ctx.birthdate; def param2 = (param1 instanceof String ? LocalDate.parse((param1).replace(\"/\", \"-\"), DateTimeFormatter.ofPattern(\"yyyy-MM-dd\")).atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); def param4 = LocalDate.from(param3).atStartOfDay(ZoneId.of('Z')); def param5 = Long.valueOf((((param4.getYear() * 12L + param4.getMonthValue()) * 4294967296L + (param4.getDayOfMonth() - 1) * 86400000L + param4.getLong(ChronoField.MILLI_OF_DAY)) - ((param2.getYear() * 12L + param2.getMonthValue()) * 4294967296L + (param2.getDayOfMonth() - 1) * 86400000L + param2.getLong(ChronoField.MILLI_OF_DAY))) / 4294967296L / 12); ctx.age = (param1 == null) ? null : param5","ignore_failure":true}},{"set":{"description":"ingested_at SET DEFAULT _ingest.timestamp","field":"ingested_at","ignore_failure":true,"value":"{{_ingest.timestamp}}","if":"ctx.ingested_at == null"}},{"script":{"description":"profile.seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY))","lang":"painless","source":"def param1 = ctx.profile?.join_date; def param2 = (param1 instanceof String ? LocalDate.parse((param1).replace(\"/\", \"-\"), DateTimeFormatter.ofPattern(\"yyyy-MM-dd\")).atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); def param4 = Long.valueOf(ChronoUnit.DAYS.between(LocalDate.from(param2), LocalDate.from(param3))); ctx.profile.seniority = (param1 == null) ? null : param4","ignore_failure":true}},{"date_index_name":{"description":"PARTITION BY birthdate (MONTH)","field":"birthdate","date_rounding":"M","date_formats":["yyyy-MM"],"index_name_prefix":"users-","ignore_failure":true}},{"set":{"description":"PRIMARY KEY (id)","field":"_id","value":"{{id}}","ignore_failure":false,"ignore_empty_value":false}}]}""" val indexMappings = schema.indexMappings println(indexMappings) - indexMappings.toString shouldBe """{"properties":{"id":{"type":"integer"},"name":{"type":"text","fields":{"raw":{"type":"keyword"}},"analyzer":"french","search_analyzer":"french"},"birthdate":{"type":"date"},"age":{"type":"integer"},"ingested_at":{"type":"date"},"profile":{"type":"object","properties":{"bio":{"type":"text"},"followers":{"type":"integer"},"join_date":{"type":"date"},"seniority":{"type":"integer"}}}},"dynamic":false,"_meta":{"primary_key":["id"],"partition_by":{"column":"birthdate","granularity":"M"},"columns":{"id":{"data_type":"INT","not_null":"true","comment":"user identifier"},"name":{"data_type":"VARCHAR","not_null":"false","default_value":"anonymous","multi_fields":{"raw":{"data_type":"KEYWORD","not_null":"false","comment":"sortable"}}},"birthdate":{"data_type":"DATE","not_null":"false"},"age":{"data_type":"INT","not_null":"false","script":{"sql":"DATE_DIFF(birthdate, CURRENT_DATE, YEAR)","column":"age","painless":"def param1 = ctx.birthdate; def param2 = (param1 instanceof String ? LocalDate.parse((param1).replace(\"/\", \"-\"), DateTimeFormatter.ofPattern(\"yyyy-MM-dd\")).atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); def param4 = Long.valueOf(ChronoUnit.YEARS.between(param2, param3)); ctx.age = (param1 == null) ? null : param4"}},"ingested_at":{"data_type":"TIMESTAMP","not_null":"false","default_value":"_ingest.timestamp"},"profile":{"data_type":"STRUCT","not_null":"false","comment":"user profile","multi_fields":{"bio":{"data_type":"VARCHAR","not_null":"false"},"followers":{"data_type":"INT","not_null":"false"},"join_date":{"data_type":"DATE","not_null":"false"},"seniority":{"data_type":"INT","not_null":"false","script":{"sql":"DATE_DIFF(profile.join_date, CURRENT_DATE, DAY)","column":"profile.seniority","painless":"def param1 = ctx.profile?.join_date; def param2 = (param1 instanceof String ? LocalDate.parse((param1).replace(\"/\", \"-\"), DateTimeFormatter.ofPattern(\"yyyy-MM-dd\")).atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); def param4 = Long.valueOf(ChronoUnit.DAYS.between(param2, param3)); ctx.profile.seniority = (param1 == null) ? null : param4"}}}}},"type":"regular"}}""".stripMargin + indexMappings.toString shouldBe """{"properties":{"id":{"type":"integer"},"name":{"type":"text","fields":{"raw":{"type":"keyword"}},"analyzer":"french","search_analyzer":"french"},"birthdate":{"type":"date"},"age":{"type":"integer"},"ingested_at":{"type":"date"},"profile":{"type":"object","properties":{"bio":{"type":"text"},"followers":{"type":"integer"},"join_date":{"type":"date"},"seniority":{"type":"integer"}}}},"dynamic":false,"_meta":{"primary_key":["id"],"partition_by":{"column":"birthdate","granularity":"M"},"columns":{"id":{"data_type":"INT","not_null":"true","comment":"user identifier"},"name":{"data_type":"VARCHAR","not_null":"false","default_value":"anonymous","multi_fields":{"raw":{"data_type":"KEYWORD","not_null":"false","comment":"sortable"}}},"birthdate":{"data_type":"DATE","not_null":"false"},"age":{"data_type":"INT","not_null":"false","script":{"sql":"TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)","column":"age","painless":"def param1 = ctx.birthdate; def param2 = (param1 instanceof String ? LocalDate.parse((param1).replace(\"/\", \"-\"), DateTimeFormatter.ofPattern(\"yyyy-MM-dd\")).atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); def param4 = LocalDate.from(param3).atStartOfDay(ZoneId.of('Z')); def param5 = Long.valueOf((((param4.getYear() * 12L + param4.getMonthValue()) * 4294967296L + (param4.getDayOfMonth() - 1) * 86400000L + param4.getLong(ChronoField.MILLI_OF_DAY)) - ((param2.getYear() * 12L + param2.getMonthValue()) * 4294967296L + (param2.getDayOfMonth() - 1) * 86400000L + param2.getLong(ChronoField.MILLI_OF_DAY))) / 4294967296L / 12); ctx.age = (param1 == null) ? null : param5"}},"ingested_at":{"data_type":"TIMESTAMP","not_null":"false","default_value":"_ingest.timestamp"},"profile":{"data_type":"STRUCT","not_null":"false","comment":"user profile","multi_fields":{"bio":{"data_type":"VARCHAR","not_null":"false"},"followers":{"data_type":"INT","not_null":"false"},"join_date":{"data_type":"DATE","not_null":"false"},"seniority":{"data_type":"INT","not_null":"false","script":{"sql":"DATE_DIFF(CURRENT_DATE, profile.join_date, DAY)","column":"profile.seniority","painless":"def param1 = ctx.profile?.join_date; def param2 = (param1 instanceof String ? LocalDate.parse((param1).replace(\"/\", \"-\"), DateTimeFormatter.ofPattern(\"yyyy-MM-dd\")).atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); def param4 = Long.valueOf(ChronoUnit.DAYS.between(LocalDate.from(param2), LocalDate.from(param3))); ctx.profile.seniority = (param1 == null) ? null : param4"}}}}},"type":"regular"}}""".stripMargin val indexSettings = schema.indexSettings println(indexSettings) indexSettings.toString shouldBe """{"index":{}}""" val pipeline = schema.defaultPipelineNode println(pipeline) - pipeline.toString shouldBe """{"description":"CREATE OR REPLACE PIPELINE users_ddl_default_pipeline WITH PROCESSORS (name SET DEFAULT 'anonymous', age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR)), ingested_at SET DEFAULT _ingest.timestamp, profile.seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY)), PARTITION BY birthdate (MONTH), PRIMARY KEY (id))","processors":[{"set":{"description":"name SET DEFAULT 'anonymous'","field":"name","ignore_failure":true,"value":"anonymous","if":"ctx.name == null"}},{"script":{"description":"age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR))","lang":"painless","source":"def param1 = ctx.birthdate; def param2 = (param1 instanceof String ? LocalDate.parse((param1).replace(\"/\", \"-\"), DateTimeFormatter.ofPattern(\"yyyy-MM-dd\")).atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); def param4 = Long.valueOf(ChronoUnit.YEARS.between(param2, param3)); ctx.age = (param1 == null) ? null : param4","ignore_failure":true}},{"set":{"description":"ingested_at SET DEFAULT _ingest.timestamp","field":"ingested_at","ignore_failure":true,"value":"{{_ingest.timestamp}}","if":"ctx.ingested_at == null"}},{"script":{"description":"profile.seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY))","lang":"painless","source":"def param1 = ctx.profile?.join_date; def param2 = (param1 instanceof String ? LocalDate.parse((param1).replace(\"/\", \"-\"), DateTimeFormatter.ofPattern(\"yyyy-MM-dd\")).atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); def param4 = Long.valueOf(ChronoUnit.DAYS.between(param2, param3)); ctx.profile.seniority = (param1 == null) ? null : param4","ignore_failure":true}},{"date_index_name":{"description":"PARTITION BY birthdate (MONTH)","field":"birthdate","date_rounding":"M","date_formats":["yyyy-MM"],"index_name_prefix":"users-","ignore_failure":true}},{"set":{"description":"PRIMARY KEY (id)","field":"_id","value":"{{id}}","ignore_failure":false,"ignore_empty_value":false}}]}""" + pipeline.toString shouldBe """{"description":"CREATE OR REPLACE PIPELINE users_ddl_default_pipeline WITH PROCESSORS (name SET DEFAULT 'anonymous', age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)), ingested_at SET DEFAULT _ingest.timestamp, profile.seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY)), PARTITION BY birthdate (MONTH), PRIMARY KEY (id))","processors":[{"set":{"description":"name SET DEFAULT 'anonymous'","field":"name","ignore_failure":true,"value":"anonymous","if":"ctx.name == null"}},{"script":{"description":"age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE))","lang":"painless","source":"def param1 = ctx.birthdate; def param2 = (param1 instanceof String ? LocalDate.parse((param1).replace(\"/\", \"-\"), DateTimeFormatter.ofPattern(\"yyyy-MM-dd\")).atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); def param4 = LocalDate.from(param3).atStartOfDay(ZoneId.of('Z')); def param5 = Long.valueOf((((param4.getYear() * 12L + param4.getMonthValue()) * 4294967296L + (param4.getDayOfMonth() - 1) * 86400000L + param4.getLong(ChronoField.MILLI_OF_DAY)) - ((param2.getYear() * 12L + param2.getMonthValue()) * 4294967296L + (param2.getDayOfMonth() - 1) * 86400000L + param2.getLong(ChronoField.MILLI_OF_DAY))) / 4294967296L / 12); ctx.age = (param1 == null) ? null : param5","ignore_failure":true}},{"set":{"description":"ingested_at SET DEFAULT _ingest.timestamp","field":"ingested_at","ignore_failure":true,"value":"{{_ingest.timestamp}}","if":"ctx.ingested_at == null"}},{"script":{"description":"profile.seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY))","lang":"painless","source":"def param1 = ctx.profile?.join_date; def param2 = (param1 instanceof String ? LocalDate.parse((param1).replace(\"/\", \"-\"), DateTimeFormatter.ofPattern(\"yyyy-MM-dd\")).atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); def param4 = Long.valueOf(ChronoUnit.DAYS.between(LocalDate.from(param2), LocalDate.from(param3))); ctx.profile.seniority = (param1 == null) ? null : param4","ignore_failure":true}},{"date_index_name":{"description":"PARTITION BY birthdate (MONTH)","field":"birthdate","date_rounding":"M","date_formats":["yyyy-MM"],"index_name_prefix":"users-","ignore_failure":true}},{"set":{"description":"PRIMARY KEY (id)","field":"_id","value":"{{id}}","ignore_failure":false,"ignore_empty_value":false}}]}""" // Reconstruct EsIndex val mappings = mapper.createObjectNode() mappings.set("mappings", indexMappings) @@ -2271,7 +2274,7 @@ class ParserSpec extends AnyFlatSpec with Matchers { | value = "anonymous" | ), | SCRIPT ( - | description = "age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR))", + | description = "age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE))", | lang = "painless", | source = "def param1 = ctx.birthdate; def param2 = ZonedDateTime.now(ZoneId.of('Z')).toLocalDate(); ctx.age = (param1 == null) ? null : Long.valueOf(ChronoUnit.YEARS.between(param1, param2))", | ignore_failure = true @@ -2284,7 +2287,7 @@ class ParserSpec extends AnyFlatSpec with Matchers { | value = "_ingest.timestamp" | ), | SCRIPT ( - | description = "profile.seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY))", + | description = "profile.seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY))", | lang = "painless", | source = "def param1 = ctx.profile?.join_date; def param2 = ZonedDateTime.now(ZoneId.of('Z')).toLocalDate(); ctx.profile.seniority = (param1 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param1, param2))", | ignore_failure = true @@ -2342,8 +2345,8 @@ class ParserSpec extends AnyFlatSpec with Matchers { case Some( ScriptProcessor( IngestPipelineType.Default, - Some("age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR))"), - "DATE_DIFF(birthdate, CURRENT_DATE, YEAR)", + Some("age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE))"), + "TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)", "age", SQLTypes.Int, source, @@ -2378,9 +2381,9 @@ class ParserSpec extends AnyFlatSpec with Matchers { ScriptProcessor( IngestPipelineType.Default, Some( - "profile.seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY))" + "profile.seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY))" ), - "DATE_DIFF(profile.join_date, CURRENT_DATE, DAY)", + "DATE_DIFF(CURRENT_DATE, profile.join_date, DAY)", "profile.seniority", SQLTypes.Int, source, diff --git a/sql/src/test/scala/app/softnetwork/elastic/sql/parser/TableauDialectSpec.scala b/sql/src/test/scala/app/softnetwork/elastic/sql/parser/TableauDialectSpec.scala index 5263e64ac..758d34b80 100644 --- a/sql/src/test/scala/app/softnetwork/elastic/sql/parser/TableauDialectSpec.scala +++ b/sql/src/test/scala/app/softnetwork/elastic/sql/parser/TableauDialectSpec.scala @@ -99,20 +99,21 @@ class TableauDialectSpec extends AnyFlatSpec with Matchers { ) } - /** MySQL 8.4 defines `TIMESTAMPDIFF(unit, dt1, dt2)` as `dt2 - dt1`, and `date_diff_transact_sql` - * binds `(unit, d1, d2)` to `DateDiff(start = d1, end = d2)`, i.e. `between(d1, d2)` — the same + /** MySQL 8.4 defines `TIMESTAMPDIFF(unit, dt1, dt2)` as `dt2 - dt1` in whole units elapsed, and + * `date_diff_transact_sql` binds `(unit, d1, d2)` to `DateDiff(start = d1, end = d2)` — the same * answer with the same sign. The MySQL/ODBC order is the one that matters, so it is the one * asserted. */ - "TIMESTAMPDIFF" should "be a spelling of DATE_DIFF, in the MySQL/ODBC unit-first order" in { - // 🔴 The ARGUMENT LIST, not just the function name: `include("DATE_DIFF(")` is satisfied by - // the sign-inverted render `DATE_DIFF(MONTH, end_date, start_date)` too, so it would prove - // nothing about the very binding this test exists to pin. + "TIMESTAMPDIFF" should "keep its name and the MySQL/ODBC unit-first order" in { + // 🔴 The ARGUMENT LIST, not just the function name: `include("TIMESTAMPDIFF(")` is satisfied by + // the sign-inverted render `TIMESTAMPDIFF(MONTH, end_date, start_date)` too, so it would prove + // nothing about the very binding this test exists to pin. The NAME is kept because it counts + // elapsed units where `DATE_DIFF` counts boundaries. canonicalises( "SELECT TIMESTAMPDIFF(MONTH, start_date, end_date) AS d FROM t", - "DATE_DIFF(MONTH, start_date, end_date)" + "TIMESTAMPDIFF(MONTH, start_date, end_date)" ) - canonicalises("SELECT TIMESTAMPDIFF(DAY, a, b) AS d FROM t", "DATE_DIFF(DAY, a, b)") + canonicalises("SELECT TIMESTAMPDIFF(DAY, a, b) AS d FROM t", "TIMESTAMPDIFF(DAY, a, b)") } it should "leave the bare unit names alone" in { diff --git a/sql/src/test/scala/app/softnetwork/elastic/sql/query/HavingAliasResolutionSpec.scala b/sql/src/test/scala/app/softnetwork/elastic/sql/query/HavingAliasResolutionSpec.scala index 2adff4078..997b74a93 100644 --- a/sql/src/test/scala/app/softnetwork/elastic/sql/query/HavingAliasResolutionSpec.scala +++ b/sql/src/test/scala/app/softnetwork/elastic/sql/query/HavingAliasResolutionSpec.scala @@ -179,7 +179,7 @@ class HavingAliasResolutionSpec extends AnyFlatSpec with Matchers { "DATE_DIFF(DAY, MAX(d) - INTERVAL 1 DAY, '2024-01-01')" ), having + "TIMESTAMPDIFF(DAY, MAX(d)::DATE, '2024-01-01') > 1" -> inline( - "DATE_DIFF(DAY, MAX(d)::DATE, '2024-01-01')" + "TIMESTAMPDIFF(DAY, MAX(d)::DATE, '2024-01-01')" ), "SELECT g, ISNULL(MAX(a)::DOUBLE) AS x FROM t GROUP BY g" -> chain, "SELECT g, ISNULL(MAX(d) - INTERVAL 1 DAY) AS x FROM t GROUP BY g" -> chain, diff --git a/sql/src/test/scala/app/softnetwork/elastic/sql/query/MaterializedViewHavingSpec.scala b/sql/src/test/scala/app/softnetwork/elastic/sql/query/MaterializedViewHavingSpec.scala index 6b214a87b..4aa65b69b 100644 --- a/sql/src/test/scala/app/softnetwork/elastic/sql/query/MaterializedViewHavingSpec.scala +++ b/sql/src/test/scala/app/softnetwork/elastic/sql/query/MaterializedViewHavingSpec.scala @@ -1225,16 +1225,18 @@ class MaterializedViewHavingSpec extends AnyFlatSpec with Matchers { // string literal as a String: each operand is converted to a date first, an aggregate ONCE (it // is the instant in UTC already). Both venues render ONE calculation (the item's context-free // rendering); the RUN is GroupByCompletenessSpec's. - def date(metric: String) = - s"Instant.ofEpochMilli(((long) params.$metric)).atZone(ZoneId.of('Z')).toLocalDate()" + def instant(metric: String) = + s"Instant.ofEpochMilli(((long) params.$metric)).atZone(ZoneId.of('Z'))" + def date(metric: String) = s"${instant(metric)}.toLocalDate()" val literal = """LocalDate.parse(("2024-01-01").replace("/", "-"), DateTimeFormatter.ofPattern("yyyy-MM-dd"))""" Seq( - // MySQL's DATEDIFF is `first - second`, every other spelling `end - start` + // the dates first are `first - second`, the unit first `last - middle`; DATEDIFF and + // DATE_DIFF count calendar days, TIMESTAMPDIFF whole days elapsed between two instants "DATEDIFF(MAX(d), '2024-01-01')" -> s"Long.valueOf(ChronoUnit.DAYS.between($literal, ${date("mx")}))", - "DATE_DIFF(MAX(d), MIN(d), DAY)" -> s"Long.valueOf(ChronoUnit.DAYS.between(${date("mx")}, ${date("mn")}))", + "DATE_DIFF(MAX(d), MIN(d), DAY)" -> s"Long.valueOf(ChronoUnit.DAYS.between(${date("mn")}, ${date("mx")}))", "TIMESTAMPDIFF(DAY, MAX(d), '2024-01-01')" -> - s"Long.valueOf(ChronoUnit.DAYS.between(${date("mx")}, $literal))" + s"Long.valueOf(ChronoUnit.DAYS.between(${instant("mx")}, $literal.atStartOfDay(ZoneId.of('Z'))))" ).foreach { case (expression, script) => val body = s"SELECT city, MAX(d) AS mx, MIN(d) AS mn, $expression AS x FROM customers GROUP BY city HAVING x > 1" diff --git a/sql/src/test/scala/app/softnetwork/elastic/sql/schema/IngestTemporalSpec.scala b/sql/src/test/scala/app/softnetwork/elastic/sql/schema/IngestTemporalSpec.scala index 944cd2945..1dac75888 100644 --- a/sql/src/test/scala/app/softnetwork/elastic/sql/schema/IngestTemporalSpec.scala +++ b/sql/src/test/scala/app/softnetwork/elastic/sql/schema/IngestTemporalSpec.scala @@ -14,8 +14,9 @@ import org.scalatest.prop.TableDrivenPropertyChecks * no `get(ChronoField)`, the script threw, `ignore_failure: true` swallowed the throw, and the * computed column was simply ABSENT from the stored document. Measured on real Elasticsearch for * `YEAR`, `MONTH`, `DATE_TRUNC`, `DATE_ADD`/`DATE_SUB`, `DATE_FORMAT` and `DATE_DIFF` — and - * `DATE_DIFF(birthdate, CURRENT_DATE, YEAR)` is a PUBLISHED example - * (`documentation/sql/ddl_statements.md`, and the REPL testkit's `users` table). + * `DATE_DIFF(birthdate, CURRENT_DATE, YEAR)` was a PUBLISHED example + * (`documentation/sql/ddl_statements.md`, and the REPL testkit's `users` table), published since + * 0.24.0 as `TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)` (pinned below). * * 🔴 Three separate things had to be true for that example to work, and only the first was known: * @@ -35,8 +36,9 @@ import org.scalatest.prop.TableDrivenPropertyChecks * `SQLTypeUtils.runtimeType` already applies to a query. * * ⚠️ Every emission below was executed as a real ingest pipeline on ES 6.8.23, 7.17.29, 8.18.3 AND - * 9.0.3, against both document shapes. `DATE_DIFF(birthdate, CURRENT_DATE, YEAR)` stores `age: 36` - * for `{"birthdate":"1990-05-20"}` and for `{"birthdate":643161600000}` on all four. + * 9.0.3, against both document shapes. `DATE_DIFF(birthdate, CURRENT_DATE, YEAR)` stored `age: 36` + * for `{"birthdate":"1990-05-20"}` and for `{"birthdate":643161600000}` on all four, before 0.24.0 + * aligned its direction and counting with the vendors' (see the published example's pin below). * * KNOWN AND UNCHANGED: a document MISSING the source field still leaves the computed column * absent. `instanceof` on null throws and `ignore_failure: true` swallows it, which is the same @@ -138,9 +140,10 @@ class IngestTemporalSpec extends AnyFlatSpec with Matchers with TableDrivenPrope } it should "keep the ZonedDateTime for CURRENT_DATE too" in { - // Narrowing to a `LocalDate` here is what made `ChronoUnit.YEARS.between` refuse the pair in - // the published DATE_DIFF example. A query is unaffected and still narrows -- asserted in - // ParserSpec, whose query-side pins did not move. + // Narrowing the clock's parameter to a `LocalDate` here is what made `ChronoUnit.YEARS.between` + // refuse the pair in the published DATE_DIFF example. The CALL reads each operand's calendar + // date with `LocalDate.from`, which takes a `ZonedDateTime` and a `LocalDate` alike, so both + // sides of `between` are one type whatever the parameter holds. val s = source( "CREATE TABLE t (birthdate DATE, age INTEGER SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR)))", "age" @@ -150,11 +153,13 @@ class IngestTemporalSpec extends AnyFlatSpec with Matchers with TableDrivenPrope } it should "emit the published example exactly as it was executed" in { - // The whole point, pinned end to end. This byte string was run as an ingest pipeline on ES - // 6.8.23, 7.17.29, 8.18.3 and 9.0.3: `{"birthdate":"1990-05-20"}` and - // `{"birthdate":643161600000}` both store `age: 36`. + // The whole point, pinned end to end. The published age is `TIMESTAMPDIFF(YEAR, birthdate, + // CURRENT_DATE)`: the whole years elapsed from `birthdate` to the start of `CURRENT_DATE` + // (UTC), which the processor holds as the current instant and reads as a date. Run as an + // ingest pipeline on ES 8.18.3, `{"birthdate":"1990-05-20"}` and `{"birthdate":643161600000}` + // both store that age. source( - "CREATE TABLE t (birthdate DATE, age INTEGER SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR)))", + "CREATE TABLE t (birthdate DATE, age INTEGER SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)))", "age" ) shouldBe "def param1 = ctx.birthdate; " + @@ -163,8 +168,12 @@ class IngestTemporalSpec extends AnyFlatSpec with Matchers with TableDrivenPrope ".atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); " + "def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), " + "ZoneId.of('Z')); " + - "def param4 = Long.valueOf(ChronoUnit.YEARS.between(param2, param3)); " + - "ctx.age = (param1 == null) ? null : param4" + "def param4 = LocalDate.from(param3).atStartOfDay(ZoneId.of('Z')); " + + "def param5 = Long.valueOf((((param4.getYear() * 12L + param4.getMonthValue()) * 4294967296L " + + "+ (param4.getDayOfMonth() - 1) * 86400000L + param4.getLong(ChronoField.MILLI_OF_DAY)) - " + + "((param2.getYear() * 12L + param2.getMonthValue()) * 4294967296L + (param2.getDayOfMonth() " + + "- 1) * 86400000L + param2.getLong(ChronoField.MILLI_OF_DAY))) / 4294967296L / 12); " + + "ctx.age = (param1 == null) ? null : param5" } /** 🔴 Issue #384 — a COMPARISON inside a computed column must not be given the query path's diff --git a/testkit/src/main/scala/app/softnetwork/elastic/client/BiDialectExecutionSpec.scala b/testkit/src/main/scala/app/softnetwork/elastic/client/BiDialectExecutionSpec.scala index 7870a28e8..08424fc44 100644 --- a/testkit/src/main/scala/app/softnetwork/elastic/client/BiDialectExecutionSpec.scala +++ b/testkit/src/main/scala/app/softnetwork/elastic/client/BiDialectExecutionSpec.scala @@ -220,26 +220,44 @@ trait BiDialectExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKit w ) shouldBe -31L } - it should "stay the opposite of DATE_DIFF, which keeps BigQuery's order" in { - val mysql = longAt( + /** Every vendor computes `end - start`; with the dates first the END comes first (MySQL's + * `DATEDIFF(expr1, expr2)`, BigQuery's `DATE_DIFF(end_date, start_date, part)`), with the unit + * first it comes last. Until 0.24.0 the dates-first forms WITH a unit answered the opposite + * sign, on a misreading of BigQuery's signature (issue #363). + */ + it should "agree with DATE_DIFF, the dates first being first - second at every arity" in { + Seq( + s"SELECT DATEDIFF('2025-01-10'::DATE, '2025-01-01'::DATE) AS n FROM $index LIMIT 1", + s"SELECT DATEDIFF('2025-01-10'::DATE, '2025-01-01'::DATE, DAY) AS n FROM $index LIMIT 1", + s"SELECT DATE_DIFF('2025-01-10'::DATE, '2025-01-01'::DATE, DAY) AS n FROM $index LIMIT 1" + ).foreach(sql => withClue(s"[$sql] ")(longAt(rowsOf(sql).head, "n") shouldBe 9L)) + // BigQuery's own documented example. + longAt( rowsOf( - s"SELECT DATEDIFF('2025-01-10'::DATE, '2025-01-01'::DATE) AS n FROM $index LIMIT 1" + s"SELECT DATE_DIFF('2010-07-07'::DATE, '2008-12-25'::DATE, DAY) AS n FROM $index LIMIT 1" ).head, "n" - ) - val bigQuery = longAt( + ) shouldBe 559L + } + + it should "count the calendar boundaries crossed, on SQL Server's and DuckDB's documented examples" in { + // SQL Server: one year boundary between the last instant of 2005 and the first of 2006. + longAt( rowsOf( - s"SELECT DATE_DIFF('2025-01-10'::DATE, '2025-01-01'::DATE, DAY) AS n FROM $index LIMIT 1" + s"SELECT DATEDIFF(YEAR, '2005-12-31 23:59:59.9999999', '2006-01-01 00:00:00.0000000') AS n FROM $index LIMIT 1" ).head, "n" - ) - mysql shouldBe 9L - bigQuery shouldBe -9L - mysql shouldBe -bigQuery + ) shouldBe 1L + // DuckDB's date_diff: two month boundaries, though not two whole months. + longAt( + rowsOf( + s"SELECT DATEDIFF(MONTH, '1992-09-15'::DATE, '1992-11-14'::DATE) AS n FROM $index LIMIT 1" + ).head, + "n" + ) shouldBe 2L } - it should "agree with TIMESTAMPDIFF, which MySQL defines as dt2 - dt1" in { - // MySQL is itself asymmetric here, and we match it on BOTH names. + it should "agree with TIMESTAMPDIFF on the sign, which MySQL defines as dt2 - dt1" in { longAt( rowsOf( s"SELECT TIMESTAMPDIFF(DAY, '2025-01-01'::DATE, '2025-01-10'::DATE) AS n FROM $index LIMIT 1" @@ -248,24 +266,18 @@ trait BiDialectExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKit w ) shouldBe 9L } - /** The lead ruling, executed: on this ONE spelling the arity changes the sign, because the - * 3-argument form is this engine's own extension and is not MySQL. - */ - it should "keep this engine's order in the three-argument form, unlike the two-argument one" in { - val two = longAt( - rowsOf( - s"SELECT DATEDIFF('2025-01-10'::DATE, '2025-01-01'::DATE) AS n FROM $index LIMIT 1" - ).head, - "n" - ) - val three = longAt( - rowsOf( - s"SELECT DATEDIFF('2025-01-10'::DATE, '2025-01-01'::DATE, DAY) AS n FROM $index LIMIT 1" - ).head, - "n" - ) - two shouldBe 9L - three shouldBe -9L + "TIMESTAMPDIFF" should "count the whole units elapsed, on MySQL's own documented examples" in { + Seq( + "TIMESTAMPDIFF(MONTH, '2003-02-01', '2003-05-01')" -> 3L, + "TIMESTAMPDIFF(YEAR, '2002-05-01', '2001-01-01')" -> -1L, + "TIMESTAMPDIFF(MINUTE, '2003-02-01', '2003-05-01 12:05:55')" -> 128885L, + // two month boundaries, one whole month: the counting DATEDIFF does not share + "TIMESTAMPDIFF(MONTH, '1992-09-15', '1992-11-14')" -> 1L + ).foreach { case (expression, expected) => + withClue(s"[$expression] ") { + longAt(rowsOf(s"SELECT $expression AS n FROM $index LIMIT 1").head, "n") shouldBe expected + } + } } "TIMESTAMPADD" should "execute as DATETIME_ADD does, on the same statement" in { diff --git a/testkit/src/main/scala/app/softnetwork/elastic/client/DateFunctionExecutionSpec.scala b/testkit/src/main/scala/app/softnetwork/elastic/client/DateFunctionExecutionSpec.scala index 86f825cc6..fac73fe92 100644 --- a/testkit/src/main/scala/app/softnetwork/elastic/client/DateFunctionExecutionSpec.scala +++ b/testkit/src/main/scala/app/softnetwork/elastic/client/DateFunctionExecutionSpec.scala @@ -115,6 +115,9 @@ trait DateFunctionExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKi override def afterAll(): Unit = { client.deleteIndex(index) if (diffIndexCreated) client.deleteIndex(diffIndex) + createdTables.foreach(t => + Try(Await.result(client.run(s"DROP TABLE IF EXISTS $t"), 60.seconds)) + ) super.afterAll() } @@ -207,24 +210,29 @@ trait DateFunctionExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKi } // ----------------------------------------------------------------------------------------------- - // The DATEDIFF family against its documented semantics, computed HERE: + // The DATEDIFF family against the vendors' definitions, computed HERE, independently of the + // engine: // - // - HOUR, MINUTE and SECOND count the ELAPSED whole units between two instants (UTC), truncated - // toward zero; a DATE operand is the start of its day; - // - DAY and every larger unit compare two CALENDAR dates (UTC): days, weeks (days / 7), months - // (complete once the day of month is reached), quarters (months / 3), years (months / 12), - // truncated toward zero; - // - the answer is `second - first`, except MySQL's two-argument DATEDIFF: `first - second`; - // - a string literal is the temporal it spells, a DATE alone or a TIMESTAMP (UTC unless it names - // a zone); NULL when an operand has no value. + // - DIRECTION, by layout: `end - start`, where the dates first (`DATE_DIFF(a, b[, unit])`, + // `DATEDIFF(a, b[, unit])`, `TIMESTAMPDIFF(a, b[, unit])`) is `a - b` and the unit first + // (`DATE_DIFF(unit, a, b)`, `DATEDIFF(unit, a, b)`, `TIMESTAMPDIFF(unit, a, b)`) is `b - a`; + // - COUNTING, by name: `DATEDIFF` and `DATE_DIFF` count the calendar BOUNDARIES crossed, the + // difference of two UTC calendar-unit indices, a week starting on the ISO Monday; + // `TIMESTAMPDIFF` counts the whole units ELAPSED, truncated toward zero, a month (quarter, + // year) once the same day of month and time of day are reached; + // - a DATE operand is the start of its day (UTC); a string literal is the temporal it spells, a + // DATE alone or a TIMESTAMP (UTC unless it names a zone); NULL when an operand has no value. // - // The population is every spelling (DATE_DIFF / DATEDIFF with the unit last or first, - // TIMESTAMPDIFF, DATE_DIFF without a unit, MySQL's DATEDIFF), every unit, every operand kind (a - // date column, a timestamp column, a DATE literal, two TIMESTAMP literals, a date and a timestamp - // aggregate) and every venue: a SELECT item per document, a WHERE, a SELECT item per group and a - // HAVING through that item's alias. Before, HOUR / MINUTE / SECOND over a column or an aggregate - // failed with `Unsupported unit: Hours`, QUARTER did not compile, and a literal failed at row - // level and, with a time of day, per group. + // The fixture sits where the two countings and the two week starts part: one second (200 ms for + // a second) across a year, quarter, month, ISO week, day, hour, minute and second end; a Saturday + // to Sunday second, which is no ISO week; a span with more boundaries than whole units in every + // unit; the vendors' own 1992-09-15 -> 1992-11-14 (two month boundaries, one month elapsed) and + // 2003-02-01 -> 2003-05-01 (three months); the same instant; a DATE against a TIMESTAMP. The + // population is every spelling and unit over every ORDERED operand pair -- so every edge in both + // directions -- of a date column, a timestamp column, a DATE literal, two TIMESTAMP literals (one + // with a zone and milliseconds) and a TIMESTAMP-typed one, and of a date and a timestamp aggregate, + // in every venue: a SELECT item per document, a WHERE, a SELECT item per group and a HAVING + // through that item's alias. // ----------------------------------------------------------------------------------------------- private val diffIndex = "date_diff_semantics" @@ -233,18 +241,37 @@ trait DateFunctionExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKi private type DiffDoc = (String, String, Option[String], Option[String]) - /** (id, group, `d` a calendar date, `ts` an instant); `None`: the document lacks it. Group g1 has - * no value at all, g4 has some. + /** (id, group, `d` a calendar date, `ts` an instant); `None`: the document lacks it. Group g01 + * has no value at all, g15 several documents with some, g18 a date and no instant; every other + * group is one document, so its aggregates are that document's edge. */ private val diffDocs: Seq[DiffDoc] = Seq( - ("r1", "g1", None, None), - ("r2", "g1", None, None), - ("r3", "g2", Some("2024-01-02"), Some("2024-01-01T23:30:00Z")), - ("r4", "g3", Some("2023-11-15"), Some("2024-01-01T10:30:15Z")), - ("r5", "g3", Some("2024-02-29"), Some("2024-03-31T00:00:45Z")), - ("r6", "g4", Some("2024-01-31"), None), - ("r7", "g4", None, Some("2023-12-31T23:59:59Z")), - ("r8", "g4", Some("2025-06-15"), Some("2024-06-14T12:00:00Z")) + ("r01", "g01", None, None), + ("r02", "g01", None, None), + // one second before the date: a year (Tuesday to Wednesday), a quarter (Monday to Tuesday), a + // month (Friday to Saturday), an ISO week (Sunday to Monday), a Sunday that starts no ISO week + // (Saturday to Sunday), a day (Wednesday to Thursday) + ("r03", "g03", Some("2025-01-01"), Some("2024-12-31T23:59:59Z")), + ("r04", "g04", Some("2025-04-01"), Some("2025-03-31T23:59:59Z")), + ("r05", "g05", Some("2025-02-01"), Some("2025-01-31T23:59:59Z")), + ("r06", "g06", Some("2025-01-13"), Some("2025-01-12T23:59:59Z")), + ("r07", "g07", Some("2025-01-12"), Some("2025-01-11T23:59:59Z")), + ("r08", "g08", Some("2025-01-16"), Some("2025-01-15T23:59:59Z")), + // one second before an hour ('2025-01-15 11:00:00'); 200 ms across a minute and a second + // ('2025-01-15T10:01:00.100Z') + ("r09", "g09", Some("2025-01-15"), Some("2025-01-15T10:59:59Z")), + ("r10", "g10", None, Some("2025-01-15T10:00:59.900Z")), + // more boundaries than whole units, in every unit from SECOND to YEAR + ("r11", "g11", Some("2025-01-02"), Some("2023-11-30T22:59:30.900Z")), + // DuckDB's date_diff example, then MySQL's TIMESTAMPDIFF one + ("r12", "g12", Some("1992-09-15"), Some("1992-11-14T00:00:00Z")), + ("r13", "g13", Some("2003-02-01"), Some("2003-05-01T00:00:00Z")), + // the same instant + ("r14", "g14", Some("2025-01-15"), Some("2025-01-15T00:00:00Z")), + ("r15", "g15", Some("2025-06-15"), None), + ("r16", "g15", None, Some("2025-06-14T12:00:00Z")), + ("r17", "g15", Some("2024-01-31"), Some("2025-07-01T00:00:00Z")), + ("r18", "g18", Some("2024-02-29"), None) ) private lazy val diffIndexLoaded: Unit = { @@ -286,9 +313,11 @@ trait DateFunctionExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKi DiffOperand(sql, aggregate = false)(_ => Some(value)) private val diffLiterals: Seq[DiffOperand] = Seq( - constant("'2024-01-01'", OnDate(LocalDate.parse("2024-01-01"))), - constant("'2024-01-01 10:00:00'", AtInstant(Instant.parse("2024-01-01T10:00:00Z"))), - constant("'2023-12-31T22:15:30Z'", AtInstant(Instant.parse("2023-12-31T22:15:30Z"))) + constant("'2025-01-01'", OnDate(LocalDate.parse("2025-01-01"))), + constant("'2025-01-15 11:00:00'", AtInstant(Instant.parse("2025-01-15T11:00:00Z"))), + constant("'2025-01-15T10:01:00.100Z'", AtInstant(Instant.parse("2025-01-15T10:01:00.100Z"))), + // a typed literal, whose time of day a calendar unit must not count + constant("'2024-12-31 23:59:59'::TIMESTAMP", AtInstant(Instant.parse("2024-12-31T23:59:59Z"))) ) private val diffRowOperands: Seq[DiffOperand] = Seq( @@ -311,59 +340,110 @@ trait DateFunctionExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKi ) ) ++ diffLiterals - private def utcDate(value: DiffValue): LocalDate = value match { - case OnDate(date) => date - case AtInstant(instant) => instant.atZone(ZoneOffset.UTC).toLocalDate + /** The instant an operand denotes: a DATE is the start of its day, in UTC. */ + private def instantOf(value: DiffValue): Instant = value match { + case OnDate(date) => date.atStartOfDay(ZoneOffset.UTC).toInstant + case AtInstant(instant) => instant } - private def utcSeconds(value: DiffValue): Long = value match { - case OnDate(date) => date.toEpochDay * 86400L - case AtInstant(instant) => instant.getEpochSecond + /** The UTC calendar unit an instant falls in, as an index: two indices differ by the boundaries + * crossed between them. 1970-01-01, epoch day 0, is a Thursday, so `epochDay + 3` counts the + * days from the Monday before it. + */ + private def unitIndex(unit: String, value: DiffValue): Long = { + val instant = instantOf(value) + val utc = instant.atOffset(ZoneOffset.UTC) + unit match { + case "YEAR" => utc.getYear.toLong + case "QUARTER" => utc.getYear * 4L + (utc.getMonthValue - 1) / 3 + case "MONTH" => utc.getYear * 12L + utc.getMonthValue - 1 + case "WEEK" => Math.floorDiv(utc.toLocalDate.toEpochDay + 3, 7L) + case "DAY" => utc.toLocalDate.toEpochDay + case "HOUR" => Math.floorDiv(instant.toEpochMilli, 3600000L) + case "MINUTE" => Math.floorDiv(instant.toEpochMilli, 60000L) + case "SECOND" => Math.floorDiv(instant.toEpochMilli, 1000L) + } } - /** The complete months from `start` to `end`: a month counts once its day of month is reached. */ - private def wholeMonths(start: LocalDate, end: LocalDate): Long = { - def packed(date: LocalDate): Long = - (date.getYear * 12L + date.getMonthValue - 1) * 32L + date.getDayOfMonth - (packed(end) - packed(start)) / 32L + /** The week a SUNDAY-start count would put an instant in (1970-01-04 is a Sunday). */ + private def sundayWeek(value: DiffValue): Long = + Math.floorDiv(instantOf(value).atOffset(ZoneOffset.UTC).toLocalDate.toEpochDay + 4, 7L) + + /** The calendar boundaries crossed from `start` to `end`. */ + private def boundaries(unit: String, start: DiffValue, end: DiffValue): Long = + unitIndex(unit, end) - unitIndex(unit, start) + + /** The whole months from `from` to `to`, `from` not after `to`: a month counts once `to` reaches + * `from`'s day of month and time of day. + */ + private def wholeMonths(from: Instant, to: Instant): Long = { + val (a, b) = (from.atOffset(ZoneOffset.UTC), to.atOffset(ZoneOffset.UTC)) + val months = (b.getYear * 12L + b.getMonthValue) - (a.getYear * 12L + a.getMonthValue) + val reached = b.getDayOfMonth > a.getDayOfMonth || + (b.getDayOfMonth == a.getDayOfMonth && !b.toLocalTime.isBefore(a.toLocalTime)) + if (reached) months else months - 1 } - /** `end - start` in `unit`, truncated toward zero (Scala's `Long` division). */ - private def documentedDiff(unit: String, start: DiffValue, end: DiffValue): Long = unit match { - case "SECOND" => utcSeconds(end) - utcSeconds(start) - case "MINUTE" => (utcSeconds(end) - utcSeconds(start)) / 60L - case "HOUR" => (utcSeconds(end) - utcSeconds(start)) / 3600L - case "DAY" => utcDate(end).toEpochDay - utcDate(start).toEpochDay - case "WEEK" => (utcDate(end).toEpochDay - utcDate(start).toEpochDay) / 7L - case "MONTH" => wholeMonths(utcDate(start), utcDate(end)) - case "QUARTER" => wholeMonths(utcDate(start), utcDate(end)) / 3L - case "YEAR" => wholeMonths(utcDate(start), utcDate(end)) / 12L + /** The whole units elapsed from `start` to `end`, truncated toward zero. */ + private def elapsed(unit: String, start: DiffValue, end: DiffValue): Long = { + val (from, to) = (instantOf(start), instantOf(end)) + if (to.isBefore(from)) -elapsed(unit, end, start) + else { + val millis = to.toEpochMilli - from.toEpochMilli + unit match { + case "YEAR" => wholeMonths(from, to) / 12L + case "QUARTER" => wholeMonths(from, to) / 3L + case "MONTH" => wholeMonths(from, to) + case "WEEK" => millis / 604800000L + case "DAY" => millis / 86400000L + case "HOUR" => millis / 3600000L + case "MINUTE" => millis / 60000L + case "SECOND" => millis / 1000L + } + } } - /** A spelling of the family: its SQL over two operands, its unit, and whether it is MySQL's - * two-argument DATEDIFF (`first - second`). + /** A spelling of the family: its SQL over two operands in WRITTEN order, its unit, its layout + * (the dates first: `first - second`; the unit first: `last - middle`) and its counting. */ - private final case class DiffSpelling(unit: String, mysql: Boolean)( + private final case class DiffSpelling(unit: String, datesFirst: Boolean, countsElapsed: Boolean)( val sql: (String, String) => String ) + /** What a spelling answers for its two operands, in written order. */ + private def answer(spelling: DiffSpelling, first: DiffValue, second: DiffValue): Long = { + val (start, end) = if (spelling.datesFirst) (second, first) else (first, second) + if (spelling.countsElapsed) elapsed(spelling.unit, start, end) + else boundaries(spelling.unit, start, end) + } + private val diffUnits: Seq[String] = Seq("YEAR", "QUARTER", "MONTH", "WEEK", "DAY", "HOUR", "MINUTE", "SECOND") private val diffSpellings: Seq[DiffSpelling] = diffUnits.flatMap { u => Seq( - DiffSpelling(u, mysql = false)((a, b) => s"DATE_DIFF($a, $b, $u)"), - DiffSpelling(u, mysql = false)((a, b) => s"DATEDIFF($a, $b, $u)"), - DiffSpelling(u, mysql = false)((a, b) => s"DATE_DIFF($u, $a, $b)"), - DiffSpelling(u, mysql = false)((a, b) => s"DATEDIFF($u, $a, $b)"), - DiffSpelling(u, mysql = false)((a, b) => s"TIMESTAMPDIFF($u, $a, $b)") + DiffSpelling(u, datesFirst = true, countsElapsed = false)((a, b) => s"DATE_DIFF($a, $b, $u)"), + DiffSpelling(u, datesFirst = true, countsElapsed = false)((a, b) => s"DATEDIFF($a, $b, $u)"), + DiffSpelling(u, datesFirst = false, countsElapsed = false)((a, b) => + s"DATE_DIFF($u, $a, $b)" + ), + DiffSpelling(u, datesFirst = false, countsElapsed = false)((a, b) => s"DATEDIFF($u, $a, $b)"), + DiffSpelling(u, datesFirst = false, countsElapsed = true)((a, b) => + s"TIMESTAMPDIFF($u, $a, $b)" + ) ) } ++ Seq( - DiffSpelling("DAY", mysql = false)((a, b) => s"DATE_DIFF($a, $b)"), - DiffSpelling("DAY", mysql = true)((a, b) => s"DATEDIFF($a, $b)") + DiffSpelling("DAY", datesFirst = true, countsElapsed = false)((a, b) => s"DATE_DIFF($a, $b)"), + DiffSpelling("DAY", datesFirst = true, countsElapsed = false)((a, b) => s"DATEDIFF($a, $b)") + ) ++ diffUnits.map { u => + // No vendor writes `TIMESTAMPDIFF` with the dates first, and it is accepted all the same: the + // two rules give it its meaning, `first - second` counted in whole units elapsed. + DiffSpelling(u, datesFirst = true, countsElapsed = true)((a, b) => s"TIMESTAMPDIFF($a, $b, $u)") + } :+ DiffSpelling("DAY", datesFirst = true, countsElapsed = true)((a, b) => + s"TIMESTAMPDIFF($a, $b)" ) - /** A statement, the venue it runs in, and the documented answer for each document or group. */ + /** A statement, the venue it runs in, and the answer for each document or group. */ private final case class DiffStatement( sql: String, venue: String, @@ -374,26 +454,31 @@ trait DateFunctionExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKi def kept: Set[String] = expected.collect { case (k, Some(v)) if v > 0 => k }.toSet } + private val diffPerDocument: Seq[(String, Seq[DiffDoc])] = diffDocs.map(doc => doc._1 -> Seq(doc)) + + private val diffPerGroup: Seq[(String, Seq[DiffDoc])] = diffDocs.groupBy(_._2).toSeq.sortBy(_._1) + + private val diffRowPairs: Seq[(DiffOperand, DiffOperand)] = + for (a <- diffRowOperands; b <- diffRowOperands) yield (a, b) + + private val diffGroupPairs: Seq[(DiffOperand, DiffOperand)] = + for (a <- diffGroupOperands; b <- diffGroupOperands if a.aggregate || b.aggregate) + yield (a, b) + private lazy val diffPopulation: Seq[DiffStatement] = { - val perDocument: Seq[(String, Seq[DiffDoc])] = diffDocs.map(doc => doc._1 -> Seq(doc)) - val perGroup: Seq[(String, Seq[DiffDoc])] = diffDocs.groupBy(_._2).toSeq.sortBy(_._1) - val rowPairs = for (a <- diffRowOperands; b <- diffRowOperands) yield (a, b) // A WHERE over two literals reads no column: it is a constant predicate, rendered as a bare // comparison of the function's boxed result, which fails to compile whatever the function is // (`WHERE ABS(-3) > 0` too). It is not a DATEDIFF question, so it is left out. - val wherePairs = rowPairs.filterNot { case (a, b) => + val wherePairs = diffRowPairs.filterNot { case (a, b) => diffLiterals.contains(a) && diffLiterals.contains(b) } - val groupPairs = - for (a <- diffGroupOperands; b <- diffGroupOperands if a.aggregate || b.aggregate) - yield (a, b) for { spelling <- diffSpellings (venue, pairs, scopes) <- Seq( - ("SELECT", rowPairs, perDocument), - ("WHERE", wherePairs, perDocument), - ("GROUP", groupPairs, perGroup), - ("HAVING", groupPairs, perGroup) + ("SELECT", diffRowPairs, diffPerDocument), + ("WHERE", wherePairs, diffPerDocument), + ("GROUP", diffGroupPairs, diffPerGroup), + ("HAVING", diffGroupPairs, diffPerGroup) ) (a, b) <- pairs } yield { @@ -408,14 +493,22 @@ trait DateFunctionExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKi key -> (for { first <- a.value(docs) second <- b.value(docs) - } yield - if (spelling.mysql) documentedDiff(spelling.unit, second, first) - else documentedDiff(spelling.unit, first, second)) + } yield answer(spelling, first, second)) }.toMap DiffStatement(sql, venue, spelling.unit, expected) } } + /** Every pair of operand values the population computes over, in written order. */ + private lazy val diffValuePairs: Seq[(DiffValue, DiffValue)] = + (for { + (pairs, scopes) <- Seq(diffRowPairs -> diffPerDocument, diffGroupPairs -> diffPerGroup) + (a, b) <- pairs + (_, docs) <- scopes + first <- a.value(docs) + second <- b.value(docs) + } yield (first, second)).distinct + /** The rows of a statement through the gateway, or why there are none: a streamed result fails * while it is consumed, so the whole read is guarded. */ @@ -443,13 +536,13 @@ trait DateFunctionExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKi } } - "the DATEDIFF family" should "answer its documented semantics for every spelling, unit, operand and venue" in { + "the DATEDIFF family" should "answer the vendors' definitions for every spelling, unit, edge, operand and venue" in { diffIndexLoaded val population = diffPopulation // Non-vacuity, computed over the material: every spelling in every venue over every operand // pair, every unit answering a negative, a zero and a positive somewhere, NULL in every venue, // and the filters keeping some and dropping some. - population.size shouldBe diffSpellings.size * (25 + 16 + 16 + 16) + population.size shouldBe diffSpellings.size * (36 + 20 + 20 + 20) population.foreach(st => withClue(s"[${st.sql}] ")(Parser(st.sql).isRight shouldBe true)) diffUnits.foreach { u => val answers = population.filter(_.unit == u).flatMap(_.expected.values.flatten) @@ -468,6 +561,27 @@ trait DateFunctionExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKi .filter(_.filters) .exists(st => st.kept.nonEmpty && st.kept != st.expected.keySet) shouldBe true + // ...and the edges are there, in both directions: in every unit a boundary crossed with no + // whole unit elapsed, and more boundaries than whole units; a Sunday that a Sunday-start week + // would count and an ISO week does not; the same instant written as a DATE and a TIMESTAMP. + val pairs = diffValuePairs + diffUnits.foreach { u => + withClue(s"$u edges: ") { + Seq( + pairs.exists { case (x, y) => boundaries(u, x, y) == 1 && elapsed(u, x, y) == 0 }, + pairs.exists { case (x, y) => boundaries(u, x, y) == -1 && elapsed(u, x, y) == 0 }, + pairs.exists { case (x, y) => + elapsed(u, x, y) > 0 && boundaries(u, x, y) > elapsed(u, x, y) + } + ) shouldBe Seq(true, true, true) + } + } + pairs.exists { case (x, y) => + boundaries("WEEK", x, y) == 0 && sundayWeek(y) - sundayWeek(x) == 1 + } shouldBe true + pairs.exists { case (x, y) => + x.isInstanceOf[OnDate] && y.isInstanceOf[AtInstant] && instantOf(x) == instantOf(y) + } shouldBe true val outcomes: Seq[(DiffStatement, Either[String, String])] = population.map { st => st -> diffRows(st.sql).flatMap { rows => @@ -511,4 +625,393 @@ trait DateFunctionExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKi errors shouldBe empty } } + + // ----------------------------------------------------------------------------------------------- + // TIMESTAMPDIFF over an operand of ANOTHER runtime type. The function compares instants, and an + // operand's declared type does not say which Java type it holds: `'…'::DATETIME` is a + // `LocalDateTime`, `CURRENT_DATE` and `ts::DATE` a `LocalDate`, `NOW()` a `ZonedDateTime`, and + // `COALESCE(ts, CURRENT_DATE)` or a CASE over `CURRENT_DATE` hold one type on one row and another + // on the next. Each is the instant it denotes, in UTC (a DATE is the start of its day), in every + // venue, for every unit and in both directions. The clock is read where the statement reads it: + // a query's when its request is built, so the instants that bracket the call bound it; a computed + // column's on the node at ingest, bounded the same way with a margin for the two clocks. + // ----------------------------------------------------------------------------------------------- + + /** The tables the DDL tests below create, dropped after the suite. */ + @volatile private var createdTables: List[String] = Nil + + /** An operand whose value may read the clock: over a document or a group, at `now`. */ + private final case class ClockOperand(sql: String, aggregate: Boolean)( + val value: (Seq[DiffDoc], Instant) => Option[DiffValue] + ) + + private def utcDate(instant: Instant): LocalDate = instant.atOffset(ZoneOffset.UTC).toLocalDate + + private def clockFree(operand: DiffOperand): ClockOperand = + ClockOperand(operand.sql, operand.aggregate)((docs, _) => operand.value(docs)) + + private val datetimeLiteral: ClockOperand = + ClockOperand("'2025-01-20 11:00:00'::DATETIME", aggregate = false)((_, _) => + Some(AtInstant(Instant.parse("2025-01-20T11:00:00Z"))) + ) + + /** A `LocalDate`, read without the clock. */ + private val dateLiteral: ClockOperand = + ClockOperand("'2025-06-01'::DATE", aggregate = false)((_, _) => + Some(OnDate(LocalDate.parse("2025-06-01"))) + ) + + private val currentDate: ClockOperand = + ClockOperand("CURRENT_DATE", aggregate = false)((_, now) => Some(OnDate(utcDate(now)))) + + private val currentInstant: ClockOperand = + ClockOperand("NOW()", aggregate = false)((_, now) => Some(AtInstant(now))) + + /** A `ZonedDateTime` on a document with `ts`, a `LocalDate` on one without. */ + private val coalesced: ClockOperand = + ClockOperand("COALESCE(ts, CURRENT_DATE)", aggregate = false)((docs, now) => + Some(docs.head._4.map(x => AtInstant(Instant.parse(x))).getOrElse(OnDate(utcDate(now)))) + ) + + /** A `LocalDate`: the clock's on one document, a literal's on the others. */ + private val chosen: ClockOperand = + ClockOperand( + "CASE WHEN id = 'r03' THEN CURRENT_DATE ELSE '2025-06-01'::DATE END", + aggregate = false + )((docs, now) => + Some(OnDate(if (docs.head._1 == "r03") utcDate(now) else LocalDate.parse("2025-06-01"))) + ) + + /** A column narrowed to its calendar date: a `LocalDate`. */ + private val castToDate: ClockOperand = + ClockOperand("ts::DATE", aggregate = false)((docs, _) => + docs.head._4.map(x => OnDate(utcDate(Instant.parse(x)))) + ) + + private val runtimeColumns: Seq[ClockOperand] = + diffRowOperands.filter(o => o.sql == "d" || o.sql == "ts").map(clockFree) + + private val runtimeAggregates: Seq[ClockOperand] = + diffGroupOperands.filter(_.aggregate).map(clockFree) + + /** Each anchor against each operand, both ways round. */ + private def runtimePairs( + anchors: Seq[ClockOperand], + operands: Seq[ClockOperand] + ): Seq[(ClockOperand, ClockOperand)] = + for { + anchor <- anchors + operand <- operands + pair <- Seq(anchor -> operand, operand -> anchor) + } yield pair + + /** A statement, the venue it runs in, and the answer for each document or group at `now`. */ + private final case class RuntimeStatement(sql: String, venue: String, unit: String)( + val expected: Instant => Map[String, Option[Long]] + ) + + private lazy val runtimePopulation: Seq[RuntimeStatement] = { + // Per group the operands are constants, whatever function reads them: a COALESCE over an + // aggregate reads its raw metric, a CASE over one does not parse, and a `bucket_script` is + // sent without its script parameters, so the request clock (`params.__now__`) is null there + // and `CURRENT_DATE` and `NOW()` cannot be read. + val rowPairs = runtimePairs( + runtimeColumns, + Seq(datetimeLiteral, currentDate, currentInstant, coalesced, chosen, castToDate) + ) + val groupPairs = runtimePairs(runtimeAggregates, Seq(datetimeLiteral, dateLiteral)) + for { + unit <- diffUnits + (venue, pairs, scopes) <- Seq( + ("SELECT", rowPairs, diffPerDocument), + ("WHERE", rowPairs, diffPerDocument), + ("GROUP", groupPairs, diffPerGroup), + ("HAVING", groupPairs, diffPerGroup) + ) + (a, b) <- pairs + } yield { + val e = s"TIMESTAMPDIFF($unit, ${a.sql}, ${b.sql})" + val sql = venue match { + case "SELECT" => s"SELECT id, $e AS x FROM $diffIndex" + case "WHERE" => s"SELECT id FROM $diffIndex WHERE $e > 0" + case "GROUP" => s"SELECT g, $e AS x FROM $diffIndex GROUP BY g" + case _ => s"SELECT g, $e AS x FROM $diffIndex GROUP BY g HAVING x > 0" + } + RuntimeStatement(sql, venue, unit)(now => + scopes.map { case (key, docs) => + key -> (for { + start <- a.value(docs, now) + end <- b.value(docs, now) + } yield elapsed(unit, start, end)) + }.toMap + ) + } + } + + /** The milliseconds clock the engine reads, as an instant. */ + private def clockNow(): Instant = Instant.ofEpochMilli(System.currentTimeMillis()) + + /** Does `actual` lie between the answers at the two instants that bracket the statement? An + * answer that does not read the clock is the same at both. + */ + private def bracketed(low: Option[Long], high: Option[Long], actual: Option[Long]): Boolean = + (low, high, actual) match { + case (Some(l), Some(h), Some(v)) => v >= math.min(l, h) && v <= math.max(l, h) + case (None, None, None) => true + case _ => false + } + + "TIMESTAMPDIFF" should "compare instants whatever the runtime type of an operand, in every venue" in { + diffIndexLoaded + val population = runtimePopulation + population.size shouldBe diffUnits.size * (24 + 24 + 8 + 8) + population.foreach(st => withClue(s"[${st.sql}] ")(Parser(st.sql).isRight shouldBe true)) + // Non-vacuity: in every venue a non-zero answer and a NULL, and the filters keep some rows and + // drop others. + val probe = clockNow() + Seq("SELECT", "WHERE", "GROUP", "HAVING").foreach { venue => + val answers = population.filter(_.venue == venue).flatMap(_.expected(probe).values) + withClue(s"$venue: ") { + Seq( + answers.exists(_.exists(_ < 0)), + answers.exists(_.exists(_ > 0)), + answers.contains(None) + ) shouldBe + Seq(true, true, true) + } + } + + val outcomes: Seq[(RuntimeStatement, Either[String, String])] = population.map { st => + val before = clockNow() + val result = diffRows(st.sql) + val after = clockNow() + val (low, high) = (st.expected(before), st.expected(after)) + st -> result.flatMap { rows => + val key = if (st.venue == "SELECT" || st.venue == "WHERE") "id" else "g" + if (st.venue == "WHERE" || st.venue == "HAVING") { + val actual = rows.map(_.getOrElse(key, "?").toString).toSet + val misplaced = (low.keySet.filter { k => + val kept = actual.contains(k) + kept != low(k).exists(_ > 0) && kept != high(k).exists(_ > 0) + } ++ (actual -- low.keySet)).toSeq.sorted + Right( + if (misplaced.isEmpty) "" + else s"kept ${actual.toSeq.sorted.mkString(",")}, misplaced ${misplaced.mkString(",")}" + ) + } else { + val values = + rows.map(r => r.getOrElse(key, "?").toString -> whole(r.getOrElse("x", null))) + values.collectFirst { case (_, Left(error)) => error } match { + case Some(error) => Left(error) + case None => + val actual = values.collect { case (k, Right(v)) => k -> v }.toMap + val off = (low.keySet ++ actual.keySet).filterNot { k => + bracketed( + low.getOrElse(k, None), + high.getOrElse(k, None), + actual.getOrElse(k, None) + ) + } + Right( + if (off.isEmpty) "" + else + s"answered ${actual.toSeq.sortBy(_._1).mkString(" ")}, expected " + + low.toSeq.sortBy(_._1).mkString(" ") + ) + } + } + } + } + val wrong = outcomes.collect { case (st, Right(diff)) if diff.nonEmpty => s"[${st.sql}] $diff" } + val errors = outcomes.collect { case (st, Left(error)) => s"[${st.sql}] $error" } + info( + s"TIMESTAMPDIFF runtime types: ${population.size} statements; wrong: ${wrong.size}; " + + s"errors: ${errors.size}" + ) + wrong.foreach(info(_)) + errors.foreach(info(_)) + withClue( + s"${wrong.size} wrong statements, ${errors.size} errors:\n" + + (wrong.take(20) ++ errors.take(10)).mkString("\n") + "\n" + ) { + wrong shouldBe empty + errors shouldBe empty + } + } + + private val runtimeTable = "timestampdiff_runtime_types" + + it should "compare instants whatever the runtime type of an operand, in a computed column" in { + // In an ingest script a COALESCE hands on the RAW JSON value of a column it reads, and a cast + // column is parsed a second time once parsed, whatever function holds either: neither + // `COALESCE(ts, CURRENT_DATE)` nor `ts::DATE` is one of the operands here. + val pairs = + runtimePairs(runtimeColumns, Seq(datetimeLiteral, currentDate, currentInstant, chosen)) + val columns: Seq[(String, String, ClockOperand, ClockOperand)] = for { + (unit, u) <- diffUnits.zipWithIndex + ((a, b), p) <- pairs.zipWithIndex + } yield (s"c${u}_$p", unit, a, b) + columns.size shouldBe diffUnits.size * 16 + // a document missing a column leaves the computed column absent, not NULL: both are present + val docs = diffDocs.filter { case (_, _, d, ts) => d.isDefined && ts.isDefined } + docs.map(_._1) should contain("r03") + val ddl = + s"CREATE TABLE $runtimeTable (id KEYWORD, d DATE, ts TIMESTAMP, " + + columns + .map { case (c, unit, a, b) => + s"$c BIGINT SCRIPT AS (TIMESTAMPDIFF($unit, ${a.sql}, ${b.sql}))" + } + .mkString(", ") + ", PRIMARY KEY (id))" + Await.result(client.run(ddl), 120.seconds) match { + case ElasticSuccess(_) => createdTables = runtimeTable :: createdTables + case ElasticFailure(error) => fail(s"CREATE TABLE $runtimeTable failed: ${error.message}") + } + val values = docs.map { case (id, _, d, ts) => s"('$id', '${d.get}', '${ts.get}')" } + // the node's clock against this JVM's: two seconds either side + val before = clockNow().minusSeconds(2) + Await.result( + client.run(s"INSERT INTO $runtimeTable (id, d, ts) VALUES ${values.mkString(", ")}"), + 120.seconds + ) match { + case ElasticSuccess(dml: DmlResult) => dml.inserted shouldBe docs.size.toLong + case other => fail(s"INSERT INTO $runtimeTable: $other") + } + val after = clockNow().plusSeconds(2) + client.refresh(runtimeTable) + val stored: Map[String, ListMap[String, Any]] = diffRows(s"SELECT * FROM $runtimeTable") match { + case Right(rows) => rows.map(r => r.getOrElse("id", "?").toString -> r).toMap + case Left(error) => fail(s"SELECT * FROM $runtimeTable: $error") + } + stored.keySet shouldBe docs.map(_._1).toSet + val wrong = for { + (column, unit, a, b) <- columns + doc <- docs + expectedAt = (now: Instant) => + for { + start <- a.value(Seq(doc), now) + end <- b.value(Seq(doc), now) + } yield elapsed(unit, start, end) + actual = stored(doc._1).get(column).flatMap(v => whole(v).toOption.flatten) + if !bracketed(expectedAt(before), expectedAt(after), actual) + } yield s"$column = TIMESTAMPDIFF($unit, ${a.sql}, ${b.sql}) on ${doc._1}: stored " + + s"${actual.getOrElse("nothing")}, expected ${expectedAt(before).getOrElse("NULL")}" + info( + s"TIMESTAMPDIFF runtime types, computed columns: ${columns.size} columns x ${docs.size} " + + s"documents; wrong: ${wrong.size}" + ) + wrong.foreach(info(_)) + withClue(s"${wrong.size} wrong:\n" + wrong.take(20).mkString("\n") + "\n") { + wrong shouldBe empty + } + } + + // ----------------------------------------------------------------------------------------------- + // `DATE_TRUNC(…, WEEK)` is the ISO Monday that starts the week, at 00:00 UTC, in every venue: a + // SELECT item over a TIMESTAMP and a DATE column, a WHERE, a GROUP BY (Elasticsearch's calendar + // `date_histogram`) and a computed column. Each weekday, with times of day, across a year end. + // ----------------------------------------------------------------------------------------------- + + private val weekTable = "date_trunc_week" + + private val weekDocs: Seq[(String, String)] = Seq( + "w0" -> "2024-12-29T23:59:59Z", // a Sunday: the week before + "w1" -> "2024-12-30T08:00:00Z", // Monday + "w2" -> "2024-12-31T23:59:59Z", // Tuesday, the last second of 2024 + "w3" -> "2025-01-01T00:00:00Z", // Wednesday, the first second of 2025 + "w4" -> "2025-01-02T12:30:00Z", // Thursday + "w5" -> "2025-01-03T18:00:00Z", // Friday + "w6" -> "2025-01-04T00:00:01Z", // Saturday + "w7" -> "2025-01-05T23:59:59Z", // Sunday + "w8" -> "2025-01-06T00:00:00Z" // Monday: the week after + ) + + /** The ISO Monday of an instant's UTC date: 1970-01-01, epoch day 0, is a Thursday. */ + private def isoMonday(iso: String): LocalDate = { + val day = utcDate(Instant.parse(iso)).toEpochDay + LocalDate.ofEpochDay(day - Math.floorMod(day + 3, 7L)) + } + + private val Midnight = """(\d{4}-\d{2}-\d{2})(T00:00(:00(\.0+)?)?(Z|\+00:00)?)?""".r + + /** A truncated value as its date when it is that date at 00:00 UTC; anything else as it came. */ + private def midnight(value: Any): String = + scalar(value) match { + case Midnight(date, _*) => date + case other => other + } + + "DATE_TRUNC(…, WEEK)" should "truncate to the ISO Monday, 00:00 UTC, in every venue" in { + val expected: Map[String, String] = + weekDocs.map { case (id, ts) => id -> isoMonday(ts).toString }.toMap + // every weekday, both years, three Mondays + weekDocs.map { case (_, ts) => utcDate(Instant.parse(ts)).getDayOfWeek }.toSet.size shouldBe 7 + expected.values.toSet shouldBe Set("2024-12-23", "2024-12-30", "2025-01-06") + + val ddl = + s"CREATE TABLE $weekTable (id KEYWORD, ts TIMESTAMP, d DATE, " + + "wt TIMESTAMP SCRIPT AS (DATE_TRUNC(ts, WEEK)), wd DATE SCRIPT AS (DATE_TRUNC(d, WEEK)), " + + "PRIMARY KEY (id))" + Await.result(client.run(ddl), 60.seconds) match { + case ElasticSuccess(_) => createdTables = weekTable :: createdTables + case ElasticFailure(error) => fail(s"CREATE TABLE $weekTable failed: ${error.message}") + } + val values = weekDocs.map { case (id, ts) => s"('$id', '$ts', '${ts.take(10)}')" } + Await.result( + client.run(s"INSERT INTO $weekTable (id, ts, d) VALUES ${values.mkString(", ")}"), + 60.seconds + ) match { + case ElasticSuccess(dml: DmlResult) => dml.inserted shouldBe weekDocs.size.toLong + case other => fail(s"INSERT INTO $weekTable: $other") + } + client.refresh(weekTable) + + // Every venue is read before anything is asserted, so a failure names each venue that differs. + def byId(sql: String, column: String): Either[String, Map[String, String]] = + diffRows(sql).map( + _.map(r => r.getOrElse("id", "?").toString -> midnight(r.getOrElse(column, null))).toMap + ) + + val mondays = expected.values.toSeq.distinct.sorted + val counts: Map[String, Either[String, Option[Long]]] = + expected.values.groupBy(identity).map { case (m, ms) => m -> Right(Some(ms.size.toLong)) } + val verdicts: Seq[(String, Boolean, String)] = + Seq( + s"SELECT id, DATE_TRUNC(ts, WEEK) AS w FROM $weekTable" -> "w", + s"SELECT id, DATE_TRUNC(WEEK, ts) AS w FROM $weekTable" -> "w", + s"SELECT id, DATE_TRUNC(d, WEEK) AS w FROM $weekTable" -> "w", + s"SELECT id, wt FROM $weekTable" -> "wt", + s"SELECT id, wd FROM $weekTable" -> "wd" + ).map { case (sql, column) => + val got = byId(sql, column) + (sql, got == Right(expected), got.fold(identity, _.toSeq.sorted.mkString(" "))) + } ++ + // a WHERE keeps each document under its own Monday and under no other + (for (column <- Seq("ts", "d"); monday <- mondays) yield { + val sql = + s"SELECT id FROM $weekTable WHERE DATE_TRUNC($column, WEEK) = CAST('$monday' AS DATE)" + val kept = diffRows(sql).map(_.map(_.getOrElse("id", "?").toString).toSet) + val want = expected.collect { case (id, m) if m == monday => id }.toSet + (sql, kept == Right(want), kept.fold(identity, _.toSeq.sorted.mkString(","))) + }) ++ + // a GROUP BY buckets on the same Mondays + Seq("ts", "d").map { column => + val sql = + s"SELECT DATE_TRUNC($column, WEEK) AS w, COUNT(*) AS c FROM $weekTable " + + s"GROUP BY DATE_TRUNC($column, WEEK)" + val buckets = diffRows(sql).map( + _.map(r => midnight(r.getOrElse("w", null)) -> whole(r.getOrElse("c", null))).toMap + ) + (sql, buckets == Right(counts), buckets.fold(identity, _.toSeq.sortBy(_._1).mkString(" "))) + } + verdicts.foreach { case (sql, ok, got) => + info(s"${if (ok) "RIGHT" else "WRONG"} [$sql] $got") + } + val wrong = verdicts.collect { case (sql, false, got) => s"[$sql] $got" } + withClue( + s"expected ${expected.toSeq.sorted.mkString(" ")}; ${wrong.size} venues differ:\n" + + wrong.mkString("\n") + "\n" + ) { + wrong shouldBe empty + } + } } diff --git a/testkit/src/main/scala/app/softnetwork/elastic/client/GatewayApiIntegrationSpec.scala b/testkit/src/main/scala/app/softnetwork/elastic/client/GatewayApiIntegrationSpec.scala index d934febbe..7ae2f2f30 100644 --- a/testkit/src/main/scala/app/softnetwork/elastic/client/GatewayApiIntegrationSpec.scala +++ b/testkit/src/main/scala/app/softnetwork/elastic/client/GatewayApiIntegrationSpec.scala @@ -27,7 +27,7 @@ import app.softnetwork.elastic.sql.query.SelectStatement import com.typesafe.config.ConfigFactory import java.time.temporal.ChronoUnit -import java.time.{LocalDate, ZoneOffset, ZonedDateTime} +import java.time.{LocalDate, ZoneOffset} // --------------------------------------------------------------------------- // Base test trait — to be mixed with ElasticDockerTestKit @@ -130,13 +130,13 @@ trait GatewayApiIntegrationSpec extends GatewayIntegrationTestKit { | id INT NOT NULL COMMENT 'user identifier', | name VARCHAR FIELDS(raw Keyword COMMENT 'sortable') DEFAULT 'anonymous' OPTIONS (analyzer = 'french', search_analyzer = 'french'), | birthdate DATE, - | age INT SCRIPT AS (DATEDIFF(birthdate, CURRENT_DATE, YEAR)), + | age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)), | ingested_at TIMESTAMP DEFAULT _ingest.timestamp, | profile STRUCT FIELDS( | bio VARCHAR, | followers INT, | join_date DATE, - | seniority INT SCRIPT AS (DATEDIFF(profile.join_date, CURRENT_DATE, DAY)) + | seniority INT SCRIPT AS (DATEDIFF(CURRENT_DATE, profile.join_date, DAY)) | ) COMMENT 'user profile', | PRIMARY KEY (id) |) PARTITION BY birthdate (MONTH), OPTIONS (mappings = (dynamic = false));""".stripMargin @@ -157,13 +157,13 @@ trait GatewayApiIntegrationSpec extends GatewayIntegrationTestKit { // of conflating the two facts. The runtime type now lives in `SQLTypeUtils.runtimeType`, // reached only through `GenericIdentifier.baseType`, so no release note is owed. ddl should include("birthdate DATE") - ddl should include("age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR))") + ddl should include("age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE))") ddl should include("ingested_at TIMESTAMP DEFAULT _ingest.timestamp") ddl should include("profile STRUCT FIELDS (") ddl should include("bio VARCHAR") ddl should include("followers INT") ddl should include("join_date DATE") - ddl should include("seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY))") + ddl should include("seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY))") ddl should include("PRIMARY KEY (id)") ddl should include("PARTITION BY birthdate (MONTH)") } @@ -1479,7 +1479,7 @@ trait GatewayApiIntegrationSpec extends GatewayIntegrationTestKit { | bio VARCHAR, | followers INT, | join_date DATE, - | seniority INT SCRIPT AS (DATEDIFF(profile.join_date, CURRENT_DATE, DAY)) + | seniority INT SCRIPT AS (DATEDIFF(CURRENT_DATE, profile.join_date, DAY)) | ) |);""".stripMargin @@ -1491,7 +1491,7 @@ trait GatewayApiIntegrationSpec extends GatewayIntegrationTestKit { | bio VARCHAR, | followers INT, | join_date DATE, - | seniority INT SCRIPT AS (DATEDIFF(profile.join_date, CURRENT_DATE, DAY)), + | seniority INT SCRIPT AS (DATEDIFF(CURRENT_DATE, profile.join_date, DAY)), | reputation DOUBLE DEFAULT 0.0 | );""".stripMargin @@ -1534,7 +1534,7 @@ trait GatewayApiIntegrationSpec extends GatewayIntegrationTestKit { | id INT NOT NULL, | join_date DATE, | reputation DOUBLE DEFAULT 0.0, - | seniority INT SCRIPT AS (DATEDIFF(join_date, CURRENT_DATE, DAY)) + | seniority INT SCRIPT AS (DATEDIFF(CURRENT_DATE, join_date, DAY)) |);""".stripMargin assertDdl(System.nanoTime(), client.run(create).futureValue) @@ -2756,7 +2756,7 @@ trait GatewayApiIntegrationSpec extends GatewayIntegrationTestKit { * the document being indexed, NOT the temporal object `doc['d'].value` hands a query — that part * of the earlier reading was right, and `SQLTypeUtils.coerce` still guards its temporal arms on * `isProcessorContext` for exactly that reason. What was wrong was the conclusion drawn from it: - * that `DATEDIFF(d, CURRENT_DATE, DAY)` therefore CANNOT compute at ingest. It can, once the + * that `DATEDIFF(CURRENT_DATE, d, DAY)` therefore CANNOT compute at ingest. It can, once the * operand is parsed first — and until story 21.8 Part C it did not, so `ignore_failure` left the * column unset and the value was silently missing from every stored document. * @@ -2768,8 +2768,9 @@ trait GatewayApiIntegrationSpec extends GatewayIntegrationTestKit { * `ZonedDateTime` throughout instead of narrowing `CURRENT_DATE` to a `LocalDate`. * * ⚠️ The expected value is COMPUTED, not pinned: it is a distance from `now`, so a literal would - * have been correct for one day. It is derived the way the ingest script derives it, and the ±1 - * tolerance covers an ingest and an assertion that straddle UTC midnight. + * have been correct for one day. It is derived the way the ingest script derives it — the dates + * come first, so it is `CURRENT_DATE - d` in calendar days (UTC) — and the ±1 tolerance covers + * an ingest and an assertion that straddle UTC midnight. */ it should "record what an ingest script sees for a DATE column (ctx runtime type)" in { val create = @@ -2777,7 +2778,7 @@ trait GatewayApiIntegrationSpec extends GatewayIntegrationTestKit { | id INT NOT NULL, | d DATE, | label KEYWORD, - | days INT SCRIPT AS (DATEDIFF(d, CURRENT_DATE, DAY)) + | days INT SCRIPT AS (DATEDIFF(CURRENT_DATE, d, DAY)) |);""".stripMargin assertDdl(System.nanoTime(), client.run(create).futureValue) @@ -2803,8 +2804,8 @@ trait GatewayApiIntegrationSpec extends GatewayIntegrationTestKit { val days = row.get("days").map(_ => scalarOf(row, "days")).orNull val expected = ChronoUnit.DAYS.between( - LocalDate.of(2024, 3, 15).atStartOfDay(ZoneOffset.UTC), - ZonedDateTime.now(ZoneOffset.UTC) + LocalDate.of(2024, 3, 15), + LocalDate.now(ZoneOffset.UTC) ) withClue(s"ingest-computed days = [$days], expected ~$expected: ") { // The measurement. Asserted, not merely printed, so a change in either direction is loud — @@ -3276,7 +3277,7 @@ trait GatewayApiIntegrationSpec extends GatewayIntegrationTestKit { | value = "anonymous" | ), | SCRIPT ( - | description = "age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR))", + | description = "age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE))", | lang = "painless", | source = "def param1 = ctx.birthdate; def param2 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(ctx['_ingest']['timestamp']), ZoneId.of('Z')).toLocalDate(); ctx.age = (param1 == null) ? null : Long.valueOf(ChronoUnit.YEARS.between(param1, param2))", | ignore_failure = true @@ -3289,7 +3290,7 @@ trait GatewayApiIntegrationSpec extends GatewayIntegrationTestKit { | value = "_ingest.timestamp" | ), | SCRIPT ( - | description = "profile.seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY))", + | description = "profile.seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY))", | lang = "painless", | source = "def param1 = ctx.profile?.join_date; def param2 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(ctx['_ingest']['timestamp']), ZoneId.of('Z')).toLocalDate(); ctx.profile.seniority = (param1 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param1, param2))", | ignore_failure = true diff --git a/testkit/src/main/scala/app/softnetwork/elastic/client/GroupByCompletenessSpec.scala b/testkit/src/main/scala/app/softnetwork/elastic/client/GroupByCompletenessSpec.scala index 3def597fc..1fe755ec7 100644 --- a/testkit/src/main/scala/app/softnetwork/elastic/client/GroupByCompletenessSpec.scala +++ b/testkit/src/main/scala/app/softnetwork/elastic/client/GroupByCompletenessSpec.scala @@ -2151,7 +2151,8 @@ trait GroupByCompletenessSpec extends AnyFlatSpecLike with ElasticDockerTestKit } /** A spelling of the family over two operands, and the days it answers for them: `end - start`, - * except MySQL's two-argument DATEDIFF, which is `first - second`. + * where the dates first is `first - second` and the unit first `last - middle`. Between two + * calendar dates, the days crossed and the whole days elapsed are the same number. */ private final case class DateDiffForm( sql: (String, String) => String, @@ -2162,9 +2163,9 @@ trait GroupByCompletenessSpec extends AnyFlatSpecLike with ElasticDockerTestKit def between(start: LocalDate, end: LocalDate): Long = ChronoUnit.DAYS.between(start, end) Seq( DateDiffForm((a, b) => s"DATEDIFF($a, $b)", (a, b) => between(b, a)), - DateDiffForm((a, b) => s"DATEDIFF($a, $b, DAY)", between), + DateDiffForm((a, b) => s"DATEDIFF($a, $b, DAY)", (a, b) => between(b, a)), DateDiffForm((a, b) => s"DATEDIFF(DAY, $a, $b)", between), - DateDiffForm((a, b) => s"DATE_DIFF($a, $b, DAY)", between), + DateDiffForm((a, b) => s"DATE_DIFF($a, $b, DAY)", (a, b) => between(b, a)), DateDiffForm((a, b) => s"DATE_DIFF(DAY, $a, $b)", between), DateDiffForm((a, b) => s"TIMESTAMPDIFF(DAY, $a, $b)", between) ) diff --git a/testkit/src/main/scala/app/softnetwork/elastic/client/repl/ReplGatewayIntegrationSpec.scala b/testkit/src/main/scala/app/softnetwork/elastic/client/repl/ReplGatewayIntegrationSpec.scala index 3fd3e7625..3451b251b 100644 --- a/testkit/src/main/scala/app/softnetwork/elastic/client/repl/ReplGatewayIntegrationSpec.scala +++ b/testkit/src/main/scala/app/softnetwork/elastic/client/repl/ReplGatewayIntegrationSpec.scala @@ -127,13 +127,13 @@ trait ReplGatewayIntegrationSpec extends ReplIntegrationTestKit { | id INT NOT NULL COMMENT 'user identifier', | name VARCHAR FIELDS(raw Keyword COMMENT 'sortable') DEFAULT 'anonymous' OPTIONS (analyzer = 'french', search_analyzer = 'french'), | birthdate DATE, - | age INT SCRIPT AS (DATEDIFF(birthdate, CURRENT_DATE, YEAR)), + | age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)), | ingested_at TIMESTAMP DEFAULT _ingest.timestamp, | profile STRUCT FIELDS( | bio VARCHAR, | followers INT, | join_date DATE, - | seniority INT SCRIPT AS (DATEDIFF(profile.join_date, CURRENT_DATE, DAY)) + | seniority INT SCRIPT AS (DATEDIFF(CURRENT_DATE, profile.join_date, DAY)) | ) COMMENT 'user profile', | PRIMARY KEY (id) |) PARTITION BY birthdate (MONTH), OPTIONS (mappings = (dynamic = false))""".stripMargin @@ -146,7 +146,7 @@ trait ReplGatewayIntegrationSpec extends ReplIntegrationTestKit { ddl should include("CREATE OR REPLACE TABLE users") ddl should include("id INT NOT NULL COMMENT 'user identifier'") ddl should include("birthdate DATE") - ddl should include("age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR))") + ddl should include("age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE))") ddl should include("PRIMARY KEY (id)") ddl should include("PARTITION BY birthdate (MONTH)") } From ef77e7ed9e04e73d3d088f6631930c1dbc1c998b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?St=C3=A9phane=20Manciot?= Date: Sat, 3 Oct 2026 09:04:09 +0200 Subject: [PATCH 2/4] fix(sql): date functions read Elasticsearch 6.8's date values ES 6.8 hands scripts a JodaCompatibleZonedDateTime, which implements no java.time interface: TIMESTAMPDIFF's month count (getLong) and DATEDIFF / DATE_DIFF over a non-column operand such as COALESCE(ts, d) (LocalDate.from) failed there. Such operands are now brought to a real ZonedDateTime with withZoneSameInstant, which 6.8 allows and which reads the same instant on every major. DATEDIFF over COALESCE(ts, d) never worked on 6.8, v0.23.0 included. Co-Authored-By: Claude Opus 5.5 (1M context) --- .../elastic/sql/function/time/package.scala | 75 +++++++++++++------ 1 file changed, 54 insertions(+), 21 deletions(-) diff --git a/sql/src/main/scala/app/softnetwork/elastic/sql/function/time/package.scala b/sql/src/main/scala/app/softnetwork/elastic/sql/function/time/package.scala index c5a61641d..89b330323 100644 --- a/sql/src/main/scala/app/softnetwork/elastic/sql/function/time/package.scala +++ b/sql/src/main/scala/app/softnetwork/elastic/sql/function/time/package.scala @@ -606,7 +606,10 @@ package object time { /** A temporal as the instant it denotes, in UTC, decided by its RUNTIME type: a `LocalDate` is * the start of its day, a `LocalDateTime` that wall-clock time in UTC, anything zoned the same * instant. `TIMESTAMPDIFF` compares instants, and an operand's declared type does not say - * which of the three it holds: `COALESCE(ts, CURRENT_DATE)` is either, row by row. + * which of the three it holds: `COALESCE(ts, CURRENT_DATE)` is either, row by row. `DATEDIFF` + * and `DATE_DIFF` read a calendar date through it too, off a row-level operand that is not a + * column: a zoned value there may be Elasticsearch 6.8's doc value, which is no `java.time` + * type until `withZoneSameInstant` (as in [[utcZoned]]) returns the `ZonedDateTime` it wraps. * * `ref` is read up to four times, so it is a name or an expression cast to `def` (the methods * below are resolved at runtime). NULL stays NULL: the MONTH, QUARTER and YEAR calls bind this @@ -615,6 +618,24 @@ package object time { private[time] def utcInstant(ref: String): String = s"($ref == null ? null : $ref instanceof LocalDate ? $ref.atStartOfDay(ZoneId.of('Z')) : $ref instanceof LocalDateTime ? $ref.atZone(ZoneId.of('Z')) : $ref.withZoneSameInstant(ZoneId.of('Z')))" + /** A column a SEARCH script reads, as a genuine `java.time.ZonedDateTime` in UTC; NULL stays + * NULL (the MONTH, QUARTER and YEAR calls bind it in the prologue, before the null guard). + * + * The column's script parameter is shared by every call that reads the column, and holds what + * the FIRST of them made of it (the parameter-identity family, issue #370): the UTC instant + * `DateDiff.in` folds onto it, or the raw doc value when a function over the same column read + * it first (`TIMESTAMPDIFF(MONTH, ts::DATE, ts)`). On Elasticsearch 6.8 that value is a + * `JodaCompatibleZonedDateTime`, which implements no `java.time` interface: it has no + * `getLong` ([[monthKey]] failed with `dynamic method [..., getLong/1] not found`) and is no + * `Temporal` (`ChronoUnit.between` failed with a `ClassCastException`). `withZoneSameInstant` + * is on its allow-list and returns the `ZonedDateTime` it wraps; on 7.x, 8.x and 9.x it reads + * the same instant, so every major counts the same units. + */ + private[time] def utcZoned(ref: String): String = { + val r = if (postfixable(ref)) ref else s"($ref)" + s"($r == null ? null : $r.withZoneSameInstant(ZoneId.of('Z')))" + } + /** Whether a method can be appended to a rendered operand as it stands: outside parentheses, * brackets and string literals it holds nothing but names, digits and dots. Anything else — a * ternary (`p4 ? p3 : p1`, which is how a CASE operand renders), a cast (`(long) x`) — is @@ -825,11 +846,13 @@ package object time { * comparison is the start of its day, as at row level — `CURRENT_DATE` included, which a * processor holds as the current instant. * - At row level a column operand holds its UTC instant ([[in]]), so its calendar date is - * `toLocalDate()`. An operand with no column of its own (`'2025-01-10'::DATE`, - * `CURRENT_DATE`, `NOW()`) is read by `LocalDate.from` for a calendar date, which takes - * the `LocalDate` and the UTC `ZonedDateTime` alike; for a time-of-day comparison a DATE - * is the start of its day (a `LocalDate` has no hours) and anything else renders as it - * did. + * `toLocalDate()`. Any other operand (`'2025-01-10'::DATE`, `CURRENT_DATE`, `NOW()`, a + * COALESCE, a CASE) is read for a calendar date by `LocalDate.from`, once it is a UTC + * `java.time` value by its RUNTIME type ([[DateDiff.utcInstant]]): a COALESCE or a CASE + * over a column hands on the column's raw doc value, which on Elasticsearch 6.8 is no + * `java.time` type, and `LocalDate.from` refused it (`ClassCastException`). For a + * time-of-day comparison a DATE is the start of its day (a `LocalDate` has no hours) and + * anything else renders as it did. * - `TIMESTAMPDIFF` takes [[elapsedOperand]] instead, in every venue. */ private def operand( @@ -873,9 +896,11 @@ package object time { if comparedIn == SQLTypes.Timestamp && value.baseType == SQLTypes.Date => SQLTypeUtils.coerce(rendered, SQLTypes.Date, comparedIn, nullable = false, None) // a `ZonedDateTime` here (`'…'::TIMESTAMP`, `NOW()`, a CASE) would keep its time of day - // through `unitStart`, and a calendar unit would count it - case _ if comparedIn == SQLTypes.Date => s"LocalDate.from($rendered)" - case _ => rendered + // through `unitStart`, and a calendar unit would count it; read by its runtime type + // first, for a COALESCE or a CASE may hand on a column's raw 6.8 doc value + case _ if comparedIn == SQLTypes.Date => + s"LocalDate.from(${DateDiff.utcInstant(bound(rendered, context))})" + case _ => rendered } case None => rendered } @@ -892,10 +917,13 @@ package object time { * - An ingest processor reads a DATE-typed operand as the start of its calendar day * (`LocalDate.from`): it holds `CURRENT_DATE` as the current INSTANT, so there the * declared type, not the runtime one, is what says "a date". - * - A column is an instant already, in a script that reads it: row level reads it as its UTC - * instant ([[in]]), an ingest processor parses it into one - * (`SQLTypeUtils.processorTemporal`). A function over it need not be one: `ts::DATE` is a - * `LocalDate`. + * - A column is an instant in a script that reads it, though not always a `java.time` one. + * An ingest processor parses it into a UTC `ZonedDateTime` + * (`SQLTypeUtils.processorTemporal`). Row level reads its column parameter, which holds + * its UTC instant ([[in]]) unless a function over the same column read it first: then it + * is the raw doc value, which on Elasticsearch 6.8 is no `java.time` type, so it goes + * through [[DateDiff.utcZoned]]. A function over a column need not be an instant at all: + * `ts::DATE` is a `LocalDate`. * - Any other operand is converted by its RUNTIME type ([[DateDiff.utcInstant]]), which its * declared type does not give: `COALESCE(ts, CURRENT_DATE)`, or a CASE over `CURRENT_DATE` * and a DATE, is a `LocalDate` on one row and a `ZonedDateTime` on another, a `::DATETIME` @@ -911,16 +939,21 @@ package object time { s"LocalDate.from($rendered).atStartOfDay(ZoneId.of('Z'))" case column: Identifier if context.isDefined && column.name.trim.nonEmpty && column.functions.isEmpty => - rendered - case _ => - val ref = context match { - case Some(_) if FunctionN.isName(rendered) => rendered - case Some(ctx) => ctx.addParam(LiteralParam(rendered)).getOrElse(s"((def) ($rendered))") - case None => s"((def) ($rendered))" - } - DateDiff.utcInstant(ref) + if (context.exists(_.isProcessor)) rendered else DateDiff.utcZoned(rendered) + case _ => DateDiff.utcInstant(bound(rendered, context)) } + /** A rendered operand as [[DateDiff.utcInstant]] may read it, up to four times: a name as it + * stands, any other expression evaluated once into a parameter where the script can bind one, + * cast to `def` where it cannot. + */ + private def bound(rendered: String, context: Option[PainlessContext]): String = + context match { + case Some(_) if FunctionN.isName(rendered) => rendered + case Some(ctx) => ctx.addParam(LiteralParam(rendered)).getOrElse(s"((def) ($rendered))") + case None => s"((def) ($rendered))" + } + /** `ChronoUnit` has no `QUARTERS`, so a `QUARTER` call failed to compile in Elasticsearch. The * ISO quarter-year unit counts the whole quarters between two calendar dates, which between * two quarter starts ([[unitStart]]) are the quarter boundaries crossed. From 15af12213711bb713abe92a82a3bd96a24bb6ba3 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?St=C3=A9phane=20Manciot?= Date: Sat, 3 Oct 2026 09:04:09 +0200 Subject: [PATCH 3/4] fix(es6-jest): CREATE TABLE and legacy templates work on Elasticsearch 6.8 From 6.8 the engine sends typeless mappings, and ES 6.x reads a mapping as typed unless the request says include_type_name=false. The REST client says it; the Jest client did not, so every CREATE TABLE (and every PARTITION BY template) failed on 6.8 with "Root mapping definition has unsupported parameters", v0.23.0 included. The Jest client now adds include_type_name=false on 6.8 when it creates an index with a mapping or a legacy template; 6.7.x and 7+ are unchanged. Template.Create gains a field (binary-incompatible: downstream projects rebuild). Co-Authored-By: Claude Opus 5.5 (1M context) --- .../elastic/client/jest/JestIndicesApi.scala | 4 +++ .../elastic/client/jest/JestTemplateApi.scala | 2 +- .../elastic/client/jest/JestVersionApi.scala | 27 ++++++++++++++++++- .../client/jest/actions/Template.scala | 10 +++++-- 4 files changed, 39 insertions(+), 4 deletions(-) diff --git a/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/JestIndicesApi.scala b/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/JestIndicesApi.scala index a9d70c8f8..013ebbe5f 100644 --- a/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/JestIndicesApi.scala +++ b/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/JestIndicesApi.scala @@ -42,6 +42,9 @@ trait JestIndicesApi extends IndicesApi with JestClientHelpers { with JestClientCompanion => /** Create an index with the given settings. + * + * A mapping says `include_type_name=false` where Elasticsearch must be told it is typeless + * ([[sendsTypelessMappings]]): without it 6.8 refused every `CREATE TABLE`. * @see * [[IndicesApi.createIndex]] */ @@ -67,6 +70,7 @@ trait JestIndicesApi extends IndicesApi with JestClientHelpers { } mappings.foreach { mapping => builder.mappings(mapping) + if (sendsTypelessMappings) builder.setParameter("include_type_name", "false") } builder.build() } diff --git a/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/JestTemplateApi.scala b/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/JestTemplateApi.scala index 3f4c2b312..2b8cd53e6 100644 --- a/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/JestTemplateApi.scala +++ b/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/JestTemplateApi.scala @@ -103,7 +103,7 @@ trait JestTemplateApi extends TemplateApi with JestClientHelpers { operation = "createLegacyTemplate", retryable = false // Creation can not be retried ) { - Template.Create(templateName, templateDefinition) + Template.Create(templateName, templateDefinition, typelessMappings = sendsTypelessMappings) } override private[client] def executeDeleteLegacyTemplate( diff --git a/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/JestVersionApi.scala b/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/JestVersionApi.scala index 2640a629f..aa4b2181c 100644 --- a/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/JestVersionApi.scala +++ b/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/JestVersionApi.scala @@ -16,13 +16,38 @@ package app.softnetwork.elastic.client.jest -import app.softnetwork.elastic.client.{result, VersionApi} +import app.softnetwork.elastic.client.{result, ElasticsearchVersion, VersionApi} import io.searchbox.core.Cat import org.json4s.DefaultFormats import org.json4s.jackson.JsonMethods trait JestVersionApi extends VersionApi with JestClientHelpers { _: JestClientCompanion => + + /** Whether a request that carries a mapping must say `include_type_name=false`: on Elasticsearch + * 6.8, the one version that receives the mapping TYPELESS yet reads a mapping as TYPED by + * default. + * + * The mapping's shape is decided upstream, per version, by + * [[ElasticsearchVersion.requiresDocTypeWrapper]] (`MappingConverter`, `TemplateConverter`): + * wrapped in `_doc` before 6.8, typeless from 6.8. Every 6.x defaults `include_type_name` to + * `true`, so 6.8 read the typeless mapping's top-level keys as TYPE names and refused the + * `_meta` every table writes (`Failed to parse mapping [_meta]: Root mapping definition has + * unsupported parameters`): no `CREATE TABLE` ran through this client on 6.8, partitioned or + * not. The ES 6 REST client sends the same parameter on the same requests (its typeless + * `CreateIndexRequest` and `PutIndexTemplateRequest`). Not before 6.8, which refuses the + * parameter with a `_doc`-wrapped mapping; not from 7.0, where a mapping is typeless by default. + * + * A put mapping needs none: it names the type in its path (`PUT //_doc/_mapping`), which + * 6.8 takes with a typeless body, and with the parameter set it would refuse the type. + */ + private[client] def sendsTypelessMappings: Boolean = + version match { + case result.ElasticSuccess(v) => + ElasticsearchVersion.isEs6(v) && !ElasticsearchVersion.requiresDocTypeWrapper(v) + case _ => false + } + override private[client] def executeVersion(): result.ElasticResult[String] = executeJestAction( "version", diff --git a/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/actions/Template.scala b/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/actions/Template.scala index 1c38c2fab..df6893610 100644 --- a/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/actions/Template.scala +++ b/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/actions/Template.scala @@ -24,14 +24,20 @@ object Template { import io.searchbox.client.JestResult import com.google.gson.Gson - case class Create(templateName: String, json: String) extends AbstractAction[JestResult] { + /** `PUT /_template/`. `typelessMappings` says `include_type_name=false`, without which + * Elasticsearch 6.8 reads the template's typeless mappings as typed and refuses them + * (`JestVersionApi.sendsTypelessMappings`). + */ + case class Create(templateName: String, json: String, typelessMappings: Boolean = false) + extends AbstractAction[JestResult] { payload = json override def getRestMethodName: String = "PUT" override def getURI(elasticsearchVersion: ElasticsearchVersion): String = - s"/_template/$templateName" + if (typelessMappings) s"/_template/$templateName?include_type_name=false" + else s"/_template/$templateName" override def createNewElasticSearchResult( json: String, From 033ab7c25094194cf678036974294162487cc300 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?St=C3=A9phane=20Manciot?= Date: Sat, 3 Oct 2026 10:54:01 +0200 Subject: [PATCH 4/4] test(es6-jest): run the Jest specs on Elasticsearch 6.8; check DATEDIFF over COALESCE on every major Six es6 Jest specs were pinned to ES 6.7.2 because CREATE TABLE failed through the Jest client on 6.8. The previous commit fixes that, so they run on 6.8.23 again (252 passed, 0 failed; the 5 cancelled tests are ES 6 platform gates: enrich policies, composable templates, regex flags before 7.0). The DateFunctionExecutionSpec population gains the 88 DATEDIFF / DATE_DIFF over COALESCE(ts, d) statements, so CI checks the 6.8 fix on every major. Co-Authored-By: Claude Opus 5.5 (1M context) --- .../client/JestClientInsertByQuerySpec.scala | 2 - ...JestClientPerIndexSchemaCacheTtlSpec.scala | 10 +- .../elastic/client/JestClientSpec.scala | 4 +- .../client/JestClientTemplateApiSpec.scala | 2 - .../elastic/client/JestGatewayApiSpec.scala | 2 - .../repl/JestReplGatewayIntegrationSpec.scala | 2 - .../client/DateFunctionExecutionSpec.scala | 152 +++++++++++++++--- 7 files changed, 132 insertions(+), 42 deletions(-) diff --git a/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientInsertByQuerySpec.scala b/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientInsertByQuerySpec.scala index d22df129b..a7846efef 100644 --- a/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientInsertByQuerySpec.scala +++ b/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientInsertByQuerySpec.scala @@ -5,6 +5,4 @@ import app.softnetwork.elastic.scalatest.ElasticDockerTestKit class JestClientInsertByQuerySpec extends InsertByQuerySpec with ElasticDockerTestKit { override lazy val client: ElasticClientApi = new JestClientSpi().client(elasticConfig) - - override def elasticVersion: String = "6.7.2" } diff --git a/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientPerIndexSchemaCacheTtlSpec.scala b/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientPerIndexSchemaCacheTtlSpec.scala index 1bffee4a8..d2818a376 100644 --- a/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientPerIndexSchemaCacheTtlSpec.scala +++ b/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientPerIndexSchemaCacheTtlSpec.scala @@ -16,12 +16,4 @@ package app.softnetwork.elastic.client -/** Pinned to 6.7.2 like every other jest spec that runs DDL (`JestGatewayApiSpec`, - * `JestClientTemplateApiSpec`, …): on 6.8 this client sends a typed mapping in which `_meta` is - * read as the TYPE name, so any `CREATE TABLE` — which always writes `_meta.columns` — is rejected - * with `Root mapping definition has unsupported parameters`. Pre-existing and unrelated to the - * TTL; the ES 6 rest client on 6.8 runs the same DDL fine. - */ -class JestClientPerIndexSchemaCacheTtlSpec extends PerIndexSchemaCacheTtlSpec { - override def elasticVersion: String = "6.7.2" -} +class JestClientPerIndexSchemaCacheTtlSpec extends PerIndexSchemaCacheTtlSpec diff --git a/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientSpec.scala b/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientSpec.scala index 8808ca47b..fbbf9b0eb 100644 --- a/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientSpec.scala +++ b/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientSpec.scala @@ -1,5 +1,3 @@ package app.softnetwork.elastic.client -class JestClientSpec extends ElasticClientSpec { - override def elasticVersion: String = "6.7.2" -} +class JestClientSpec extends ElasticClientSpec diff --git a/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientTemplateApiSpec.scala b/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientTemplateApiSpec.scala index e6160b4a3..fab25f7a7 100644 --- a/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientTemplateApiSpec.scala +++ b/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientTemplateApiSpec.scala @@ -5,6 +5,4 @@ import app.softnetwork.elastic.scalatest.ElasticDockerTestKit class JestClientTemplateApiSpec extends TemplateApiSpec with ElasticDockerTestKit { override lazy val client: TemplateApi with VersionApi = new JestClientSpi().client(elasticConfig) - - override def elasticVersion: String = "6.7.2" } diff --git a/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestGatewayApiSpec.scala b/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestGatewayApiSpec.scala index ded66b82c..9a6b0dced 100644 --- a/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestGatewayApiSpec.scala +++ b/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestGatewayApiSpec.scala @@ -5,6 +5,4 @@ import app.softnetwork.elastic.scalatest.ElasticDockerTestKit class JestGatewayApiSpec extends GatewayApiIntegrationSpec with ElasticDockerTestKit { override lazy val client: GatewayApi = new JestClientSpi().client(elasticConfig) - - override def elasticVersion: String = "6.7.2" } diff --git a/es6/jest/src/test/scala/app/softnetwork/elastic/client/repl/JestReplGatewayIntegrationSpec.scala b/es6/jest/src/test/scala/app/softnetwork/elastic/client/repl/JestReplGatewayIntegrationSpec.scala index c8fcd1255..a3207254d 100644 --- a/es6/jest/src/test/scala/app/softnetwork/elastic/client/repl/JestReplGatewayIntegrationSpec.scala +++ b/es6/jest/src/test/scala/app/softnetwork/elastic/client/repl/JestReplGatewayIntegrationSpec.scala @@ -7,6 +7,4 @@ import app.softnetwork.elastic.scalatest.ElasticDockerTestKit class JestReplGatewayIntegrationSpec extends ReplGatewayIntegrationSpec with ElasticDockerTestKit { override lazy val gateway: GatewayApi = new JestClientSpi().client(elasticConfig) - - override def elasticVersion: String = "6.7.2" } diff --git a/testkit/src/main/scala/app/softnetwork/elastic/client/DateFunctionExecutionSpec.scala b/testkit/src/main/scala/app/softnetwork/elastic/client/DateFunctionExecutionSpec.scala index fac73fe92..f34dd3dbb 100644 --- a/testkit/src/main/scala/app/softnetwork/elastic/client/DateFunctionExecutionSpec.scala +++ b/testkit/src/main/scala/app/softnetwork/elastic/client/DateFunctionExecutionSpec.scala @@ -320,10 +320,13 @@ trait DateFunctionExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKi constant("'2024-12-31 23:59:59'::TIMESTAMP", AtInstant(Instant.parse("2024-12-31T23:59:59Z"))) ) - private val diffRowOperands: Seq[DiffOperand] = Seq( - DiffOperand("d", aggregate = false)(_.head._3.map(x => OnDate(LocalDate.parse(x)))), + private val diffDate: DiffOperand = + DiffOperand("d", aggregate = false)(_.head._3.map(x => OnDate(LocalDate.parse(x)))) + + private val diffInstant: DiffOperand = DiffOperand("ts", aggregate = false)(_.head._4.map(x => AtInstant(Instant.parse(x)))) - ) ++ diffLiterals + + private val diffRowOperands: Seq[DiffOperand] = Seq(diffDate, diffInstant) ++ diffLiterals private val diffGroupOperands: Seq[DiffOperand] = Seq( DiffOperand("MAX(d)", aggregate = true)( @@ -465,6 +468,30 @@ trait DateFunctionExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKi for (a <- diffGroupOperands; b <- diffGroupOperands if a.aggregate || b.aggregate) yield (a, b) + /** A spelling over two operands in a venue, and its answer for each document or group. */ + private def diffStatement( + spelling: DiffSpelling, + venue: String, + a: DiffOperand, + b: DiffOperand, + scopes: Seq[(String, Seq[DiffDoc])] + ): DiffStatement = { + val e = spelling.sql(a.sql, b.sql) + val sql = venue match { + case "SELECT" => s"SELECT id, $e AS x FROM $diffIndex" + case "WHERE" => s"SELECT id FROM $diffIndex WHERE $e > 0" + case "GROUP" => s"SELECT g, $e AS x FROM $diffIndex GROUP BY g" + case _ => s"SELECT g, $e AS x FROM $diffIndex GROUP BY g HAVING x > 0" + } + val expected = scopes.map { case (key, docs) => + key -> (for { + first <- a.value(docs) + second <- b.value(docs) + } yield answer(spelling, first, second)) + }.toMap + DiffStatement(sql, venue, spelling.unit, expected) + } + private lazy val diffPopulation: Seq[DiffStatement] = { // A WHERE over two literals reads no column: it is a constant predicate, rendered as a bare // comparison of the function's boxed result, which fails to compile whatever the function is @@ -481,22 +508,7 @@ trait DateFunctionExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKi ("HAVING", diffGroupPairs, diffPerGroup) ) (a, b) <- pairs - } yield { - val e = spelling.sql(a.sql, b.sql) - val sql = venue match { - case "SELECT" => s"SELECT id, $e AS x FROM $diffIndex" - case "WHERE" => s"SELECT id FROM $diffIndex WHERE $e > 0" - case "GROUP" => s"SELECT g, $e AS x FROM $diffIndex GROUP BY g" - case _ => s"SELECT g, $e AS x FROM $diffIndex GROUP BY g HAVING x > 0" - } - val expected = scopes.map { case (key, docs) => - key -> (for { - first <- a.value(docs) - second <- b.value(docs) - } yield answer(spelling, first, second)) - }.toMap - DiffStatement(sql, venue, spelling.unit, expected) - } + } yield diffStatement(spelling, venue, a, b, scopes) } /** Every pair of operand values the population computes over, in written order. */ @@ -583,6 +595,13 @@ trait DateFunctionExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKi x.isInstanceOf[OnDate] && y.isInstanceOf[AtInstant] && instantOf(x) == instantOf(y) } shouldBe true + assertDiffPopulation("DATEDIFF family", population) + } + + /** Every statement of a population through the gateway, against its answer: the documents or + * groups a filter keeps, or each one's value; `label` names the population in the report. + */ + private def assertDiffPopulation(label: String, population: Seq[DiffStatement]): Unit = { val outcomes: Seq[(DiffStatement, Either[String, String])] = population.map { st => st -> diffRows(st.sql).flatMap { rows => val key = if (st.venue == "SELECT" || st.venue == "WHERE") "id" else "g" @@ -612,9 +631,7 @@ trait DateFunctionExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKi } val wrong = outcomes.collect { case (st, Right(diff)) if diff.nonEmpty => s"[${st.sql}] $diff" } val errors = outcomes.collect { case (st, Left(error)) => s"[${st.sql}] $error" } - info( - s"DATEDIFF family: ${population.size} statements; wrong: ${wrong.size}; errors: ${errors.size}" - ) + info(s"$label: ${population.size} statements; wrong: ${wrong.size}; errors: ${errors.size}") wrong.foreach(info(_)) errors.foreach(info(_)) withClue( @@ -626,6 +643,97 @@ trait DateFunctionExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKi } } + // ----------------------------------------------------------------------------------------------- + // DATE_DIFF and DATEDIFF over an operand that holds a column's value without being the column, + // `COALESCE(ts, d)` or `COALESCE(d, ts)`. A calendar unit compares such an operand as a date, + // from the value the node hands the script for the column; Elasticsearch 6.8 hands one of its + // own, which is no `java.time` value, and until the engine brought it to one first, every + // statement below failed there with a `class_cast_exception`. The oracle above answers them as + // it answers any operand, a COALESCE being its first argument where the document has a value for + // it and its second where it has not: `DATE_DIFF(a, b, unit)` and `DATEDIFF(unit, a, b)` in each + // calendar unit, and `DATEDIFF(a, b)`, each COALESCE against the column it falls back to, both + // ways round, as a SELECT item and in a WHERE. + // ----------------------------------------------------------------------------------------------- + + /** `COALESCE(x, y)` over a document: `x`'s value where the document has one, else `y`'s. */ + private def coalesce(x: DiffOperand, y: DiffOperand): DiffOperand = + DiffOperand(s"COALESCE(${x.sql}, ${y.sql})", aggregate = false)(docs => + x.value(docs).orElse(y.value(docs)) + ) + + /** The arguments of each COALESCE: the timestamp column first, then the date column first. */ + private val diffCoalesces: Seq[(DiffOperand, DiffOperand)] = + Seq(diffInstant -> diffDate, diffDate -> diffInstant) + + /** Each COALESCE against the column it falls back to, both ways round. */ + private val diffCoalescedPairs: Seq[(DiffOperand, DiffOperand)] = + diffCoalesces.flatMap { case (x, y) => + val c = coalesce(x, y) + Seq(c -> y, y -> c) + } + + /** `DATE_DIFF(a, b, unit)` and `DATEDIFF(unit, a, b)` per calendar unit, and `DATEDIFF(a, b)`. */ + private val diffCoalescedSpellings: Seq[DiffSpelling] = + Seq("YEAR", "QUARTER", "MONTH", "WEEK", "DAY").flatMap { u => + Seq( + DiffSpelling(u, datesFirst = true, countsElapsed = false)((a, b) => + s"DATE_DIFF($a, $b, $u)" + ), + DiffSpelling(u, datesFirst = false, countsElapsed = false)((a, b) => + s"DATEDIFF($u, $a, $b)" + ) + ) + } :+ DiffSpelling("DAY", datesFirst = true, countsElapsed = false)((a, b) => + s"DATEDIFF($a, $b)" + ) + + private lazy val diffCoalescedPopulation: Seq[DiffStatement] = + for { + spelling <- diffCoalescedSpellings + venue <- Seq("SELECT", "WHERE") + (a, b) <- diffCoalescedPairs + } yield diffStatement(spelling, venue, a, b, diffPerDocument) + + it should "answer the same definitions over a COALESCE of the date and the timestamp column" in { + diffIndexLoaded + val population = diffCoalescedPopulation + // Non-vacuity, computed over the material: 11 spellings x 4 operand pairs x 2 venues, every + // unit answering a negative, a zero and a positive somewhere, each COALESCE answered by its + // first argument on some document and by its second on another, NULL in both venues, and the + // filters keeping some and dropping some. + population.size shouldBe 11 * 4 * 2 + population.foreach(st => withClue(s"[${st.sql}] ")(Parser(st.sql).isRight shouldBe true)) + diffCoalescedSpellings.map(_.unit).distinct.foreach { u => + val answers = population.filter(_.unit == u).flatMap(_.expected.values.flatten) + withClue(s"$u: ") { + Seq(answers.exists(_ < 0), answers.contains(0L), answers.exists(_ > 0)) shouldBe + Seq(true, true, true) + } + } + diffCoalesces.foreach { case (x, y) => + withClue(s"COALESCE(${x.sql}, ${y.sql}): ") { + Seq( + diffPerDocument.exists { case (_, docs) => x.value(docs).isDefined }, + diffPerDocument.exists { case (_, docs) => + x.value(docs).isEmpty && y.value(docs).isDefined + } + ) shouldBe Seq(true, true) + } + } + Seq("SELECT", "WHERE").foreach { venue => + withClue(s"$venue: ") { + population.filter(_.venue == venue).exists(_.expected.values.exists(_.isEmpty)) shouldBe + true + } + } + population + .filter(_.filters) + .exists(st => st.kept.nonEmpty && st.kept != st.expected.keySet) shouldBe + true + + assertDiffPopulation("DATEDIFF family over COALESCE", population) + } + // ----------------------------------------------------------------------------------------------- // TIMESTAMPDIFF over an operand of ANOTHER runtime type. The function compares instants, and an // operand's declared type does not say which Java type it holds: `'…'::DATETIME` is a