diff --git a/bridge/src/test/scala/app/softnetwork/elastic/sql/SQLQuerySpec.scala b/bridge/src/test/scala/app/softnetwork/elastic/sql/SQLQuerySpec.scala index 785d3234d..b0f1f2888 100644 --- a/bridge/src/test/scala/app/softnetwork/elastic/sql/SQLQuerySpec.scala +++ b/bridge/src/test/scala/app/softnetwork/elastic/sql/SQLQuerySpec.scala @@ -1421,13 +1421,13 @@ class SQLQuerySpec extends AnyFlatSpec with Matchers { | "w": { | "script": { | "lang": "painless", - | "source": "def param1 = (doc['lastUpdated'].size() == 0 ? null : doc['lastUpdated'].value.toInstant().atZone(ZoneId.of('Z')).with(DayOfWeek.SUNDAY).truncatedTo(ChronoUnit.DAYS)); def param2 = DateTimeFormatter.ofPattern(\"yyyy-MM-dd\"); (param1 == null) ? null : param2.format(param1)" + | "source": "def param1 = (doc['lastUpdated'].size() == 0 ? null : doc['lastUpdated'].value.toInstant().atZone(ZoneId.of('Z')).with(DayOfWeek.MONDAY).truncatedTo(ChronoUnit.DAYS)); def param2 = DateTimeFormatter.ofPattern(\"yyyy-MM-dd\"); (param1 == null) ? null : param2.format(param1)" | } | }, | "w2": { | "script": { | "lang": "painless", - | "source": "def param1 = (doc['lastUpdated'].size() == 0 ? null : doc['lastUpdated'].value.toInstant().atZone(ZoneId.of('Z')).with(DayOfWeek.SUNDAY).truncatedTo(ChronoUnit.DAYS)); def param2 = DateTimeFormatter.ofPattern(\"yyyy-MM-dd\"); (param1 == null) ? null : param2.format(param1)" + | "source": "def param1 = (doc['lastUpdated'].size() == 0 ? null : doc['lastUpdated'].value.toInstant().atZone(ZoneId.of('Z')).with(DayOfWeek.MONDAY).truncatedTo(ChronoUnit.DAYS)); def param2 = DateTimeFormatter.ofPattern(\"yyyy-MM-dd\"); (param1 == null) ? null : param2.format(param1)" | } | }, | "d": { @@ -1655,7 +1655,7 @@ class SQLQuerySpec extends AnyFlatSpec with Matchers { | "diff": { | "script": { | "lang": "painless", - | "source": "def param1 = (doc['createdAt'].size() == 0 ? null : doc['createdAt'].value.toInstant().atZone(ZoneId.of('Z'))); def param2 = (doc['updatedAt'].size() == 0 ? null : doc['updatedAt'].value.toInstant().atZone(ZoneId.of('Z'))); (param1 == null || param2 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param1.toLocalDate(), param2.toLocalDate()))" + | "source": "def param1 = (doc['updatedAt'].size() == 0 ? null : doc['updatedAt'].value.toInstant().atZone(ZoneId.of('Z'))); def param2 = (doc['createdAt'].size() == 0 ? null : doc['createdAt'].value.toInstant().atZone(ZoneId.of('Z'))); (param1 == null || param2 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param1.toLocalDate(), param2.toLocalDate()))" | } | } | }, @@ -1712,7 +1712,7 @@ class SQLQuerySpec extends AnyFlatSpec with Matchers { | "max": { | "script": { | "lang": "painless", - | "source": "def param1 = (doc['createdAt'].size() == 0 ? null : doc['createdAt'].value); def param2 = (doc['updatedAt'].size() == 0 ? null : doc['updatedAt'].value.toInstant().atZone(ZoneId.of('Z'))); def param3 = (param1 == null) ? null : ZonedDateTime.parse(param1, new DateTimeFormatterBuilder().appendPattern(\"yyyy-MM-dd HH:mm:ss\").appendFraction(ChronoField.NANO_OF_SECOND, 0, 9, true).toFormatter().withZone(ZoneId.of('Z'))); (param1 == null || param2 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param3.toLocalDate(), param2.toLocalDate()))" + | "source": "def param1 = (doc['updatedAt'].size() == 0 ? null : doc['updatedAt'].value.toInstant().atZone(ZoneId.of('Z'))); def param2 = (doc['createdAt'].size() == 0 ? null : doc['createdAt'].value); def param3 = (param2 == null) ? null : ZonedDateTime.parse(param2, new DateTimeFormatterBuilder().appendPattern(\"yyyy-MM-dd HH:mm:ss\").appendFraction(ChronoField.NANO_OF_SECOND, 0, 9, true).toFormatter().withZone(ZoneId.of('Z'))); (param1 == null || param2 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param1.toLocalDate(), param3.toLocalDate()))" | } | } | } diff --git a/core/src/main/resources/help/commands/ddl/create_table.json b/core/src/main/resources/help/commands/ddl/create_table.json index ef23b06ef..3af497039 100644 --- a/core/src/main/resources/help/commands/ddl/create_table.json +++ b/core/src/main/resources/help/commands/ddl/create_table.json @@ -126,7 +126,7 @@ { "title": "Table with computed column", "description": "Create table with script-based computed field", - "sql": "CREATE TABLE employees (\n id INT NOT NULL,\n birthdate DATE,\n age INT SCRIPT AS (DATEDIFF(birthdate, CURRENT_DATE, YEAR)),\n hire_date DATE,\n tenure INT SCRIPT AS (DATEDIFF(hire_date, CURRENT_DATE, DAY)),\n PRIMARY KEY (id)\n)" + "sql": "CREATE TABLE employees (\n id INT NOT NULL,\n birthdate DATE,\n age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)),\n hire_date DATE,\n tenure INT SCRIPT AS (DATEDIFF(CURRENT_DATE, hire_date, DAY)),\n PRIMARY KEY (id)\n)" }, { "title": "Partitioned table", diff --git a/core/src/main/resources/help/functions/date/date_diff.json b/core/src/main/resources/help/functions/date/date_diff.json index aee75ba0e..aeec1fa14 100644 --- a/core/src/main/resources/help/functions/date/date_diff.json +++ b/core/src/main/resources/help/functions/date/date_diff.json @@ -7,26 +7,26 @@ "DATE_DIFF(unit, date1, date2)", "TIMESTAMPDIFF(unit, date1, date2)" ], - "description": "Returns the difference between two dates in the specified unit, as date2 - date1: the first argument is the START and the second is the END.", + "description": "Returns the difference between two dates in the specified unit. With the dates first it is date1 - date2, as in BigQuery's DATE_DIFF(end_date, start_date, unit); with the unit first it is date2 - date1. DATE_DIFF counts the calendar boundaries crossed; TIMESTAMPDIFF counts the whole units elapsed, as MySQL does.", "parameters": [ { "name": "date1", "type": "DATE/TIMESTAMP", - "description": "First date (start)", + "description": "First date: the end when the dates come first, the start when the unit comes first", "optional": false, "defaultValue": null }, { "name": "date2", "type": "DATE/TIMESTAMP", - "description": "Second date (end)", + "description": "Second date: the start when the dates come first, the end when the unit comes first", "optional": false, "defaultValue": null }, { "name": "unit", "type": "KEYWORD", - "description": "Unit for result (YEAR, MONTH, WEEK, DAY, HOUR, MINUTE, SECOND)", + "description": "Unit for result (YEAR, QUARTER, MONTH, WEEK, DAY, HOUR, MINUTE, SECOND)", "optional": true, "defaultValue": "DAY" } @@ -34,33 +34,39 @@ "returnType": "BIGINT", "examples": [ { - "title": "Calculate age", - "description": "Years between birthdate and now", - "sql": "SELECT DATE_DIFF(birthdate, CURRENT_DATE, YEAR) as age FROM users" + "title": "BigQuery's own documented example", + "description": "The end date first: returns 559", + "sql": "SELECT DATE_DIFF('2010-07-07'::DATE, '2008-12-25'::DATE, DAY) as days FROM orders" }, { "title": "Days since event", - "description": "Calculate days elapsed", - "sql": "SELECT DATE_DIFF(event_date, CURRENT_DATE, DAY) as days_ago FROM events" + "description": "Calendar days from the event to today, positive for a past event", + "sql": "SELECT DATE_DIFF(CURRENT_DATE, event_date, DAY) as days_ago FROM events" }, { "title": "Hours between timestamps", - "description": "Calculate duration", - "sql": "SELECT DATE_DIFF(start_time, end_time, HOUR) as duration_hours FROM sessions" + "description": "The hour boundaries crossed from start_time to end_time", + "sql": "SELECT DATE_DIFF(HOUR, start_time, end_time) as duration_hours FROM sessions" + }, + { + "title": "Calculate age", + "description": "The whole years elapsed since birthdate, which TIMESTAMPDIFF counts", + "sql": "SELECT TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE) as age FROM users" } ], "notes": [ - "Returns a negative value if date2 is before date1", - "Truncates to whole units (no fractions)", - "DATEDIFF is NOT an alias of this function. It is MySQL's, and subtracts the other way round (expr1 - expr2). See HELP DATEDIFF.", - "TIMESTAMPDIFF is the ODBC/JDBC spelling and takes the unit first; MySQL defines it as dt2 - dt1, which is what this function already computes." + "Every vendor computes end - start, and the layout decides where the end sits: the dates first is date1 - date2, the unit first is date2 - date1.", + "DATE_DIFF and DATEDIFF count the calendar boundaries crossed: one second across a year end is 1 YEAR, and 1992-09-15 to 1992-11-14 is 2 MONTHs.", + "TIMESTAMPDIFF counts the whole units elapsed, truncated toward zero: 1992-09-15 to 1992-11-14 is 1 MONTH. A DATE is the datetime at 00:00:00.", + "Boundaries are in UTC. A WEEK starts on Monday (ISO 8601); a QUARTER on January 1, April 1, July 1 and October 1.", + "DATEDIFF follows the same rules, and also takes MySQL's two-argument DATEDIFF(date1, date2), in days. See HELP DATEDIFF.", + "TIMESTAMPDIFF is not an alias of DATE_DIFF: it takes the same layouts but counts the whole units elapsed, where DATE_DIFF counts the boundaries crossed. TIMESTAMPDIFF(YEAR, '2025-12-31 23:59:59', '2026-01-01 00:00:00') is 0; DATE_DIFF(YEAR, '2025-12-31 23:59:59', '2026-01-01 00:00:00') is 1.", + "Changed in 0.24.0: the dates-first forms returned date2 - date1, and the units from WEEK up counted the whole units elapsed between two calendar dates. HOUR, MINUTE and SECOND count the boundaries crossed too: DATE_DIFF(HOUR, '2025-01-10 10:59:00', '2025-01-10 11:00:00') is 1." ], "seeAlso": [ "DATEDIFF", "DATE_ADD", "DATE_SUB" ], - "aliases": [ - "TIMESTAMPDIFF" - ] + "aliases": [] } diff --git a/core/src/main/resources/help/functions/date/datediff.json b/core/src/main/resources/help/functions/date/datediff.json index 965348d52..60bcad8c0 100644 --- a/core/src/main/resources/help/functions/date/datediff.json +++ b/core/src/main/resources/help/functions/date/datediff.json @@ -1,32 +1,32 @@ { "name": "DATEDIFF", "category": "Date/Time", - "shortDescription": "MySQL day difference: expr1 - expr2", + "shortDescription": "Difference between dates: date1 - date2 with the dates first", "syntax": [ "DATEDIFF(date1, date2)", "DATEDIFF(date1, date2, unit)", "DATEDIFF(unit, date1, date2)" ], - "description": "MySQL's day-difference function. DATEDIFF(date1, date2) returns date1 - date2 in DAYS, which is the opposite subtraction from DATE_DIFF. They are different functions with similar names.", + "description": "DATEDIFF(date1, date2) is MySQL's: date1 - date2 in DAYS. With a unit, the dates first is still date1 - date2, and the unit first - the spelling of SQL Server, Snowflake, Redshift, DuckDB and Elasticsearch SQL - is date2 - date1. It counts the calendar boundaries crossed, as DATE_DIFF does.", "parameters": [ { "name": "date1", "type": "DATE/TIMESTAMP", - "description": "Date subtracted FROM", + "description": "The date subtracted FROM when the dates come first; the start when the unit comes first", "optional": false, "defaultValue": null }, { "name": "date2", "type": "DATE/TIMESTAMP", - "description": "Date subtracted", + "description": "The date subtracted when the dates come first; the end when the unit comes first", "optional": false, "defaultValue": null }, { "name": "unit", "type": "KEYWORD", - "description": "Unit for result. NOT MySQL - this engine's own extension, and it reverses the subtraction to date2 - date1", + "description": "Unit for result (YEAR, QUARTER, MONTH, WEEK, DAY, HOUR, MINUTE, SECOND)", "optional": true, "defaultValue": "DAY" } @@ -44,16 +44,15 @@ "sql": "SELECT DATEDIFF(due_date, CURRENT_DATE) as days_left FROM tasks" }, { - "title": "The three-argument form reverses it", - "description": "This form is not MySQL and follows DATE_DIFF instead", - "sql": "SELECT DATEDIFF(start_date, end_date, MONTH) as months FROM projects" + "title": "The unit first", + "description": "The month boundaries crossed from start_date to end_date, as SQL Server counts them", + "sql": "SELECT DATEDIFF(MONTH, start_date, end_date) as months FROM projects" } ], "notes": [ - "WARNING: the two-argument and three-argument forms subtract in OPPOSITE directions. DATEDIFF(a, b) is a - b (MySQL, days only); DATEDIFF(a, b, unit) is b - a (this engine's own extension). Adding a unit reverses the sign.", - "If you want a unit, prefer DATE_DIFF(start, end, unit) or TIMESTAMPDIFF(unit, start, end), where one rule holds at every arity.", - "DATEDIFF(unit, date1, date2) - the unit FIRST - is T-SQL's and DuckDB's spelling and means date2 - date1, like TIMESTAMPDIFF. Only the two-argument form is MySQL's.", - "Before engine 0.24.0 this name was an alias of DATE_DIFF and returned the opposite sign. Statements written against the old behaviour need their arguments swapped." + "The dates first is date1 - date2, with or without a unit; the unit first is date2 - date1.", + "DATEDIFF counts the calendar boundaries crossed, in UTC, a WEEK starting on Monday: DATEDIFF(YEAR, '2025-12-31 23:59:59', '2026-01-01 00:00:00') is 1. For the whole units elapsed, use TIMESTAMPDIFF(unit, date1, date2).", + "Changed in 0.24.0: this name was an alias of DATE_DIFF; every dates-first form returned date2 - date1, and from WEEK up it counted the whole units elapsed between the two calendar dates. Swapping the dates of a statement written against the old behaviour restores its sign, not its count." ], "seeAlso": [ "DATE_DIFF", diff --git a/documentation/sql/ddl_statements.md b/documentation/sql/ddl_statements.md index 852c189c5..3c032b3eb 100644 --- a/documentation/sql/ddl_statements.md +++ b/documentation/sql/ddl_statements.md @@ -176,7 +176,7 @@ CREATE TABLE users ( zip VARCHAR ), join_date DATE, - seniority INT SCRIPT AS (DATEDIFF(profile.join_date, CURRENT_DATE, DAY)) + seniority INT SCRIPT AS (DATEDIFF(CURRENT_DATE, profile.join_date, DAY)) ) ) ``` @@ -304,7 +304,7 @@ CREATE TABLE users ( id INT, name VARCHAR DEFAULT 'anonymous', birthdate DATE, - age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR)), + age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)), PRIMARY KEY (id) ); ``` @@ -399,7 +399,7 @@ value derived for it. CREATE TABLE users ( id INT, birthdate DATE, - age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR)) + age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)) ); ``` @@ -410,7 +410,7 @@ computing it here: CREATE TABLE users_view ( id INT, birthdate DATE, - age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR)) STORED + age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)) STORED ); ``` @@ -705,7 +705,7 @@ WITH PROCESSORS ( value = "anonymous" ), SCRIPT ( - description = "age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR))", + description = "age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE))", lang = "painless", source = "...", ignore_failure = true diff --git a/documentation/sql/dql_statements.md b/documentation/sql/dql_statements.md index 85fe3cd07..1e5db4dc3 100644 --- a/documentation/sql/dql_statements.md +++ b/documentation/sql/dql_statements.md @@ -1297,7 +1297,7 @@ FROM dql_users; | `DATE_SUB(date, INTERVAL n unit)` | Subtract interval | | `DATETIME_ADD(ts, INTERVAL n unit)` | Add interval to timestamp | | `DATETIME_SUB(ts, INTERVAL n unit)` | Subtract interval from timestamp | -| `DATE_DIFF(date1, date2, unit)` | Difference in units: elapsed `HOUR` / `MINUTE` / `SECOND`, calendar dates from `DAY` up (UTC) | +| `DATE_DIFF(date1, date2, unit)` | `date1 - date2` in units: the calendar boundaries crossed (UTC, weeks start on Monday) | | `DATE_TRUNC(date, unit)` | Truncate to unit | ##### **Formatting & parsing:** @@ -1327,7 +1327,7 @@ SELECT id, MONTH(CURRENT_DATE) AS current_month, DAY(CURRENT_DATE) AS current_day, YEAR(birthdate) AS year_b, - DATE_DIFF(birthdate, CURRENT_DATE, YEAR) AS diff_years, + DATE_DIFF(CURRENT_DATE, birthdate, YEAR) AS diff_years, DATE_TRUNC(birthdate, MONTH) AS trunc_month, DATETIME_FORMAT(birthdate, '%Y-%m-%d') AS birth_str FROM dql_users; @@ -1561,13 +1561,13 @@ CREATE TABLE IF NOT EXISTS users ( id INT NOT NULL COMMENT 'user identifier', name VARCHAR FIELDS(raw Keyword COMMENT 'sortable') DEFAULT 'anonymous' OPTIONS (analyzer = 'french', search_analyzer = 'french'), birthdate DATE, - age INT SCRIPT AS (DATEDIFF(birthdate, CURRENT_DATE, YEAR)), + age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)), ingested_at TIMESTAMP DEFAULT _ingest.timestamp, profile STRUCT FIELDS( bio VARCHAR, followers INT, join_date DATE, - seniority INT SCRIPT AS (DATEDIFF(profile.join_date, CURRENT_DATE, DAY)) + seniority INT SCRIPT AS (DATEDIFF(CURRENT_DATE, profile.join_date, DAY)) ) COMMENT 'user profile', PRIMARY KEY (id) ) PARTITION BY birthdate (MONTH), OPTIONS (mappings = (dynamic = false)); @@ -1579,14 +1579,14 @@ SHOW TABLE users; | Field | Type | Null | Key | Default | Comment | Script | Extra | |-------------------|-----------|------|-----|-------------------|-----------------|-------------------------------------------------|---------------------------------------------------| -| age | INT | yes | | NULL | | DATE_DIFF(birthdate, CURRENT_DATE, YEAR) | () | +| age | INT | yes | | NULL | | TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE) | () | | birthdate | DATE | yes | | NULL | | | () | | id | INT | no | PRI | NULL | user identifier | | () | | ingested_at | TIMESTAMP | yes | | _ingest.timestamp | | | () | | name | VARCHAR | yes | | anonymous | | | (analyzer = "french", search_analyzer = "french") | | name.raw | KEYWORD | yes | | NULL | sortable | | () | | profile | STRUCT | yes | | NULL | user profile | | () | -| profile.seniority | INT | yes | | NULL | | DATE_DIFF(profile.join_date, CURRENT_DATE, DAY) | () | +| profile.seniority | INT | yes | | NULL | | DATE_DIFF(CURRENT_DATE, profile.join_date, DAY) | () | | profile.join_date | DATE | yes | | NULL | | | () | | profile.followers | INT | yes | | NULL | | | () | | profile.bio | VARCHAR | yes | | NULL | | | () | @@ -1606,7 +1606,7 @@ _meta: (primary_key = ('id'), partition_by = (column = 'birthdate', granularity 📝 DDL: ```sql CREATE OR REPLACE TABLE users ( - age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR)), + age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)), birthdate DATE, id INT NOT NULL COMMENT 'user identifier', ingested_at TIMESTAMP DEFAULT _ingest.timestamp, @@ -1614,7 +1614,7 @@ CREATE OR REPLACE TABLE users ( raw KEYWORD COMMENT 'sortable' ) DEFAULT 'anonymous' OPTIONS (analyzer = "french", search_analyzer = "french"), profile STRUCT FIELDS ( - seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY)), + seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY)), join_date DATE, followers INT, bio VARCHAR @@ -1640,7 +1640,7 @@ SHOW CREATE TABLE users; ```sql CREATE OR REPLACE TABLE users ( - age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR)), + age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)), birthdate DATE, id INT NOT NULL COMMENT 'user identifier', ingested_at TIMESTAMP DEFAULT _ingest.timestamp, @@ -1648,7 +1648,7 @@ CREATE OR REPLACE TABLE users ( raw KEYWORD COMMENT 'sortable' ) DEFAULT 'anonymous' OPTIONS (analyzer = "french", search_analyzer = "french"), profile STRUCT FIELDS ( - seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY)), + seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY)), join_date DATE, followers INT, bio VARCHAR @@ -1682,14 +1682,14 @@ Returns the **normalized SQL schema**, including : | Field | Type | Null | Key | Default | Comment | Script | Extra | |-------------------|-----------|------|-----|-------------------|-----------------|-------------------------------------------------|---------------------------------------------------| -| age | INT | yes | | NULL | | DATE_DIFF(birthdate, CURRENT_DATE, YEAR) | () | +| age | INT | yes | | NULL | | TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE) | () | | birthdate | DATE | yes | | NULL | | | () | | id | INT | no | PRI | NULL | user identifier | | () | | ingested_at | TIMESTAMP | yes | | _ingest.timestamp | | | () | | name | VARCHAR | yes | | anonymous | | | (analyzer = "french", search_analyzer = "french") | | name.raw | KEYWORD | yes | | NULL | sortable | | () | | profile | STRUCT | yes | | NULL | user profile | | () | -| profile.seniority | INT | yes | | NULL | | DATE_DIFF(profile.join_date, CURRENT_DATE, DAY) | () | +| profile.seniority | INT | yes | | NULL | | DATE_DIFF(CURRENT_DATE, profile.join_date, DAY) | () | | profile.join_date | DATE | yes | | NULL | | | () | | profile.followers | INT | yes | | NULL | | | () | | profile.bio | VARCHAR | yes | | NULL | | | () | @@ -1787,9 +1787,9 @@ Processors: (6) | processor_type | description | field | ignore_failure | options | |-----------------|-----------------------------------------------------------------------------------|-------------------|----------------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| | set | DEFAULT 'anonymous' | name | yes | (value = "anonymous", if = "ctx.name == null") | -| script | age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR)) | age | yes | (lang = "painless", source = "def param1 = ctx.birthdate; def param2 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(ctx['_ingest']['timestamp']), ZoneId.of('Z')).toLocalDate(); ctx.age = (param1 == null) ? null : Long.valueOf(ChronoUnit.YEARS.between(param1, param2))") | +| script | age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)) | age | yes | (lang = "painless", source = "def param1 = ctx.birthdate; def param2 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(ctx['_ingest']['timestamp']), ZoneId.of('Z')).toLocalDate(); ctx.age = (param1 == null) ? null : Long.valueOf(ChronoUnit.YEARS.between(param1, param2))") | | set | DEFAULT _ingest.timestamp | ingested_at | yes | (value = "_ingest.timestamp", if = "ctx.ingested_at == null") | -| script | profile.seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY)) | profile.seniority | yes | (lang = "painless", source = "def param1 = ctx.profile?.join_date; def param2 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(ctx['_ingest']['timestamp']), ZoneId.of('Z')).toLocalDate(); ctx.profile.seniority = (param1 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param1, param2))") | +| script | profile.seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY)) | profile.seniority | yes | (lang = "painless", source = "def param1 = ctx.profile?.join_date; def param2 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(ctx['_ingest']['timestamp']), ZoneId.of('Z')).toLocalDate(); ctx.profile.seniority = (param1 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param1, param2))") | | date_index_name | PARTITION BY birthdate (MONTH) | birthdate | yes | (date_rounding = "M", date_formats = ["yyyy-MM"], index_name_prefix = "users-") | | set | PRIMARY KEY (id) | _id | no | (value = "{{id}}", ignore_empty_value = false) | @@ -1804,7 +1804,7 @@ CREATE OR REPLACE PIPELINE user_pipeline WITH PROCESSORS ( if = "ctx.name == null" ), SCRIPT( - description = "age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR))", + description = "age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE))", lang = "painless", source = "...", ignore_failure = true @@ -1817,7 +1817,7 @@ CREATE OR REPLACE PIPELINE user_pipeline WITH PROCESSORS ( if = "ctx.ingested_at == null" ), SCRIPT( - description = "profile.seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY))", + description = "profile.seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY))", lang = "painless", source = "...", ignore_failure = true @@ -1868,7 +1868,7 @@ CREATE OR REPLACE PIPELINE user_pipeline WITH PROCESSORS ( if = "ctx.name == null" ), SCRIPT( - description = "age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR))", + description = "age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE))", lang = "painless", source = "...", ignore_failure = true @@ -1881,7 +1881,7 @@ CREATE OR REPLACE PIPELINE user_pipeline WITH PROCESSORS ( if = "ctx.ingested_at == null" ), SCRIPT( - description = "profile.seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY))", + description = "profile.seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY))", lang = "painless", source = "...", ignore_failure = true @@ -1928,9 +1928,9 @@ DESCRIBE PIPELINE user_pipeline; | processor_type | description | field | ignore_failure | options | |-----------------|-----------------------------------------------------------------------------------|-------------------|----------------|-----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------| | set | DEFAULT 'anonymous' | name | yes | (value = "anonymous", if = "ctx.name == null") | -| script | age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR)) | age | yes | (lang = "painless", source = "def param1 = ctx.birthdate; def param2 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(ctx['_ingest']['timestamp']), ZoneId.of('Z')).toLocalDate(); ctx.age = (param1 == null) ? null : Long.valueOf(ChronoUnit.YEARS.between(param1, param2))") | +| script | age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)) | age | yes | (lang = "painless", source = "def param1 = ctx.birthdate; def param2 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(ctx['_ingest']['timestamp']), ZoneId.of('Z')).toLocalDate(); ctx.age = (param1 == null) ? null : Long.valueOf(ChronoUnit.YEARS.between(param1, param2))") | | set | DEFAULT _ingest.timestamp | ingested_at | yes | (value = "_ingest.timestamp", if = "ctx.ingested_at == null") | -| script | profile.seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY)) | profile.seniority | yes | (lang = "painless", source = "def param1 = ctx.profile?.join_date; def param2 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(ctx['_ingest']['timestamp']), ZoneId.of('Z')).toLocalDate(); ctx.profile.seniority = (param1 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param1, param2))") | +| script | profile.seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY)) | profile.seniority | yes | (lang = "painless", source = "def param1 = ctx.profile?.join_date; def param2 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(ctx['_ingest']['timestamp']), ZoneId.of('Z')).toLocalDate(); ctx.profile.seniority = (param1 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param1, param2))") | | date_index_name | PARTITION BY birthdate (MONTH) | birthdate | yes | (date_rounding = "M", date_formats = ["yyyy-MM"], index_name_prefix = "users-") | | set | PRIMARY KEY (id) | _id | no | (value = "{{id}}", ignore_empty_value = false) | 📊 6 row(s) (1ms) diff --git a/documentation/sql/functions_date_time.md b/documentation/sql/functions_date_time.md index b8e312834..baecc694f 100644 --- a/documentation/sql/functions_date_time.md +++ b/documentation/sql/functions_date_time.md @@ -316,70 +316,158 @@ SELECT DATETIME_SUB('2025-01-10T12:00:00Z'::TIMESTAMP, INTERVAL 1 MONTH) AS last ### Date/Time Difference Functions -#### DATEDIFF / DATE_DIFF +#### DATEDIFF / DATE_DIFF / TIMESTAMPDIFF -Difference between 2 dates (date2 - date1) in the specified time unit: `date1` is the start and `date2` the end. -MySQL's two-argument `DATEDIFF(a, b)` gives `a - b`. +Difference between 2 dates in a time unit. These functions follow the definitions of the databases they come from: the **layout** of the arguments decides which date is subtracted from which, and the **name** decides what is counted. **Syntax:** ```sql -DATEDIFF(date1, date2) +-- the dates first: date1 - date2 +DATEDIFF(date1, date2) -- MySQL: in days DATEDIFF(date1, date2, unit) -DATE_DIFF(date1, date2) +DATE_DIFF(date1, date2) -- BigQuery: DATE_DIFF(end_date, start_date, unit) DATE_DIFF(date1, date2, unit) + +-- the unit first: date2 - date1 +DATEDIFF(unit, date1, date2) -- SQL Server, Snowflake, Redshift, DuckDB, Elasticsearch SQL +DATE_DIFF(unit, date1, date2) +TIMESTAMPDIFF(unit, date1, date2) -- MySQL ``` **Inputs:** -- `date1` - `DATE` or `DATETIME` -- `date2` - `DATE` or `DATETIME` -- `unit` (optional) - One of: `YEAR`, `QUARTER`, `MONTH`, `WEEK`, `DAY`, `HOUR`, `MINUTE`, `SECOND` - - Default: `DAY` +- `date1` - `DATE`, `DATETIME` or `TIMESTAMP` +- `date2` - `DATE`, `DATETIME` or `TIMESTAMP` +- `unit` - One of: `YEAR`, `QUARTER`, `MONTH`, `WEEK`, `DAY`, `HOUR`, `MINUTE`, `SECOND` (or their ODBC names, `SQL_TSI_YEAR` …) + - Default, when the dates come first: `DAY` **Output:** - `BIGINT` -**Units:** -- `HOUR`, `MINUTE`, `SECOND` count the elapsed whole units between the two instants, in UTC, truncated toward zero. A `DATE` operand counts from the start of its day (00:00 UTC). -- `DAY`, `WEEK`, `MONTH`, `QUARTER`, `YEAR` compare the two calendar dates, in UTC, whatever the time of day: `2025-01-10T23:30:00Z` and `2025-01-11T00:30:00Z` are 1 day apart, as MySQL's `DATEDIFF` counts them. A week is 7 whole days, a month counts once its day of month is reached, a quarter is 3 whole months and a year 12; every count is truncated toward zero. +**Direction, by layout:** every one of these databases computes *end − start*; the layout decides where the end sits. +- The dates first: `date1 - date2`. `DATEDIFF('2025-01-10', '2025-01-01')` is `9`, as in MySQL; `DATE_DIFF('2010-07-07', '2008-12-25', DAY)` is `559`, as in BigQuery. +- The unit first: `date2 - date1`. `DATEDIFF(DAY, '2025-01-01', '2025-01-10')` is `9`. + +**Counting, by name:** +- `DATEDIFF` and `DATE_DIFF` count the calendar **boundaries crossed**, as SQL Server, Snowflake, Redshift, BigQuery and DuckDB do: one second across a year end is 1 `YEAR`, `1992-09-15` to `1992-11-14` is 2 `MONTH`s, `10:59` to `11:00` is 1 `HOUR`. `DAY` counts calendar days: `2025-01-10T23:30:00Z` and `2025-01-11T00:30:00Z` are 1 day apart. +- `TIMESTAMPDIFF` counts the **whole units elapsed**, truncated toward zero, as MySQL does: a month counts once the same day of month and time of day are reached, so `2003-02-01` to `2003-05-01` is 3 `MONTH`s and `1992-09-15` to `1992-11-14` is 1. A `DATE` operand is the datetime at `00:00:00`. + +**Calendar:** +- Every boundary is in UTC. +- A `WEEK` starts on **Monday** (ISO 8601): the `WEEK` boundaries are the Mondays crossed, so a Saturday and the next Sunday are 0 weeks apart. A `QUARTER` starts on January 1, April 1, July 1 and October 1. **Literals:** - A string literal is read as the temporal it spells: `'2025-01-10'` (or `'2025/01/10'`) is a `DATE`; a literal with a time of day (`'2025-01-10 14:00:00'`, `'2025-01-10T14:00:00Z'`) is a `TIMESTAMP`, in UTC unless it names a zone. -- This holds in every clause, for each row and for each group: `DATEDIFF(MAX(created_at), '2025-01-10 08:00:00', HOUR)`. +- This holds in every clause, for each row and for each group, and in a computed column (`SCRIPT AS`): `DATEDIFF(MAX(created_at), '2025-01-10 08:00:00', HOUR)`. - A `NULL` operand, or an aggregate over a group that has no value, gives `NULL`. +> 🔴 **Changed in 0.24.0 — these functions follow their vendors' definitions.** Statements written +> against the old behaviour return other values, without an error: +> - **The dates first subtract the second date from the first.** `DATEDIFF(a, b)`, +> `DATEDIFF(a, b, unit)`, `DATE_DIFF(a, b)` and `DATE_DIFF(a, b, unit)` returned `b - a`; they +> return `a - b`, as MySQL's `DATEDIFF` and BigQuery's `DATE_DIFF` do. The unit-first forms +> (`DATEDIFF(unit, a, b)`, `DATE_DIFF(unit, a, b)`) keep `b - a`. +> - **`DATEDIFF` and `DATE_DIFF` count the boundaries crossed.** From `WEEK` up they counted the +> whole units elapsed between the two calendar dates (a week was 7 days, a month counted once its +> day of month was reached): `DATE_DIFF(MONTH, '1992-09-15', '1992-11-14')` was `1` and is `2`, and +> a `WEEK` boundary is now a Monday. `HOUR`, `MINUTE` and `SECOND` count the boundaries crossed too: +> `DATEDIFF(HOUR, '2025-01-10 10:59:00', '2025-01-10 11:00:00')` is `1`. `DAY` still counts +> calendar days. +> - **A computed column answers what a query answers.** In a `SCRIPT AS` column these functions +> counted the whole units elapsed between two instants whatever the unit (a `DAY` was 24 hours), and +> a string literal operand made the `CREATE TABLE` fail. +> +> New in 0.24.0, where they failed or were refused before: `TIMESTAMPDIFF`, the `QUARTER` unit, +> `HOUR` / `MINUTE` / `SECOND` over a column or an aggregate, and a string literal operand in a query. +> +> **What was deployed before 0.24.0 keeps its 0.23 results until it is re-created.** A computed +> column (`SCRIPT AS`), an ingest pipeline and a materialized view keep the script they were +> deployed with: adding a column with `ALTER TABLE`, `SHOW CREATE` and re-running the pipeline text +> that `SHOW CREATE PIPELINE` returns leave it as it was. +> - Re-creating it from the same SQL adopts the new definition, and these shapes then return other +> values: `DATEDIFF(a, b[, unit])` and `DATE_DIFF(a, b[, unit])` (the sign, and the count), +> `DATE_DIFF(unit, a, b)` (the count), `DATE_TRUNC(x, WEEK)` (the Monday), and `DAY` in a computed +> column (calendar days). +> - `CREATE OR REPLACE MATERIALIZED VIEW` with an unchanged text keeps the old results: to adopt the +> new definition, `DROP` the view and `CREATE` it. +> - `SHOW CREATE` shows the stored text, and that text now means the new definition. +> - To keep a 0.23 value, swap the dates: `DATE_DIFF(a, b, unit)` becomes `DATE_DIFF(b, a, unit)`, +> which restores the sign. Where it counted whole units elapsed, use `TIMESTAMPDIFF`: +> `TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)` is the age that +> `DATE_DIFF(birthdate, CURRENT_DATE, YEAR)` returned. + **Examples:** ```sql --- Difference in days (default), MySQL's two-argument form: date1 - date2 +-- MySQL's two-argument form, in days: date1 - date2 SELECT DATEDIFF('2025-01-10'::DATE, '2025-01-01'::DATE) AS diff; -- Result: 9 --- Difference in days (explicit) -SELECT DATEDIFF('2025-01-01'::DATE, '2025-01-10'::DATE, DAY) AS diff_days; +-- The dates first, with a unit: still date1 - date2 +SELECT DATEDIFF('2025-01-10'::DATE, '2025-01-01'::DATE, DAY) AS diff_days; -- Result: 9 --- Difference in weeks -SELECT DATE_DIFF('2025-01-01'::DATE, '2025-01-31'::DATE, WEEK) AS diff_weeks; +-- BigQuery's own example: the end date first +SELECT DATE_DIFF('2010-07-07'::DATE, '2008-12-25'::DATE, DAY) AS diff_days; +-- Result: 559 + +-- The unit first: date2 - date1 +SELECT DATEDIFF(DAY, '2025-01-01'::DATE, '2025-01-10'::DATE) AS diff_days; +-- Result: 9 + +-- Difference in weeks: the Mondays crossed (January 6, 13, 20 and 27) +SELECT DATE_DIFF('2025-01-31'::DATE, '2025-01-01'::DATE, WEEK) AS diff_weeks; -- Result: 4 +-- A week starts on Monday: Saturday to Sunday crosses none, Sunday to Monday one +SELECT DATEDIFF(WEEK, '2025-01-11'::DATE, '2025-01-12'::DATE) AS sat_to_sun, + DATEDIFF(WEEK, '2025-01-12'::DATE, '2025-01-13'::DATE) AS sun_to_mon; +-- Result: 0, 1 + -- Difference in months -SELECT DATEDIFF('2025-01-01'::DATE, '2025-06-01'::DATE, MONTH) AS diff_months; +SELECT DATEDIFF('2025-06-01'::DATE, '2025-01-01'::DATE, MONTH) AS diff_months; -- Result: 5 +-- Two month boundaries are crossed, one whole month elapses +SELECT DATE_DIFF(MONTH, '1992-09-15'::DATE, '1992-11-14'::DATE) AS boundaries, + TIMESTAMPDIFF(MONTH, '1992-09-15'::DATE, '1992-11-14'::DATE) AS elapsed; +-- Result: 2, 1 + +-- MySQL's own example +SELECT TIMESTAMPDIFF(MONTH, '2003-02-01', '2003-05-01') AS diff_months; +-- Result: 3 + -- Difference in years -SELECT DATEDIFF('2025-01-01'::DATE, '2027-01-01'::DATE, YEAR) AS diff_years; +SELECT DATEDIFF('2027-01-01'::DATE, '2025-01-01'::DATE, YEAR) AS diff_years; -- Result: 2 +-- One second across a year end +SELECT DATEDIFF(YEAR, '2025-12-31 23:59:59', '2026-01-01 00:00:00') AS boundaries, + TIMESTAMPDIFF(YEAR, '2025-12-31 23:59:59', '2026-01-01 00:00:00') AS elapsed; +-- Result: 1, 0 + -- Difference in hours (with timestamps) -SELECT DATEDIFF('2025-01-10T12:00:00Z'::TIMESTAMP, '2025-01-10T14:00:00Z'::TIMESTAMP, HOUR) AS diff_hours; +SELECT DATEDIFF('2025-01-10T14:00:00Z'::TIMESTAMP, '2025-01-10T12:00:00Z'::TIMESTAMP, HOUR) AS diff_hours; -- Result: 2 +-- One minute across an hour +SELECT DATEDIFF(HOUR, '2025-01-10 10:59:00', '2025-01-10 11:00:00') AS boundaries, + TIMESTAMPDIFF(HOUR, '2025-01-10 10:59:00', '2025-01-10 11:00:00') AS elapsed; +-- Result: 1, 0 + +-- One hour across midnight +SELECT DATEDIFF(DAY, '2025-01-10 23:30:00', '2025-01-11 00:30:00') AS boundaries, + TIMESTAMPDIFF(DAY, '2025-01-10 23:30:00', '2025-01-11 00:30:00') AS elapsed; +-- Result: 1, 0 + -- Difference in minutes -SELECT DATEDIFF('2025-01-10T12:00:00Z'::TIMESTAMP, '2025-01-10T12:30:00Z'::TIMESTAMP, MINUTE) AS diff_minutes; +SELECT DATEDIFF('2025-01-10T12:30:00Z'::TIMESTAMP, '2025-01-10T12:00:00Z'::TIMESTAMP, MINUTE) AS diff_minutes; -- Result: 30 -- Difference in seconds -SELECT DATEDIFF('2025-01-10T12:00:00Z'::TIMESTAMP, '2025-01-10T12:00:45Z'::TIMESTAMP, SECOND) AS diff_seconds; +SELECT DATEDIFF('2025-01-10T12:00:45Z'::TIMESTAMP, '2025-01-10T12:00:00Z'::TIMESTAMP, SECOND) AS diff_seconds; -- Result: 45 + +-- An age in whole years: the years elapsed, not the year boundaries crossed +SELECT TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE) AS age FROM users; ``` --- @@ -618,6 +706,14 @@ DATE_TRUNC(date_or_datetime_expr, unit) **Output:** - `DATE` or `DATETIME` (same type as input) +**Week:** +- `WEEK` truncates to the **Monday** that starts the ISO-8601 week, at 00:00 UTC, in every clause: a `GROUP BY DATE_TRUNC(…, WEEK)` buckets on that same Monday. + +> 🔴 **Changed in 0.24.0 — `DATE_TRUNC(…, WEEK)` is the Monday that starts the week.** Before 0.24.0, a +> `SELECT` item, a `WHERE` condition or a computed column (`SCRIPT AS`) moved a date FORWARD to the +> Sunday that ends its week (`2025-01-15` gave `2025-01-19`), while a `GROUP BY` of the same +> expression bucketed on the Monday. + **Examples:** ```sql -- Truncate to start of month diff --git a/documentation/sql/functions_math.md b/documentation/sql/functions_math.md index 2964cc482..0181f9eb4 100644 --- a/documentation/sql/functions_math.md +++ b/documentation/sql/functions_math.md @@ -206,7 +206,7 @@ SELECT FLOOR(123.999) AS f; SELECT user_id, name, - FLOOR(DATEDIFF(birth_date, CURRENT_DATE, DAY) / 365.25) AS age + FLOOR(DATEDIFF(CURRENT_DATE, birth_date, DAY) / 365.25) AS age FROM users; -- Bucket values diff --git a/es6/bridge/src/test/scala/app/softnetwork/elastic/sql/SQLQuerySpec.scala b/es6/bridge/src/test/scala/app/softnetwork/elastic/sql/SQLQuerySpec.scala index ab09f06a3..6f31f8661 100644 --- a/es6/bridge/src/test/scala/app/softnetwork/elastic/sql/SQLQuerySpec.scala +++ b/es6/bridge/src/test/scala/app/softnetwork/elastic/sql/SQLQuerySpec.scala @@ -1421,13 +1421,13 @@ class SQLQuerySpec extends AnyFlatSpec with Matchers { | "w": { | "script": { | "lang": "painless", - | "source": "def param1 = (doc['lastUpdated'].size() == 0 ? null : doc['lastUpdated'].value.toInstant().atZone(ZoneId.of('Z')).with(DayOfWeek.SUNDAY).truncatedTo(ChronoUnit.DAYS)); def param2 = DateTimeFormatter.ofPattern(\"yyyy-MM-dd\"); (param1 == null) ? null : param2.format(param1)" + | "source": "def param1 = (doc['lastUpdated'].size() == 0 ? null : doc['lastUpdated'].value.toInstant().atZone(ZoneId.of('Z')).with(DayOfWeek.MONDAY).truncatedTo(ChronoUnit.DAYS)); def param2 = DateTimeFormatter.ofPattern(\"yyyy-MM-dd\"); (param1 == null) ? null : param2.format(param1)" | } | }, | "w2": { | "script": { | "lang": "painless", - | "source": "def param1 = (doc['lastUpdated'].size() == 0 ? null : doc['lastUpdated'].value.toInstant().atZone(ZoneId.of('Z')).with(DayOfWeek.SUNDAY).truncatedTo(ChronoUnit.DAYS)); def param2 = DateTimeFormatter.ofPattern(\"yyyy-MM-dd\"); (param1 == null) ? null : param2.format(param1)" + | "source": "def param1 = (doc['lastUpdated'].size() == 0 ? null : doc['lastUpdated'].value.toInstant().atZone(ZoneId.of('Z')).with(DayOfWeek.MONDAY).truncatedTo(ChronoUnit.DAYS)); def param2 = DateTimeFormatter.ofPattern(\"yyyy-MM-dd\"); (param1 == null) ? null : param2.format(param1)" | } | }, | "d": { @@ -1655,7 +1655,7 @@ class SQLQuerySpec extends AnyFlatSpec with Matchers { | "diff": { | "script": { | "lang": "painless", - | "source": "def param1 = (doc['createdAt'].size() == 0 ? null : doc['createdAt'].value.toInstant().atZone(ZoneId.of('Z'))); def param2 = (doc['updatedAt'].size() == 0 ? null : doc['updatedAt'].value.toInstant().atZone(ZoneId.of('Z'))); (param1 == null || param2 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param1.toLocalDate(), param2.toLocalDate()))" + | "source": "def param1 = (doc['updatedAt'].size() == 0 ? null : doc['updatedAt'].value.toInstant().atZone(ZoneId.of('Z'))); def param2 = (doc['createdAt'].size() == 0 ? null : doc['createdAt'].value.toInstant().atZone(ZoneId.of('Z'))); (param1 == null || param2 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param1.toLocalDate(), param2.toLocalDate()))" | } | } | }, @@ -1712,7 +1712,7 @@ class SQLQuerySpec extends AnyFlatSpec with Matchers { | "max": { | "script": { | "lang": "painless", - | "source": "def param1 = (doc['createdAt'].size() == 0 ? null : doc['createdAt'].value); def param2 = (doc['updatedAt'].size() == 0 ? null : doc['updatedAt'].value.toInstant().atZone(ZoneId.of('Z'))); def param3 = (param1 == null) ? null : ZonedDateTime.parse(param1, new DateTimeFormatterBuilder().appendPattern(\"yyyy-MM-dd HH:mm:ss\").appendFraction(ChronoField.NANO_OF_SECOND, 0, 9, true).toFormatter().withZone(ZoneId.of('Z'))); (param1 == null || param2 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param3.toLocalDate(), param2.toLocalDate()))" + | "source": "def param1 = (doc['updatedAt'].size() == 0 ? null : doc['updatedAt'].value.toInstant().atZone(ZoneId.of('Z'))); def param2 = (doc['createdAt'].size() == 0 ? null : doc['createdAt'].value); def param3 = (param2 == null) ? null : ZonedDateTime.parse(param2, new DateTimeFormatterBuilder().appendPattern(\"yyyy-MM-dd HH:mm:ss\").appendFraction(ChronoField.NANO_OF_SECOND, 0, 9, true).toFormatter().withZone(ZoneId.of('Z'))); (param1 == null || param2 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param1.toLocalDate(), param3.toLocalDate()))" | } | } | } diff --git a/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/JestIndicesApi.scala b/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/JestIndicesApi.scala index a9d70c8f8..013ebbe5f 100644 --- a/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/JestIndicesApi.scala +++ b/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/JestIndicesApi.scala @@ -42,6 +42,9 @@ trait JestIndicesApi extends IndicesApi with JestClientHelpers { with JestClientCompanion => /** Create an index with the given settings. + * + * A mapping says `include_type_name=false` where Elasticsearch must be told it is typeless + * ([[sendsTypelessMappings]]): without it 6.8 refused every `CREATE TABLE`. * @see * [[IndicesApi.createIndex]] */ @@ -67,6 +70,7 @@ trait JestIndicesApi extends IndicesApi with JestClientHelpers { } mappings.foreach { mapping => builder.mappings(mapping) + if (sendsTypelessMappings) builder.setParameter("include_type_name", "false") } builder.build() } diff --git a/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/JestTemplateApi.scala b/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/JestTemplateApi.scala index 3f4c2b312..2b8cd53e6 100644 --- a/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/JestTemplateApi.scala +++ b/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/JestTemplateApi.scala @@ -103,7 +103,7 @@ trait JestTemplateApi extends TemplateApi with JestClientHelpers { operation = "createLegacyTemplate", retryable = false // Creation can not be retried ) { - Template.Create(templateName, templateDefinition) + Template.Create(templateName, templateDefinition, typelessMappings = sendsTypelessMappings) } override private[client] def executeDeleteLegacyTemplate( diff --git a/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/JestVersionApi.scala b/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/JestVersionApi.scala index 2640a629f..aa4b2181c 100644 --- a/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/JestVersionApi.scala +++ b/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/JestVersionApi.scala @@ -16,13 +16,38 @@ package app.softnetwork.elastic.client.jest -import app.softnetwork.elastic.client.{result, VersionApi} +import app.softnetwork.elastic.client.{result, ElasticsearchVersion, VersionApi} import io.searchbox.core.Cat import org.json4s.DefaultFormats import org.json4s.jackson.JsonMethods trait JestVersionApi extends VersionApi with JestClientHelpers { _: JestClientCompanion => + + /** Whether a request that carries a mapping must say `include_type_name=false`: on Elasticsearch + * 6.8, the one version that receives the mapping TYPELESS yet reads a mapping as TYPED by + * default. + * + * The mapping's shape is decided upstream, per version, by + * [[ElasticsearchVersion.requiresDocTypeWrapper]] (`MappingConverter`, `TemplateConverter`): + * wrapped in `_doc` before 6.8, typeless from 6.8. Every 6.x defaults `include_type_name` to + * `true`, so 6.8 read the typeless mapping's top-level keys as TYPE names and refused the + * `_meta` every table writes (`Failed to parse mapping [_meta]: Root mapping definition has + * unsupported parameters`): no `CREATE TABLE` ran through this client on 6.8, partitioned or + * not. The ES 6 REST client sends the same parameter on the same requests (its typeless + * `CreateIndexRequest` and `PutIndexTemplateRequest`). Not before 6.8, which refuses the + * parameter with a `_doc`-wrapped mapping; not from 7.0, where a mapping is typeless by default. + * + * A put mapping needs none: it names the type in its path (`PUT //_doc/_mapping`), which + * 6.8 takes with a typeless body, and with the parameter set it would refuse the type. + */ + private[client] def sendsTypelessMappings: Boolean = + version match { + case result.ElasticSuccess(v) => + ElasticsearchVersion.isEs6(v) && !ElasticsearchVersion.requiresDocTypeWrapper(v) + case _ => false + } + override private[client] def executeVersion(): result.ElasticResult[String] = executeJestAction( "version", diff --git a/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/actions/Template.scala b/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/actions/Template.scala index 1c38c2fab..df6893610 100644 --- a/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/actions/Template.scala +++ b/es6/jest/src/main/scala/app/softnetwork/elastic/client/jest/actions/Template.scala @@ -24,14 +24,20 @@ object Template { import io.searchbox.client.JestResult import com.google.gson.Gson - case class Create(templateName: String, json: String) extends AbstractAction[JestResult] { + /** `PUT /_template/`. `typelessMappings` says `include_type_name=false`, without which + * Elasticsearch 6.8 reads the template's typeless mappings as typed and refuses them + * (`JestVersionApi.sendsTypelessMappings`). + */ + case class Create(templateName: String, json: String, typelessMappings: Boolean = false) + extends AbstractAction[JestResult] { payload = json override def getRestMethodName: String = "PUT" override def getURI(elasticsearchVersion: ElasticsearchVersion): String = - s"/_template/$templateName" + if (typelessMappings) s"/_template/$templateName?include_type_name=false" + else s"/_template/$templateName" override def createNewElasticSearchResult( json: String, diff --git a/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientInsertByQuerySpec.scala b/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientInsertByQuerySpec.scala index d22df129b..a7846efef 100644 --- a/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientInsertByQuerySpec.scala +++ b/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientInsertByQuerySpec.scala @@ -5,6 +5,4 @@ import app.softnetwork.elastic.scalatest.ElasticDockerTestKit class JestClientInsertByQuerySpec extends InsertByQuerySpec with ElasticDockerTestKit { override lazy val client: ElasticClientApi = new JestClientSpi().client(elasticConfig) - - override def elasticVersion: String = "6.7.2" } diff --git a/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientPerIndexSchemaCacheTtlSpec.scala b/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientPerIndexSchemaCacheTtlSpec.scala index 1bffee4a8..d2818a376 100644 --- a/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientPerIndexSchemaCacheTtlSpec.scala +++ b/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientPerIndexSchemaCacheTtlSpec.scala @@ -16,12 +16,4 @@ package app.softnetwork.elastic.client -/** Pinned to 6.7.2 like every other jest spec that runs DDL (`JestGatewayApiSpec`, - * `JestClientTemplateApiSpec`, …): on 6.8 this client sends a typed mapping in which `_meta` is - * read as the TYPE name, so any `CREATE TABLE` — which always writes `_meta.columns` — is rejected - * with `Root mapping definition has unsupported parameters`. Pre-existing and unrelated to the - * TTL; the ES 6 rest client on 6.8 runs the same DDL fine. - */ -class JestClientPerIndexSchemaCacheTtlSpec extends PerIndexSchemaCacheTtlSpec { - override def elasticVersion: String = "6.7.2" -} +class JestClientPerIndexSchemaCacheTtlSpec extends PerIndexSchemaCacheTtlSpec diff --git a/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientSpec.scala b/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientSpec.scala index 8808ca47b..fbbf9b0eb 100644 --- a/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientSpec.scala +++ b/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientSpec.scala @@ -1,5 +1,3 @@ package app.softnetwork.elastic.client -class JestClientSpec extends ElasticClientSpec { - override def elasticVersion: String = "6.7.2" -} +class JestClientSpec extends ElasticClientSpec diff --git a/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientTemplateApiSpec.scala b/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientTemplateApiSpec.scala index e6160b4a3..fab25f7a7 100644 --- a/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientTemplateApiSpec.scala +++ b/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestClientTemplateApiSpec.scala @@ -5,6 +5,4 @@ import app.softnetwork.elastic.scalatest.ElasticDockerTestKit class JestClientTemplateApiSpec extends TemplateApiSpec with ElasticDockerTestKit { override lazy val client: TemplateApi with VersionApi = new JestClientSpi().client(elasticConfig) - - override def elasticVersion: String = "6.7.2" } diff --git a/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestGatewayApiSpec.scala b/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestGatewayApiSpec.scala index ded66b82c..9a6b0dced 100644 --- a/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestGatewayApiSpec.scala +++ b/es6/jest/src/test/scala/app/softnetwork/elastic/client/JestGatewayApiSpec.scala @@ -5,6 +5,4 @@ import app.softnetwork.elastic.scalatest.ElasticDockerTestKit class JestGatewayApiSpec extends GatewayApiIntegrationSpec with ElasticDockerTestKit { override lazy val client: GatewayApi = new JestClientSpi().client(elasticConfig) - - override def elasticVersion: String = "6.7.2" } diff --git a/es6/jest/src/test/scala/app/softnetwork/elastic/client/repl/JestReplGatewayIntegrationSpec.scala b/es6/jest/src/test/scala/app/softnetwork/elastic/client/repl/JestReplGatewayIntegrationSpec.scala index c8fcd1255..a3207254d 100644 --- a/es6/jest/src/test/scala/app/softnetwork/elastic/client/repl/JestReplGatewayIntegrationSpec.scala +++ b/es6/jest/src/test/scala/app/softnetwork/elastic/client/repl/JestReplGatewayIntegrationSpec.scala @@ -7,6 +7,4 @@ import app.softnetwork.elastic.scalatest.ElasticDockerTestKit class JestReplGatewayIntegrationSpec extends ReplGatewayIntegrationSpec with ElasticDockerTestKit { override lazy val gateway: GatewayApi = new JestClientSpi().client(elasticConfig) - - override def elasticVersion: String = "6.7.2" } diff --git a/sql/src/main/scala/app/softnetwork/elastic/sql/function/time/package.scala b/sql/src/main/scala/app/softnetwork/elastic/sql/function/time/package.scala index cfc9e5687..89b330323 100644 --- a/sql/src/main/scala/app/softnetwork/elastic/sql/function/time/package.scala +++ b/sql/src/main/scala/app/softnetwork/elastic/sql/function/time/package.scala @@ -266,6 +266,18 @@ package object time { else Today.sql } + /** The start of the ISO-8601 week a temporal falls in: its MONDAY. It is what `DATE_TRUNC(…, + * WEEK)` truncates to and where `DATEDIFF` / `DATE_DIFF` count a WEEK boundary, and it is what + * Elasticsearch's calendar week (`date_histogram`, `/w` date math) rounds to, so every venue + * that truncates to a week agrees. + * + * 🔴 `with(DayOfWeek.SUNDAY)`, rendered before, is NOT a Sunday-start week: `DayOfWeek` adjusts + * within the Monday-to-Sunday week, so it moved a Wednesday FORWARD to the Sunday that ENDS its + * week (2025-01-08 -> 2025-01-12, JDK-verified), while a `GROUP BY` of the same expression went + * back to the Monday. + */ + private[sql] val isoWeekStart: String = ".with(DayOfWeek.MONDAY)" + case object DateTrunc extends Expr("DATE_TRUNC") with TokenRegex with PainlessScript { override def painless(context: Option[PainlessContext]): String = ".truncatedTo" override lazy val words: List[String] = List(sql, "DATETRUNC") @@ -341,7 +353,7 @@ package object time { unit match { case TimeUnit.YEARS => s".withDayOfYear(1)$truncateTime" case TimeUnit.MONTHS => s".withDayOfMonth(1)$truncateTime" - case TimeUnit.WEEKS => s".with(DayOfWeek.SUNDAY)$truncateTime" + case TimeUnit.WEEKS => s"$isoWeekStart$truncateTime" case TimeUnit.QUARTERS => context match { case Some(ctx) => @@ -546,6 +558,12 @@ package object time { override def painless(context: Option[PainlessContext]): String = ".between" override lazy val words: List[String] = List(sql, "TIMESTAMPDIFF") + /** MySQL's `TIMESTAMPDIFF`, this token's second word. It takes the layouts `DATE_DIFF` takes + * but COUNTS differently (whole units elapsed, see [[DateDiffSpelling]]), so the parser asks + * which word it read and the render writes this one back. + */ + lazy val timestampDiff: String = words(1) + /** A calendar date alone, in the two separators the DATE parse accepts (`SQLTypeUtils.coerce`). */ private val CalendarDate = """\d{4}[-/]\d{2}[-/]\d{2}""".r @@ -566,44 +584,145 @@ package object time { } case _ => None } + + /** The scale of [[monthKey]]: more milliseconds than any month holds (31 days: 2678400000). */ + private[time] val MonthKeyScale: Long = 4294967296L + + /** A temporal as one number whose difference with another, divided by [[MonthKeyScale]] and + * truncated toward zero, is MySQL's whole months elapsed between them (`TIMESTAMPDIFF`): the + * months between the two, less one when the later one's day of month and time of day come + * before the earlier one's. + * + * 🔴 Not `ChronoUnit.MONTHS.between`, which differs on one edge. It first moves an end whose + * time of day comes before the start's back by one DAY — on the 1st of a month, into the month + * before — and then counts that month short as well: `2024-12-31 23:59:59` to `2025-07-01 + * 00:00:00` is 5 months there and 6 in MySQL (found by the testkit's population). + */ + private[time] def monthKey(rendered: String): String = { + val t = if (postfixable(rendered)) rendered else s"($rendered)" + s"(($t.getYear() * 12L + $t.getMonthValue()) * ${MonthKeyScale}L + ($t.getDayOfMonth() - 1) * 86400000L + $t.getLong(ChronoField.MILLI_OF_DAY))" + } + + /** A temporal as the instant it denotes, in UTC, decided by its RUNTIME type: a `LocalDate` is + * the start of its day, a `LocalDateTime` that wall-clock time in UTC, anything zoned the same + * instant. `TIMESTAMPDIFF` compares instants, and an operand's declared type does not say + * which of the three it holds: `COALESCE(ts, CURRENT_DATE)` is either, row by row. `DATEDIFF` + * and `DATE_DIFF` read a calendar date through it too, off a row-level operand that is not a + * column: a zoned value there may be Elasticsearch 6.8's doc value, which is no `java.time` + * type until `withZoneSameInstant` (as in [[utcZoned]]) returns the `ZonedDateTime` it wraps. + * + * `ref` is read up to four times, so it is a name or an expression cast to `def` (the methods + * below are resolved at runtime). NULL stays NULL: the MONTH, QUARTER and YEAR calls bind this + * value in the script's prologue, which runs before the null guard. + */ + private[time] def utcInstant(ref: String): String = + s"($ref == null ? null : $ref instanceof LocalDate ? $ref.atStartOfDay(ZoneId.of('Z')) : $ref instanceof LocalDateTime ? $ref.atZone(ZoneId.of('Z')) : $ref.withZoneSameInstant(ZoneId.of('Z')))" + + /** A column a SEARCH script reads, as a genuine `java.time.ZonedDateTime` in UTC; NULL stays + * NULL (the MONTH, QUARTER and YEAR calls bind it in the prologue, before the null guard). + * + * The column's script parameter is shared by every call that reads the column, and holds what + * the FIRST of them made of it (the parameter-identity family, issue #370): the UTC instant + * `DateDiff.in` folds onto it, or the raw doc value when a function over the same column read + * it first (`TIMESTAMPDIFF(MONTH, ts::DATE, ts)`). On Elasticsearch 6.8 that value is a + * `JodaCompatibleZonedDateTime`, which implements no `java.time` interface: it has no + * `getLong` ([[monthKey]] failed with `dynamic method [..., getLong/1] not found`) and is no + * `Temporal` (`ChronoUnit.between` failed with a `ClassCastException`). `withZoneSameInstant` + * is on its allow-list and returns the `ZonedDateTime` it wraps; on 7.x, 8.x and 9.x it reads + * the same instant, so every major counts the same units. + */ + private[time] def utcZoned(ref: String): String = { + val r = if (postfixable(ref)) ref else s"($ref)" + s"($r == null ? null : $r.withZoneSameInstant(ZoneId.of('Z')))" + } + + /** Whether a method can be appended to a rendered operand as it stands: outside parentheses, + * brackets and string literals it holds nothing but names, digits and dots. Anything else — a + * ternary (`p4 ? p3 : p1`, which is how a CASE operand renders), a cast (`(long) x`) — is + * parenthesised first, or the method would bind to its last term alone. + */ + private[time] def postfixable(rendered: String): Boolean = { + var depth = 0 + var quote: Char = 0 + var i = 0 + while (i < rendered.length) { + val c = rendered.charAt(i) + if (quote != 0) { + if (c == '\\') i += 1 + else if (c == quote) quote = 0 + } else if (c == '"' || c == '\'') quote = c + else if (c == '(' || c == '[') depth += 1 + else if (c == ')' || c == ']') depth -= 1 + else if (depth == 0 && !(c.isLetterOrDigit || c == '_' || c == '.')) return false + i += 1 + } + true + } } - /** MySQL's `DATEDIFF`, which is a DIFFERENT function from the one above and needs its own token - * so the parser can tell them apart (issue #363). - * - * MySQL 8.4 defines `DATEDIFF(expr1, expr2)` as `expr1 - expr2`; `DATE_DIFF(start, end, unit)` - * is BigQuery's and is `end - start`. While `DATEDIFF` was merely a WORD of the token above it - * inherited BigQuery's order, so it returned the opposite sign from the function it is named - * after — MySQL's own documented `DATEDIFF('2007-12-31','2007-12-30') -> 1` answered `-1` here. + /** `DATEDIFF`, the name MySQL, SQL Server, Snowflake, Redshift, DuckDB and Elasticsearch SQL give + * the function. It needs its own token because its two-argument form is MySQL's, which the + * `DATE_DIFF` productions do not parse (issue #363). * * 🔴 Neither spelling is a prefix of the other (`DATE_DIFF` has an underscore where `DATEDIFF` * has a `D`), so the two regexes cannot shadow one another whatever order they are tried in. */ case object MySqlDateDiff extends Expr("DATEDIFF") with TokenRegex - /** Which spelling a `DateDiff` was written as. It decides the RENDER and nothing else — the node - * itself always means `end - start`, so every consumer (`args`, `left`/`right`, the Painless - * emission, validation, `update`) has exactly ONE encoding of the decision to read. + /** How a `DateDiff` was written, which decides its RENDER and what it COUNTS. The node itself + * always means `end - start`: the parser stores the operands so, and every other consumer + * (`args`, `left`/`right`, the Painless emission, validation, `update`) reads that one encoding. + * + * The vendors' definitions, fetched from their references (2026-10-02), and the lead's ruling to + * follow them on both axes: + * + * - DIRECTION, by layout. Every vendor computes `end - start`; the layout decides where the + * end sits. Dates first is `first - second`: MySQL's `DATEDIFF(expr1, expr2)`, BigQuery's + * `DATE_DIFF(end_date, start_date, part)` (`DATE_DIFF('2010-07-07', '2008-12-25', DAY)` is + * 559). Unit first is `last - middle`: `DATEDIFF(unit, start, end)` in SQL Server, + * Snowflake, Redshift, DuckDB and Elasticsearch SQL, and MySQL's `TIMESTAMPDIFF(unit, dt1, + * dt2)`. + * - COUNTING, by name. `DATEDIFF` and `DATE_DIFF` count the calendar BOUNDARIES crossed, as + * SQL Server, Snowflake, Redshift, BigQuery and DuckDB's `date_diff` do: one second across a + * year end is one YEAR. `TIMESTAMPDIFF` counts the whole units ELAPSED, truncated toward + * zero, as MySQL does: a month counts once its day and time of day are reached. * - * A `Boolean` cannot carry three forms, and silently widening one is how the old `transactSql` - * flag would have rotted; being sealed, the compiler now forces every render arm. + * 🔴 Issue #363 read BigQuery's `DATE_DIFF` as `(start, end, unit)`, so until 0.24.0 the + * dates-first forms with a unit subtracted the other way round, and every unit from WEEK up + * counted elapsed units. Sealed, so the compiler forces every arm of the render. */ - sealed trait DateDiffSpelling + sealed trait DateDiffSpelling { + + /** `TIMESTAMPDIFF` counts whole units ELAPSED; every other spelling counts BOUNDARIES. */ + def elapsed: Boolean = false + } object DateDiffSpelling { - /** `DATE_DIFF(start, end, unit)` — BigQuery's order, this engine's canonical render. */ + /** `DATE_DIFF(end, start[, unit])` (BigQuery's order) and `DATEDIFF(end, start, unit)`: dates + * first, so the parser stores the SECOND operand as the start. + */ case object DateFirst extends DateDiffSpelling - /** `DATE_DIFF(unit, start, end)` / `TIMESTAMPDIFF(unit, start, end)` — the ODBC/T-SQL and MySQL - * `TIMESTAMPDIFF` order. MySQL defines that one as `dt2 - dt1`, which is what this engine - * already computed, so it needed no change. - */ + /** `DATE_DIFF(unit, start, end)` / `DATEDIFF(unit, start, end)`: unit first. */ case object UnitFirst extends DateDiffSpelling - /** `DATEDIFF(expr1, expr2)` — MySQL's, `expr1 - expr2`, days only. The parser stores it with - * `start`/`end` SWAPPED, so the node still means `end - start`; the render swaps them back. + /** `DATEDIFF(end, start)`: MySQL's, in days. It means what `DATE_DIFF(end, start, DAY)` means; + * the spelling is kept so that `SHOW CREATE …` hands back what was written. */ case object MySql extends DateDiffSpelling + + /** `TIMESTAMPDIFF(unit, start, end)`: MySQL's, whole units ELAPSED. */ + case object TimestampDiff extends DateDiffSpelling { + override val elapsed: Boolean = true + } + + /** `TIMESTAMPDIFF(end, start[, unit])`. No vendor writes it; it parses because `TIMESTAMPDIFF` + * is a word of the `DATE_DIFF` token, and the two rules above give it its meaning: dates + * first, whole units elapsed. + */ + case object TimestampDiffDateFirst extends DateDiffSpelling { + override val elapsed: Boolean = true + } } case class DateDiff( @@ -628,22 +747,24 @@ package object time { * `MaterializedViewExtension` persists this text and re-runs `client.run(alter.sql)`, `SHOW * CREATE MATERIALIZED VIEW` echoes it, and `SCRIPT AS` stores it beside the Painless. * - * 🔴 Keeping the `DATEDIFF` spelling is a READABILITY choice, not a correctness one, and the - * distinction is worth stating because the opposite claim is easy to reach for. Because the - * parser already stored MySQL's operands swapped, rendering this node as `DATE_DIFF(start, - * end, unit)` would ALSO re-parse to the same node and mean the same thing — a mutation that - * does exactly that reddens one assertion here, and it is the spelling one. What the swap - * protects against is the OTHER design, the one where the node keeps the operands as written - * and reverses them at emission: there the canonical render really does flip the sign of a - * stored statement on its next round trip (issue #363). + * 🔴 The dates-first forms are stored with their operands SWAPPED (the second is the start), + * and the render swaps them back. What the swap protects against is the other design, the one + * where the node keeps the operands as written and reverses them at emission: there the + * canonical render flips the sign of a stored statement on its next round trip (issue #363). * - * The spelling is preserved so that `SHOW CREATE …` hands back what was written, rather than - * the same statement with its two arguments visibly exchanged. + * 🔴 `TIMESTAMPDIFF` keeps its name, and that is correctness, not readability: it counts + * elapsed units where `DATE_DIFF` counts boundaries, so a render as `DATE_DIFF` would change + * what a stored statement answers. `DATEDIFF(a, b)` keeps its name for readability only (it + * means `DATE_DIFF(a, b, DAY)`), so that `SHOW CREATE …` hands back what was written. */ override def toSQL(base: String): String = spelling match { case DateDiffSpelling.UnitFirst => s"$sql(${unit.sql}, ${start.sql}, ${end.sql})" case DateDiffSpelling.MySql => s"${MySqlDateDiff.sql}(${end.sql}, ${start.sql})" - case DateDiffSpelling.DateFirst => s"$sql(${start.sql}, ${end.sql}, ${unit.sql})" + case DateDiffSpelling.DateFirst => s"$sql(${end.sql}, ${start.sql}, ${unit.sql})" + case DateDiffSpelling.TimestampDiff => + s"${DateDiff.timestampDiff}(${unit.sql}, ${start.sql}, ${end.sql})" + case DateDiffSpelling.TimestampDiffDateFirst => + s"${DateDiff.timestampDiff}(${end.sql}, ${start.sql}, ${unit.sql})" } /** The type a COLUMN operand is read as, whatever the unit: the instant it denotes, in UTC @@ -659,20 +780,39 @@ package object time { */ override def in: SQLType = SQLTypes.Timestamp - /** The type the two operands are COMPARED in, and the UNIT decides it. + /** The type the two operands are COMPARED in, in UTC; the counting and the unit decide it. * - * - `HOUR`, `MINUTE` and `SECOND` count the ELAPSED whole units between two instants (UTC), - * truncated toward zero; a DATE operand is the start of its day. They used to be compared - * as a `LocalDate`, which has no time of day, so every such call failed with `Unsupported - * unit: Hours`. - * - `DAY` and every larger unit compare the two CALENDAR dates (UTC): 23:30 and 00:30 the - * next day are one day apart, as MySQL's `DATEDIFF` answers. That is what they computed. + * - `TIMESTAMPDIFF` counts the whole units ELAPSED between two instants, whatever the unit: + * a DATE operand is the start of its day. + * - `DATEDIFF` / `DATE_DIFF` count BOUNDARIES: between two calendar dates from `DAY` up + * (23:30 and 00:30 the next day are one day apart), between two instants for `HOUR`, + * `MINUTE` and `SECOND`, which a `LocalDate` cannot hold (`Unsupported unit: Hours`). */ private def comparedIn: SQLType = unit match { + case _ if spelling.elapsed => SQLTypes.Timestamp case TimeUnit.HOURS | TimeUnit.MINUTES | TimeUnit.SECONDS => SQLTypes.Timestamp case _ => SQLTypes.Date } + /** What each operand is moved to before the units between them are counted: for a BOUNDARY + * count, the start of the unit it falls in (UTC), so that the whole units elapsed between the + * two starts are the boundaries crossed between the operands — 2005-12-31 23:59:59 and + * 2006-01-01 00:00:00 are one YEAR apart, 1992-09-15 and 1992-11-14 two MONTHs, 10:59 and + * 11:00 one HOUR. A WEEK starts on the ISO Monday ([[isoWeekStart]]), a QUARTER on January, + * April, July or October 1st. An ELAPSED count moves nothing. + */ + private def unitStart: String = + if (spelling.elapsed) "" + else + unit match { + case TimeUnit.YEARS => ".withDayOfYear(1)" + case TimeUnit.QUARTERS => ".with(java.time.temporal.IsoFields.DAY_OF_QUARTER, 1)" + case TimeUnit.MONTHS => ".withDayOfMonth(1)" + case TimeUnit.WEEKS => isoWeekStart + case TimeUnit.DAYS => "" + case subDay => s".truncatedTo(${subDay.painless(None)})" + } + /** A context-free rendering over an aggregate is the PER-GROUP calculation: the `bucket_script` * of a SELECT item (`DATEDIFF(MAX(d), '2024-01-01') AS x`), which a HAVING over `x` reads by * name and a materialized view's pivot computes too (`SingleSearch.transformBucketScripts`). @@ -688,32 +828,40 @@ package object time { case _ => super[BinaryFunction].painless(context) } - /** One operand, brought to [[comparedIn]] through the coercion arms a CAST uses. + /** One operand, brought to [[comparedIn]] through the coercion arms a CAST uses, then to the + * start of its unit ([[unitStart]]). * * - A string LITERAL is first read as the temporal it spells ([[DateDiff.literalType]]: UTC - * unless it names a zone). Row level used to hand Painless the bare string - * (`between("2024-01-01", param1)`, which it cannot call), and per group parsed it as a - * DATE whatever it held, so a time of day failed the search. + * unless it names a zone), in every venue. Row level used to hand Painless the bare string + * (`between("2024-01-01", param1)`, which it cannot call), and an ingest processor still + * did. * - Per group, an aggregate is the metric Elasticsearch computed, which a `bucket_script` * receives as a `java.lang.Double` holding EPOCH MILLIS (`(long)` is required: Painless * refuses to cast a `def` double to `long` implicitly). Any other operand is converted * from its own type. + * - In an INGEST processor any other operand is a UTC temporal the processor made itself: a + * column parsed at runtime by `SQLTypeUtils.processorTemporal`, the clock, a `::DATE` + * literal. The first two are `ZonedDateTime`s whatever their type, the last a `LocalDate`, + * and `LocalDate.from` reads the calendar date of either. A DATE under a time-of-day + * comparison is the start of its day, as at row level — `CURRENT_DATE` included, which a + * processor holds as the current instant. * - At row level a column operand holds its UTC instant ([[in]]), so its calendar date is - * `toLocalDate()`. An operand with no column of its own (`'2025-01-10'::DATE`, - * `CURRENT_DATE`, `NOW()`) renders as it did, except a DATE under a sub-day unit, which is - * the start of its day: a `LocalDate` has no hours. - * - * 🔴 An INGEST processor renders as it did: there a column is the raw document value, parsed - * at runtime by `SQLTypeUtils.processorTemporal`. + * `toLocalDate()`. Any other operand (`'2025-01-10'::DATE`, `CURRENT_DATE`, `NOW()`, a + * COALESCE, a CASE) is read for a calendar date by `LocalDate.from`, once it is a UTC + * `java.time` value by its RUNTIME type ([[DateDiff.utcInstant]]): a COALESCE or a CASE + * over a column hands on the column's raw doc value, which on Elasticsearch 6.8 is no + * `java.time` type, and `LocalDate.from` refused it (`ClassCastException`). For a + * time-of-day comparison a DATE is the start of its day (a `LocalDate` has no hours) and + * anything else renders as it did. + * - `TIMESTAMPDIFF` takes [[elapsedOperand]] instead, in every venue. */ private def operand( arg: PainlessScript, rendered: String, context: Option[PainlessContext], perGroup: Boolean - ): String = - DateDiff.literalType(arg) match { - case _ if context.exists(_.isProcessor) => rendered + ): String = { + val compared = DateDiff.literalType(arg) match { case Some(literal) => val value = SQLTypeUtils.coerce(rendered, SQLTypes.Varchar, literal, nullable = false, None) @@ -724,15 +872,22 @@ package object time { val epochMillis = SQLTypeUtils .coerce(rendered, SQLTypes.Double, SQLTypes.BigInt, nullable = false, None) // The instant in UTC already (`Instant.ofEpochMilli(...).atZone(ZoneId.of('Z'))`), so - // DAY and above read its calendar date off it: ONE conversion per aggregate. The - // TIMESTAMP -> DATE arm would normalise it to UTC a second time - // (`.toInstant().atZone(ZoneId.of('Z'))`), as it must an operand of unknown zone. + // a calendar date is read off it: ONE conversion per aggregate. The TIMESTAMP -> DATE + // arm would normalise it to UTC a second time (`.toInstant().atZone(ZoneId.of('Z'))`), + // as it must an operand of unknown zone. val utc = SQLTypeUtils .coerce(epochMillis, SQLTypes.BigInt, SQLTypes.Timestamp, nullable = false, None) if (comparedIn == SQLTypes.Date) s"$utc.toLocalDate()" else utc + case _ if spelling.elapsed => elapsedOperand(arg, rendered, context) case other => SQLTypeUtils.coerce(rendered, other.baseType, comparedIn, nullable = false, None) } + case None if spelling.elapsed => elapsedOperand(arg, rendered, context) + case None if context.exists(_.isProcessor) => + if (comparedIn == SQLTypes.Date) s"LocalDate.from($rendered)" + else if (arg.baseType == SQLTypes.Date) + s"LocalDate.from($rendered).atStartOfDay(ZoneId.of('Z'))" + else rendered case None if context.isDefined => arg match { case column: Identifier if column.name.trim.nonEmpty => @@ -740,20 +895,86 @@ package object time { case value: Identifier if comparedIn == SQLTypes.Timestamp && value.baseType == SQLTypes.Date => SQLTypeUtils.coerce(rendered, SQLTypes.Date, comparedIn, nullable = false, None) + // a `ZonedDateTime` here (`'…'::TIMESTAMP`, `NOW()`, a CASE) would keep its time of day + // through `unitStart`, and a calendar unit would count it; read by its runtime type + // first, for a COALESCE or a CASE may hand on a column's raw 6.8 doc value + case _ if comparedIn == SQLTypes.Date => + s"LocalDate.from(${DateDiff.utcInstant(bound(rendered, context))})" case _ => rendered } case None => rendered } + if (unitStart.isEmpty) compared + else if (DateDiff.postfixable(compared)) s"$compared$unitStart" + else s"($compared)$unitStart" + } + + /** A `TIMESTAMPDIFF` operand, as the instant it denotes in UTC. The function compares INSTANTS: + * `ChronoUnit.between` reads its end as the type of its start, so a `ZonedDateTime` start + * refused a `LocalDate` or a `LocalDateTime` end, and the month count ([[DateDiff.monthKey]]) + * reads a time of day a `LocalDate` does not have. Either way the statement failed. + * + * - An ingest processor reads a DATE-typed operand as the start of its calendar day + * (`LocalDate.from`): it holds `CURRENT_DATE` as the current INSTANT, so there the + * declared type, not the runtime one, is what says "a date". + * - A column is an instant in a script that reads it, though not always a `java.time` one. + * An ingest processor parses it into a UTC `ZonedDateTime` + * (`SQLTypeUtils.processorTemporal`). Row level reads its column parameter, which holds + * its UTC instant ([[in]]) unless a function over the same column read it first: then it + * is the raw doc value, which on Elasticsearch 6.8 is no `java.time` type, so it goes + * through [[DateDiff.utcZoned]]. A function over a column need not be an instant at all: + * `ts::DATE` is a `LocalDate`. + * - Any other operand is converted by its RUNTIME type ([[DateDiff.utcInstant]]), which its + * declared type does not give: `COALESCE(ts, CURRENT_DATE)`, or a CASE over `CURRENT_DATE` + * and a DATE, is a `LocalDate` on one row and a `ZonedDateTime` on another, a `::DATETIME` + * a `LocalDateTime`. It is evaluated once, bound to a name where the script can bind one; + * a `bucket_script` cannot, and per group such an operand is a constant of the request. + */ + private def elapsedOperand( + arg: PainlessScript, + rendered: String, + context: Option[PainlessContext] + ): String = arg match { + case _ if context.exists(_.isProcessor) && arg.baseType == SQLTypes.Date => + s"LocalDate.from($rendered).atStartOfDay(ZoneId.of('Z'))" + case column: Identifier + if context.isDefined && column.name.trim.nonEmpty && column.functions.isEmpty => + if (context.exists(_.isProcessor)) rendered else DateDiff.utcZoned(rendered) + case _ => DateDiff.utcInstant(bound(rendered, context)) + } + + /** A rendered operand as [[DateDiff.utcInstant]] may read it, up to four times: a name as it + * stands, any other expression evaluated once into a parameter where the script can bind one, + * cast to `def` where it cannot. + */ + private def bound(rendered: String, context: Option[PainlessContext]): String = + context match { + case Some(_) if FunctionN.isName(rendered) => rendered + case Some(ctx) => ctx.addParam(LiteralParam(rendered)).getOrElse(s"((def) ($rendered))") + case None => s"((def) ($rendered))" + } /** `ChronoUnit` has no `QUARTERS`, so a `QUARTER` call failed to compile in Elasticsearch. The - * ISO quarter-year unit counts the whole quarters between two calendar dates: the whole months - * between them, divided by 3. + * ISO quarter-year unit counts the whole quarters between two calendar dates, which between + * two quarter starts ([[unitStart]]) are the quarter boundaries crossed. */ private def unitPainless(context: Option[PainlessContext]): String = unit match { case TimeUnit.QUARTERS => "java.time.temporal.IsoFields.QUARTER_YEARS" case other => other.painless(context) } + /** `TIMESTAMPDIFF`'s MONTH, QUARTER and YEAR are whole months elapsed, divided by 1, 3 or 12. + */ + private def monthsPerUnit: Option[Int] = + if (!spelling.elapsed) None + else + unit match { + case TimeUnit.MONTHS => Some(1) + case TimeUnit.QUARTERS => Some(3) + case TimeUnit.YEARS => Some(12) + case _ => None + } + /** The function's ONE rendering: row level reaches it from `FunctionN.painless` with its * operands rendered, per group from [[painless]]; [[operand]] converts each one. */ @@ -765,8 +986,17 @@ package object time { val operands = args.zip(callArgs).map { case (arg, rendered) => operand(arg, rendered, context, perGroup) } - val ret = - s"Long.valueOf(${unitPainless(context)}${DateDiff.painless(context)}(${operands.mkString(", ")}))" + val ret = monthsPerUnit match { + case Some(perUnit) => + // each operand is read four times by `monthKey`: bound once where a script can bind it + val bound = + operands.map(op => context.flatMap(_.addParam(LiteralParam(op))).getOrElse(op)) + val months = + s"(${DateDiff.monthKey(bound(1))} - ${DateDiff.monthKey(bound.head)}) / ${DateDiff.MonthKeyScale}L" + s"Long.valueOf(${if (perUnit == 1) months else s"$months / $perUnit"})" + case None => + s"Long.valueOf(${unitPainless(context)}${DateDiff.painless(context)}(${operands.mkString(", ")}))" + } context match { case Some(ctx) if ctx.isProcessor => // to fix bug in painless script processor context with elasticsearch v6 diff --git a/sql/src/main/scala/app/softnetwork/elastic/sql/parser/function/time/package.scala b/sql/src/main/scala/app/softnetwork/elastic/sql/parser/function/time/package.scala index 909a0a62f..0d8233b89 100644 --- a/sql/src/main/scala/app/softnetwork/elastic/sql/parser/function/time/package.scala +++ b/sql/src/main/scala/app/softnetwork/elastic/sql/parser/function/time/package.scala @@ -204,51 +204,61 @@ package object time { // (`Parser.operandIdentifier`), as `ISNULL`'s is: an aggregate in them is the aggregate every // other position builds, not the bare `MAX` token -- which a view's transform did not // recognise, and whose rendered name (`MAX(d)`) did not match a qualified SELECT item's. + // + // The node means `end - start`, and the LAYOUT decides which operand is the end (see + // `DateDiffSpelling`): dates first is `first - second`, so the SECOND operand is stored as the + // start; unit first is `last - middle`. The NAME decides the counting: `TIMESTAMPDIFF` is the + // second word of `DateDiff.regex`, so the word read is kept. lazy val date_diff: PackratParser[BinaryFunction[_, _, _]] = DateDiff.regex ~ start ~ operandIdentifier ~ separator ~ operandIdentifier ~ (separator ~ time_unit).? ~ end ^^ { - case _ ~ _ ~ d1 ~ _ ~ d2 ~ u ~ _ => + case name ~ _ ~ d1 ~ _ ~ d2 ~ u ~ _ => DateDiff( - d1, d2, + d1, u match { case Some(_ ~ unit) => unit case None => TimeUnit.DAYS - } + }, + if (name.equalsIgnoreCase(DateDiff.timestampDiff)) + DateDiffSpelling.TimestampDiffDateFirst + else DateDiffSpelling.DateFirst ) } - /** 🔴 BOTH names. `DATEDIFF(unit, start, end)` is T-SQL's and DuckDB's spelling — what a BI - * tool set to a SQL Server or DuckDB dialect emits — and it is `end - start` in both, which is - * exactly what this production binds. Issue #363 moved `DATEDIFF` onto its own token to give - * the TWO-argument MySQL form its own meaning, and keying this production on `DateDiff.regex` - * alone silently dropped the unit-first spelling with it. + /** 🔴 BOTH names. `DATEDIFF(unit, start, end)` is the spelling of SQL Server, Snowflake, + * Redshift, DuckDB and Elasticsearch SQL — what a BI tool set to one of those dialects emits — + * and it is `end - start` in all of them, which is exactly what this production binds. Issue + * #363 moved `DATEDIFF` onto its own token to give the TWO-argument MySQL form its own + * meaning, and keying this production on `DateDiff.regex` alone silently dropped the + * unit-first spelling with it. * * Ordering is safe: this production is tried BEFORE `mysql_date_diff`, and `time_unit` fails * on a plain column, so `DATEDIFF(a, b)` still falls through to the MySQL form. */ lazy val date_diff_transact_sql: PackratParser[BinaryFunction[_, _, _]] = - (DateDiff.regex | MySqlDateDiff.regex) ~ start ~> time_unit ~ separator ~ operandIdentifier ~ separator ~ operandIdentifier <~ end ^^ { - case u ~ _ ~ d1 ~ _ ~ d2 => - DateDiff(d1, d2, u, DateDiffSpelling.UnitFirst) + (DateDiff.regex | MySqlDateDiff.regex) ~ start ~ time_unit ~ separator ~ operandIdentifier ~ separator ~ operandIdentifier ~ end ^^ { + case name ~ _ ~ u ~ _ ~ d1 ~ _ ~ d2 ~ _ => + DateDiff( + d1, + d2, + u, + if (name.equalsIgnoreCase(DateDiff.timestampDiff)) DateDiffSpelling.TimestampDiff + else DateDiffSpelling.UnitFirst + ) } - /** MySQL's `DATEDIFF` (issue #363), which is NOT the function `date_diff` above parses. - * - * Two arguments — the shape MySQL actually defines — means `expr1 - expr2`, so the operands - * are stored SWAPPED and the node keeps its single meaning of `end - start`. The dialect lives - * here and in `toSQL`, nowhere else. + /** `DATEDIFF` with the dates first, which `date_diff` above does not parse (issue #363). * - * 🔴 The THREE-argument `DATEDIFF(a, b, unit)` is NOT MySQL — MySQL's is days-only — it is - * this engine's own extension, and a lead ruling keeps it exactly as it behaved before: `end - - * start`, rendered as `DATE_DIFF`. So on this ONE spelling the arity changes the sign, and - * `DateDiffSignSpec` pins both readings side by side so that stays deliberate and visible - * rather than discovered. + * Two arguments is MySQL's `DATEDIFF(expr1, expr2)`, `expr1 - expr2` in days; with a unit it + * is the same layout, so the same direction: `first - second` either way, and the operands are + * stored SWAPPED so that the node keeps its single meaning of `end - start`. Before 0.24.0 the + * three-argument form subtracted the other way round (see `DateDiffSpelling`). */ lazy val mysql_date_diff: PackratParser[BinaryFunction[_, _, _]] = MySqlDateDiff.regex ~ start ~ operandIdentifier ~ separator ~ operandIdentifier ~ (separator ~ time_unit).? ~ end ^^ { case _ ~ _ ~ d1 ~ _ ~ d2 ~ u ~ _ => u match { - case Some(_ ~ unit) => DateDiff(d1, d2, unit, DateDiffSpelling.DateFirst) + case Some(_ ~ unit) => DateDiff(d2, d1, unit, DateDiffSpelling.DateFirst) case None => DateDiff(d2, d1, TimeUnit.DAYS, DateDiffSpelling.MySql) } } diff --git a/sql/src/test/scala/app/softnetwork/elastic/sql/parser/DateDiffSignSpec.scala b/sql/src/test/scala/app/softnetwork/elastic/sql/parser/DateDiffSignSpec.scala index 9da2b432d..d28ca73cc 100644 --- a/sql/src/test/scala/app/softnetwork/elastic/sql/parser/DateDiffSignSpec.scala +++ b/sql/src/test/scala/app/softnetwork/elastic/sql/parser/DateDiffSignSpec.scala @@ -20,15 +20,23 @@ import app.softnetwork.elastic.sql.function.time.{DateDiff, DateDiffSpelling} import org.scalatest.flatspec.AnyFlatSpec import org.scalatest.matchers.should.Matchers -/** Issue #363 — `DATEDIFF` is MySQL's function and must return MySQL's sign. +/** `DATEDIFF`, `DATE_DIFF` and `TIMESTAMPDIFF` subtract and count as the vendors define them. * - * MySQL 8.4 defines `DATEDIFF(expr1, expr2)` as `expr1 - expr2`; BigQuery's `DATE_DIFF(start, end, - * unit)` is `end - start`. While `DATEDIFF` was only a WORD of the `DATE_DIFF` token it inherited - * BigQuery's order and answered the opposite sign. + * Every vendor computes `end - start`, and the LAYOUT decides where the end sits. With the dates + * first it is the FIRST operand: MySQL's `DATEDIFF(expr1, expr2)`, BigQuery's `DATE_DIFF(end_date, + * start_date, part)` (`DATE_DIFF('2010-07-07', '2008-12-25', DAY)` is 559). With the unit first it + * is the LAST: `DATEDIFF(unit, start, end)` in SQL Server, Snowflake, Redshift, DuckDB and + * Elasticsearch SQL, and MySQL's `TIMESTAMPDIFF(unit, dt1, dt2)`. The NAME decides the counting: + * `TIMESTAMPDIFF` counts whole units elapsed, the other two calendar boundaries. + * + * 🔴 Issue #363 rested on a wrong premise, that BigQuery's `DATE_DIFF` takes `(start, end, unit)`. + * Until 0.24.0 the dates-first forms with a unit (`DATE_DIFF(a, b, unit)`, `DATEDIFF(a, b, unit)`) + * therefore answered `b - a`, and adding a unit flipped `DATEDIFF`'s sign. * * 🔴 The assertions here are on the AST's `start`/`end`, never on "it parsed". The node always - * means `end - start`, so which operand landed in which field IS the semantics — and a test that - * only checked `isRight`, or only the rendered text, would pass against the defect. + * means `end - start`, so which operand landed in which field IS the direction, and a test that + * only checked `isRight`, or only the rendered text, would pass against the defect. The answers + * themselves are RUN in `DateFunctionExecutionSpec`. */ class DateDiffSignSpec extends AnyFlatSpec with Matchers { @@ -50,51 +58,68 @@ class DateDiffSignSpec extends AnyFlatSpec with Matchers { private def renderOf(sql: String): String = Parser(s"SELECT $sql AS v FROM t").map(_.sql).getOrElse(fail(s"[$sql] rejected")) - // ─────────────────────────────── the sign itself ──────────────────────────────── + // ──────────────────────────── the direction, by layout ───────────────────────────── - /** MySQL's own two documented examples. Before this fix they answered -1 and 31. */ - "DATEDIFF with two arguments" should "compute expr1 - expr2, as MySQL does" in { - val d = dateDiffOf("DATEDIFF(a, b)") - // `end - start` is what the node means, so MySQL's `a - b` requires start = b, end = a. - d.start.sql shouldBe "b" - d.end.sql shouldBe "a" - d.spelling shouldBe DateDiffSpelling.MySql + "the dates first" should "subtract the second operand from the first, in both names and at every arity" in { + Seq( + "DATEDIFF(a, b)", + "DATEDIFF(a, b, DAY)", + "DATEDIFF(a, b, MONTH)", + "DATE_DIFF(a, b)", + "DATE_DIFF(a, b, YEAR)" + ).foreach { sql => + withClue(s"[$sql] ") { + val d = dateDiffOf(sql) + // `end - start` is what the node means, so `a - b` requires start = b, end = a. + (d.start.sql, d.end.sql) shouldBe (("b", "a")) + } + } } - "DATE_DIFF" should "keep BigQuery's end - start, untouched by this change" in { - val d = dateDiffOf("DATE_DIFF(a, b)") - d.start.sql shouldBe "a" - d.end.sql shouldBe "b" + it should "read BigQuery's own documented example with its end first" in { + // DATE_DIFF('2010-07-07', '2008-12-25', DAY) is 559, a positive number: the end comes first. + val d = dateDiffOf("DATE_DIFF('2010-07-07', '2008-12-25', DAY)") + (d.start.sql, d.end.sql) shouldBe (("'2008-12-25'", "'2010-07-07'")) d.spelling shouldBe DateDiffSpelling.DateFirst } - "TIMESTAMPDIFF" should "keep MySQL's dt2 - dt1, which it already matched" in { - val d = dateDiffOf("TIMESTAMPDIFF(DAY, a, b)") - d.start.sql shouldBe "a" - d.end.sql shouldBe "b" - d.spelling shouldBe DateDiffSpelling.UnitFirst + it should "no longer change DATEDIFF's sign when a unit is added" in { + // The pair #363 pinned as deliberately different: one layout, one direction. + val two = dateDiffOf("DATEDIFF(a, b)") + val three = dateDiffOf("DATEDIFF(a, b, DAY)") + (two.start.sql, two.end.sql) shouldBe ((three.start.sql, three.end.sql)) + two.spelling shouldBe DateDiffSpelling.MySql + three.spelling shouldBe DateDiffSpelling.DateFirst } - /** 🔴 The regression an independent review caught, and the reason it is pinned here. + "the unit first" should "subtract the middle operand from the last, in all three names" in { + Seq( + "DATEDIFF(DAY, a, b)", + "DATEDIFF(MONTH, a, b)", + "DATE_DIFF(YEAR, a, b)", + "TIMESTAMPDIFF(DAY, a, b)" + ).foreach { sql => + withClue(s"[$sql] ") { + val d = dateDiffOf(sql) + (d.start.sql, d.end.sql) shouldBe (("a", "b")) + } + } + } + + /** 🔴 The regression an independent review caught on #363, and the reason it is pinned here. * - * `DATEDIFF(unit, start, end)` is T-SQL's and DuckDB's spelling — what a BI tool set to a SQL - * Server or DuckDB dialect emits — and both define it as `end - start`, which is what this - * engine already computed. Moving `DATEDIFF` onto its own token for the MySQL form silently took + * `DATEDIFF(unit, start, end)` is what a BI tool set to a SQL Server, Snowflake, Redshift or + * DuckDB dialect emits. Moving `DATEDIFF` onto its own token for the MySQL form silently took * the unit-first spelling with it, because `date_diff_transact_sql` keyed on `DateDiff.regex` - * alone. It parsed on `main` and stopped parsing here. + * alone. * - * 🔴 `GrammarDiffProbe` did NOT catch it: its corpus contains no `DATEDIFF(unit, a, b)` input. - * That is the second time that probe has returned a clean differential over a real narrowing, so - * it is necessary and not sufficient — a spelling removed from a `words` list needs its own - * assertion, not a corpus sweep. + * 🔴 `GrammarDiffProbe` did NOT catch it: its corpus contains no `DATEDIFF(unit, a, b)` input. A + * spelling removed from a `words` list needs its own assertion, not a corpus sweep. */ - "DATEDIFF with the unit first" should "keep T-SQL's and DuckDB's spelling, and their order" in { + "DATEDIFF with the unit first" should "keep its spelling" in { Seq("DAY", "MONTH", "YEAR").foreach { unit => withClue(s"[DATEDIFF($unit, a, b)] ") { - val d = dateDiffOf(s"DATEDIFF($unit, a, b)") - d.start.sql shouldBe "a" - d.end.sql shouldBe "b" - d.spelling shouldBe DateDiffSpelling.UnitFirst + dateDiffOf(s"DATEDIFF($unit, a, b)").spelling shouldBe DateDiffSpelling.UnitFirst } } } @@ -102,28 +127,35 @@ class DateDiffSignSpec extends AnyFlatSpec with Matchers { it should "still let the two-argument MySQL form through, which follows it in the alternation" in { // `time_unit` fails on a plain column, so `DATEDIFF(a, b)` falls past the unit-first // production to the MySQL one. If that ordering ever breaks, this goes red rather than the - // sign silently reverting. + // direction silently changing. val d = dateDiffOf("DATEDIFF(a, b)") d.spelling shouldBe DateDiffSpelling.MySql (d.start.sql, d.end.sql) shouldBe (("b", "a")) } - /** 🔴 The lead ruling, pinned BOTH WAYS so it stays deliberate: on this one spelling, adding a - * unit changes the sign, because the 3-argument form is this engine's own extension and is not - * MySQL. If either half of this ever moves, it should move on purpose. - */ - "DATEDIFF with three arguments" should "keep this engine's end - start, unlike the two-arg form" in { - val two = dateDiffOf("DATEDIFF(a, b)") - val three = dateDiffOf("DATEDIFF(a, b, DAY)") - withClue("the 2-arg MySQL form swaps: ") { - (two.start.sql, two.end.sql) shouldBe ("b", "a") - } - withClue("the 3-arg extension does not: ") { - (three.start.sql, three.end.sql) shouldBe ("a", "b") + // ──────────────────────────── the counting, by name ───────────────────────────── + + "TIMESTAMPDIFF" should "count whole units elapsed, where DATEDIFF and DATE_DIFF count boundaries" in { + dateDiffOf("TIMESTAMPDIFF(MONTH, a, b)").spelling shouldBe DateDiffSpelling.TimestampDiff + dateDiffOf("TIMESTAMPDIFF(MONTH, a, b)").spelling.elapsed shouldBe true + Seq( + "DATEDIFF(a, b)", + "DATEDIFF(a, b, MONTH)", + "DATEDIFF(MONTH, a, b)", + "DATE_DIFF(a, b, MONTH)", + "DATE_DIFF(MONTH, a, b)" + ).foreach { sql => + withClue(s"[$sql] ")(dateDiffOf(sql).spelling.elapsed shouldBe false) } - three.spelling shouldBe DateDiffSpelling.DateFirst - // Stated as a difference, so the test fails if they are ever quietly unified. - (two.start.sql, two.end.sql) should not be (three.start.sql, three.end.sql) + } + + it should "take the dates first too, as no vendor writes it: first - second, elapsed" in { + // It parses because TIMESTAMPDIFF is a word of the DATE_DIFF token, and the two rules give it + // its meaning. + val d = dateDiffOf("TIMESTAMPDIFF(a, b, DAY)") + (d.start.sql, d.end.sql) shouldBe (("b", "a")) + d.spelling shouldBe DateDiffSpelling.TimestampDiffDateFirst + d.spelling.elapsed shouldBe true } // ────────────────────── the render must re-parse to the SAME node ────────────────────── @@ -131,27 +163,27 @@ class DateDiffSignSpec extends AnyFlatSpec with Matchers { /** 🔴 THIS is the correctness property, and it is the one the naive fix would have broken: with * the operands kept as written and reversed at emission, the canonical render flips the sign * back on a round trip. It matters because `MaterializedViewExtension` PERSISTS the render and - * re-runs it. - * - * Note what this does NOT say: it does not say the `DATEDIFF` spelling must survive the render. - * Because the parser stores MySQL's operands swapped, `DATE_DIFF(start, end, unit)` would - * re-parse to this same node too. That is a separate, weaker assertion below. + * re-runs it. The counting must survive it too, which is why `TIMESTAMPDIFF` keeps its name. */ - "the render" should "re-parse to a node with the same start and end, for every spelling" in { + "the render" should "re-parse to a node with the same start, end and counting, for every spelling" in { Seq( "DATEDIFF(a, b)", "DATEDIFF(a, b, DAY)", + "DATEDIFF(a, b, MONTH)", "DATE_DIFF(a, b)", "DATE_DIFF(a, b, MONTH)", + "DATEDIFF(WEEK, a, b)", + "DATE_DIFF(DAY, a, b)", "TIMESTAMPDIFF(DAY, a, b)", - "DATE_DIFF(DAY, a, b)" + "TIMESTAMPDIFF(a, b, QUARTER)" ).foreach { sql => withClue(s"[$sql] ") { val once = dateDiffOf(sql) val rendered = renderOf(sql) val twice = dateDiffOf(rendered.substring(rendered.indexOf("SELECT ") + 7).split(" AS ").head) - (twice.start.sql, twice.end.sql) shouldBe ((once.start.sql, once.end.sql)) + (twice.start.sql, twice.end.sql, twice.unit, twice.spelling.elapsed) shouldBe + ((once.start.sql, once.end.sql, once.unit, once.spelling.elapsed)) withClue(s"render [$rendered] must be a fixed point - ") { renderOf( rendered.substring(rendered.indexOf("SELECT ") + 7).split(" AS ").head @@ -162,16 +194,21 @@ class DateDiffSignSpec extends AnyFlatSpec with Matchers { } /** A READABILITY choice, pinned so it is not changed by accident — not a correctness one. - * Normalising to `DATE_DIFF(b, a, DAY)` would be equally correct; it would just hand the user - * back their own statement with the two arguments visibly exchanged. + * `DATE_DIFF(a, b, DAY)` means the same; it would just hand the user back another function. */ it should "keep MySQL's spelling, so SHOW CREATE hands back what was written" in { renderOf("DATEDIFF(a, b)") should include("DATEDIFF(a, b)") renderOf("DATEDIFF(a, b)") should not include "DATE_DIFF" } - it should "render the two ODBC/BigQuery spellings as DATE_DIFF, as before" in { + it should "keep TIMESTAMPDIFF's name, which decides what it counts" in { + renderOf("TIMESTAMPDIFF(DAY, a, b)") should include("TIMESTAMPDIFF(DAY, a, b)") + renderOf("TIMESTAMPDIFF(a, b, DAY)") should include("TIMESTAMPDIFF(a, b, DAY)") + } + + it should "render every other spelling as DATE_DIFF, its operands in the order written" in { renderOf("DATE_DIFF(a, b)") should include("DATE_DIFF(a, b, DAY)") - renderOf("TIMESTAMPDIFF(DAY, a, b)") should include("DATE_DIFF(DAY, a, b)") + renderOf("DATEDIFF(a, b, MONTH)") should include("DATE_DIFF(a, b, MONTH)") + renderOf("DATEDIFF(DAY, a, b)") should include("DATE_DIFF(DAY, a, b)") } } diff --git a/sql/src/test/scala/app/softnetwork/elastic/sql/parser/ParserSpec.scala b/sql/src/test/scala/app/softnetwork/elastic/sql/parser/ParserSpec.scala index 177e0d3ea..c092bdb25 100644 --- a/sql/src/test/scala/app/softnetwork/elastic/sql/parser/ParserSpec.scala +++ b/sql/src/test/scala/app/softnetwork/elastic/sql/parser/ParserSpec.scala @@ -1441,13 +1441,13 @@ class ParserSpec extends AnyFlatSpec with Matchers { | id INT NOT NULL COMMENT 'user identifier', | name VARCHAR FIELDS(raw Keyword COMMENT 'sortable') DEFAULT 'anonymous' OPTIONS (analyzer = 'french', search_analyzer = 'french'), | birthdate DATE, - | age INT SCRIPT AS (DATEDIFF(birthdate, CURRENT_DATE, YEAR)), + | age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)), | ingested_at TIMESTAMP DEFAULT _ingest.timestamp, | profile STRUCT FIELDS( | bio VARCHAR, | followers INT, | join_date DATE, - | seniority INT SCRIPT AS (DATEDIFF(profile.join_date, CURRENT_DATE, DAY)) + | seniority INT SCRIPT AS (DATEDIFF(CURRENT_DATE, profile.join_date, DAY)) | ) COMMENT 'user profile', | PRIMARY KEY (id) |) PARTITION BY birthdate (MONTH), OPTIONS (mappings = (dynamic = false))""".stripMargin @@ -1490,13 +1490,15 @@ class ParserSpec extends AnyFlatSpec with Matchers { // to a `LocalDate` in a processor, so both sides of `between` are the same type. """def param1 = ctx.birthdate; |def param2 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); - |def param3 = Long.valueOf(ChronoUnit.YEARS.between(param1, param2)); - |ctx.age = (param1 == null) ? null : param3""".stripMargin.replaceAll("\n", " ") + |def param3 = LocalDate.from(param2).atStartOfDay(ZoneId.of('Z')); + |def param4 = Long.valueOf((((param3.getYear() * 12L + param3.getMonthValue()) * 4294967296L + (param3.getDayOfMonth() - 1) * 86400000L + param3.getLong(ChronoField.MILLI_OF_DAY)) - ((param1.getYear() * 12L + param1.getMonthValue()) * 4294967296L + (param1.getDayOfMonth() - 1) * 86400000L + param1.getLong(ChronoField.MILLI_OF_DAY))) / 4294967296L / 12); + |ctx.age = (param1 == null) ? null : param4""".stripMargin.replaceAll("\n", " ") ) // The RESOLVED processor, which is what `CREATE TABLE` deploys. The operand is parsed // first and its runtime shape decided by `instanceof`, because `ctx.birthdate` is the raw - // JSON value of the document being indexed. This exact pipeline was run on all four - // majors: {"birthdate":"1990-05-20"} and {"birthdate":643161600000} both store `age: 36`. + // JSON value of the document being indexed. The column is the whole years elapsed from + // `birthdate` to the start of `CURRENT_DATE` (UTC), a positive age; `IngestTemporalSpec` + // pins the same processor, which was run as a real ingest pipeline. ct.schema.columns .find(_.name == "age") .flatMap(_.script) @@ -1505,8 +1507,9 @@ class ParserSpec extends AnyFlatSpec with Matchers { """def param1 = ctx.birthdate; |def param2 = (param1 instanceof String ? LocalDate.parse((param1).replace("/", "-"), DateTimeFormatter.ofPattern("yyyy-MM-dd")).atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); |def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); - |def param4 = Long.valueOf(ChronoUnit.YEARS.between(param2, param3)); - |ctx.age = (param1 == null) ? null : param4""".stripMargin.replaceAll("\n", " ") + |def param4 = LocalDate.from(param3).atStartOfDay(ZoneId.of('Z')); + |def param5 = Long.valueOf((((param4.getYear() * 12L + param4.getMonthValue()) * 4294967296L + (param4.getDayOfMonth() - 1) * 86400000L + param4.getLong(ChronoField.MILLI_OF_DAY)) - ((param2.getYear() * 12L + param2.getMonthValue()) * 4294967296L + (param2.getDayOfMonth() - 1) * 86400000L + param2.getLong(ChronoField.MILLI_OF_DAY))) / 4294967296L / 12); + |ctx.age = (param1 == null) ? null : param5""".stripMargin.replaceAll("\n", " ") ) cols.find(_.name == "ingested_at").get.defaultValue.map(_.value) shouldBe Some( "_ingest.timestamp" @@ -1518,16 +1521,16 @@ class ParserSpec extends AnyFlatSpec with Matchers { println(schema.defaultPipeline.ddl) val json = schema.defaultPipeline.json println(json) - json shouldBe """{"description":"CREATE OR REPLACE PIPELINE users_ddl_default_pipeline WITH PROCESSORS (name SET DEFAULT 'anonymous', age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR)), ingested_at SET DEFAULT _ingest.timestamp, profile.seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY)), PARTITION BY birthdate (MONTH), PRIMARY KEY (id))","processors":[{"set":{"description":"name SET DEFAULT 'anonymous'","field":"name","ignore_failure":true,"value":"anonymous","if":"ctx.name == null"}},{"script":{"description":"age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR))","lang":"painless","source":"def param1 = ctx.birthdate; def param2 = (param1 instanceof String ? LocalDate.parse((param1).replace(\"/\", \"-\"), DateTimeFormatter.ofPattern(\"yyyy-MM-dd\")).atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); def param4 = Long.valueOf(ChronoUnit.YEARS.between(param2, param3)); ctx.age = (param1 == null) ? null : param4","ignore_failure":true}},{"set":{"description":"ingested_at SET DEFAULT _ingest.timestamp","field":"ingested_at","ignore_failure":true,"value":"{{_ingest.timestamp}}","if":"ctx.ingested_at == null"}},{"script":{"description":"profile.seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY))","lang":"painless","source":"def param1 = ctx.profile?.join_date; def param2 = (param1 instanceof String ? LocalDate.parse((param1).replace(\"/\", \"-\"), DateTimeFormatter.ofPattern(\"yyyy-MM-dd\")).atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); def param4 = Long.valueOf(ChronoUnit.DAYS.between(param2, param3)); ctx.profile.seniority = (param1 == null) ? null : param4","ignore_failure":true}},{"date_index_name":{"description":"PARTITION BY birthdate (MONTH)","field":"birthdate","date_rounding":"M","date_formats":["yyyy-MM"],"index_name_prefix":"users-","ignore_failure":true}},{"set":{"description":"PRIMARY KEY (id)","field":"_id","value":"{{id}}","ignore_failure":false,"ignore_empty_value":false}}]}""" + json shouldBe """{"description":"CREATE OR REPLACE PIPELINE users_ddl_default_pipeline WITH PROCESSORS (name SET DEFAULT 'anonymous', age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)), ingested_at SET DEFAULT _ingest.timestamp, profile.seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY)), PARTITION BY birthdate (MONTH), PRIMARY KEY (id))","processors":[{"set":{"description":"name SET DEFAULT 'anonymous'","field":"name","ignore_failure":true,"value":"anonymous","if":"ctx.name == null"}},{"script":{"description":"age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE))","lang":"painless","source":"def param1 = ctx.birthdate; def param2 = (param1 instanceof String ? LocalDate.parse((param1).replace(\"/\", \"-\"), DateTimeFormatter.ofPattern(\"yyyy-MM-dd\")).atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); def param4 = LocalDate.from(param3).atStartOfDay(ZoneId.of('Z')); def param5 = Long.valueOf((((param4.getYear() * 12L + param4.getMonthValue()) * 4294967296L + (param4.getDayOfMonth() - 1) * 86400000L + param4.getLong(ChronoField.MILLI_OF_DAY)) - ((param2.getYear() * 12L + param2.getMonthValue()) * 4294967296L + (param2.getDayOfMonth() - 1) * 86400000L + param2.getLong(ChronoField.MILLI_OF_DAY))) / 4294967296L / 12); ctx.age = (param1 == null) ? null : param5","ignore_failure":true}},{"set":{"description":"ingested_at SET DEFAULT _ingest.timestamp","field":"ingested_at","ignore_failure":true,"value":"{{_ingest.timestamp}}","if":"ctx.ingested_at == null"}},{"script":{"description":"profile.seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY))","lang":"painless","source":"def param1 = ctx.profile?.join_date; def param2 = (param1 instanceof String ? LocalDate.parse((param1).replace(\"/\", \"-\"), DateTimeFormatter.ofPattern(\"yyyy-MM-dd\")).atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); def param4 = Long.valueOf(ChronoUnit.DAYS.between(LocalDate.from(param2), LocalDate.from(param3))); ctx.profile.seniority = (param1 == null) ? null : param4","ignore_failure":true}},{"date_index_name":{"description":"PARTITION BY birthdate (MONTH)","field":"birthdate","date_rounding":"M","date_formats":["yyyy-MM"],"index_name_prefix":"users-","ignore_failure":true}},{"set":{"description":"PRIMARY KEY (id)","field":"_id","value":"{{id}}","ignore_failure":false,"ignore_empty_value":false}}]}""" val indexMappings = schema.indexMappings println(indexMappings) - indexMappings.toString shouldBe """{"properties":{"id":{"type":"integer"},"name":{"type":"text","fields":{"raw":{"type":"keyword"}},"analyzer":"french","search_analyzer":"french"},"birthdate":{"type":"date"},"age":{"type":"integer"},"ingested_at":{"type":"date"},"profile":{"type":"object","properties":{"bio":{"type":"text"},"followers":{"type":"integer"},"join_date":{"type":"date"},"seniority":{"type":"integer"}}}},"dynamic":false,"_meta":{"primary_key":["id"],"partition_by":{"column":"birthdate","granularity":"M"},"columns":{"id":{"data_type":"INT","not_null":"true","comment":"user identifier"},"name":{"data_type":"VARCHAR","not_null":"false","default_value":"anonymous","multi_fields":{"raw":{"data_type":"KEYWORD","not_null":"false","comment":"sortable"}}},"birthdate":{"data_type":"DATE","not_null":"false"},"age":{"data_type":"INT","not_null":"false","script":{"sql":"DATE_DIFF(birthdate, CURRENT_DATE, YEAR)","column":"age","painless":"def param1 = ctx.birthdate; def param2 = (param1 instanceof String ? LocalDate.parse((param1).replace(\"/\", \"-\"), DateTimeFormatter.ofPattern(\"yyyy-MM-dd\")).atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); def param4 = Long.valueOf(ChronoUnit.YEARS.between(param2, param3)); ctx.age = (param1 == null) ? null : param4"}},"ingested_at":{"data_type":"TIMESTAMP","not_null":"false","default_value":"_ingest.timestamp"},"profile":{"data_type":"STRUCT","not_null":"false","comment":"user profile","multi_fields":{"bio":{"data_type":"VARCHAR","not_null":"false"},"followers":{"data_type":"INT","not_null":"false"},"join_date":{"data_type":"DATE","not_null":"false"},"seniority":{"data_type":"INT","not_null":"false","script":{"sql":"DATE_DIFF(profile.join_date, CURRENT_DATE, DAY)","column":"profile.seniority","painless":"def param1 = ctx.profile?.join_date; def param2 = (param1 instanceof String ? LocalDate.parse((param1).replace(\"/\", \"-\"), DateTimeFormatter.ofPattern(\"yyyy-MM-dd\")).atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); def param4 = Long.valueOf(ChronoUnit.DAYS.between(param2, param3)); ctx.profile.seniority = (param1 == null) ? null : param4"}}}}},"type":"regular"}}""".stripMargin + indexMappings.toString shouldBe """{"properties":{"id":{"type":"integer"},"name":{"type":"text","fields":{"raw":{"type":"keyword"}},"analyzer":"french","search_analyzer":"french"},"birthdate":{"type":"date"},"age":{"type":"integer"},"ingested_at":{"type":"date"},"profile":{"type":"object","properties":{"bio":{"type":"text"},"followers":{"type":"integer"},"join_date":{"type":"date"},"seniority":{"type":"integer"}}}},"dynamic":false,"_meta":{"primary_key":["id"],"partition_by":{"column":"birthdate","granularity":"M"},"columns":{"id":{"data_type":"INT","not_null":"true","comment":"user identifier"},"name":{"data_type":"VARCHAR","not_null":"false","default_value":"anonymous","multi_fields":{"raw":{"data_type":"KEYWORD","not_null":"false","comment":"sortable"}}},"birthdate":{"data_type":"DATE","not_null":"false"},"age":{"data_type":"INT","not_null":"false","script":{"sql":"TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)","column":"age","painless":"def param1 = ctx.birthdate; def param2 = (param1 instanceof String ? LocalDate.parse((param1).replace(\"/\", \"-\"), DateTimeFormatter.ofPattern(\"yyyy-MM-dd\")).atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); def param4 = LocalDate.from(param3).atStartOfDay(ZoneId.of('Z')); def param5 = Long.valueOf((((param4.getYear() * 12L + param4.getMonthValue()) * 4294967296L + (param4.getDayOfMonth() - 1) * 86400000L + param4.getLong(ChronoField.MILLI_OF_DAY)) - ((param2.getYear() * 12L + param2.getMonthValue()) * 4294967296L + (param2.getDayOfMonth() - 1) * 86400000L + param2.getLong(ChronoField.MILLI_OF_DAY))) / 4294967296L / 12); ctx.age = (param1 == null) ? null : param5"}},"ingested_at":{"data_type":"TIMESTAMP","not_null":"false","default_value":"_ingest.timestamp"},"profile":{"data_type":"STRUCT","not_null":"false","comment":"user profile","multi_fields":{"bio":{"data_type":"VARCHAR","not_null":"false"},"followers":{"data_type":"INT","not_null":"false"},"join_date":{"data_type":"DATE","not_null":"false"},"seniority":{"data_type":"INT","not_null":"false","script":{"sql":"DATE_DIFF(CURRENT_DATE, profile.join_date, DAY)","column":"profile.seniority","painless":"def param1 = ctx.profile?.join_date; def param2 = (param1 instanceof String ? LocalDate.parse((param1).replace(\"/\", \"-\"), DateTimeFormatter.ofPattern(\"yyyy-MM-dd\")).atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); def param4 = Long.valueOf(ChronoUnit.DAYS.between(LocalDate.from(param2), LocalDate.from(param3))); ctx.profile.seniority = (param1 == null) ? null : param4"}}}}},"type":"regular"}}""".stripMargin val indexSettings = schema.indexSettings println(indexSettings) indexSettings.toString shouldBe """{"index":{}}""" val pipeline = schema.defaultPipelineNode println(pipeline) - pipeline.toString shouldBe """{"description":"CREATE OR REPLACE PIPELINE users_ddl_default_pipeline WITH PROCESSORS (name SET DEFAULT 'anonymous', age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR)), ingested_at SET DEFAULT _ingest.timestamp, profile.seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY)), PARTITION BY birthdate (MONTH), PRIMARY KEY (id))","processors":[{"set":{"description":"name SET DEFAULT 'anonymous'","field":"name","ignore_failure":true,"value":"anonymous","if":"ctx.name == null"}},{"script":{"description":"age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR))","lang":"painless","source":"def param1 = ctx.birthdate; def param2 = (param1 instanceof String ? LocalDate.parse((param1).replace(\"/\", \"-\"), DateTimeFormatter.ofPattern(\"yyyy-MM-dd\")).atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); def param4 = Long.valueOf(ChronoUnit.YEARS.between(param2, param3)); ctx.age = (param1 == null) ? null : param4","ignore_failure":true}},{"set":{"description":"ingested_at SET DEFAULT _ingest.timestamp","field":"ingested_at","ignore_failure":true,"value":"{{_ingest.timestamp}}","if":"ctx.ingested_at == null"}},{"script":{"description":"profile.seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY))","lang":"painless","source":"def param1 = ctx.profile?.join_date; def param2 = (param1 instanceof String ? LocalDate.parse((param1).replace(\"/\", \"-\"), DateTimeFormatter.ofPattern(\"yyyy-MM-dd\")).atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); def param4 = Long.valueOf(ChronoUnit.DAYS.between(param2, param3)); ctx.profile.seniority = (param1 == null) ? null : param4","ignore_failure":true}},{"date_index_name":{"description":"PARTITION BY birthdate (MONTH)","field":"birthdate","date_rounding":"M","date_formats":["yyyy-MM"],"index_name_prefix":"users-","ignore_failure":true}},{"set":{"description":"PRIMARY KEY (id)","field":"_id","value":"{{id}}","ignore_failure":false,"ignore_empty_value":false}}]}""" + pipeline.toString shouldBe """{"description":"CREATE OR REPLACE PIPELINE users_ddl_default_pipeline WITH PROCESSORS (name SET DEFAULT 'anonymous', age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)), ingested_at SET DEFAULT _ingest.timestamp, profile.seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY)), PARTITION BY birthdate (MONTH), PRIMARY KEY (id))","processors":[{"set":{"description":"name SET DEFAULT 'anonymous'","field":"name","ignore_failure":true,"value":"anonymous","if":"ctx.name == null"}},{"script":{"description":"age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE))","lang":"painless","source":"def param1 = ctx.birthdate; def param2 = (param1 instanceof String ? LocalDate.parse((param1).replace(\"/\", \"-\"), DateTimeFormatter.ofPattern(\"yyyy-MM-dd\")).atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); def param4 = LocalDate.from(param3).atStartOfDay(ZoneId.of('Z')); def param5 = Long.valueOf((((param4.getYear() * 12L + param4.getMonthValue()) * 4294967296L + (param4.getDayOfMonth() - 1) * 86400000L + param4.getLong(ChronoField.MILLI_OF_DAY)) - ((param2.getYear() * 12L + param2.getMonthValue()) * 4294967296L + (param2.getDayOfMonth() - 1) * 86400000L + param2.getLong(ChronoField.MILLI_OF_DAY))) / 4294967296L / 12); ctx.age = (param1 == null) ? null : param5","ignore_failure":true}},{"set":{"description":"ingested_at SET DEFAULT _ingest.timestamp","field":"ingested_at","ignore_failure":true,"value":"{{_ingest.timestamp}}","if":"ctx.ingested_at == null"}},{"script":{"description":"profile.seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY))","lang":"painless","source":"def param1 = ctx.profile?.join_date; def param2 = (param1 instanceof String ? LocalDate.parse((param1).replace(\"/\", \"-\"), DateTimeFormatter.ofPattern(\"yyyy-MM-dd\")).atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), ZoneId.of('Z')); def param4 = Long.valueOf(ChronoUnit.DAYS.between(LocalDate.from(param2), LocalDate.from(param3))); ctx.profile.seniority = (param1 == null) ? null : param4","ignore_failure":true}},{"date_index_name":{"description":"PARTITION BY birthdate (MONTH)","field":"birthdate","date_rounding":"M","date_formats":["yyyy-MM"],"index_name_prefix":"users-","ignore_failure":true}},{"set":{"description":"PRIMARY KEY (id)","field":"_id","value":"{{id}}","ignore_failure":false,"ignore_empty_value":false}}]}""" // Reconstruct EsIndex val mappings = mapper.createObjectNode() mappings.set("mappings", indexMappings) @@ -2271,7 +2274,7 @@ class ParserSpec extends AnyFlatSpec with Matchers { | value = "anonymous" | ), | SCRIPT ( - | description = "age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR))", + | description = "age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE))", | lang = "painless", | source = "def param1 = ctx.birthdate; def param2 = ZonedDateTime.now(ZoneId.of('Z')).toLocalDate(); ctx.age = (param1 == null) ? null : Long.valueOf(ChronoUnit.YEARS.between(param1, param2))", | ignore_failure = true @@ -2284,7 +2287,7 @@ class ParserSpec extends AnyFlatSpec with Matchers { | value = "_ingest.timestamp" | ), | SCRIPT ( - | description = "profile.seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY))", + | description = "profile.seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY))", | lang = "painless", | source = "def param1 = ctx.profile?.join_date; def param2 = ZonedDateTime.now(ZoneId.of('Z')).toLocalDate(); ctx.profile.seniority = (param1 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param1, param2))", | ignore_failure = true @@ -2342,8 +2345,8 @@ class ParserSpec extends AnyFlatSpec with Matchers { case Some( ScriptProcessor( IngestPipelineType.Default, - Some("age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR))"), - "DATE_DIFF(birthdate, CURRENT_DATE, YEAR)", + Some("age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE))"), + "TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)", "age", SQLTypes.Int, source, @@ -2378,9 +2381,9 @@ class ParserSpec extends AnyFlatSpec with Matchers { ScriptProcessor( IngestPipelineType.Default, Some( - "profile.seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY))" + "profile.seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY))" ), - "DATE_DIFF(profile.join_date, CURRENT_DATE, DAY)", + "DATE_DIFF(CURRENT_DATE, profile.join_date, DAY)", "profile.seniority", SQLTypes.Int, source, diff --git a/sql/src/test/scala/app/softnetwork/elastic/sql/parser/TableauDialectSpec.scala b/sql/src/test/scala/app/softnetwork/elastic/sql/parser/TableauDialectSpec.scala index 5263e64ac..758d34b80 100644 --- a/sql/src/test/scala/app/softnetwork/elastic/sql/parser/TableauDialectSpec.scala +++ b/sql/src/test/scala/app/softnetwork/elastic/sql/parser/TableauDialectSpec.scala @@ -99,20 +99,21 @@ class TableauDialectSpec extends AnyFlatSpec with Matchers { ) } - /** MySQL 8.4 defines `TIMESTAMPDIFF(unit, dt1, dt2)` as `dt2 - dt1`, and `date_diff_transact_sql` - * binds `(unit, d1, d2)` to `DateDiff(start = d1, end = d2)`, i.e. `between(d1, d2)` — the same + /** MySQL 8.4 defines `TIMESTAMPDIFF(unit, dt1, dt2)` as `dt2 - dt1` in whole units elapsed, and + * `date_diff_transact_sql` binds `(unit, d1, d2)` to `DateDiff(start = d1, end = d2)` — the same * answer with the same sign. The MySQL/ODBC order is the one that matters, so it is the one * asserted. */ - "TIMESTAMPDIFF" should "be a spelling of DATE_DIFF, in the MySQL/ODBC unit-first order" in { - // 🔴 The ARGUMENT LIST, not just the function name: `include("DATE_DIFF(")` is satisfied by - // the sign-inverted render `DATE_DIFF(MONTH, end_date, start_date)` too, so it would prove - // nothing about the very binding this test exists to pin. + "TIMESTAMPDIFF" should "keep its name and the MySQL/ODBC unit-first order" in { + // 🔴 The ARGUMENT LIST, not just the function name: `include("TIMESTAMPDIFF(")` is satisfied by + // the sign-inverted render `TIMESTAMPDIFF(MONTH, end_date, start_date)` too, so it would prove + // nothing about the very binding this test exists to pin. The NAME is kept because it counts + // elapsed units where `DATE_DIFF` counts boundaries. canonicalises( "SELECT TIMESTAMPDIFF(MONTH, start_date, end_date) AS d FROM t", - "DATE_DIFF(MONTH, start_date, end_date)" + "TIMESTAMPDIFF(MONTH, start_date, end_date)" ) - canonicalises("SELECT TIMESTAMPDIFF(DAY, a, b) AS d FROM t", "DATE_DIFF(DAY, a, b)") + canonicalises("SELECT TIMESTAMPDIFF(DAY, a, b) AS d FROM t", "TIMESTAMPDIFF(DAY, a, b)") } it should "leave the bare unit names alone" in { diff --git a/sql/src/test/scala/app/softnetwork/elastic/sql/query/HavingAliasResolutionSpec.scala b/sql/src/test/scala/app/softnetwork/elastic/sql/query/HavingAliasResolutionSpec.scala index 2adff4078..997b74a93 100644 --- a/sql/src/test/scala/app/softnetwork/elastic/sql/query/HavingAliasResolutionSpec.scala +++ b/sql/src/test/scala/app/softnetwork/elastic/sql/query/HavingAliasResolutionSpec.scala @@ -179,7 +179,7 @@ class HavingAliasResolutionSpec extends AnyFlatSpec with Matchers { "DATE_DIFF(DAY, MAX(d) - INTERVAL 1 DAY, '2024-01-01')" ), having + "TIMESTAMPDIFF(DAY, MAX(d)::DATE, '2024-01-01') > 1" -> inline( - "DATE_DIFF(DAY, MAX(d)::DATE, '2024-01-01')" + "TIMESTAMPDIFF(DAY, MAX(d)::DATE, '2024-01-01')" ), "SELECT g, ISNULL(MAX(a)::DOUBLE) AS x FROM t GROUP BY g" -> chain, "SELECT g, ISNULL(MAX(d) - INTERVAL 1 DAY) AS x FROM t GROUP BY g" -> chain, diff --git a/sql/src/test/scala/app/softnetwork/elastic/sql/query/MaterializedViewHavingSpec.scala b/sql/src/test/scala/app/softnetwork/elastic/sql/query/MaterializedViewHavingSpec.scala index 6b214a87b..4aa65b69b 100644 --- a/sql/src/test/scala/app/softnetwork/elastic/sql/query/MaterializedViewHavingSpec.scala +++ b/sql/src/test/scala/app/softnetwork/elastic/sql/query/MaterializedViewHavingSpec.scala @@ -1225,16 +1225,18 @@ class MaterializedViewHavingSpec extends AnyFlatSpec with Matchers { // string literal as a String: each operand is converted to a date first, an aggregate ONCE (it // is the instant in UTC already). Both venues render ONE calculation (the item's context-free // rendering); the RUN is GroupByCompletenessSpec's. - def date(metric: String) = - s"Instant.ofEpochMilli(((long) params.$metric)).atZone(ZoneId.of('Z')).toLocalDate()" + def instant(metric: String) = + s"Instant.ofEpochMilli(((long) params.$metric)).atZone(ZoneId.of('Z'))" + def date(metric: String) = s"${instant(metric)}.toLocalDate()" val literal = """LocalDate.parse(("2024-01-01").replace("/", "-"), DateTimeFormatter.ofPattern("yyyy-MM-dd"))""" Seq( - // MySQL's DATEDIFF is `first - second`, every other spelling `end - start` + // the dates first are `first - second`, the unit first `last - middle`; DATEDIFF and + // DATE_DIFF count calendar days, TIMESTAMPDIFF whole days elapsed between two instants "DATEDIFF(MAX(d), '2024-01-01')" -> s"Long.valueOf(ChronoUnit.DAYS.between($literal, ${date("mx")}))", - "DATE_DIFF(MAX(d), MIN(d), DAY)" -> s"Long.valueOf(ChronoUnit.DAYS.between(${date("mx")}, ${date("mn")}))", + "DATE_DIFF(MAX(d), MIN(d), DAY)" -> s"Long.valueOf(ChronoUnit.DAYS.between(${date("mn")}, ${date("mx")}))", "TIMESTAMPDIFF(DAY, MAX(d), '2024-01-01')" -> - s"Long.valueOf(ChronoUnit.DAYS.between(${date("mx")}, $literal))" + s"Long.valueOf(ChronoUnit.DAYS.between(${instant("mx")}, $literal.atStartOfDay(ZoneId.of('Z'))))" ).foreach { case (expression, script) => val body = s"SELECT city, MAX(d) AS mx, MIN(d) AS mn, $expression AS x FROM customers GROUP BY city HAVING x > 1" diff --git a/sql/src/test/scala/app/softnetwork/elastic/sql/schema/IngestTemporalSpec.scala b/sql/src/test/scala/app/softnetwork/elastic/sql/schema/IngestTemporalSpec.scala index 944cd2945..1dac75888 100644 --- a/sql/src/test/scala/app/softnetwork/elastic/sql/schema/IngestTemporalSpec.scala +++ b/sql/src/test/scala/app/softnetwork/elastic/sql/schema/IngestTemporalSpec.scala @@ -14,8 +14,9 @@ import org.scalatest.prop.TableDrivenPropertyChecks * no `get(ChronoField)`, the script threw, `ignore_failure: true` swallowed the throw, and the * computed column was simply ABSENT from the stored document. Measured on real Elasticsearch for * `YEAR`, `MONTH`, `DATE_TRUNC`, `DATE_ADD`/`DATE_SUB`, `DATE_FORMAT` and `DATE_DIFF` — and - * `DATE_DIFF(birthdate, CURRENT_DATE, YEAR)` is a PUBLISHED example - * (`documentation/sql/ddl_statements.md`, and the REPL testkit's `users` table). + * `DATE_DIFF(birthdate, CURRENT_DATE, YEAR)` was a PUBLISHED example + * (`documentation/sql/ddl_statements.md`, and the REPL testkit's `users` table), published since + * 0.24.0 as `TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)` (pinned below). * * 🔴 Three separate things had to be true for that example to work, and only the first was known: * @@ -35,8 +36,9 @@ import org.scalatest.prop.TableDrivenPropertyChecks * `SQLTypeUtils.runtimeType` already applies to a query. * * ⚠️ Every emission below was executed as a real ingest pipeline on ES 6.8.23, 7.17.29, 8.18.3 AND - * 9.0.3, against both document shapes. `DATE_DIFF(birthdate, CURRENT_DATE, YEAR)` stores `age: 36` - * for `{"birthdate":"1990-05-20"}` and for `{"birthdate":643161600000}` on all four. + * 9.0.3, against both document shapes. `DATE_DIFF(birthdate, CURRENT_DATE, YEAR)` stored `age: 36` + * for `{"birthdate":"1990-05-20"}` and for `{"birthdate":643161600000}` on all four, before 0.24.0 + * aligned its direction and counting with the vendors' (see the published example's pin below). * * KNOWN AND UNCHANGED: a document MISSING the source field still leaves the computed column * absent. `instanceof` on null throws and `ignore_failure: true` swallows it, which is the same @@ -138,9 +140,10 @@ class IngestTemporalSpec extends AnyFlatSpec with Matchers with TableDrivenPrope } it should "keep the ZonedDateTime for CURRENT_DATE too" in { - // Narrowing to a `LocalDate` here is what made `ChronoUnit.YEARS.between` refuse the pair in - // the published DATE_DIFF example. A query is unaffected and still narrows -- asserted in - // ParserSpec, whose query-side pins did not move. + // Narrowing the clock's parameter to a `LocalDate` here is what made `ChronoUnit.YEARS.between` + // refuse the pair in the published DATE_DIFF example. The CALL reads each operand's calendar + // date with `LocalDate.from`, which takes a `ZonedDateTime` and a `LocalDate` alike, so both + // sides of `between` are one type whatever the parameter holds. val s = source( "CREATE TABLE t (birthdate DATE, age INTEGER SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR)))", "age" @@ -150,11 +153,13 @@ class IngestTemporalSpec extends AnyFlatSpec with Matchers with TableDrivenPrope } it should "emit the published example exactly as it was executed" in { - // The whole point, pinned end to end. This byte string was run as an ingest pipeline on ES - // 6.8.23, 7.17.29, 8.18.3 and 9.0.3: `{"birthdate":"1990-05-20"}` and - // `{"birthdate":643161600000}` both store `age: 36`. + // The whole point, pinned end to end. The published age is `TIMESTAMPDIFF(YEAR, birthdate, + // CURRENT_DATE)`: the whole years elapsed from `birthdate` to the start of `CURRENT_DATE` + // (UTC), which the processor holds as the current instant and reads as a date. Run as an + // ingest pipeline on ES 8.18.3, `{"birthdate":"1990-05-20"}` and `{"birthdate":643161600000}` + // both store that age. source( - "CREATE TABLE t (birthdate DATE, age INTEGER SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR)))", + "CREATE TABLE t (birthdate DATE, age INTEGER SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)))", "age" ) shouldBe "def param1 = ctx.birthdate; " + @@ -163,8 +168,12 @@ class IngestTemporalSpec extends AnyFlatSpec with Matchers with TableDrivenPrope ".atStartOfDay(ZoneId.of('Z')) : Instant.ofEpochMilli(param1).atZone(ZoneId.of('Z'))); " + "def param3 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(System.currentTimeMillis()), " + "ZoneId.of('Z')); " + - "def param4 = Long.valueOf(ChronoUnit.YEARS.between(param2, param3)); " + - "ctx.age = (param1 == null) ? null : param4" + "def param4 = LocalDate.from(param3).atStartOfDay(ZoneId.of('Z')); " + + "def param5 = Long.valueOf((((param4.getYear() * 12L + param4.getMonthValue()) * 4294967296L " + + "+ (param4.getDayOfMonth() - 1) * 86400000L + param4.getLong(ChronoField.MILLI_OF_DAY)) - " + + "((param2.getYear() * 12L + param2.getMonthValue()) * 4294967296L + (param2.getDayOfMonth() " + + "- 1) * 86400000L + param2.getLong(ChronoField.MILLI_OF_DAY))) / 4294967296L / 12); " + + "ctx.age = (param1 == null) ? null : param5" } /** 🔴 Issue #384 — a COMPARISON inside a computed column must not be given the query path's diff --git a/testkit/src/main/scala/app/softnetwork/elastic/client/BiDialectExecutionSpec.scala b/testkit/src/main/scala/app/softnetwork/elastic/client/BiDialectExecutionSpec.scala index 7870a28e8..08424fc44 100644 --- a/testkit/src/main/scala/app/softnetwork/elastic/client/BiDialectExecutionSpec.scala +++ b/testkit/src/main/scala/app/softnetwork/elastic/client/BiDialectExecutionSpec.scala @@ -220,26 +220,44 @@ trait BiDialectExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKit w ) shouldBe -31L } - it should "stay the opposite of DATE_DIFF, which keeps BigQuery's order" in { - val mysql = longAt( + /** Every vendor computes `end - start`; with the dates first the END comes first (MySQL's + * `DATEDIFF(expr1, expr2)`, BigQuery's `DATE_DIFF(end_date, start_date, part)`), with the unit + * first it comes last. Until 0.24.0 the dates-first forms WITH a unit answered the opposite + * sign, on a misreading of BigQuery's signature (issue #363). + */ + it should "agree with DATE_DIFF, the dates first being first - second at every arity" in { + Seq( + s"SELECT DATEDIFF('2025-01-10'::DATE, '2025-01-01'::DATE) AS n FROM $index LIMIT 1", + s"SELECT DATEDIFF('2025-01-10'::DATE, '2025-01-01'::DATE, DAY) AS n FROM $index LIMIT 1", + s"SELECT DATE_DIFF('2025-01-10'::DATE, '2025-01-01'::DATE, DAY) AS n FROM $index LIMIT 1" + ).foreach(sql => withClue(s"[$sql] ")(longAt(rowsOf(sql).head, "n") shouldBe 9L)) + // BigQuery's own documented example. + longAt( rowsOf( - s"SELECT DATEDIFF('2025-01-10'::DATE, '2025-01-01'::DATE) AS n FROM $index LIMIT 1" + s"SELECT DATE_DIFF('2010-07-07'::DATE, '2008-12-25'::DATE, DAY) AS n FROM $index LIMIT 1" ).head, "n" - ) - val bigQuery = longAt( + ) shouldBe 559L + } + + it should "count the calendar boundaries crossed, on SQL Server's and DuckDB's documented examples" in { + // SQL Server: one year boundary between the last instant of 2005 and the first of 2006. + longAt( rowsOf( - s"SELECT DATE_DIFF('2025-01-10'::DATE, '2025-01-01'::DATE, DAY) AS n FROM $index LIMIT 1" + s"SELECT DATEDIFF(YEAR, '2005-12-31 23:59:59.9999999', '2006-01-01 00:00:00.0000000') AS n FROM $index LIMIT 1" ).head, "n" - ) - mysql shouldBe 9L - bigQuery shouldBe -9L - mysql shouldBe -bigQuery + ) shouldBe 1L + // DuckDB's date_diff: two month boundaries, though not two whole months. + longAt( + rowsOf( + s"SELECT DATEDIFF(MONTH, '1992-09-15'::DATE, '1992-11-14'::DATE) AS n FROM $index LIMIT 1" + ).head, + "n" + ) shouldBe 2L } - it should "agree with TIMESTAMPDIFF, which MySQL defines as dt2 - dt1" in { - // MySQL is itself asymmetric here, and we match it on BOTH names. + it should "agree with TIMESTAMPDIFF on the sign, which MySQL defines as dt2 - dt1" in { longAt( rowsOf( s"SELECT TIMESTAMPDIFF(DAY, '2025-01-01'::DATE, '2025-01-10'::DATE) AS n FROM $index LIMIT 1" @@ -248,24 +266,18 @@ trait BiDialectExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKit w ) shouldBe 9L } - /** The lead ruling, executed: on this ONE spelling the arity changes the sign, because the - * 3-argument form is this engine's own extension and is not MySQL. - */ - it should "keep this engine's order in the three-argument form, unlike the two-argument one" in { - val two = longAt( - rowsOf( - s"SELECT DATEDIFF('2025-01-10'::DATE, '2025-01-01'::DATE) AS n FROM $index LIMIT 1" - ).head, - "n" - ) - val three = longAt( - rowsOf( - s"SELECT DATEDIFF('2025-01-10'::DATE, '2025-01-01'::DATE, DAY) AS n FROM $index LIMIT 1" - ).head, - "n" - ) - two shouldBe 9L - three shouldBe -9L + "TIMESTAMPDIFF" should "count the whole units elapsed, on MySQL's own documented examples" in { + Seq( + "TIMESTAMPDIFF(MONTH, '2003-02-01', '2003-05-01')" -> 3L, + "TIMESTAMPDIFF(YEAR, '2002-05-01', '2001-01-01')" -> -1L, + "TIMESTAMPDIFF(MINUTE, '2003-02-01', '2003-05-01 12:05:55')" -> 128885L, + // two month boundaries, one whole month: the counting DATEDIFF does not share + "TIMESTAMPDIFF(MONTH, '1992-09-15', '1992-11-14')" -> 1L + ).foreach { case (expression, expected) => + withClue(s"[$expression] ") { + longAt(rowsOf(s"SELECT $expression AS n FROM $index LIMIT 1").head, "n") shouldBe expected + } + } } "TIMESTAMPADD" should "execute as DATETIME_ADD does, on the same statement" in { diff --git a/testkit/src/main/scala/app/softnetwork/elastic/client/DateFunctionExecutionSpec.scala b/testkit/src/main/scala/app/softnetwork/elastic/client/DateFunctionExecutionSpec.scala index 86f825cc6..f34dd3dbb 100644 --- a/testkit/src/main/scala/app/softnetwork/elastic/client/DateFunctionExecutionSpec.scala +++ b/testkit/src/main/scala/app/softnetwork/elastic/client/DateFunctionExecutionSpec.scala @@ -115,6 +115,9 @@ trait DateFunctionExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKi override def afterAll(): Unit = { client.deleteIndex(index) if (diffIndexCreated) client.deleteIndex(diffIndex) + createdTables.foreach(t => + Try(Await.result(client.run(s"DROP TABLE IF EXISTS $t"), 60.seconds)) + ) super.afterAll() } @@ -207,24 +210,29 @@ trait DateFunctionExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKi } // ----------------------------------------------------------------------------------------------- - // The DATEDIFF family against its documented semantics, computed HERE: + // The DATEDIFF family against the vendors' definitions, computed HERE, independently of the + // engine: // - // - HOUR, MINUTE and SECOND count the ELAPSED whole units between two instants (UTC), truncated - // toward zero; a DATE operand is the start of its day; - // - DAY and every larger unit compare two CALENDAR dates (UTC): days, weeks (days / 7), months - // (complete once the day of month is reached), quarters (months / 3), years (months / 12), - // truncated toward zero; - // - the answer is `second - first`, except MySQL's two-argument DATEDIFF: `first - second`; - // - a string literal is the temporal it spells, a DATE alone or a TIMESTAMP (UTC unless it names - // a zone); NULL when an operand has no value. + // - DIRECTION, by layout: `end - start`, where the dates first (`DATE_DIFF(a, b[, unit])`, + // `DATEDIFF(a, b[, unit])`, `TIMESTAMPDIFF(a, b[, unit])`) is `a - b` and the unit first + // (`DATE_DIFF(unit, a, b)`, `DATEDIFF(unit, a, b)`, `TIMESTAMPDIFF(unit, a, b)`) is `b - a`; + // - COUNTING, by name: `DATEDIFF` and `DATE_DIFF` count the calendar BOUNDARIES crossed, the + // difference of two UTC calendar-unit indices, a week starting on the ISO Monday; + // `TIMESTAMPDIFF` counts the whole units ELAPSED, truncated toward zero, a month (quarter, + // year) once the same day of month and time of day are reached; + // - a DATE operand is the start of its day (UTC); a string literal is the temporal it spells, a + // DATE alone or a TIMESTAMP (UTC unless it names a zone); NULL when an operand has no value. // - // The population is every spelling (DATE_DIFF / DATEDIFF with the unit last or first, - // TIMESTAMPDIFF, DATE_DIFF without a unit, MySQL's DATEDIFF), every unit, every operand kind (a - // date column, a timestamp column, a DATE literal, two TIMESTAMP literals, a date and a timestamp - // aggregate) and every venue: a SELECT item per document, a WHERE, a SELECT item per group and a - // HAVING through that item's alias. Before, HOUR / MINUTE / SECOND over a column or an aggregate - // failed with `Unsupported unit: Hours`, QUARTER did not compile, and a literal failed at row - // level and, with a time of day, per group. + // The fixture sits where the two countings and the two week starts part: one second (200 ms for + // a second) across a year, quarter, month, ISO week, day, hour, minute and second end; a Saturday + // to Sunday second, which is no ISO week; a span with more boundaries than whole units in every + // unit; the vendors' own 1992-09-15 -> 1992-11-14 (two month boundaries, one month elapsed) and + // 2003-02-01 -> 2003-05-01 (three months); the same instant; a DATE against a TIMESTAMP. The + // population is every spelling and unit over every ORDERED operand pair -- so every edge in both + // directions -- of a date column, a timestamp column, a DATE literal, two TIMESTAMP literals (one + // with a zone and milliseconds) and a TIMESTAMP-typed one, and of a date and a timestamp aggregate, + // in every venue: a SELECT item per document, a WHERE, a SELECT item per group and a HAVING + // through that item's alias. // ----------------------------------------------------------------------------------------------- private val diffIndex = "date_diff_semantics" @@ -233,18 +241,37 @@ trait DateFunctionExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKi private type DiffDoc = (String, String, Option[String], Option[String]) - /** (id, group, `d` a calendar date, `ts` an instant); `None`: the document lacks it. Group g1 has - * no value at all, g4 has some. + /** (id, group, `d` a calendar date, `ts` an instant); `None`: the document lacks it. Group g01 + * has no value at all, g15 several documents with some, g18 a date and no instant; every other + * group is one document, so its aggregates are that document's edge. */ private val diffDocs: Seq[DiffDoc] = Seq( - ("r1", "g1", None, None), - ("r2", "g1", None, None), - ("r3", "g2", Some("2024-01-02"), Some("2024-01-01T23:30:00Z")), - ("r4", "g3", Some("2023-11-15"), Some("2024-01-01T10:30:15Z")), - ("r5", "g3", Some("2024-02-29"), Some("2024-03-31T00:00:45Z")), - ("r6", "g4", Some("2024-01-31"), None), - ("r7", "g4", None, Some("2023-12-31T23:59:59Z")), - ("r8", "g4", Some("2025-06-15"), Some("2024-06-14T12:00:00Z")) + ("r01", "g01", None, None), + ("r02", "g01", None, None), + // one second before the date: a year (Tuesday to Wednesday), a quarter (Monday to Tuesday), a + // month (Friday to Saturday), an ISO week (Sunday to Monday), a Sunday that starts no ISO week + // (Saturday to Sunday), a day (Wednesday to Thursday) + ("r03", "g03", Some("2025-01-01"), Some("2024-12-31T23:59:59Z")), + ("r04", "g04", Some("2025-04-01"), Some("2025-03-31T23:59:59Z")), + ("r05", "g05", Some("2025-02-01"), Some("2025-01-31T23:59:59Z")), + ("r06", "g06", Some("2025-01-13"), Some("2025-01-12T23:59:59Z")), + ("r07", "g07", Some("2025-01-12"), Some("2025-01-11T23:59:59Z")), + ("r08", "g08", Some("2025-01-16"), Some("2025-01-15T23:59:59Z")), + // one second before an hour ('2025-01-15 11:00:00'); 200 ms across a minute and a second + // ('2025-01-15T10:01:00.100Z') + ("r09", "g09", Some("2025-01-15"), Some("2025-01-15T10:59:59Z")), + ("r10", "g10", None, Some("2025-01-15T10:00:59.900Z")), + // more boundaries than whole units, in every unit from SECOND to YEAR + ("r11", "g11", Some("2025-01-02"), Some("2023-11-30T22:59:30.900Z")), + // DuckDB's date_diff example, then MySQL's TIMESTAMPDIFF one + ("r12", "g12", Some("1992-09-15"), Some("1992-11-14T00:00:00Z")), + ("r13", "g13", Some("2003-02-01"), Some("2003-05-01T00:00:00Z")), + // the same instant + ("r14", "g14", Some("2025-01-15"), Some("2025-01-15T00:00:00Z")), + ("r15", "g15", Some("2025-06-15"), None), + ("r16", "g15", None, Some("2025-06-14T12:00:00Z")), + ("r17", "g15", Some("2024-01-31"), Some("2025-07-01T00:00:00Z")), + ("r18", "g18", Some("2024-02-29"), None) ) private lazy val diffIndexLoaded: Unit = { @@ -286,15 +313,20 @@ trait DateFunctionExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKi DiffOperand(sql, aggregate = false)(_ => Some(value)) private val diffLiterals: Seq[DiffOperand] = Seq( - constant("'2024-01-01'", OnDate(LocalDate.parse("2024-01-01"))), - constant("'2024-01-01 10:00:00'", AtInstant(Instant.parse("2024-01-01T10:00:00Z"))), - constant("'2023-12-31T22:15:30Z'", AtInstant(Instant.parse("2023-12-31T22:15:30Z"))) + constant("'2025-01-01'", OnDate(LocalDate.parse("2025-01-01"))), + constant("'2025-01-15 11:00:00'", AtInstant(Instant.parse("2025-01-15T11:00:00Z"))), + constant("'2025-01-15T10:01:00.100Z'", AtInstant(Instant.parse("2025-01-15T10:01:00.100Z"))), + // a typed literal, whose time of day a calendar unit must not count + constant("'2024-12-31 23:59:59'::TIMESTAMP", AtInstant(Instant.parse("2024-12-31T23:59:59Z"))) ) - private val diffRowOperands: Seq[DiffOperand] = Seq( - DiffOperand("d", aggregate = false)(_.head._3.map(x => OnDate(LocalDate.parse(x)))), + private val diffDate: DiffOperand = + DiffOperand("d", aggregate = false)(_.head._3.map(x => OnDate(LocalDate.parse(x)))) + + private val diffInstant: DiffOperand = DiffOperand("ts", aggregate = false)(_.head._4.map(x => AtInstant(Instant.parse(x)))) - ) ++ diffLiterals + + private val diffRowOperands: Seq[DiffOperand] = Seq(diffDate, diffInstant) ++ diffLiterals private val diffGroupOperands: Seq[DiffOperand] = Seq( DiffOperand("MAX(d)", aggregate = true)( @@ -311,59 +343,110 @@ trait DateFunctionExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKi ) ) ++ diffLiterals - private def utcDate(value: DiffValue): LocalDate = value match { - case OnDate(date) => date - case AtInstant(instant) => instant.atZone(ZoneOffset.UTC).toLocalDate + /** The instant an operand denotes: a DATE is the start of its day, in UTC. */ + private def instantOf(value: DiffValue): Instant = value match { + case OnDate(date) => date.atStartOfDay(ZoneOffset.UTC).toInstant + case AtInstant(instant) => instant } - private def utcSeconds(value: DiffValue): Long = value match { - case OnDate(date) => date.toEpochDay * 86400L - case AtInstant(instant) => instant.getEpochSecond + /** The UTC calendar unit an instant falls in, as an index: two indices differ by the boundaries + * crossed between them. 1970-01-01, epoch day 0, is a Thursday, so `epochDay + 3` counts the + * days from the Monday before it. + */ + private def unitIndex(unit: String, value: DiffValue): Long = { + val instant = instantOf(value) + val utc = instant.atOffset(ZoneOffset.UTC) + unit match { + case "YEAR" => utc.getYear.toLong + case "QUARTER" => utc.getYear * 4L + (utc.getMonthValue - 1) / 3 + case "MONTH" => utc.getYear * 12L + utc.getMonthValue - 1 + case "WEEK" => Math.floorDiv(utc.toLocalDate.toEpochDay + 3, 7L) + case "DAY" => utc.toLocalDate.toEpochDay + case "HOUR" => Math.floorDiv(instant.toEpochMilli, 3600000L) + case "MINUTE" => Math.floorDiv(instant.toEpochMilli, 60000L) + case "SECOND" => Math.floorDiv(instant.toEpochMilli, 1000L) + } } - /** The complete months from `start` to `end`: a month counts once its day of month is reached. */ - private def wholeMonths(start: LocalDate, end: LocalDate): Long = { - def packed(date: LocalDate): Long = - (date.getYear * 12L + date.getMonthValue - 1) * 32L + date.getDayOfMonth - (packed(end) - packed(start)) / 32L + /** The week a SUNDAY-start count would put an instant in (1970-01-04 is a Sunday). */ + private def sundayWeek(value: DiffValue): Long = + Math.floorDiv(instantOf(value).atOffset(ZoneOffset.UTC).toLocalDate.toEpochDay + 4, 7L) + + /** The calendar boundaries crossed from `start` to `end`. */ + private def boundaries(unit: String, start: DiffValue, end: DiffValue): Long = + unitIndex(unit, end) - unitIndex(unit, start) + + /** The whole months from `from` to `to`, `from` not after `to`: a month counts once `to` reaches + * `from`'s day of month and time of day. + */ + private def wholeMonths(from: Instant, to: Instant): Long = { + val (a, b) = (from.atOffset(ZoneOffset.UTC), to.atOffset(ZoneOffset.UTC)) + val months = (b.getYear * 12L + b.getMonthValue) - (a.getYear * 12L + a.getMonthValue) + val reached = b.getDayOfMonth > a.getDayOfMonth || + (b.getDayOfMonth == a.getDayOfMonth && !b.toLocalTime.isBefore(a.toLocalTime)) + if (reached) months else months - 1 } - /** `end - start` in `unit`, truncated toward zero (Scala's `Long` division). */ - private def documentedDiff(unit: String, start: DiffValue, end: DiffValue): Long = unit match { - case "SECOND" => utcSeconds(end) - utcSeconds(start) - case "MINUTE" => (utcSeconds(end) - utcSeconds(start)) / 60L - case "HOUR" => (utcSeconds(end) - utcSeconds(start)) / 3600L - case "DAY" => utcDate(end).toEpochDay - utcDate(start).toEpochDay - case "WEEK" => (utcDate(end).toEpochDay - utcDate(start).toEpochDay) / 7L - case "MONTH" => wholeMonths(utcDate(start), utcDate(end)) - case "QUARTER" => wholeMonths(utcDate(start), utcDate(end)) / 3L - case "YEAR" => wholeMonths(utcDate(start), utcDate(end)) / 12L + /** The whole units elapsed from `start` to `end`, truncated toward zero. */ + private def elapsed(unit: String, start: DiffValue, end: DiffValue): Long = { + val (from, to) = (instantOf(start), instantOf(end)) + if (to.isBefore(from)) -elapsed(unit, end, start) + else { + val millis = to.toEpochMilli - from.toEpochMilli + unit match { + case "YEAR" => wholeMonths(from, to) / 12L + case "QUARTER" => wholeMonths(from, to) / 3L + case "MONTH" => wholeMonths(from, to) + case "WEEK" => millis / 604800000L + case "DAY" => millis / 86400000L + case "HOUR" => millis / 3600000L + case "MINUTE" => millis / 60000L + case "SECOND" => millis / 1000L + } + } } - /** A spelling of the family: its SQL over two operands, its unit, and whether it is MySQL's - * two-argument DATEDIFF (`first - second`). + /** A spelling of the family: its SQL over two operands in WRITTEN order, its unit, its layout + * (the dates first: `first - second`; the unit first: `last - middle`) and its counting. */ - private final case class DiffSpelling(unit: String, mysql: Boolean)( + private final case class DiffSpelling(unit: String, datesFirst: Boolean, countsElapsed: Boolean)( val sql: (String, String) => String ) + /** What a spelling answers for its two operands, in written order. */ + private def answer(spelling: DiffSpelling, first: DiffValue, second: DiffValue): Long = { + val (start, end) = if (spelling.datesFirst) (second, first) else (first, second) + if (spelling.countsElapsed) elapsed(spelling.unit, start, end) + else boundaries(spelling.unit, start, end) + } + private val diffUnits: Seq[String] = Seq("YEAR", "QUARTER", "MONTH", "WEEK", "DAY", "HOUR", "MINUTE", "SECOND") private val diffSpellings: Seq[DiffSpelling] = diffUnits.flatMap { u => Seq( - DiffSpelling(u, mysql = false)((a, b) => s"DATE_DIFF($a, $b, $u)"), - DiffSpelling(u, mysql = false)((a, b) => s"DATEDIFF($a, $b, $u)"), - DiffSpelling(u, mysql = false)((a, b) => s"DATE_DIFF($u, $a, $b)"), - DiffSpelling(u, mysql = false)((a, b) => s"DATEDIFF($u, $a, $b)"), - DiffSpelling(u, mysql = false)((a, b) => s"TIMESTAMPDIFF($u, $a, $b)") + DiffSpelling(u, datesFirst = true, countsElapsed = false)((a, b) => s"DATE_DIFF($a, $b, $u)"), + DiffSpelling(u, datesFirst = true, countsElapsed = false)((a, b) => s"DATEDIFF($a, $b, $u)"), + DiffSpelling(u, datesFirst = false, countsElapsed = false)((a, b) => + s"DATE_DIFF($u, $a, $b)" + ), + DiffSpelling(u, datesFirst = false, countsElapsed = false)((a, b) => s"DATEDIFF($u, $a, $b)"), + DiffSpelling(u, datesFirst = false, countsElapsed = true)((a, b) => + s"TIMESTAMPDIFF($u, $a, $b)" + ) ) } ++ Seq( - DiffSpelling("DAY", mysql = false)((a, b) => s"DATE_DIFF($a, $b)"), - DiffSpelling("DAY", mysql = true)((a, b) => s"DATEDIFF($a, $b)") + DiffSpelling("DAY", datesFirst = true, countsElapsed = false)((a, b) => s"DATE_DIFF($a, $b)"), + DiffSpelling("DAY", datesFirst = true, countsElapsed = false)((a, b) => s"DATEDIFF($a, $b)") + ) ++ diffUnits.map { u => + // No vendor writes `TIMESTAMPDIFF` with the dates first, and it is accepted all the same: the + // two rules give it its meaning, `first - second` counted in whole units elapsed. + DiffSpelling(u, datesFirst = true, countsElapsed = true)((a, b) => s"TIMESTAMPDIFF($a, $b, $u)") + } :+ DiffSpelling("DAY", datesFirst = true, countsElapsed = true)((a, b) => + s"TIMESTAMPDIFF($a, $b)" ) - /** A statement, the venue it runs in, and the documented answer for each document or group. */ + /** A statement, the venue it runs in, and the answer for each document or group. */ private final case class DiffStatement( sql: String, venue: String, @@ -374,48 +457,70 @@ trait DateFunctionExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKi def kept: Set[String] = expected.collect { case (k, Some(v)) if v > 0 => k }.toSet } + private val diffPerDocument: Seq[(String, Seq[DiffDoc])] = diffDocs.map(doc => doc._1 -> Seq(doc)) + + private val diffPerGroup: Seq[(String, Seq[DiffDoc])] = diffDocs.groupBy(_._2).toSeq.sortBy(_._1) + + private val diffRowPairs: Seq[(DiffOperand, DiffOperand)] = + for (a <- diffRowOperands; b <- diffRowOperands) yield (a, b) + + private val diffGroupPairs: Seq[(DiffOperand, DiffOperand)] = + for (a <- diffGroupOperands; b <- diffGroupOperands if a.aggregate || b.aggregate) + yield (a, b) + + /** A spelling over two operands in a venue, and its answer for each document or group. */ + private def diffStatement( + spelling: DiffSpelling, + venue: String, + a: DiffOperand, + b: DiffOperand, + scopes: Seq[(String, Seq[DiffDoc])] + ): DiffStatement = { + val e = spelling.sql(a.sql, b.sql) + val sql = venue match { + case "SELECT" => s"SELECT id, $e AS x FROM $diffIndex" + case "WHERE" => s"SELECT id FROM $diffIndex WHERE $e > 0" + case "GROUP" => s"SELECT g, $e AS x FROM $diffIndex GROUP BY g" + case _ => s"SELECT g, $e AS x FROM $diffIndex GROUP BY g HAVING x > 0" + } + val expected = scopes.map { case (key, docs) => + key -> (for { + first <- a.value(docs) + second <- b.value(docs) + } yield answer(spelling, first, second)) + }.toMap + DiffStatement(sql, venue, spelling.unit, expected) + } + private lazy val diffPopulation: Seq[DiffStatement] = { - val perDocument: Seq[(String, Seq[DiffDoc])] = diffDocs.map(doc => doc._1 -> Seq(doc)) - val perGroup: Seq[(String, Seq[DiffDoc])] = diffDocs.groupBy(_._2).toSeq.sortBy(_._1) - val rowPairs = for (a <- diffRowOperands; b <- diffRowOperands) yield (a, b) // A WHERE over two literals reads no column: it is a constant predicate, rendered as a bare // comparison of the function's boxed result, which fails to compile whatever the function is // (`WHERE ABS(-3) > 0` too). It is not a DATEDIFF question, so it is left out. - val wherePairs = rowPairs.filterNot { case (a, b) => + val wherePairs = diffRowPairs.filterNot { case (a, b) => diffLiterals.contains(a) && diffLiterals.contains(b) } - val groupPairs = - for (a <- diffGroupOperands; b <- diffGroupOperands if a.aggregate || b.aggregate) - yield (a, b) for { spelling <- diffSpellings (venue, pairs, scopes) <- Seq( - ("SELECT", rowPairs, perDocument), - ("WHERE", wherePairs, perDocument), - ("GROUP", groupPairs, perGroup), - ("HAVING", groupPairs, perGroup) + ("SELECT", diffRowPairs, diffPerDocument), + ("WHERE", wherePairs, diffPerDocument), + ("GROUP", diffGroupPairs, diffPerGroup), + ("HAVING", diffGroupPairs, diffPerGroup) ) (a, b) <- pairs - } yield { - val e = spelling.sql(a.sql, b.sql) - val sql = venue match { - case "SELECT" => s"SELECT id, $e AS x FROM $diffIndex" - case "WHERE" => s"SELECT id FROM $diffIndex WHERE $e > 0" - case "GROUP" => s"SELECT g, $e AS x FROM $diffIndex GROUP BY g" - case _ => s"SELECT g, $e AS x FROM $diffIndex GROUP BY g HAVING x > 0" - } - val expected = scopes.map { case (key, docs) => - key -> (for { - first <- a.value(docs) - second <- b.value(docs) - } yield - if (spelling.mysql) documentedDiff(spelling.unit, second, first) - else documentedDiff(spelling.unit, first, second)) - }.toMap - DiffStatement(sql, venue, spelling.unit, expected) - } + } yield diffStatement(spelling, venue, a, b, scopes) } + /** Every pair of operand values the population computes over, in written order. */ + private lazy val diffValuePairs: Seq[(DiffValue, DiffValue)] = + (for { + (pairs, scopes) <- Seq(diffRowPairs -> diffPerDocument, diffGroupPairs -> diffPerGroup) + (a, b) <- pairs + (_, docs) <- scopes + first <- a.value(docs) + second <- b.value(docs) + } yield (first, second)).distinct + /** The rows of a statement through the gateway, or why there are none: a streamed result fails * while it is consumed, so the whole read is guarded. */ @@ -443,13 +548,13 @@ trait DateFunctionExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKi } } - "the DATEDIFF family" should "answer its documented semantics for every spelling, unit, operand and venue" in { + "the DATEDIFF family" should "answer the vendors' definitions for every spelling, unit, edge, operand and venue" in { diffIndexLoaded val population = diffPopulation // Non-vacuity, computed over the material: every spelling in every venue over every operand // pair, every unit answering a negative, a zero and a positive somewhere, NULL in every venue, // and the filters keeping some and dropping some. - population.size shouldBe diffSpellings.size * (25 + 16 + 16 + 16) + population.size shouldBe diffSpellings.size * (36 + 20 + 20 + 20) population.foreach(st => withClue(s"[${st.sql}] ")(Parser(st.sql).isRight shouldBe true)) diffUnits.foreach { u => val answers = population.filter(_.unit == u).flatMap(_.expected.values.flatten) @@ -468,7 +573,35 @@ trait DateFunctionExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKi .filter(_.filters) .exists(st => st.kept.nonEmpty && st.kept != st.expected.keySet) shouldBe true + // ...and the edges are there, in both directions: in every unit a boundary crossed with no + // whole unit elapsed, and more boundaries than whole units; a Sunday that a Sunday-start week + // would count and an ISO week does not; the same instant written as a DATE and a TIMESTAMP. + val pairs = diffValuePairs + diffUnits.foreach { u => + withClue(s"$u edges: ") { + Seq( + pairs.exists { case (x, y) => boundaries(u, x, y) == 1 && elapsed(u, x, y) == 0 }, + pairs.exists { case (x, y) => boundaries(u, x, y) == -1 && elapsed(u, x, y) == 0 }, + pairs.exists { case (x, y) => + elapsed(u, x, y) > 0 && boundaries(u, x, y) > elapsed(u, x, y) + } + ) shouldBe Seq(true, true, true) + } + } + pairs.exists { case (x, y) => + boundaries("WEEK", x, y) == 0 && sundayWeek(y) - sundayWeek(x) == 1 + } shouldBe true + pairs.exists { case (x, y) => + x.isInstanceOf[OnDate] && y.isInstanceOf[AtInstant] && instantOf(x) == instantOf(y) + } shouldBe true + + assertDiffPopulation("DATEDIFF family", population) + } + /** Every statement of a population through the gateway, against its answer: the documents or + * groups a filter keeps, or each one's value; `label` names the population in the report. + */ + private def assertDiffPopulation(label: String, population: Seq[DiffStatement]): Unit = { val outcomes: Seq[(DiffStatement, Either[String, String])] = population.map { st => st -> diffRows(st.sql).flatMap { rows => val key = if (st.venue == "SELECT" || st.venue == "WHERE") "id" else "g" @@ -498,8 +631,310 @@ trait DateFunctionExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKi } val wrong = outcomes.collect { case (st, Right(diff)) if diff.nonEmpty => s"[${st.sql}] $diff" } val errors = outcomes.collect { case (st, Left(error)) => s"[${st.sql}] $error" } + info(s"$label: ${population.size} statements; wrong: ${wrong.size}; errors: ${errors.size}") + wrong.foreach(info(_)) + errors.foreach(info(_)) + withClue( + s"${wrong.size} wrong statements, ${errors.size} errors:\n" + + (wrong.take(20) ++ errors.take(10)).mkString("\n") + "\n" + ) { + wrong shouldBe empty + errors shouldBe empty + } + } + + // ----------------------------------------------------------------------------------------------- + // DATE_DIFF and DATEDIFF over an operand that holds a column's value without being the column, + // `COALESCE(ts, d)` or `COALESCE(d, ts)`. A calendar unit compares such an operand as a date, + // from the value the node hands the script for the column; Elasticsearch 6.8 hands one of its + // own, which is no `java.time` value, and until the engine brought it to one first, every + // statement below failed there with a `class_cast_exception`. The oracle above answers them as + // it answers any operand, a COALESCE being its first argument where the document has a value for + // it and its second where it has not: `DATE_DIFF(a, b, unit)` and `DATEDIFF(unit, a, b)` in each + // calendar unit, and `DATEDIFF(a, b)`, each COALESCE against the column it falls back to, both + // ways round, as a SELECT item and in a WHERE. + // ----------------------------------------------------------------------------------------------- + + /** `COALESCE(x, y)` over a document: `x`'s value where the document has one, else `y`'s. */ + private def coalesce(x: DiffOperand, y: DiffOperand): DiffOperand = + DiffOperand(s"COALESCE(${x.sql}, ${y.sql})", aggregate = false)(docs => + x.value(docs).orElse(y.value(docs)) + ) + + /** The arguments of each COALESCE: the timestamp column first, then the date column first. */ + private val diffCoalesces: Seq[(DiffOperand, DiffOperand)] = + Seq(diffInstant -> diffDate, diffDate -> diffInstant) + + /** Each COALESCE against the column it falls back to, both ways round. */ + private val diffCoalescedPairs: Seq[(DiffOperand, DiffOperand)] = + diffCoalesces.flatMap { case (x, y) => + val c = coalesce(x, y) + Seq(c -> y, y -> c) + } + + /** `DATE_DIFF(a, b, unit)` and `DATEDIFF(unit, a, b)` per calendar unit, and `DATEDIFF(a, b)`. */ + private val diffCoalescedSpellings: Seq[DiffSpelling] = + Seq("YEAR", "QUARTER", "MONTH", "WEEK", "DAY").flatMap { u => + Seq( + DiffSpelling(u, datesFirst = true, countsElapsed = false)((a, b) => + s"DATE_DIFF($a, $b, $u)" + ), + DiffSpelling(u, datesFirst = false, countsElapsed = false)((a, b) => + s"DATEDIFF($u, $a, $b)" + ) + ) + } :+ DiffSpelling("DAY", datesFirst = true, countsElapsed = false)((a, b) => + s"DATEDIFF($a, $b)" + ) + + private lazy val diffCoalescedPopulation: Seq[DiffStatement] = + for { + spelling <- diffCoalescedSpellings + venue <- Seq("SELECT", "WHERE") + (a, b) <- diffCoalescedPairs + } yield diffStatement(spelling, venue, a, b, diffPerDocument) + + it should "answer the same definitions over a COALESCE of the date and the timestamp column" in { + diffIndexLoaded + val population = diffCoalescedPopulation + // Non-vacuity, computed over the material: 11 spellings x 4 operand pairs x 2 venues, every + // unit answering a negative, a zero and a positive somewhere, each COALESCE answered by its + // first argument on some document and by its second on another, NULL in both venues, and the + // filters keeping some and dropping some. + population.size shouldBe 11 * 4 * 2 + population.foreach(st => withClue(s"[${st.sql}] ")(Parser(st.sql).isRight shouldBe true)) + diffCoalescedSpellings.map(_.unit).distinct.foreach { u => + val answers = population.filter(_.unit == u).flatMap(_.expected.values.flatten) + withClue(s"$u: ") { + Seq(answers.exists(_ < 0), answers.contains(0L), answers.exists(_ > 0)) shouldBe + Seq(true, true, true) + } + } + diffCoalesces.foreach { case (x, y) => + withClue(s"COALESCE(${x.sql}, ${y.sql}): ") { + Seq( + diffPerDocument.exists { case (_, docs) => x.value(docs).isDefined }, + diffPerDocument.exists { case (_, docs) => + x.value(docs).isEmpty && y.value(docs).isDefined + } + ) shouldBe Seq(true, true) + } + } + Seq("SELECT", "WHERE").foreach { venue => + withClue(s"$venue: ") { + population.filter(_.venue == venue).exists(_.expected.values.exists(_.isEmpty)) shouldBe + true + } + } + population + .filter(_.filters) + .exists(st => st.kept.nonEmpty && st.kept != st.expected.keySet) shouldBe + true + + assertDiffPopulation("DATEDIFF family over COALESCE", population) + } + + // ----------------------------------------------------------------------------------------------- + // TIMESTAMPDIFF over an operand of ANOTHER runtime type. The function compares instants, and an + // operand's declared type does not say which Java type it holds: `'…'::DATETIME` is a + // `LocalDateTime`, `CURRENT_DATE` and `ts::DATE` a `LocalDate`, `NOW()` a `ZonedDateTime`, and + // `COALESCE(ts, CURRENT_DATE)` or a CASE over `CURRENT_DATE` hold one type on one row and another + // on the next. Each is the instant it denotes, in UTC (a DATE is the start of its day), in every + // venue, for every unit and in both directions. The clock is read where the statement reads it: + // a query's when its request is built, so the instants that bracket the call bound it; a computed + // column's on the node at ingest, bounded the same way with a margin for the two clocks. + // ----------------------------------------------------------------------------------------------- + + /** The tables the DDL tests below create, dropped after the suite. */ + @volatile private var createdTables: List[String] = Nil + + /** An operand whose value may read the clock: over a document or a group, at `now`. */ + private final case class ClockOperand(sql: String, aggregate: Boolean)( + val value: (Seq[DiffDoc], Instant) => Option[DiffValue] + ) + + private def utcDate(instant: Instant): LocalDate = instant.atOffset(ZoneOffset.UTC).toLocalDate + + private def clockFree(operand: DiffOperand): ClockOperand = + ClockOperand(operand.sql, operand.aggregate)((docs, _) => operand.value(docs)) + + private val datetimeLiteral: ClockOperand = + ClockOperand("'2025-01-20 11:00:00'::DATETIME", aggregate = false)((_, _) => + Some(AtInstant(Instant.parse("2025-01-20T11:00:00Z"))) + ) + + /** A `LocalDate`, read without the clock. */ + private val dateLiteral: ClockOperand = + ClockOperand("'2025-06-01'::DATE", aggregate = false)((_, _) => + Some(OnDate(LocalDate.parse("2025-06-01"))) + ) + + private val currentDate: ClockOperand = + ClockOperand("CURRENT_DATE", aggregate = false)((_, now) => Some(OnDate(utcDate(now)))) + + private val currentInstant: ClockOperand = + ClockOperand("NOW()", aggregate = false)((_, now) => Some(AtInstant(now))) + + /** A `ZonedDateTime` on a document with `ts`, a `LocalDate` on one without. */ + private val coalesced: ClockOperand = + ClockOperand("COALESCE(ts, CURRENT_DATE)", aggregate = false)((docs, now) => + Some(docs.head._4.map(x => AtInstant(Instant.parse(x))).getOrElse(OnDate(utcDate(now)))) + ) + + /** A `LocalDate`: the clock's on one document, a literal's on the others. */ + private val chosen: ClockOperand = + ClockOperand( + "CASE WHEN id = 'r03' THEN CURRENT_DATE ELSE '2025-06-01'::DATE END", + aggregate = false + )((docs, now) => + Some(OnDate(if (docs.head._1 == "r03") utcDate(now) else LocalDate.parse("2025-06-01"))) + ) + + /** A column narrowed to its calendar date: a `LocalDate`. */ + private val castToDate: ClockOperand = + ClockOperand("ts::DATE", aggregate = false)((docs, _) => + docs.head._4.map(x => OnDate(utcDate(Instant.parse(x)))) + ) + + private val runtimeColumns: Seq[ClockOperand] = + diffRowOperands.filter(o => o.sql == "d" || o.sql == "ts").map(clockFree) + + private val runtimeAggregates: Seq[ClockOperand] = + diffGroupOperands.filter(_.aggregate).map(clockFree) + + /** Each anchor against each operand, both ways round. */ + private def runtimePairs( + anchors: Seq[ClockOperand], + operands: Seq[ClockOperand] + ): Seq[(ClockOperand, ClockOperand)] = + for { + anchor <- anchors + operand <- operands + pair <- Seq(anchor -> operand, operand -> anchor) + } yield pair + + /** A statement, the venue it runs in, and the answer for each document or group at `now`. */ + private final case class RuntimeStatement(sql: String, venue: String, unit: String)( + val expected: Instant => Map[String, Option[Long]] + ) + + private lazy val runtimePopulation: Seq[RuntimeStatement] = { + // Per group the operands are constants, whatever function reads them: a COALESCE over an + // aggregate reads its raw metric, a CASE over one does not parse, and a `bucket_script` is + // sent without its script parameters, so the request clock (`params.__now__`) is null there + // and `CURRENT_DATE` and `NOW()` cannot be read. + val rowPairs = runtimePairs( + runtimeColumns, + Seq(datetimeLiteral, currentDate, currentInstant, coalesced, chosen, castToDate) + ) + val groupPairs = runtimePairs(runtimeAggregates, Seq(datetimeLiteral, dateLiteral)) + for { + unit <- diffUnits + (venue, pairs, scopes) <- Seq( + ("SELECT", rowPairs, diffPerDocument), + ("WHERE", rowPairs, diffPerDocument), + ("GROUP", groupPairs, diffPerGroup), + ("HAVING", groupPairs, diffPerGroup) + ) + (a, b) <- pairs + } yield { + val e = s"TIMESTAMPDIFF($unit, ${a.sql}, ${b.sql})" + val sql = venue match { + case "SELECT" => s"SELECT id, $e AS x FROM $diffIndex" + case "WHERE" => s"SELECT id FROM $diffIndex WHERE $e > 0" + case "GROUP" => s"SELECT g, $e AS x FROM $diffIndex GROUP BY g" + case _ => s"SELECT g, $e AS x FROM $diffIndex GROUP BY g HAVING x > 0" + } + RuntimeStatement(sql, venue, unit)(now => + scopes.map { case (key, docs) => + key -> (for { + start <- a.value(docs, now) + end <- b.value(docs, now) + } yield elapsed(unit, start, end)) + }.toMap + ) + } + } + + /** The milliseconds clock the engine reads, as an instant. */ + private def clockNow(): Instant = Instant.ofEpochMilli(System.currentTimeMillis()) + + /** Does `actual` lie between the answers at the two instants that bracket the statement? An + * answer that does not read the clock is the same at both. + */ + private def bracketed(low: Option[Long], high: Option[Long], actual: Option[Long]): Boolean = + (low, high, actual) match { + case (Some(l), Some(h), Some(v)) => v >= math.min(l, h) && v <= math.max(l, h) + case (None, None, None) => true + case _ => false + } + + "TIMESTAMPDIFF" should "compare instants whatever the runtime type of an operand, in every venue" in { + diffIndexLoaded + val population = runtimePopulation + population.size shouldBe diffUnits.size * (24 + 24 + 8 + 8) + population.foreach(st => withClue(s"[${st.sql}] ")(Parser(st.sql).isRight shouldBe true)) + // Non-vacuity: in every venue a non-zero answer and a NULL, and the filters keep some rows and + // drop others. + val probe = clockNow() + Seq("SELECT", "WHERE", "GROUP", "HAVING").foreach { venue => + val answers = population.filter(_.venue == venue).flatMap(_.expected(probe).values) + withClue(s"$venue: ") { + Seq( + answers.exists(_.exists(_ < 0)), + answers.exists(_.exists(_ > 0)), + answers.contains(None) + ) shouldBe + Seq(true, true, true) + } + } + + val outcomes: Seq[(RuntimeStatement, Either[String, String])] = population.map { st => + val before = clockNow() + val result = diffRows(st.sql) + val after = clockNow() + val (low, high) = (st.expected(before), st.expected(after)) + st -> result.flatMap { rows => + val key = if (st.venue == "SELECT" || st.venue == "WHERE") "id" else "g" + if (st.venue == "WHERE" || st.venue == "HAVING") { + val actual = rows.map(_.getOrElse(key, "?").toString).toSet + val misplaced = (low.keySet.filter { k => + val kept = actual.contains(k) + kept != low(k).exists(_ > 0) && kept != high(k).exists(_ > 0) + } ++ (actual -- low.keySet)).toSeq.sorted + Right( + if (misplaced.isEmpty) "" + else s"kept ${actual.toSeq.sorted.mkString(",")}, misplaced ${misplaced.mkString(",")}" + ) + } else { + val values = + rows.map(r => r.getOrElse(key, "?").toString -> whole(r.getOrElse("x", null))) + values.collectFirst { case (_, Left(error)) => error } match { + case Some(error) => Left(error) + case None => + val actual = values.collect { case (k, Right(v)) => k -> v }.toMap + val off = (low.keySet ++ actual.keySet).filterNot { k => + bracketed( + low.getOrElse(k, None), + high.getOrElse(k, None), + actual.getOrElse(k, None) + ) + } + Right( + if (off.isEmpty) "" + else + s"answered ${actual.toSeq.sortBy(_._1).mkString(" ")}, expected " + + low.toSeq.sortBy(_._1).mkString(" ") + ) + } + } + } + } + val wrong = outcomes.collect { case (st, Right(diff)) if diff.nonEmpty => s"[${st.sql}] $diff" } + val errors = outcomes.collect { case (st, Left(error)) => s"[${st.sql}] $error" } info( - s"DATEDIFF family: ${population.size} statements; wrong: ${wrong.size}; errors: ${errors.size}" + s"TIMESTAMPDIFF runtime types: ${population.size} statements; wrong: ${wrong.size}; " + + s"errors: ${errors.size}" ) wrong.foreach(info(_)) errors.foreach(info(_)) @@ -511,4 +946,180 @@ trait DateFunctionExecutionSpec extends AnyFlatSpecLike with ElasticDockerTestKi errors shouldBe empty } } + + private val runtimeTable = "timestampdiff_runtime_types" + + it should "compare instants whatever the runtime type of an operand, in a computed column" in { + // In an ingest script a COALESCE hands on the RAW JSON value of a column it reads, and a cast + // column is parsed a second time once parsed, whatever function holds either: neither + // `COALESCE(ts, CURRENT_DATE)` nor `ts::DATE` is one of the operands here. + val pairs = + runtimePairs(runtimeColumns, Seq(datetimeLiteral, currentDate, currentInstant, chosen)) + val columns: Seq[(String, String, ClockOperand, ClockOperand)] = for { + (unit, u) <- diffUnits.zipWithIndex + ((a, b), p) <- pairs.zipWithIndex + } yield (s"c${u}_$p", unit, a, b) + columns.size shouldBe diffUnits.size * 16 + // a document missing a column leaves the computed column absent, not NULL: both are present + val docs = diffDocs.filter { case (_, _, d, ts) => d.isDefined && ts.isDefined } + docs.map(_._1) should contain("r03") + val ddl = + s"CREATE TABLE $runtimeTable (id KEYWORD, d DATE, ts TIMESTAMP, " + + columns + .map { case (c, unit, a, b) => + s"$c BIGINT SCRIPT AS (TIMESTAMPDIFF($unit, ${a.sql}, ${b.sql}))" + } + .mkString(", ") + ", PRIMARY KEY (id))" + Await.result(client.run(ddl), 120.seconds) match { + case ElasticSuccess(_) => createdTables = runtimeTable :: createdTables + case ElasticFailure(error) => fail(s"CREATE TABLE $runtimeTable failed: ${error.message}") + } + val values = docs.map { case (id, _, d, ts) => s"('$id', '${d.get}', '${ts.get}')" } + // the node's clock against this JVM's: two seconds either side + val before = clockNow().minusSeconds(2) + Await.result( + client.run(s"INSERT INTO $runtimeTable (id, d, ts) VALUES ${values.mkString(", ")}"), + 120.seconds + ) match { + case ElasticSuccess(dml: DmlResult) => dml.inserted shouldBe docs.size.toLong + case other => fail(s"INSERT INTO $runtimeTable: $other") + } + val after = clockNow().plusSeconds(2) + client.refresh(runtimeTable) + val stored: Map[String, ListMap[String, Any]] = diffRows(s"SELECT * FROM $runtimeTable") match { + case Right(rows) => rows.map(r => r.getOrElse("id", "?").toString -> r).toMap + case Left(error) => fail(s"SELECT * FROM $runtimeTable: $error") + } + stored.keySet shouldBe docs.map(_._1).toSet + val wrong = for { + (column, unit, a, b) <- columns + doc <- docs + expectedAt = (now: Instant) => + for { + start <- a.value(Seq(doc), now) + end <- b.value(Seq(doc), now) + } yield elapsed(unit, start, end) + actual = stored(doc._1).get(column).flatMap(v => whole(v).toOption.flatten) + if !bracketed(expectedAt(before), expectedAt(after), actual) + } yield s"$column = TIMESTAMPDIFF($unit, ${a.sql}, ${b.sql}) on ${doc._1}: stored " + + s"${actual.getOrElse("nothing")}, expected ${expectedAt(before).getOrElse("NULL")}" + info( + s"TIMESTAMPDIFF runtime types, computed columns: ${columns.size} columns x ${docs.size} " + + s"documents; wrong: ${wrong.size}" + ) + wrong.foreach(info(_)) + withClue(s"${wrong.size} wrong:\n" + wrong.take(20).mkString("\n") + "\n") { + wrong shouldBe empty + } + } + + // ----------------------------------------------------------------------------------------------- + // `DATE_TRUNC(…, WEEK)` is the ISO Monday that starts the week, at 00:00 UTC, in every venue: a + // SELECT item over a TIMESTAMP and a DATE column, a WHERE, a GROUP BY (Elasticsearch's calendar + // `date_histogram`) and a computed column. Each weekday, with times of day, across a year end. + // ----------------------------------------------------------------------------------------------- + + private val weekTable = "date_trunc_week" + + private val weekDocs: Seq[(String, String)] = Seq( + "w0" -> "2024-12-29T23:59:59Z", // a Sunday: the week before + "w1" -> "2024-12-30T08:00:00Z", // Monday + "w2" -> "2024-12-31T23:59:59Z", // Tuesday, the last second of 2024 + "w3" -> "2025-01-01T00:00:00Z", // Wednesday, the first second of 2025 + "w4" -> "2025-01-02T12:30:00Z", // Thursday + "w5" -> "2025-01-03T18:00:00Z", // Friday + "w6" -> "2025-01-04T00:00:01Z", // Saturday + "w7" -> "2025-01-05T23:59:59Z", // Sunday + "w8" -> "2025-01-06T00:00:00Z" // Monday: the week after + ) + + /** The ISO Monday of an instant's UTC date: 1970-01-01, epoch day 0, is a Thursday. */ + private def isoMonday(iso: String): LocalDate = { + val day = utcDate(Instant.parse(iso)).toEpochDay + LocalDate.ofEpochDay(day - Math.floorMod(day + 3, 7L)) + } + + private val Midnight = """(\d{4}-\d{2}-\d{2})(T00:00(:00(\.0+)?)?(Z|\+00:00)?)?""".r + + /** A truncated value as its date when it is that date at 00:00 UTC; anything else as it came. */ + private def midnight(value: Any): String = + scalar(value) match { + case Midnight(date, _*) => date + case other => other + } + + "DATE_TRUNC(…, WEEK)" should "truncate to the ISO Monday, 00:00 UTC, in every venue" in { + val expected: Map[String, String] = + weekDocs.map { case (id, ts) => id -> isoMonday(ts).toString }.toMap + // every weekday, both years, three Mondays + weekDocs.map { case (_, ts) => utcDate(Instant.parse(ts)).getDayOfWeek }.toSet.size shouldBe 7 + expected.values.toSet shouldBe Set("2024-12-23", "2024-12-30", "2025-01-06") + + val ddl = + s"CREATE TABLE $weekTable (id KEYWORD, ts TIMESTAMP, d DATE, " + + "wt TIMESTAMP SCRIPT AS (DATE_TRUNC(ts, WEEK)), wd DATE SCRIPT AS (DATE_TRUNC(d, WEEK)), " + + "PRIMARY KEY (id))" + Await.result(client.run(ddl), 60.seconds) match { + case ElasticSuccess(_) => createdTables = weekTable :: createdTables + case ElasticFailure(error) => fail(s"CREATE TABLE $weekTable failed: ${error.message}") + } + val values = weekDocs.map { case (id, ts) => s"('$id', '$ts', '${ts.take(10)}')" } + Await.result( + client.run(s"INSERT INTO $weekTable (id, ts, d) VALUES ${values.mkString(", ")}"), + 60.seconds + ) match { + case ElasticSuccess(dml: DmlResult) => dml.inserted shouldBe weekDocs.size.toLong + case other => fail(s"INSERT INTO $weekTable: $other") + } + client.refresh(weekTable) + + // Every venue is read before anything is asserted, so a failure names each venue that differs. + def byId(sql: String, column: String): Either[String, Map[String, String]] = + diffRows(sql).map( + _.map(r => r.getOrElse("id", "?").toString -> midnight(r.getOrElse(column, null))).toMap + ) + + val mondays = expected.values.toSeq.distinct.sorted + val counts: Map[String, Either[String, Option[Long]]] = + expected.values.groupBy(identity).map { case (m, ms) => m -> Right(Some(ms.size.toLong)) } + val verdicts: Seq[(String, Boolean, String)] = + Seq( + s"SELECT id, DATE_TRUNC(ts, WEEK) AS w FROM $weekTable" -> "w", + s"SELECT id, DATE_TRUNC(WEEK, ts) AS w FROM $weekTable" -> "w", + s"SELECT id, DATE_TRUNC(d, WEEK) AS w FROM $weekTable" -> "w", + s"SELECT id, wt FROM $weekTable" -> "wt", + s"SELECT id, wd FROM $weekTable" -> "wd" + ).map { case (sql, column) => + val got = byId(sql, column) + (sql, got == Right(expected), got.fold(identity, _.toSeq.sorted.mkString(" "))) + } ++ + // a WHERE keeps each document under its own Monday and under no other + (for (column <- Seq("ts", "d"); monday <- mondays) yield { + val sql = + s"SELECT id FROM $weekTable WHERE DATE_TRUNC($column, WEEK) = CAST('$monday' AS DATE)" + val kept = diffRows(sql).map(_.map(_.getOrElse("id", "?").toString).toSet) + val want = expected.collect { case (id, m) if m == monday => id }.toSet + (sql, kept == Right(want), kept.fold(identity, _.toSeq.sorted.mkString(","))) + }) ++ + // a GROUP BY buckets on the same Mondays + Seq("ts", "d").map { column => + val sql = + s"SELECT DATE_TRUNC($column, WEEK) AS w, COUNT(*) AS c FROM $weekTable " + + s"GROUP BY DATE_TRUNC($column, WEEK)" + val buckets = diffRows(sql).map( + _.map(r => midnight(r.getOrElse("w", null)) -> whole(r.getOrElse("c", null))).toMap + ) + (sql, buckets == Right(counts), buckets.fold(identity, _.toSeq.sortBy(_._1).mkString(" "))) + } + verdicts.foreach { case (sql, ok, got) => + info(s"${if (ok) "RIGHT" else "WRONG"} [$sql] $got") + } + val wrong = verdicts.collect { case (sql, false, got) => s"[$sql] $got" } + withClue( + s"expected ${expected.toSeq.sorted.mkString(" ")}; ${wrong.size} venues differ:\n" + + wrong.mkString("\n") + "\n" + ) { + wrong shouldBe empty + } + } } diff --git a/testkit/src/main/scala/app/softnetwork/elastic/client/GatewayApiIntegrationSpec.scala b/testkit/src/main/scala/app/softnetwork/elastic/client/GatewayApiIntegrationSpec.scala index d934febbe..7ae2f2f30 100644 --- a/testkit/src/main/scala/app/softnetwork/elastic/client/GatewayApiIntegrationSpec.scala +++ b/testkit/src/main/scala/app/softnetwork/elastic/client/GatewayApiIntegrationSpec.scala @@ -27,7 +27,7 @@ import app.softnetwork.elastic.sql.query.SelectStatement import com.typesafe.config.ConfigFactory import java.time.temporal.ChronoUnit -import java.time.{LocalDate, ZoneOffset, ZonedDateTime} +import java.time.{LocalDate, ZoneOffset} // --------------------------------------------------------------------------- // Base test trait — to be mixed with ElasticDockerTestKit @@ -130,13 +130,13 @@ trait GatewayApiIntegrationSpec extends GatewayIntegrationTestKit { | id INT NOT NULL COMMENT 'user identifier', | name VARCHAR FIELDS(raw Keyword COMMENT 'sortable') DEFAULT 'anonymous' OPTIONS (analyzer = 'french', search_analyzer = 'french'), | birthdate DATE, - | age INT SCRIPT AS (DATEDIFF(birthdate, CURRENT_DATE, YEAR)), + | age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)), | ingested_at TIMESTAMP DEFAULT _ingest.timestamp, | profile STRUCT FIELDS( | bio VARCHAR, | followers INT, | join_date DATE, - | seniority INT SCRIPT AS (DATEDIFF(profile.join_date, CURRENT_DATE, DAY)) + | seniority INT SCRIPT AS (DATEDIFF(CURRENT_DATE, profile.join_date, DAY)) | ) COMMENT 'user profile', | PRIMARY KEY (id) |) PARTITION BY birthdate (MONTH), OPTIONS (mappings = (dynamic = false));""".stripMargin @@ -157,13 +157,13 @@ trait GatewayApiIntegrationSpec extends GatewayIntegrationTestKit { // of conflating the two facts. The runtime type now lives in `SQLTypeUtils.runtimeType`, // reached only through `GenericIdentifier.baseType`, so no release note is owed. ddl should include("birthdate DATE") - ddl should include("age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR))") + ddl should include("age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE))") ddl should include("ingested_at TIMESTAMP DEFAULT _ingest.timestamp") ddl should include("profile STRUCT FIELDS (") ddl should include("bio VARCHAR") ddl should include("followers INT") ddl should include("join_date DATE") - ddl should include("seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY))") + ddl should include("seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY))") ddl should include("PRIMARY KEY (id)") ddl should include("PARTITION BY birthdate (MONTH)") } @@ -1479,7 +1479,7 @@ trait GatewayApiIntegrationSpec extends GatewayIntegrationTestKit { | bio VARCHAR, | followers INT, | join_date DATE, - | seniority INT SCRIPT AS (DATEDIFF(profile.join_date, CURRENT_DATE, DAY)) + | seniority INT SCRIPT AS (DATEDIFF(CURRENT_DATE, profile.join_date, DAY)) | ) |);""".stripMargin @@ -1491,7 +1491,7 @@ trait GatewayApiIntegrationSpec extends GatewayIntegrationTestKit { | bio VARCHAR, | followers INT, | join_date DATE, - | seniority INT SCRIPT AS (DATEDIFF(profile.join_date, CURRENT_DATE, DAY)), + | seniority INT SCRIPT AS (DATEDIFF(CURRENT_DATE, profile.join_date, DAY)), | reputation DOUBLE DEFAULT 0.0 | );""".stripMargin @@ -1534,7 +1534,7 @@ trait GatewayApiIntegrationSpec extends GatewayIntegrationTestKit { | id INT NOT NULL, | join_date DATE, | reputation DOUBLE DEFAULT 0.0, - | seniority INT SCRIPT AS (DATEDIFF(join_date, CURRENT_DATE, DAY)) + | seniority INT SCRIPT AS (DATEDIFF(CURRENT_DATE, join_date, DAY)) |);""".stripMargin assertDdl(System.nanoTime(), client.run(create).futureValue) @@ -2756,7 +2756,7 @@ trait GatewayApiIntegrationSpec extends GatewayIntegrationTestKit { * the document being indexed, NOT the temporal object `doc['d'].value` hands a query — that part * of the earlier reading was right, and `SQLTypeUtils.coerce` still guards its temporal arms on * `isProcessorContext` for exactly that reason. What was wrong was the conclusion drawn from it: - * that `DATEDIFF(d, CURRENT_DATE, DAY)` therefore CANNOT compute at ingest. It can, once the + * that `DATEDIFF(CURRENT_DATE, d, DAY)` therefore CANNOT compute at ingest. It can, once the * operand is parsed first — and until story 21.8 Part C it did not, so `ignore_failure` left the * column unset and the value was silently missing from every stored document. * @@ -2768,8 +2768,9 @@ trait GatewayApiIntegrationSpec extends GatewayIntegrationTestKit { * `ZonedDateTime` throughout instead of narrowing `CURRENT_DATE` to a `LocalDate`. * * ⚠️ The expected value is COMPUTED, not pinned: it is a distance from `now`, so a literal would - * have been correct for one day. It is derived the way the ingest script derives it, and the ±1 - * tolerance covers an ingest and an assertion that straddle UTC midnight. + * have been correct for one day. It is derived the way the ingest script derives it — the dates + * come first, so it is `CURRENT_DATE - d` in calendar days (UTC) — and the ±1 tolerance covers + * an ingest and an assertion that straddle UTC midnight. */ it should "record what an ingest script sees for a DATE column (ctx runtime type)" in { val create = @@ -2777,7 +2778,7 @@ trait GatewayApiIntegrationSpec extends GatewayIntegrationTestKit { | id INT NOT NULL, | d DATE, | label KEYWORD, - | days INT SCRIPT AS (DATEDIFF(d, CURRENT_DATE, DAY)) + | days INT SCRIPT AS (DATEDIFF(CURRENT_DATE, d, DAY)) |);""".stripMargin assertDdl(System.nanoTime(), client.run(create).futureValue) @@ -2803,8 +2804,8 @@ trait GatewayApiIntegrationSpec extends GatewayIntegrationTestKit { val days = row.get("days").map(_ => scalarOf(row, "days")).orNull val expected = ChronoUnit.DAYS.between( - LocalDate.of(2024, 3, 15).atStartOfDay(ZoneOffset.UTC), - ZonedDateTime.now(ZoneOffset.UTC) + LocalDate.of(2024, 3, 15), + LocalDate.now(ZoneOffset.UTC) ) withClue(s"ingest-computed days = [$days], expected ~$expected: ") { // The measurement. Asserted, not merely printed, so a change in either direction is loud — @@ -3276,7 +3277,7 @@ trait GatewayApiIntegrationSpec extends GatewayIntegrationTestKit { | value = "anonymous" | ), | SCRIPT ( - | description = "age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR))", + | description = "age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE))", | lang = "painless", | source = "def param1 = ctx.birthdate; def param2 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(ctx['_ingest']['timestamp']), ZoneId.of('Z')).toLocalDate(); ctx.age = (param1 == null) ? null : Long.valueOf(ChronoUnit.YEARS.between(param1, param2))", | ignore_failure = true @@ -3289,7 +3290,7 @@ trait GatewayApiIntegrationSpec extends GatewayIntegrationTestKit { | value = "_ingest.timestamp" | ), | SCRIPT ( - | description = "profile.seniority INT SCRIPT AS (DATE_DIFF(profile.join_date, CURRENT_DATE, DAY))", + | description = "profile.seniority INT SCRIPT AS (DATE_DIFF(CURRENT_DATE, profile.join_date, DAY))", | lang = "painless", | source = "def param1 = ctx.profile?.join_date; def param2 = ZonedDateTime.ofInstant(Instant.ofEpochMilli(ctx['_ingest']['timestamp']), ZoneId.of('Z')).toLocalDate(); ctx.profile.seniority = (param1 == null) ? null : Long.valueOf(ChronoUnit.DAYS.between(param1, param2))", | ignore_failure = true diff --git a/testkit/src/main/scala/app/softnetwork/elastic/client/GroupByCompletenessSpec.scala b/testkit/src/main/scala/app/softnetwork/elastic/client/GroupByCompletenessSpec.scala index 3def597fc..1fe755ec7 100644 --- a/testkit/src/main/scala/app/softnetwork/elastic/client/GroupByCompletenessSpec.scala +++ b/testkit/src/main/scala/app/softnetwork/elastic/client/GroupByCompletenessSpec.scala @@ -2151,7 +2151,8 @@ trait GroupByCompletenessSpec extends AnyFlatSpecLike with ElasticDockerTestKit } /** A spelling of the family over two operands, and the days it answers for them: `end - start`, - * except MySQL's two-argument DATEDIFF, which is `first - second`. + * where the dates first is `first - second` and the unit first `last - middle`. Between two + * calendar dates, the days crossed and the whole days elapsed are the same number. */ private final case class DateDiffForm( sql: (String, String) => String, @@ -2162,9 +2163,9 @@ trait GroupByCompletenessSpec extends AnyFlatSpecLike with ElasticDockerTestKit def between(start: LocalDate, end: LocalDate): Long = ChronoUnit.DAYS.between(start, end) Seq( DateDiffForm((a, b) => s"DATEDIFF($a, $b)", (a, b) => between(b, a)), - DateDiffForm((a, b) => s"DATEDIFF($a, $b, DAY)", between), + DateDiffForm((a, b) => s"DATEDIFF($a, $b, DAY)", (a, b) => between(b, a)), DateDiffForm((a, b) => s"DATEDIFF(DAY, $a, $b)", between), - DateDiffForm((a, b) => s"DATE_DIFF($a, $b, DAY)", between), + DateDiffForm((a, b) => s"DATE_DIFF($a, $b, DAY)", (a, b) => between(b, a)), DateDiffForm((a, b) => s"DATE_DIFF(DAY, $a, $b)", between), DateDiffForm((a, b) => s"TIMESTAMPDIFF(DAY, $a, $b)", between) ) diff --git a/testkit/src/main/scala/app/softnetwork/elastic/client/repl/ReplGatewayIntegrationSpec.scala b/testkit/src/main/scala/app/softnetwork/elastic/client/repl/ReplGatewayIntegrationSpec.scala index 3fd3e7625..3451b251b 100644 --- a/testkit/src/main/scala/app/softnetwork/elastic/client/repl/ReplGatewayIntegrationSpec.scala +++ b/testkit/src/main/scala/app/softnetwork/elastic/client/repl/ReplGatewayIntegrationSpec.scala @@ -127,13 +127,13 @@ trait ReplGatewayIntegrationSpec extends ReplIntegrationTestKit { | id INT NOT NULL COMMENT 'user identifier', | name VARCHAR FIELDS(raw Keyword COMMENT 'sortable') DEFAULT 'anonymous' OPTIONS (analyzer = 'french', search_analyzer = 'french'), | birthdate DATE, - | age INT SCRIPT AS (DATEDIFF(birthdate, CURRENT_DATE, YEAR)), + | age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE)), | ingested_at TIMESTAMP DEFAULT _ingest.timestamp, | profile STRUCT FIELDS( | bio VARCHAR, | followers INT, | join_date DATE, - | seniority INT SCRIPT AS (DATEDIFF(profile.join_date, CURRENT_DATE, DAY)) + | seniority INT SCRIPT AS (DATEDIFF(CURRENT_DATE, profile.join_date, DAY)) | ) COMMENT 'user profile', | PRIMARY KEY (id) |) PARTITION BY birthdate (MONTH), OPTIONS (mappings = (dynamic = false))""".stripMargin @@ -146,7 +146,7 @@ trait ReplGatewayIntegrationSpec extends ReplIntegrationTestKit { ddl should include("CREATE OR REPLACE TABLE users") ddl should include("id INT NOT NULL COMMENT 'user identifier'") ddl should include("birthdate DATE") - ddl should include("age INT SCRIPT AS (DATE_DIFF(birthdate, CURRENT_DATE, YEAR))") + ddl should include("age INT SCRIPT AS (TIMESTAMPDIFF(YEAR, birthdate, CURRENT_DATE))") ddl should include("PRIMARY KEY (id)") ddl should include("PARTITION BY birthdate (MONTH)") }