From 71916d379b6f4e4e9bbd02da633e5097abbf4ed3 Mon Sep 17 00:00:00 2001 From: YukiAoto <84632330+aoto-tech@users.noreply.github.com> Date: Sat, 12 Sep 2026 19:06:21 +0900 Subject: [PATCH 1/2] feat: implement regexp_split_to_array I added PostgreSQL's regexp_split_to_array function and the Rust expression helper on the current ScalarUDF API. It supports Utf8, LargeUtf8, and Utf8View, including scalar broadcasting, common string coercion, strict NULL handling, and PostgreSQL-style zero-width delimiter boundaries. I reused the regex cache and compile constant delimiters at the first non-NULL row. I also check output byte and child offsets before appending, and added unit tests, SQL tests, and generated documentation. Regex syntax and flags still follow DataFusion's engine, so the known differences from PostgreSQL are documented. Closes apache/datafusion#13050. References apache/datafusion#13110. --- datafusion/functions/src/regex/mod.rs | 15 + .../functions/src/regex/regexpsplittoarray.rs | 586 ++++++++++++++++++ .../regexp/regexp_split_to_array.slt | 114 ++++ .../source/user-guide/sql/scalar_functions.md | 26 + 4 files changed, 741 insertions(+) create mode 100644 datafusion/functions/src/regex/regexpsplittoarray.rs create mode 100644 datafusion/sqllogictest/test_files/regexp/regexp_split_to_array.slt diff --git a/datafusion/functions/src/regex/mod.rs b/datafusion/functions/src/regex/mod.rs index 3877251c66a45..a75116798a0b7 100644 --- a/datafusion/functions/src/regex/mod.rs +++ b/datafusion/functions/src/regex/mod.rs @@ -30,6 +30,7 @@ pub mod regexpinstr; pub mod regexplike; pub mod regexpmatch; pub mod regexpreplace; +pub mod regexpsplittoarray; /// Arrow's regex kernels treat null flags as no flags, but reject empty flags. /// Normalize empty strings without copying the string buffers. @@ -45,6 +46,10 @@ make_udf_function!(regexpinstr::RegexpInstrFunc, regexp_instr); make_udf_function!(regexpmatch::RegexpMatchFunc, regexp_match); make_udf_function!(regexplike::RegexpLikeFunc, regexp_like); make_udf_function!(regexpreplace::RegexpReplaceFunc, regexp_replace); +make_udf_function!( + regexpsplittoarray::RegexpSplitToArrayFunc, + regexp_split_to_array +); pub mod expr_fn { use datafusion_expr::Expr; @@ -67,6 +72,15 @@ pub mod expr_fn { super::regexp_count().call(args) } + /// Splits a string into an array using a regular-expression delimiter. + pub fn regexp_split_to_array(values: Expr, regex: Expr, flags: Option) -> Expr { + let mut args = vec![values, regex]; + if let Some(flags) = flags { + args.push(flags); + } + super::regexp_split_to_array().call(args) + } + /// Returns a list of regular expression matches in a string. pub fn regexp_match(values: Expr, regex: Expr, flags: Option) -> Expr { let mut args = vec![values, regex]; @@ -136,6 +150,7 @@ pub fn functions() -> Vec> { regexp_instr(), regexp_like(), regexp_replace(), + regexp_split_to_array(), ] } diff --git a/datafusion/functions/src/regex/regexpsplittoarray.rs b/datafusion/functions/src/regex/regexpsplittoarray.rs new file mode 100644 index 0000000000000..5f6056cc4ff2d --- /dev/null +++ b/datafusion/functions/src/regex/regexpsplittoarray.rs @@ -0,0 +1,586 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! Split strings into lists using regular-expression delimiters. + +use std::collections::HashMap; +use std::sync::Arc; + +use arrow::array::{ + Array, ArrayBuilder, ArrayRef, AsArray, LargeStringBuilder, ListBuilder, + StringArrayType, StringBuilder, StringViewBuilder, new_null_array, +}; +use arrow::datatypes::{DataType, Field}; +use datafusion_common::{Result, ScalarValue, exec_err}; +use datafusion_expr::{ + ColumnarValue, Documentation, ScalarFunctionArgs, ScalarUDFImpl, Signature, + TypeSignature, Volatility, +}; +use datafusion_macros::user_doc; + +use super::{compile_and_cache_regex, compile_regex}; + +#[user_doc( + doc_section(label = "Regular Expression Functions"), + description = "Splits a string into an array using a [regular expression](https://docs.rs/regex/latest/regex/#syntax) as the delimiter. Zero-length matches at the beginning or end of the string, or immediately after a previous match, are ignored. A NULL argument returns NULL without validating the delimiter or flags. Uses DataFusion's regular expression engine, whose syntax, flags, and alternation matching differ from PostgreSQL; for example, `a|ab` matches `a` first rather than the longest alternative `ab`.", + syntax_example = "regexp_split_to_array(str, regexp[, flags])", + standard_argument(name = "str", prefix = "String"), + argument( + name = "regexp", + description = "Regular expression to use as the delimiter. Can be a constant, column, or function." + ), + argument( + name = "flags", + description = "Optional regular expression flags. Refer to the flags reference above for supported flags. The global flag 'g' is not supported." + ), + sql_example = r#"```sql +> SELECT regexp_split_to_array('one,two;three', '[,;]') AS parts; ++-------------------+ +| parts | ++-------------------+ +| [one, two, three] | ++-------------------+ +```"# +)] +#[derive(Debug, PartialEq, Eq, Hash)] +pub struct RegexpSplitToArrayFunc { + signature: Signature, +} + +impl Default for RegexpSplitToArrayFunc { + fn default() -> Self { + Self::new() + } +} + +impl RegexpSplitToArrayFunc { + pub fn new() -> Self { + Self { + signature: Signature::one_of( + vec![TypeSignature::String(2), TypeSignature::String(3)], + Volatility::Immutable, + ), + } + } +} + +impl ScalarUDFImpl for RegexpSplitToArrayFunc { + fn name(&self) -> &str { + "regexp_split_to_array" + } + + fn signature(&self) -> &Signature { + &self.signature + } + + fn return_type(&self, arg_types: &[DataType]) -> Result { + match arg_types.first() { + Some(kind @ (DataType::Utf8 | DataType::LargeUtf8 | DataType::Utf8View)) => { + Ok(DataType::List(Arc::new(Field::new_list_field( + kind.clone(), + true, + )))) + } + _ => exec_err!("regexp_split_to_array requires string arguments"), + } + } + + fn is_strict(&self) -> bool { + true + } + + fn invoke_with_args(&self, args: ScalarFunctionArgs) -> Result { + if !(2..=3).contains(&args.args.len()) { + return exec_err!("regexp_split_to_array requires 2 or 3 arguments"); + } + let array_length = args.args.iter().find_map(|arg| match arg { + ColumnarValue::Array(array) => Some(array.len()), + ColumnarValue::Scalar(_) => None, + }); + let row_count = array_length.unwrap_or(1); + for arg in &args.args { + if let ColumnarValue::Array(array) = arg + && array.len() != row_count + { + return exec_err!( + "regexp_split_to_array arguments must have matching array lengths" + ); + } + } + let result = if args + .args + .iter() + .any(|arg| matches!(arg, ColumnarValue::Scalar(value) if value.is_null())) + { + new_null_array(args.return_type(), row_count) + } else { + // Keep scalar arguments at length one, rather than copying a literal + // pattern or string once per row. The kernel broadcasts these values. + let arrays = args + .args + .iter() + .map(|arg| match arg { + ColumnarValue::Scalar(value) => value.to_array_of_size(1), + ColumnarValue::Array(array) => Ok(Arc::clone(array)), + }) + .collect::>>()?; + split_arrays(&arrays, row_count)? + }; + if array_length.is_none() { + Ok(ColumnarValue::Scalar(ScalarValue::try_from_array( + &result, 0, + )?)) + } else { + Ok(ColumnarValue::Array(result)) + } + } + + fn documentation(&self) -> Option<&Documentation> { + self.doc() + } +} + +fn split_arrays(args: &[ArrayRef], row_count: usize) -> Result { + let kind = args[0].data_type(); + if args.iter().any(|arg| arg.data_type() != kind) { + return exec_err!("regexp_split_to_array requires matching string types"); + } + macro_rules! dispatch { + ($accessor:ident $(::<$offset:ty>)?, $builder:expr, $byte_limit:expr) => { + split_inner( + args[0].$accessor $(::<$offset>)?(), + args[1].$accessor $(::<$offset>)?(), + args.get(2).map(|arg| arg.$accessor $(::<$offset>)?()), + row_count, + $builder, + $byte_limit, + |builder, value| builder.append_value(value), + ) + }; + } + match kind { + DataType::Utf8 => { + dispatch!(as_string::, StringBuilder::new(), i32::MAX as usize) + } + DataType::LargeUtf8 => dispatch!( + as_string::, + LargeStringBuilder::new(), + usize::try_from(i64::MAX).unwrap_or(usize::MAX) + ), + DataType::Utf8View => { + dispatch!(as_string_view, StringViewBuilder::new(), usize::MAX) + } + _ => exec_err!("regexp_split_to_array requires string arguments"), + } +} + +fn split_inner<'a, S: StringArrayType<'a> + Copy, B: ArrayBuilder>( + values: S, + patterns: S, + flags: Option, + row_count: usize, + builder: B, + byte_limit: usize, + mut append_value: impl FnMut(&mut B, &str), +) -> Result { + let mut output = ListBuilder::with_capacity(builder, row_count); + let mut cache = HashMap::new(); + let constant_pattern = + patterns.len() == 1 && flags.is_none_or(|array| array.len() == 1); + let mut constant_regex = None; + let mut size = SplitOutputSize { + bytes: 0, + elements: 0, + byte_limit, + }; + for row in 0..row_count { + let value_index = if values.len() == 1 { 0 } else { row }; + let pattern_index = if patterns.len() == 1 { 0 } else { row }; + let flag_index = flags + .as_ref() + .map(|array| if array.len() == 1 { 0 } else { row }); + if values.is_null(value_index) + || patterns.is_null(pattern_index) + || flags + .as_ref() + .zip(flag_index) + .is_some_and(|(array, index)| array.is_null(index)) + { + output.append(false); + continue; + } + let value = values.value(value_index); + let pattern = patterns.value(pattern_index); + let flags = flags + .as_ref() + .zip(flag_index) + .map(|(array, index)| array.value(index)); + // Compile a constant delimiter once, lazily: NULL rows and empty + // batches must not evaluate an otherwise unused invalid delimiter. + // Unlike regexp_like's eager scalar flag validation, this deliberately + // follows PostgreSQL's strict NULL behavior for splitting. + let regex = if constant_pattern { + match constant_regex { + Some(ref regex) => regex, + None => { + validate_flags(flags)?; + constant_regex.insert(compile_regex(pattern, flags)?) + } + } + } else { + validate_flags(flags)?; + compile_and_cache_regex(pattern, flags, &mut cache)? + }; + let mut end = 0; + for matched in regex.find_iter(value) { + // PostgreSQL's split semantics exclude zero-width delimiters at + // the edges and immediately after the previous delimiter. + if matched.is_empty() + && (matched.start() == 0 + || matched.start() == value.len() + || matched.start() == end) + { + continue; + } + size.add(matched.start() - end)?; + append_value(output.values(), &value[end..matched.start()]); + end = matched.end(); + } + size.add(value.len() - end)?; + append_value(output.values(), &value[end..]); + output.append(true); + } + Ok(Arc::new(output.finish())) +} + +fn validate_flags(flags: Option<&str>) -> Result<()> { + if flags.is_some_and(|flags| flags.contains('g')) { + return exec_err!("regexp_split_to_array does not support the global flag"); + } + Ok(()) +} + +// Check actual output growth before touching the builders. An estimate based +// on input sizes could reject a valid result when delimiters remove most bytes. +struct SplitOutputSize { + bytes: usize, + elements: usize, + byte_limit: usize, +} + +impl SplitOutputSize { + fn add(&mut self, bytes: usize) -> Result<()> { + let Some(total) = self.bytes.checked_add(bytes) else { + return exec_err!("regexp_split_to_array output byte offset overflow"); + }; + if total > self.byte_limit || self.elements >= i32::MAX as usize { + return exec_err!("regexp_split_to_array output offset overflow"); + } + self.bytes = total; + self.elements += 1; + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use arrow::compute::cast; + use datafusion_common::config::ConfigOptions; + + fn strings(values: &[Option<&str>], kind: &DataType) -> ArrayRef { + cast(&arrow::array::StringArray::from(values.to_vec()), kind).unwrap() + } + + fn invoke(args: Vec) -> Result { + let udf = RegexpSplitToArrayFunc::new(); + let types = args + .iter() + .map(ColumnarValue::data_type) + .collect::>(); + udf.invoke_with_args(ScalarFunctionArgs { + arg_fields: types + .iter() + .map(|kind| Arc::new(Field::new("arg", kind.clone(), true))) + .collect(), + return_field: Arc::new(Field::new("result", udf.return_type(&types)?, true)), + args, + number_rows: 3, + config_options: Arc::new(ConfigOptions::default()), + }) + } + + fn assert_rows( + result: ColumnarValue, + expected: &[Option>], + kind: &DataType, + ) { + let array = match result { + ColumnarValue::Array(array) => array, + ColumnarValue::Scalar(value) => value.to_array_of_size(1).unwrap(), + }; + assert_eq!( + array.data_type(), + &DataType::List(Arc::new(Field::new_list_field(kind.clone(), true))) + ); + let lists = array.as_list::(); + assert_eq!(lists.len(), expected.len()); + for (row, expected) in expected.iter().enumerate() { + match expected { + None => assert!(lists.is_null(row), "row {row} must be a null list"), + Some(values) => { + assert!(!lists.is_null(row)); + let actual = cast(&lists.value(row), &DataType::Utf8).unwrap(); + assert_eq!( + actual.as_string::().iter().collect::>(), + values.iter().map(|s| Some(*s)).collect::>() + ); + } + } + } + } + + #[test] + fn postgres_split_boundaries() { + // Expected values checked against PostgreSQL, including Unicode and + // zero-width matches that Regex::split handles differently. + let cases = [ + ("", "", vec![""]), + ("", "x", vec![""]), + ("abc", "", vec!["a", "b", "c"]), + ("abc", "^|$", vec!["abc"]), + ("abc", "a*", vec!["", "b", "c"]), + ("aaab", "a*", vec!["", "b"]), + (" a b ", r"\s+", vec!["", "a", "b", ""]), + ("a,b,", ",", vec!["a", "b", ""]), + ("é猫", "", vec!["é", "猫"]), + ("one,two;three", "([,;])", vec!["one", "two", "three"]), + ("abc", "z", vec!["abc"]), + ]; + for kind in [DataType::Utf8, DataType::LargeUtf8, DataType::Utf8View] { + let values = strings( + &cases.iter().map(|(s, _, _)| Some(*s)).collect::>(), + &kind, + ); + let patterns = strings( + &cases.iter().map(|(_, p, _)| Some(*p)).collect::>(), + &kind, + ); + let expected = cases + .iter() + .map(|(_, _, parts)| Some(parts.clone())) + .collect::>(); + assert_rows( + invoke(vec![ + ColumnarValue::Array(values), + ColumnarValue::Array(patterns), + ]) + .unwrap(), + &expected, + &kind, + ); + } + } + + #[test] + fn scalar_and_array_arguments() { + for kind in [DataType::Utf8, DataType::LargeUtf8, DataType::Utf8View] { + for mask in 0..8 { + let args = ["aBcB", "b", "i"] + .into_iter() + .enumerate() + .map(|(i, value)| { + let scalar = + ScalarValue::try_from_string(value.to_string(), &kind) + .unwrap(); + if mask & (1 << i) == 0 { + ColumnarValue::Scalar(scalar) + } else { + ColumnarValue::Array(scalar.to_array_of_size(3).unwrap()) + } + }) + .collect(); + let result = invoke(args).unwrap(); + assert_eq!(matches!(result, ColumnarValue::Scalar(_)), mask == 0); + assert_rows( + result, + &vec![Some(vec!["a", "c", ""]); if mask == 0 { 1 } else { 3 }], + &kind, + ); + } + } + } + + #[test] + fn row_flags_and_nulls() { + for kind in [DataType::Utf8, DataType::LargeUtf8, DataType::Utf8View] { + let args = [ + vec![ + Some("aBc"), + Some("aBc"), + Some("aBc"), + None, + Some("aBc"), + Some("aBc"), + ], + vec![Some("b"), Some("b"), Some("b"), Some("b"), None, Some("b")], + vec![Some("i"), Some(""), Some("i"), Some("i"), Some("i"), None], + ] + .map(|values| ColumnarValue::Array(strings(&values, &kind))); + assert_rows( + invoke(args.to_vec()).unwrap(), + &[ + Some(vec!["a", "c"]), + Some(vec!["aBc"]), + Some(vec!["a", "c"]), + None, + None, + None, + ], + &kind, + ); + for null_arg in 0..3 { + let mut args = ["abc", "b", "i"].map(|value| { + ColumnarValue::Scalar( + ScalarValue::try_from_string(value.to_string(), &kind).unwrap(), + ) + }); + args[null_arg] = + ColumnarValue::Scalar(ScalarValue::try_from(&kind).unwrap()); + assert_rows(invoke(args.to_vec()).unwrap(), &[None], &kind); + } + } + } + + #[test] + fn empty_batches_and_errors() { + for kind in [DataType::Utf8, DataType::LargeUtf8, DataType::Utf8View] { + assert_rows( + invoke(vec![ + ColumnarValue::Array(strings(&[], &kind)), + ColumnarValue::Scalar( + ScalarValue::try_from_string(",".to_string(), &kind).unwrap(), + ), + ]) + .unwrap(), + &[], + &kind, + ); + } + for (pattern, flags, message) in [ + ("[", "", "Regular expression did not compile"), + ("b", "g", "global flag"), + ("b", "!", "Regular expression did not compile"), + ] { + let args = ["abc", pattern, flags].map(|value| { + ColumnarValue::Scalar(ScalarValue::Utf8(Some(value.to_string()))) + }); + assert!( + invoke(args.to_vec()) + .unwrap_err() + .to_string() + .contains(message) + ); + } + for count in [0, 1, 4] { + let result = + RegexpSplitToArrayFunc::new().invoke_with_args(ScalarFunctionArgs { + args: vec![ + ColumnarValue::Scalar(ScalarValue::Utf8(Some( + "a".to_string() + ))); + count + ], + arg_fields: (0..count) + .map(|_| Arc::new(Field::new("arg", DataType::Utf8, true))) + .collect(), + number_rows: 1, + return_field: Arc::new(Field::new( + "result", + DataType::List(Arc::new(Field::new_list_field( + DataType::Utf8, + true, + ))), + true, + )), + config_options: Arc::new(ConfigOptions::default()), + }); + assert!( + result + .unwrap_err() + .to_string() + .contains("requires 2 or 3 arguments") + ); + } + let args = vec![ + ColumnarValue::Array(strings(&[Some("a"), Some("b")], &DataType::Utf8)), + ColumnarValue::Array(strings(&[Some(",")], &DataType::Utf8)), + ]; + assert!( + invoke(args) + .unwrap_err() + .to_string() + .contains("matching array lengths") + ); + } + + #[test] + fn unused_invalid_delimiters_are_not_evaluated() { + for kind in [DataType::Utf8, DataType::LargeUtf8, DataType::Utf8View] { + for rows in [vec![], vec![None, None]] { + for (pattern, flags) in [("[", ""), ("b", "g")] { + let args = vec![ + ColumnarValue::Array(strings(&rows, &kind)), + ColumnarValue::Scalar( + ScalarValue::try_from_string(pattern.to_string(), &kind) + .unwrap(), + ), + ColumnarValue::Scalar( + ScalarValue::try_from_string(flags.to_string(), &kind) + .unwrap(), + ), + ]; + assert_rows(invoke(args).unwrap(), &vec![None; rows.len()], &kind); + } + } + } + } + + #[test] + fn output_offset_limits() { + let mut size = SplitOutputSize { + bytes: i32::MAX as usize - 1, + elements: 0, + byte_limit: i32::MAX as usize, + }; + size.add(1).unwrap(); + assert!(size.add(1).is_err()); + let mut size = SplitOutputSize { + bytes: 0, + elements: i32::MAX as usize - 1, + byte_limit: usize::MAX, + }; + size.add(0).unwrap(); + assert!(size.add(0).is_err()); + let mut size = SplitOutputSize { + bytes: usize::MAX, + elements: 0, + byte_limit: usize::MAX, + }; + assert!(size.add(1).is_err()); + } +} diff --git a/datafusion/sqllogictest/test_files/regexp/regexp_split_to_array.slt b/datafusion/sqllogictest/test_files/regexp/regexp_split_to_array.slt new file mode 100644 index 0000000000000..d999970659827 --- /dev/null +++ b/datafusion/sqllogictest/test_files/regexp/regexp_split_to_array.slt @@ -0,0 +1,114 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +query ? +SELECT regexp_split_to_array('one,two;three', '[,;]'); +---- +[one, two, three] + +query ? +SELECT regexp_split_to_array('abc', ''); +---- +[a, b, c] + +query ? +SELECT regexp_split_to_array('abc', 'a*'); +---- +[, b, c] + +query ? +SELECT regexp_split_to_array('é猫', ''); +---- +[é, 猫] + +query I +SELECT cardinality(regexp_split_to_array('', '')); +---- +1 + +query ? +SELECT regexp_split_to_array('a,b,', ','); +---- +[a, b, ] + +query ??? +SELECT regexp_split_to_array(NULL, ','), regexp_split_to_array('abc', NULL), regexp_split_to_array('abc', 'b', NULL); +---- +NULL NULL NULL + +# Vary flags while reusing the same pattern in one batch. +query ? +SELECT regexp_split_to_array(s, p, f) FROM (VALUES + (1, 'aBc', 'b', 'i'), (2, 'aBc', 'b', ''), + (3, 'aBc', 'b', 'i'), (4, NULL, 'b', 'i'), + (5, 'aBc', NULL, 'i'), (6, 'aBc', 'b', NULL) +) AS t(id, s, p, f) ORDER BY id; +---- +[a, c] +[aBc] +[a, c] +NULL +NULL +NULL + +# All string representations preserve their element type. +query T +SELECT arrow_typeof(regexp_split_to_array(arrow_cast('a,b', 'Utf8View'), arrow_cast(',', 'Utf8View'))); +---- +List(Utf8View) + +query T +SELECT arrow_typeof(regexp_split_to_array(arrow_cast('a,b', 'LargeUtf8'), arrow_cast(',', 'LargeUtf8'))); +---- +List(LargeUtf8) + +query T +SELECT arrow_typeof(regexp_split_to_array(arrow_cast('a,b', 'Utf8'), arrow_cast(',', 'Utf8'))); +---- +List(Utf8) + +statement error regexp_split_to_array does not support the global flag +SELECT regexp_split_to_array('abc', 'b', 'g'); + +statement error Regular expression did not compile +SELECT regexp_split_to_array('abc', '['); + +# A narrow literal must not narrow a LargeUtf8 input. +query T +SELECT arrow_typeof(regexp_split_to_array(arrow_cast('a,b', 'LargeUtf8'), arrow_cast(',', 'Utf8'))); +---- +List(LargeUtf8) + +statement error failed to match any signature +SELECT regexp_split_to_array(12321, '2'); + +statement error failed to match any signature +SELECT regexp_split_to_array('12321', 2); + +statement error failed to match any signature +SELECT regexp_split_to_array('12321', '2', 1); + +query ? +SELECT regexp_split_to_array('abc', 'a|ab'); +---- +[, bc] + +# Dictionary strings are decoded to the common string representation. +query ?T +SELECT regexp_split_to_array(arrow_cast('a,b', 'Dictionary(Int32, Utf8)'), ','), arrow_typeof(regexp_split_to_array(arrow_cast('a,b', 'Dictionary(Int32, Utf8)'), arrow_cast(',', 'Utf8'))); +---- +[a, b] List(Utf8) diff --git a/docs/source/user-guide/sql/scalar_functions.md b/docs/source/user-guide/sql/scalar_functions.md index 6726b7e145524..7a59a0c6e86e6 100644 --- a/docs/source/user-guide/sql/scalar_functions.md +++ b/docs/source/user-guide/sql/scalar_functions.md @@ -2212,6 +2212,7 @@ The following regular expression functions are supported: - [regexp_like](#regexp_like) - [regexp_match](#regexp_match) - [regexp_replace](#regexp_replace) +- [regexp_split_to_array](#regexp_split_to_array) ### `regexp_count` @@ -2369,6 +2370,31 @@ SELECT regexp_replace('aBc', '(b|d)', 'Ab\\1a', 'i'); Additional examples can be found [here](https://github.com/apache/datafusion/blob/main/datafusion-examples/examples/builtin_functions/regexp.rs) +### `regexp_split_to_array` + +Splits a string into an array using a [regular expression](https://docs.rs/regex/latest/regex/#syntax) as the delimiter. Zero-length matches at the beginning or end of the string, or immediately after a previous match, are ignored. A NULL argument returns NULL without validating the delimiter or flags. Uses DataFusion's regular expression engine, whose syntax, flags, and alternation matching differ from PostgreSQL; for example, `a|ab` matches `a` first rather than the longest alternative `ab`. + +```sql +regexp_split_to_array(str, regexp[, flags]) +``` + +#### Arguments + +- **str**: String expression to operate on. Can be a constant, column, or function, and any combination of operators. +- **regexp**: Regular expression to use as the delimiter. Can be a constant, column, or function. +- **flags**: Optional regular expression flags. Refer to the flags reference above for supported flags. The global flag 'g' is not supported. + +#### Example + +```sql +> SELECT regexp_split_to_array('one,two;three', '[,;]') AS parts; ++-------------------+ +| parts | ++-------------------+ +| [one, two, three] | ++-------------------+ +``` + ## Time and Date Functions - [current_date](#current_date) From 7f4e26ad743f3389cacf3057fd47e0832e5935ea Mon Sep 17 00:00:00 2001 From: pushnanashi2 Date: Mon, 14 Sep 2026 21:39:11 +0900 Subject: [PATCH 2/2] test: cover regexp_split_to_array helpers and errors --- datafusion/functions/src/regex/mod.rs | 28 +++++++++++++++ .../functions/src/regex/regexpsplittoarray.rs | 34 +++++++++++++++++++ 2 files changed, 62 insertions(+) diff --git a/datafusion/functions/src/regex/mod.rs b/datafusion/functions/src/regex/mod.rs index a75116798a0b7..d132136e8f82a 100644 --- a/datafusion/functions/src/regex/mod.rs +++ b/datafusion/functions/src/regex/mod.rs @@ -210,6 +210,34 @@ pub fn compile_regex(regex: &str, flags: Option<&str>) -> Result