diff --git a/deepdiff/search.py b/deepdiff/search.py index 9b1b11a1..ab081f6d 100644 --- a/deepdiff/search.py +++ b/deepdiff/search.py @@ -106,7 +106,7 @@ def __init__(self, self.obj: Any = obj self.case_sensitive: bool = case_sensitive if isinstance(item, strings) else True - item = item if self.case_sensitive else (item.lower() if isinstance(item, str) else item) + item = item if self.case_sensitive or use_regexp else (item.lower() if isinstance(item, str) else item) _exclude_exact, self.exclude_glob_paths = separate_wildcard_and_exact_paths(set(exclude_paths) if exclude_paths else None) self.exclude_paths: SetOrdered = SetOrdered(_exclude_exact) if _exclude_exact else SetOrdered() self.exclude_regex_paths: List[Pattern[str]] = [re.compile(exclude_regex_path) for exclude_regex_path in exclude_regex_paths] @@ -127,7 +127,8 @@ def __init__(self, item = str(item) if self.use_regexp: try: - item = re.compile(item) + flags = 0 if self.case_sensitive else re.IGNORECASE + item = re.compile(item, flags=flags) except TypeError as e: raise TypeError(f"The passed item of {item} is not usable for regex: {e}") from None self.strict_checking: bool = strict_checking @@ -236,7 +237,7 @@ def __search_dict(self, parents_ids_added = add_to_frozen_set(parents_ids, item_id) new_parent = parent_text % (parent, item_key_str) - new_parent_cased = new_parent if self.case_sensitive else new_parent.lower() + new_parent_cased = new_parent if self.case_sensitive or self.use_regexp else new_parent.lower() str_item = str(item) if (self.match_string and str_item == new_parent_cased) or\ @@ -282,7 +283,10 @@ def __search_iterable(self, def __search_str(self, obj: Union[str, bytes, memoryview], item: Union[str, bytes, memoryview, Pattern[str]], parent: str) -> None: """Compare strings""" - obj_text = obj if self.case_sensitive else (obj.lower() if isinstance(obj, str) else obj) + obj_text = ( + obj if self.case_sensitive or self.use_regexp + else (obj.lower() if isinstance(obj, str) else obj) + ) is_matched = False if self.use_regexp and isinstance(item, type(re.compile(''))): diff --git a/tests/test_search.py b/tests/test_search.py index 3984349a..78451296 100644 --- a/tests/test_search.py +++ b/tests/test_search.py @@ -2,6 +2,7 @@ import pytest import ipaddress import logging +import re from typing import Union from deepdiff import DeepSearch, grep from datetime import datetime @@ -369,6 +370,49 @@ def test_regex_in_string(self): result = {"matched_values": {"root"}} assert DeepSearch(obj, item, verbose_level=1, use_regexp=True) == result + @pytest.mark.parametrize('pattern, values, indexes', [ + (r'\S+', ['word', ' '], [0]), + (r'\W+', ['word', '---'], [1]), + (r'\D+', ['123', 'letters'], [1]), + (r'\N{LATIN CAPITAL LETTER A}', ['a', 'A', 'b'], [0, 1]), + (r'\U0001F680', ['rocket', '\U0001F680'], [1]), + (r'(?-i:ABC)', ['ABC', 'abc'], [0]), + (r'(?-i:ABC)def', ['ABCdef', 'ABCDEF', 'abcdef'], [0, 1]), + ]) + def test_case_insensitive_regex_preserves_syntax(self, pattern, values, indexes): + expected = {'matched_values': {f'root[{index}]': values[index] for index in indexes}} + assert DeepSearch(values, pattern, use_regexp=True, verbose_level=2) == expected + + @pytest.mark.parametrize('case_sensitive, indexes', [(False, [0, 1]), (True, [0])]) + def test_regex_case_sensitivity(self, case_sensitive, indexes): + values = ['ABC', 'abc', 'other'] + expected = {'matched_values': {f'root[{index}]': values[index] for index in indexes}} + assert DeepSearch(values, 'ABC', use_regexp=True, case_sensitive=case_sensitive, verbose_level=2) == expected + + @pytest.mark.parametrize('pattern, keys', [ + (r"\['\D+'\]$", ['Name', 'name']), + (r"\['(?-i:Name)'\]$", ['Name']), + (r"\['name'\]$", ['Name', 'name']), + ]) + def test_case_insensitive_regex_preserves_dictionary_paths(self, pattern, keys): + obj = {'Name': 1, 'name': 2, '123': 3} + expected = {'matched_paths': {f"root['{key}']": obj[key] for key in keys}} + assert DeepSearch(obj, pattern, use_regexp=True, verbose_level=2) == expected + + @pytest.mark.parametrize('flags, indexes', [(0, [0]), (re.IGNORECASE, [0, 1])]) + def test_compiled_regex_keeps_its_flags(self, flags, indexes): + values = ['ABC', 'abc'] + expected = {'matched_values': {f'root[{index}]': values[index] for index in indexes}} + assert DeepSearch(values, re.compile('ABC', flags), use_regexp=True, verbose_level=2) == expected + + @pytest.mark.parametrize('case_sensitive, expected', [ + (False, {'matched_values': {'root[0]': r'\D+', 'root[1]': r'\d+'}}), + (True, {'matched_values': {'root[0]': r'\D+'}}), + ]) + def test_literal_search_keeps_case_handling(self, case_sensitive, expected): + values = [r'\D+', r'\d+', '123'] + assert DeepSearch(values, r'\D+', case_sensitive=case_sensitive, verbose_level=2) == expected + def test_regex_does_not_match_the_regex_string_itself(self): obj = ["We like python", "but not (?:p|t)ython"] item = "(?:p|t)ython"