From 5e77cac26af912ba0a2d8a7e23b0a304a8f0b5f7 Mon Sep 17 00:00:00 2001 From: Serhii A Date: Wed, 16 Sep 2026 10:12:06 +0200 Subject: [PATCH] Find a date written next to a relative one without a comma (#695) (#1385) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * Improve Czech (cs) language data (#1172) - Add month locative/dative forms (lednu, únoru, březnu, dubnu, květnu, červnu, červenci, srpnu, říjnu, prosinci) for dates like "v lednu 2023" - Add July abbreviation "črv" - Fix "za měsíc/týden/rok" not parsing: add accusative simplification patterns (previous patterns only matched instrumental -em forms) - Add tests covering the new expressions Co-Authored-By: Claude Sonnet 4.6 (1M context) * Find a date written next to a relative one without a comma (#695) search_dates('Today 13 Feb 2020') found no dates at all, while 'Today, 13 Feb 2020' found both. A chunk that does not parse as one date is split by looking for a separator that the translation and the text it was translated from hold the same number of times, and "today" is translated into "0 day ago": the translation gains two spaces the text never had, so the whitespace splitter is refused and the whole chunk is dropped. The comma form works because the comma survives the translation one for one. Cut such a chunk where the relative expression ends instead. The cut is taken only when the expression sits at one end of the chunk, when the rest of the translation keeps one word per remaining original word, when the word at that end really is the one the expression was translated from -- which the locale is asked to confirm by translating it on its own -- and when what is left of the chunk is a date. A chunk that fails any of these is left alone rather than cut at a guess, which is why 'Today 13 Feb 2020 and 14 Mar 2021', whose "and" the translation drops, still finds nothing. The cut is only looked for when no separator lined up, so a chunk that splits today keeps splitting exactly the way it does now. Fixes #695 Co-Authored-By: Claude Opus 5 (1M context) --------- Co-authored-by: Claude Sonnet 4.6 (1M context) --- dateparser/search/search.py | 69 ++++++++++++++++++++++++++++++-- tests/test_search.py | 79 +++++++++++++++++++++++++++++++++++++ 2 files changed, 145 insertions(+), 3 deletions(-) diff --git a/dateparser/search/search.py b/dateparser/search/search.py index 80ee08463..c818f2522 100644 --- a/dateparser/search/search.py +++ b/dateparser/search/search.py @@ -6,6 +6,7 @@ from dateparser.conf import apply_settings, check_settings from dateparser.custom_language_detection.language_mapping import map_languages from dateparser.date import DateDataParser +from dateparser.freshness_date_parser import _UNITS from dateparser.languages.loader import LocaleDataLoader from dateparser.search.ngram_search import _NgramDateSearch from dateparser.search.text_detection import FullTextLanguageDetector @@ -13,6 +14,13 @@ RELATIVE_REG = re.compile("(ago|in|from now|tomorrow|today|yesterday)") +# How a relative expression looks once translated. A single word such as "today" +# turns into several words ("0 day ago"), which is why the translation of a +# chunk can hold more words than the text it was translated from. +TRANSLATED_RELATIVE_REG = re.compile( + r"\bin \d+ (?:{units})s?\b|\b\d+ (?:{units})s? ago\b".format(units=_UNITS) +) + def date_is_relative(translation): return re.search(RELATIVE_REG, translation) is not None @@ -104,12 +112,61 @@ def split_by(self, item, original, splitter): all_possible_splits.append([item_partially_split, original_partially_split]) return all_possible_splits - def split_if_not_parsed(self, item, original): + def split_by_relative_expression(self, parser, item, original, language, settings): + """Split a chunk into a relative expression and the date next to it. + + A word translated into a multi-word relative expression gives the + translation more separators than the text it was translated from, which + is what keeps the splitters above from lining the two up. Cutting the + chunk in two where the expression ends lines them up again, once the + word the expression was translated from is known and the rest of the + chunk turns out to be the date it was written next to. + """ + words = original.split() + possible_splits = [] + for match in TRANSLATED_RELATIVE_REG.finditer(item): + before = item[: match.start()].strip() + after = item[match.end() :].strip() + if bool(before) == bool(after): + # The expression is the whole chunk, or sits between two parts of + # it, which leaves no boundary the original text can be cut at. + continue + rest = after or before + if len(rest.split()) != len(words) - 1: + # What is left of the translation does not keep one word per + # remaining original one, so it would not line up either. + continue + index = 1 if after else len(words) - 1 + expression = words[0] if after else words[-1] + if language.translate(expression, settings=settings) != match.group(): + # The word next to the boundary is not the one the expression was + # translated from: some other word of the chunk also changed the + # word count, and only made the one above add up. + continue + if parser.get_date_data(rest)["date_obj"] is None: + # There is no date next to the expression, so cutting the chunk + # would report the expression and drop the rest of it. + continue + possible_splits.append( + [ + [match.group(), after] if after else [before, match.group()], + [" ".join(words[:index]), " ".join(words[index:])], + ] + ) + return possible_splits + + def split_if_not_parsed(self, parser, item, original, language, settings): splitters = [",", "،", "——", "—", "–", ".", " "] possible_splits = [] for splitter in splitters: if splitter in item and item.count(splitter) == original.count(splitter): possible_splits.extend(self.split_by(item, original, splitter)) + if not possible_splits: + # Only when no splitter lines up, so that a chunk which already + # splits keeps being split exactly the way it is split today. + possible_splits = self.split_by_relative_expression( + parser, item, original, language, settings + ) return possible_splits def parse_item(self, parser, item, translated_item, parsed, need_relative_base): @@ -127,7 +184,9 @@ def parse_item(self, parser, item, translated_item, parsed, need_relative_base): parsed_item = parser.get_date_data(item) return parsed_item, is_relative - def parse_found_objects(self, parser, to_parse, original, translated, settings): + def parse_found_objects( + self, parser, to_parse, original, translated, settings, language + ): parsed = [] substrings = [] need_relative_base = True @@ -145,7 +204,9 @@ def parse_found_objects(self, parser, to_parse, original, translated, settings): substrings.append(original[i].strip(" .,:()[]-'")) continue - possible_splits = self.split_if_not_parsed(item, original[i]) + possible_splits = self.split_if_not_parsed( + parser, item, original[i], language, settings + ) if not possible_splits: continue @@ -179,6 +240,7 @@ def parse_found_objects(self, parser, to_parse, original, translated, settings): return parsed, substrings def search_parse(self, shortname, text, settings): + language = self.get_current_language(shortname) translated, original = self.search(shortname, text, settings) bad_translate_with_search = [ "vi", @@ -198,6 +260,7 @@ def search_parse(self, shortname, text, settings): original=original, translated=translated, settings=settings, + language=language, ) results = list(zip(substrings, [i[0]["date_obj"] for i in parsed])) diff --git a/tests/test_search.py b/tests/test_search.py index 1428e9e14..b1e634400 100644 --- a/tests/test_search.py +++ b/tests/test_search.py @@ -12,6 +12,7 @@ from tests import BaseTestCase today = datetime.datetime.now(tz=pytz.timezone("UTC")) +relative_base = datetime.datetime(2020, 2, 13, 20, 7, 6) class TestTranslateSearch(BaseTestCase): @@ -1048,6 +1049,84 @@ def test_date_search_function(self, text, languages, settings, expected): result = search_dates(text, languages=languages, settings=settings) self.assertEqual(result, expected) + @parameterized.expand( + [ + # The comma used to be what told the expression and the date apart + param( + text="Today 13 Feb 2020", + languages=["en"], + expected=[ + ("Today", relative_base), + ("13 Feb 2020", datetime.datetime(2020, 2, 13, 0, 0)), + ], + ), + param( + text="Today, 13 Feb 2020", + languages=["en"], + expected=[ + ("Today", relative_base), + ("13 Feb 2020", datetime.datetime(2020, 2, 13, 0, 0)), + ], + ), + # The expression written after the date instead of before it + param( + text="13 Feb 2020 Today", + languages=["en"], + expected=[ + ("13 Feb 2020", datetime.datetime(2020, 2, 13, 0, 0)), + ("Today", relative_base), + ], + ), + # An expression translated into "in 1 day" instead of "0 day ago" + param( + text="Tomorrow 13 Feb 2020", + languages=["en"], + expected=[ + ("Tomorrow", datetime.datetime(2020, 2, 14, 20, 7, 6)), + ("13 Feb 2020", datetime.datetime(2020, 2, 13, 0, 0)), + ], + ), + param( + text="hoy 13 feb 2020", + languages=["es"], + expected=[ + ("hoy", relative_base), + ("13 feb 2020", datetime.datetime(2020, 2, 13, 0, 0)), + ], + ), + # "and" is written by the text but not by its translation, so the + # words left of the expression no longer line up with the ones + # they were translated from + param( + text="Today 13 Feb 2020 and 14 Mar 2021", + languages=["en"], + expected=None, + ), + # "3 days ago" is written as three words and translated into three, + # which the two "tomorrow" adds back: the counts line up although + # the expression was not translated from "3" alone + param( + text="3 days ago tomorrow 13 Feb 2020", + languages=["en"], + expected=None, + ), + # Cutting "today" off would leave "13 feb 2020 before", which is no + # longer a date, so the expression was not written next to one + param( + text="13 feb 2020 before today", + languages=["en"], + expected=None, + ), + ] + ) + def test_search_dates_with_a_relative_expression_next_to_a_date( + self, text, languages, expected + ): + result = search_dates( + text, languages=languages, settings={"RELATIVE_BASE": relative_base} + ) + self.assertEqual(result, expected) + @parameterized.expand( [ param(