From 5cb74bdba196a7806b86e59903fb2739dd250582 Mon Sep 17 00:00:00 2001 From: dhairya Date: Sat, 3 Oct 2026 02:49:58 +0530 Subject: [PATCH 1/3] Fix consumed character tracking in Jaro-Winkler matching --- strings/jaro_winkler.py | 11 ++++++----- 1 file changed, 6 insertions(+), 5 deletions(-) diff --git a/strings/jaro_winkler.py b/strings/jaro_winkler.py index 0ce5d83b3c41..fdfbfbb60e08 100644 --- a/strings/jaro_winkler.py +++ b/strings/jaro_winkler.py @@ -27,15 +27,16 @@ def jaro_winkler(str1: str, str2: str) -> float: def get_matched_characters(_str1: str, _str2: str) -> str: matched = [] + matched_indices: set[int] = set() limit = min(len(_str1), len(_str2)) // 2 for i, char in enumerate(_str1): left = int(max(0, i - limit)) right = int(min(i + limit + 1, len(_str2))) - if char in _str2[left:right]: - matched.append(char) - _str2 = ( - f"{_str2[0 : _str2.index(char)]} {_str2[_str2.index(char) + 1 :]}" - ) + for index in range(left, right): + if index not in matched_indices and char == _str2[index]: + matched.append(char) + matched_indices.add(index) + break return "".join(matched) From ab709a20f1423847d90dc9f7dc4a6fe5e04894d5 Mon Sep 17 00:00:00 2001 From: dhairya Date: Sat, 3 Oct 2026 09:47:49 +0530 Subject: [PATCH 2/3] Add Jaro-Winkler regression tests and a large-input benchmark Preserve all eight existing examples and append three regressions that fail before the fix. Add an optional 2,000-character timing comparison without changing the production implementation. --- strings/jaro_winkler.py | 30 ++++++++++++++++++++++++++++++ 1 file changed, 30 insertions(+) diff --git a/strings/jaro_winkler.py b/strings/jaro_winkler.py index fdfbfbb60e08..e5cdaa8521c8 100644 --- a/strings/jaro_winkler.py +++ b/strings/jaro_winkler.py @@ -23,6 +23,16 @@ def jaro_winkler(str1: str, str2: str) -> float: 0.4666666666666666 >>> jaro_winkler("hell**o", "*world") 0.4365079365079365 + + Matched positions cannot be reused as literal spaces. + >>> round(jaro_winkler("aa ", "aa"), 6) + 0.911111 + >>> round(jaro_winkler("a a", "aa"), 6) + 0.9 + + Repeated characters must be consumed inside the matching window. + >>> round(jaro_winkler("aabb", "bab"), 6) + 0.722222 """ def get_matched_characters(_str1: str, _str2: str) -> str: @@ -76,6 +86,26 @@ def get_matched_characters(_str1: str, _str2: str) -> str: if __name__ == "__main__": import doctest + import sys + from timeit import repeat doctest.testmod() print(jaro_winkler("hello", "world")) + + # Run with --benchmark on each revision using the same 2,000-character inputs. + if "--benchmark" in sys.argv: + size = 2000 + cases = ( + ("identical", "a" * size, "a" * size), + ("no matches", "a" * size, "b" * size), + ("trailing space", "a" * (size - 1) + " ", "a" * (size - 1)), + ) + for name, first, second in cases: + score = jaro_winkler(first, second) + timings = repeat( + "jaro_winkler(first, second)", repeat=5, number=1, globals=globals() + ) + print( + f"{name}, {len(first)}/{len(second)} characters: " + f"{min(timings):.6f} seconds (best of 5), score={score}" + ) From 9172815668e882109a13e2d7c074e0f790bf7f72 Mon Sep 17 00:00:00 2001 From: dhairya Date: Sat, 3 Oct 2026 12:01:00 +0530 Subject: [PATCH 3/3] Optimize Jaro-Winkler matching with occurrence queues Preserve earliest unused matches while avoiding repeated window scans. Retain all existing regression doctests and the benchmark. --- strings/jaro_winkler.py | 21 +++++++++++++++------ 1 file changed, 15 insertions(+), 6 deletions(-) diff --git a/strings/jaro_winkler.py b/strings/jaro_winkler.py index e5cdaa8521c8..2933d05057f9 100644 --- a/strings/jaro_winkler.py +++ b/strings/jaro_winkler.py @@ -1,5 +1,7 @@ """https://en.wikipedia.org/wiki/Jaro%E2%80%93Winkler_distance""" +from collections import defaultdict, deque + def jaro_winkler(str1: str, str2: str) -> float: """ @@ -37,16 +39,23 @@ def jaro_winkler(str1: str, str2: str) -> float: def get_matched_characters(_str1: str, _str2: str) -> str: matched = [] - matched_indices: set[int] = set() + character_positions: defaultdict[str, deque[int]] = defaultdict(deque) + for index, char in enumerate(_str2): + character_positions[char].append(index) + limit = min(len(_str1), len(_str2)) // 2 for i, char in enumerate(_str1): + positions = character_positions.get(char) + if not positions: + continue left = int(max(0, i - limit)) right = int(min(i + limit + 1, len(_str2))) - for index in range(left, right): - if index not in matched_indices and char == _str2[index]: - matched.append(char) - matched_indices.add(index) - break + # Left edges only advance, so earlier positions cannot match later. + while positions and positions[0] < left: + positions.popleft() + if positions and positions[0] < right: + matched.append(char) + positions.popleft() return "".join(matched)