Accept single-edit typos in correction gates

difflib's SequenceMatcher underrates adjacent transpositions, so the
flagship examples (gti->git 0.667, greo->grep 0.75, psuh->push 0.75)
fell below the 0.8 ratio gates and were declined. Per the plan
amendment, the token and candidate gates in the history resolver,
guess_from_path and the help resolver additionally accept a
first-char-equal single edit (one substitution, insertion, deletion
or adjacent transposition; new thefuck/typo.py), with the ratio
cutoffs, the exact-one-candidate rule and the 0.79 boundary (edit
distance >= 2) unchanged.
This commit is contained in:
Alexander
2026-09-13 21:00:20 +02:00
parent 572d301bfe
commit 8d2f2fa11c
8 changed files with 218 additions and 26 deletions
+17 -1
View File
@@ -10,7 +10,7 @@ import time
import pytest
from thefuck import shell_ast
from thefuck import shell_ast, typo
from thefuck.resolvers.help_resolver import get_help_correction
@@ -95,6 +95,22 @@ def test_returns_correction_when_subcommand_misspelled(
assert _spawned(argv_log) == ['--help']
def test_corrects_transposed_subcommand(fake_bin, argv_log, enable_cache):
# bulid -> build scores 2 * 4 / 10 = 0.8, right at the cutoff,
# and is also a single transposition (li <-> il).
fake_bin('fake')
assert get_help_correction('fake bulid .') == 'fake build .'
assert typo.single_edit('bulid', 'build')
def test_corrects_transposition_under_cutoff(
fake_bin, argv_log, enable_cache):
# psuh -> push scores 2 * 3 / 8 = 0.75, under the cutoff, so only
# the first-char-equal single-edit match fixes it.
fake_bin('fake', commands=('push', 'build', 'clean'))
assert get_help_correction('fake psuh x') == 'fake push x'
def test_respawns_when_binary_mtime_changes(
fake_bin, argv_log, enable_cache):
path = fake_bin('fake')
+19 -11
View File
@@ -2,7 +2,7 @@ import difflib
import pytest
from thefuck import shell_ast
from thefuck import shell_ast, typo
from thefuck.resolvers.history_resolver import (
_CANDIDATES,
_TOKEN_CUTOFF,
@@ -42,20 +42,26 @@ def test_corrects_two_diverged_tokens_across_segments(history):
assert get_history_correction(command) == 'git status | grep -i foo'
def test_declines_tokens_below_cutoff(history):
# The plan's flagship pair lands under the mandated token cutoff:
# SequenceMatcher(None, 'psuh', 'push').ratio() == 0.75 and
# SequenceMatcher(None, 'greo', 'grep').ratio() == 0.75, both
# below _TOKEN_CUTOFF, so the safety gate must decline them.
@requires_ast
def test_corrects_transposed_tokens_under_ratio_cutoff(history):
# psuh -> push and greo -> grep both score 2 * 3 / 8 = 0.75,
# under _TOKEN_CUTOFF, but each is a first-char-equal single
# transposition, which the token gate accepts alongside the
# ratio.
history(['git push | grep -i foo'])
command = Command('git psuh | greo -i foo', '')
assert get_history_correction(command) is None
assert get_history_correction(command) == 'git push | grep -i foo'
def test_declines_token_just_below_cutoff(history):
# '--ignore-whitespaces' vs '--ignore-all-spaces' scores
# 2 * 15 / 39 = 0.769 (under _TOKEN_CUTOFF) AND spans two edits
# (pinned below), so even the amended single-edit gate declines.
history(['git diff --ignore-all-spaces HEAD'])
command = Command('git diff --ignore-whitespaces HEAD', '')
assert get_history_correction(command) is None
assert not typo.single_edit('--ignore-whitespaces',
'--ignore-all-spaces')
def test_token_cutoff_boundary_arithmetic():
@@ -115,13 +121,15 @@ def test_declines_when_history_is_only_unparseable(history):
assert get_history_correction(command) is None
def test_flat_mode_declines_multi_segment_scripts(history, monkeypatch):
# With the parser unavailable the pipeline cannot be verified
# structurally and both diverged tokens fall under the cutoff.
def test_flat_mode_corrects_single_edit_tokens(history, monkeypatch):
# With the parser unavailable both scripts take the flat view;
# token counts still line up and both diverged tokens are
# first-char-equal single edits, so the amended gate corrects
# them there too.
monkeypatch.setattr(shell_ast, 'AST_AVAILABLE', False)
history(['git push | grep -i foo'])
command = Command('git psuh | greo -i foo', '')
assert get_history_correction(command) is None
assert get_history_correction(command) == 'git push | grep -i foo'
def test_declines_history_below_prefilter_cutoff(history):
+16
View File
@@ -180,6 +180,22 @@ class TestGuessFromPathSegments(object):
assert (learned.guess_from_path('gi psuh | gre -i foo')
== 'git psuh | grep -i foo')
def test_fixes_single_edit_typos_under_cutoff(self, learned,
path_bins):
# gti -> git scores 2 * 2 / 6 = 0.667 and greo -> grep scores
# 2 * 3 / 8 = 0.75, both under GUESS_CUTOFF; each is still a
# first-char-equal single edit with one executable match.
path_bins(executables=['git', 'grep', 'sed'])
assert (learned.guess_from_path('gti psuh | greo -i foo')
== 'git psuh | grep -i foo')
def test_declines_single_edit_ambiguity(self, learned, path_bins):
# gti is one edit from both git (transposition) and gui
# (substitution), each scoring 2 * 2 / 6 = 0.667, so two
# candidates survive and the segment is skipped.
path_bins(executables=['git', 'gui'])
assert learned.guess_from_path('gti psuh') is None
def test_fixes_only_segment_with_unknown_head(self, learned, path_bins):
path_bins(executables=['git', 'grep', 'sed'], existing=['git'])
assert (learned.guess_from_path('git psuh | gre -i foo')
+84
View File
@@ -0,0 +1,84 @@
from thefuck.typo import single_edit
class TestSubstitution(object):
def test_single_substitution(self):
assert single_edit('greo', 'grep')
def test_single_substitution_one_char_words(self):
assert single_edit('a', 'b')
class TestTransposition(object):
def test_adjacent_transposition_gti(self):
assert single_edit('gti', 'git')
def test_adjacent_transposition_psuh(self):
assert single_edit('psuh', 'push')
def test_adjacent_transposition_bulid(self):
assert single_edit('bulid', 'build')
def test_transposition_is_symmetric(self):
assert single_edit('git', 'gti')
def test_non_adjacent_swap_is_two_edits(self):
# Swapping two NON-adjacent characters takes two moves:
# abcd -> badc diverges at every position.
assert not single_edit('abcd', 'badc')
def test_two_differing_positions_not_crossed(self):
# Two substitutions shaped like a swap but with distinct
# characters are two edits, not one transposition.
assert not single_edit('ab', 'cd')
class TestInsertionAndDeletion(object):
def test_single_insertion(self):
assert single_edit('clea', 'clear')
def test_single_deletion(self):
assert single_edit('clear', 'clea')
def test_insertion_in_the_middle(self):
assert single_edit('gt', 'git')
def test_two_insertions(self):
assert not single_edit('cl', 'clear')
class TestBoundaries(object):
def test_equal_strings(self):
assert not single_edit('git', 'git')
def test_empty_strings(self):
assert not single_edit('', '')
def test_empty_to_single_char(self):
assert single_edit('', 'a')
def test_single_char_to_empty(self):
assert single_edit('a', '')
def test_empty_to_two_chars(self):
assert not single_edit('', 'ab')
def test_length_difference_over_one(self):
assert not single_edit('g', 'git')
def test_shifted_words_are_two_edits(self):
# abc -> bcd keeps no common alignment: a deletion plus an
# insertion, i.e. two edits.
assert not single_edit('abc', 'bcd')
def test_two_substitutions(self):
# Adjacent but not crossed: two substitutions, not one
# transposition.
assert not single_edit('abcd', 'abxy')
def test_help_flag_boundary_pair(self):
# The pair the history-resolver boundary test pins: under the
# 0.8 ratio cutoff (0.769) AND not a single edit, so the
# amended gate still declines it.
assert not single_edit('--ignore-whitespaces',
'--ignore-all-spaces')
+12 -5
View File
@@ -4,7 +4,7 @@ import shelve
import time
from difflib import get_close_matches
from . import logs, shell_ast
from . import logs, shell_ast, typo
from .utils import get_all_executables, which
try:
@@ -164,12 +164,19 @@ class LearnedCorrections(object):
token = token[1:-1]
if not token or '/' in token or '.' in token or which(token):
continue
candidates = [cmd for cmd in get_close_matches(
token, get_all_executables(), n=5, cutoff=GUESS_CUTOFF)
if cmd.startswith(token[0])]
executables = get_all_executables()
# The ratio cutoff underrates transpositions (gti -> git
# at 0.667), so same-first-char single edits join the
# difflib candidates; exactly one distinct survivor wins.
candidates = set(cmd for cmd in get_close_matches(
token, executables, n=5, cutoff=GUESS_CUTOFF)
if cmd.startswith(token[0]))
candidates.update(cmd for cmd in executables
if cmd[:1] == token[:1]
and typo.single_edit(token, cmd))
if len(candidates) != 1:
continue
replacements.append((start, end, candidates[0]))
replacements.append((start, end, candidates.pop()))
if not replacements:
return None
return shell_ast.splice(script, replacements)
+10 -4
View File
@@ -18,7 +18,7 @@ import os
import subprocess
from difflib import get_close_matches
from thefuck import shell_ast
from thefuck import shell_ast, typo
from thefuck.utils import cache, which
@@ -72,12 +72,18 @@ def _token_replacement(word, binary_name, binary_path):
token, start, end = word
if token in commands:
return None
matches = [match for match in get_close_matches(
# Same gate shape as learned.guess_from_path: difflib candidates
# plus same-first-char single edits (transpositions score under
# the cutoff), exactly one distinct survivor.
matches = set(match for match in get_close_matches(
token, commands, n=_MATCHES, cutoff=_CUTOFF)
if match.startswith(token[0])]
if match.startswith(token[0]))
matches.update(command for command in commands
if command[:1] == token[:1]
and typo.single_edit(token, command))
if len(matches) != 1:
return None
return start, end, matches[0]
return start, end, matches.pop()
def _get_commands(binary_path, binary_name):
+19 -5
View File
@@ -7,7 +7,7 @@ existing `history` rule keeps offering choices instead of auto-running.
"""
import difflib
from thefuck import shell_ast
from thefuck import shell_ast, typo
from thefuck.utils import get_valid_history_without_current
_PREFILTER_CUTOFF = 0.5
@@ -57,12 +57,27 @@ def _prefilter(script, history):
return [line for _, line in scored[:_CANDIDATES]]
def _similar(token, candidate_token):
"""Returns True when a diverged token pair passes the similarity gate.
The ratio cutoff accepts substitutions; `single_edit` adds the
adjacent transpositions the ratio underrates (psuh -> push at
0.75) while keeping every two-edit pair out. The first-char
equality keeps the single-edit path as narrow as the typo
intent: a slipped key, not a different word.
"""
return (difflib.SequenceMatcher(
None, token, candidate_token).ratio() >= _TOKEN_CUTOFF
or (token[:1] == candidate_token[:1]
and typo.single_edit(token, candidate_token)))
def _correct(script, segments, candidate):
"""Returns the spliced correction when candidate passes every gate.
The gates: identical segment and per-segment token counts, at
most `_MAX_DIVERGED` diverged tokens, every diverged pair at
least `_TOKEN_CUTOFF` similar, and a candidate different from
most `_MAX_DIVERGED` diverged tokens, every diverged pair
similar enough (see `_similar`), and a candidate different from
the script itself.
"""
if candidate == script:
@@ -82,8 +97,7 @@ def _correct(script, segments, candidate):
continue
if len(replacements) >= _MAX_DIVERGED:
return None
if difflib.SequenceMatcher(
None, token, candidate_token).ratio() < _TOKEN_CUTOFF:
if not _similar(token, candidate_token):
return None
replacements.append((start, end, candidate_token))
if not replacements:
+41
View File
@@ -0,0 +1,41 @@
"""Single-edit (Damerau distance 1) typo predicate.
`difflib.SequenceMatcher` rates adjacent transpositions (gti -> git)
well below similarity cutoffs that accept substitutions, so gates
that only use the ratio decline the most common keyboard slips.
Callers accept `single_edit(a, b)` matches as an additional path.
"""
def single_edit(a, b):
"""Returns True when `a` and `b` differ by exactly one edit.
One edit is a substitution, insertion, deletion or adjacent
transposition (restricted Damerau distance of 1); equal strings
are zero edits and return False.
"""
if a == b:
return False
if abs(len(a) - len(b)) > 1:
return False
if len(a) == len(b):
return _equal_length_single_edit(a, b)
if len(a) > len(b):
return _contains_single_insertion(a, b)
return _contains_single_insertion(b, a)
def _equal_length_single_edit(a, b):
diffs = [i for i in range(len(a)) if a[i] != b[i]]
if len(diffs) == 1:
return True
return (len(diffs) == 2 and diffs[1] == diffs[0] + 1
and a[diffs[0]] == b[diffs[1]]
and a[diffs[1]] == b[diffs[0]])
def _contains_single_insertion(longer, shorter):
for i in range(len(shorter)):
if shorter[i] != longer[i]:
return shorter[i:] == longer[i + 1:]
return True