Add history-similarity resolver with correction-only gates
get_history_correction aligns a failed script with near-identical history commands segment by segment (shell_ast words with byte-true offsets) and splices the corrected tokens back in place. It only returns a fix when exactly one candidate matches the structure with at most two diverged tokens, each at least 80 percent similar; looser or ambiguous matches decline so the asking history rule stays in charge of them. Ultraworked with [Sisyphus](https://github.com/code-yeongyu/oh-my-openagent) Co-authored-by: Sisyphus <clio-agent@sisyphuslabs.ai>
This commit is contained in:
@@ -0,0 +1,145 @@
|
||||
import difflib
|
||||
|
||||
import pytest
|
||||
|
||||
from thefuck import shell_ast
|
||||
from thefuck.resolvers.history_resolver import (
|
||||
_CANDIDATES,
|
||||
_TOKEN_CUTOFF,
|
||||
_prefilter,
|
||||
get_history_correction,
|
||||
)
|
||||
from thefuck.types import Command
|
||||
|
||||
pytestmark = pytest.mark.usefixtures('no_memoize')
|
||||
|
||||
requires_ast = pytest.mark.skipif(not shell_ast.AST_AVAILABLE,
|
||||
reason='bashlex is not available')
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def history(mocker):
|
||||
def _history(lines):
|
||||
return mocker.patch(
|
||||
'thefuck.resolvers.history_resolver.'
|
||||
'get_valid_history_without_current',
|
||||
return_value=lines)
|
||||
return _history
|
||||
|
||||
|
||||
def test_corrects_single_diverged_token(history):
|
||||
history(['docker build -t foo .'])
|
||||
command = Command('docker bilud -t foo .', '')
|
||||
assert get_history_correction(command) == 'docker build -t foo .'
|
||||
|
||||
|
||||
@requires_ast
|
||||
def test_corrects_two_diverged_tokens_across_segments(history):
|
||||
history(['git status | grep -i foo'])
|
||||
command = Command('git statuz | greep -i foo', '')
|
||||
# statuz -> status scores 2 * 6 / 12 = 0.833 and
|
||||
# greep -> grep scores 2 * 4 / 9 = 0.889, both above cutoff.
|
||||
assert get_history_correction(command) == 'git status | grep -i foo'
|
||||
|
||||
|
||||
def test_declines_tokens_below_cutoff(history):
|
||||
# The plan's flagship pair lands under the mandated token cutoff:
|
||||
# SequenceMatcher(None, 'psuh', 'push').ratio() == 0.75 and
|
||||
# SequenceMatcher(None, 'greo', 'grep').ratio() == 0.75, both
|
||||
# below _TOKEN_CUTOFF, so the safety gate must decline them.
|
||||
history(['git push | grep -i foo'])
|
||||
command = Command('git psuh | greo -i foo', '')
|
||||
assert get_history_correction(command) is None
|
||||
|
||||
|
||||
def test_declines_token_just_below_cutoff(history):
|
||||
history(['git diff --ignore-all-spaces HEAD'])
|
||||
command = Command('git diff --ignore-whitespaces HEAD', '')
|
||||
assert get_history_correction(command) is None
|
||||
|
||||
|
||||
def test_token_cutoff_boundary_arithmetic():
|
||||
# 20 + 19 chars with 15 matched: 2 * 15 / 39 = 0.769..., i.e.
|
||||
# above the 0.75 the declined flagship typos score but strictly
|
||||
# under _TOKEN_CUTOFF, so the boundary is provably tight.
|
||||
ratio = difflib.SequenceMatcher(
|
||||
None, '--ignore-whitespaces', '--ignore-all-spaces').ratio()
|
||||
assert 0.75 <= ratio < _TOKEN_CUTOFF
|
||||
|
||||
|
||||
def test_declines_three_diverged_tokens(history):
|
||||
# Every pair alone passes the token cutoff (0.833, 0.909 and
|
||||
# 0.8), so only the max-diverged gate rejects this candidate.
|
||||
history(['docker status branch build'])
|
||||
command = Command('docker statuz brnch bilud', '')
|
||||
assert get_history_correction(command) is None
|
||||
|
||||
|
||||
def test_declines_identical_history_line(history):
|
||||
# get_valid_history_without_current already drops lines equal to
|
||||
# the script; this pins the resolver-side no-op guard for callers
|
||||
# that bypass that filter.
|
||||
history(['docker bilud -t foo .'])
|
||||
command = Command('docker bilud -t foo .', '')
|
||||
assert get_history_correction(command) is None
|
||||
|
||||
|
||||
def test_declines_two_equally_close_candidates(history):
|
||||
# bilud -> build (0.8) and bilud -> bild (0.889) both pass every
|
||||
# gate, so the resolver declines and the existing history rule
|
||||
# keeps offering the choice instead of auto-running one.
|
||||
history(['docker build -t foo .', 'docker bild -t foo .'])
|
||||
command = Command('docker bilud -t foo .', '')
|
||||
assert get_history_correction(command) is None
|
||||
|
||||
|
||||
def test_repeated_history_line_is_one_candidate(history):
|
||||
history(['docker build -t foo .'] * 3)
|
||||
command = Command('docker bilud -t foo .', '')
|
||||
assert get_history_correction(command) == 'docker build -t foo .'
|
||||
|
||||
|
||||
def test_skips_unparseable_history_line(history):
|
||||
# '"foo bar .' never closes its quote, so bashlex refuses the
|
||||
# line and shell_ast falls back to a flat view whose token count
|
||||
# no longer matches; it is skipped without raising and the valid
|
||||
# line still corrects the script.
|
||||
history(['docker build -t "foo bar .', 'docker build -t foo .'])
|
||||
command = Command('docker bilud -t foo .', '')
|
||||
assert get_history_correction(command) == 'docker build -t foo .'
|
||||
|
||||
|
||||
def test_declines_when_history_is_only_unparseable(history):
|
||||
history(['docker build -t "foo bar .'])
|
||||
command = Command('docker bilud -t foo .', '')
|
||||
assert get_history_correction(command) is None
|
||||
|
||||
|
||||
def test_flat_mode_declines_multi_segment_scripts(history, monkeypatch):
|
||||
# With the parser unavailable the pipeline cannot be verified
|
||||
# structurally and both diverged tokens fall under the cutoff.
|
||||
monkeypatch.setattr(shell_ast, 'AST_AVAILABLE', False)
|
||||
history(['git push | grep -i foo'])
|
||||
command = Command('git psuh | greo -i foo', '')
|
||||
assert get_history_correction(command) is None
|
||||
|
||||
|
||||
def test_declines_history_below_prefilter_cutoff(history):
|
||||
history(['totally unrelated command here'])
|
||||
command = Command('docker bilud -t foo .', '')
|
||||
assert get_history_correction(command) is None
|
||||
|
||||
|
||||
def test_returns_none_with_empty_history(history):
|
||||
history([])
|
||||
command = Command('docker bilud -t foo .', '')
|
||||
assert get_history_correction(command) is None
|
||||
|
||||
|
||||
def test_prefilter_caps_and_orders_candidates():
|
||||
script = 'docker build -t foo . 3'
|
||||
lines = ['docker build -t foo . {}'.format(index)
|
||||
for index in range(_CANDIDATES + 5)]
|
||||
selected = _prefilter(script, lines)
|
||||
assert len(selected) == _CANDIDATES
|
||||
assert selected[0] == 'docker build -t foo . 3'
|
||||
@@ -0,0 +1,93 @@
|
||||
"""History-similarity resolver with correction-only auto-run gates.
|
||||
|
||||
Corrects a failed script from a structurally-aligned history command
|
||||
when exactly one candidate differs in at most a couple of highly
|
||||
similar tokens; anything looser or ambiguous is declined so the
|
||||
existing `history` rule keeps offering choices instead of auto-running.
|
||||
"""
|
||||
import difflib
|
||||
|
||||
from thefuck import shell_ast
|
||||
from thefuck.utils import get_valid_history_without_current
|
||||
|
||||
_PREFILTER_CUTOFF = 0.5
|
||||
_CANDIDATES = 10
|
||||
_MAX_DIVERGED = 2
|
||||
_TOKEN_CUTOFF = 0.8
|
||||
|
||||
|
||||
def get_history_correction(command):
|
||||
"""Returns the corrected script for `command`, or None.
|
||||
|
||||
:type command: thefuck.types.Command
|
||||
:rtype: str | None
|
||||
|
||||
"""
|
||||
candidates = _prefilter(command.script,
|
||||
get_valid_history_without_current(command))
|
||||
if not candidates:
|
||||
return None
|
||||
segments = shell_ast.parse(command.script)
|
||||
matches = []
|
||||
for candidate in candidates:
|
||||
correction = _correct(command.script, segments, candidate)
|
||||
if correction is not None:
|
||||
matches.append(correction)
|
||||
if len(matches) > 1:
|
||||
return None
|
||||
return matches[0] if len(matches) == 1 else None
|
||||
|
||||
|
||||
def _prefilter(script, history):
|
||||
"""Returns up to `_CANDIDATES` distinct closest history lines.
|
||||
|
||||
Duplicate lines collapse into one candidate: repeats of the same
|
||||
command in history are one option, not ambiguity.
|
||||
"""
|
||||
scored = []
|
||||
seen = set()
|
||||
for line in history:
|
||||
if line in seen:
|
||||
continue
|
||||
seen.add(line)
|
||||
ratio = difflib.SequenceMatcher(None, script, line).ratio()
|
||||
if ratio >= _PREFILTER_CUTOFF:
|
||||
scored.append((ratio, line))
|
||||
scored.sort(key=lambda scored_line: scored_line[0], reverse=True)
|
||||
return [line for _, line in scored[:_CANDIDATES]]
|
||||
|
||||
|
||||
def _correct(script, segments, candidate):
|
||||
"""Returns the spliced correction when candidate passes every gate.
|
||||
|
||||
The gates: identical segment and per-segment token counts, at
|
||||
most `_MAX_DIVERGED` diverged tokens, every diverged pair at
|
||||
least `_TOKEN_CUTOFF` similar, and a candidate different from
|
||||
the script itself.
|
||||
"""
|
||||
if candidate == script:
|
||||
return None
|
||||
candidate_segments = shell_ast.parse(candidate)
|
||||
if len(candidate_segments) != len(segments):
|
||||
return None
|
||||
replacements = []
|
||||
for segment, candidate_segment in zip(segments, candidate_segments):
|
||||
if len(segment.words) != len(candidate_segment.words):
|
||||
return None
|
||||
for word, candidate_word in zip(segment.words,
|
||||
candidate_segment.words):
|
||||
token, start, end = word
|
||||
candidate_token = candidate_word[0]
|
||||
if token == candidate_token:
|
||||
continue
|
||||
if len(replacements) >= _MAX_DIVERGED:
|
||||
return None
|
||||
if difflib.SequenceMatcher(
|
||||
None, token, candidate_token).ratio() < _TOKEN_CUTOFF:
|
||||
return None
|
||||
replacements.append((start, end, candidate_token))
|
||||
if not replacements:
|
||||
# A whitespace-only difference yields no replacements and a
|
||||
# correction equal to the script would be a no-op run.
|
||||
return None
|
||||
return shell_ast.splice(script, replacements)
|
||||
Reference in New Issue
Block a user