test(train): strip ANSI + widen terminal in reward-hack help assertions (v0.71.26)

The two TestRewardHackMitigationCli help-text assertions grepped the raw
`soup train --help` output for the new flags. CI runners emit color codes
that split long option names into non-contiguous characters (same failure
mode documented in test_eval_gate.py for --gate), so the raw substring check
failed on every OS/Python combo while passing locally on a no-color terminal.

Fix follows the established repo pattern: render with COLUMNS=200 (no option
wrapping) and strip ANSI escapes before the membership check. Verified under
FORCE_COLOR=1: both assertions pass. Test-only change, no version bump.
This commit is contained in:
Alpamys 2026-07-01 18:16:53 +05:00
parent 2ffc3743ae
commit 0db7404a6a
1 changed files with 15 additions and 7 deletions

View File

@ -1017,10 +1017,15 @@ class TestRewardHackMitigationCli:
return CliRunner(), app
def test_help_shows_flag(self):
import re
runner, app = self._runner()
result = runner.invoke(app, ["train", "--help"])
assert result.exit_code == 0, result.output
assert "--reward-hack-mitigation" in result.output
# Wide terminal avoids option-name line-wrapping; ANSI strip handles the
# color codes CI runners inject mid-token (see test_eval_gate.py).
result = runner.invoke(app, ["train", "--help"], env={"COLUMNS": "200"})
assert result.exit_code == 0, (result.output, repr(result.exception))
cleaned = re.sub(r"\x1b\[[0-9;]*m", "", result.output)
assert "--reward-hack-mitigation" in cleaned
def test_override_without_detector_rejected(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
@ -1048,11 +1053,14 @@ class TestRewardHackMitigationCli:
assert "cfg.training.reward_hack_mitigation" in src
def test_help_shows_detector_and_halt_flags(self):
import re
runner, app = self._runner()
result = runner.invoke(app, ["train", "--help"])
assert result.exit_code == 0, result.output
assert "--reward-hack-detector" in result.output
assert "--reward-hack-halt" in result.output
result = runner.invoke(app, ["train", "--help"], env={"COLUMNS": "200"})
assert result.exit_code == 0, (result.output, repr(result.exception))
cleaned = re.sub(r"\x1b\[[0-9;]*m", "", result.output)
assert "--reward-hack-detector" in cleaned
assert "--reward-hack-halt" in cleaned
def test_bad_detector_value_rejected(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)