soup/tests/test_v07127.py

2191 lines
87 KiB
Python

"""v0.71.27 "Fine-tune Doctor" — chat-template doctor + loss-mask X-ray +
preference linter (closes the release's three headline items, plus the
diagnose.py O_NOFOLLOW evidence-loader housekeeping rider).
- ``soup data doctor`` — template-compat report (EOS-missing / BOS
duplication / no-system-role / unknown-role / truncation risk) +
``--show-mask N`` loss-mask X-ray through the REAL collator path.
- ``soup data lint`` — preference-data linter (dpo/kto): length bias
(Cohen's d effect size), label imbalance, near-dup pairs (MinHash),
chosen==rejected, prompt-leaked-into-completion.
- ``commands/diagnose.py::_load_evidence`` — O_NOFOLLOW + fstat hardening
(backport of the v0.71.25 ``soup ship`` pattern).
"""
from __future__ import annotations
import json
import os
from pathlib import Path
import pytest
from typer.testing import CliRunner
runner = CliRunner()
_PROJECT_ROOT = Path(__file__).resolve().parents[1]
try:
import datasketch # noqa: F401
_HAS_DATASKETCH = True
except ImportError:
_HAS_DATASKETCH = False
# ---------------------------------------------------------------------------
# Fake tokenizer — char-level, extends the v0.36.0 test_assistant_mask.py
# `_FakeTokenizer` pattern with controllable EOS / BOS / system-role knobs.
# ---------------------------------------------------------------------------
_BOS_ID = -1001
_EOS_ID = -1002
class _FakeTokenizer:
"""Deterministic char-level fake tokenizer for doctor/lint unit tests.
Renders ``<role>:content\\n`` per turn; each char maps to
``(ord(c) % 200) + 1`` so ids never collide with the negative BOS/EOS
sentinels. Knobs simulate the real-world footguns the doctor detects:
``emit_eos=False`` (chat template never appends EOS after assistant
turns), ``double_bos=True`` (template + tokenizer both prepend BOS),
``reject_system=True`` (Mistral-style templates without a system role).
"""
bos_token_id = _BOS_ID
pad_token_id = 0
def __init__(
self,
*,
supports_assistant_mask: bool = True,
has_chat_template: bool = True,
emit_eos: bool = True,
double_bos: bool = False,
reject_system: bool = False,
fail_render: bool = False,
eos_token_id=_EOS_ID,
trailing_trained: bool = False,
):
self.supports_assistant_mask = supports_assistant_mask
self.chat_template = "fake-template" if has_chat_template else None
self.emit_eos = emit_eos
self.double_bos = double_bos
self.reject_system = reject_system
self.fail_render = fail_render
self.eos_token_id = eos_token_id
# v0.71.27 real-tokenizer smoke finding: SmolLM2's ChatML template
# renders `<|im_end|>\n` for the LAST assistant turn with the
# trailing `\n` landing INSIDE the trained span too (no later
# message to fold it into a masked prefix on the fallback delta
# path) — EOS is present but is not the literal last trained token.
self.trailing_trained = trailing_trained
def _render(self, messages):
if self.fail_render:
raise ValueError("fake render failure")
if self.reject_system and any(m.get("role") == "system" for m in messages):
raise ValueError("System role not supported")
ids: list[int] = []
mask: list[int] = []
for _ in range(2 if self.double_bos else 1):
ids.append(self.bos_token_id)
mask.append(0)
for msg in messages:
role = msg.get("role", "")
content = msg.get("content", "")
is_assistant = role == "assistant"
for c in f"<{role}>:":
ids.append((ord(c) % 200) + 1)
mask.append(0)
for c in str(content):
ids.append((ord(c) % 200) + 1)
mask.append(1 if is_assistant else 0)
if is_assistant and self.emit_eos and self.eos_token_id is not None:
# A real tokenizer's chat template emits exactly ONE
# concrete eos id even when eos_token_id is a multi-entry
# list/tuple (e.g. Llama-3's [128001, 128009]) — pick the
# first entry so this fake's rendered ids stay flat ints.
eos_value = self.eos_token_id
if isinstance(eos_value, (list, tuple)):
eos_value = eos_value[0]
ids.append(eos_value)
mask.append(1)
ids.append((ord("\n") % 200) + 1)
mask.append(1 if (is_assistant and self.trailing_trained) else 0)
return ids, mask
def apply_chat_template(
self,
messages,
tokenize=False,
add_generation_prompt=False,
return_assistant_tokens_mask=False,
return_dict=False,
**kwargs,
):
ids, mask = self._render(messages)
if not tokenize:
return "".join(f"[{i}]" for i in ids)
if return_assistant_tokens_mask and return_dict:
if not self.supports_assistant_mask:
raise TypeError("return_assistant_tokens_mask unsupported")
return {"input_ids": ids, "assistant_masks": mask}
if return_dict:
return {"input_ids": ids}
return ids
def __call__(self, text, add_special_tokens=False, return_offsets_mapping=False, **kwargs):
ids = [(ord(c) % 200) + 1 for c in str(text)]
out = {"input_ids": ids}
if return_offsets_mapping:
out["offset_mapping"] = [(i, i + 1) for i in range(len(ids))]
return out
def encode(self, text, add_special_tokens=False):
return [(ord(c) % 200) + 1 for c in str(text)]
def decode(self, ids):
if isinstance(ids, int):
ids = [ids]
return "".join(f"[{i}]" for i in ids)
def convert_ids_to_tokens(self, ids):
return [f"t{i}" for i in ids]
class _JinjaLikeError(RuntimeError):
"""Stand-in for ``jinja2.exceptions.TemplateError`` (neither ValueError
nor TypeError — a real Mistral-style template's ``raise_exception()``
raises this shape, not one of the two the checks used to catch)."""
class _OddRaisingTokenizer(_FakeTokenizer):
"""Raises a non-ValueError/TypeError from ``apply_chat_template`` for
any row containing a system message — mirrors the real
jinja2.exceptions.TemplateError a Mistral-style no-system-role template
raises (v0.71.27 real-SmolLM2 smoke finding: this must be a per-row
skip, not an unhandled crash of the whole report)."""
def apply_chat_template(self, messages, **kwargs):
if any(m.get("role") == "system" for m in messages):
raise _JinjaLikeError("System role not supported")
return super().apply_chat_template(messages, **kwargs)
class _TypeErrorOnKwargTokenizer(_FakeTokenizer):
"""``.encode`` has no ``add_special_tokens`` parameter at all (an older
tokenizer) — calling it WITH that kwarg raises a genuine Python
TypeError, calling it WITHOUT (positional-only) succeeds. Exercises the
`soup data lint --model` length_fn's inner TypeError fallback."""
def encode(self, text): # noqa: D102 — deliberately no add_special_tokens param
return [(ord(c) % 200) + 1 for c in str(text)]
class _AlwaysFailEncodeTokenizer(_FakeTokenizer):
"""``.encode(...)`` always raises, regardless of args — exercises the
length_fn word-count degrade-gracefully fallback."""
def encode(self, text, add_special_tokens=False):
raise RuntimeError("encode is broken")
class _PerRowEosTokenizer(_FakeTokenizer):
"""EOS emission varies PER ROW (via a "NOEOS" marker in the assistant
content) rather than being all-or-nothing for the whole tokenizer
instance — needed to construct an EXACT boundary fraction (e.g. 2/4 =
50%) for a threshold check's cutoff, not just "all rows" or "no rows".
"""
def _render(self, messages):
skip = any(m.get("content") == "NOEOS" for m in messages if m.get("role") == "assistant")
original = self.emit_eos
self.emit_eos = not skip
try:
return super()._render(messages)
finally:
self.emit_eos = original
def _chat_row(*turns):
return {"messages": [{"role": r, "content": c} for r, c in turns]}
def _write_jsonl(path: Path, rows: list[dict]) -> None:
with open(path, "w", encoding="utf-8") as fh:
for row in rows:
fh.write(json.dumps(row) + "\n")
# ---------------------------------------------------------------------------
# data_doctor.py — dataclasses + taxonomy
# ---------------------------------------------------------------------------
class TestDoctorCheck:
def test_valid_check(self):
from soup_cli.utils.data_doctor import DoctorCheck
c = DoctorCheck(name="chat_template", verdict="OK", message="fine", evidence="")
assert c.verdict == "OK"
def test_rejects_unknown_name(self):
from soup_cli.utils.data_doctor import DoctorCheck
with pytest.raises(ValueError, match="unknown"):
DoctorCheck(name="not_a_real_check", verdict="OK", message="x")
def test_rejects_unknown_verdict(self):
from soup_cli.utils.data_doctor import DoctorCheck
with pytest.raises(ValueError, match="verdict"):
DoctorCheck(name="chat_template", verdict="BAD", message="x")
def test_rejects_null_byte(self):
from soup_cli.utils.data_doctor import DoctorCheck
with pytest.raises(ValueError, match="null byte"):
DoctorCheck(name="chat_template", verdict="OK", message="a\x00b")
def test_rejects_oversize_message(self):
from soup_cli.utils.data_doctor import DoctorCheck
with pytest.raises(ValueError, match="too long"):
DoctorCheck(name="chat_template", verdict="OK", message="x" * 3000)
def test_is_frozen(self):
from soup_cli.utils.data_doctor import DoctorCheck
c = DoctorCheck(name="chat_template", verdict="OK", message="x")
with pytest.raises(Exception): # noqa: PT011 — dataclasses.FrozenInstanceError
c.verdict = "MAJOR"
class TestOverallVerdict:
def test_empty_is_ok(self):
from soup_cli.utils.data_doctor import overall_verdict
assert overall_verdict([]) == "OK"
def test_worst_wins(self):
from soup_cli.utils.data_doctor import DoctorCheck, overall_verdict
checks = [
DoctorCheck(name="chat_template", verdict="OK", message="x"),
DoctorCheck(name="bos_duplication", verdict="MINOR", message="x"),
DoctorCheck(name="eos_in_labels", verdict="MAJOR", message="x"),
]
assert overall_verdict(checks) == "MAJOR"
def test_rejects_non_check_entries(self):
from soup_cli.utils.data_doctor import overall_verdict
with pytest.raises(TypeError):
overall_verdict(["not a check"])
class TestDoctorReport:
def test_to_dict_roundtrip_shape(self):
from soup_cli.utils.data_doctor import DoctorCheck, compose_doctor_report
checks = [DoctorCheck(name="chat_template", verdict="OK", message="fine", evidence="e")]
report = compose_doctor_report(checks, rows_scanned=5, total_rows=10)
d = report.to_dict()
assert d["overall"] == "OK"
assert d["rows_scanned"] == 5
assert d["total_rows"] == 10
assert d["checks"][0]["name"] == "chat_template"
def test_rows_scanned_cannot_exceed_total(self):
from soup_cli.utils.data_doctor import DoctorReport
with pytest.raises(ValueError, match="cannot exceed"):
DoctorReport(checks=(), overall="OK", rows_scanned=10, total_rows=5)
def test_negative_counts_rejected(self):
from soup_cli.utils.data_doctor import DoctorReport
with pytest.raises(ValueError):
DoctorReport(checks=(), overall="OK", rows_scanned=-1, total_rows=5)
class TestSampleIndices:
def test_empty(self):
from soup_cli.utils.data_doctor import sample_indices
assert sample_indices(0, 10) == []
def test_n_greater_than_total_returns_all(self):
from soup_cli.utils.data_doctor import sample_indices
assert sample_indices(3, 100) == [0, 1, 2]
def test_spans_full_range(self):
from soup_cli.utils.data_doctor import sample_indices
idxs = sample_indices(1000, 10)
assert len(idxs) <= 10
assert idxs == sorted(set(idxs))
assert idxs[0] < 100 # not just the tail
assert idxs[-1] > 800 # not just the head — evenly spaced
def test_exact_indices_evenly_divisible(self):
"""Regression (tdd-review): sample_indices is a pure, deterministic
function that gates row sampling for all 13 checks — property-only
assertions (count/sorted/range) would still pass under a changed
step formula (e.g. round() instead of int(), or // instead of /)
that silently samples different rows. Assert the exact list."""
from soup_cli.utils.data_doctor import sample_indices
assert sample_indices(1000, 10) == [0, 100, 200, 300, 400, 500, 600, 700, 800, 900]
assert sample_indices(10, 3) == [0, 3, 6]
def test_exact_indices_non_round_ratio(self):
"""int()-truncation on a non-round total/n ratio can skip an index
(7/5=1.4 -> 0,1,2,4,5 — index 3 is never produced); this is an
accepted property of the even-spacing approach, not a bug, but it
must not silently change."""
from soup_cli.utils.data_doctor import sample_indices
assert sample_indices(7, 5) == [0, 1, 2, 4, 5]
# ---------------------------------------------------------------------------
# resolve_tokenizer
# ---------------------------------------------------------------------------
class TestResolveTokenizer:
def test_passthrough_object(self):
from soup_cli.utils.data_doctor import resolve_tokenizer
tok = _FakeTokenizer()
assert resolve_tokenizer(tok) is tok
def test_rejects_non_string_non_tokenizer(self):
from soup_cli.utils.data_doctor import resolve_tokenizer
with pytest.raises(TypeError):
resolve_tokenizer(12345)
def test_rejects_empty_string(self):
from soup_cli.utils.data_doctor import resolve_tokenizer
with pytest.raises(ValueError, match="non-empty"):
resolve_tokenizer("")
def test_delegates_to_real_auto_tokenizer_from_pretrained(self, monkeypatch):
"""Regression (tdd-review): every other test either passes a
duck-typed fake (hits the early passthrough) or monkeypatches
resolve_tokenizer away entirely — the actual
transformers.AutoTokenizer.from_pretrained delegation was never
exercised, string id in, trust_remote_code kwarg forwarded, and all.
"""
import transformers
captured = {}
def fake_from_pretrained(model_id, trust_remote_code=False):
captured["model_id"] = model_id
captured["trust_remote_code"] = trust_remote_code
return _FakeTokenizer()
monkeypatch.setattr(
transformers.AutoTokenizer, "from_pretrained", fake_from_pretrained
)
from soup_cli.utils.data_doctor import resolve_tokenizer
result = resolve_tokenizer("some-org/some-model", trust_remote_code=True)
assert isinstance(result, _FakeTokenizer)
assert captured == {"model_id": "some-org/some-model", "trust_remote_code": True}
def test_wraps_a_real_load_failure_with_model_id_and_exception_type(self, monkeypatch):
import transformers
def fake_from_pretrained(model_id, trust_remote_code=False):
raise OSError("model not found on the Hub")
monkeypatch.setattr(
transformers.AutoTokenizer, "from_pretrained", fake_from_pretrained
)
from soup_cli.utils.data_doctor import resolve_tokenizer
with pytest.raises(ValueError, match="nonexistent/model") as excinfo:
resolve_tokenizer("nonexistent/model")
assert "OSError" in str(excinfo.value)
def test_missing_transformers_dependency_is_a_friendly_error(self, monkeypatch):
"""Mirrors TestCheckNearDuplicatesNoDep's ImportError-simulation
pattern for the datasketch optional dep."""
import builtins
real_import = builtins.__import__
def fake_import(name, *args, **kwargs):
if name == "transformers":
raise ImportError("no transformers")
return real_import(name, *args, **kwargs)
monkeypatch.setattr(builtins, "__import__", fake_import)
from soup_cli.utils.data_doctor import resolve_tokenizer
with pytest.raises(ValueError, match="transformers"):
resolve_tokenizer("some-org/some-model")
# ---------------------------------------------------------------------------
# Individual checks
# ---------------------------------------------------------------------------
class TestCheckChatTemplate:
def test_present(self):
from soup_cli.utils.data_doctor import check_chat_template
c = check_chat_template(_FakeTokenizer(has_chat_template=True))
assert c.verdict == "OK"
def test_missing_is_major(self):
from soup_cli.utils.data_doctor import check_chat_template
c = check_chat_template(_FakeTokenizer(has_chat_template=False))
assert c.verdict == "MAJOR"
class TestCheckTemplateRender:
def test_all_render_ok(self):
from soup_cli.utils.data_doctor import check_template_render
rows = [_chat_row(("user", "hi"), ("assistant", "hello"))]
c = check_template_render(_FakeTokenizer(), rows)
assert c.verdict == "OK"
def test_render_failures_flagged(self):
from soup_cli.utils.data_doctor import check_template_render
rows = [_chat_row(("system", "s"), ("user", "hi"), ("assistant", "hello"))] * 5
c = check_template_render(_FakeTokenizer(reject_system=True), rows)
assert c.verdict == "MAJOR" # 5/5 fail -> frac=1.0 >= the 10% MAJOR cutoff
assert "5/5" in c.message
def test_no_template_skips(self):
from soup_cli.utils.data_doctor import check_template_render
tok = _FakeTokenizer(has_chat_template=False)
c = check_template_render(tok, [_chat_row(("user", "hi"))])
assert c.verdict == "OK"
assert "skipped" in c.message
class TestCheckGenerationMarkers:
def test_supported(self):
from soup_cli.utils.data_doctor import check_generation_markers
c = check_generation_markers(_FakeTokenizer(supports_assistant_mask=True))
assert c.verdict == "OK"
def test_unsupported_is_minor(self):
from soup_cli.utils.data_doctor import check_generation_markers
c = check_generation_markers(_FakeTokenizer(supports_assistant_mask=False))
assert c.verdict == "MINOR"
assert "generation" in c.message.lower() or "heuristic" in c.message.lower()
class TestEosTokenIds:
"""Regression (tdd-review): `_eos_token_ids`'s list/tuple normalisation
exists specifically for real multi-EOS models (Llama-3's
eos_token_id=[128001, 128009]) but was never exercised — every test
tokenizer used a bare int or None."""
def test_list_eos_token_id_normalised(self):
from soup_cli.utils.data_doctor import _eos_token_ids
tok = _FakeTokenizer(eos_token_id=[128001, 128009])
assert _eos_token_ids(tok) == {128001, 128009}
def test_tuple_eos_token_id_normalised(self):
from soup_cli.utils.data_doctor import _eos_token_ids
tok = _FakeTokenizer(eos_token_id=(2, 3))
assert _eos_token_ids(tok) == {2, 3}
def test_list_with_non_int_entries_filtered(self):
from soup_cli.utils.data_doctor import _eos_token_ids
tok = _FakeTokenizer(eos_token_id=[2, "not-an-id", None, True])
assert _eos_token_ids(tok) == {2}
def test_bool_eos_token_id_rejected(self):
from soup_cli.utils.data_doctor import _eos_token_ids
tok = _FakeTokenizer(eos_token_id=True)
assert _eos_token_ids(tok) == set()
def test_multi_eos_list_actually_used_by_the_check(self):
"""Not just _eos_token_ids in isolation — a real multi-EOS model's
list must actually satisfy check_eos_in_labels's span search."""
from soup_cli.utils.data_doctor import check_eos_in_labels
tok = _FakeTokenizer(emit_eos=True, eos_token_id=[999, _EOS_ID])
rows = [_chat_row(("user", "hi"), ("assistant", "hello"))] * 4
c = check_eos_in_labels(tok, rows, max_length=2048)
assert c.verdict == "OK"
class TestCheckEosInLabels:
"""The #1 'model never stops generating' bug."""
def test_eos_present_is_ok(self):
from soup_cli.utils.data_doctor import check_eos_in_labels
rows = [_chat_row(("user", "hi"), ("assistant", "hello"))] * 4
c = check_eos_in_labels(_FakeTokenizer(emit_eos=True), rows, max_length=2048)
assert c.verdict == "OK"
def test_eos_missing_is_flagged(self):
from soup_cli.utils.data_doctor import check_eos_in_labels
rows = [_chat_row(("user", "hi"), ("assistant", "hello"))] * 4
c = check_eos_in_labels(_FakeTokenizer(emit_eos=False), rows, max_length=2048)
assert c.verdict == "MAJOR" # 4/4 missing -> frac=1.0 >= the 50% MAJOR cutoff
assert "4/4" in c.message
assert "stop" in c.message.lower()
def test_mixed_missing_is_proportional(self):
from soup_cli.utils.data_doctor import check_eos_in_labels
bad = _FakeTokenizer(emit_eos=False)
rows = [_chat_row(("user", "hi"), ("assistant", "hello"))] * 10
# Every row is checked against a tokenizer that never emits EOS at
# all, so every row should be flagged (proportional to the 4/4 case
# above at a different N — proves the fraction, not just a fixed count).
c = check_eos_in_labels(bad, rows, max_length=2048)
assert "10/10" in c.message
def test_exactly_at_the_major_boundary(self):
"""Regression (tdd-review): every prior fixture was 0% or 100%
missing — none exercised the documented >=50% MAJOR cutoff at the
exact boundary or just below it. An off-by-one (>= vs >) would not
be caught by any test before this one."""
from soup_cli.utils.data_doctor import check_eos_in_labels
tok = _PerRowEosTokenizer(emit_eos=True)
rows = [
_chat_row(("user", "hi"), ("assistant", "hello")),
_chat_row(("user", "hi"), ("assistant", "NOEOS")),
_chat_row(("user", "hi"), ("assistant", "hello")),
_chat_row(("user", "hi"), ("assistant", "NOEOS")),
]
c = check_eos_in_labels(tok, rows, max_length=2048)
assert "2/4" in c.message
assert c.verdict == "MAJOR" # frac=0.5 >= 0.5
def test_just_below_the_major_boundary(self):
from soup_cli.utils.data_doctor import check_eos_in_labels
tok = _PerRowEosTokenizer(emit_eos=True)
rows = [
_chat_row(("user", "hi"), ("assistant", "hello")),
_chat_row(("user", "hi"), ("assistant", "hello")),
_chat_row(("user", "hi"), ("assistant", "hello")),
_chat_row(("user", "hi"), ("assistant", "NOEOS")),
]
c = check_eos_in_labels(tok, rows, max_length=2048)
assert "1/4" in c.message
assert c.verdict == "MINOR" # frac=0.25 < 0.5, but > 0
def test_no_eos_token_id_is_advisory(self):
from soup_cli.utils.data_doctor import check_eos_in_labels
tok = _FakeTokenizer(eos_token_id=None)
row = _chat_row(("user", "hi"), ("assistant", "hello"))
c = check_eos_in_labels(tok, [row], max_length=2048)
assert c.verdict == "MINOR"
def test_no_template_skips(self):
from soup_cli.utils.data_doctor import check_eos_in_labels
c = check_eos_in_labels(_FakeTokenizer(has_chat_template=False), [], max_length=2048)
assert c.verdict == "OK"
def test_eos_present_but_not_literal_last_token_is_ok(self):
"""Regression (found via real SmolLM2-135M-Instruct smoke test,
v0.71.27): ChatML-style templates render `<|im_end|>\\n` for the
last assistant turn, and on the fallback delta-masking path the
trailing `\\n` has no later message to fold into, so it stays
INSIDE the trained span too — EOS is present but is not the
literal last trained id. Must not be flagged as missing.
"""
from soup_cli.utils.data_doctor import check_eos_in_labels
tok = _FakeTokenizer(emit_eos=True, trailing_trained=True)
rows = [_chat_row(("user", "hi"), ("assistant", "hello"))] * 4
c = check_eos_in_labels(tok, rows, max_length=2048)
assert c.verdict == "OK", c.message
def test_non_valueerror_render_failure_is_skipped_not_raised(self):
"""Regression (v0.71.27 real-SmolLM2 smoke test): a row whose
template rejects it with a non-ValueError/TypeError exception (a
real jinja2.TemplateError shape) must be skipped, not propagate
and crash the whole doctor report.
"""
from soup_cli.utils.data_doctor import check_eos_in_labels
rows = [
_chat_row(("system", "s"), ("user", "hi"), ("assistant", "hello")),
_chat_row(("user", "hi"), ("assistant", "hello")),
]
c = check_eos_in_labels(_OddRaisingTokenizer(), rows, max_length=2048)
assert c.verdict == "OK"
def test_eos_missing_on_an_earlier_turn_is_still_flagged(self):
"""Regression: checking only the LAST trained span would miss a
template bug that drops EOS on an EARLIER assistant turn while the
final turn still closes correctly — every turn boundary matters for
multi-turn inference, not just the last."""
from soup_cli.utils.data_doctor import check_eos_in_labels
class _FirstTurnMissingEos(_FakeTokenizer):
def _render(self, messages):
ids, mask = [], []
ids.append(self.bos_token_id)
mask.append(0)
assistant_seen = 0
for msg in messages:
role = msg.get("role", "")
content = msg.get("content", "")
is_assistant = role == "assistant"
for c in f"<{role}>:":
ids.append((ord(c) % 200) + 1)
mask.append(0)
for c in str(content):
ids.append((ord(c) % 200) + 1)
mask.append(1 if is_assistant else 0)
if is_assistant:
# Only the FIRST assistant turn skips EOS.
if assistant_seen > 0:
ids.append(self.eos_token_id)
mask.append(1)
assistant_seen += 1
ids.append((ord("\n") % 200) + 1)
mask.append(0)
return ids, mask
multi_turn_row = {
"messages": [
{"role": "user", "content": "hi"},
{"role": "assistant", "content": "first reply"},
{"role": "user", "content": "again"},
{"role": "assistant", "content": "second reply"},
]
}
c = check_eos_in_labels(_FirstTurnMissingEos(), [multi_turn_row], max_length=2048)
assert c.verdict == "MAJOR", c.message # 1/1 missing -> frac=1.0 >= the 50% cutoff
assert "1/1" in c.message
def test_truncated_row_excluded_not_flagged_as_missing_eos(self):
"""Regression (code-review MEDIUM): data.loss_mask._truncate keeps
only the FIRST max_length tokens, which can cut off the trailing
EOS a long row would otherwise have. That's a max_length problem
(already surfaced by truncation_risk), not a template bug — a
truncated row must not count against eos_in_labels at all."""
from soup_cli.utils.data_doctor import check_eos_in_labels
tok = _FakeTokenizer(emit_eos=True)
long_row = _chat_row(("user", "hi"), ("assistant", "x" * 200))
# A tiny max_length guarantees truncation lands well before the
# trailing EOS this fake always appends after assistant content.
c = check_eos_in_labels(tok, [long_row], max_length=10)
assert c.verdict == "OK", c.message
assert "no assistant turns to check" in c.message
class TestCheckBosDuplication:
def test_single_bos_ok(self):
from soup_cli.utils.data_doctor import check_bos_duplication
rows = [_chat_row(("user", "hi"), ("assistant", "hello"))] * 4
c = check_bos_duplication(_FakeTokenizer(double_bos=False), rows, max_length=2048)
assert c.verdict == "OK"
def test_double_bos_flagged(self):
from soup_cli.utils.data_doctor import check_bos_duplication
rows = [_chat_row(("user", "hi"), ("assistant", "hello"))] * 4
c = check_bos_duplication(_FakeTokenizer(double_bos=True), rows, max_length=2048)
assert c.verdict == "MAJOR" # 4/4 duplicated -> frac=1.0 >= the 50% MAJOR cutoff
assert "4/4" in c.message
def test_no_bos_token_id_skips(self):
from soup_cli.utils.data_doctor import check_bos_duplication
class _NoBos(_FakeTokenizer):
bos_token_id = None
row = _chat_row(("user", "hi"), ("assistant", "hi"))
c = check_bos_duplication(_NoBos(), [row], max_length=2048)
assert c.verdict == "OK"
def test_non_valueerror_render_failure_is_skipped_not_raised(self):
"""Regression (v0.71.27 real-SmolLM2 smoke test) — see the matching
eos_in_labels test for the full rationale."""
from soup_cli.utils.data_doctor import check_bos_duplication
rows = [
_chat_row(("system", "s"), ("user", "hi"), ("assistant", "hello")),
_chat_row(("user", "hi"), ("assistant", "hello")),
]
c = check_bos_duplication(_OddRaisingTokenizer(), rows, max_length=2048)
assert c.verdict == "OK"
class TestCheckSystemRole:
def test_no_system_rows_is_ok(self):
from soup_cli.utils.data_doctor import check_system_role
rows = [_chat_row(("user", "hi"), ("assistant", "hello"))]
c = check_system_role(_FakeTokenizer(reject_system=True), rows)
assert c.verdict == "OK"
def test_system_supported(self):
from soup_cli.utils.data_doctor import check_system_role
rows = [_chat_row(("system", "s"), ("user", "hi"), ("assistant", "hello"))]
c = check_system_role(_FakeTokenizer(reject_system=False), rows)
assert c.verdict == "OK"
def test_system_unsupported_is_major(self):
from soup_cli.utils.data_doctor import check_system_role
rows = [_chat_row(("system", "s"), ("user", "hi"), ("assistant", "hello"))] * 3
c = check_system_role(_FakeTokenizer(reject_system=True), rows)
assert c.verdict == "MAJOR"
assert "3 rows" in c.message
class TestCheckUnknownRoles:
def test_known_roles_ok(self):
from soup_cli.utils.data_doctor import check_unknown_roles
rows = [_chat_row(("system", "s"), ("user", "hi"), ("assistant", "hello"))]
c = check_unknown_roles(rows)
assert c.verdict == "OK"
def test_unknown_role_flagged(self):
from soup_cli.utils.data_doctor import check_unknown_roles
rows = [_chat_row(("human", "hi"), ("assistant", "hello"))] * 3
c = check_unknown_roles(rows)
assert c.verdict == "MAJOR" # 3/3 unknown -> frac=1.0 >= the 10% MAJOR cutoff
assert "human" in c.evidence
class TestCheckTruncationRisk:
def test_short_rows_ok(self):
from soup_cli.utils.data_doctor import check_truncation_risk
rows = [_chat_row(("user", "hi"), ("assistant", "ok"))] * 5
c = check_truncation_risk(_FakeTokenizer(), rows, max_length=2048)
assert c.verdict == "OK"
def test_long_rows_flagged(self):
from soup_cli.utils.data_doctor import check_truncation_risk
long_content = "x" * 500
rows = [_chat_row(("user", long_content), ("assistant", long_content))] * 10
c = check_truncation_risk(_FakeTokenizer(), rows, max_length=32)
assert c.verdict == "MAJOR" # 10/10 over -> frac_over=1.0 >= the 20% MAJOR cutoff
assert "32" in c.message
def test_p95_within_bound_is_ok_despite_a_truncated_outlier(self):
"""Regression (tdd-review): this is a COMPOUND gate, not a plain
fraction — `if p95 <= max_length: OK` short-circuits regardless of
frac_over. 19 short rows + 1 huge outlier keeps p95 within bound
even though that one row (5%) would individually be truncated;
empirically verified p95=52 for this exact fixture."""
from soup_cli.utils.data_doctor import check_truncation_risk
rows = [_chat_row(("user", "hi"), ("assistant", "ok"))] * 19 + [
_chat_row(("user", "hi"), ("assistant", "x" * 500))
]
c = check_truncation_risk(_FakeTokenizer(), rows, max_length=52)
assert c.verdict == "OK", c.message # p95(52) <= max_length(52)
assert "1/20" in c.message # the outlier is still named, just not gating
def test_p95_just_over_bound_flips_to_minor(self):
from soup_cli.utils.data_doctor import check_truncation_risk
rows = [_chat_row(("user", "hi"), ("assistant", "ok"))] * 19 + [
_chat_row(("user", "hi"), ("assistant", "x" * 500))
]
c = check_truncation_risk(_FakeTokenizer(), rows, max_length=51)
assert c.verdict == "MINOR", c.message # p95(52) > max_length(51), frac_over 5% < 20%
# ---------------------------------------------------------------------------
# run_doctor — end to end
# ---------------------------------------------------------------------------
class TestRunDoctor:
def test_happy_path(self):
from soup_cli.utils.data_doctor import run_doctor
rows = [_chat_row(("user", "hi"), ("assistant", "hello"))] * 5
report = run_doctor(rows, _FakeTokenizer(), fmt="chatml", max_length=2048)
assert report.overall == "OK"
assert report.rows_scanned == 5
assert report.total_rows == 5
names = {c.name for c in report.checks}
assert {"chat_template", "eos_in_labels", "bos_duplication", "truncation_risk"} <= names
def test_missing_eos_yields_major_overall(self):
from soup_cli.utils.data_doctor import run_doctor
rows = [_chat_row(("user", "hi"), ("assistant", "hello"))] * 5
report = run_doctor(rows, _FakeTokenizer(emit_eos=False), fmt="chatml", max_length=2048)
assert report.overall == "MAJOR"
def test_minor_only_check_yields_minor_overall(self):
"""Regression (tdd-review): every existing 'bad tokenizer' fixture
happens to trip a MAJOR check, so no test proves overall correctly
surfaces MINOR when that's genuinely the worst verdict present."""
from soup_cli.utils.data_doctor import run_doctor
tok = _FakeTokenizer(supports_assistant_mask=False) # only generation_markers fires
rows = [_chat_row(("user", "hi"), ("assistant", "hello"))] * 5
report = run_doctor(rows, tok, fmt="chatml", max_length=2048)
names_to_verdicts = {c.name: c.verdict for c in report.checks}
assert names_to_verdicts["generation_markers"] == "MINOR"
others = {k: v for k, v in names_to_verdicts.items() if k != "generation_markers"}
assert all(v == "OK" for v in others.values())
assert report.overall == "MINOR"
def test_worst_wins_against_a_real_competing_minor(self):
"""A single-bad-check test can't distinguish 'worst wins' from
'first non-OK wins' — this fixture makes generation_markers MINOR
and eos_in_labels MAJOR simultaneously, so only a correct rank
table produces MAJOR overall."""
from soup_cli.utils.data_doctor import run_doctor
tok = _FakeTokenizer(supports_assistant_mask=False, emit_eos=False)
rows = [_chat_row(("user", "hi"), ("assistant", "hello"))] * 5
report = run_doctor(rows, tok, fmt="chatml", max_length=2048)
names_to_verdicts = {c.name: c.verdict for c in report.checks}
assert names_to_verdicts["generation_markers"] == "MINOR"
assert names_to_verdicts["eos_in_labels"] == "MAJOR"
assert report.overall == "MAJOR"
def test_zero_convertible_rows_raises(self):
from soup_cli.utils.data_doctor import run_doctor
rows = [{"prompt": "x", "chosen": "y", "rejected": "z"}] * 3
with pytest.raises(ValueError, match="soup data lint"):
run_doctor(rows, _FakeTokenizer(), fmt="dpo", max_length=2048)
def test_empty_dataset_does_not_raise(self):
from soup_cli.utils.data_doctor import run_doctor
report = run_doctor([], _FakeTokenizer(), fmt="chatml", max_length=2048)
assert report.total_rows == 0
def test_sample_size_bounds(self):
from soup_cli.utils.data_doctor import run_doctor
rows = [_chat_row(("user", "hi"), ("assistant", "ho"))] * 3
with pytest.raises(ValueError):
run_doctor(rows, _FakeTokenizer(), fmt="chatml", sample_size=0)
with pytest.raises(ValueError):
run_doctor(rows, _FakeTokenizer(), fmt="chatml", sample_size=10**9)
def test_max_length_bounds(self):
from soup_cli.utils.data_doctor import run_doctor
rows = [_chat_row(("user", "hi"), ("assistant", "ho"))]
with pytest.raises(ValueError):
run_doctor(rows, _FakeTokenizer(), fmt="chatml", max_length=0)
def test_masking_strategy_flags_are_actually_threaded_into_checks(self):
"""Regression (code-review MEDIUM): eos_in_labels/bos_duplication
must use the SAME masking strategy as --show-mask, not hard-code
assistant-only — verified by making the two strategies disagree.
build_assistant_only_labels always trains role=="assistant"
regardless of any 'train' key; build_per_message_train_labels reads
'train' directly. Marking the assistant turn train=False and the
user turn train=True flips which span gets checked — and since this
fake only ever appends EOS after an assistant turn, that flip must
turn an OK verdict into a flagged one.
"""
from soup_cli.utils.data_doctor import run_doctor
rows = [
{
"messages": [
{"role": "user", "content": "hi", "train": True},
{"role": "assistant", "content": "hello", "train": False},
]
}
]
tok = _FakeTokenizer(emit_eos=True)
default_report = run_doctor(rows, tok, fmt="chatml", max_length=2048)
default_eos = next(c for c in default_report.checks if c.name == "eos_in_labels")
assert default_eos.verdict == "OK", default_eos.message
train_field_report = run_doctor(
rows, tok, fmt="chatml", max_length=2048,
train_on_messages_with_train_field=True,
)
train_field_eos = next(c for c in train_field_report.checks if c.name == "eos_in_labels")
# 1/1 missing -> frac=1.0 >= the 50% MAJOR cutoff
assert train_field_eos.verdict == "MAJOR", train_field_eos.message
def test_masking_strategy_flag_type_validation(self):
from soup_cli.utils.data_doctor import run_doctor
rows = [_chat_row(("user", "hi"), ("assistant", "ho"))]
with pytest.raises(TypeError):
run_doctor(rows, _FakeTokenizer(), fmt="chatml", train_on_responses_only="yes")
with pytest.raises(TypeError):
run_doctor(
rows, _FakeTokenizer(), fmt="chatml", train_on_messages_with_train_field=1
)
# ---------------------------------------------------------------------------
# --show-mask: MaskedToken / MaskPreviewRow / render_mask_preview
# ---------------------------------------------------------------------------
class TestMaskedToken:
def test_valid(self):
from soup_cli.utils.data_doctor import MaskedToken
t = MaskedToken(text="hi", trained=True)
assert t.weight == 1.0
def test_rejects_negative_weight(self):
from soup_cli.utils.data_doctor import MaskedToken
with pytest.raises(ValueError):
MaskedToken(text="hi", trained=True, weight=-1.0)
class TestRenderMaskPreview:
def test_assistant_only_strategy(self):
from soup_cli.utils.data_doctor import render_mask_preview
rows = [_chat_row(("user", "hi"), ("assistant", "hello"))]
previews = render_mask_preview(rows, _FakeTokenizer(), fmt="chatml", n=1)
assert len(previews) == 1
assert previews[0].strategy == "assistant_only"
assert any(t.trained for t in previews[0].tokens)
assert any(not t.trained for t in previews[0].tokens)
def test_per_message_train_field_strategy(self):
from soup_cli.utils.data_doctor import render_mask_preview
rows = [_chat_row(("user", "hi"), ("assistant", "hello"))]
previews = render_mask_preview(
rows, _FakeTokenizer(), fmt="chatml", n=1, train_on_messages_with_train_field=True,
)
assert previews[0].strategy == "per_message_train"
def test_legacy_text_strategy_trains_everything(self):
from soup_cli.utils.data_doctor import render_mask_preview
rows = [_chat_row(("user", "hi"), ("assistant", "hello"))]
previews = render_mask_preview(
rows, _FakeTokenizer(), fmt="chatml", n=1, train_on_responses_only=False,
)
assert previews[0].strategy == "legacy_text"
assert all(t.trained for t in previews[0].tokens)
def test_legacy_text_strategy_truncates_to_max_length(self):
"""Regression (code-review LOW): legacy_text must truncate like the
other two strategies (both delegate to data.loss_mask._truncate) —
otherwise --show-mask could render more tokens than the trainer
would actually see."""
from soup_cli.utils.data_doctor import render_mask_preview
rows = [_chat_row(("user", "hi"), ("assistant", "x" * 200))]
previews = render_mask_preview(
rows, _FakeTokenizer(), fmt="chatml", n=1,
train_on_responses_only=False, max_length=10,
)
assert len(previews) == 1
assert len(previews[0].tokens) == 10
def test_raft_strategy_uses_loss_weights(self):
from soup_cli.utils.data_doctor import render_mask_preview
rows = [
{
"query": "What is the capital of France?",
"golden_doc": "Paris is the capital of France.",
"distractor_docs": ["Berlin is the capital of Germany."],
"answer": "Paris",
}
]
previews = render_mask_preview(rows, _FakeTokenizer(), fmt="raft", n=1)
assert len(previews) == 1
assert previews[0].strategy == "raft"
assert any(t.trained for t in previews[0].tokens)
assert any(not t.trained for t in previews[0].tokens)
def test_stops_at_n(self):
from soup_cli.utils.data_doctor import render_mask_preview
rows = [_chat_row(("user", "hi"), ("assistant", "hello"))] * 10
previews = render_mask_preview(rows, _FakeTokenizer(), fmt="chatml", n=3)
assert len(previews) == 3
def test_skips_unconvertible_rows(self):
from soup_cli.utils.data_doctor import render_mask_preview
rows = [{"garbage": True}, _chat_row(("user", "hi"), ("assistant", "hello"))]
previews = render_mask_preview(rows, _FakeTokenizer(), fmt="chatml", n=5)
assert len(previews) == 1
def test_row_index_reflects_real_source_position_not_display_order(self):
"""Regression (code-review LOW): row_index must be the row's real
index into raw_rows (what a user would find in their JSONL), not a
0,1,2... display-order counter that drifts once an earlier row is
skipped for failing to convert/render."""
from soup_cli.utils.data_doctor import render_mask_preview
rows = [
{"garbage": True}, # index 0 — skipped
{"garbage": True}, # index 1 — skipped
_chat_row(("user", "hi"), ("assistant", "hello")), # index 2 — survives
]
previews = render_mask_preview(rows, _FakeTokenizer(), fmt="chatml", n=5)
assert len(previews) == 1
assert previews[0].row_index == 2
def test_non_valueerror_render_failure_is_skipped_not_raised(self):
"""Regression (v0.71.27 real-SmolLM2 smoke test) — see the matching
eos_in_labels test for the full rationale."""
from soup_cli.utils.data_doctor import render_mask_preview
rows = [
_chat_row(("system", "s"), ("user", "hi"), ("assistant", "hello")),
_chat_row(("user", "hi"), ("assistant", "hello")),
]
previews = render_mask_preview(rows, _OddRaisingTokenizer(), fmt="chatml", n=5)
assert len(previews) == 1
def test_n_bounds(self):
from soup_cli.utils.data_doctor import render_mask_preview
rows = [_chat_row(("user", "hi"), ("assistant", "hello"))]
with pytest.raises(ValueError):
render_mask_preview(rows, _FakeTokenizer(), fmt="chatml", n=0)
with pytest.raises(ValueError):
render_mask_preview(rows, _FakeTokenizer(), fmt="chatml", n=10_000)
# ---------------------------------------------------------------------------
# No top-level heavy imports (fast CLI convention)
# ---------------------------------------------------------------------------
class TestNoTopLevelHeavyImport:
@pytest.mark.parametrize(
"relpath",
[
"src/soup_cli/utils/data_doctor.py",
"src/soup_cli/utils/data_lint.py",
"src/soup_cli/commands/data_doctor.py",
],
)
def test_no_torch_or_transformers_at_module_level(self, relpath):
source = (_PROJECT_ROOT / relpath).read_text(encoding="utf-8")
# Column-0 (unindented) lines only — a lazy import inside a function
# body is indented and must NOT trip this check.
top_level_lines = [ln for ln in source.splitlines() if ln[:1] not in ("", " ", "\t")]
for ln in top_level_lines:
assert not ln.startswith("import torch"), relpath
assert not ln.startswith("from torch"), relpath
assert not ln.startswith("import transformers"), relpath
assert not ln.startswith("from transformers"), relpath
# ===========================================================================
# data_lint.py
# ===========================================================================
class TestLintCheck:
"""Regression (tdd-review): the doctor/lint 'twins' had asymmetric
coverage — DoctorCheck/DoctorReport had a dedicated test class,
LintCheck/LintReport did not (0% coverage on their __post_init__
validation per the coverage report)."""
def test_valid_check(self):
from soup_cli.utils.data_lint import LintCheck
c = LintCheck(name="length_bias", verdict="OK", message="fine", evidence="")
assert c.verdict == "OK"
def test_rejects_unknown_name(self):
from soup_cli.utils.data_lint import LintCheck
with pytest.raises(ValueError, match="unknown"):
LintCheck(name="not_a_real_check", verdict="OK", message="x")
def test_rejects_unknown_verdict(self):
from soup_cli.utils.data_lint import LintCheck
with pytest.raises(ValueError, match="verdict"):
LintCheck(name="length_bias", verdict="BAD", message="x")
def test_rejects_null_byte(self):
from soup_cli.utils.data_lint import LintCheck
with pytest.raises(ValueError, match="null byte"):
LintCheck(name="length_bias", verdict="OK", message="a\x00b")
def test_rejects_oversize_message(self):
from soup_cli.utils.data_lint import LintCheck
with pytest.raises(ValueError, match="too long"):
LintCheck(name="length_bias", verdict="OK", message="x" * 3000)
def test_is_frozen(self):
from soup_cli.utils.data_lint import LintCheck
c = LintCheck(name="length_bias", verdict="OK", message="x")
with pytest.raises(Exception): # noqa: PT011 — dataclasses.FrozenInstanceError
c.verdict = "MAJOR"
class TestLintReport:
def test_to_dict_roundtrip_shape(self):
from soup_cli.utils.data_lint import LintCheck, compose_lint_report
checks = [LintCheck(name="length_bias", verdict="OK", message="fine", evidence="e")]
report = compose_lint_report(checks, fmt="dpo", rows_scanned=5, total_rows=10)
d = report.to_dict()
assert d["fmt"] == "dpo"
assert d["overall"] == "OK"
assert d["rows_scanned"] == 5
assert d["total_rows"] == 10
assert d["checks"][0]["name"] == "length_bias"
def test_rows_scanned_cannot_exceed_total(self):
from soup_cli.utils.data_lint import LintReport
with pytest.raises(ValueError, match="cannot exceed"):
LintReport(checks=(), overall="OK", fmt="dpo", rows_scanned=10, total_rows=5)
def test_negative_counts_rejected(self):
from soup_cli.utils.data_lint import LintReport
with pytest.raises(ValueError):
LintReport(checks=(), overall="OK", fmt="dpo", rows_scanned=-1, total_rows=5)
def test_rejects_unsupported_fmt(self):
from soup_cli.utils.data_lint import LintReport
with pytest.raises(ValueError, match="fmt"):
LintReport(checks=(), overall="OK", fmt="chatml", rows_scanned=0, total_rows=0)
class TestExtractPrefText:
def test_string_passthrough(self):
from soup_cli.utils.data_lint import extract_pref_text
assert extract_pref_text("hello") == "hello"
def test_messages_list_joins_content(self):
from soup_cli.utils.data_lint import extract_pref_text
val = [{"role": "assistant", "content": "hi"}, {"role": "user", "content": "there"}]
out = extract_pref_text(val)
assert "hi" in out and "there" in out
def test_none_becomes_empty(self):
from soup_cli.utils.data_lint import extract_pref_text
assert extract_pref_text(None) == ""
class TestCohensD:
def test_identical_distributions_zero(self):
from soup_cli.utils.data_lint import cohens_d
assert cohens_d([10.0] * 20, [10.0] * 20) == 0.0
def test_large_shift_large_effect(self):
from soup_cli.utils.data_lint import cohens_d
a = [100.0 + i for i in range(20)]
b = [10.0 + i for i in range(20)]
d = cohens_d(a, b)
assert d > 0.8
def test_direction_sign(self):
from soup_cli.utils.data_lint import cohens_d
a = [5.0] * 10
b = [50.0] * 10
assert cohens_d(a, b) < 0
def test_empty_raises(self):
from soup_cli.utils.data_lint import cohens_d
with pytest.raises(ValueError):
cohens_d([], [1.0])
def test_single_sample_each_no_crash(self):
from soup_cli.utils.data_lint import cohens_d
assert cohens_d([5.0], [1.0]) > 0
def test_asymmetric_degenerate_n_takes_sign_only_path(self):
"""Regression (tdd-review): the n<2 fallback is a single combined
`n_a < 2 or n_b < 2` condition — only the SYMMETRIC (both n=1) case
was tested. A larger, non-degenerate second sample must still take
the sign-only path when the OTHER side is degenerate."""
from soup_cli.utils.data_lint import cohens_d
assert cohens_d([5.0], [1.0, 2.0, 9.0]) > 0
assert cohens_d([1.0, 2.0, 9.0], [5.0]) < 0
def test_degenerate_tie_is_exactly_zero(self):
"""Regression (tdd-review): the n<2 fallback's `mean_a == mean_b`
True branch is only ever hit incidentally (inside a 1-row DPO
fixture that never inspects length_bias) — assert it directly."""
from soup_cli.utils.data_lint import cohens_d
assert cohens_d([5.0], [5.0]) == 0.0
class TestCheckLengthBias:
def test_balanced_lengths_ok(self):
from soup_cli.utils.data_lint import check_length_bias
rows = [{"chosen": "a" * 20, "rejected": "b" * 20} for _ in range(20)]
c = check_length_bias(rows, length_fn=len)
assert c.verdict == "OK"
def test_chosen_systematically_longer_flagged(self):
from soup_cli.utils.data_lint import check_length_bias
rows = [{"chosen": "a" * 200, "rejected": "b" * 10} for _ in range(20)]
c = check_length_bias(rows, length_fn=len)
# Constant, zero-variance lengths -> cohens_d's sign-only fallback = 1.0 >= 0.8
assert c.verdict == "MAJOR"
assert "chosen" in c.message.lower()
def test_just_at_or_above_the_major_boundary(self):
"""Regression (tdd-review): no test hit the documented |d| >= 0.8
MAJOR cutoff near the exact boundary. Equal-n, equal-variance
samples make Cohen's d hand-computable via mean_diff/pooled_std —
note 0.8 itself has no exact binary float representation, so this
targets a value confirmed (empirically, not just hand-derived) to
land just above it, rather than an unreliable exact match. Uses a
controlled length_fn (parses the row as a float) for exact numeric
control instead of string length."""
from soup_cli.utils.data_lint import check_length_bias
# chosen: mean=10, var=4 (std=2); rejected: mean=8.36, var=4 (std=2)
# equal n + equal variance -> pooled_std=2 -> d=1.64/2=0.82 (verified: 0.8200000000000004)
rows = [
{"chosen": "8.0", "rejected": "6.36"},
{"chosen": "10.0", "rejected": "8.36"},
{"chosen": "12.0", "rejected": "10.36"},
]
c = check_length_bias(rows, length_fn=float)
assert c.verdict == "MAJOR", c.message
def test_just_below_the_major_boundary(self):
from soup_cli.utils.data_lint import check_length_bias
# Same shape, mean diff 1.6 -> d=1.6/2=0.8 mathematically, but
# verified to resolve to 0.7999999999999998 in float64 (0.8 has no
# exact binary representation) -> MINOR, not MAJOR. This IS the
# near-boundary case, not a workaround for it.
rows = [
{"chosen": "8.0", "rejected": "6.4"},
{"chosen": "10.0", "rejected": "8.4"},
{"chosen": "12.0", "rejected": "10.4"},
]
c = check_length_bias(rows, length_fn=float)
assert c.verdict == "MINOR", c.message
def test_empty_is_ok(self):
from soup_cli.utils.data_lint import check_length_bias
c = check_length_bias([], length_fn=len)
assert c.verdict == "OK"
class TestCheckLabelImbalance:
def test_balanced_ok(self):
from soup_cli.utils.data_lint import check_label_imbalance
rows = [{"label": True}] * 15 + [{"label": False}] * 15
c = check_label_imbalance(rows)
assert c.verdict == "OK"
def test_severe_imbalance_flagged(self):
from soup_cli.utils.data_lint import check_label_imbalance
rows = [{"label": True}] * 99 + [{"label": False}] * 1
c = check_label_imbalance(rows)
assert c.verdict == "MAJOR" # minority_frac=0.01 < the 5% MAJOR cutoff
def test_empty_is_ok(self):
from soup_cli.utils.data_lint import check_label_imbalance
assert check_label_imbalance([]).verdict == "OK"
class TestCheckIdenticalPairs:
def test_no_dupes_ok(self):
from soup_cli.utils.data_lint import check_identical_pairs
rows = [{"chosen": "a", "rejected": "b"} for _ in range(5)]
c = check_identical_pairs(rows)
assert c.verdict == "OK"
def test_identical_pair_is_major(self):
from soup_cli.utils.data_lint import check_identical_pairs
rows = [{"chosen": "same text", "rejected": "same text"}] + [
{"chosen": "a", "rejected": "b"} for _ in range(9)
]
c = check_identical_pairs(rows)
assert c.verdict == "MAJOR"
assert "1/10" in c.message
def test_messages_list_pairs_compared(self):
from soup_cli.utils.data_lint import check_identical_pairs
msgs = [{"role": "assistant", "content": "x"}]
rows = [{"chosen": msgs, "rejected": list(msgs)}]
c = check_identical_pairs(rows)
assert c.verdict == "MAJOR"
class TestCheckPromptLeak:
def test_no_leak_ok(self):
from soup_cli.utils.data_lint import check_prompt_leak
rows = [{"prompt": "x" * 60, "chosen": "totally unrelated", "rejected": "also unrelated"}]
c = check_prompt_leak(rows, fmt="dpo")
assert c.verdict == "OK"
def test_leak_in_chosen_flagged(self):
from soup_cli.utils.data_lint import check_prompt_leak
prompt = "Please summarize the following article: " + ("lorem ipsum " * 5)
rows = [{"prompt": prompt, "chosen": prompt + " Here is a summary.", "rejected": "n/a"}]
c = check_prompt_leak(rows, fmt="dpo")
assert c.verdict == "MAJOR" # 1/1 flagged -> frac=1.0 >= the 10% MAJOR cutoff
def test_kto_checks_completion(self):
from soup_cli.utils.data_lint import check_prompt_leak
prompt = "Please summarize the following article: " + ("lorem ipsum " * 5)
rows = [{"prompt": prompt, "completion": prompt + " Summary.", "label": True}]
c = check_prompt_leak(rows, fmt="kto")
assert c.verdict == "MAJOR" # 1/1 flagged -> frac=1.0 >= the 10% MAJOR cutoff
def test_short_prompts_not_flagged(self):
from soup_cli.utils.data_lint import check_prompt_leak
rows = [{"prompt": "hi", "chosen": "hi there", "rejected": "no"}]
c = check_prompt_leak(rows, fmt="dpo")
assert c.verdict == "OK"
@pytest.mark.skipif(not _HAS_DATASKETCH, reason="datasketch not installed")
class TestCheckNearDuplicates:
def test_no_dupes_ok(self):
from soup_cli.utils.data_lint import check_near_duplicates
rows = [
{
"prompt": f"unique question number {i} about topic {i}",
"chosen": "x",
"rejected": "y",
}
for i in range(10)
]
c = check_near_duplicates(rows, key_fn=lambda r: r["prompt"])
assert c.verdict == "OK"
def test_duplicates_flagged(self):
from soup_cli.utils.data_lint import check_near_duplicates
base = "what is the capital of france and why is it important historically"
rows = [{"prompt": base, "chosen": "x", "rejected": "y"} for _ in range(10)]
c = check_near_duplicates(rows, key_fn=lambda r: r["prompt"])
# 10 byte-identical prompts -> every row matches every other -> frac=1.0
assert c.verdict == "MAJOR"
class TestCheckNearDuplicatesNoDep:
def test_missing_dep_degrades_gracefully(self, monkeypatch):
import builtins
real_import = builtins.__import__
def fake_import(name, *args, **kwargs):
if name == "datasketch":
raise ImportError("no datasketch")
return real_import(name, *args, **kwargs)
monkeypatch.setattr(builtins, "__import__", fake_import)
from soup_cli.utils.data_lint import check_near_duplicates
rows = [{"prompt": "x", "chosen": "a", "rejected": "b"}]
c = check_near_duplicates(rows, key_fn=lambda r: r["prompt"])
assert c.verdict == "OK"
assert "skip" in c.message.lower()
class TestRunLint:
def test_dpo_happy_path(self):
from soup_cli.utils.data_lint import run_lint
rows = [
{"prompt": f"q{i}", "chosen": "answer text here", "rejected": "different answer"}
for i in range(10)
]
report = run_lint(rows, fmt="dpo")
assert report.fmt == "dpo"
names = {c.name for c in report.checks}
assert "length_bias" in names
assert "identical_pairs" in names
assert "label_imbalance" not in names
def test_kto_happy_path(self):
from soup_cli.utils.data_lint import run_lint
rows = [
{"prompt": f"q{i}", "completion": "some completion", "label": i % 2 == 0}
for i in range(10)
]
report = run_lint(rows, fmt="kto")
names = {c.name for c in report.checks}
assert "label_imbalance" in names
assert "length_bias" not in names
def test_unsupported_format_raises(self):
from soup_cli.utils.data_lint import run_lint
with pytest.raises(ValueError, match="dpo|kto"):
run_lint([{"messages": []}], fmt="chatml")
def test_sample_size_bounds(self):
"""Regression (tdd-review): run_doctor has this bounds test;
run_lint's identical validation was untested."""
from soup_cli.utils.data_lint import run_lint
rows = [{"prompt": "q", "chosen": "a", "rejected": "b"}] * 3
with pytest.raises(ValueError):
run_lint(rows, fmt="dpo", sample_size=0)
with pytest.raises(ValueError):
run_lint(rows, fmt="dpo", sample_size=10**9)
def test_identical_pairs_overall_major(self):
from soup_cli.utils.data_lint import run_lint
rows = [{"prompt": "q", "chosen": "same", "rejected": "same"}]
report = run_lint(rows, fmt="dpo")
assert report.overall == "MAJOR"
def test_worst_wins_against_a_real_competing_minor(self):
"""Regression (tdd-review): the only prior multi-row fixture puts
every row in the SAME bucket, so it can't distinguish 'worst wins'
from 'first non-OK wins'. Mix a MAJOR identical_pairs row (a binary
any-occurrence rule) with a separate MINOR prompt_leak row (a
fractional rule, <10% here) and confirm MAJOR — not MINOR — wins.
"""
from soup_cli.utils.data_lint import run_lint
clean_rows = [
{
"prompt": f"distinct question {i}",
"chosen": f"distinct answer {i}",
"rejected": f"other {i}",
}
for i in range(13)
]
leak_prompt = "Please summarize the following unique article: " + ("lorem ipsum " * 5)
leak_row = {"prompt": leak_prompt, "chosen": leak_prompt + " done.", "rejected": "n/a"}
identical_row = {"prompt": "q-identical", "chosen": "same text", "rejected": "same text"}
rows = clean_rows + [leak_row, identical_row]
report = run_lint(rows, fmt="dpo")
names_to_verdicts = {c.name: c.verdict for c in report.checks}
assert names_to_verdicts["identical_pairs"] == "MAJOR"
assert names_to_verdicts["prompt_leak"] == "MINOR", names_to_verdicts
assert report.overall == "MAJOR"
def test_auto_detects_format(self):
from soup_cli.utils.data_lint import run_lint
rows = [{"prompt": "q", "chosen": "a", "rejected": "b"}]
report = run_lint(rows, fmt="auto")
assert report.fmt == "dpo"
def test_all_rows_unparseable_raises_not_silent_ok(self):
"""Regression (code-review HIGH, empirically reproduced): a dataset
where every row is structurally broken (e.g. chosen/rejected
corrupted to null by an upstream export bug) must not silently
report a false "OK, 0 rows scanned" — mirrors run_doctor's
`test_zero_convertible_rows_raises` guard."""
from soup_cli.utils.data_lint import run_lint
rows = [{"prompt": "q", "chosen": None, "rejected": None}] * 5
with pytest.raises(ValueError, match="no rows converted"):
run_lint(rows, fmt="dpo")
def test_empty_dataset_does_not_raise_zero_rows_guard(self):
"""The zero-rows guard is for 'every row failed to convert', not
'there were no rows to begin with' — an empty dataset must still
hit the earlier, more specific error."""
from soup_cli.utils.data_lint import run_lint
with pytest.raises(ValueError, match="empty dataset"):
run_lint([], fmt="auto")
# ===========================================================================
# CLI layer — soup data doctor / soup data lint
# ===========================================================================
def _fake_resolve_tokenizer_factory(tok):
def _fake(model, *, trust_remote_code=False):
return tok
return _fake
def _patch_tokenizer(monkeypatch, tok=None):
"""Make ``commands/data_doctor.py``'s ``engine.resolve_tokenizer`` return ``tok``."""
import soup_cli.utils.data_doctor as engine
fake = _fake_resolve_tokenizer_factory(tok or _FakeTokenizer())
monkeypatch.setattr(engine, "resolve_tokenizer", fake)
class TestForTerminal:
def test_strips_esc_and_bell(self):
from soup_cli.commands.data_doctor import _for_terminal
assert _for_terminal("\x1b]0;PWNED\x07human") == "]0;PWNEDhuman"
def test_preserves_tab_newline_cr(self):
from soup_cli.commands.data_doctor import _for_terminal
assert _for_terminal("a\tb\nc\rd") == "a\tb\nc\rd"
def test_strips_del(self):
from soup_cli.commands.data_doctor import _for_terminal
assert _for_terminal("a\x7fb") == "ab"
def test_plain_text_unchanged(self):
from soup_cli.commands.data_doctor import _for_terminal
assert _for_terminal("nothing weird here") == "nothing weird here"
class TestLoadTokenizerUsesResolvedTrust:
def test_resolved_trust_value_is_what_gets_passed_through(self, monkeypatch):
"""Regression (security-review LOW): _load_tokenizer must use
resolve_trust_remote_code's RETURN value (mirrors chat.py/diff.py/
merge.py/...), not silently re-use the raw CLI-supplied bool —
a latent footgun if that function's contract ever transforms the
flag rather than just gating/warning on it."""
import soup_cli.commands.data_doctor as cli_mod
captured = {}
def fake_resolve_trust(model, requested, console, requires):
return "SENTINEL" # deliberately not the raw bool, to prove it's threaded through
def fake_resolve_tokenizer(model, *, trust_remote_code):
captured["trust_remote_code"] = trust_remote_code
return _FakeTokenizer()
monkeypatch.setattr(cli_mod, "resolve_trust_remote_code", fake_resolve_trust)
monkeypatch.setattr(cli_mod.engine, "resolve_tokenizer", fake_resolve_tokenizer)
cli_mod._load_tokenizer("fake/model", False)
assert captured["trust_remote_code"] == "SENTINEL"
class TestDoctorCli:
def test_help(self):
from soup_cli.cli import app
result = runner.invoke(app, ["data", "doctor", "--help"])
assert result.exit_code == 0
assert "model" in result.output.lower()
def test_happy_path_exits_zero(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "train.jsonl"
_write_jsonl(data_path, [_chat_row(("user", "hi"), ("assistant", "hello"))] * 5)
_patch_tokenizer(monkeypatch)
from soup_cli.cli import app
result = runner.invoke(app, ["data", "doctor", str(data_path), "--model", "fake/model"])
assert result.exit_code == 0, result.output
def test_trust_remote_code_shows_warning_panel(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "train.jsonl"
_write_jsonl(data_path, [_chat_row(("user", "hi"), ("assistant", "hello"))])
_patch_tokenizer(monkeypatch)
from soup_cli.cli import app
result = runner.invoke(
app,
["data", "doctor", str(data_path), "--model", "some-org/model", "--trust-remote-code"],
)
assert result.exit_code == 0, result.output
assert "remote code" in result.output.lower()
def test_local_model_requiring_remote_code_is_refused_without_opt_in(
self, tmp_path, monkeypatch
):
"""Regression (code-review MEDIUM): _load_tokenizer must probe
model_requires_trust_remote_code like every other model-loading
command (chat.py, diff.py, export.py, ...) — a local model dir with
auto_map set must be refused with the standard gate message unless
--trust-remote-code is passed, not silently attempted."""
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "train.jsonl"
_write_jsonl(data_path, [_chat_row(("user", "hi"), ("assistant", "hello"))])
model_dir = tmp_path / "custom-model"
model_dir.mkdir()
(model_dir / "config.json").write_text(
json.dumps({"auto_map": {"AutoTokenizer": "custom.CustomTokenizer"}}),
encoding="utf-8",
)
_patch_tokenizer(monkeypatch)
from soup_cli.cli import app
result = runner.invoke(app, ["data", "doctor", str(data_path), "--model", str(model_dir)])
assert result.exit_code == 1, result.output
assert "trust" in result.output.lower() and "remote" in result.output.lower()
def test_control_characters_in_dataset_content_stripped_from_terminal(
self, tmp_path, monkeypatch
):
"""Regression (security-review MEDIUM): rich.markup.escape() only
neutralises [...] tag syntax, not raw control bytes — an attacker-
controlled 'role' field carrying a literal ESC byte (e.g. an OSC
title-spoof / cursor-trick sequence) must never reach the terminal
raw, even though it survives escape() untouched."""
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "train.jsonl"
evil_role = "\x1b]0;PWNED\x07human"
_write_jsonl(
data_path,
[
{
"messages": [
{"role": evil_role, "content": "hi"},
{"role": "assistant", "content": "hello"},
]
}
],
)
_patch_tokenizer(monkeypatch)
from soup_cli.cli import app
result = runner.invoke(app, ["data", "doctor", str(data_path), "--model", "fake/model"])
assert "\x1b" not in result.output
assert "\x07" not in result.output
def test_empty_model_value_is_friendly_error(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "train.jsonl"
_write_jsonl(data_path, [_chat_row(("user", "hi"), ("assistant", "hello"))])
from soup_cli.cli import app
result = runner.invoke(app, ["data", "doctor", str(data_path), "--model", ""])
assert result.exit_code == 1, result.output
assert not isinstance(result.exception, (KeyError, AttributeError))
def test_missing_eos_exits_two(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "train.jsonl"
_write_jsonl(data_path, [_chat_row(("user", "hi"), ("assistant", "hello"))] * 5)
_patch_tokenizer(monkeypatch, _FakeTokenizer(emit_eos=False))
from soup_cli.cli import app
result = runner.invoke(app, ["data", "doctor", str(data_path), "--model", "fake/model"])
assert result.exit_code == 2, result.output
assert "stop" in result.output.lower() or "eos" in result.output.lower()
def test_missing_file_exits_nonzero(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
from soup_cli.cli import app
result = runner.invoke(app, ["data", "doctor", "nope.jsonl", "--model", "fake/model"])
assert result.exit_code == 1
def test_dpo_format_routes_to_lint_message(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "pref.jsonl"
_write_jsonl(data_path, [{"prompt": "q", "chosen": "a", "rejected": "b"}] * 3)
_patch_tokenizer(monkeypatch)
from soup_cli.cli import app
result = runner.invoke(app, ["data", "doctor", str(data_path), "--model", "fake/model"])
assert result.exit_code == 2
assert "lint" in result.output.lower()
def test_raft_without_show_mask_rejected(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "raft.jsonl"
_write_jsonl(
data_path,
[
{
"query": "q",
"golden_doc": "the answer is here",
"distractor_docs": ["irrelevant"],
"answer": "here",
}
]
* 3,
)
_patch_tokenizer(monkeypatch)
from soup_cli.cli import app
result = runner.invoke(
app, ["data", "doctor", str(data_path), "--model", "fake/model", "--format", "raft"]
)
assert result.exit_code == 2
assert "show-mask" in result.output.lower()
def test_show_mask_renders_rows(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "train.jsonl"
_write_jsonl(data_path, [_chat_row(("user", "hi"), ("assistant", "hello"))] * 3)
_patch_tokenizer(monkeypatch)
from soup_cli.cli import app
result = runner.invoke(
app, ["data", "doctor", str(data_path), "--model", "fake/model", "--show-mask", "2"]
)
assert result.exit_code == 0, result.output
def test_raft_show_mask_works_without_full_report(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "raft.jsonl"
_write_jsonl(
data_path,
[
{
"query": "q",
"golden_doc": "the answer is here",
"distractor_docs": ["irrelevant"],
"answer": "here",
}
]
* 3,
)
_patch_tokenizer(monkeypatch)
from soup_cli.cli import app
result = runner.invoke(
app,
[
"data", "doctor", str(data_path), "--model", "fake/model",
"--format", "raft", "--show-mask", "1",
],
)
assert result.exit_code == 0, result.output
def test_output_json_written(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "train.jsonl"
_write_jsonl(data_path, [_chat_row(("user", "hi"), ("assistant", "hello"))] * 3)
out_path = tmp_path / "report.json"
_patch_tokenizer(monkeypatch)
from soup_cli.cli import app
result = runner.invoke(
app,
["data", "doctor", str(data_path), "--model", "fake/model", "--output", str(out_path)],
)
assert result.exit_code == 0, result.output
payload = json.loads(out_path.read_text(encoding="utf-8"))
assert payload["overall"] == "OK"
def test_output_path_outside_cwd_rejected(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "train.jsonl"
_write_jsonl(data_path, [_chat_row(("user", "hi"), ("assistant", "hello"))])
_patch_tokenizer(monkeypatch)
from soup_cli.cli import app
result = runner.invoke(
app,
[
"data",
"doctor",
str(data_path),
"--model",
"fake/model",
"--output",
"../outside.json",
],
)
assert result.exit_code == 1
def test_empty_dataset_file_is_friendly_error(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "empty.jsonl"
data_path.write_text("", encoding="utf-8")
from soup_cli.cli import app
result = runner.invoke(app, ["data", "doctor", str(data_path), "--model", "fake/model"])
assert result.exit_code == 1, result.output
assert "empty" in result.output.lower()
def test_format_auto_detect_failure_is_friendly_error(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "undetectable.jsonl"
_write_jsonl(data_path, [{"some_unknown_key": "value"}])
from soup_cli.cli import app
result = runner.invoke(app, ["data", "doctor", str(data_path), "--model", "fake/model"])
assert result.exit_code == 1, result.output
def test_show_mask_out_of_bounds_is_friendly_error(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "train.jsonl"
_write_jsonl(data_path, [_chat_row(("user", "hi"), ("assistant", "hello"))])
_patch_tokenizer(monkeypatch)
from soup_cli.cli import app
result = runner.invoke(
app, ["data", "doctor", str(data_path), "--model", "fake/model", "--show-mask", "0"]
)
assert result.exit_code == 1, result.output
def test_sample_zero_is_friendly_error(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "train.jsonl"
_write_jsonl(data_path, [_chat_row(("user", "hi"), ("assistant", "hello"))])
_patch_tokenizer(monkeypatch)
from soup_cli.cli import app
result = runner.invoke(
app, ["data", "doctor", str(data_path), "--model", "fake/model", "--sample", "0"]
)
assert result.exit_code == 1, result.output
def test_show_mask_no_renderable_rows_prints_warning(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "train.jsonl"
_write_jsonl(data_path, [_chat_row(("user", "hi"), ("assistant", "hello"))])
_patch_tokenizer(monkeypatch, _FakeTokenizer(has_chat_template=False))
from soup_cli.cli import app
result = runner.invoke(
app, ["data", "doctor", str(data_path), "--model", "fake/model", "--show-mask", "1"]
)
# No chat_template -> check_chat_template is MAJOR (exit 2) AND
# render_mask_preview finds nothing to render (the warning path).
assert result.exit_code == 2, result.output
assert "no rows could be rendered" in result.output.lower()
class TestLintCli:
def test_help(self):
from soup_cli.cli import app
result = runner.invoke(app, ["data", "lint", "--help"])
assert result.exit_code == 0
def test_happy_path_exits_zero(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "pref.jsonl"
_write_jsonl(
data_path,
[
{
"prompt": f"q{i}",
"chosen": "a reasonable answer",
"rejected": "a different answer",
}
for i in range(10)
],
)
from soup_cli.cli import app
result = runner.invoke(app, ["data", "lint", str(data_path)])
assert result.exit_code == 0, result.output
def test_identical_pairs_exits_two(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "pref.jsonl"
row = {"prompt": "q", "chosen": "same text here", "rejected": "same text here"}
_write_jsonl(data_path, [row])
from soup_cli.cli import app
result = runner.invoke(app, ["data", "lint", str(data_path)])
assert result.exit_code == 2, result.output
def test_output_json_written(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "pref.jsonl"
rows = [{"prompt": f"q{i}", "chosen": "answer", "rejected": "other"} for i in range(5)]
_write_jsonl(data_path, rows)
out_path = tmp_path / "lint.json"
from soup_cli.cli import app
result = runner.invoke(app, ["data", "lint", str(data_path), "--output", str(out_path)])
assert result.exit_code == 0, result.output
payload = json.loads(out_path.read_text(encoding="utf-8"))
assert payload["fmt"] == "dpo"
def test_unsupported_format_message(self, tmp_path, monkeypatch):
"""Regression (code-review MEDIUM): the 'wrong command for this
format' routing case is a dedicated exit 2 (mirroring doctor's
symmetric dpo/kto-routes-to-lint check), not a generic exit 1 —
the module docstring documents this exact exit-code contract."""
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "chat.jsonl"
_write_jsonl(data_path, [_chat_row(("user", "hi"), ("assistant", "hello"))])
from soup_cli.cli import app
result = runner.invoke(app, ["data", "lint", str(data_path)])
assert result.exit_code == 2, result.output
assert "dpo" in result.output.lower() or "kto" in result.output.lower()
def test_missing_file_exits_nonzero(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
from soup_cli.cli import app
result = runner.invoke(app, ["data", "lint", "nope.jsonl"])
assert result.exit_code == 1
def test_empty_dataset_file_is_friendly_error(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "empty.jsonl"
data_path.write_text("", encoding="utf-8")
from soup_cli.cli import app
result = runner.invoke(app, ["data", "lint", str(data_path)])
assert result.exit_code == 1, result.output
assert "empty" in result.output.lower()
def test_output_path_outside_cwd_rejected(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "pref.jsonl"
_write_jsonl(data_path, [{"prompt": "q", "chosen": "a", "rejected": "b"}])
from soup_cli.cli import app
result = runner.invoke(
app, ["data", "lint", str(data_path), "--output", "../outside.json"]
)
assert result.exit_code == 1
def test_model_flag_uses_tokenizer_for_length_bias(self, tmp_path, monkeypatch):
"""Exercises the --model / length_fn closure end to end (previously
0% covered) — a real (fake) tokenizer must drive the length_bias
effect size instead of the default word-count proxy."""
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "pref.jsonl"
_write_jsonl(
data_path,
[
{"prompt": f"q{i}", "chosen": "a" * 200, "rejected": "b"}
for i in range(10)
],
)
_patch_tokenizer(monkeypatch)
from soup_cli.cli import app
result = runner.invoke(
app, ["data", "lint", str(data_path), "--model", "fake/model"]
)
assert result.exit_code == 2, result.output
assert "length_bias" in result.output.lower() or "cohen" in result.output.lower()
def test_model_flag_not_loaded_for_kto(self, tmp_path, monkeypatch):
"""kto's checks never consult length_fn — --model must not even be
resolved (mirrors doctor's format-before-tokenizer-load ordering)."""
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "pref.jsonl"
_write_jsonl(
data_path,
[{"prompt": f"q{i}", "completion": "c", "label": i % 2 == 0} for i in range(10)],
)
import soup_cli.utils.data_doctor as engine
def _boom(model, *, trust_remote_code=False):
raise AssertionError("tokenizer should not be loaded for kto lint")
monkeypatch.setattr(engine, "resolve_tokenizer", _boom)
from soup_cli.cli import app
result = runner.invoke(
app, ["data", "lint", str(data_path), "--format", "kto", "--model", "fake/model"]
)
assert result.exit_code == 0, result.output
def test_model_length_fn_type_error_fallback(self, tmp_path, monkeypatch):
"""length_fn's inner TypeError fallback (an older tokenizer whose
.encode has no add_special_tokens parameter) must still succeed."""
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "pref.jsonl"
_write_jsonl(
data_path, [{"prompt": f"q{i}", "chosen": "aaaa", "rejected": "b"} for i in range(10)]
)
_patch_tokenizer(monkeypatch, _TypeErrorOnKwargTokenizer())
from soup_cli.cli import app
result = runner.invoke(
app, ["data", "lint", str(data_path), "--model", "fake/model"]
)
# chosen="aaaa" (4 char-tokens) vs rejected="b" (1) via the TypeError
# fallback's char-level tokenizer -> constant, zero-variance Cohen's
# d=1.0 -> MAJOR.
assert result.exit_code == 2, result.output
def test_model_length_fn_encode_failure_degrades_to_word_count(self, tmp_path, monkeypatch):
"""length_fn's outer Exception fallback (a tokenizer whose .encode
is completely broken) must degrade to word count, not crash."""
monkeypatch.chdir(tmp_path)
data_path = tmp_path / "pref.jsonl"
_write_jsonl(
data_path, [{"prompt": f"q{i}", "chosen": "aaaa", "rejected": "b"} for i in range(10)]
)
_patch_tokenizer(monkeypatch, _AlwaysFailEncodeTokenizer())
from soup_cli.cli import app
result = runner.invoke(
app, ["data", "lint", str(data_path), "--model", "fake/model"]
)
# Both "aaaa" and "b" are 1 word -> word-count fallback gives d=0.0 -> OK.
assert result.exit_code == 0, result.output
assert "length_bias" in result.output.lower()
# ===========================================================================
# Housekeeping rider — diagnose.py O_NOFOLLOW evidence loader (v0.71.25
# known-limitation (4)); ships alongside the doctor/lint feature work.
# ===========================================================================
class TestDiagnoseEvidenceHardening:
def test_source_uses_o_nofollow_and_fstat(self):
source = (_PROJECT_ROOT / "src" / "soup_cli" / "commands" / "diagnose.py").read_text(
encoding="utf-8"
)
assert "O_NOFOLLOW" in source
assert "os.fstat(" in source
def test_happy_path_still_loads(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
ev = tmp_path / "ev.json"
ev.write_text(json.dumps({"scores": {}}), encoding="utf-8")
from soup_cli.commands.diagnose import _load_evidence
payload = _load_evidence(str(ev))
assert payload == {"scores": {}}
def test_size_cap_still_enforced_via_fstat(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
ev = tmp_path / "ev.json"
ev.write_text("{}", encoding="utf-8")
from unittest.mock import MagicMock, patch
from soup_cli.commands.diagnose import _load_evidence
fake_stat = MagicMock(st_size=20 * 1024 * 1024)
with patch("os.fstat", return_value=fake_stat):
with pytest.raises(Exception, match="exceeds"):
_load_evidence(str(ev))
def test_not_a_dict_rejected(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
ev = tmp_path / "ev.json"
ev.write_text("[1, 2, 3]", encoding="utf-8")
from soup_cli.commands.diagnose import _load_evidence
with pytest.raises(Exception, match="JSON object"):
_load_evidence(str(ev))
def test_missing_file_raises(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
from soup_cli.commands.diagnose import _load_evidence
with pytest.raises(Exception, match="unreadable"):
_load_evidence(str(tmp_path / "missing.json"))
@pytest.mark.skipif(os.name == "nt", reason="POSIX symlink semantics")
def test_symlink_still_rejected(self, tmp_path, monkeypatch):
monkeypatch.chdir(tmp_path)
target = tmp_path / "real.json"
target.write_text("{}", encoding="utf-8")
link = tmp_path / "link.json"
os.symlink(target, link)
from soup_cli.commands.diagnose import _load_evidence
with pytest.raises(ValueError, match="symlink"):
_load_evidence(str(link))