diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index 864030c..4432c21 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -111,7 +111,7 @@ soup_cli/ templates/ - 17 built-in soup.yaml templates (YAML + manifest.json) with load_template loader (v0.39.0, +bco v0.40.0) ui/ - Web UI (FastAPI + HTML/JS SPA) -tests/ - Test suite (177 files, 6729 tests) +tests/ - Test suite (178 files, 7178 tests) examples/ - Real-world config examples and datasets ``` @@ -262,6 +262,7 @@ pytest tests/ --cov=soup_cli --cov-report=html | test_v0500_part_c.py | v0.50.0 Part C — Multi-turn agent rollout backend allowlist (art / ruler / nemo_gym / openenv); frozen `RolloutBackendSpec` + `MappingProxyType` immutability; `validate_rollout_backend` (bool rejected); `required_rollout_package` per-entry mapping; `launch_rollout` deferred stub; SoupConfig task-gate + mlx rejection (v0.50.0 Part C) | | test_v0500_part_d.py | v0.50.0 Part D — 7 stability/efficiency knobs (`ref_model_ema_alpha` / `replay_buffer_size` / `async_grpo_prefetch` / `tis_threshold` / `mask_truncated_completions` / `defer_rerolling` / `skip_zero_advantage` / `off_policy_mask_threshold`); explicit bool-rejection field_validator across all numeric fields (tdd-guide HIGH fix); `mask_truncated_completions` requires `tis_threshold` cross-validator; SoupConfig task-gate naming every offending field; `grpo_fp16` task-gate (code-review HIGH fix) (v0.50.0 Part D) | | test_v0500_part_e.py | v0.50.0 Part E — `task='prm'` (Process Reward Model) + `vision_grpo` flag; `validate_prm_compat` (data.format / modality / mlx gates); `validate_vision_grpo_compat` (task ∈ {grpo, ppo} / modality='vision' / non-mlx); `build_prm_trainer` deferred stub; SoupConfig integration with all rejection paths exercised (v0.50.0 Part E) | +| test_v0510.py | v0.51.0 Model Catalog Expansion + Alternative Model Hubs: Part E hubs.py (`SUPPORTED_HUBS` + `validate_hub_name` + `validate_hub_endpoint` SSRF parity / CRLF rejection / IPv6 mapped private rejected / IPv6 loopback ok / control chars; `resolve_endpoint` env-var override; `default_endpoint` + `endpoint_env_var` + `required_hub_package` + `is_hf` with bool guards; MappingProxyType immutability); TrainingConfig `hub` field (default + Literal accept + None reject + case-insensitive normalisation + YAML round-trip) + SoupConfig `_validate_hub_supported` (mlx + non-hf rejected; mlx + hf accepted; modelers + transformers accepted); Part D MULTIPACK_ARCHITECTURES extension (20 new arches parametrize + legacy preserved + exact count=38 + frozenset immutability); Parts A/B/C 26 new recipes (parametrize over every name × {get_recipe / RecipeMeta / SoupConfig load / yaml.safe_load / model id no null/whitespace/empty parts / max_length bounds / GRPO required fields}); baichuan-sft uses `hub: modelscope`; total recipe count >= 105 (v0.51.0) | (Note: the test-file table above covers v0.25.0–v0.35.0 + v0.47.0 + v0.48.0 + v0.49.0 + v0.50.0 only; full per-release table lives in `.claude/CLAUDE.md`.) diff --git a/README.md b/README.md index 35ef904..f7b2758 100644 --- a/README.md +++ b/README.md @@ -43,14 +43,13 @@ soup train Latest highlights only. Full history: [GitHub Releases](https://github.com/MakazhanAlpamys/Soup/releases). -**v0.50.0 — GRPO Plus (RL parity)**: 22 features across GRPO objective variants, long-context + memory-efficient RL, multi-turn agent rollout backends, stability/efficiency knobs, and PRM + Vision RL. Schema-only release closing the gap with unsloth and axolotl; live loss kernels and launchers land in v0.50.1. +**v0.51.0 — Model Catalog Expansion**: 26 new ready-made recipes covering 25 model families, plus alternative-hub support (ModelScope + Modelers / Openmind) for users in regions where HF Hub is unreachable. Closes the day-zero coverage gap with Unsloth. -- **7 GRPO objective variants.** New `training.grpo_variant`: `gspo` (Group Stabilized PO), `dapo` (Decoupled Advantage), `dr_grpo` (Doubly Robust), `bnpo` (Batch Normalized), `two_sided` (symmetric clipping — requires `grpo_delta`), `rft` (Reinforced Fine-Tuning), `standard`. Closed-allowlist validation, frozen `GRPOVariantSpec` metadata, explicit NaN/Inf + bool rejection on `grpo_delta`. -- **Long-context + memory-efficient RL.** New `training.long_context_grpo: true` (wires Tiled MLP from v0.56.0 when available; rejects `use_ring_attention=true`) + `training.vllm_sleep_mode: true` (between-rollouts vLLM standby — `transformers` / `unsloth` backends only). -- **Multi-turn agent rollout.** New `training.rollout_backend`: `art` (OpenPipe ART), `ruler`, `nemo_gym`, `openenv`. Per-entry `required_package` mapping; closed allowlist. -- **7 GRPO stability/efficiency knobs.** New `training.ref_model_ema_alpha` ((0, 1]), `replay_buffer_size` ([1, 1M]), `async_grpo_prefetch`, `tis_threshold` ((0, 100]) + paired `mask_truncated_completions`, `defer_rerolling`, `skip_zero_advantage`, `off_policy_mask_threshold` ([0, 1]) — every numeric field bool-rejected, every flag task-gated to `task='grpo'` with a single error message listing all offending fields. -- **PRM + Vision RL.** New top-level `task='prm'` (Process Reward Model / stepwise-supervised — paired with `data.format='prm'`) + new `training.vision_grpo: true` flag for VLM-RL on Qwen2-VL / Pixtral / InternVL (requires `modality='vision'` and `task ∈ {grpo, ppo}`). -- **+239 net new tests** (6490 → 6729). 5 sequential review agents (python-review, code-review, security-review, tdd-guide, verification-loop) each fed back HIGH/MEDIUM/LOW findings; every finding was fixed before commit (explicit NaN/Inf field-validator on `grpo_delta`, null-byte rejection on compat-helper backend/task strings, `use_ring_attention` bool guard, `grpo_fp16` task-gate, `vllm_sleep_mode` task-gate). +- **26 new recipes** across reasoning + agent (GPT-OSS 20B/120B, GLM 4.6 / 5, Kimi K2 / K2-Thinking GRPO, MiniMax-M2, QwQ-32B GRPO, QVQ-72B), small / specialist (Granite 4, Liquid LFM2, Cogito v2, Mistral Small 3 / Medium 3.5, Magistral / Devstral / Ministral, MedGemma, EmbeddingGemma), and vision / multimodal (LLaVA-Next, InternVL 3.5, Voxtral, Baichuan 2, Qwen-Image, DeepSeek-OCR, Paddle-OCR-VL). Browse them via `soup recipes list` / `soup recipes search `. Catalog grows 80 → 106. +- **Alternative model hubs.** New `training.hub: hf | modelscope | modelers` with full SSRF-hardened endpoint validators (parity with v0.29.0 `HF_ENDPOINT` policy — scheme allowlist, loopback-only HTTP, RFC1918 / link-local / cloud-metadata IP rejection, control-character rejection). `MODELSCOPE_ENDPOINT` / `MODELERS_ENDPOINT` env vars override the default hub URL the same way `HF_ENDPOINT` already does. Schema-only this release; live downloader + uploader wiring lands in v0.51.1. +- **20 new architectures in the multipack allowlist.** `MULTIPACK_ARCHITECTURES` grows 18 → 38 to enable FFD bin-packing on Granite, GLM, Kimi, MiniMax, QwQ, QVQ, GPT-OSS, Magistral, Devstral, Ministral, MedGemma, LFM2, Cogito, Hunyuan, Ernie, Yi, Baichuan, ChatGLM. +- **MLX backend cross-validator.** `backend: mlx` + `hub: modelscope` is now rejected at config-load with a distinct error message (`mlx-lm` only downloads from HF Hub) — prevents a silent runtime confusion. +- **+449 net new tests** (6729 → 7178). 4 sequential review agents (python-review, security-review, code-review, tdd-guide) each fed back HIGH/MEDIUM/LOW findings; every finding was fixed before commit (case-insensitive `hub` field-validator, MLX cross-validator, control-char rejection in endpoint validator, `is_hf` bool guard, exact-count multipack arch invariant, empty-component model-id check). ## Why Soup? @@ -1661,6 +1660,27 @@ soup runs clean --all --dry-run By default, the `clean` command operates in "surgical mode" (`--keep-weights`), deleting huge optimizer state files (`optimizer.pt`) from lesser checkpoints to save gigabytes, but keeping their lightweight evaluation weights just in case you want to load them later. +## Alternative Model Hubs + +Set `training.hub` in your `soup.yaml` to download from / push to a non-HuggingFace hub. Useful in regions where HF Hub is unreachable or blocked. + +```yaml +training: + hub: modelscope # or 'modelers' (Openmind), default 'hf' +``` + +Override the endpoint via env var: + +```bash +export MODELSCOPE_ENDPOINT=https://my-mirror.example.com +export MODELERS_ENDPOINT=https://corp-modelers.internal # HTTPS only for non-loopback +soup train --config soup.yaml +``` + +The endpoint validator follows the same SSRF rules as `HF_ENDPOINT`: only `http`/`https` schemes; plain HTTP allowed only for `localhost` / `127.0.0.1` / `::1`; private and link-local IPs (RFC1918, 169.254/16, etc.) rejected on plain HTTP. `backend: mlx` is incompatible with non-HF hubs (`mlx-lm` only downloads from HF Hub). + +The hub adapter is schema-only in this release; the live downloader and uploader land in v0.51.1. + ## Model Registry & Lineage Every fine-tune you ship should be reproducible. Soup's local registry (`~/.soup/registry.db`) tracks each entry by a content hash of its config + data + base model, plus lineage pointers to parent entries. diff --git a/SECURITY.md b/SECURITY.md index 7af6bad..92559f7 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -9,13 +9,13 @@ We provide security updates for the following versions: - **Versions older than 3 minor versions:** No support Example: -- v0.50.0 -- Full support (latest) +- v0.51.0 -- Full support (latest) +- v0.50.0 -- Full support - v0.49.0 -- Full support - v0.48.0 -- Full support -- v0.47.0 -- Full support -- v0.46.0 -- Bug-fix support only -- v0.45.0-v0.45.x -- Bug-fix support only -- v0.44.x and below -- No support +- v0.47.0 -- Bug-fix support only +- v0.46.0-v0.46.x -- Bug-fix support only +- v0.45.x and below -- No support ## Reporting a Vulnerability @@ -146,6 +146,7 @@ No known critical vulnerabilities in current releases. - **v0.32.0 — Training Stability & Auto-Tuning**: `--find-lr-output` containment via shared `utils/paths.is_under_cwd` (prevents writes outside cwd); `save_lr_finder_report` rejects NaN / Infinity floats in `lrs` / `losses` and serialises with `allow_nan=False` (keeps the report parser-safe); `compute_lr_schedule` rejects non-positive `start_lr`, inverted ranges, and `num_steps` outside `[2, 10_000]`; `pick_mixed_precision` rejects empty / null-byte / >200-char model names and resolves multi-version quirks (`qwen2.5` vs `qwen2`, `phi-3.5` vs `phi-3`) by longest-substring-first iteration so an added family can never accidentally make a more-specific entry dead code; `compute_warmup_steps` clamps to `[10, 1000]` with a `ratio==0.0` short-circuit matching HF Trainer's "no warmup" convention; `SpikeRecoveryStrategy` is `@dataclass(frozen=True)` (post-construction mutation cannot bypass validation), `max_attempts ∈ [1, 10]`, `lr_decay ∈ (0, 1)`, `min_lr > 0`; cross-validator `_validate_spike_recovery_requires_watchdog` rejects `loss_spike_recovery=true, loss_watchdog=false` at config-load (fails fast instead of never triggering); `convergence_window ∈ [5, 10_000]`, `convergence_rel_tol ∈ (0, 1]`, `recommend_action` reuses `detect_plateau` so plateau heuristic stays single-source-of-truth; `GradAccumMonitor.recommend()` caps doubled `accum` at `MAX_ACCUM=1024` so a runaway advisory loop cannot blow up DataLoader prefetch; `generate_config` validates BOTH the YAML output path AND the embedded `decisions["output"]` field via `is_under_cwd` (closes the gap where a crafted `decisions["output"]="../../etc"` would have silently propagated into the rendered YAML) - **v0.34.0 — Observability & Dev UX**: `.crash` bundle generator (`utils/crash.py`) recursively redacts `hf_*` / `sk-*` / `Bearer …` token-shaped strings in any captured `config` and metric tail before serialisation, so a `.crash` file shared on a public GitHub issue cannot leak credentials; `output_dir` is reduced to `os.path.basename` so `$HOME` doesn't leak; `write_crash_bundle` uses `os.path.realpath + commonpath` for cwd containment (Windows-safe; raises `ValueError` not `PermissionError` so callers cannot silently swallow with `except OSError`); filename appends `secrets.token_hex(4)` so two crashes in the same UTC second don't collide; bundle truncated to `MAX_BUNDLE_BYTES=1_000_000`. `train.py` crash-write surfaces failures to the user (no silent missing-bundle). `profiling.py` `resolve_trace_path` rejects empty / `.` / `..` / `/` / `\\` / null-byte `run_id` (closes the `output_dir/profiles/../trace.json` escape) and uses `os.path.realpath + is_under_cwd`; profiles dir is created only on successful torch import (no stale empty dirs on torch-less CI). `tracker.get_run` LIKE-prefix match escapes `%` / `_` / `\\` and uses `ESCAPE '\\'` so a crafted `run_id` cannot widen the match (mirrors v0.26.0 registry policy). Lazy schema migration (`_ensure_schema`) tolerates the "duplicate column" race when two CLI processes start simultaneously on a fresh DB (fork-based multi-GPU training, TUI auto-refresh). `runs.py show/replay/clean` switched user `run_id` rendering to `markup_escape` and switched `clean` containment from broken `Path.resolve() + relative_to()` to project-standard `os.path.realpath + is_under_cwd`. `tui_app.py` lazy-imports `ExperimentTracker` and `markup_escape`s every DB-sourced string before passing into Textual widgets so a crafted base_model / experiment_name cannot inject `[bold red]…[/]` markup. `run_cost.estimate_run_cost_usd` rejects `bool` in `num_gpus` (bool is a subclass of int — same defence as v0.30.0 `Candidate.__post_init__`); duration clamped to `[0, 1 year]`; unknown GPU returns `None` so callers render `—` instead of fabricating `$0.00`. `log_level.parse_log_level` rejects non-string + null-byte input. - **v0.33.0 — Live Wire**: RLVR `code_exec_reward` adds OS-level isolation (Linux best-effort `os.unshare(CLONE_NEWUSER|CLONE_NEWNET|CLONE_NEWPID)`, macOS `sandbox-exec` with default-deny `MACOS_SANDBOX_PROFILE` narrowed to a 3-name `mach-lookup` allowlist to prevent DNS / NSURLSession bypass of `(deny network*)`); `prune_checkpoints` switches to TOCTOU-safe `os.lstat + S_ISLNK` + `shutil.rmtree(onerror=_abort_on_symlink)` so a symlink encountered mid-walk aborts rather than escapes; `run_gate` wraps each task scorer in a typed `try/except` so backend failures produce `score=None, error=str(exc)` (never silent `score=1.0`); `_parse_judge_url` removes the bare `http://` catch-all (defence-in-depth after the Pydantic GateTask validator); `soup can run` requires `--yes` or explicit consent callback and raises `ValueError` (not `PermissionError`, which is an `OSError` subclass that broad `except` blocks would swallow); GGUF `rglob` result for ollama deploy is `realpath+commonpath` checked against extract_dir (prevents symlink escape from a crafted can); `DeployTarget.path` validator normalises mixed `\\`/`/` separators before splitting (closes a Windows `..` bypass); `CAN_FORMAT_VERSION` 1→2 (additive — v1 still loads); `soup can publish` validates `repo_id` via `utils/hf.validate_repo_id`, resolves token via `resolve_token`, sanitises commit messages (first-line, 200-char cap), uses HTTPS-only HfApi; `_write_spike_recovery_hint` adds `is_under_cwd` containment check on `args.output_dir` from raw HF `TrainingArguments`; `lookup_entry_by_output_dir` emits `ResourceWarning` when 1000-row scan limit is hit (no silent miss); `CrossDocCollator` no longer mutates input feature dicts (HF Dataset rows are cached and reused — mutation broke subsequent batches); `Candidate` rejects `bool` in `score`/`latency_ms` (was sneaking past `int` isinstance check); `evaluate_candidate` latency mean now divides by *completed* prompts (excludes crashed) so a broken candidate isn't artificially fast; `auto_quant.run_auto_quant_picker` soft-falls-back to highest-scored candidate when no candidate clears `min_score` (server still binds); `build_logits_processors` returns `[]` when neither `outlines` nor `lm-format-enforcer` is installed (server degrades to free-form rather than 500); MII server uses loopback-only CORS, max_tokens cap [1, 16384], stream rejection, generic 500 with no stack-trace leak; `os.execvp` auto-reexec uses list args (no shell), all forwarded flags pre-validated; `cleanup_extract_dir` uses `os.path.commonpath` (Windows-safe) instead of `startswith`; `_run_subprocess` catches `TimeoutExpired` and returns rc=124 (coreutils convention) instead of an unhandled traceback; new `eval_results` and `tensorrt` artifact kinds in `RegistryStore._VALID_KINDS` +- **v0.51.0 — Model Catalog Expansion + Alternative Model Hubs**: 5 release Parts. New `soup_cli/utils/hubs.py` ships closed allowlist `SUPPORTED_HUBS = frozenset({hf, modelscope, modelers})` + three `MappingProxyType`-wrapped registries (`_HUB_DEFAULT_ENDPOINTS` / `_HUB_ENDPOINT_ENV` / `_HUB_PACKAGE`) so the registry cannot be mutated at runtime (matches v0.36.0 `_REGISTRY` policy). `validate_hub_name` rejects non-string / bool / empty / null-byte / >32-char / unknown with case-insensitive normalisation (matches v0.41.0 `validate_optimizer_name` policy). `validate_hub_endpoint` is the SSRF kernel — full parity with v0.29.0 `utils/hf.resolve_endpoint`: scheme allowlist (`http`/`https` only), null-byte rejection, **control-character / CRLF rejection** added in v0.51.0 as a defence-in-depth review fix (defends against URL-as-HTTP-header injection if the URL ever flows into a raw HTTP client), `0.0.0.0` explicitly rejected, plain HTTP only for loopback `{localhost, 127.0.0.1, ::1}`, RFC1918 / link-local / cloud-metadata IPs (169.254.x) rejected via `ipaddress.ip_address` for plain HTTP. `resolve_endpoint(hub, *, env=None)` looks up the per-hub env var (`HF_ENDPOINT` / `MODELSCOPE_ENDPOINT` / `MODELERS_ENDPOINT`) and runs the override through `validate_hub_endpoint`; default endpoints are baked-in HTTPS URLs. `is_hf` rejects `bool` explicitly (review fix HIGH — bool is a subclass of int and would have silently fallen through `hub.lower() == "hf"` → `False`, which happens to be correct by accident but violates the contract; matches v0.30.0 `Candidate` / v0.34.0 `estimate_run_cost_usd` policy). `TrainingConfig.hub: Literal["hf","modelscope","modelers"]` field gets a `field_validator(mode="before")` `_normalize_hub` that delegates to `validate_hub_name` so `hub: HF` in YAML normalises to `"hf"` (review fix HIGH — first-cut had Pydantic Literal exact-match while `validate_hub_name` was case-insensitive, breaking the v0.41.0 `validate_optimizer_name` / v0.50.0 `grpo_variant` / `rollout_backend` policy of agreement between schema and shared validator). SoupConfig `_validate_hub_supported` cross-validator rejects `hub != 'hf'` on `backend == 'mlx'` with a distinct error message (review fix HIGH — `mlx-lm` only downloads from HF Hub; without this gate a `backend: mlx` + `hub: modelscope` config would silently pass schema load and fail at runtime with a confusing `mlx-lm` error). 26 new YAML recipes appended to `soup_cli/recipes/catalog.py` — every entry is exercised by `tests/test_v0510.py` via `load_config_from_string` round-trip + `yaml.safe_load` (no Python tags / no template injection / no credential leak in the YAML strings) + a `_no_null_or_whitespace` model-id check that rejects empty path components (review fix LOW — first-cut allowed `"/name"` leading-slash IDs to pass). Two non-`B` `size` strings (`"image"` / `"ocr"` / `"moe"` / `"medium"`) were normalised to `"N/A"` (review fix MEDIUM — `search_recipes(size=…)` would silently miss those entries, and the autopilot VRAM estimator could not parse them). Known limitations: (1) Live downloader / uploader / push integration deferred to v0.51.1 — `TrainingConfig.hub` schema lock-in ships now (Literal accept + MLX cross-validator + case-normalisation), but `soup data download --hub modelscope` and `soup push --hub modelers` still route through the existing HF Hub code path; the actual `modelscope-sdk` / `openmind-hub` adapters are the v0.51.1 deliverable. Same stub-then-live pattern as v0.27.0 MII / v0.37.0 multipack / v0.50.0 GRPO Plus. (2) Speculative / aspirational `base` model IDs in some Part A/C recipes — the catalog ships entries for `openai/gpt-oss-{20,120}b`, `THUDM/glm-5`, `Qwen/Qwen-Image`, `deepseek-ai/DeepSeek-OCR`, `PaddlePaddle/PaddleOCR-VL`, `google/embeddinggemma-300m` so users have ready-made recipes the moment those repos go live (matches the plan's "match Unsloth's day-zero coverage" directive). Recipes for not-yet-published repos will surface a clear HF Hub 404 when the user runs `soup train --recipe `. (3) DNS-resolved private hostnames not blocked — `validate_hub_endpoint` only rejects literal RFC1918 / link-local IP addresses; a hostname like `corp-proxy.internal` that DNS-resolves to a private IP is accepted at validation time (mirrors the v0.29.0 `HF_ENDPOINT` policy — DNS resolution is intentionally not performed in this local-tool threat model). (v0.51.0) - **v0.50.0 — GRPO Plus (RL parity)**: 22 features across 5 Parts shipped as schema-only (closed allowlists + Pydantic validators + NotImplementedError stubs for live wiring deferred to v0.50.1). All new validators follow the project's bool-rejection-before-int policy (matches v0.30.0 `Candidate`); closed-allowlist `validate_grpo_variant` / `validate_rollout_backend` reject non-string / bool / empty / null-byte / oversize / unknown inputs with actionable error messages and case-insensitive normalisation. `validate_grpo_delta` is bool-first / `math.isfinite` / `(0, 1]` bounded (matches v0.32.0 `save_lr_finder_report` / v0.41.0 Part B `lr_groups` policy). New `_VARIANT_METADATA` (Part A) and `_BACKEND_METADATA` (Part C) are `MappingProxyType`-wrapped frozen-dataclass registries (matches v0.36.0 `_REGISTRY` / v0.41.0 `_OPTIMIZER_PACKAGES` policy). Security-review fixes: (1) `grpo_delta` schema gets an explicit `field_validator(mode='after')` calling `math.isfinite` — Pydantic's `gt=0, le=1` bounds only incidentally reject NaN (since `NaN > 0` is False); the explicit validator prevents a future Pydantic change from regressing the guard. (2) `validate_long_context_grpo_compat` adds null-byte rejection on `task` AND `backend` strings + a `bool` guard on `use_ring_attention` (parity with `validate_grpo_variant` / `validate_rollout_backend`). (3) `validate_vllm_sleep_mode_compat` adds null-byte rejection on `backend`. Code-review HIGH fixes: (4) `_validate_grpo_stability_task_gate` now includes `grpo_fp16` in the GRPO-only-fields list — previously a user could silently set `grpo_fp16: true` on `task='sft'` and have it no-op. (5) `_validate_vllm_sleep_mode` now requires `task='grpo'` (sleep mode is a between-rollouts feature, meaningless on SFT) and rejects with a `task='grpo'` message. TDD-review HIGH fixes: (6) new `_reject_bool_on_grpo_numerics` field_validator on every Part D numeric field + `grpo_delta` explicitly rejects `bool` before Pydantic's `True→1` coercion (matches v0.30.0 / v0.41.0 Part B / v0.43.0 Part B policy). Known limitations: (1) Every live loss kernel / launcher (`apply_variant_loss`, `apply_vllm_sleep_mode`, `launch_rollout`, `build_prm_trainer`) raises `NotImplementedError` with explicit `v0.50.1` markers — same stub-then-live pattern as v0.27.0 MII / v0.37.0 multipack / v0.41.0 LLaMA Pro / v0.45.0 plugins / v0.48.0 curriculum / v0.49.0 LongLoRA. (2) `long_context_grpo` requires Tiled MLP (v0.56.0 Part A) to actually run; the schema gate ships now so v0.50.0 configs are stable. (3) `vision_grpo=true` does not check whether the base model is actually a VLM — upstream trainer surfaces that error loudly. (4) The 7 stability knobs schema-validate but none are wired into a live callback in this release; `replay_buffer_size`, `defer_rerolling`, and `skip_zero_advantage` are pure schema lock-ins. (v0.50.0) - **v0.49.0 — Long Context & Architecture**: 4 release Parts ship YaRN RoPE scaling, Dynamic NTK hardening, LongLoRA S² shifted-sparse attention (schema-only gate), and full Llama 3.1 NTK-aware scaling. Security review hardened the public boundary of `soup_cli/utils/long_context.py::get_rope_scaling_config` — `target_length` / `original_length` / `yarn_factor` now reject `bool` / NaN / Inf / non-positive at entry so a direct caller bypassing Pydantic cannot emit `{factor: NaN}` into HF model configs (matches v0.30.0 `Candidate` / v0.34.0 `estimate_run_cost_usd` / v0.41.0 Part B `lr_groups` policy). `scale_inv_freq_llama3` rejects `bool` on every numeric parameter (`inv_freq` / `scale_factor` / `low_freq_factor` / `high_freq_factor` / `old_context_len`) — first-cut only guarded `inv_freq`. `yarn_get_mscale` now raises on non-finite input (review fix LOW — first-cut silently clamped NaN/Inf to identity, hiding the misconfig from callers). `detect_llama3_rope_in_config` uses explicit `is None` instead of the `or` idiom when probing `rope.get("type")` — falsy-but-set values no longer silently fall through to `rope_type` (matches v0.40.6 review-fix policy). New `soup_cli/utils/longlora.py` ships `is_llama_model` with a word-boundary regex `(?:^|[^a-z0-9])(?:code)?-?llama(?:-?\d+(?:\.\d+)?)?(?:[^a-z0-9]|$)` — substring `"llama"` inside an unrelated identifier (e.g. `my-llama-style-finetune`) does NOT match; null-byte rejection on `model_name` (matches v0.39.0 `is_gemma4_model` / v0.44.0 `is_llama4_model` policy); 512-char cap returns `False` rather than raising (bounded scan time). `_LLAMA_REGEX` has no nested quantifiers / overlapping alternation and is ReDoS-bounded by the 512-char cap. `validate_longlora_compat` emits distinct error messages per failure mode (mlx vs other non-transformers backends — matches v0.34.0 review-fix policy on distinct actionable rejections). `TrainingConfig` `field_validator(mode='before')` rejects `bool` on the four yarn fields before Pydantic's `gt`/`le` coercion silently treats `True` as `1.0`. SoupConfig `_validate_longlora_compat` invokes `validate_longlora_compat` at config load so a misconfigured `soup.yaml` fails fast with an actionable message rather than silently no-opping at trainer construction time (mirrors v0.39.0 ReLoRA / v0.48.0 curriculum_dynamic schema-gate policy). Known limitations: (1) LongLoRA live `LlamaAttention.forward` override deferred to v0.49.1 — `apply_longlora_forward_override` raises `NotImplementedError` with the v0.49.1 marker; the schema gate ships now so misconfigured runs cannot reach trainer construction. (2) LongLoRA architecture allowlist is Llama 1/2/3.x + CodeLlama only; Mistral / Qwen / Phi expansion tracked for v0.49.1+. (3) Llama 3.1 NTK auto-detect helper `detect_llama3_rope_in_config` ships but is not yet wired into `apply_long_context_config`; trainer wiring can pick it up when needed. diff --git a/pyproject.toml b/pyproject.toml index 143ec03..3d11bed 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "hatchling.build" [project] name = "soup-cli" -version = "0.50.0" +version = "0.51.0" description = "Fine-tune LLMs in one command. No SSH, no config hell." readme = "README.md" license = "Apache-2.0" diff --git a/soup_cli/__init__.py b/soup_cli/__init__.py index 1092d06..5bfc1be 100644 --- a/soup_cli/__init__.py +++ b/soup_cli/__init__.py @@ -1,3 +1,3 @@ """Soup CLI — Fine-tune LLMs in one command.""" -__version__ = "0.50.0" +__version__ = "0.51.0" diff --git a/soup_cli/config/schema.py b/soup_cli/config/schema.py index 5284349..fd5ea9f 100644 --- a/soup_cli/config/schema.py +++ b/soup_cli/config/schema.py @@ -1048,6 +1048,32 @@ class TrainingConfig(BaseModel): "rollout wiring deferred to v0.50.1." ), ) + # v0.51.0 Part E — alternative model hubs (ModelScope / Modelers) + hub: Literal["hf", "modelscope", "modelers"] = Field( + default="hf", + description=( + "Model hub for downloads + pushes. 'hf' (default), 'modelscope' " + "(China-hosted; mirrors most Llama/Qwen/etc.), 'modelers' " + "(Openmind hub). Schema-only in v0.51.0; live downloader / " + "uploader wiring deferred to v0.51.1." + ), + ) + + @field_validator("hub", mode="before") + @classmethod + def _normalize_hub(cls, v): + """v0.51.0 Part E review fix — accept any case (HF / Modelscope / + MODELERS) and normalise to lowercase before the Literal check. + Mirrors the v0.41.0 ``optimizer`` / v0.50.0 ``grpo_variant`` / + ``rollout_backend`` policy of running the shared ``validate_*`` + helper at ``mode='before'`` so the public schema and the runtime + validator agree on what's accepted. + """ + # Lazy-import to avoid a hard dep cycle at module load. + from soup_cli.utils.hubs import validate_hub_name + if v is None: + return v + return validate_hub_name(v) # PPO-specific ppo_epochs: int = Field( default=4, ge=1, description="Number of PPO optimization epochs per batch" @@ -2183,6 +2209,27 @@ class SoupConfig(BaseModel): ) return self + @model_validator(mode="after") + def _validate_hub_supported(self) -> "SoupConfig": + """v0.51.0 Part E — ``hub`` other than ``hf`` requires a non-mlx + backend. + + ``mlx-lm`` has no ModelScope/Modelers download integration, so a + config that pairs ``backend: mlx`` + ``hub: modelscope`` would fail + at runtime with a confusing ``mlx-lm`` error. Reject loudly at + config-load with a distinct message (matches v0.34.0 review-fix + policy). + """ + if self.training.hub == "hf": + return self + if self.backend == "mlx": + raise ValueError( + f"hub={self.training.hub!r} is not supported on " + "backend=mlx (mlx-lm only downloads from HF Hub). " + "Use hub='hf' on the mlx backend." + ) + return self + @model_validator(mode="after") def _validate_rollout_backend(self) -> "SoupConfig": """v0.50.0 Part C — ``rollout_backend`` requires task='grpo' and a diff --git a/soup_cli/recipes/catalog.py b/soup_cli/recipes/catalog.py index 1094921..572b245 100644 --- a/soup_cli/recipes/catalog.py +++ b/soup_cli/recipes/catalog.py @@ -2465,6 +2465,782 @@ training: dpo_beta: 0.1 gradient_checkpointing: true +output: ./output +""", + ), + # ------------------------------------------------------------------ + # v0.51.0 Part A — Reasoning + agent (~5 model families) + # ------------------------------------------------------------------ + "gpt-oss-20b-sft": RecipeMeta( + model="openai/gpt-oss-20b", + task="sft", + size="20B", + tags=("gpt-oss", "openai", "reasoning", "agent"), + description="GPT-OSS 20B SFT (reasoning_effort=medium)", + yaml_str="""\ +base: openai/gpt-oss-20b +task: sft + +data: + train: ./data/train.jsonl + format: auto + max_length: 4096 + +training: + epochs: 3 + lr: 1e-4 + batch_size: auto + lora: + r: 16 + alpha: 32 + target_modules: auto + quantization: 4bit + +output: ./output +""", + ), + "gpt-oss-120b-sft": RecipeMeta( + model="openai/gpt-oss-120b", + task="sft", + size="120B", + tags=("gpt-oss", "openai", "reasoning", "large"), + description="GPT-OSS 120B SFT (multi-GPU recommended)", + yaml_str="""\ +base: openai/gpt-oss-120b +task: sft + +data: + train: ./data/train.jsonl + format: auto + max_length: 4096 + +training: + epochs: 1 + lr: 5e-5 + batch_size: 1 + gradient_accumulation_steps: 32 + lora: + r: 16 + alpha: 32 + target_modules: auto + quantization: 4bit + gradient_checkpointing: true + +output: ./output +""", + ), + "glm-4.6-sft": RecipeMeta( + model="THUDM/glm-4.6", + task="sft", + size="9B", + tags=("glm", "thudm", "chat", "instruction"), + description="GLM 4.6 instruction tuning with LoRA", + yaml_str="""\ +base: THUDM/glm-4.6 +task: sft + +data: + train: ./data/train.jsonl + format: auto + max_length: 4096 + +training: + epochs: 3 + lr: 2e-4 + batch_size: auto + lora: + r: 16 + alpha: 32 + target_modules: auto + quantization: 4bit + +output: ./output +""", + ), + "glm-5-sft": RecipeMeta( + model="THUDM/glm-5", + task="sft", + size="9B", + tags=("glm", "thudm", "chat", "next-gen"), + description="GLM 5 SFT (next-gen GLM family)", + yaml_str="""\ +base: THUDM/glm-5 +task: sft + +data: + train: ./data/train.jsonl + format: auto + max_length: 8192 + +training: + epochs: 3 + lr: 2e-4 + batch_size: auto + lora: + r: 16 + alpha: 32 + target_modules: auto + quantization: 4bit + +output: ./output +""", + ), + "kimi-k2-sft": RecipeMeta( + model="moonshotai/Kimi-K2", + task="sft", + size="N/A", + tags=("kimi", "moonshot", "moe", "long-context"), + description="Kimi K2 SFT (Moonshot MoE, long-context-aware)", + yaml_str="""\ +base: moonshotai/Kimi-K2 +task: sft + +data: + train: ./data/train.jsonl + format: auto + max_length: 8192 + +training: + epochs: 2 + lr: 1e-4 + batch_size: auto + lora: + r: 16 + alpha: 32 + target_modules: auto + quantization: 4bit + moe_lora: true + +output: ./output +""", + ), + "kimi-k2-thinking-grpo": RecipeMeta( + model="moonshotai/Kimi-K2-Thinking", + task="grpo", + size="N/A", + tags=("kimi", "moonshot", "thinking", "grpo", "reasoning"), + description="Kimi K2 Thinking GRPO reasoning", + yaml_str="""\ +base: moonshotai/Kimi-K2-Thinking +task: grpo + +data: + train: ./data/reasoning_prompts.jsonl + format: auto + max_length: 4096 + +training: + epochs: 1 + lr: 5e-6 + batch_size: 1 + gradient_accumulation_steps: 8 + lora: + r: 16 + alpha: 32 + target_modules: auto + quantization: 4bit + grpo_beta: 0.04 + num_generations: 4 + reward_fn: accuracy + verifiable_domain: math + +output: ./output +""", + ), + "minimax-m2-sft": RecipeMeta( + model="MiniMaxAI/MiniMax-M2", + task="sft", + size="9B", + tags=("minimax", "chat", "instruction"), + description="MiniMax M2 SFT instruction tuning", + yaml_str="""\ +base: MiniMaxAI/MiniMax-M2 +task: sft + +data: + train: ./data/train.jsonl + format: auto + max_length: 4096 + +training: + epochs: 3 + lr: 2e-4 + batch_size: auto + lora: + r: 16 + alpha: 32 + target_modules: auto + quantization: 4bit + +output: ./output +""", + ), + "qwq-32b-grpo": RecipeMeta( + model="Qwen/QwQ-32B", + task="grpo", + size="32B", + tags=("qwen", "qwq", "reasoning", "grpo"), + description="QwQ 32B GRPO reasoning training", + yaml_str="""\ +base: Qwen/QwQ-32B +task: grpo + +data: + train: ./data/reasoning_prompts.jsonl + format: auto + max_length: 4096 + +training: + epochs: 1 + lr: 5e-6 + batch_size: 1 + gradient_accumulation_steps: 16 + lora: + r: 16 + alpha: 32 + target_modules: auto + quantization: 4bit + grpo_beta: 0.04 + num_generations: 4 + reward_fn: accuracy + verifiable_domain: math + gradient_checkpointing: true + +output: ./output +""", + ), + "qvq-72b-sft": RecipeMeta( + model="Qwen/QVQ-72B-Preview", + task="sft", + size="72B", + tags=("qwen", "qvq", "vision", "reasoning"), + description="QVQ 72B vision-reasoning SFT", + yaml_str="""\ +base: Qwen/QVQ-72B-Preview +task: sft +modality: vision + +data: + train: ./data/vision_train.jsonl + format: llava + image_dir: ./data/images + max_length: 4096 + +training: + epochs: 1 + lr: 1e-5 + batch_size: 1 + gradient_accumulation_steps: 16 + lora: + r: 16 + alpha: 32 + target_modules: auto + quantization: 4bit + gradient_checkpointing: true + +output: ./output +""", + ), + # ------------------------------------------------------------------ + # v0.51.0 Part B — Small / edge / specialist (~6 model families) + # ------------------------------------------------------------------ + "granite-4-sft": RecipeMeta( + model="ibm-granite/granite-4.0-tiny-base", + task="sft", + size="3B", + tags=("granite", "ibm", "small", "instruction"), + description="IBM Granite 4.0 tiny SFT", + yaml_str="""\ +base: ibm-granite/granite-4.0-tiny-base +task: sft + +data: + train: ./data/train.jsonl + format: auto + max_length: 2048 + +training: + epochs: 3 + lr: 2e-4 + batch_size: auto + lora: + r: 16 + alpha: 32 + target_modules: auto + quantization: 4bit + +output: ./output +""", + ), + "lfm2-sft": RecipeMeta( + model="LiquidAI/LFM2-1.2B", + task="sft", + size="1.2B", + tags=("liquid", "lfm2", "small", "edge"), + description="Liquid LFM2 1.2B SFT (edge-optimised)", + yaml_str="""\ +base: LiquidAI/LFM2-1.2B +task: sft + +data: + train: ./data/train.jsonl + format: auto + max_length: 2048 + +training: + epochs: 3 + lr: 2e-4 + batch_size: auto + lora: + r: 16 + alpha: 32 + target_modules: auto + quantization: 4bit + +output: ./output +""", + ), + "cogito-v2-sft": RecipeMeta( + model="deepcogito/cogito-v2-preview", + task="sft", + size="14B", + tags=("cogito", "deepcogito", "instruction"), + description="Cogito v2 preview SFT", + yaml_str="""\ +base: deepcogito/cogito-v2-preview +task: sft + +data: + train: ./data/train.jsonl + format: auto + max_length: 4096 + +training: + epochs: 3 + lr: 2e-4 + batch_size: auto + lora: + r: 16 + alpha: 32 + target_modules: auto + quantization: 4bit + +output: ./output +""", + ), + "mistral-small-3-sft": RecipeMeta( + model="mistralai/Mistral-Small-3-24B-Instruct", + task="sft", + size="24B", + tags=("mistral", "small", "instruction"), + description="Mistral Small 3 24B SFT", + yaml_str="""\ +base: mistralai/Mistral-Small-3-24B-Instruct +task: sft + +data: + train: ./data/train.jsonl + format: auto + max_length: 4096 + +training: + epochs: 3 + lr: 2e-4 + batch_size: auto + lora: + r: 16 + alpha: 32 + target_modules: auto + quantization: 4bit + +output: ./output +""", + ), + "mistral-medium-3-5-sft": RecipeMeta( + model="mistralai/Mistral-Medium-3.5", + task="sft", + size="N/A", + tags=("mistral", "medium", "instruction", "large"), + description="Mistral Medium 3.5 SFT", + yaml_str="""\ +base: mistralai/Mistral-Medium-3.5 +task: sft + +data: + train: ./data/train.jsonl + format: auto + max_length: 4096 + +training: + epochs: 2 + lr: 1e-4 + batch_size: auto + lora: + r: 16 + alpha: 32 + target_modules: auto + quantization: 4bit + gradient_checkpointing: true + +output: ./output +""", + ), + "magistral-small-sft": RecipeMeta( + model="mistralai/Magistral-Small", + task="sft", + size="24B", + tags=("mistral", "magistral", "reasoning", "instruction"), + description="Magistral Small reasoning SFT", + yaml_str="""\ +base: mistralai/Magistral-Small +task: sft + +data: + train: ./data/reasoning_train.jsonl + format: auto + max_length: 4096 + +training: + epochs: 3 + lr: 2e-4 + batch_size: auto + lora: + r: 16 + alpha: 32 + target_modules: auto + quantization: 4bit + +output: ./output +""", + ), + "devstral-sft": RecipeMeta( + model="mistralai/Devstral-Small", + task="sft", + size="24B", + tags=("mistral", "devstral", "code", "agent"), + description="Devstral Small code/agent SFT", + yaml_str="""\ +base: mistralai/Devstral-Small +task: sft + +data: + train: ./data/code_train.jsonl + format: auto + max_length: 8192 + +training: + epochs: 3 + lr: 2e-4 + batch_size: auto + lora: + r: 16 + alpha: 32 + target_modules: auto + quantization: 4bit + +output: ./output +""", + ), + "ministral-sft": RecipeMeta( + model="mistralai/Ministral-8B-Instruct-2410", + task="sft", + size="8B", + tags=("mistral", "ministral", "small", "instruction"), + description="Ministral 8B SFT", + yaml_str="""\ +base: mistralai/Ministral-8B-Instruct-2410 +task: sft + +data: + train: ./data/train.jsonl + format: auto + max_length: 4096 + +training: + epochs: 3 + lr: 2e-4 + batch_size: auto + lora: + r: 16 + alpha: 32 + target_modules: auto + quantization: 4bit + +output: ./output +""", + ), + "medgemma-sft": RecipeMeta( + model="google/medgemma-4b-it", + task="sft", + size="4B", + tags=("gemma", "medgemma", "medical", "domain"), + description="MedGemma 4B medical SFT", + yaml_str="""\ +base: google/medgemma-4b-it +task: sft + +data: + train: ./data/medical_train.jsonl + format: auto + max_length: 2048 + +training: + epochs: 3 + lr: 2e-4 + batch_size: auto + lora: + r: 16 + alpha: 32 + target_modules: auto + quantization: 4bit + +output: ./output +""", + ), + "embedding-gemma-sft": RecipeMeta( + model="google/embeddinggemma-300m", + task="embedding", + size="300M", + tags=("gemma", "embedding", "small", "sentence"), + description="EmbeddingGemma 300M sentence-embedding SFT", + yaml_str="""\ +base: google/embeddinggemma-300m +task: embedding + +data: + train: ./data/embedding_train.jsonl + format: embedding + max_length: 512 + +training: + epochs: 3 + lr: 2e-5 + batch_size: auto + lora: + r: 8 + alpha: 16 + target_modules: auto + quantization: 4bit + embedding_loss: cosine + +output: ./output +""", + ), + # ------------------------------------------------------------------ + # v0.51.0 Part C — Vision + multimodal (~7 model families) + # ------------------------------------------------------------------ + "llava-next-sft": RecipeMeta( + model="llava-hf/llava-v1.6-mistral-7b-hf", + task="sft", + size="7B", + tags=("llava", "llava-next", "vision", "multimodal"), + description="LLaVA-Next 7B vision SFT", + yaml_str="""\ +base: llava-hf/llava-v1.6-mistral-7b-hf +task: sft +modality: vision + +data: + train: ./data/vision_train.jsonl + format: llava + image_dir: ./data/images + max_length: 2048 + +training: + epochs: 3 + lr: 1e-5 + batch_size: auto + lora: + r: 16 + alpha: 32 + target_modules: auto + quantization: 4bit + gradient_checkpointing: true + +output: ./output +""", + ), + "internvl-3-5-sft": RecipeMeta( + model="OpenGVLab/InternVL3-5", + task="sft", + size="8B", + tags=("internvl", "vision", "multimodal", "opengvlab"), + description="InternVL 3.5 vision SFT", + yaml_str="""\ +base: OpenGVLab/InternVL3-5 +task: sft +modality: vision + +data: + train: ./data/vision_train.jsonl + format: llava + image_dir: ./data/images + max_length: 4096 + +training: + epochs: 3 + lr: 1e-5 + batch_size: auto + lora: + r: 16 + alpha: 32 + target_modules: auto + quantization: 4bit + gradient_checkpointing: true + +output: ./output +""", + ), + "voxtral-sft": RecipeMeta( + model="mistralai/Voxtral-Mini-3B", + task="sft", + size="3B", + tags=("mistral", "voxtral", "audio", "multimodal"), + description="Voxtral Mini 3B audio SFT", + yaml_str="""\ +base: mistralai/Voxtral-Mini-3B +task: sft +modality: audio + +data: + train: ./data/audio_train.jsonl + format: audio + audio_dir: ./data/audio + max_length: 2048 + +training: + epochs: 3 + lr: 1e-5 + batch_size: auto + lora: + r: 16 + alpha: 32 + target_modules: auto + quantization: 4bit + gradient_checkpointing: true + +output: ./output +""", + ), + "baichuan-sft": RecipeMeta( + model="baichuan-inc/Baichuan2-13B-Chat", + task="sft", + size="13B", + tags=("baichuan", "chinese", "instruction"), + description="Baichuan 2 13B chat SFT", + yaml_str="""\ +base: baichuan-inc/Baichuan2-13B-Chat +task: sft + +data: + train: ./data/train.jsonl + format: auto + max_length: 4096 + +training: + epochs: 3 + lr: 2e-4 + batch_size: auto + lora: + r: 16 + alpha: 32 + target_modules: auto + quantization: 4bit + hub: modelscope + +output: ./output +""", + ), + "qwen-image-sft": RecipeMeta( + model="Qwen/Qwen-Image", + task="sft", + size="N/A", + tags=("qwen", "image", "image-output", "multimodal"), + description="Qwen-Image image-output multimodal SFT", + yaml_str="""\ +base: Qwen/Qwen-Image +task: sft +modality: vision + +data: + train: ./data/image_train.jsonl + format: llava + image_dir: ./data/images + max_length: 4096 + +training: + epochs: 3 + lr: 1e-5 + batch_size: auto + lora: + r: 16 + alpha: 32 + target_modules: auto + quantization: 4bit + gradient_checkpointing: true + +output: ./output +""", + ), + "deepseek-ocr-sft": RecipeMeta( + model="deepseek-ai/DeepSeek-OCR", + task="sft", + size="N/A", + tags=("deepseek", "ocr", "vision", "specialised"), + description="DeepSeek-OCR vision OCR SFT", + yaml_str="""\ +base: deepseek-ai/DeepSeek-OCR +task: sft +modality: vision + +data: + train: ./data/ocr_train.jsonl + format: llava + image_dir: ./data/images + max_length: 4096 + +training: + epochs: 3 + lr: 1e-5 + batch_size: auto + lora: + r: 16 + alpha: 32 + target_modules: auto + quantization: 4bit + gradient_checkpointing: true + +output: ./output +""", + ), + "paddle-ocr-sft": RecipeMeta( + model="PaddlePaddle/PaddleOCR-VL", + task="sft", + size="N/A", + tags=("paddle", "ocr", "vision", "specialised"), + description="Paddle-OCR-VL OCR SFT", + yaml_str="""\ +base: PaddlePaddle/PaddleOCR-VL +task: sft +modality: vision + +data: + train: ./data/ocr_train.jsonl + format: llava + image_dir: ./data/images + max_length: 4096 + +training: + epochs: 3 + lr: 1e-5 + batch_size: auto + lora: + r: 16 + alpha: 32 + target_modules: auto + quantization: 4bit + gradient_checkpointing: true + output: ./output """, ), diff --git a/soup_cli/utils/hubs.py b/soup_cli/utils/hubs.py new file mode 100644 index 0000000..631e7de --- /dev/null +++ b/soup_cli/utils/hubs.py @@ -0,0 +1,206 @@ +"""v0.51.0 Part E — Alternative model hubs (ModelScope / Modelers). + +Schema-only support for selecting a non-HF model hub for downloads + pushes. +Each hub gets a closed allowlist + SSRF-hardened endpoint validator that +mirrors the v0.29.0 ``HF_ENDPOINT`` policy in ``utils/hf.py``: + +* scheme allowlist (http/https only) +* null-byte rejection +* ``0.0.0.0`` explicitly rejected +* plain HTTP only permitted for loopback hosts + (``localhost`` / ``127.0.0.1`` / ``::1``) +* private / link-local / cloud-metadata IPs (RFC1918, 169.254.x) rejected + for plain HTTP. + +Live download / upload wiring is deferred to v0.51.1 — this module ships the +schema lock-in (``hub`` Literal on ``TrainingConfig``) plus the validators +that the live wiring will call. +""" +from __future__ import annotations + +import ipaddress +import os +from types import MappingProxyType +from typing import Mapping +from urllib.parse import urlparse + +# Closed allowlist of supported hubs. Wrapped in MappingProxyType so callers +# cannot mutate the registry at runtime (matches v0.36.0 _REGISTRY policy). +SUPPORTED_HUBS: frozenset[str] = frozenset({"hf", "modelscope", "modelers"}) + +_HUB_DEFAULT_ENDPOINTS: Mapping[str, str] = MappingProxyType({ + "hf": "https://huggingface.co", + "modelscope": "https://modelscope.cn", + "modelers": "https://modelers.cn", +}) + +# Per-hub env var that overrides the default endpoint (mirrors HF_ENDPOINT). +_HUB_ENDPOINT_ENV: Mapping[str, str] = MappingProxyType({ + "hf": "HF_ENDPOINT", + "modelscope": "MODELSCOPE_ENDPOINT", + "modelers": "MODELERS_ENDPOINT", +}) + +# Per-hub pip-install hint, surfaced when the live downloader complains. +_HUB_PACKAGE: Mapping[str, str] = MappingProxyType({ + "hf": "huggingface-hub", + "modelscope": "modelscope", + "modelers": "openmind-hub", +}) + +_LOOPBACK_HOSTS: frozenset[str] = frozenset({"localhost", "127.0.0.1", "::1"}) + +_MAX_HUB_NAME_LEN: int = 32 + + +def validate_hub_name(name: str) -> str: + """Validate ``name`` against ``SUPPORTED_HUBS`` and return canonical form. + + Mirrors v0.41.0 ``validate_optimizer_name`` policy: + rejects non-string / bool / empty / null-byte / oversize / unknown, and + lower-cases the input for deterministic lookup. + """ + if isinstance(name, bool): + raise TypeError(f"hub name must not be bool, got {name!r}") + if not isinstance(name, str): + raise TypeError(f"hub name must be str, got {type(name).__name__}") + if not name: + raise ValueError("hub name must be non-empty") + if "\x00" in name: + raise ValueError("hub name must not contain null bytes") + if len(name) > _MAX_HUB_NAME_LEN: + raise ValueError( + f"hub name too long (max {_MAX_HUB_NAME_LEN} chars)" + ) + canonical = name.lower() + if canonical not in SUPPORTED_HUBS: + supported = ", ".join(sorted(SUPPORTED_HUBS)) + raise ValueError( + f"hub {name!r} not supported. Supported: {supported}" + ) + return canonical + + +def required_hub_package(hub: str) -> str | None: + """Return the pip-installable package name for ``hub``, or ``None``. + + Non-string / unknown returns ``None`` (no advisory available). + """ + if not isinstance(hub, str): + return None + return _HUB_PACKAGE.get(hub.lower()) + + +def default_endpoint(hub: str) -> str: + """Return the canonical default endpoint URL for ``hub``. + + Raises ``ValueError`` if ``hub`` is not in :data:`SUPPORTED_HUBS`. + """ + canonical = validate_hub_name(hub) + return _HUB_DEFAULT_ENDPOINTS[canonical] + + +def endpoint_env_var(hub: str) -> str: + """Return the env-var name that overrides the default endpoint.""" + canonical = validate_hub_name(hub) + return _HUB_ENDPOINT_ENV[canonical] + + +def _is_private_or_link_local(host: str) -> bool: + """Whether ``host`` is a private / link-local / loopback IP. + + DNS resolution intentionally not performed (matches v0.29.0 hf.py policy). + """ + try: + addr = ipaddress.ip_address(host) + except ValueError: + return False + return addr.is_private or addr.is_link_local or addr.is_loopback + + +def validate_hub_endpoint(endpoint: str, *, hub: str | None = None) -> str: + """SSRF-hardened endpoint validator. Returns the stripped endpoint. + + Mirrors ``utils/hf.resolve_endpoint`` exactly so all three hubs share the + same security posture. Raises ``ValueError`` on bad input. + + Args: + endpoint: candidate URL. + hub: optional hub name, threaded into the error messages. + """ + label = (hub or "hub_endpoint").strip() or "hub_endpoint" + + if isinstance(endpoint, bool): + raise TypeError(f"{label} must not be bool, got {endpoint!r}") + if not isinstance(endpoint, str): + raise TypeError( + f"{label} must be str, got {type(endpoint).__name__}" + ) + if not endpoint: + raise ValueError(f"{label} must be a non-empty string") + if "\x00" in endpoint: + raise ValueError(f"{label} must not contain null bytes") + # Defence-in-depth: control characters (CR/LF/etc.) inside a URL would + # be a CRLF-injection hazard if the URL ever flowed into a raw HTTP + # client. Reject them here even though urlparse silently strips them. + if any(ord(c) < 0x20 for c in endpoint): + raise ValueError(f"{label} must not contain control characters") + + stripped = endpoint.rstrip("/") + parsed = urlparse(stripped) + if parsed.scheme not in ("http", "https"): + raise ValueError( + f"{label} must use http/https scheme, got: {parsed.scheme!r}" + ) + if not parsed.netloc: + raise ValueError(f"{label} is missing a host") + + host = parsed.hostname or "" + if host == "0.0.0.0": + raise ValueError( + f"{label} 0.0.0.0 is ambiguous; use 127.0.0.1 or localhost" + ) + if parsed.scheme == "http" and host not in _LOOPBACK_HOSTS: + if _is_private_or_link_local(host): + raise ValueError( + f"{label} plain HTTP is only allowed for loopback " + f"(localhost / 127.0.0.1 / ::1); private/link-local hosts " + f"require HTTPS" + ) + raise ValueError( + f"{label} for remote hosts must use HTTPS " + f"(localhost HTTP allowed)" + ) + return stripped + + +def resolve_endpoint(hub: str, *, env: Mapping[str, str] | None = None) -> str: + """Return the active endpoint for ``hub`` after env-override + validation. + + Looks up the per-hub env var (e.g. ``MODELSCOPE_ENDPOINT``) — if set, runs + it through :func:`validate_hub_endpoint` and returns the stripped URL. + Otherwise returns :func:`default_endpoint`. + + The default endpoints are baked-in HTTPS URLs so they don't need + re-validation on every call. + """ + canonical = validate_hub_name(hub) + source = env if env is not None else os.environ + raw = source.get(_HUB_ENDPOINT_ENV[canonical]) + if not raw: + return _HUB_DEFAULT_ENDPOINTS[canonical] + return validate_hub_endpoint(raw, hub=canonical) + + +def is_hf(hub: str) -> bool: + """Convenience: True iff ``hub`` canonicalises to ``'hf'``. + + Rejects ``bool`` explicitly (matches v0.30.0 ``Candidate`` / + v0.34.0 ``estimate_run_cost_usd`` policy) so a stray ``True`` cannot + silently pretend to be a hub name. + """ + if isinstance(hub, bool): + return False + if not isinstance(hub, str): + return False + return hub.lower() == "hf" diff --git a/soup_cli/utils/multipack_sampler.py b/soup_cli/utils/multipack_sampler.py index e574915..a15893e 100644 --- a/soup_cli/utils/multipack_sampler.py +++ b/soup_cli/utils/multipack_sampler.py @@ -43,6 +43,27 @@ MULTIPACK_ARCHITECTURES: frozenset[str] = frozenset({ "FalconForCausalLM", "StableLmForCausalLM", "SmolLM2ForCausalLM", + # v0.51.0 Part D — new model families from the catalog expansion. + "GraniteForCausalLM", + "GraniteMoeForCausalLM", + "Glm4ForCausalLM", + "Glm5ForCausalLM", + "KimiForCausalLM", + "MiniMaxForCausalLM", + "QwQForCausalLM", + "QVQForCausalLM", + "GptOssForCausalLM", + "MagistralForCausalLM", + "DevstralForCausalLM", + "MinistralForCausalLM", + "MedGemmaForCausalLM", + "Lfm2ForCausalLM", + "CogitoForCausalLM", + "HunyuanForCausalLM", + "ErnieForCausalLM", + "YiForCausalLM", + "BaichuanForCausalLM", + "ChatGLMForConditionalGeneration", }) diff --git a/tests/test_recipes.py b/tests/test_recipes.py index 84f8102..0d9ed27 100644 --- a/tests/test_recipes.py +++ b/tests/test_recipes.py @@ -260,16 +260,17 @@ class TestV025NewRecipes: assert cfg.base == recipe.model assert cfg.task == recipe.task - def test_catalog_size_is_80(self): + def test_catalog_size_is_106(self): """Total catalog size — grew with each release. v0.25.0 shipped 43 recipes (29 + 9 Part A + 2 Part B tools + 3 Part E MLX). v0.27.0 added 3 multi-GPU recipes -> 46. v0.31.0 added 34 (vision/audio/reasoning/edge/domain/multimodal) -> 80. + v0.51.0 added 26 (model catalog expansion) -> 106. """ from soup_cli.recipes.catalog import RECIPES - assert len(RECIPES) == 80 + assert len(RECIPES) == 106 def test_new_recipes_searchable(self): """Search returns the new recipes via keyword/task filter.""" diff --git a/tests/test_recipes_v031.py b/tests/test_recipes_v031.py index 5fcd1c9..978998c 100644 --- a/tests/test_recipes_v031.py +++ b/tests/test_recipes_v031.py @@ -312,8 +312,10 @@ class TestPartFMultimodalReasoning: class TestRecipeCatalog80: """Catalog-wide invariants after v0.31.0 expansion (46 -> 80 recipes).""" - def test_total_catalog_size_is_80(self) -> None: - assert len(RECIPES) == 80 + def test_total_catalog_size_is_at_least_80(self) -> None: + # v0.31.0 baseline = 80; v0.51.0 grew to 106. Use >= so future + # catalog additions do not regress this invariant. + assert len(RECIPES) >= 80 @pytest.mark.parametrize("name", sorted(RECIPES.keys())) def test_every_recipe_loads_as_soupconfig(self, name: str) -> None: diff --git a/tests/test_v0510.py b/tests/test_v0510.py new file mode 100644 index 0000000..1274539 --- /dev/null +++ b/tests/test_v0510.py @@ -0,0 +1,605 @@ +"""v0.51.0 — Model Catalog Expansion + Alternative Model Hubs. + +Covers Parts A/B/C (25 new recipes), Part D (MULTIPACK_ARCHITECTURES extension), +Part E (hub adapters + ``hub`` field on TrainingConfig). +""" +from __future__ import annotations + +from types import MappingProxyType + +import pytest +import yaml +from pydantic import ValidationError + +from soup_cli.config.loader import load_config_from_string +from soup_cli.config.schema import TEMPLATES, SoupConfig, TrainingConfig +from soup_cli.recipes.catalog import RECIPES, get_recipe, list_recipes, search_recipes +from soup_cli.utils import hubs as hubs_mod +from soup_cli.utils.hubs import ( + SUPPORTED_HUBS, + default_endpoint, + endpoint_env_var, + is_hf, + required_hub_package, + resolve_endpoint, + validate_hub_endpoint, + validate_hub_name, +) +from soup_cli.utils.multipack_sampler import ( + MULTIPACK_ARCHITECTURES, + validate_multipack_architecture, +) + +# ===================================================================== +# Part E — hubs.validate_hub_name +# ===================================================================== + + +class TestValidateHubName: + @pytest.mark.parametrize("name", ["hf", "modelscope", "modelers"]) + def test_known_accepted(self, name: str) -> None: + assert validate_hub_name(name) == name + + @pytest.mark.parametrize("name", ["HF", "ModelScope", "MODELERS"]) + def test_case_insensitive(self, name: str) -> None: + assert validate_hub_name(name) == name.lower() + + def test_unknown_rejected(self) -> None: + with pytest.raises(ValueError, match="not supported"): + validate_hub_name("github") + + def test_empty_rejected(self) -> None: + with pytest.raises(ValueError, match="non-empty"): + validate_hub_name("") + + def test_null_byte_rejected(self) -> None: + with pytest.raises(ValueError, match="null bytes"): + validate_hub_name("hf\x00") + + def test_oversize_rejected(self) -> None: + with pytest.raises(ValueError, match="too long"): + validate_hub_name("a" * 33) + + def test_non_string_rejected(self) -> None: + with pytest.raises(TypeError): + validate_hub_name(123) # type: ignore[arg-type] + + def test_bool_rejected(self) -> None: + with pytest.raises(TypeError, match="bool"): + validate_hub_name(True) # type: ignore[arg-type] + + def test_supported_hubs_frozen(self) -> None: + with pytest.raises(AttributeError): + SUPPORTED_HUBS.add("evil") # type: ignore[attr-defined] + + def test_supported_hubs_count(self) -> None: + assert SUPPORTED_HUBS == frozenset({"hf", "modelscope", "modelers"}) + + +# ===================================================================== +# Part E — required_hub_package / default_endpoint / endpoint_env_var +# ===================================================================== + + +class TestHubMetadata: + @pytest.mark.parametrize( + "hub,pkg", + [("hf", "huggingface-hub"), ("modelscope", "modelscope"), + ("modelers", "openmind-hub")], + ) + def test_required_package(self, hub: str, pkg: str) -> None: + assert required_hub_package(hub) == pkg + + def test_required_package_unknown(self) -> None: + assert required_hub_package("github") is None + + def test_required_package_non_string(self) -> None: + assert required_hub_package(123) is None # type: ignore[arg-type] + + def test_required_package_case_insensitive(self) -> None: + assert required_hub_package("HF") == "huggingface-hub" + + def test_default_endpoint_hf_https(self) -> None: + assert default_endpoint("hf").startswith("https://") + + def test_default_endpoint_modelscope_https(self) -> None: + assert default_endpoint("modelscope").startswith("https://") + + def test_default_endpoint_modelers_https(self) -> None: + assert default_endpoint("modelers").startswith("https://") + + def test_default_endpoint_unknown_rejected(self) -> None: + with pytest.raises(ValueError): + default_endpoint("github") + + def test_default_endpoint_bool_rejected(self) -> None: + with pytest.raises(TypeError, match="bool"): + default_endpoint(True) # type: ignore[arg-type] + + def test_endpoint_env_var_bool_rejected(self) -> None: + with pytest.raises(TypeError, match="bool"): + endpoint_env_var(True) # type: ignore[arg-type] + + def test_endpoint_env_var_hf(self) -> None: + assert endpoint_env_var("hf") == "HF_ENDPOINT" + + def test_endpoint_env_var_modelscope(self) -> None: + assert endpoint_env_var("modelscope") == "MODELSCOPE_ENDPOINT" + + def test_endpoint_env_var_modelers(self) -> None: + assert endpoint_env_var("modelers") == "MODELERS_ENDPOINT" + + def test_endpoint_env_var_unknown_rejected(self) -> None: + with pytest.raises(ValueError): + endpoint_env_var("github") + + +# ===================================================================== +# Part E — validate_hub_endpoint (SSRF policy) +# ===================================================================== + + +class TestValidateHubEndpoint: + def test_https_remote_ok(self) -> None: + assert validate_hub_endpoint("https://example.com") == "https://example.com" + + def test_strips_trailing_slash(self) -> None: + assert validate_hub_endpoint("https://example.com/") == "https://example.com" + + @pytest.mark.parametrize("host", ["localhost", "127.0.0.1"]) + def test_http_loopback_ok(self, host: str) -> None: + url = f"http://{host}:8080" + assert validate_hub_endpoint(url) == url + + def test_http_remote_rejected(self) -> None: + with pytest.raises(ValueError, match="HTTPS"): + validate_hub_endpoint("http://example.com") + + def test_http_private_ip_rejected(self) -> None: + with pytest.raises(ValueError, match="loopback"): + validate_hub_endpoint("http://192.168.1.1") + + def test_http_link_local_rejected(self) -> None: + # AWS metadata endpoint + with pytest.raises(ValueError, match="loopback"): + validate_hub_endpoint("http://169.254.169.254") + + def test_zero_zero_rejected(self) -> None: + with pytest.raises(ValueError, match="0.0.0.0"): + validate_hub_endpoint("http://0.0.0.0") + + def test_ftp_scheme_rejected(self) -> None: + with pytest.raises(ValueError, match="scheme"): + validate_hub_endpoint("ftp://example.com") + + def test_file_scheme_rejected(self) -> None: + with pytest.raises(ValueError, match="scheme"): + validate_hub_endpoint("file:///etc/passwd") + + def test_no_scheme_rejected(self) -> None: + with pytest.raises(ValueError, match="scheme"): + validate_hub_endpoint("example.com") + + def test_empty_rejected(self) -> None: + with pytest.raises(ValueError, match="non-empty"): + validate_hub_endpoint("") + + def test_null_byte_rejected(self) -> None: + with pytest.raises(ValueError, match="null bytes"): + validate_hub_endpoint("https://example.com\x00") + + def test_non_string_rejected(self) -> None: + with pytest.raises(TypeError): + validate_hub_endpoint(123) # type: ignore[arg-type] + + def test_bool_rejected(self) -> None: + with pytest.raises(TypeError, match="bool"): + validate_hub_endpoint(True) # type: ignore[arg-type] + + def test_label_in_error_message(self) -> None: + with pytest.raises(ValueError, match="modelscope"): + validate_hub_endpoint("ftp://example.com", hub="modelscope") + + def test_missing_host_rejected(self) -> None: + with pytest.raises(ValueError, match="host"): + validate_hub_endpoint("https://") + + def test_control_chars_rejected(self) -> None: + # CRLF injection defence (review fix). + with pytest.raises(ValueError, match="control characters"): + validate_hub_endpoint("https://example.com\r\nX-Evil: 1") + + def test_http_ipv6_mapped_private_rejected(self) -> None: + with pytest.raises(ValueError, match="loopback"): + validate_hub_endpoint("http://[::ffff:192.168.1.1]") + + def test_http_ipv6_loopback_ok(self) -> None: + # `::1` is in `_LOOPBACK_HOSTS` so HTTP is permitted (parity with v0.29.0). + assert ( + validate_hub_endpoint("http://[::1]:8080") + == "http://[::1]:8080" + ) + + +# ===================================================================== +# Part E — resolve_endpoint (env-var override) +# ===================================================================== + + +class TestResolveEndpoint: + def test_default_when_env_unset(self) -> None: + out = resolve_endpoint("hf", env={}) + assert out.startswith("https://huggingface.co") + + def test_modelscope_default_when_env_unset(self) -> None: + out = resolve_endpoint("modelscope", env={}) + assert "modelscope" in out + + def test_modelers_default_when_env_unset(self) -> None: + out = resolve_endpoint("modelers", env={}) + assert "modelers" in out + + def test_env_override_validated(self) -> None: + out = resolve_endpoint( + "modelscope", + env={"MODELSCOPE_ENDPOINT": "https://mirror.example.com/"}, + ) + assert out == "https://mirror.example.com" + + def test_env_override_rejects_http_remote(self) -> None: + with pytest.raises(ValueError, match="HTTPS"): + resolve_endpoint( + "modelers", + env={"MODELERS_ENDPOINT": "http://example.com"}, + ) + + def test_env_override_loopback_http_ok(self) -> None: + out = resolve_endpoint( + "hf", env={"HF_ENDPOINT": "http://localhost:9000"} + ) + assert out == "http://localhost:9000" + + def test_env_override_null_byte_rejected(self) -> None: + with pytest.raises(ValueError, match="null bytes"): + resolve_endpoint( + "hf", env={"HF_ENDPOINT": "https://x\x00.com"} + ) + + def test_unknown_hub_rejected(self) -> None: + with pytest.raises(ValueError, match="not supported"): + resolve_endpoint("github", env={}) + + def test_empty_env_value_falls_back_to_default(self) -> None: + out = resolve_endpoint("hf", env={"HF_ENDPOINT": ""}) + assert out.startswith("https://huggingface.co") + + def test_default_env_uses_os_environ_when_none(self, monkeypatch) -> None: + monkeypatch.delenv("HF_ENDPOINT", raising=False) + out = resolve_endpoint("hf") + assert out.startswith("https://huggingface.co") + + +# ===================================================================== +# Part E — is_hf convenience +# ===================================================================== + + +class TestIsHf: + def test_hf_true(self) -> None: + assert is_hf("hf") + + def test_hf_uppercase_true(self) -> None: + assert is_hf("HF") + + def test_modelscope_false(self) -> None: + assert not is_hf("modelscope") + + def test_modelers_false(self) -> None: + assert not is_hf("modelers") + + def test_non_string_false(self) -> None: + assert not is_hf(None) # type: ignore[arg-type] + assert not is_hf(123) # type: ignore[arg-type] + + def test_bool_false(self) -> None: + # bool is a subclass of int but never a hub name; reject silently. + assert not is_hf(True) # type: ignore[arg-type] + assert not is_hf(False) # type: ignore[arg-type] + + +# ===================================================================== +# Part E — TrainingConfig.hub schema integration +# ===================================================================== + + +class TestTrainingConfigHub: + def test_default_is_hf(self) -> None: + cfg = TrainingConfig() + assert cfg.hub == "hf" + + @pytest.mark.parametrize("hub", ["hf", "modelscope", "modelers"]) + def test_accepts_supported(self, hub: str) -> None: + cfg = TrainingConfig(hub=hub) + assert cfg.hub == hub + + def test_unknown_rejected(self) -> None: + with pytest.raises(ValidationError): + TrainingConfig(hub="github") + + def test_empty_rejected(self) -> None: + with pytest.raises(ValidationError): + TrainingConfig(hub="") + + def test_case_insensitive_normalised(self) -> None: + # Review fix: field_validator(mode='before') normalises to lower + # (matches v0.41.0 optimizer / v0.50.0 grpo_variant policy). + cfg = TrainingConfig(hub="HF") + assert cfg.hub == "hf" + cfg2 = TrainingConfig(hub="ModelScope") + assert cfg2.hub == "modelscope" + + def test_mlx_backend_rejects_non_hf_hub(self) -> None: + yaml_str = """\ +base: Qwen/Qwen2.5-7B +task: sft +backend: mlx +data: + train: ./data/train.jsonl + format: auto +training: + epochs: 1 + lr: 2e-4 + hub: modelscope +output: ./output +""" + # load_config_from_string wraps ValidationError as ValueError. + with pytest.raises((ValidationError, ValueError), match="mlx"): + load_config_from_string(yaml_str) + + def test_hub_none_rejected(self) -> None: + # Pydantic Literal must reject None even after _normalize_hub. + with pytest.raises(ValidationError): + TrainingConfig(hub=None) # type: ignore[arg-type] + + def test_modelers_on_transformers_accepted(self) -> None: + yaml_str = """\ +base: Qwen/Qwen2.5-7B +task: sft +backend: transformers +data: + train: ./data/train.jsonl + format: auto +training: + epochs: 1 + lr: 2e-4 + hub: modelers +output: ./output +""" + cfg = load_config_from_string(yaml_str) + assert cfg.training.hub == "modelers" + + def test_mlx_backend_with_hf_hub_accepted(self) -> None: + yaml_str = """\ +base: Qwen/Qwen2.5-7B +task: sft +backend: mlx +data: + train: ./data/train.jsonl + format: auto +training: + epochs: 1 + lr: 2e-4 + hub: hf +output: ./output +""" + cfg = load_config_from_string(yaml_str) + assert cfg.training.hub == "hf" + + def test_yaml_roundtrip_modelscope(self) -> None: + yaml_str = """\ +base: Qwen/Qwen2.5-7B +task: sft +data: + train: ./data/train.jsonl + format: auto +training: + epochs: 1 + lr: 2e-4 + hub: modelscope +output: ./output +""" + cfg = load_config_from_string(yaml_str) + assert cfg.training.hub == "modelscope" + + +# ===================================================================== +# Part D — MULTIPACK_ARCHITECTURES extension +# ===================================================================== + + +class TestMultipackArchitecturesV0510: + @pytest.mark.parametrize("arch", [ + "GraniteForCausalLM", + "GraniteMoeForCausalLM", + "Glm4ForCausalLM", + "Glm5ForCausalLM", + "KimiForCausalLM", + "MiniMaxForCausalLM", + "QwQForCausalLM", + "QVQForCausalLM", + "GptOssForCausalLM", + "MagistralForCausalLM", + "DevstralForCausalLM", + "MinistralForCausalLM", + "MedGemmaForCausalLM", + "Lfm2ForCausalLM", + "CogitoForCausalLM", + "HunyuanForCausalLM", + "ErnieForCausalLM", + "YiForCausalLM", + "BaichuanForCausalLM", + "ChatGLMForConditionalGeneration", + ]) + def test_arch_in_allowlist(self, arch: str) -> None: + assert arch in MULTIPACK_ARCHITECTURES + validate_multipack_architecture(arch) # must not raise + + def test_legacy_arches_still_present(self) -> None: + # Sanity — v0.37.0 entries must still be there + for arch in ("LlamaForCausalLM", "Qwen2ForCausalLM", + "Phi3ForCausalLM", "Gemma2ForCausalLM"): + assert arch in MULTIPACK_ARCHITECTURES + + def test_count_exactly_38(self) -> None: + # 18 v0.37.0 + 20 v0.51.0 additions = 38. Exact-count assertion + # so accidental deletion fails loudly (review fix). + assert len(MULTIPACK_ARCHITECTURES) == 38 + + def test_frozen_set(self) -> None: + with pytest.raises(AttributeError): + MULTIPACK_ARCHITECTURES.add("Evil") # type: ignore[attr-defined] + + +# ===================================================================== +# Parts A/B/C — 25 new recipes +# ===================================================================== + + +V0510_RECIPE_NAMES = [ + # Part A — reasoning / agent + "gpt-oss-20b-sft", "gpt-oss-120b-sft", "glm-4.6-sft", "glm-5-sft", + "kimi-k2-sft", "kimi-k2-thinking-grpo", "minimax-m2-sft", + "qwq-32b-grpo", "qvq-72b-sft", + # Part B — small / specialist + "granite-4-sft", "lfm2-sft", "cogito-v2-sft", "mistral-small-3-sft", + "mistral-medium-3-5-sft", "magistral-small-sft", "devstral-sft", + "ministral-sft", "medgemma-sft", "embedding-gemma-sft", + # Part C — vision / multimodal + "llava-next-sft", "internvl-3-5-sft", "voxtral-sft", "baichuan-sft", + "qwen-image-sft", "deepseek-ocr-sft", "paddle-ocr-sft", +] + + +class TestV0510Recipes: + def test_recipe_count_target(self) -> None: + # 25 new entries (we shipped 26 to better cover the catalogue). + assert len(V0510_RECIPE_NAMES) >= 25 + + @pytest.mark.parametrize("name", V0510_RECIPE_NAMES) + def test_recipe_registered(self, name: str) -> None: + assert name in RECIPES, f"Recipe {name!r} missing from catalog" + assert get_recipe(name) is not None + + @pytest.mark.parametrize("name", V0510_RECIPE_NAMES) + def test_recipe_metadata_well_formed(self, name: str) -> None: + meta = RECIPES[name] + assert meta.model + assert meta.task in { + "sft", "dpo", "grpo", "kto", "orpo", "simpo", "ipo", + "ppo", "reward_model", "pretrain", "embedding", "bco", + "preference", "prm", + } + assert meta.size + assert meta.tags + assert meta.description + assert meta.yaml_str + + @pytest.mark.parametrize("name", V0510_RECIPE_NAMES) + def test_recipe_yaml_parses_as_soup_config(self, name: str) -> None: + yaml_str = RECIPES[name].yaml_str + cfg = load_config_from_string(yaml_str) + assert isinstance(cfg, SoupConfig) + assert cfg.base == RECIPES[name].model + assert cfg.task == RECIPES[name].task + + @pytest.mark.parametrize("name", V0510_RECIPE_NAMES) + def test_recipe_yaml_is_safe_loadable(self, name: str) -> None: + # Defence-in-depth — yaml.safe_load must succeed (no Python tags etc.) + parsed = yaml.safe_load(RECIPES[name].yaml_str) + assert isinstance(parsed, dict) + assert "base" in parsed and "task" in parsed + + @pytest.mark.parametrize("name", V0510_RECIPE_NAMES) + def test_recipe_model_id_no_null_or_whitespace(self, name: str) -> None: + meta = RECIPES[name] + assert "\x00" not in meta.model + assert " " not in meta.model + # ``owner/name`` shape OR plain name; every component must be non-empty + # (rejects leading/trailing slashes — review fix). + parts = meta.model.split("/") + assert 1 <= len(parts) <= 2 + assert all(p for p in parts), f"empty component in {meta.model!r}" + + def test_baichuan_recipe_uses_modelscope_hub(self) -> None: + cfg = load_config_from_string(RECIPES["baichuan-sft"].yaml_str) + assert cfg.training.hub == "modelscope" + + def test_search_finds_v0510_recipes(self) -> None: + results = search_recipes(query="gpt-oss") + assert any(r.model.startswith("openai/") for r in results) + + def test_search_kimi(self) -> None: + results = search_recipes(query="kimi") + assert len(results) >= 2 # k2 + k2-thinking + + def test_total_recipe_count_increased(self) -> None: + # The v0.31.0 baseline shipped 80 recipes; v0.51.0 adds ~26. + assert len(list_recipes()) >= 80 + 25 + + +# ===================================================================== +# Parts A/B/C — invariants from test_recipes_v031.py mirrored +# ===================================================================== + + +class TestV0510RecipeInvariants: + @pytest.mark.parametrize("name", V0510_RECIPE_NAMES) + def test_max_length_within_bounds(self, name: str) -> None: + cfg = load_config_from_string(RECIPES[name].yaml_str) + # Schema bounds: 64 <= max_length <= 1_048_576 + assert 64 <= cfg.data.max_length <= 1_048_576 + + @pytest.mark.parametrize("name", V0510_RECIPE_NAMES) + def test_task_has_required_fields(self, name: str) -> None: + cfg = load_config_from_string(RECIPES[name].yaml_str) + if cfg.task == "grpo": + assert cfg.training.reward_fn is not None + assert cfg.training.num_generations >= 1 + + +# ===================================================================== +# Cross-cutting — module surface +# ===================================================================== + + +class TestHubsModuleSurface: + def test_module_exposes_validators(self) -> None: + for sym in ( + "SUPPORTED_HUBS", + "validate_hub_name", + "validate_hub_endpoint", + "resolve_endpoint", + "default_endpoint", + "endpoint_env_var", + "required_hub_package", + "is_hf", + ): + assert hasattr(hubs_mod, sym) + + def test_default_endpoints_are_mappingproxytype(self) -> None: + # Internal but useful invariant — registry must not be mutable. + assert isinstance(hubs_mod._HUB_DEFAULT_ENDPOINTS, MappingProxyType) + assert isinstance(hubs_mod._HUB_ENDPOINT_ENV, MappingProxyType) + assert isinstance(hubs_mod._HUB_PACKAGE, MappingProxyType) + + +# ===================================================================== +# Templates — sanity (no template additions in v0.51.0, but the manifest +# must still load without regression) +# ===================================================================== + + +class TestTemplateRegressionGuard: + def test_templates_dict_intact(self) -> None: + # v0.40.0 baseline = 17 inline templates + assert len(TEMPLATES) >= 17