[build-system] requires = ["hatchling"] build-backend = "hatchling.build" [project] name = "soup-cli" version = "0.53.10" description = "Fine-tune LLMs in one command. No SSH, no config hell." readme = "README.md" license = "Apache-2.0" requires-python = ">=3.9" authors = [ { name = "Soup Team" }, ] keywords = ["llm", "fine-tuning", "lora", "qlora", "machine-learning"] classifiers = [ "Development Status :: 3 - Alpha", "Intended Audience :: Developers", "Intended Audience :: Science/Research", "License :: OSI Approved :: Apache Software License", "Programming Language :: Python :: 3", "Topic :: Scientific/Engineering :: Artificial Intelligence", ] dependencies = [ "typer>=0.9.0,<0.21.0", "rich>=13.0.0", "pydantic>=2.0.0", "pyyaml>=6.0", "torch>=2.0.0", "transformers>=4.36.0,<5.0.0", "peft>=0.7.0", "trl>=0.7.0", "datasets>=2.14.0", "bitsandbytes>=0.41.0", "accelerate>=0.25.0", "huggingface-hub>=0.16.0", "plotext>=5.2.0", ] [project.optional-dependencies] eval = ["lm-eval>=0.4.0"] data = ["datasketch>=1.6.0"] wandb = ["wandb>=0.15.0,<0.18.0"] dev = ["pytest>=7.0", "ruff>=0.1.0", "pytest-cov>=4.0", "httpx>=0.24.0"] ui = ["fastapi>=0.104.0", "uvicorn>=0.24.0"] serve = ["fastapi>=0.104.0", "uvicorn>=0.24.0"] serve-fast = ["vllm>=0.4.0", "fastapi>=0.104.0", "uvicorn>=0.24.0"] generate = ["httpx>=0.24.0"] deepspeed = ["deepspeed>=0.12.0"] fast = ["unsloth>=2024.8"] vision = ["Pillow>=9.0.0"] qat = ["torchao>=0.4.0"] liger = ["liger-kernel>=0.3.0"] ring-attn = ["ring-flash-attn>=0.1.0"] onnx = ["optimum[onnxruntime]>=1.16.0"] tensorrt = ["tensorrt_llm>=0.9.0"] audio = ["librosa>=0.10.0", "soundfile>=0.12.0"] awq = ["autoawq>=0.2.0"] gptq = ["auto-gptq>=0.7.0"] sglang = ["sglang>=0.2.0", "fastapi>=0.104.0", "uvicorn>=0.24.0"] mlx = ["mlx>=0.20.0", "mlx-lm>=0.20.0"] cce = ["cut-cross-entropy>=24.10.0"] tui = ["textual>=0.50.0"] # v0.53.8 #89 — bundle MLflow / SwanLab / Trackio for `--tracker` users. trackers = ["mlflow>=2.0.0", "swanlab>=0.3.0", "trackio>=0.0.1"] # v0.53.8 #85 — fsspec backends for remote dataset loading (s3 / gs / az / oci). remote = ["fsspec>=2024.1.0", "s3fs>=2024.1.0", "gcsfs>=2024.1.0", "adlfs>=2024.1.0"] # v0.53.10 #150 — bundle scikit-optimize so `soup data mix --optimize` runs the # Bayesian-style loop instead of falling back to the v0.48.0 Dirichlet sampler. mix = ["scikit-optimize>=0.9.0"] # v0.53.10 #113 — production-grade data quality: langdetect (language) + # presidio-analyzer (PII). Llama-Guard-3-1B is documented as a manual recipe # (license + ~600 MB weight blob too large to bundle by default). data-pro = ["langdetect>=1.0.9", "presidio-analyzer>=2.2.0"] [project.scripts] soup = "soup_cli.cli:run" [project.urls] Homepage = "https://github.com/MakazhanAlpamys/Soup" Repository = "https://github.com/MakazhanAlpamys/Soup" Issues = "https://github.com/MakazhanAlpamys/Soup/issues" [tool.hatch.build.targets.wheel] packages = ["soup_cli"] # v0.53.8 #93 — include bundled fixture JSONLs as package data so # `soup data demo` works in zipapp / namespace-package installs. # Hatchling's ``packages = ["soup_cli"]`` already recurses into the # package directory, so we use the artifacts directive (NOT # force-include, which double-shipped the files in v0.53.8 and produced # a "duplicate filename in local headers" 400 from PyPI upload). artifacts = ["soup_cli/data/_fixtures/*.jsonl"] [tool.ruff] target-version = "py39" line-length = 100 [tool.ruff.lint] select = ["E", "F", "I", "N", "W"] [tool.pytest.ini_options] testpaths = ["tests"] markers = ["smoke: slow smoke tests that download models and run training (run with: pytest -m smoke)"] addopts = "-m 'not smoke' --cov=soup_cli --cov-fail-under=50 --cov-report=term-missing:skip-covered"