Updates for trl 0.16.0 - mostly for GRPO (#2437) [skip ci]

* add grpo scale_rewards config for trl#3135 * options to connect to vllm server directly w grpo trl#3094 * temperature support trl#3029 * sampling/generation kwargs for grpo trl#2989 * make vllm_enable_prefix_caching a config param trl#2900 * grpo multi-step optimizeations trl#2899 * remove overrides for grpo trainer * bump trl to 0.16.0 * add cli to start vllm-serve via trl * call the python module directly * update to use vllm with 2.6.0 too now and call trl vllm serve from module * vllm 0.8.1 * use python3 * use sys.executable * remove context and wait for start * fixes to make it actually work * fixes so the grpo tests pass with new vllm paradigm * explicit host/port and check in start vllm * make sure that vllm doesn't hang by setting quiet so outouts go to dev null * also bump bnb to latest release * add option for wait from cli and nccl debugging for ci * grpo + vllm test on separate devices for now * make sure grpo + vllm tests runs single worker since pynccl comms would conflict * fix cli * remove wait and add caching for argilla dataset * refactoring configs * chore: lint * add vllm config * fixup vllm grpo args * fix one more incorrect schema/config path * fix another vlllm reference and increase timeout * make the tests run a bit faster * change mbsz back so it is correct for grpo * another change mbsz back so it is correct for grpo * fixing cli args * nits * adding docs * docs * include tensor parallel size for vllm in pydantic schema * moving start_vllm, more docs * limit output len for grpo vllm * vllm enable_prefix_caching isn't a bool cli arg * fix env ordering in tests and also use pid check when looking for vllm --------- Co-authored-by: Salman Mohammadi <salman.mohammadi@outlook.com>
2025-03-31 15:47:11 -04:00
parent b35992262e
commit b6fc46ada8
24 changed files with 703 additions and 349 deletions
--- a/setup.py
+++ b/setup.py
@@ -10,7 +10,7 @@ from pathlib import Path
 from setuptools import find_packages, setup


-def parse_requirements():
+def parse_requirements(extras_require_map):
    _install_requires = []
    _dependency_links = []
    with open("./requirements.txt", encoding="utf-8") as requirements_file:
@@ -67,6 +67,7 @@ def parse_requirements():
            if (major, minor) >= (2, 6):
                _install_requires.pop(_install_requires.index(xformers_version))
                _install_requires.append("xformers==0.0.29.post2")
+                extras_require_map["vllm"] = ["vllm==0.8.1"]
            elif (major, minor) >= (2, 5):
                _install_requires.pop(_install_requires.index(xformers_version))
                if patch == 0:
@@ -86,7 +87,7 @@ def parse_requirements():

    except PackageNotFoundError:
        pass
-    return _install_requires, _dependency_links
+    return _install_requires, _dependency_links, extras_require_map


 def get_package_version():
@@ -103,7 +104,46 @@ def get_package_version():
    return version_


-install_requires, dependency_links = parse_requirements()
+extras_require = {
+    "flash-attn": ["flash-attn==2.7.4.post1"],
+    "ring-flash-attn": ["ring-flash-attn>=0.1.4", "yunchang==0.6.0"],
+    "deepspeed": [
+        "deepspeed==0.16.4",
+        "deepspeed-kernels",
+    ],
+    "mamba-ssm": [
+        "mamba-ssm==1.2.0.post1",
+        "causal_conv1d",
+    ],
+    "auto-gptq": [
+        "auto-gptq==0.5.1",
+    ],
+    "mlflow": [
+        "mlflow",
+    ],
+    "galore": [
+        "galore_torch",
+    ],
+    "apollo": [
+        "apollo-torch",
+    ],
+    "optimizers": [
+        "galore_torch",
+        "apollo-torch",
+        "lomo-optim==0.1.1",
+        "torch-optimi==0.2.1",
+    ],
+    "ray": [
+        "ray[train]",
+    ],
+    "vllm": [
+        "vllm==0.7.2",
+    ],
+}
+
+install_requires, dependency_links, extras_require_build = parse_requirements(
+    extras_require
+)

 setup(
    version=get_package_version(),
@@ -116,40 +156,5 @@ setup(
            "axolotl=axolotl.cli.main:main",
        ],
    },
-    extras_require={
-        "flash-attn": ["flash-attn==2.7.4.post1"],
-        "ring-flash-attn": ["ring-flash-attn>=0.1.4", "yunchang==0.6.0"],
-        "deepspeed": [
-            "deepspeed==0.16.4",
-            "deepspeed-kernels",
-        ],
-        "mamba-ssm": [
-            "mamba-ssm==1.2.0.post1",
-            "causal_conv1d",
-        ],
-        "auto-gptq": [
-            "auto-gptq==0.5.1",
-        ],
-        "mlflow": [
-            "mlflow",
-        ],
-        "galore": [
-            "galore_torch",
-        ],
-        "apollo": [
-            "apollo-torch",
-        ],
-        "optimizers": [
-            "galore_torch",
-            "apollo-torch",
-            "lomo-optim==0.1.1",
-            "torch-optimi==0.2.1",
-        ],
-        "ray": [
-            "ray[train]",
-        ],
-        "vllm": [
-            "vllm==0.7.2",
-        ],
-    },
+    extras_require=extras_require_build,
 )