migrate example configs to canonical attn_implementation

2026-04-23 22:15:07 +00:00
parent 2d64d009d8
commit 39226623d2
222 changed files with 209 additions and 243 deletions
--- a/examples/gpt-oss/gpt-oss-120b-fft-fsdp2-offload.yaml
+++ b/examples/gpt-oss/gpt-oss-120b-fft-fsdp2-offload.yaml
@@ -47,7 +47,7 @@ learning_rate: 2e-5
 bf16: true
 tf32: true

-flash_attention: true
+attn_implementation: flash_attention_2
 attn_implementation: kernels-community/vllm-flash-attn3  # this is not needed if using flash_attn >= 2.8.3

 gradient_checkpointing: true
--- a/examples/gpt-oss/gpt-oss-20b-fft-deepspeed-zero3.yaml
+++ b/examples/gpt-oss/gpt-oss-20b-fft-deepspeed-zero3.yaml
@@ -43,7 +43,7 @@ learning_rate: 2e-5
 bf16: true
 tf32: true

-flash_attention: true
+attn_implementation: flash_attention_2
 attn_implementation: kernels-community/vllm-flash-attn3  # this is not needed if using flash_attn >= 2.8.3

 gradient_checkpointing: true
--- a/examples/gpt-oss/gpt-oss-20b-fft-fsdp2-offload.yaml
+++ b/examples/gpt-oss/gpt-oss-20b-fft-fsdp2-offload.yaml
@@ -44,7 +44,7 @@ learning_rate: 2e-5
 bf16: true
 tf32: true

-flash_attention: true
+attn_implementation: flash_attention_2
 attn_implementation: kernels-community/vllm-flash-attn3  # this is not needed if using flash_attn >= 2.8.3

 gradient_checkpointing: true
--- a/examples/gpt-oss/gpt-oss-20b-fft-fsdp2.yaml
+++ b/examples/gpt-oss/gpt-oss-20b-fft-fsdp2.yaml
@@ -43,7 +43,7 @@ learning_rate: 2e-5
 bf16: true
 tf32: true

-flash_attention: true
+attn_implementation: flash_attention_2
 attn_implementation: kernels-community/vllm-flash-attn3  # this is not needed if using flash_attn >= 2.8.3

 gradient_checkpointing: true
--- a/examples/gpt-oss/gpt-oss-20b-sft-lora-singlegpu.yaml
+++ b/examples/gpt-oss/gpt-oss-20b-sft-lora-singlegpu.yaml
@@ -56,7 +56,7 @@ learning_rate: 2e-4
 bf16: true
 tf32: true

-flash_attention: true
+attn_implementation: flash_attention_2
 attn_implementation: kernels-community/vllm-flash-attn3  # this is not needed if using flash_attn >= 2.8.3

 gradient_checkpointing: true
--- a/examples/gpt-oss/gpt-oss-safeguard-20b-sft-lora-singlegpu.yaml
+++ b/examples/gpt-oss/gpt-oss-safeguard-20b-sft-lora-singlegpu.yaml
@@ -56,7 +56,7 @@ learning_rate: 2e-4
 bf16: true
 tf32: true

-flash_attention: true
+attn_implementation: flash_attention_2
 attn_implementation: kernels-community/vllm-flash-attn3  # this is not needed if using flash_attn >= 2.8.3

 gradient_checkpointing: true