Add wandb_entity to wandb options, update example configs, update README (#361)

* Update wandb_entity and add wandb descriptions * add wandb to config section * remove trailing whitespace for pre-commit hook * remove trailing whitespace for pre-commit hook --------- Co-authored-by: Morgan McGuire <morganmcguire@Morgans-MacBook-Pro.local> Co-authored-by: Wing Lian <wing.lian@gmail.com>
2023-08-12 17:17:11 +01:00
parent 96bd6ae1c4
commit 7019509daa
20 changed files with 37 additions and 4 deletions
--- a/README.md
+++ b/README.md
@@ -401,11 +401,12 @@ lora_out_dir:
 lora_fan_in_fan_out: false

 # wandb configuration if you're using it
-wandb_mode:
-wandb_project:
+wandb_mode: # "offline" to save run metadata locally and not sync to the server, "disabled" to turn off wandb
+wandb_project: # your wandb project name
+wandb_entity: # a wandb Team name if using a Team
 wandb_watch:
-wandb_run_id:
-wandb_log_model: # 'checkpoint'
+wandb_run_id: # set the name of your wandb run
+wandb_log_model: # "checkpoint" to log model to wandb Artifacts every `save_steps` or "end" to log only at the end of training

 # where to save the finished model to
 output_dir: ./completed-model
@@ -553,6 +554,18 @@ fsdp_config:

 - llama Deepspeed: append `ACCELERATE_USE_DEEPSPEED=true` in front of finetune command

+##### Weights & Biases Logging
+
+- wandb options
+```yaml
+wandb_mode:
+wandb_project:
+wandb_entity:
+wandb_watch:
+wandb_run_id:
+wandb_log_model:
+```
+
 ### Inference

 Pass the appropriate flag to the train command:
--- a/examples/cerebras/qlora.yml
+++ b/examples/cerebras/qlora.yml
@@ -23,6 +23,7 @@ lora_target_modules:
 lora_target_linear:
 lora_fan_in_fan_out:
 wandb_project:
+wandb_entity:
 wandb_watch:
 wandb_run_id:
 wandb_log_model:
--- a/examples/falcon/config-7b-lora.yml
+++ b/examples/falcon/config-7b-lora.yml
@@ -24,6 +24,7 @@ lora_target_modules:
 lora_target_linear: true
 lora_fan_in_fan_out:
 wandb_project:
+wandb_entity:
 wandb_watch:
 wandb_run_id:
 wandb_log_model:
--- a/examples/falcon/config-7b-qlora.yml
+++ b/examples/falcon/config-7b-qlora.yml
@@ -38,6 +38,7 @@ lora_target_linear: true
 lora_fan_in_fan_out:

 wandb_project:
+wandb_entity:
 wandb_watch:
 wandb_run_id:
 wandb_log_model:
--- a/examples/falcon/config-7b.yml
+++ b/examples/falcon/config-7b.yml
@@ -24,6 +24,7 @@ lora_target_modules:
 lora_target_linear: true
 lora_fan_in_fan_out:
 wandb_project:
+wandb_entity:
 wandb_watch:
 wandb_run_id:
 wandb_log_model:
--- a/examples/gptj/qlora.yml
+++ b/examples/gptj/qlora.yml
@@ -20,6 +20,7 @@ lora_target_modules:
 lora_target_linear: true
 lora_fan_in_fan_out:
 wandb_project:
+wandb_entity:
 wandb_watch:
 wandb_run_id:
 wandb_log_model:
--- a/examples/gptq-lora-7b/config.yml
+++ b/examples/gptq-lora-7b/config.yml
@@ -22,6 +22,7 @@ lora_target_modules:
  - v_proj
 lora_fan_in_fan_out: false
 wandb_project: llama-7b-lora-int4
+wandb_entity:
 wandb_watch:
 wandb_run_id:
 wandb_log_model:
--- a/examples/jeopardy-bot/config.yml
+++ b/examples/jeopardy-bot/config.yml
@@ -18,6 +18,7 @@ lora_dropout:
 lora_target_modules:
 lora_fan_in_fan_out: false
 wandb_project:
+wandb_entity:
 wandb_watch:
 wandb_run_id:
 wandb_log_model:
--- a/examples/llama-2/lora.yml
+++ b/examples/llama-2/lora.yml
@@ -26,6 +26,7 @@ lora_target_linear: true
 lora_fan_in_fan_out:

 wandb_project:
+wandb_entity:
 wandb_watch:
 wandb_run_id:
 wandb_log_model:
--- a/examples/llama-2/qlora.yml
+++ b/examples/llama-2/qlora.yml
@@ -27,6 +27,7 @@ lora_target_linear: true
 lora_fan_in_fan_out:

 wandb_project:
+wandb_entity:
 wandb_watch:
 wandb_run_id:
 wandb_log_model:
--- a/examples/mpt-7b/config.yml
+++ b/examples/mpt-7b/config.yml
@@ -20,6 +20,7 @@ lora_target_modules:
  - v_proj
 lora_fan_in_fan_out: false
 wandb_project: mpt-alpaca-7b
+wandb_entity:
 wandb_watch:
 wandb_run_id:
 wandb_log_model:
--- a/examples/openllama-3b/config.yml
+++ b/examples/openllama-3b/config.yml
@@ -22,6 +22,7 @@ lora_target_modules:
 lora_target_linear:
 lora_fan_in_fan_out:
 wandb_project:
+wandb_entity:
 wandb_watch:
 wandb_run_id:
 wandb_log_model:
--- a/examples/openllama-3b/lora.yml
+++ b/examples/openllama-3b/lora.yml
@@ -28,6 +28,7 @@ lora_target_modules:
  - o_proj
 lora_fan_in_fan_out:
 wandb_project:
+wandb_entity:
 wandb_watch:
 wandb_run_id:
 wandb_log_model:
--- a/examples/openllama-3b/qlora.yml
+++ b/examples/openllama-3b/qlora.yml
@@ -22,6 +22,7 @@ lora_target_modules:
 lora_target_linear: true
 lora_fan_in_fan_out:
 wandb_project:
+wandb_entity:
 wandb_watch:
 wandb_run_id:
 wandb_log_model:
--- a/examples/pythia-12b/config.yml
+++ b/examples/pythia-12b/config.yml
@@ -23,6 +23,7 @@ lora_target_modules:
 lora_target_linear: true
 lora_fan_in_fan_out: true  # pythia/GPTNeoX lora specific
 wandb_project:
+wandb_entity:
 wandb_watch:
 wandb_run_id:
 wandb_log_model:
--- a/examples/pythia/lora.yml
+++ b/examples/pythia/lora.yml
@@ -17,6 +17,7 @@ lora_target_modules:
 lora_target_linear:
 lora_fan_in_fan_out: true  # pythia/GPTNeoX lora specific
 wandb_project:
+wandb_entity:
 wandb_watch:
 wandb_run_id:
 wandb_log_model:
--- a/examples/redpajama/config-3b.yml
+++ b/examples/redpajama/config-3b.yml
@@ -21,6 +21,7 @@ lora_target_modules:
  - v_proj
 lora_fan_in_fan_out: false
 wandb_project: redpajama-alpaca-3b
+wandb_entity:
 wandb_watch:
 wandb_run_id:
 wandb_log_model:
--- a/examples/replit-3b/config-lora.yml
+++ b/examples/replit-3b/config-lora.yml
@@ -20,6 +20,7 @@ lora_target_modules:
  - mlp_down
 lora_fan_in_fan_out:
 wandb_project: lora-replit
+wandb_entity:
 wandb_watch:
 wandb_run_id:
 wandb_log_model:
--- a/examples/xgen-7b/xgen-7b-8k-qlora.yml
+++ b/examples/xgen-7b/xgen-7b-8k-qlora.yml
@@ -37,6 +37,7 @@ lora_target_linear: true
 lora_fan_in_fan_out:

 wandb_project:
+wandb_entity:
 wandb_watch:
 wandb_run_id:
 wandb_log_model:
--- a/src/axolotl/utils/wandb.py
+++ b/src/axolotl/utils/wandb.py
@@ -9,6 +9,8 @@ def setup_wandb_env_vars(cfg):
    elif cfg.wandb_project and len(cfg.wandb_project) > 0:
        os.environ["WANDB_PROJECT"] = cfg.wandb_project
        cfg.use_wandb = True
+        if cfg.wandb_entity and len(cfg.wandb_entity) > 0:
+            os.environ["WANDB_ENTITY"] = cfg.wandb_entity
        if cfg.wandb_watch and len(cfg.wandb_watch) > 0:
            os.environ["WANDB_WATCH"] = cfg.wandb_watch
        if cfg.wandb_log_model and len(cfg.wandb_log_model) > 0: