Unbound fix (#105)

* is some error
```
                                [rank1]:   File "/mnt/ceph/develop/jiawei/cogvideox-distillation/training/cogvideox_image_to_video_lora.py", line 1005, in <module>                                                                                                             [rank1]:     main(args)                                                                                                 [rank1]:   File "/mnt/ceph/develop/jiawei/cogvideox-distillation/training/cogvideox_image_to_video_lora.py", line 884, in main                                                                                                                  [rank1]:     "gradient_norm_before_clip": gradient_norm_before_clip,                                                    [rank1]: UnboundLocalError: local variable 'gradient_norm_before_clip' referenced before assignment
```
ref1 :https://github.com/a-r-r-o-w/cogvideox-factory/pull/84
ref2: https://github.com/a-r-r-o-w/cogvideox-factory/pull/100
It seems that there is a bug in the accelerator.is_main_process task and accelerator.distributed_type. Is_main_process has scheduling problems during the training initiation and training stages.

* fix
This commit is contained in:
glide-the
2024-12-01 04:28:27 +08:00
committed by GitHub
parent 099c9c35c4
commit ec3ec47891
3 changed files with 6 additions and 6 deletions
+3 -3
View File
@@ -641,7 +641,7 @@ def main(args):
# We need to initialize the trackers we use, and also store our configuration.
# The trackers initializes automatically on the main process.
if accelerator.is_main_process:
if accelerator.distributed_type == DistributedType.DEEPSPEED or accelerator.is_main_process:
tracker_name = args.tracker_name or "cogvideox-lora"
accelerator.init_trackers(tracker_name, config=vars(args))
@@ -824,7 +824,7 @@ def main(args):
loss = loss.mean()
accelerator.backward(loss)
if accelerator.sync_gradients:
if accelerator.sync_gradients and accelerator.distributed_type != DistributedType.DEEPSPEED:
gradient_norm_before_clip = get_gradient_norm(transformer.parameters())
accelerator.clip_grad_norm_(transformer.parameters(), args.max_grad_norm)
gradient_norm_after_clip = get_gradient_norm(transformer.parameters())
@@ -878,7 +878,7 @@ def main(args):
last_lr = lr_scheduler.get_last_lr()[0] if lr_scheduler is not None else args.learning_rate
logs = {"loss": loss.detach().item(), "lr": last_lr}
# gradnorm + deepspeed: https://github.com/microsoft/DeepSpeed/issues/4555
if accelerator.distributed_type != DistributedType.DEEPSPEED:
if accelerator.sync_gradients and accelerator.distributed_type != DistributedType.DEEPSPEED:
logs.update(
{
"gradient_norm_before_clip": gradient_norm_before_clip,
+2 -2
View File
@@ -732,7 +732,7 @@ def main(args):
loss = loss.mean()
accelerator.backward(loss)
if accelerator.sync_gradients:
if accelerator.sync_gradients and accelerator.distributed_type != DistributedType.DEEPSPEED:
gradient_norm_before_clip = get_gradient_norm(transformer.parameters())
accelerator.clip_grad_norm_(transformer.parameters(), args.max_grad_norm)
gradient_norm_after_clip = get_gradient_norm(transformer.parameters())
@@ -778,7 +778,7 @@ def main(args):
last_lr = lr_scheduler.get_last_lr()[0] if lr_scheduler is not None else args.learning_rate
logs = {"loss": loss.detach().item(), "lr": last_lr}
# gradnorm + deepspeed: https://github.com/microsoft/DeepSpeed/issues/4555
if accelerator.distributed_type != DistributedType.DEEPSPEED:
if accelerator.sync_gradients and accelerator.distributed_type != DistributedType.DEEPSPEED:
logs.update(
{
"gradient_norm_before_clip": gradient_norm_before_clip,
+1 -1
View File
@@ -698,7 +698,7 @@ def main(args):
loss = loss.mean()
accelerator.backward(loss)
if accelerator.sync_gradients:
if accelerator.sync_gradients and accelerator.distributed_type != DistributedType.DEEPSPEED:
gradient_norm_before_clip = get_gradient_norm(transformer.parameters())
accelerator.clip_grad_norm_(transformer.parameters(), args.max_grad_norm)
gradient_norm_after_clip = get_gradient_norm(transformer.parameters())