mirror of
https://github.com/wassname/pytorch-lightning.git
synced 2026-09-11 12:31:23 +08:00
Fix Horovod distributed backend to set the root_gpu property (#1669)
* params * drop acc * Fix Horovod distributed backend to set the root_gpu * Fixed test * Fixed tests * Fixed lint * Set root_gpu during initialization * chlog Co-authored-by: Jirka <jirka.borovec@seznam.cz>
This commit is contained in:
+3
-1
@@ -18,10 +18,12 @@ The format is based on [Keep a Changelog](http://keepachangelog.com/en/1.0.0/).
|
||||
|
||||
### Fixed
|
||||
|
||||
- Fixed broken link in PR template ([#1675](https://github.com/PyTorchLightning/pytorch-lightning/pull/1675))
|
||||
- Fixed ModelCheckpoint not None checking filepath ([1654](https://github.com/PyTorchLightning/pytorch-lightning/pull/1654))
|
||||
|
||||
- Trainer now calls `on_load_checkpoint()` when resuming from a checkpoint ([1666](https://github.com/PyTorchLightning/pytorch-lightning/pull/1666))
|
||||
|
||||
- Fixed Horovod distributed backend to set the `root_gpu` property ([#1669](https://github.com/PyTorchLightning/pytorch-lightning/pull/1669))
|
||||
|
||||
|
||||
## [0.7.5] - 2020-04-27
|
||||
|
||||
|
||||
@@ -194,8 +194,7 @@ class TrainerDDPMixin(ABC):
|
||||
|
||||
if distributed_backend is None:
|
||||
if self.has_horovodrun():
|
||||
self.check_horovod()
|
||||
self.use_horovod = True
|
||||
self._set_horovod_backend()
|
||||
elif self.num_gpus == 0:
|
||||
if self.num_nodes > 1 or self.num_processes > 1:
|
||||
self.use_ddp = True # ddp_cpu
|
||||
@@ -235,8 +234,7 @@ class TrainerDDPMixin(ABC):
|
||||
self.data_parallel_device_ids = None
|
||||
self.on_gpu = False
|
||||
elif distributed_backend == 'horovod':
|
||||
self.check_horovod()
|
||||
self.use_horovod = True
|
||||
self._set_horovod_backend()
|
||||
|
||||
# throw error to force user ddp or ddp2 choice
|
||||
if self.num_nodes > 1 and not (self.use_ddp2 or self.use_ddp):
|
||||
@@ -421,6 +419,16 @@ class TrainerDDPMixin(ABC):
|
||||
|
||||
return root_node
|
||||
|
||||
def _set_horovod_backend(self):
|
||||
self.check_horovod()
|
||||
self.use_horovod = True
|
||||
|
||||
# Initialize Horovod to get rank / size info
|
||||
hvd.init()
|
||||
if self.on_gpu:
|
||||
# Horovod assigns one local GPU per process
|
||||
self.root_gpu = hvd.local_rank()
|
||||
|
||||
def check_horovod(self):
|
||||
"""Raises a `MisconfigurationException` if the Trainer is not configured correctly for Horovod."""
|
||||
if not HOROVOD_AVAILABLE:
|
||||
|
||||
@@ -570,13 +570,11 @@ class TrainerDPMixin(ABC):
|
||||
model.forward = model_autocast_original_forward
|
||||
|
||||
def horovod_train(self, model):
|
||||
# Horovod: initialize library
|
||||
hvd.init()
|
||||
|
||||
if torch.cuda.is_available() and self.on_gpu:
|
||||
# Horovod: pin GPU to local rank
|
||||
torch.cuda.set_device(hvd.local_rank())
|
||||
model.cuda(hvd.local_rank())
|
||||
assert self.root_gpu == hvd.local_rank()
|
||||
torch.cuda.set_device(self.root_gpu)
|
||||
model.cuda(self.root_gpu)
|
||||
|
||||
# Only show progress bar from the first worker
|
||||
self.progress_bar_refresh_rate = self.progress_bar_refresh_rate if hvd.rank() == 0 else 0
|
||||
|
||||
@@ -27,12 +27,14 @@ PATH_HERE = os.path.abspath(os.path.dirname(__file__))
|
||||
PATH_ROOT = os.path.join(PATH_HERE, '..', '..', '..', '..')
|
||||
sys.path.insert(0, os.path.abspath(PATH_ROOT))
|
||||
|
||||
from pytorch_lightning import Trainer # noqa: E402
|
||||
from pytorch_lightning.callbacks import ModelCheckpoint # noqa: E402
|
||||
import tests.base.utils as tutils # noqa: E402
|
||||
|
||||
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument('--trainer-options', required=True)
|
||||
parser.add_argument('--on-gpu', action='store_true', default=False)
|
||||
|
||||
|
||||
def run_test_from_config(trainer_options):
|
||||
@@ -44,11 +46,15 @@ def run_test_from_config(trainer_options):
|
||||
trainer_options['checkpoint_callback'] = ModelCheckpoint(ckpt_path)
|
||||
|
||||
model, hparams = tutils.get_default_model()
|
||||
tutils.run_model_test(trainer_options, model, version=0, with_hpc=False)
|
||||
tutils.run_model_test(trainer_options, model, on_gpu=args.on_gpu, version=0, with_hpc=False)
|
||||
|
||||
# Horovod should be initialized following training. If not, this will raise an exception.
|
||||
assert hvd.size() == 2
|
||||
|
||||
if args.on_gpu:
|
||||
# Test the root_gpu property
|
||||
assert Trainer(gpus=1, distributed_backend='horovod', max_epochs=1).root_gpu == hvd.local_rank()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
args = parser.parse_args()
|
||||
|
||||
@@ -38,10 +38,12 @@ def _nccl_available():
|
||||
return False
|
||||
|
||||
|
||||
def _run_horovod(trainer_options):
|
||||
def _run_horovod(trainer_options, on_gpu=False):
|
||||
"""Execute the training script across multiple workers in parallel."""
|
||||
cmdline = ['horovodrun', '-np', '2', sys.executable, TEST_SCRIPT,
|
||||
'--trainer-options', shlex.quote(json.dumps(trainer_options))]
|
||||
if on_gpu:
|
||||
cmdline += ['--on-gpu']
|
||||
exit_code = subprocess.call(' '.join(cmdline), shell=True, env=os.environ.copy())
|
||||
assert exit_code == 0
|
||||
|
||||
@@ -93,7 +95,7 @@ def test_horovod_multi_gpu(tmpdir):
|
||||
gpus=1,
|
||||
distributed_backend='horovod'
|
||||
)
|
||||
_run_horovod(trainer_options)
|
||||
_run_horovod(trainer_options, on_gpu=True)
|
||||
|
||||
|
||||
@pytest.mark.skipif(sys.version_info >= (3, 8), reason="Horovod not yet supported in Python 3.8")
|
||||
@@ -159,5 +161,6 @@ def test_horovod_multi_optimizer(tmpdir):
|
||||
def get_optimizer_params(optimizer):
|
||||
return set([p for group in optimizer.param_groups for p in group.get('params', [])])
|
||||
|
||||
assert get_model_params(model.generator) != get_model_params(model.discriminator)
|
||||
assert get_model_params(model.generator) == get_optimizer_params(trainer.optimizers[0])
|
||||
assert get_model_params(model.discriminator) == get_optimizer_params(trainer.optimizers[1])
|
||||
|
||||
Reference in New Issue
Block a user