已合并
[pytorch][bugfix] intercepting the vpp feature during inference #4018
yanzhixiao创建于 1月7日
[pytorch][bugfix] intercepting the vpp feature during inference #4018
已合并
共 1 个文件变更+3-0
| @@ -48,6 +48,9 @@ def model_provider(pre_process=True, post_process=True) -> Union[GPTModelInfer, | |||
| 48 | if args.sequence_parallel and args.use_kv_cache: | 48 | if args.sequence_parallel and args.use_kv_cache: |
| 49 | raise AssertionError('Use_kv_cache can not be true in sequence_parallel mode.') | 49 | raise AssertionError('Use_kv_cache can not be true in sequence_parallel mode.') |
| 50 | 50 | ||
| 51 | + if args.num_layers_per_virtual_pipeline_stage is not None: | ||
| 52 | + raise AssertionError('VPP is not supported for inference.') | ||
| 53 | + | ||
| 51 | print_rank_0('building GPT model ...') | 54 | print_rank_0('building GPT model ...') |
| 52 | # Experimental loading arguments from yaml | 55 | # Experimental loading arguments from yaml |
| 53 | if args.yaml_cfg is not None: | 56 | if args.yaml_cfg is not None: |