已合并
[pytorch][bugfix] intercepting the vpp feature during inference #4018
[pytorch][bugfix] intercepting the vpp feature during inference #4018
已合并
yanzhixiao创建于 1月7日
共 1 个文件变更+3-0
@@ -48,6 +48,9 @@ def model_provider(pre_process=True, post_process=True) -> Union[GPTModelInfer,
48 if args.sequence_parallel and args.use_kv_cache:48 if args.sequence_parallel and args.use_kv_cache:
49 raise AssertionError('Use_kv_cache can not be true in sequence_parallel mode.')49 raise AssertionError('Use_kv_cache can not be true in sequence_parallel mode.')
50 50 
51+ if args.num_layers_per_virtual_pipeline_stage is not None:
52+ raise AssertionError('VPP is not supported for inference.')
53+ 
51 print_rank_0('building GPT model ...')54 print_rank_0('building GPT model ...')
52 # Experimental loading arguments from yaml55 # Experimental loading arguments from yaml
53 if args.yaml_cfg is not None:56 if args.yaml_cfg is not None: