diff --git a/mindspore/python/mindspore/profiler/parser/integrator.py b/mindspore/python/mindspore/profiler/parser/integrator.py index 726721e2b86..800bfb0f29c 100644 --- a/mindspore/python/mindspore/profiler/parser/integrator.py +++ b/mindspore/python/mindspore/profiler/parser/integrator.py @@ -18,8 +18,10 @@ import json import os import stat from decimal import Decimal +from enum import Enum from mindspore import log as logger +from mindspore import context from mindspore.context import get_auto_parallel_context from mindspore.profiler.common.exceptions.exceptions import ProfilerIOException, \ ProfilerFileNotFoundException, ProfilerRawFileException, ProfilerParamValueErrorException @@ -508,6 +510,13 @@ class Integrator: self._display_col_names_detail.append(self._col_names_detail[5]) +class DeviceTarget(Enum): + """The device target enum.""" + CPU = 'cpu' + GPU = 'gpu' + ASCEND = 'ascend' + + class BaseTimelineGenerator: """ Analyse timeline data from file. @@ -524,6 +533,8 @@ class BaseTimelineGenerator: _HOST_CPU_PID = 11000 _OP_OVERLAP_PID = 12000 + _OP_GPU_ACTIVITY_PID = 13000 + _RECEIVE_ALONE = 7997 _ALLREDUCE_ALONE = 7998 _MERGED_COMPUTATION_TID = 7999 @@ -536,7 +547,7 @@ class BaseTimelineGenerator: _HOST_CPU_OP_TID = 100003 _SINGLE_TID = 0 - _STEPS_SORT_INDEX = -1 + _STEPS_SORT_INDEX = -4 _map_tid_name_to_int = { "Steps": (-4, _STEPS_TID), @@ -553,32 +564,45 @@ class BaseTimelineGenerator: } _op_name_idx, _tid_idx, _start_time_idx, _duration_idx = 0, 1, 2, 3 _max_scope_name_num = 0 - _host_cpu_op_label = 'HostCpuOps' + _host_cpu_op_label = 'Host CPU OP' + _gpu_op_label = "GPU Op" + _ascend_op_label = "Ascend Op" + _aicore_op_label = "AICORE OP" + _aicpu_op_label = "AICPU OP" _device_id = 0 _profiling_dir = "" _timeline_summary_filename = "" _display_filename = "" + _op_name_list = [] + _device_target = DeviceTarget.ASCEND.value + _model = context.GRAPH_MODE - def __init__(self): + def __init__(self, device_target, model): self._tid_dict = { "computation_op": (self._MERGED_COMPUTATION_TID, self._OP_OVERLAP_PID), "communication_not_overlapped": (self._PURE_COMMUNICATION_TID, self._OP_OVERLAP_PID), "communication": (self._MERGED_COMMUNICATION_TID, self._OP_OVERLAP_PID), "free_time": (self._FREE_TIME_TID, self._OP_OVERLAP_PID) } + self._device_target = str(device_target).lower() + self._model = model self._step_end_op_name = "" def get_thread_label_name(self): """Get process and thread config.""" + device_process_label = self._get_device_process_label() return [ - {"name": "process_labels", "ph": "M", "pid": self._device_id, "args": {"labels": "AI Core Op"}}, - {"name": "process_labels", "ph": "M", "pid": self._AI_CPU_PID, "args": {"labels": "AI CPU Op"}}, + {"name": "process_labels", "ph": "M", "pid": self._device_id, "args": {"labels": device_process_label}}, + {"name": "process_labels", "ph": "M", "pid": self._AI_CPU_PID, "args": {"labels": self._aicpu_op_label}}, {"name": "process_labels", "ph": "M", "pid": self._COMMUNICATION_OP_PID, "args": {"labels": "Communication Op"}}, - {"name": "process_labels", "ph": "M", "pid": self._HOST_CPU_PID, "args": {"labels": "Host CPU Op"}}, + {"name": "process_labels", "ph": "M", "pid": self._HOST_CPU_PID, + "args": {"labels": self._host_cpu_op_label}}, {"name": "process_labels", "ph": "M", "pid": self._OP_OVERLAP_PID, "args": {"labels": "Op Overlap Analyse"}}, + {"name": "process_labels", "ph": "M", "pid": self._OP_GPU_ACTIVITY_PID, + "args": {"labels": "Activity Op"}}, {"name": "process_sort_index", "ph": "M", "pid": self._device_id, "args": {"sort_index": 0}}, {"name": "process_sort_index", "ph": "M", "pid": self._AI_CPU_PID, "args": {"sort_index": 10}}, @@ -613,6 +637,20 @@ class BaseTimelineGenerator: "args": {"sort_index": self._STEPS_SORT_INDEX}}, ] + def _get_device_process_label(self): + """Get device process label.""" + device_process_label = self._aicore_op_label + if self._device_target == DeviceTarget.ASCEND.value: + if self._model == context.GRAPH_MODE: + device_process_label = self._aicore_op_label + elif self._model == context.PYNATIVE_MODE: + device_process_label = self._ascend_op_label + elif self._device_target == DeviceTarget.GPU.value: + device_process_label = self._gpu_op_label + elif self._device_target == DeviceTarget.CPU.value: + device_process_label = self._host_cpu_op_label + return device_process_label + def _get_merged_time_list(self, time_list, get_interval_time=False, display_name="computation_op", factor=1): """ Get merged time segment list. @@ -874,8 +912,6 @@ class BaseTimelineGenerator: "is not supported in offline parse mode.") parallel_mode = "data_parallel" stage_num = 1 - finally: - pass if stage_num > 1: parallel_mode = "pipeline-parallel" elif parallel_mode != "data_parallel": @@ -917,6 +953,12 @@ class BaseTimelineGenerator: logger.warning(f'Failed to save {cluster_analyse_file_path}. {err}') raise ProfilerIOException + def _register_op_name(self, timeline_list): + """Register op name to op name list.""" + for timeline in timeline_list: + if timeline and timeline[self._op_name_idx] not in self._op_name_list: + self._op_name_list.append(timeline[self._op_name_idx]) + class GpuTimelineGenerator(BaseTimelineGenerator): """Generate gpu Timeline data from file.""" @@ -929,8 +971,8 @@ class GpuTimelineGenerator(BaseTimelineGenerator): _cluster_analyse_filename = 'gpu_cluster_analyse_{}_{}_{}_{}.csv' _activity_keys_list = [] - def __init__(self, profiling_dir, device_id, rank_size): - super().__init__() + def __init__(self, profiling_dir, device_id, rank_size, model): + super().__init__(DeviceTarget.GPU.value, model) self._device_id = device_id self._rank_size = rank_size self._profiling_dir = profiling_dir @@ -1011,7 +1053,7 @@ class GpuTimelineGenerator(BaseTimelineGenerator): timeline_list, communication_info = self._load_op_data(op_file_path, reduce_op_type) communication_info.sort(key=lambda x: float(x[2])) # Add host cpu op timeline. - cpu_timeline_generator = CpuTimelineGenerator(self._profiling_dir, self._device_id, self._rank_size) + cpu_timeline_generator = CpuTimelineGenerator(self._profiling_dir, self._model) cpu_timeline_list = cpu_timeline_generator.load_cpu_op_data() if cpu_timeline_list: self._clock_synchronize_to_gpu(cpu_timeline_list) @@ -1309,8 +1351,6 @@ class GpuTimelineGenerator(BaseTimelineGenerator): computation_time.append(step_info[step][self._duration_idx] - comm_alone_time[step]) except IndexError as e: logger.error(e) - finally: - pass metrices_per_step_list = [computation_time, comm_alone_time, stage_time, recieve_alone_time, collective_comm_alone_time] @@ -1401,8 +1441,8 @@ class AscendTimelineGenerator(BaseTimelineGenerator): _timeline_summary_filename = 'ascend_timeline_summary_{}.json' _cluster_analyse_filename = 'ascend_cluster_analyse_{}_{}_{}_{}.csv' - def __init__(self, profiling_dir, device_id, rank_id, rank_size): - super().__init__() + def __init__(self, profiling_dir, device_id, rank_id, rank_size, model): + super().__init__(DeviceTarget.ASCEND.value, model) self._profiling_dir = profiling_dir self._device_id = device_id self._rank_id = rank_id @@ -1410,6 +1450,16 @@ class AscendTimelineGenerator(BaseTimelineGenerator): self._display_filename = self._display_filename.format(rank_id) self._timeline_summary_filename = self._timeline_summary_filename.format(rank_id) + @staticmethod + def _get_all_reduce_names(communication_info): + names = [] + for info in communication_info: + # all_reduce_name format: stream_stream_id_stream_op_index_opname + all_reduce_name = info[0][info[0].rindex('_') + 1:] + if all_reduce_name not in names: + names.append(all_reduce_name) + return names + def _parse_timeline_data(self, timeline, min_cycle_counter): """Parse timeline data.""" # factor to convert the time unit from 1ms to 1us for timeline display @@ -1440,16 +1490,6 @@ class AscendTimelineGenerator(BaseTimelineGenerator): self._update_format_meta_data(timeline_dict) self._timeline_meta.append(timeline_dict) - @staticmethod - def _get_all_reduce_names(communication_info): - names = [] - for info in communication_info: - # all_reduce_name format: stream_stream_id_stream_op_index_opname - all_reduce_name = info[0][info[0].rindex('_') + 1:] - if all_reduce_name not in names: - names.append(all_reduce_name) - return names - def _get_op_timeline(self, communication_info, source_path): """get ai_core and cpu timeline.""" all_reduce_names = AscendTimelineGenerator._get_all_reduce_names(communication_info) @@ -1457,7 +1497,7 @@ class AscendTimelineGenerator(BaseTimelineGenerator): for timeline in timeline_list: timeline[self._tid_idx] = f"Stream #{timeline[self._tid_idx]}" - cpu_timeline_generator = CpuTimelineGenerator(self._profiling_dir, self._rank_id, self._rank_size) + cpu_timeline_generator = CpuTimelineGenerator(self._profiling_dir, self._model) cpu_timeline_list = cpu_timeline_generator.get_timeline_data() if cpu_timeline_list: self._clock_synchronize_to_device(cpu_timeline_list, source_path) @@ -1706,8 +1746,6 @@ class AscendTimelineGenerator(BaseTimelineGenerator): computation_time.append(step_info[step][self._duration_idx] - comm_alone_time[step]) except IndexError as err: logger.error(err) - finally: - pass metrices_per_step_list = [computation_time, comm_alone_time, stage_time, recieve_alone_time, collective_comm_alone_time] if step_num > 1: @@ -1775,15 +1813,17 @@ class AscendTimelineGenerator(BaseTimelineGenerator): def init_pynative_timeline(self): """Init timeline for pynative model.""" timeline_list = OPIntermediateParser(self._profiling_dir, self._rank_id).get_timeline_data() - cpu_timeline_generator = CpuTimelineGenerator(self._profiling_dir, self._rank_id, self._rank_size) + cpu_timeline_generator = CpuTimelineGenerator(self._profiling_dir, self._model) cpu_timeline_list = cpu_timeline_generator.load_cpu_op_data() if cpu_timeline_list: self._pynative_clock_synchronize(cpu_timeline_list) timeline_list.extend(cpu_timeline_list) + self._register_op_name(timeline_list) self._timeline_summary['op_exe_times'] = len(timeline_list) self._max_scope_name_num = self._get_max_scope_name_num(timeline_list) self._timeline_summary['max_scope_name_num'] = self._max_scope_name_num + self._timeline_summary['num_of_ops'] = len(self._op_name_list) timeline_list.sort(key=lambda x: float(x[self._start_time_idx])) min_cycle_counter = float(timeline_list[0][self._start_time_idx]) @@ -1840,8 +1880,6 @@ class AscendTimelineGenerator(BaseTimelineGenerator): except (IOError, OSError) as err: logger.critical(f'Error occurred when read {start_time_file_path}: {err}') raise ProfilerIOException() - finally: - pass time_diff = gpu_start_time * 1000 - host_monotonic_start_time for idx, time_item in enumerate(timeline_list): timeline_list[idx][self._start_time_idx] = int(time_item[self._start_time_idx]) + time_diff @@ -1855,6 +1893,10 @@ class CpuTimelineGenerator(GpuTimelineGenerator): _display_filename = 'cpu_timeline_display_{}.json' _timeline_summary_filename = 'cpu_timeline_summary_{}.json' + def __init__(self, profiling_dir, model): + super().__init__(profiling_dir, 0, 0, model) + self._device_target = DeviceTarget.CPU.value + def _get_and_validate_path(self, file_name): """Generate op or activity file path from file name, and validate this path.""" file_path = os.path.join( diff --git a/mindspore/python/mindspore/profiler/parser/minddata_pipeline_parser.py b/mindspore/python/mindspore/profiler/parser/minddata_pipeline_parser.py index adf8502dd85..85cbd49c655 100644 --- a/mindspore/python/mindspore/profiler/parser/minddata_pipeline_parser.py +++ b/mindspore/python/mindspore/profiler/parser/minddata_pipeline_parser.py @@ -20,8 +20,8 @@ import stat from queue import Queue from mindspore.profiler.common.exceptions.exceptions import \ - ProfilerPathErrorException, ProfilerFileNotFoundException, \ - ProfilerDirNotFoundException, ProfilerRawFileException + ProfilerPathErrorException, ProfilerRawFileException, \ + ProfilerDirNotFoundException from mindspore import log as logger from mindspore.profiler.common.validator.validate_path import \ validate_and_normalize_path @@ -39,8 +39,6 @@ class MinddataPipelineParser: Raises: ProfilerPathErrorException: If the minddata pipeline file path or the output path is invalid. - ProfilerFileNotFoundException: If the minddata pipeline file or - the output dir does not exist. """ _raw_pipeline_file_name = 'pipeline_profiling_{}.json' _parsed_pipeline_file_name = 'minddata_pipeline_raw_{}.csv' @@ -73,6 +71,8 @@ class MinddataPipelineParser: ProfilerRawFileException: If fails to parse the raw file of minddata pipeline or the file is empty. """ + if not self._pipeline_path: + return with open(self._pipeline_path, 'r') as file: try: pipeline_info = json.load(file) @@ -113,7 +113,7 @@ class MinddataPipelineParser: logger.warning( 'The minddata pipeline file <%s> not found.', pipeline_path ) - raise ProfilerFileNotFoundException(pipeline_path) + pipeline_path = "" return pipeline_path diff --git a/mindspore/python/mindspore/profiler/profiling.py b/mindspore/python/mindspore/profiler/profiling.py index 910d87a973d..8031d6ceba5 100644 --- a/mindspore/python/mindspore/profiler/profiling.py +++ b/mindspore/python/mindspore/profiler/profiling.py @@ -34,7 +34,7 @@ from mindspore.profiler.common.validator.validate_path import \ from mindspore.profiler.parser.aicpu_data_parser import DataPreProcessParser from mindspore.profiler.parser.framework_parser import FrameworkParser from mindspore.profiler.parser.hwts_log_parser import HWTSLogParser -from mindspore.profiler.parser.integrator import Integrator +from mindspore.profiler.parser.integrator import Integrator, DeviceTarget from mindspore.profiler.parser.integrator import GpuTimelineGenerator, CpuTimelineGenerator, AscendTimelineGenerator from mindspore.profiler.parser.memory_usage_parser import MemoryUsageParser from mindspore.profiler.parser.minddata_parser import MinddataParser @@ -139,7 +139,8 @@ class Profiler: _aicpu_op_output_filename_target = "output_data_preprocess_aicpu_" _has_analysed = False _has_initialized = False - _ascend_profiling_options = {} + _ascend_profiling_options = "" + _ascend_job_id = "" def __init__(self, **kwargs): if Profiler._has_initialized: @@ -165,22 +166,28 @@ class Profiler: self._cpu_profiler = cpu_profiler.get_instance() self._cpu_profiler.init(self._output_path) - if self._device_target and self._device_target == "CPU": + if self._device_target and self._device_target == DeviceTarget.CPU.value: + if context.get_context("mode") == context.PYNATIVE_MODE: + raise RuntimeError("Pynative model is not supported on CPU currently.") + self.start_profile = kwargs.pop("start_profile", True) if not isinstance(self.start_profile, bool): raise TypeError(f"For '{self.__class__.__name__}', the parameter start_profile must be bool, " f"but got type {type(self.start_profile)}") - if self._device_target and self._device_target == "GPU": + if self._device_target and self._device_target == DeviceTarget.GPU.value: + if context.get_context("mode") == context.PYNATIVE_MODE: + raise RuntimeError("Pynative model is not supported on GPU currently.") + self._parse_parameter_for_gpu(**kwargs) + gpu_profiler = c_expression.GPUProfiler self._gpu_profiler = gpu_profiler.get_instance() self._gpu_profiler.init(self._output_path) if GlobalComm.WORLD_COMM_GROUP == "nccl_world_group": self._dev_id = str(get_rank()) os.environ['DEVICE_ID'] = self._dev_id - self._parse_parameter_for_gpu(**kwargs) - elif self._device_target and self._device_target == "Ascend": + elif self._device_target and self._device_target == DeviceTarget.ASCEND.value: self._init_time = int(time.time() * 10000000) logger.info("Profiling: profiling init time: %d", self._init_time) self._parse_parameter_for_ascend(**kwargs) @@ -196,6 +203,8 @@ class Profiler: if self.start_profile: self.start() + elif context.get_context("mode") == context.PYNATIVE_MODE: + raise RuntimeError("Pynative model does not support conditional collection of performance data.") def _construct_profiling_options(self): """ @@ -296,7 +305,7 @@ class Profiler: def _is_offline_parser(self): """Return whether offline parser or online parser.""" - if self._device_target and self._device_target == "Ascend": + if self._device_target and self._device_target == DeviceTarget.ASCEND.value: return bool(self._ascend_job_id) return False @@ -313,13 +322,13 @@ class Profiler: self._cpu_profiler.stop() - if self._device_target and self._device_target == "CPU": + if self._device_target and self._device_target == DeviceTarget.CPU.value: self._cpu_analyse() - if self._device_target and self._device_target == "GPU": + if self._device_target and self._device_target == DeviceTarget.GPU.value: self._gpu_analyse() - elif self._device_target and self._device_target == "Ascend": + elif self._device_target and self._device_target == DeviceTarget.ASCEND.value: self._ascend_analyse() logger.info("Profiling: all the data have been analyzed.") @@ -335,8 +344,12 @@ class Profiler: source_path = os.path.join(self._output_path, job_id) MinddataParser.execute(source_path, self._output_path, self._rank_id) + pipeline_parser = MinddataPipelineParser(self._output_path, self._rank_id, self._output_path) + logger.info("Profiling: analyzing the minddata pipeline operator and queue.") + pipeline_parser.parse() + timeline_analyser = AscendTimelineGenerator(self._output_path, self._dev_id, self._rank_id, - self._rank_size) + self._rank_size, context.get_context("mode")) timeline_analyser.init_pynative_timeline() size_limit = 100 * 1024 * 1024 # 100MB timeline_analyser.write_timeline(size_limit) @@ -571,9 +584,9 @@ class Profiler: self._md_profiler.start() self._cpu_profiler.step_profiling_enable(True) - if self._device_target and self._device_target == "GPU": + if self._device_target and self._device_target == DeviceTarget.GPU.value: self._gpu_profiler.step_profiling_enable(True) - elif self._device_target and self._device_target == "Ascend": + elif self._device_target and self._device_target == DeviceTarget.ASCEND.value: if context.get_context("mode") == context.PYNATIVE_MODE: self._ascend_pynative_start() else: @@ -645,9 +658,9 @@ class Profiler: self._md_profiler.stop() self._md_profiler.save(self._output_path) - if self._device_target and self._device_target == "GPU": + if self._device_target and self._device_target == DeviceTarget.GPU.value: self._gpu_profiler.stop() - elif self._device_target and self._device_target == "Ascend": + elif self._device_target and self._device_target == DeviceTarget.ASCEND.value: if context.get_context("mode") == context.PYNATIVE_MODE: self._pynative_profiler.stop() self._ascend_profiler.stop() @@ -725,7 +738,7 @@ class Profiler: try: size_limit = 100 * 1024 * 1024 # 100MB - timeline_generator = CpuTimelineGenerator(self._output_path, 0, 1) + timeline_generator = CpuTimelineGenerator(self._output_path, context.get_context("mode")) timeline_generator.init_timeline() timeline_generator.write_timeline(size_limit) timeline_generator.write_timeline_summary() @@ -746,7 +759,7 @@ class Profiler: """ logger.info("Begin to parse step trace.") # construct output path - dev_id = self._rank_id if self._device_target == "Ascend" else self._dev_id + dev_id = self._rank_id if self._device_target == DeviceTarget.ASCEND.value else self._dev_id step_trace_intermediate_file_path = os.path.join( self._output_path, f'step_trace_raw_{dev_id}_detail_time.csv' @@ -758,7 +771,7 @@ class Profiler: step_trace_intermediate_file_path = validate_and_normalize_path(step_trace_intermediate_file_path) point_info_file_path = validate_and_normalize_path(point_info_file_path) - if self._device_target and self._device_target == 'GPU': + if self._device_target and self._device_target == DeviceTarget.GPU.value: input_file_path = os.path.join(self._output_path, f'step_trace_profiling_{self._dev_id}.txt') input_file_path = validate_and_normalize_path(input_file_path) parser = GpuStepTraceParser(input_dir=input_file_path, @@ -799,7 +812,8 @@ class Profiler: optime_parser (OPComputeTimeParserParser): The parser instance for AI Core operator execution time calculation. """ - timeline_analyser = AscendTimelineGenerator(self._output_path, self._dev_id, self._rank_id, self._rank_size) + timeline_analyser = AscendTimelineGenerator(self._output_path, self._dev_id, self._rank_id, + self._rank_size, context.get_context("mode")) # Get framework info integrator = Integrator(self._output_path, self._rank_id) aicore_detail_data = integrator.get_aicore_detail_data() @@ -831,7 +845,8 @@ class Profiler: """Used for gpu, generate timeline info, write to json format file.""" try: size_limit = 100 * 1024 * 1024 # 100MB - timeline_generator = GpuTimelineGenerator(self._output_path, self._dev_id, self._rank_size) + timeline_generator = GpuTimelineGenerator(self._output_path, self._dev_id, self._rank_size, + context.get_context("mode")) timeline_generator.init_timeline(reduce_op_type) timeline_generator.write_timeline(size_limit) timeline_generator.write_timeline_summary() @@ -1010,7 +1025,7 @@ class Profiler: rank_id = "" try: dev_id = str(context.get_context("device_id")) - device_target = context.get_context("device_target") + device_target = context.get_context("device_target").lower() except ValueError as err: logger.error("Profiling: fail to get context, %s", err) @@ -1020,7 +1035,8 @@ class Profiler: dev_id = "0" logger.warning("Fail to get DEVICE_ID, use 0 instead.") - if device_target and device_target not in ["Ascend", "GPU", "CPU"]: + if device_target and device_target not in [DeviceTarget.ASCEND.value, DeviceTarget.GPU.value, + DeviceTarget.CPU.value]: msg = "Profiling: unsupported backend: %s" % device_target raise RuntimeError(msg) @@ -1031,7 +1047,7 @@ class Profiler: f"use 0 instead.") self._dev_id = dev_id - self._device_target = device_target + self._device_target = device_target.lower() self._rank_id = rank_id def _get_output_path(self, kwargs):