linzhao-amd commited on
Commit
1d7594a
·
verified ·
1 Parent(s): 014fc7f

Delete quark_profile.yaml

Browse files
Files changed (1) hide show
  1. quark_profile.yaml +0 -177
quark_profile.yaml DELETED
@@ -1,177 +0,0 @@
1
- # Quark Profiling Results
2
-
3
- memory_usage:
4
- - step: "Start"
5
- timestamp: 1778206913.5518243
6
- relative_time_secs: 0.0
7
- cpu_memory_mb: 4589.59
8
- gpu_memory_mb: 777.22
9
- disk_read_mb: 0.0
10
- disk_write_mb: 0.0
11
- - step: "Model Loading Start"
12
- timestamp: 1778206913.7514918
13
- relative_time_secs: 0.19966745376586914
14
- cpu_memory_mb: 4589.59
15
- gpu_memory_mb: 777.22
16
- disk_read_mb: 0.0
17
- disk_write_mb: 0.0
18
- - step: "Model Loading End"
19
- timestamp: 1778207378.5582051
20
- relative_time_secs: 465.0063807964325
21
- cpu_memory_mb: 6537.77
22
- gpu_memory_mb: 63960.58
23
- disk_read_mb: 436296.35
24
- disk_write_mb: 0.09
25
- - step: "Dataset Loading Start"
26
- timestamp: 1778207394.8392925
27
- relative_time_secs: 481.28746819496155
28
- cpu_memory_mb: 27673.3
29
- gpu_memory_mb: 65073.95
30
- disk_read_mb: 437001.61
31
- disk_write_mb: 8369.02
32
- - step: "Dataset Loading End"
33
- timestamp: 1778207398.4055986
34
- relative_time_secs: 484.8537743091583
35
- cpu_memory_mb: 27678.77
36
- gpu_memory_mb: 65073.95
37
- disk_read_mb: 437349.88
38
- disk_write_mb: 8369.02
39
- - step: "Model Quantization Start"
40
- timestamp: 1778207398.6073127
41
- relative_time_secs: 485.0554883480072
42
- cpu_memory_mb: 27678.77
43
- gpu_memory_mb: 65073.95
44
- disk_read_mb: 437349.88
45
- disk_write_mb: 8369.02
46
- - step: "Model Preparation Start"
47
- timestamp: 1778207398.749175
48
- relative_time_secs: 485.19735074043274
49
- cpu_memory_mb: 27678.77
50
- gpu_memory_mb: 65073.95
51
- disk_read_mb: 437349.88
52
- disk_write_mb: 8369.02
53
- - step: "Model Preparation End"
54
- timestamp: 1778207436.6778612
55
- relative_time_secs: 523.1260368824005
56
- cpu_memory_mb: 35183.42
57
- gpu_memory_mb: 65119.96
58
- disk_read_mb: 437349.88
59
- disk_write_mb: 8369.02
60
- - step: "Advanced Algorithms Start"
61
- timestamp: 1778207436.84824
62
- relative_time_secs: 523.2964155673981
63
- cpu_memory_mb: 35183.42
64
- gpu_memory_mb: 65119.96
65
- disk_read_mb: 437349.88
66
- disk_write_mb: 8369.02
67
- - step: "Advanced Algorithms End"
68
- timestamp: 1778207437.0057657
69
- relative_time_secs: 523.4539413452148
70
- cpu_memory_mb: 35183.42
71
- gpu_memory_mb: 65119.96
72
- disk_read_mb: 437349.88
73
- disk_write_mb: 8369.02
74
- - step: "Calibration Start"
75
- timestamp: 1778207437.273917
76
- relative_time_secs: 523.722092628479
77
- cpu_memory_mb: 35183.42
78
- gpu_memory_mb: 65119.96
79
- disk_read_mb: 437349.88
80
- disk_write_mb: 8369.02
81
- - step: "Calibration End"
82
- timestamp: 1778210047.9661236
83
- relative_time_secs: 3134.414299249649
84
- cpu_memory_mb: 36125.95
85
- gpu_memory_mb: 87323.7
86
- disk_read_mb: 437351.9
87
- disk_write_mb: 8399.16
88
- - step: "Model Quantization End"
89
- timestamp: 1778210176.061249
90
- relative_time_secs: 3262.509424686432
91
- cpu_memory_mb: 107186.82
92
- gpu_memory_mb: 87275.89
93
- disk_read_mb: 437352.28
94
- disk_write_mb: 8399.16
95
- - step: "Freeze Model Start"
96
- timestamp: 1778210176.2332122
97
- relative_time_secs: 3262.681387901306
98
- cpu_memory_mb: 107186.82
99
- gpu_memory_mb: 87275.89
100
- disk_read_mb: 437352.28
101
- disk_write_mb: 8399.16
102
- - step: "Freeze Model End"
103
- timestamp: 1778210190.9314125
104
- relative_time_secs: 3277.3795881271362
105
- cpu_memory_mb: 107521.27
106
- gpu_memory_mb: 87275.89
107
- disk_read_mb: 437352.28
108
- disk_write_mb: 8399.16
109
- - step: "Export HF Safetensors Start"
110
- timestamp: 1778210191.0775836
111
- relative_time_secs: 3277.5257592201233
112
- cpu_memory_mb: 107521.52
113
- gpu_memory_mb: 87275.89
114
- disk_read_mb: 437352.28
115
- disk_write_mb: 8399.16
116
- - step: "Export HF Safetensors End"
117
- timestamp: 1778210393.5681884
118
- relative_time_secs: 3480.016364097595
119
- cpu_memory_mb: 112519.61
120
- gpu_memory_mb: 86791.54
121
- disk_read_mb: 437357.31
122
- disk_write_mb: 136647.9
123
- - step: "Model Evaluation Start"
124
- timestamp: 1778210393.7849188
125
- relative_time_secs: 3480.2330944538116
126
- cpu_memory_mb: 112519.61
127
- gpu_memory_mb: 86791.54
128
- disk_read_mb: 437357.31
129
- disk_write_mb: 136647.9
130
- - step: "Model Evaluation End"
131
- timestamp: 1778214657.600004
132
- relative_time_secs: 7744.048179626465
133
- cpu_memory_mb: 113494.72
134
- gpu_memory_mb: 87319.95
135
- disk_read_mb: 437365.14
136
- disk_write_mb: 136653.94
137
- - step: "End"
138
- timestamp: 1778214657.857704
139
- relative_time_secs: 7744.3058795928955
140
- cpu_memory_mb: 113492.76
141
- gpu_memory_mb: 87319.95
142
- disk_read_mb: 437365.14
143
- disk_write_mb: 136653.94
144
-
145
- # Summary Metrics
146
- total_quantization_time_seconds: 7744.3059
147
- peak_memory_mb: 113495.66
148
- peak_gpu_memory_mb: 87323.7
149
- total_disk_read_mb: 437365.14
150
- total_disk_write_mb: 136653.94
151
-
152
- # Metric Definitions:
153
- #
154
- # Checkpoint Metrics (per record):
155
- # - step: Name of the profiling checkpoint. Common steps include:
156
- # - "Start": Initial state when profiling begins
157
- # - "Model Loaded": After loading the ONNX model into memory
158
- # - "Pre-process Start/End": Before and after model preprocessing
159
- # - "Calibration Start/End": Before and after calibration data collection
160
- # - "Quantization (MatMulNBits) Start/End": MatMulNBits quantization phase
161
- # - "Quantization (Static) Start/End": Static quantization phase
162
- # - "Post-process Start/End": Before and after post-processing
163
- # - "Fast Finetune Start/End": Before and after fast finetuning (if enabled)
164
- # - timestamp: Unix timestamp (seconds since epoch) when this measurement was taken. Useful for correlating with external logs or events.
165
- # - relative_time_secs: Time elapsed (in seconds) since the "Start" step. Useful for understanding the duration of each phase relative to the beginning of profiling.
166
- # - cpu_memory_mb: Current Resident Set Size (RSS) in megabytes at this step. This includes memory from the main process and all child processes. RSS represents the portion of memory held in RAM (not swapped out).
167
- # - gpu_memory_mb: Current GPU memory usage in megabytes. This represents actual GPU memory used by the process, including allocations from PyTorch, ONNX Runtime, TensorRT, and other frameworks. Only available when PyTorch with CUDA/ROCm is installed and GPU is available.
168
- # - disk_read_mb: Cumulative disk bytes read (in megabytes) since the start of profiling. Measured relative to the baseline captured at the 'Start' checkpoint, including I/O from the main process and all child processes. Only available when psutil is installed and the OS exposes per-process I/O counters (Linux /proc/<pid>/io, Windows; not available on macOS without root).
169
- # - disk_write_mb: Cumulative disk bytes written (in megabytes) since the start of profiling. Measured relative to the baseline captured at the 'Start' checkpoint, including I/O from the main process and all child processes. Only available when psutil is installed and the OS exposes per-process I/O counters (Linux /proc/<pid>/io, Windows; not available on macOS without root).
170
- #
171
- # Summary Metrics (overall):
172
- # - total_quantization_time_seconds: Total elapsed time (in seconds) from the start of profiling to the end of the quantization process.
173
- # - peak_memory_mb: Peak resident set size (RSS) in megabytes for the main process during the entire profiling session. On Linux, this is read from VmHWM (high water mark) in /proc/<pid>/status. On Windows, this is the peak working set size. This metric may not be available on all platforms.
174
- # - peak_gpu_memory_mb: Peak GPU memory usage in megabytes during the entire profiling session. This is the maximum GPU memory used, including allocations from PyTorch, ONNX Runtime, TensorRT, and other frameworks. Only available when PyTorch with CUDA/ROCm is installed and GPU is available.
175
- # - total_disk_read_mb: Total disk bytes read (in megabytes) during the entire profiling session. Computed as the difference between the final and baseline cumulative read counters, including I/O from the main process and all child processes. Only available when psutil is installed and the OS exposes per-process I/O counters (Linux /proc/<pid>/io, Windows; not available on macOS without root).
176
- # - total_disk_write_mb: Total disk bytes written (in megabytes) during the entire profiling session. Computed as the difference between the final and baseline cumulative write counters, including I/O from the main process and all child processes. Only available when psutil is installed and the OS exposes per-process I/O counters (Linux /proc/<pid>/io, Windows; not available on macOS without root).
177
- # - peak_cache_dir_disk_usage_mb: Highest peak increase in disk usage (in megabytes) among all cache directories created during the profiling session, relative to each cache directory's size when monitoring started. Sampled every 1 second by recursively summing file sizes with os.scandir().