jiaxwang commited on
Commit
e01203e
·
verified ·
1 Parent(s): dad0bf7

Delete quark_profile.yaml

Browse files
Files changed (1) hide show
  1. quark_profile.yaml +0 -176
quark_profile.yaml DELETED
@@ -1,176 +0,0 @@
1
- # Quark Profiling Results
2
-
3
- memory_usage:
4
- - step: "Start"
5
- timestamp: 1775601224.862912
6
- relative_time_secs: 0.0
7
- cpu_memory_mb: 2576.44
8
- gpu_memory_mb: 3095.88
9
- disk_read_mb: 0.0
10
- disk_write_mb: 0.0
11
- - step: "Model Loading Start"
12
- timestamp: 1775601225.3419523
13
- relative_time_secs: 0.47904038429260254
14
- cpu_memory_mb: 2576.44
15
- gpu_memory_mb: 3095.88
16
- disk_read_mb: 0.0
17
- disk_write_mb: 0.0
18
- - step: "Model Loading End"
19
- timestamp: 1775601355.0208464
20
- relative_time_secs: 130.15793442726135
21
- cpu_memory_mb: 9506.13
22
- gpu_memory_mb: 676866.96
23
- disk_read_mb: 0.0
24
- disk_write_mb: 0.0
25
- - step: "Dataset Loading Start"
26
- timestamp: 1775601361.3078878
27
- relative_time_secs: 136.4449758529663
28
- cpu_memory_mb: 14503.83
29
- gpu_memory_mb: 679259.66
30
- disk_read_mb: 0.0
31
- disk_write_mb: 1137.96
32
- - step: "Dataset Loading End"
33
- timestamp: 1775601363.4968152
34
- relative_time_secs: 138.6339032649994
35
- cpu_memory_mb: 14515.22
36
- gpu_memory_mb: 679260.05
37
- disk_read_mb: 0.0
38
- disk_write_mb: 1137.96
39
- - step: "Model Quantization Start"
40
- timestamp: 1775601364.0407772
41
- relative_time_secs: 139.17786526679993
42
- cpu_memory_mb: 14515.22
43
- gpu_memory_mb: 679260.17
44
- disk_read_mb: 0.0
45
- disk_write_mb: 1137.96
46
- - step: "Model Preparation Start"
47
- timestamp: 1775601364.4628615
48
- relative_time_secs: 139.59994959831238
49
- cpu_memory_mb: 14515.22
50
- gpu_memory_mb: 679260.46
51
- disk_read_mb: 0.0
52
- disk_write_mb: 1137.96
53
- - step: "Model Preparation End"
54
- timestamp: 1775601376.496135
55
- relative_time_secs: 151.6332230567932
56
- cpu_memory_mb: 21922.45
57
- gpu_memory_mb: 679267.33
58
- disk_read_mb: 0.0
59
- disk_write_mb: 1137.96
60
- - step: "Advanced Algorithms Start"
61
- timestamp: 1775601376.9244616
62
- relative_time_secs: 152.0615496635437
63
- cpu_memory_mb: 21922.45
64
- gpu_memory_mb: 679267.33
65
- disk_read_mb: 0.0
66
- disk_write_mb: 1137.96
67
- - step: "Advanced Algorithms End"
68
- timestamp: 1775601377.3558223
69
- relative_time_secs: 152.49291038513184
70
- cpu_memory_mb: 21922.45
71
- gpu_memory_mb: 679267.33
72
- disk_read_mb: 0.0
73
- disk_write_mb: 1137.96
74
- - step: "Calibration Start"
75
- timestamp: 1775601377.8195803
76
- relative_time_secs: 152.9566683769226
77
- cpu_memory_mb: 21922.45
78
- gpu_memory_mb: 679267.33
79
- disk_read_mb: 0.0
80
- disk_write_mb: 1137.96
81
- - step: "Calibration End"
82
- timestamp: 1775601445.2248025
83
- relative_time_secs: 220.3618905544281
84
- cpu_memory_mb: 22447.09
85
- gpu_memory_mb: 803600.8
86
- disk_read_mb: 0.0
87
- disk_write_mb: 1184.3
88
- - step: "Model Quantization End"
89
- timestamp: 1775601445.7345195
90
- relative_time_secs: 220.87160754203796
91
- cpu_memory_mb: 22447.09
92
- gpu_memory_mb: 803600.8
93
- disk_read_mb: 0.0
94
- disk_write_mb: 1184.3
95
- - step: "Freeze Model Start"
96
- timestamp: 1775601446.1352663
97
- relative_time_secs: 221.27235436439514
98
- cpu_memory_mb: 22447.09
99
- gpu_memory_mb: 803600.8
100
- disk_read_mb: 0.0
101
- disk_write_mb: 1184.3
102
- - step: "Freeze Model End"
103
- timestamp: 1775601455.716045
104
- relative_time_secs: 230.85313296318054
105
- cpu_memory_mb: 35313.63
106
- gpu_memory_mb: 803877.68
107
- disk_read_mb: 0.0
108
- disk_write_mb: 1184.34
109
- - step: "Export HF Safetensors Start"
110
- timestamp: 1775601456.2258182
111
- relative_time_secs: 231.36290621757507
112
- cpu_memory_mb: 35313.63
113
- gpu_memory_mb: 803871.5
114
- disk_read_mb: 0.0
115
- disk_write_mb: 1184.34
116
- - step: "Export HF Safetensors End"
117
- timestamp: 1775601559.6133552
118
- relative_time_secs: 334.75044322013855
119
- cpu_memory_mb: 41515.62
120
- gpu_memory_mb: 803513.64
121
- disk_read_mb: 5.55
122
- disk_write_mb: 200606.78
123
- - step: "Model Evaluation Start"
124
- timestamp: 1775601560.083004
125
- relative_time_secs: 335.22009205818176
126
- cpu_memory_mb: 41515.62
127
- gpu_memory_mb: 803513.64
128
- disk_read_mb: 5.55
129
- disk_write_mb: 200606.78
130
- - step: "Model Evaluation End"
131
- timestamp: 1775601750.331539
132
- relative_time_secs: 525.4686269760132
133
- cpu_memory_mb: 42509.07
134
- gpu_memory_mb: 824351.98
135
- disk_read_mb: 17.29
136
- disk_write_mb: 200614.07
137
- - step: "End"
138
- timestamp: 1775601751.196488
139
- relative_time_secs: 526.333575963974
140
- cpu_memory_mb: 41731.62
141
- gpu_memory_mb: 824351.67
142
- disk_read_mb: 17.48
143
- disk_write_mb: 200614.07
144
-
145
- # Summary Metrics
146
- total_quantization_time_seconds: 526.3336
147
- peak_memory_mb: 42200.82
148
- peak_gpu_memory_mb: 824351.98
149
- total_disk_read_mb: 17.48
150
- total_disk_write_mb: 200614.07
151
-
152
- # Metric Definitions:
153
- #
154
- # Checkpoint Metrics (per record):
155
- # - step: Name of the profiling checkpoint. Common steps include:
156
- # - "Start": Initial state when profiling begins
157
- # - "Model Loaded": After loading the ONNX model into memory
158
- # - "Pre-process Start/End": Before and after model preprocessing
159
- # - "Calibration Start/End": Before and after calibration data collection
160
- # - "Quantization (MatMulNBits) Start/End": MatMulNBits quantization phase
161
- # - "Quantization (Static) Start/End": Static quantization phase
162
- # - "Post-process Start/End": Before and after post-processing
163
- # - "Fast Finetune Start/End": Before and after fast finetuning (if enabled)
164
- # - timestamp: Unix timestamp (seconds since epoch) when this measurement was taken. Useful for correlating with external logs or events.
165
- # - relative_time_secs: Time elapsed (in seconds) since the "Start" step. Useful for understanding the duration of each phase relative to the beginning of profiling.
166
- # - cpu_memory_mb: Current Resident Set Size (RSS) in megabytes at this step. This includes memory from the main process and all child processes. RSS represents the portion of memory held in RAM (not swapped out).
167
- # - gpu_memory_mb: Current GPU memory usage in megabytes. This represents actual GPU memory used by the process, including allocations from PyTorch, ONNX Runtime, TensorRT, and other frameworks. Only available when PyTorch with CUDA/ROCm is installed and GPU is available.
168
- # - disk_read_mb: Cumulative disk bytes read (in megabytes) since the start of profiling. Measured relative to the baseline captured at the 'Start' checkpoint, including I/O from the main process and all child processes. Only available when psutil is installed and the OS exposes per-process I/O counters (Linux /proc/<pid>/io, Windows; not available on macOS without root).
169
- # - disk_write_mb: Cumulative disk bytes written (in megabytes) since the start of profiling. Measured relative to the baseline captured at the 'Start' checkpoint, including I/O from the main process and all child processes. Only available when psutil is installed and the OS exposes per-process I/O counters (Linux /proc/<pid>/io, Windows; not available on macOS without root).
170
- #
171
- # Summary Metrics (overall):
172
- # - total_quantization_time_seconds: Total elapsed time (in seconds) from the start of profiling to the end of the quantization process.
173
- # - peak_memory_mb: Peak resident set size (RSS) in megabytes for the main process during the entire profiling session. On Linux, this is read from VmHWM (high water mark) in /proc/<pid>/status. On Windows, this is the peak working set size. This metric may not be available on all platforms.
174
- # - peak_gpu_memory_mb: Peak GPU memory usage in megabytes during the entire profiling session. This is the maximum GPU memory used, including allocations from PyTorch, ONNX Runtime, TensorRT, and other frameworks. Only available when PyTorch with CUDA/ROCm is installed and GPU is available.
175
- # - total_disk_read_mb: Total disk bytes read (in megabytes) during the entire profiling session. Computed as the difference between the final and baseline cumulative read counters, including I/O from the main process and all child processes. Only available when psutil is installed and the OS exposes per-process I/O counters (Linux /proc/<pid>/io, Windows; not available on macOS without root).
176
- # - total_disk_write_mb: Total disk bytes written (in megabytes) during the entire profiling session. Computed as the difference between the final and baseline cumulative write counters, including I/O from the main process and all child processes. Only available when psutil is installed and the OS exposes per-process I/O counters (Linux /proc/<pid>/io, Windows; not available on macOS without root).