File size: 7,443 Bytes
c33c303
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
"""
===============================================================================
config.py β€” Central Configuration for the Machine Sound Classification Pipeline
===============================================================================

PURPOSE:
    This file is the SINGLE SOURCE OF TRUTH for every hyperparameter,
    file path, and constant used across the project. Every team member
    should import values from here instead of hard-coding numbers.

    If you need to change a value (e.g. sample rate, n_mels, batch size),
    change it HERE and it propagates everywhere automatically.

OWNER: Shared (everyone imports from here)
===============================================================================
"""

import os

# =============================================================================
# 1. PATH CONFIGURATION
# =============================================================================
# Root directory of the project (where this file lives)
PROJECT_ROOT = os.path.dirname(os.path.abspath(__file__))

# Directory where the examiner places test .wav files
# CRITICAL: infer.py reads from this exact folder
DATA_DIR = os.path.join(PROJECT_ROOT, "data")

# Directory for the training dataset (organized by class)
TRAIN_DATA_DIR = os.path.join(PROJECT_ROOT, "train_data")

PROCESSED_DATA_DIR = os.path.join(PROJECT_ROOT, "processed_features")

# Directory where trained model checkpoints are saved
CHECKPOINT_DIR = os.path.join(PROJECT_ROOT, "checkpoints")

# Output files required by the submission
RESULTS_FILE = os.path.join(PROJECT_ROOT, "results.txt")
TIME_FILE = os.path.join(PROJECT_ROOT, "time.txt")

# Directory for split metadata (train/val/test file lists)
SPLITS_DIR = os.path.join(PROJECT_ROOT, "splits")


# =============================================================================
# 2. AUDIO PREPROCESSING CONSTANTS
# =============================================================================
# Target sampling rate β€” all audio is resampled to this before processing.
# 16 kHz is standard for machine sound analysis; Nyquist limit = 8 kHz.
# Owner: JSON (resampling.py)
TARGET_SR = 16000

# Silence removal threshold in dB.
# Frames quieter than this (relative to peak) are considered silence.
# 20 dB is a safe starting point for factory recordings.
# Owner: EL sir (silence_removal.py)
SILENCE_TOP_DB = 40

# Noise reduction β€” duration (in seconds) of the noise profile sample.
# We estimate the noise floor from the first N seconds of each clip.
# Owner: EL sir (noise_reduction.py)
NOISE_PROFILE_DURATION = 0.5  # seconds


# =============================================================================
# 3. MEL SPECTROGRAM PARAMETERS
# =============================================================================
# These MUST be agreed upon by sala7 (feature extraction) and EL sir (CNN input).
# Changing n_mels here changes the CNN input height automatically.
# Owner: sala7 (mel_spectrogram.py) β€” coordinated with EL sir

# Number of mel filter banks (height of the spectrogram "image")
N_MELS = 128

# FFT window size β€” 1024 samples @ 16 kHz = 64 ms analysis window.
# For machine fault detection, 64 ms captures one full rotation of many motors.
N_FFT = 1024

# Hop length β€” 512 samples = 50% overlap between consecutive frames.
# Good time resolution without quadrupling computation.
HOP_LENGTH = 512

# Maximum frequency for the mel filterbank.
# At 16 kHz sampling rate, Nyquist = 8 kHz, so fmax = 8000.
FMAX = 8000

# Power for the mel spectrogram (2.0 = power spectrogram)
MEL_POWER = 2.0


# =============================================================================
# 4. FIXED-SIZE TENSOR PARAMETERS
# =============================================================================
# After silence removal, audio clips vary in length. We must pad/trim
# spectrograms to a uniform time dimension for batching.
#
# CALCULATION:
#   Original audio β‰ˆ 11 seconds β†’ 11 * 16000 = 176,000 samples
#   Time frames = ceil(176000 / 512) = 344 frames (full 11s)
#   After silence trimming, clips are shorter β†’ 256 frames β‰ˆ 8.2 seconds
#   is a reasonable target that captures the machine sound while
#   discarding silence.
#

# Final input shape to the CNN: (batch, channels, n_mels, time_frames)
# channels = 1 (grayscale spectrogram)
CNN_INPUT_CHANNELS = 1


# =============================================================================
# 5. DATA SPLIT RATIOS
# =============================================================================
# Stratified split ratios β€” every class appears in every split at the same proportion.
# Owner: JSON (splits.py)
TRAIN_RATIO = 0.70
VAL_RATIO = 0.15
TEST_RATIO = 0.15

# Random seed for reproducibility across all random operations
RANDOM_SEED = 42


# =============================================================================
# 6. MODEL ARCHITECTURE PARAMETERS
# =============================================================================
# Number of output classes:
#   0 = Machine 1 Normal, 1 = Machine 1 Abnormal,
#   2 = Machine 2 Normal, 3 = Machine 2 Abnormal,
#   4 = Machine 3 Normal, 5 = Machine 3 Abnormal
# Owner: EL sir (cnn.py)
NUM_CLASSES = 6

# Convolutional layer filter counts (depth progression)
# Each successive layer doubles the filters to capture more complex patterns.
CNN_FILTERS = [32, 64, 128, 256]

# Kernel size for all Conv2D layers
CNN_KERNEL_SIZE = 3

# Padding for Conv2D layers (1 = 'same' padding with kernel_size=3)
CNN_PADDING = 1

# Pool size for MaxPool2d layers
CNN_POOL_SIZE = 2

# Output size of AdaptiveAvgPool2d before the classifier head
# This makes the model accept any time-length input gracefully.
ADAPTIVE_POOL_OUTPUT = (4, 4)


# =============================================================================
# 7. TRAINING HYPERPARAMETERS
# =============================================================================
# Owner: Osama (trainer.py)

# Optimizer: AdamW (corrects weight decay application vs vanilla Adam)
LEARNING_RATE = 1e-3
WEIGHT_DECAY = 1e-4

# Batch size for training DataLoader
BATCH_SIZE = 64

# Maximum number of training epochs
MAX_EPOCHS = 100

# Early stopping β€” stop if val loss doesn't improve for this many epochs
EARLY_STOPPING_PATIENCE = 10

# Learning rate scheduler β€” cosine annealing
LR_SCHEDULER_T_MAX = MAX_EPOCHS  # period of the cosine cycle

# Number of DataLoader workers for parallel data loading
NUM_WORKERS = 8


# =============================================================================
# 8. AUGMENTATION PARAMETERS
# =============================================================================
# Owner: sala7 (augmentation.py)
# These are applied ONLY during training (not val/test).

# SpecAugment: number of frequency bands to mask
FREQ_MASK_PARAM = 20

# SpecAugment: number of time steps to mask
TIME_MASK_PARAM = 30

# Gaussian noise injection β€” standard deviation
NOISE_STD = 0.005

# Probability of applying each augmentation
AUGMENT_PROB = 0.5


# =============================================================================
# 9. INFERENCE PARAMETERS
# =============================================================================
# Path to the best saved model checkpoint (used by infer.py)
BEST_MODEL_PATH = os.path.join(CHECKPOINT_DIR, "best_model.pth")

# Device selection for inference (auto-detect GPU)
import torch
DEVICE = torch.device("cuda" if torch.cuda.is_available() else "cpu")