-
Notifications
You must be signed in to change notification settings - Fork 19
Expand file tree
/
Copy path05_preprocessing_demo.py
More file actions
163 lines (126 loc) · 5.4 KB
/
Copy path05_preprocessing_demo.py
File metadata and controls
163 lines (126 loc) · 5.4 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
"""
Demo: Data Preprocessing Operations
===================================
This demo showcases preprocessing functions in dskit.
"""
from dskit import auto_encode, auto_scale, train_test_auto
import pandas as pd
import numpy as np
def create_sample_data():
"""Create sample dataset for preprocessing"""
np.random.seed(42)
return pd.DataFrame({
'age': np.random.randint(18, 70, 200),
'salary': np.random.randint(30000, 150000, 200),
'experience': np.random.randint(0, 30, 200),
'department': np.random.choice(['IT', 'HR', 'Sales', 'Marketing'], 200),
'education': np.random.choice(['High School', 'Bachelor', 'Master', 'PhD'], 200),
'location': np.random.choice(['NYC', 'SF', 'LA', 'Chicago', 'Boston'], 200),
'performance': np.random.choice(['Low', 'Medium', 'High'], 200)
})
def demo_auto_encode():
"""Demo 1: Automatic encoding of categorical variables"""
print("=" * 60)
print("DEMO 1: Automatic Categorical Encoding")
print("=" * 60)
df = create_sample_data()
print("\n📊 Original data shape:", df.shape)
print("📊 Original columns:", list(df.columns))
print("\n📊 Categorical columns:")
cat_cols = df.select_dtypes(include='object').columns.tolist()
for col in cat_cols:
print(f" - {col}: {df[col].nunique()} unique values")
print("\n🔧 Applying automatic encoding...")
print(" (Uses One-Hot for low cardinality, Label for high cardinality)")
df_encoded = auto_encode(df, max_unique_for_onehot=10)
print("\n✓ Encoded data shape:", df_encoded.shape)
print("✓ New columns:", list(df_encoded.columns))
print(f"✓ Added {df_encoded.shape[1] - df.shape[1]} new columns")
print("\n📊 Sample of encoded data:")
print(df_encoded.head())
def demo_auto_scale():
"""Demo 2: Automatic feature scaling"""
print("\n" + "=" * 60)
print("DEMO 2: Automatic Feature Scaling")
print("=" * 60)
df = create_sample_data()
df_encoded = auto_encode(df)
print("\n📊 Original numeric ranges:")
numeric_cols = df_encoded.select_dtypes(include=[np.number]).columns[:3]
for col in numeric_cols:
print(f" {col}: [{df_encoded[col].min():.2f}, {df_encoded[col].max():.2f}]")
# Standard scaling
print("\n🔧 Applying Standard Scaling...")
df_standard = auto_scale(df_encoded, method='standard')
print("\n✓ After Standard Scaling:")
for col in numeric_cols:
print(f" {col}: mean={df_standard[col].mean():.4f}, std={df_standard[col].std():.4f}")
# MinMax scaling
print("\n🔧 Applying MinMax Scaling...")
df_minmax = auto_scale(df_encoded, method='minmax')
print("\n✓ After MinMax Scaling:")
for col in numeric_cols:
print(f" {col}: [{df_minmax[col].min():.2f}, {df_minmax[col].max():.2f}]")
# Robust scaling
print("\n🔧 Applying Robust Scaling...")
df_robust = auto_scale(df_encoded, method='robust')
print("\n✓ After Robust Scaling:")
for col in numeric_cols:
print(f" {col}: median={df_robust[col].median():.4f}")
def demo_train_test_split():
"""Demo 3: Automatic train-test splitting"""
print("\n" + "=" * 60)
print("DEMO 3: Train-Test Split")
print("=" * 60)
df = create_sample_data()
df_encoded = auto_encode(df)
df_scaled = auto_scale(df_encoded)
print("\n📊 Full dataset shape:", df_scaled.shape)
print("\n🔧 Splitting data (80-20 split)...")
X_train, X_test, y_train, y_test = train_test_auto(
df_scaled,
target='performance',
test_size=0.2,
random_state=42
)
print("\n✓ Split completed:")
print(f" Training set: {X_train.shape[0]} samples, {X_train.shape[1]} features")
print(f" Test set: {X_test.shape[0]} samples, {X_test.shape[1]} features")
print(f" Target distribution in train: {y_train.value_counts().to_dict()}")
print(f" Target distribution in test: {y_test.value_counts().to_dict()}")
def demo_complete_pipeline():
"""Demo 4: Complete preprocessing pipeline"""
print("\n" + "=" * 60)
print("DEMO 4: Complete Preprocessing Pipeline")
print("=" * 60)
print("\n📊 Starting with raw data...")
df = create_sample_data()
print(f" Shape: {df.shape}")
print("\n🔧 Step 1: Encoding categorical variables...")
df_encoded = auto_encode(df, max_unique_for_onehot=10)
print(f" ✓ Shape after encoding: {df_encoded.shape}")
print("\n🔧 Step 2: Scaling features...")
df_scaled = auto_scale(df_encoded, method='standard')
print(f" ✓ Shape after scaling: {df_scaled.shape}")
print("\n🔧 Step 3: Splitting into train/test...")
X_train, X_test, y_train, y_test = train_test_auto(
df_scaled,
target='performance',
test_size=0.2,
random_state=42
)
print(f" ✓ Train set: {X_train.shape}")
print(f" ✓ Test set: {X_test.shape}")
print("\n✅ Preprocessing pipeline completed!")
print(" Data is now ready for modeling")
if __name__ == "__main__":
print("\n" + "⚙️" * 30)
print("PREPROCESSING OPERATIONS DEMO".center(60))
print("⚙️" * 30 + "\n")
demo_auto_encode()
demo_auto_scale()
demo_train_test_split()
demo_complete_pipeline()
print("\n" + "✅" * 30)
print("ALL DEMOS COMPLETED".center(60))
print("✅" * 30 + "\n")