mrhammad12/y_data_profiling
0
1import streamlit as st
2import pandas as pd
3import requests
4import time
5import streamlit.components.v1 as components
6import sys
7import os
8import json
9
10# Add utils to path
11sys.path.append(os.path.join(os.path.dirname(__file__), 'utils'))
12
13try:
14 from data_loader import load_titanic_data, load_iris_data
15except ImportError:
16 # Fallback functions
17 def load_titanic_data():
18 url = "https://raw.githubusercontent.com/datasciencedojo/datasets/master/titanic.csv"
19 return pd.read_csv(url)
20
21 def load_iris_data():
22 url = "https://raw.githubusercontent.com/mwaskom/seaborn-data/master/iris.csv"
23 return pd.read_csv(url)
24
25# Page configuration
26st.set_page_config(
27 page_title="Data Profiler Pro",
28 page_icon="๐",
29 layout="wide",
30 initial_sidebar_state="expanded"
31)
32
33# Custom CSS for green theme and animations
34st.markdown("""
35<style>
36 .main-header {
37 font-size: 3rem;
38 color: #2E8B57;
39 text-align: center;
40 margin-bottom: 2rem;
41 font-weight: bold;
42 background: linear-gradient(90deg, #2E8B57, #3CB371);
43 -webkit-background-clip: text;
44 -webkit-text-fill-color: transparent;
45 animation: fadeIn 2s ease-in;
46 }
47
48 .sub-header {
49 font-size: 1.5rem;
50 color: #228B22;
51 margin-bottom: 1rem;
52 font-weight: 600;
53 }
54
55 .stButton button {
56 background: linear-gradient(135deg, #2E8B57, #32CD32);
57 color: white;
58 border: none;
59 padding: 0.5rem 2rem;
60 border-radius: 25px;
61 font-weight: bold;
62 transition: all 0.3s ease;
63 box-shadow: 0 4px 15px 0 rgba(46, 139, 87, 0.3);
64 }
65
66 .stButton button:hover {
67 transform: translateY(-2px);
68 box-shadow: 0 6px 20px 0 rgba(46, 139, 87, 0.5);
69 background: linear-gradient(135deg, #228B22, #2E8B57);
70 }
71
72 .dataset-card {
73 background: linear-gradient(135deg, #F8FFF8, #E8F5E8);
74 border-radius: 15px;
75 padding: 1.5rem;
76 margin: 1rem 0;
77 border-left: 5px solid #2E8B57;
78 box-shadow: 0 4px 15px 0 rgba(46, 139, 87, 0.1);
79 transition: all 0.3s ease;
80 }
81
82 .dataset-card:hover {
83 transform: translateY(-3px);
84 box-shadow: 0 6px 25px 0 rgba(46, 139, 87, 0.2);
85 }
86
87 @keyframes fadeIn {
88 from { opacity: 0; transform: translateY(-20px); }
89 to { opacity: 1; transform: translateY(0); }
90 }
91
92 @keyframes pulse {
93 0% { transform: scale(1); }
94 50% { transform: scale(1.05); }
95 100% { transform: scale(1); }
96 }
97
98 .pulse-animation {
99 animation: pulse 2s infinite;
100 }
101
102 .success-message {
103 background: linear-gradient(135deg, #90EE90, #98FB98);
104 color: #006400;
105 padding: 1rem;
106 border-radius: 10px;
107 border-left: 5px solid #32CD32;
108 margin: 1rem 0;
109 }
110
111 .error-message {
112 background: linear-gradient(135deg, #FFB6C1, #FF69B4);
113 color: #8B0000;
114 padding: 1rem;
115 border-radius: 10px;
116 border-left: 5px solid #DC143C;
117 margin: 1rem 0;
118 }
119
120 .info-message {
121 background: linear-gradient(135deg, #87CEFA, #ADD8E6);
122 color: #000080;
123 padding: 1rem;
124 border-radius: 10px;
125 border-left: 5px solid #1E90FF;
126 margin: 1rem 0;
127 }
128
129 .sidebar .sidebar-content {
130 background: linear-gradient(180deg, #F0FFF0, #E0F7E0);
131 }
132
133 /* Custom scrollbar */
134 ::-webkit-scrollbar {
135 width: 8px;
136 }
137
138 ::-webkit-scrollbar-track {
139 background: #f1f1f1;
140 }
141
142 ::-webkit-scrollbar-thumb {
143 background: #2E8B57;
144 border-radius: 4px;
145 }
146
147 ::-webkit-scrollbar-thumb:hover {
148 background: #228B22;
149 }
150</style>
151""", unsafe_allow_html=True)
152
153def load_lottie_url(url: str):
154 """Load Lottie animation from URL"""
155 try:
156 r = requests.get(url)
157 if r.status_code != 200:
158 return None
159 return r.json()
160 except:
161 return None
162
163def load_local_lottie(filepath: str):
164 """Load Lottie animation from local file"""
165 try:
166 with open(filepath, 'r') as f:
167 return json.load(f)
168 except:
169 return None
170
171def create_profiling_report(df, title):
172 """Create yData profiling report with updated parameters"""
173 try:
174 from ydata_profiling import ProfileReport
175
176 profile = ProfileReport(
177 df,
178 title=title,
179 explorative=True,
180 minimal=False,
181 progress_bar=False
182 )
183 return profile
184 except ImportError:
185 st.error("ydata-profiling not installed. Run: pip install ydata-profiling")
186 return None
187 except Exception as e:
188 st.error(f"Error creating profile report: {str(e)}")
189 return None
190
191def st_profile_report(profile):
192 """Display profile report in Streamlit"""
193 try:
194 profile_html = profile.to_html()
195 components.html(profile_html, height=800, scrolling=True)
196 except Exception as e:
197 st.error(f"Error displaying profile report: {str(e)}")
198 st.warning("Please download the full report using the download button below.")
199
200def display_basic_analysis(df, title):
201 """Display basic data analysis when full profiling fails"""
202 st.markdown(f"## ๐ Basic Analysis: {title}")
203
204 # Basic metrics
205 col1, col2, col3, col4 = st.columns(4)
206 with col1:
207 st.metric("Total Rows", df.shape[0])
208 with col2:
209 st.metric("Total Columns", df.shape[1])
210 with col3:
211 st.metric("Missing Values", df.isnull().sum().sum())
212 with col4:
213 st.metric("Duplicate Rows", df.duplicated().sum())
214
215 # Data preview
216 st.subheader("๐ Data Preview")
217 st.dataframe(df.head(10), use_container_width=True)
218
219 # Data types
220 col1, col2 = st.columns(2)
221 with col1:
222 st.subheader("๐ Data Types")
223 st.write(df.dtypes)
224
225 with col2:
226 st.subheader("๐ Basic Statistics")
227 st.write(df.describe())
228
229 # Missing values
230 st.subheader("โ Missing Values Analysis")
231 missing_data = df.isnull().sum()
232 if missing_data.sum() > 0:
233 missing_df = pd.DataFrame({
234 'Column': missing_data[missing_data > 0].index,
235 'Missing Count': missing_data[missing_data > 0].values,
236 'Missing Percentage': (missing_data[missing_data > 0].values / len(df) * 100).round(2)
237 })
238 st.dataframe(missing_df, use_container_width=True)
239 else:
240 st.success("๐ No missing values found!")
241
242 # Unique values
243 st.subheader("๐ฏ Unique Values")
244 unique_df = pd.DataFrame({
245 'Column': df.columns,
246 'Unique Values': [df[col].nunique() for col in df.columns],
247 'Data Type': df.dtypes.values
248 })
249 st.dataframe(unique_df, use_container_width=True)
250
251def main():
252 # Header section with animation
253 col1, col2, col3 = st.columns([1, 2, 1])
254
255 with col2:
256 st.markdown('<h1 class="main-header">๐ Data Profiler Pro</h1>', unsafe_allow_html=True)
257 st.markdown("### *Advanced Data Analysis with yData Profiling*")
258
259 # Loading animation
260 calibration_animation = load_lottie_url("https://assets1.lottiefiles.com/packages/lf20_Stt1R6.json")
261
262 if calibration_animation:
263 try:
264 from streamlit_lottie import st_lottie
265 with col2:
266 st_lottie(calibration_animation, height=150, key="calibration")
267 except ImportError:
268 st.info("โจ Install streamlit-lottie for animations: `pip install streamlit-lottie`")
269
270 # Sidebar
271 with st.sidebar:
272 st.markdown("## ๐ฏ Navigation")
273 st.markdown("---")
274
275 dataset_choice = st.radio(
276 "Select Dataset:",
277 ["Titanic", "Iris", "Upload Your Own"],
278 index=0
279 )
280
281 st.markdown("---")
282 st.markdown("### โ๏ธ Report Settings")
283
284 report_mode = st.radio(
285 "Report Mode:",
286 ["Complete", "Minimal"],
287 help="Complete: Full detailed report. Minimal: Faster basic report."
288 )
289
290 st.markdown("---")
291 st.markdown("### ๐ Features")
292 st.markdown("""
293 - **Automated Data Quality Assessment**
294 - **Interactive Visualizations**
295 - **Statistical Summaries**
296 - **Correlation Analysis**
297 - **Missing Values Analysis**
298 - **Data Type Detection**
299 - **Duplicate Detection**
300 """)
301
302 st.markdown("---")
303 st.markdown("### ๐ ๏ธ Built With")
304 st.markdown("""
305 - Streamlit
306 - yData Profiling
307 - Pandas
308 - Lottie Animations
309 """)
310
311 # Main content area
312 if dataset_choice == "Upload Your Own":
313 st.markdown('<div class="sub-header">๐ค Upload Your Dataset</div>', unsafe_allow_html=True)
314
315 uploaded_file = st.file_uploader(
316 "Choose a CSV file",
317 type=['csv', 'xlsx', 'xls'],
318 help="Upload your dataset in CSV or Excel format"
319 )
320
321 if uploaded_file is not None:
322 try:
323 with st.spinner('๐ฅ Loading your data...'):
324 if uploaded_file.name.endswith('.csv'):
325 df = pd.read_csv(uploaded_file)
326 else:
327 df = pd.read_excel(uploaded_file)
328
329 st.markdown(f'<div class="success-message">โ
Successfully loaded dataset with {df.shape[0]} rows and {df.shape[1]} columns</div>', unsafe_allow_html=True)
330
331 # Data preview
332 with st.expander("๐ Data Preview", expanded=True):
333 st.dataframe(df.head(), use_container_width=True)
334
335 # Generate report
336 if st.button("๐ Generate Profiling Report", key="generate_custom", type="primary"):
337 with st.spinner('๐ฌ Creating comprehensive profile report...'):
338 progress_bar = st.progress(0)
339 for i in range(100):
340 time.sleep(0.01)
341 progress_bar.progress(i + 1)
342
343 profile = create_profiling_report(df, "Custom Dataset Profiling Report")
344 if profile:
345 st.markdown('<div class="success-message">๐ Profile Report Generated Successfully!</div>', unsafe_allow_html=True)
346 st.balloons()
347 st_profile_report(profile)
348
349 # Download option
350 st.markdown("---")
351 st.markdown("### ๐พ Download Report")
352 profile_html = profile.to_html()
353
354 col1, col2 = st.columns(2)
355 with col1:
356 st.download_button(
357 label="๐ฅ Download HTML Report",
358 data=profile_html,
359 file_name="custom_dataset_profile_report.html",
360 mime="text/html",
361 use_container_width=True
362 )
363 with col2:
364 st.download_button(
365 label="๐ Download Dataset",
366 data=uploaded_file.getvalue(),
367 file_name=uploaded_file.name,
368 use_container_width=True
369 )
370 else:
371 st.markdown('<div class="error-message">โ ๏ธ Falling back to Basic Analysis</div>', unsafe_allow_html=True)
372 display_basic_analysis(df, "Custom Dataset")
373
374 except Exception as e:
375 st.error(f"Error loading file: {str(e)}")
376
377 else:
378 # Dataset selection and description
379 if dataset_choice == "Titanic":
380 st.markdown('<div class="sub-header">๐ข Titanic Dataset Analysis</div>', unsafe_allow_html=True)
381
382 col1, col2 = st.columns([2, 1])
383
384 with col1:
385 st.markdown("""
386 <div class="dataset-card">
387 <h3>About the Titanic Dataset</h3>
388 <p>The Titanic dataset contains information about passengers aboard the RMS Titanic,
389 including survival status, passenger class, age, gender, and more. This dataset is
390 commonly used for predictive modeling and data analysis exercises.</p>
391 <p><strong>Key Features:</strong> Survival, Pclass, Sex, Age, Fare, Embarked</p>
392 <p><strong>Use Cases:</strong> Classification, Survival Analysis, Feature Engineering</p>
393 </div>
394 """, unsafe_allow_html=True)
395
396 with col2:
397 titanic_animation = load_lottie_url("https://assets1.lottiefiles.com/packages/lf20_kUZwfP.json")
398 if titanic_animation:
399 try:
400 from streamlit_lottie import st_lottie
401 st_lottie(titanic_animation, height=150, key="titanic")
402 except:
403 pass
404
405 with st.spinner('๐ณ๏ธ Loading Titanic dataset...'):
406 df = load_titanic_data()
407
408 else: # Iris dataset
409 st.markdown('<div class="sub-header">๐บ Iris Dataset Analysis</div>', unsafe_allow_html=True)
410
411 col1, col2 = st.columns([2, 1])
412
413 with col1:
414 st.markdown("""
415 <div class="dataset-card">
416 <h3>About the Iris Dataset</h3>
417 <p>The Iris flower dataset is a multivariate dataset introduced by Ronald Fisher.
418 It contains measurements of iris flowers from three different species, making it
419 perfect for classification and clustering analysis.</p>
420 <p><strong>Key Features:</strong> Sepal length, Sepal width, Petal length, Petal width, Species</p>
421 <p><strong>Use Cases:</strong> Classification, Clustering, Dimensionality Reduction</p>
422 </div>
423 """, unsafe_allow_html=True)
424
425 with col2:
426 flower_animation = load_lottie_url("https://assets1.lottiefiles.com/packages/lf20_sk5h1kfn.json")
427 if flower_animation:
428 try:
429 from streamlit_lottie import st_lottie
430 st_lottie(flower_animation, height=150, key="flower")
431 except:
432 pass
433
434 with st.spinner('๐ธ Loading Iris dataset...'):
435 df = load_iris_data()
436
437 # Show dataset info
438 st.markdown(f'<div class="success-message">โ
Successfully loaded {dataset_choice} dataset with {df.shape[0]} rows and {df.shape[1]} columns</div>', unsafe_allow_html=True)
439
440 # Data preview section
441 with st.expander("๐ Quick Data Preview", expanded=True):
442 col1, col2, col3, col4 = st.columns(4)
443
444 with col1:
445 st.metric("Total Rows", df.shape[0])
446 with col2:
447 st.metric("Total Columns", df.shape[1])
448 with col3:
449 st.metric("Missing Values", df.isnull().sum().sum())
450 with col4:
451 st.metric("Data Size", f"{df.memory_usage(deep=True).sum() / 1024:.1f} KB")
452
453 st.dataframe(df.head(10), use_container_width=True)
454
455 col1, col2 = st.columns(2)
456 with col1:
457 st.write("**๐ Data Types:**")
458 st.write(df.dtypes)
459 with col2:
460 st.write("**๐ Basic Statistics:**")
461 st.write(df.describe())
462
463 # Generate profiling report
464 st.markdown("---")
465 st.markdown('<div class="sub-header">๐ Generate Comprehensive Profile Report</div>', unsafe_allow_html=True)
466
467 if st.button(f"๐ Generate {dataset_choice} Profiling Report", key=f"generate_{dataset_choice}", type="primary"):
468 with st.spinner('๐ฌ Creating comprehensive profile report...'):
469 progress_bar = st.progress(0)
470 status_text = st.empty()
471
472 for i in range(100):
473 progress_bar.progress(i + 1)
474 status_text.text(f"Processing... {i+1}%")
475 time.sleep(0.01)
476
477 status_text.text("Finalizing report...")
478
479 profile = create_profiling_report(df, f"{dataset_choice} Dataset Profiling Report")
480
481 if profile:
482 st.markdown('<div class="success-message">๐ Profile Report Generated Successfully!</div>', unsafe_allow_html=True)
483 st.balloons()
484
485 st_profile_report(profile)
486
487 st.markdown("---")
488 st.markdown("### ๐พ Download Options")
489
490 profile_html = profile.to_html()
491
492 col1, col2 = st.columns(2)
493 with col1:
494 st.download_button(
495 label="๐ฅ Download HTML Report",
496 data=profile_html,
497 file_name=f"{dataset_choice.lower()}_profile_report.html",
498 mime="text/html",
499 use_container_width=True
500 )
501 with col2:
502 csv_data = df.to_csv(index=False)
503 st.download_button(
504 label="๐ Download Dataset CSV",
505 data=csv_data,
506 file_name=f"{dataset_choice.lower()}_dataset.csv",
507 mime="text/csv",
508 use_container_width=True
509 )
510 else:
511 st.markdown('<div class="error-message">โ ๏ธ Advanced profiling failed. Showing Basic Analysis instead.</div>', unsafe_allow_html=True)
512 display_basic_analysis(df, dataset_choice)
513
514 # Footer
515 st.markdown("---")
516 st.markdown("""
517 <div style='text-align: center; color: #2E8B57;'>
518 <h3>๐ Data Profiler Pro</h3>
519 <p>Made with โค๏ธ(Hammad_Zahid) using Streamlit & yData Profiling</p>
520 <p>Professional Data Analysis Tool | Open Source</p>
521 <p>โญ Star this project on <a href="https://github.com/Hammad_Ansari/data-profiler-app" target="_blank">GitHub</a></p>
522 </div>
523 """, unsafe_allow_html=True)
524
525if __name__ == "__main__":
526 main()