guifav/data_analytics
0
1import streamlit as st2import pandas as pd3import plotly.express as px4import plotly.graph_objects as go5import numpy as np6import io7import base648import re9from datetime import datetime10 11# Page configuration12st.set_page_config(13 page_title="Data Analysis Dashboard",14 page_icon="๐",15 layout="wide",16 initial_sidebar_state="expanded"17)18 19# Custom CSS for better styling20st.markdown("""21<style>22 .main-header {23 font-size: 2.5rem;24 font-weight: bold;25 color: #1f77b4;26 text-align: center;27 margin-bottom: 2rem;28 }29 .metric-container {30 background-color: #f0f2f6;31 padding: 1rem;32 border-radius: 0.5rem;33 margin: 0.5rem 0;34 }35 .stSelectbox > div > div {36 background-color: white;37 }38 .upload-section {39 border: 2px dashed #cccccc;40 border-radius: 10px;41 padding: 2rem;42 text-align: center;43 margin: 1rem 0;44 }45</style>46""", unsafe_allow_html=True)47 48def convert_brazilian_number(value):49 """Convert Brazilian number format (xx.xxx.xxx,xx) to float"""50 if pd.isna(value) or value == '':51 return np.nan52 53 # Convert to string if not already54 str_value = str(value).strip()55 56 # Check if it's already a number57 try:58 return float(str_value)59 except ValueError:60 pass61 62 # Brazilian number pattern: can have dots as thousand separators and comma as decimal63 # Examples: "1.234.567,89", "1.234,56", "1234,56", "1234"64 brazilian_pattern = r'^-?\d{1,3}(?:\.\d{3})*(?:,\d+)?$'65 66 if re.match(brazilian_pattern, str_value):67 # Remove thousand separators (dots) and replace decimal comma with dot68 converted = str_value.replace('.', '').replace(',', '.')69 try:70 return float(converted)71 except ValueError:72 return np.nan73 74 return np.nan75 76def detect_and_convert_brazilian_numbers(df):77 """Detect and convert Brazilian number format columns to numeric"""78 converted_columns = []79 df_converted = df.copy()80 81 for col in df.columns:82 if df[col].dtype == 'object': # Only check string columns83 # Sample some non-null values to check if they look like Brazilian numbers84 sample_values = df[col].dropna().astype(str).head(10)85 86 if len(sample_values) > 0:87 # Check if most values match Brazilian number pattern88 brazilian_count = 089 total_count = 090 91 for value in sample_values:92 value = str(value).strip()93 if value and value != 'nan':94 total_count += 195 # Brazilian number pattern96 if re.match(r'^-?\d{1,3}(?:\.\d{3})*(?:,\d+)?$', value) or re.match(r'^-?\d+,\d+$', value):97 brazilian_count += 198 99 # If more than 70% of values look like Brazilian numbers, convert the column100 if total_count > 0 and (brazilian_count / total_count) > 0.7:101 converted_series = df[col].apply(convert_brazilian_number)102 103 # Only convert if we successfully converted most values104 non_null_original = df[col].notna().sum()105 non_null_converted = converted_series.notna().sum()106 107 if non_null_converted >= (non_null_original * 0.8): # At least 80% conversion success108 df_converted[col] = converted_series109 converted_columns.append(col)110 111 return df_converted, converted_columns112 113def load_sample_data():114 """Generate sample data for demonstration"""115 np.random.seed(42)116 n_samples = 1000117 118 data = {119 'Date': pd.date_range('2023-01-01', periods=n_samples, freq='D'),120 'Sales': np.random.normal(1000, 200, n_samples),121 'Profit': np.random.normal(150, 50, n_samples),122 'Category': np.random.choice(['Electronics', 'Clothing', 'Books', 'Home'], n_samples),123 'Region': np.random.choice(['North', 'South', 'East', 'West'], n_samples),124 'Customer_Age': np.random.randint(18, 80, n_samples),125 'Rating': np.random.uniform(1, 5, n_samples)126 }127 128 df = pd.DataFrame(data)129 df['Sales'] = np.where(df['Sales'] < 0, abs(df['Sales']), df['Sales'])130 df['Profit'] = np.where(df['Category'] == 'Electronics', df['Profit'] * 1.5, df['Profit'])131 132 # Add some Brazilian formatted numbers for demonstration133 df['Vendas_BR'] = df['Sales'].apply(lambda x: f"{x:,.2f}".replace(',', 'X').replace('.', ',').replace('X', '.'))134 df['Lucro_BR'] = df['Profit'].apply(lambda x: f"{x:,.2f}".replace(',', 'X').replace('.', ',').replace('X', '.'))135 136 return df137 138def get_numeric_columns(df):139 """Get numeric columns from dataframe"""140 return df.select_dtypes(include=[np.number]).columns.tolist()141 142def get_categorical_columns(df):143 """Get categorical columns from dataframe"""144 return df.select_dtypes(include=['object', 'category']).columns.tolist()145 146def create_download_link(df, filename="filtered_data.csv"):147 """Create download link for dataframe"""148 csv = df.to_csv(index=False)149 b64 = base64.b64encode(csv.encode()).decode()150 href = f'<a href="data:file/csv;base64,{b64}" download="{filename}">Download CSV File</a>'151 return href152 153def main():154 # Header155 st.markdown('<h1 class="main-header">๐ Data Analysis Dashboard</h1>', unsafe_allow_html=True)156 157 # Sidebar158 st.sidebar.title("๐ง Controls")159 st.sidebar.markdown("---")160 161 # File upload section162 st.sidebar.subheader("๐ Data Upload")163 uploaded_file = st.sidebar.file_uploader(164 "Choose a CSV file",165 type="csv",166 help="Upload a CSV file to analyze your data"167 )168 169 use_sample = st.sidebar.checkbox(170 "Use Sample Data",171 value=True if uploaded_file is None else False,172 help="Check this to use built-in sample data for demonstration"173 )174 175 # Brazilian number conversion option176 convert_brazilian = st.sidebar.checkbox(177 "๐ง๐ท Auto-convert Brazilian Numbers",178 value=True,179 help="Automatically detect and convert Brazilian number format (xx.xxx.xxx,xx) to numeric"180 )181 182 # Load data183 try:184 if uploaded_file is not None:185 df = pd.read_csv(uploaded_file)186 st.sidebar.success(f"โ
File uploaded successfully! ({len(df)} rows)")187 elif use_sample:188 df = load_sample_data()189 st.sidebar.info("๐ Using sample data")190 else:191 st.warning("Please upload a CSV file or use sample data to get started.")192 st.markdown("""193 ### ๐ Welcome to the Data Analysis Dashboard!194 195 This app helps you analyze and visualize your data with:196 - **Interactive charts** (bar, line, scatter, histogram)197 - **Dynamic filtering** and data exploration198 - **Statistical summaries** and insights199 - **Export capabilities** for data and visualizations200 - **๐ง๐ท Brazilian number format support** (xx.xxx.xxx,xx)201 202 **To get started:**203 1. Upload a CSV file using the sidebar, or204 2. Check "Use Sample Data" to explore with demo data205 """)206 return207 208 # Apply Brazilian number conversion if enabled209 if convert_brazilian:210 df_original = df.copy()211 df, converted_cols = detect_and_convert_brazilian_numbers(df)212 213 if converted_cols:214 st.sidebar.success(f"๐ง๐ท Converted {len(converted_cols)} columns from Brazilian format: {', '.join(converted_cols)}")215 216 except Exception as e:217 st.error(f"โ Error loading file: {str(e)}")218 st.info("Please make sure your file is a valid CSV format.")219 return220 221 # Data preview section222 st.subheader("๐ Data Preview")223 224 col1, col2, col3, col4 = st.columns(4)225 with col1:226 st.metric("Total Rows", len(df))227 with col2:228 st.metric("Total Columns", len(df.columns))229 with col3:230 st.metric("Numeric Columns", len(get_numeric_columns(df)))231 with col4:232 st.metric("Text Columns", len(get_categorical_columns(df)))233 234 # Show data preview235 with st.expander("๐ View Raw Data", expanded=False):236 st.dataframe(df.head(100), use_container_width=True)237 238 # Data summary239 with st.expander("๐ Statistical Summary", expanded=False):240 col1, col2 = st.columns(2)241 242 with col1:243 st.subheader("Numeric Columns")244 numeric_cols = get_numeric_columns(df)245 if numeric_cols:246 st.dataframe(df[numeric_cols].describe())247 else:248 st.info("No numeric columns found")249 250 with col2:251 st.subheader("Categorical Columns")252 cat_cols = get_categorical_columns(df)253 if cat_cols:254 for col in cat_cols[:5]: # Show first 5 categorical columns255 st.write(f"**{col}:** {df[col].nunique()} unique values")256 if df[col].nunique() <= 10:257 st.write(df[col].value_counts().head())258 else:259 st.info("No categorical columns found")260 261 # Show conversion info if Brazilian conversion was applied262 if convert_brazilian and 'converted_cols' in locals() and converted_cols:263 with st.expander("๐ง๐ท Brazilian Number Conversion Details", expanded=False):264 st.write("**Converted Columns:**")265 for col in converted_cols:266 original_sample = df_original[col].dropna().head(3).tolist()267 converted_sample = df[col].dropna().head(3).tolist()268 st.write(f"**{col}:**")269 st.write(f" - Original: {original_sample}")270 st.write(f" - Converted: {converted_sample}")271 272 # Filtering section273 st.sidebar.markdown("---")274 st.sidebar.subheader("๐ Data Filters")275 276 # Create a copy for filtering277 filtered_df = df.copy()278 279 # Numeric filters280 numeric_cols = get_numeric_columns(df)281 for col in numeric_cols:282 if df[col].dtype in ['int64', 'float64']:283 min_val = float(df[col].min())284 max_val = float(df[col].max())285 286 if min_val != max_val:287 selected_range = st.sidebar.slider(288 f"{col} Range",289 min_value=min_val,290 max_value=max_val,291 value=(min_val, max_val),292 help=f"Filter data by {col} values"293 )294 filtered_df = filtered_df[295 (filtered_df[col] >= selected_range[0]) & 296 (filtered_df[col] <= selected_range[1])297 ]298 299 # Categorical filters300 cat_cols = get_categorical_columns(df)301 for col in cat_cols:302 unique_values = df[col].unique().tolist()303 if len(unique_values) <= 50: # Only show filter for columns with reasonable number of unique values304 selected_values = st.sidebar.multiselect(305 f"Select {col}",306 options=unique_values,307 default=unique_values,308 help=f"Filter data by {col} categories"309 )310 if selected_values:311 filtered_df = filtered_df[filtered_df[col].isin(selected_values)]312 313 # Show filtered data info314 if len(filtered_df) != len(df):315 st.sidebar.info(f"Filtered: {len(filtered_df)} of {len(df)} rows")316 317 # Visualization section318 st.markdown("---")319 st.subheader("๐ Data Visualization")320 321 # Chart type selection322 chart_type = st.selectbox(323 "Select Chart Type",324 ["Bar Chart", "Line Chart", "Scatter Plot", "Histogram", "Box Plot"],325 help="Choose the type of visualization"326 )327 328 col1, col2, col3 = st.columns(3)329 330 with col1:331 if chart_type in ["Bar Chart", "Line Chart", "Scatter Plot", "Box Plot"]:332 x_column = st.selectbox(333 "X-axis Column",334 options=df.columns.tolist(),335 help="Select column for X-axis"336 )337 else:338 x_column = st.selectbox(339 "Column to Analyze",340 options=numeric_cols,341 help="Select numeric column for histogram"342 )343 344 with col2:345 if chart_type in ["Bar Chart", "Line Chart", "Scatter Plot", "Box Plot"]:346 y_column = st.selectbox(347 "Y-axis Column",348 options=numeric_cols,349 help="Select numeric column for Y-axis"350 )351 else:352 y_column = None353 354 with col3:355 if chart_type in ["Bar Chart", "Scatter Plot", "Box Plot"]:356 color_column = st.selectbox(357 "Color/Group By (Optional)",358 options=[None] + cat_cols,359 help="Select column to group/color data"360 )361 else:362 color_column = None363 364 # Create visualization365 if chart_type == "Bar Chart" and x_column and y_column:366 if x_column in cat_cols:367 # Aggregate data for categorical x-axis368 agg_df = filtered_df.groupby(x_column)[y_column].mean().reset_index()369 fig = px.bar(370 agg_df, 371 x=x_column, 372 y=y_column,373 title=f"Average {y_column} by {x_column}",374 color=color_column if color_column and color_column in agg_df.columns else None375 )376 else:377 fig = px.bar(378 filtered_df, 379 x=x_column, 380 y=y_column,381 title=f"{y_column} vs {x_column}",382 color=color_column383 )384 385 elif chart_type == "Line Chart" and x_column and y_column:386 fig = px.line(387 filtered_df, 388 x=x_column, 389 y=y_column,390 title=f"{y_column} vs {x_column}",391 color=color_column392 )393 394 elif chart_type == "Scatter Plot" and x_column and y_column:395 fig = px.scatter(396 filtered_df, 397 x=x_column, 398 y=y_column,399 title=f"{y_column} vs {x_column}",400 color=color_column,401 size=y_column if y_column in numeric_cols else None402 )403 404 elif chart_type == "Histogram" and x_column:405 fig = px.histogram(406 filtered_df, 407 x=x_column,408 title=f"Distribution of {x_column}",409 nbins=30410 )411 412 elif chart_type == "Box Plot" and x_column and y_column:413 fig = px.box(414 filtered_df, 415 x=x_column, 416 y=y_column,417 title=f"{y_column} Distribution by {x_column}",418 color=color_column419 )420 421 else:422 st.warning("Please select appropriate columns for the chosen chart type.")423 return424 425 # Update layout for better appearance426 fig.update_layout(427 height=500,428 showlegend=True,429 title_x=0.5,430 font=dict(size=12)431 )432 433 # Display chart434 st.plotly_chart(fig, use_container_width=True)435 436 # Download section437 st.markdown("---")438 st.subheader("๐พ Download Options")439 440 col1, col2 = st.columns(2)441 442 with col1:443 st.markdown("**Download Filtered Data**")444 if st.button("Generate CSV Download Link"):445 download_link = create_download_link(filtered_df, f"filtered_data_{datetime.now().strftime('%Y%m%d_%H%M%S')}.csv")446 st.markdown(download_link, unsafe_allow_html=True)447 448 with col2:449 st.markdown("**Download Chart**")450 if st.button("Download Chart as HTML"):451 html_string = fig.to_html(include_plotlyjs='cdn')452 st.download_button(453 label="Download HTML",454 data=html_string,455 file_name=f"chart_{datetime.now().strftime('%Y%m%d_%H%M%S')}.html",456 mime="text/html"457 )458 459 # Additional insights460 if len(filtered_df) > 0:461 st.markdown("---")462 st.subheader("๐ Quick Insights")463 464 col1, col2 = st.columns(2)465 466 with col1:467 st.markdown("**Data Overview**")468 st.write(f"โข Total records: {len(filtered_df):,}")469 st.write(f"โข Columns: {len(filtered_df.columns)}")470 471 if numeric_cols:472 st.write(f"โข Numeric columns: {len(numeric_cols)}")473 for col in numeric_cols[:3]:474 mean_val = filtered_df[col].mean()475 st.write(f" - {col}: avg = {mean_val:.2f}")476 477 with col2:478 st.markdown("**Missing Data**")479 missing_data = filtered_df.isnull().sum()480 if missing_data.sum() > 0:481 for col, missing in missing_data.items():482 if missing > 0:483 pct = (missing / len(filtered_df)) * 100484 st.write(f"โข {col}: {missing} ({pct:.1f}%)")485 else:486 st.write("โ
No missing data found")487 488if __name__ == "__main__":489 main()