Spaces:
Sleeping
Sleeping
| import altair as alt | |
| import numpy as np | |
| import pandas as pd | |
| import streamlit as st | |
| import requests | |
| import boto3 | |
| import os | |
| import openmeteo_requests | |
| import requests_cache | |
| # import psycopg2 | |
| import io | |
| import joblib | |
| import plotly.express as px | |
| from retry_requests import retry | |
| from sqlalchemy import create_engine | |
| from sklearn.model_selection import train_test_split, GridSearchCV | |
| from sklearn.preprocessing import StandardScaler, OneHotEncoder | |
| from sklearn.compose import ColumnTransformer | |
| from sklearn.pipeline import Pipeline | |
| from sklearn.linear_model import LinearRegression | |
| from sklearn.metrics import r2_score, mean_squared_error | |
| os.environ['AWS_ACCESS_KEY_ID'] = os.getenv('AWS_ACCESS_KEY_ID') | |
| os.environ['AWS_SECRET_ACCESS_KEY'] = os.getenv('AWS_SECRET_ACCESS_KEY') | |
| os.environ['S3_BUCKET'] = os.getenv('S3_BUCKET') | |
| os.environ['S3_BUCKET2'] = os.getenv('S3_BUCKET2') | |
| s3 = boto3.client('s3') | |
| db_user = os.getenv("DB_USER") | |
| db_password = os.getenv("DB_PASS") | |
| db_host = os.getenv("DB_HOST") | |
| db_name = os.getenv("DB_NAME") | |
| # engine = create_engine(f"postgresql+psycopg2://{db_user}:{db_password}@{db_host}/{db_name}") | |
| #------------------------------------- Code XGBOOST---------------------------------------------------- | |
| # Chargement du modèle sur le S3 | |
| bucket = os.getenv('S3_BUCKET') | |
| bucket2 = os.getenv('S3_BUCKET2') | |
| key = "mlflow/models/xgboost_model_a8bb2d98d53843fb9564d8304fbd8145.joblib" | |
| response = s3.get_object(Bucket=bucket, Key=key) | |
| buffer = io.BytesIO(response["Body"].read()) | |
| model = joblib.load(buffer) | |
| # Chargement des données météo des 7 prochains jour sur l'API météo | |
| cache_session = requests_cache.CachedSession('.cache', expire_after = 3600) | |
| retry_session = retry(cache_session, retries = 5, backoff_factor = 0.2) | |
| openmeteo = openmeteo_requests.Client(session = retry_session) | |
| # Make sure all required weather variables are listed here | |
| # The order of variables in hourly or daily is important to assign them correctly below | |
| url = "https://api.open-meteo.com/v1/forecast" | |
| params = { | |
| "latitude": 43.549999, | |
| "longitude": 1.1911389, | |
| "hourly": ["temperature_2m", "precipitation", "weather_code", "cloud_cover", "cloud_cover_low", "cloud_cover_mid", "cloud_cover_high", "global_tilted_irradiance_instant"], | |
| "timezone": "auto", | |
| "tilt": 18, | |
| "azimuth": 11, | |
| } | |
| responses = openmeteo.weather_api(url, params=params) | |
| # Process first location. Add a for-loop for multiple locations or weather models | |
| response = responses[0] | |
| print(f"Coordinates: {response.Latitude()}°N {response.Longitude()}°E") | |
| print(f"Elevation: {response.Elevation()} m asl") | |
| print(f"Timezone difference to GMT+0: {response.UtcOffsetSeconds()}s") | |
| # Process hourly data. The order of variables needs to be the same as requested. | |
| hourly = response.Hourly() | |
| hourly_temperature_2m = hourly.Variables(0).ValuesAsNumpy() | |
| hourly_precipitation = hourly.Variables(1).ValuesAsNumpy() | |
| hourly_weather_code = hourly.Variables(2).ValuesAsNumpy() | |
| hourly_cloud_cover = hourly.Variables(3).ValuesAsNumpy() | |
| hourly_cloud_cover_low = hourly.Variables(4).ValuesAsNumpy() | |
| hourly_cloud_cover_mid = hourly.Variables(5).ValuesAsNumpy() | |
| hourly_cloud_cover_high = hourly.Variables(6).ValuesAsNumpy() | |
| hourly_global_tilted_irradiance_instant = hourly.Variables(7).ValuesAsNumpy() | |
| hourly_data = {"date": pd.date_range( | |
| start = pd.to_datetime(hourly.Time(), unit = "s", utc = True), | |
| end = pd.to_datetime(hourly.TimeEnd(), unit = "s", utc = True), | |
| freq = pd.Timedelta(seconds = hourly.Interval()), | |
| inclusive = "left" | |
| )} | |
| hourly_data["temperature_2m"] = hourly_temperature_2m | |
| hourly_data["precipitation"] = hourly_precipitation | |
| hourly_data["weather_code"] = hourly_weather_code | |
| hourly_data["cloud_cover"] = hourly_cloud_cover | |
| hourly_data["cloud_cover_low"] = hourly_cloud_cover_low | |
| hourly_data["cloud_cover_mid"] = hourly_cloud_cover_mid | |
| hourly_data["cloud_cover_high"] = hourly_cloud_cover_high | |
| hourly_data["global_tilted_irradiance_instant"] = hourly_global_tilted_irradiance_instant | |
| df = pd.DataFrame(data = hourly_data) | |
| df['year'] = df['date'].dt.year | |
| df['month'] = df["date"].dt.month | |
| df['day'] = df["date"].dt.day | |
| df['hour'] = df['date'].dt.hour | |
| df.drop('date', axis=1, inplace=True) | |
| df1 = df.copy() | |
| y_prediction = model.predict(df1) | |
| y_prediction = np.clip(y_prediction, 0, None) | |
| df1['Puissance (KW)'] = (y_prediction/1000).round(2) | |
| modif_month = {1: 'Janvier', 2: 'Février', 3: 'Mars', 4: 'Avril', 5: 'Mai', 6: 'Juin', 7: 'Juillet', 8: 'Août', 9: 'Septembre', 10: 'Octobre', 11: 'Novembre', 12: 'Décembre'} | |
| st.title('Application de prédiction de production photovoltaïque') | |
| st.markdown("""Cette application à pour but de prédire la production photovoltaïque pour un équipement et des coordonnées GPS prèdéfinit. | |
| 2 méthodes ont été testé : | |
| - Utilisation d'un algorithme de machine learnig sur la base d'une regression linéaire, et en utilisant les données météo. | |
| - Simple calcule d'estimation à l'aide du GTI ( irridiation global instantané ) | |
| Pour récupérer les données météo nous utilisons l'API gratuite : https://open-meteo.com/. Cette API est alimenté par des données GPS de notre installation au niveau ville pas domicile. | |
| L'inclinaison des panneaux ainsi que leur orientation a été entré dans les critères pour le calcule du GTI. Nous Récupérons les prédictions météo des 7 prochains jours. | |
| """) | |
| st.subheader("Algorithme de prédiction entrainé") | |
| st.markdown("""Un modèle XGBoost a été entrainé et optimisé par un gridsearch pour générer des prédictions de production élèctrique. | |
| Le modèle c'est entrainé sur des données de production heure par heure depuis 2018 jusqu'en 2025, et les condtions climatique sur ces jours là. | |
| """) | |
| # Code XGBoost | |
| for i in df1.month.unique(): | |
| mask = df1['month'] == i | |
| df1_mask = df1.loc[mask] | |
| df1_mask = df1_mask[df1_mask['Puissance (KW)'] != 0] | |
| grouped_df1 = df1_mask.groupby('day')['Puissance (KW)'].sum().reset_index() | |
| fig = px.bar(grouped_df1, x='day', y='Puissance (KW)', text_auto=True, title=f"Prévision de production par jour avec XGBoost pour le mois de {modif_month[i]}") | |
| fig.update_xaxes(dtick=1) | |
| st.plotly_chart(fig, width='stretch') | |
| df1_mask2 = df1[df1['Puissance (KW)'] != 0] | |
| df1_grouped = df1_mask2.groupby('day')['Puissance (KW)'].sum().reset_index() | |
| total_semaine = df1_grouped["Puissance (KW)"].sum() | |
| st.markdown("""Production cumul jour""") | |
| st.dataframe(df1_grouped) | |
| st.markdown("""Les 3 périodes optimum des 3 prochains jour pour utiliser des appareils élèctriques gourmand :""") | |
| N = 3 | |
| Periode_max = (df1.groupby(['day', 'month'], group_keys=False).apply(lambda x:x.nlargest(N,"Puissance (KW)")).reset_index(drop=True)) | |
| Periode_max = Periode_max.sort_values(["month", "day", "Puissance (KW)"], ascending=[True, True, False]) | |
| Periode_max = Periode_max[Periode_max['Puissance (KW)'] != 0] | |
| Periode_max = Periode_max[Periode_max['day'].isin(Periode_max['day'].unique()[:3])] | |
| Periode_max.drop(columns={'temperature_2m', | |
| 'precipitation', | |
| 'weather_code', | |
| 'cloud_cover', | |
| 'cloud_cover_low', | |
| 'cloud_cover_mid', | |
| 'cloud_cover_high', | |
| 'global_tilted_irradiance_instant', | |
| 'year', | |
| 'month'}, axis=1, inplace=True) | |
| col1, col2, col3 = st.columns(3) | |
| B1 = Periode_max.iloc[0:3] | |
| fig1 = px.bar(B1, x='hour', y='Puissance (KW)', text_auto=True, title=f"Meilleur tranche du {B1['day'].iloc[0]}") | |
| B2 = Periode_max.iloc[3:6] | |
| fig2 = px.bar(B2, x='hour', y='Puissance (KW)', text_auto=True, title=f"Meilleur tranche du {B2['day'].iloc[0]}") | |
| B3 = Periode_max.iloc[6:9] | |
| fig3 = px.bar(B3, x='hour', y='Puissance (KW)', text_auto=True, title=f"Meilleur tranche du {B3['day'].iloc[0]}") | |
| with col1: | |
| st.plotly_chart(fig1, width='stretch') | |
| with col2: | |
| st.plotly_chart(fig2, width='stretch') | |
| with col3: | |
| st.plotly_chart(fig3, width='stretch') | |
| st.text(f"Production total sur 7 jours avec XGBoost: {total_semaine:,.2f} KW") | |
| st.subheader("Prédiction Grace au GTI (W/m²)") | |
| st.markdown("""Une méthode simple de prédiction uniquement lié au GTI estimé par le site météo. | |
| la formule de calcule utilisé pour estimer la production est : GTI*surface_total_panneaux*rendement. | |
| Le rendement est calculé en faisant puissance crête d'un panneau divisé par 1000 * par sa surface. | |
| """) | |
| # Code GTI | |
| df2 = df.copy() | |
| df2= df2.rename(columns={'global_tilted_irradiance_instant':'gti'}) | |
| panneaux = 9 | |
| puissance_panneaux = 0.327 | |
| surface_panneaux = 1.63 | |
| df2['gti'] = df2['gti'].astype(float) | |
| # rendement = puissance_panneaux/(1000*surface_panneaux) | |
| rendement = 0.78 | |
| df2['pv_kwh'] = ((df2['gti']/1000)*rendement*(puissance_panneaux*panneaux)).round(2) | |
| df2['pv_kwh'] = df2['pv_kwh'].clip(lower=0).fillna(0) | |
| for i in df2.month.unique(): | |
| mask = df2['month'] == i | |
| df2_mask = df2.loc[mask] | |
| df2_mask = df2_mask[df2_mask['pv_kwh'] != 0] | |
| grouped_df2 = df2_mask.groupby('day')['pv_kwh'].sum().reset_index() | |
| fig2 = px.bar(grouped_df2, x='day', y='pv_kwh', text_auto=True, title=f"Prévision de production par jour avec le GTI pour le mois de {modif_month[i]}") | |
| fig2.update_xaxes(dtick=1) | |
| st.plotly_chart(fig2, width='stretch') | |
| df2_mask2 = df2[df2['pv_kwh'] != 0] | |
| df2_grouped = df2_mask2.groupby('day')['pv_kwh'].sum().reset_index() | |
| total_semaine2 = df2_grouped['pv_kwh'].sum() | |
| st.markdown("""Production cumul jour""") | |
| st.dataframe(df2_grouped) | |
| st.markdown("""Les 3 périodes optimum des 3 prochains jour pour utiliser des appareils élèctrique gourmand :""") | |
| N = 3 | |
| Periode_max2 = (df2.groupby(['day', 'month'], group_keys=False).apply(lambda x:x.nlargest(N,'pv_kwh')).reset_index(drop=True)) | |
| Periode_max2 = Periode_max2.sort_values(["month", "day", "pv_kwh"], ascending=[True, True, False]) | |
| Periode_max2 = Periode_max2[Periode_max2['pv_kwh'] != 0] | |
| Periode_max2 = Periode_max2[Periode_max2['day'].isin(Periode_max2['day'].unique()[:3])] | |
| Periode_max2.drop(columns={'temperature_2m', | |
| 'precipitation', | |
| 'weather_code', | |
| 'cloud_cover', | |
| 'cloud_cover_low', | |
| 'cloud_cover_mid', | |
| 'cloud_cover_high', | |
| 'gti', | |
| 'year', | |
| 'month'}, axis=1, inplace=True) | |
| A1 = Periode_max2.iloc[0:3] | |
| fig3 = px.bar(A1, x='hour', y='pv_kwh', text_auto=True, title=f"Meilleur tranche du {A1['day'].iloc[0]}") | |
| A2 = Periode_max2.iloc[3:6] | |
| fig4 = px.bar(A2, x='hour', y='pv_kwh', text_auto=True, title=f"Meilleur tranche du {A2['day'].iloc[0]}") | |
| A3 = Periode_max2.iloc[6:9] | |
| fig5 = px.bar(A3, x='hour', y='pv_kwh', text_auto=True, title=f"Meilleur tranche du {A3['day'].iloc[0]}") | |
| col1, col2, col3 = st.columns(3) | |
| with col1: | |
| st.plotly_chart(fig3, width='stretch') | |
| with col2: | |
| st.plotly_chart(fig4, width='stretch') | |
| with col3: | |
| st.plotly_chart(fig5, width='stretch') | |
| st.text(f"Production total sur 7 jours avec GTI : {total_semaine2.round(2)} KW") | |
| date_now = pd.to_datetime('now') | |
| df1_mask2.to_csv(f'predictions_XGBOOST_du_{date_now}.csv', index=False) | |
| df2_mask2.to_csv(f'predictions_GTI_du_{date_now}.csv', index=False) | |
| s3.upload_file( | |
| Filename=f'predictions_XGBOOST_du_{date_now}.csv', | |
| Bucket=bucket2, | |
| Key=f'prediction/predictions_XGBOOST_du_{date_now}.csv' | |
| ) | |
| s3.upload_file( | |
| Filename=f'predictions_GTI_du_{date_now}.csv', | |
| Bucket=bucket2, | |
| Key=f'prediction/predictions_GTI_du_{date_now}.csv' | |
| ) |