process-aware-ai / backend /validate_ai_logic.py
borndeveloper's picture
Deploy Process Aware AI Dashboard without binaries
b4a2e7f
Raw
History Blame Contribute Delete
4.22 kB
import sys
import os
import pandas as pd
import statistics
# Add current directory to path so we can import app
sys.path.append(os.getcwd())
from app.services.data_service import data_service
def run_validation():
print("Loading data...")
data_service.load_data()
if data_service.master_df is None:
print("Error: Could not load data.")
return
# Get unique articles from master_df
# Column might be 'Article No' or 'Article' - check data_service.load_data
# Usually it's mapped. Let's start with 'Article' if mapped, or 'Material'
# In data_service.py, keys are likely from CSV.
# Let's verify column name. SAFE bet is to iterate what get_article_predictions expects.
# It takes article_id.
# Let's try to get unique values from the dataframe.
df = data_service.master_df
# Identify article column
article_col = 'Article' if 'Article' in df.columns else 'Material'
# If not found, try to list columns
if article_col not in df.columns:
print(f"Columns: {df.columns}")
return
all_articles = df[article_col].dropna().unique()
print(f"Found {len(all_articles)} unique articles. Running AI validation...")
results = []
# Run for all
count = 0
for article in all_articles:
count += 1
if count % 100 == 0:
print(f"Processed {count}/{len(all_articles)}...")
try:
pred = data_service.get_article_predictions(article)
ai = pred['ai_prediction']
rec = ai['recommendation']
norm = ai['norm_analysis']
stats = ai['historical_analysis']
yield_stats = ai['yield_stats']
results.append({
'Article': article,
'OrderCount': ai['historical_orders'],
'SuccessRate': stats['success_rate_pct'],
'NormPct': norm['base_norm_pct'],
'RecPct': rec['suggested_reservation_pct'],
'Adjustment': rec['ai_adjustment_pct'],
'YieldAvg': yield_stats['avg'],
'YieldStd': yield_stats['std_dev']
})
except Exception as e:
print(f"Error processing {article}: {e}")
pass
print(f"Collected results for {len(results)} articles.")
if not results:
print("No results collected! Exiting.")
return
res_df = pd.DataFrame(results)
res_df['Savings'] = res_df['NormPct'] - res_df['RecPct']
# Save to CSV for inspection
res_df.to_csv('validation_results.csv', index=False)
print("\n" + "="*40)
print(" VALIDATION SUMMARY")
print("="*40)
print(f"Total Articles Analyzed: {len(res_df)}")
print(f"Articles with Savings (> 0.5%): {len(res_df[res_df['Savings'] > 0.5])}")
print(f"Articles with More Buffer (< -0.5%): {len(res_df[res_df['Savings'] < -0.5])}")
print(f"Average Savings across plant: {res_df['Savings'].mean():.2f}%")
print("\n--- TOP 5 SAVINGS OPPORTUNITIES (Less Waste) ---")
print(res_df[res_df['OrderCount'] > 5].sort_values('Savings', ascending=False).head(5)[['Article', 'OrderCount', 'NormPct', 'RecPct', 'Savings', 'SuccessRate']])
print("\n--- TOP 5 RISK MITIGATION (More Safety) ---")
print(res_df[res_df['OrderCount'] > 5].sort_values('Savings', ascending=True).head(5)[['Article', 'OrderCount', 'NormPct', 'RecPct', 'Savings', 'SuccessRate']])
# Anomalies
neg_recs = res_df[res_df['RecPct'] < 0]
if not neg_recs.empty:
print(f"\n[CRITICAL] Found {len(neg_recs)} articles with NEGATIVE recommendation!")
print(neg_recs[['Article', 'RecPct']])
high_recs = res_df[res_df['RecPct'] > 15]
if not high_recs.empty:
print(f"\n[WARNING] Found {len(high_recs)} articles with > 15% recommendation!")
print(high_recs[['Article', 'RecPct', 'OrderCount']])
# Check specifically for the 'Partial Order' edge cases (high yields but high failure rate if not handled)
# We can't easily identify them here without looking at raw orders, but we can see if Rec % is reasonable.
if __name__ == "__main__":
run_validation()