Spaces:
Sleeping
Sleeping
File size: 6,307 Bytes
7ea279c b85f76a 7ea279c b85f76a 7ea279c b85f76a 7ea279c b85f76a 7ea279c b85f76a 7ea279c b85f76a 7ea279c b85f76a 7ea279c b85f76a 7ea279c | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 | """Generate downloadable PDF and HTML analysis reports."""
from __future__ import annotations
import base64
import os
from datetime import datetime
from fpdf import FPDF
from data_processor import TEMP_DIR
PDF_PATH = os.path.join(TEMP_DIR, "analysis_report.pdf")
HTML_PATH = os.path.join(TEMP_DIR, "analysis_report.html")
def _sanitize(text: str, max_word: int = 40) -> str:
"""fpdf core fonts are latin-1 only; also break words too wide for one line.
fpdf2 raises 'Not enough horizontal space to render a single character'
when an unbroken token (e.g. a long one-hot column name) exceeds the
remaining cell width, so long words are split into chunks.
"""
replacements = {"²": "^2", "—": "-", "–": "-", "'": "'", "'": "'", """: '"', """: '"', "×": "x"}
for a, b in replacements.items():
text = text.replace(a, b)
text = text.encode("latin-1", "replace").decode("latin-1")
words = []
for word in text.split(" "):
while len(word) > max_word:
words.append(word[:max_word])
word = word[max_word:]
words.append(word)
return " ".join(words)
def generate_pdf_report(profile: dict, steps: list[str], result: dict) -> str:
pdf = FPDF()
pdf.set_auto_page_break(auto=True, margin=15)
pdf.add_page()
pdf.set_font("Helvetica", "B", 18)
pdf.cell(0, 12, "ML Data Analysis Report", new_x="LMARGIN", new_y="NEXT")
pdf.set_font("Helvetica", "", 10)
pdf.set_text_color(120, 120, 120)
pdf.cell(0, 6, f"Generated {datetime.now():%Y-%m-%d %H:%M}", new_x="LMARGIN", new_y="NEXT")
pdf.set_text_color(0, 0, 0)
pdf.ln(4)
def section(title):
pdf.set_font("Helvetica", "B", 13)
pdf.set_fill_color(230, 236, 245)
pdf.cell(0, 9, _sanitize(title), new_x="LMARGIN", new_y="NEXT", fill=True)
pdf.ln(2)
pdf.set_font("Helvetica", "", 10)
def kv(key, value):
pdf.set_font("Helvetica", "B", 10)
pdf.cell(55, 6, _sanitize(str(key)))
pdf.set_font("Helvetica", "", 10)
# new_x must reset to the margin: fpdf2's multi_cell default leaves the
# cursor at the cell's right edge, so the next line would start off-page
pdf.multi_cell(0, 6, _sanitize(str(value)), new_x="LMARGIN", new_y="NEXT")
section("1. Dataset Overview")
kv("Rows", f"{profile['n_rows']:,}")
kv("Columns", profile["n_cols"])
kv("Numeric columns", ", ".join(profile["numeric_columns"]) or "none")
kv("Categorical columns", ", ".join(profile["categorical_columns"]) or "none")
kv("Missing values", f"{profile['missing_total']:,}")
kv("Duplicate rows", f"{profile['duplicate_rows']:,}")
pdf.ln(4)
section("2. Preprocessing Steps")
for i, step in enumerate(steps, 1):
pdf.multi_cell(0, 6, _sanitize(f"{i}. {step}"), new_x="LMARGIN", new_y="NEXT")
pdf.ln(4)
section("3. Model & Results")
kv("Model", result["model_name"])
kv("Task", result["task"])
kv("Target column", result["target"])
kv("Training samples", f"{result['n_train']:,}")
kv("Test samples", f"{result['n_test']:,}")
pdf.ln(2)
for metric, value in result["metrics"].items():
kv(metric, value)
pdf.ln(4)
for title, path in [
("4. Result Plot", result.get("plot_path")),
("5. Feature Importance", result.get("importance_path")),
]:
if path and os.path.exists(path):
section(title)
pdf.image(path, w=150)
pdf.ln(4)
pdf.output(PDF_PATH)
return PDF_PATH
def _img_b64(path):
if not path or not os.path.exists(path):
return None
with open(path, "rb") as f:
return base64.b64encode(f.read()).decode()
def generate_html_report(profile: dict, steps: list[str], result: dict) -> str:
plot_b64 = _img_b64(result.get("plot_path"))
imp_b64 = _img_b64(result.get("importance_path"))
metrics_rows = "".join(
f"<tr><td>{k}</td><td><b>{v}</b></td></tr>" for k, v in result["metrics"].items()
)
steps_html = "".join(f"<li>{s}</li>" for s in steps)
missing = {k: v for k, v in profile["missing_counts"].items() if v > 0}
missing_html = (
", ".join(f"{k} ({v})" for k, v in missing.items()) if missing else "None"
)
html = f"""<!DOCTYPE html>
<html><head><meta charset="utf-8"><title>ML Data Analysis Report</title>
<style>
body {{ font-family: -apple-system, 'Segoe UI', Roboto, sans-serif; max-width: 860px;
margin: 40px auto; padding: 0 20px; color: #1a202c; line-height: 1.6; }}
h1 {{ border-bottom: 3px solid #4C72B0; padding-bottom: 8px; }}
h2 {{ color: #4C72B0; margin-top: 32px; }}
table {{ border-collapse: collapse; width: 100%; margin: 12px 0; }}
td, th {{ border: 1px solid #e2e8f0; padding: 8px 12px; text-align: left; }}
th {{ background: #edf2f7; }}
img {{ max-width: 100%; border: 1px solid #e2e8f0; border-radius: 8px; margin: 8px 0; }}
.meta {{ color: #718096; font-size: 0.9em; }}
</style></head><body>
<h1>ML Data Analysis Report</h1>
<p class="meta">Generated {datetime.now():%Y-%m-%d %H:%M}</p>
<h2>1. Dataset Overview</h2>
<table>
<tr><th>Rows</th><td>{profile['n_rows']:,}</td></tr>
<tr><th>Columns</th><td>{profile['n_cols']}</td></tr>
<tr><th>Numeric columns</th><td>{', '.join(profile['numeric_columns']) or 'none'}</td></tr>
<tr><th>Categorical columns</th><td>{', '.join(profile['categorical_columns']) or 'none'}</td></tr>
<tr><th>Missing values</th><td>{profile['missing_total']:,} ({missing_html})</td></tr>
<tr><th>Duplicate rows</th><td>{profile['duplicate_rows']:,}</td></tr>
</table>
<h2>2. Preprocessing Steps</h2>
<ol>{steps_html}</ol>
<h2>3. Model & Results</h2>
<table>
<tr><th>Model</th><td>{result['model_name']}</td></tr>
<tr><th>Task</th><td>{result['task']}</td></tr>
<tr><th>Target</th><td>{result['target']}</td></tr>
<tr><th>Train / test samples</th><td>{result['n_train']:,} / {result['n_test']:,}</td></tr>
{metrics_rows}
</table>
"""
if plot_b64:
html += f'<h2>4. Result Plot</h2><img src="data:image/png;base64,{plot_b64}">'
if imp_b64:
html += f'<h2>5. Feature Importance</h2><img src="data:image/png;base64,{imp_b64}">'
html += "</body></html>"
with open(HTML_PATH, "w", encoding="utf-8") as f:
f.write(html)
return HTML_PATH
|