Spaces:
Running
Running
| import os | |
| import threading | |
| from pathlib import Path | |
| print("[BOOT 1/4] app.py process started", flush=True) | |
| from tracernet.config import PROJECT_ROOT, apply_env_defaults | |
| from tracernet.crn import pathways as crn_pathways | |
| from tracernet.crn.repository import CRNRepository | |
| from tracernet.services.bibliography import ( | |
| first_doi, | |
| normalize_doi, | |
| reference_contains_source, | |
| ) | |
| from tracernet.services.pathway_evidence import ( | |
| load_condition_vocabulary, | |
| parse_formula, | |
| parse_path, | |
| path_components, | |
| ) | |
| from tracernet.services.report_text import ( | |
| compose_graph_only_note, | |
| is_bad_generation_text, | |
| ) | |
| from tracernet.utils.assets import ensure_runtime_assets, find_asset | |
| from tracernet.utils.images import coerce_image_result | |
| apply_env_defaults() | |
| ensure_runtime_assets( | |
| PROJECT_ROOT, | |
| { | |
| "spectrum.db.gz": "spectrum.db", | |
| "cytoscape_network.png.gz": "cytoscape_network.png", | |
| }, | |
| search_roots=(PROJECT_ROOT, PROJECT_ROOT.parent), | |
| ) | |
| print("[BOOT 2/4] Runtime assets checked", flush=True) | |
| import gradio as gr | |
| import numpy as np | |
| import pandas as pd | |
| import logging | |
| import difflib | |
| import html | |
| import tempfile | |
| import re | |
| def _patch_template_response_compat() -> None: | |
| """ | |
| Bridge signature drift between Gradio's route call | |
| (TemplateResponse(name, context)) and Starlette variants that | |
| prefer TemplateResponse(request, name, context). | |
| """ | |
| try: | |
| from starlette.templating import Jinja2Templates | |
| except Exception: | |
| return | |
| original = getattr(Jinja2Templates, "TemplateResponse", None) | |
| if not callable(original) or getattr(original, "_compat_patched_by_tracernet", False): | |
| return | |
| def _compat_template_response(self, *args, **kwargs): | |
| try: | |
| return original(self, *args, **kwargs) | |
| except TypeError: | |
| if len(args) >= 2 and isinstance(args[0], str) and isinstance(args[1], dict): | |
| name = args[0] | |
| context = args[1] or {} | |
| request = context.get("request") or kwargs.get("request") | |
| if request is not None: | |
| adapted_kwargs = {k: v for k, v in kwargs.items() if k != "request"} | |
| return original(self, request, name, context, *args[2:], **adapted_kwargs) | |
| if len(args) >= 2 and not isinstance(args[0], str) and isinstance(args[1], str): | |
| request = args[0] | |
| name = args[1] | |
| if len(args) > 2 and isinstance(args[2], dict): | |
| context = dict(args[2]) | |
| tail = args[3:] | |
| else: | |
| context = dict(kwargs.get("context") or {}) | |
| tail = args[2:] | |
| context.setdefault("request", request) | |
| adapted_kwargs = {k: v for k, v in kwargs.items() if k not in ("context", "request")} | |
| return original(self, name, context, *tail, **adapted_kwargs) | |
| raise | |
| _compat_template_response._compat_patched_by_tracernet = True | |
| Jinja2Templates.TemplateResponse = _compat_template_response | |
| _patch_template_response_compat() | |
| LATEX_DELIMITERS = [ | |
| {"left": "$$", "right": "$$", "display": True}, | |
| {"left": "$", "right": "$", "display": False}, | |
| ] | |
| try: | |
| import matplotlib.pyplot as plt | |
| except Exception: | |
| plt = None | |
| try: | |
| from matplotlib.gridspec import GridSpec | |
| from matplotlib.colors import LinearSegmentedColormap | |
| except Exception: | |
| GridSpec = None | |
| LinearSegmentedColormap = None | |
| try: | |
| import gradio_client.utils as _gc_utils | |
| if hasattr(_gc_utils, "_json_schema_to_python_type"): | |
| _orig_json_schema = _gc_utils._json_schema_to_python_type | |
| def _safe_json_schema_to_python_type(schema, defs=None): | |
| try: | |
| if isinstance(schema, bool): | |
| return "Any" | |
| return _orig_json_schema(schema, defs) | |
| except Exception: | |
| return "Any" | |
| _gc_utils._json_schema_to_python_type = _safe_json_schema_to_python_type | |
| except Exception: | |
| pass | |
| try: | |
| from dotenv import load_dotenv | |
| load_dotenv() | |
| except Exception: | |
| pass | |
| from chemical_formatter import ( | |
| display_condition_label as _display_condition_label, | |
| formula_for_markdown, | |
| normalize_legacy_text, | |
| render_chem_text, | |
| to_unicode_subscript, | |
| ) | |
| from language_policy import ENGLISH_ONLY_MESSAGE, is_supported_english_query | |
| from mural_copilot import MuralCopilot | |
| _crn_repository = CRNRepository.discover((PROJECT_ROOT,), max_depth=0) | |
| print("[BOOT 3/4] Core application modules imported", flush=True) | |
| rag_service = None | |
| _rag_init_error = None | |
| _rag_init_lock = threading.Lock() | |
| def _initialize_rag_service() -> None: | |
| """Eagerly initialize RAG after the HTTP server starts listening.""" | |
| global rag_service, _rag_init_error | |
| with _rag_init_lock: | |
| if rag_service is not None: | |
| return | |
| print("[RAG 1/5] Importing RAG dependencies...", flush=True) | |
| try: | |
| from rag_module import RAGService | |
| print("[RAG 2/5] RAG dependencies imported", flush=True) | |
| rag_service = RAGService(crn_repository=_crn_repository) | |
| _rag_init_error = None | |
| print( | |
| "[RAG 5/5] RAG service ready " | |
| f"(mode={rag_service.retrieval_mode}, " | |
| f"documents={rag_service.collection_count}).", | |
| flush=True, | |
| ) | |
| except Exception as exc: | |
| _rag_init_error = str(exc) | |
| rag_service = None | |
| logging.exception("RAG Service initialization failed") | |
| print(f"[RAG ERROR] RAG Service init failed: {exc}", flush=True) | |
| try: | |
| from spectral_recognition import ( | |
| MultimodalFusion, | |
| SpectralRecognitionSystem, | |
| identify_spectrum_gradio, | |
| query_spectrum_gradio, | |
| ) | |
| SPECTRAL_AVAILABLE = True | |
| print("Spectral analysis module loaded") | |
| except Exception as exc: | |
| print(f"Spectral modules load warning: {exc}") | |
| SPECTRAL_AVAILABLE = False | |
| MultimodalFusion = None | |
| SpectralRecognitionSystem = None | |
| identify_spectrum_gradio = None | |
| query_spectrum_gradio = None | |
| try: | |
| from search_subgraph import CRNTracer, GRAPHML_PATH | |
| _crn_tracer = CRNTracer( | |
| GRAPHML_PATH, | |
| crn_repository=_crn_repository, | |
| ) | |
| G = _crn_tracer.G | |
| CRN_AVAILABLE = True | |
| print("CRN modules loaded successfully") | |
| except Exception as exc: | |
| print(f"CRN modules load warning: {exc}") | |
| CRN_AVAILABLE = False | |
| _crn_tracer = None | |
| G = None | |
| def _save_matplotlib_figure(fig, prefix="fusion_plot"): | |
| if fig is None or not hasattr(fig, "savefig"): | |
| return None | |
| try: | |
| tmp = tempfile.NamedTemporaryFile(delete=False, suffix=".png", prefix=f"{prefix}_") | |
| tmp.close() | |
| fig.savefig(tmp.name, dpi=1200, bbox_inches="tight") | |
| return tmp.name | |
| except Exception as e: | |
| logging.error(f"Failed to save figure for download: {e}") | |
| return None | |
| def _save_gradio_image_file(image, prefix="pathway_trace"): | |
| if image is None: | |
| return None | |
| try: | |
| if isinstance(image, str) and os.path.exists(image): | |
| return image | |
| if not hasattr(image, "save"): | |
| return None | |
| tmp = tempfile.NamedTemporaryFile(delete=False, suffix=".png", prefix=f"{prefix}_") | |
| tmp.close() | |
| image.save(tmp.name, format="PNG") | |
| return tmp.name | |
| except Exception as e: | |
| logging.error(f"Failed to save Gradio image output: {e}", exc_info=True) | |
| return None | |
| def find_cytoscape_image(): | |
| image_path = find_asset( | |
| "cytoscape_network.png", | |
| (PROJECT_ROOT, Path.cwd(), PROJECT_ROOT.parent), | |
| ) | |
| if image_path is None: | |
| print("No static Cytoscape image found") | |
| return None | |
| file_size = image_path.stat().st_size / 1024 | |
| print(f"Found Cytoscape image: {image_path} ({file_size:.1f} KB)") | |
| return str(image_path) | |
| def show_cytoscape_image(): | |
| image = coerce_image_result(find_cytoscape_image()) | |
| if image is None: | |
| print("Unable to display Cytoscape network image") | |
| return image | |
| def show_compound_spectrum(material_name: str, spectrum_type: str): | |
| logging.info(f"show_compound_spectrum called with material_name={material_name!r}, spectrum_type={spectrum_type!r}") | |
| if not SPECTRAL_AVAILABLE: | |
| logging.warning("Spectral module not available in show_compound_spectrum") | |
| return None, "Spectral analysis module not available", gr.update(choices=[], value=None, visible=False), gr.update(visible=False) | |
| def _handle_out(out): | |
| empty_choice = gr.update(choices=[], value=None, visible=False) | |
| hide_confirm = gr.update(visible=False) | |
| if not isinstance(out, tuple): | |
| return None, "Query failed (unexpected return)", empty_choice, hide_confirm | |
| if len(out) == 3: | |
| fig, info, extra = out | |
| if isinstance(extra, (list, tuple)) and extra: | |
| choices = list(extra) | |
| default_choice = str(choices[0]).strip() | |
| try: | |
| auto_out = query_spectrum_gradio(default_choice, spectrum_type) | |
| if isinstance(auto_out, tuple) and len(auto_out) >= 2 and auto_out[0] is not None: | |
| auto_fig = auto_out[0] | |
| auto_info = auto_out[1] | |
| merged_info = ( | |
| f"{auto_info}\n\n" | |
| f"Auto-selected first candidate: {default_choice}. " | |
| "If needed, choose another candidate and click Confirm." | |
| ) | |
| return auto_fig, merged_info, gr.update(choices=choices, value=default_choice, visible=True), gr.update(visible=True) | |
| except Exception: | |
| pass | |
| return None, info, gr.update(choices=choices, value=default_choice, visible=True), gr.update(visible=True) | |
| return fig, info, empty_choice, hide_confirm | |
| if len(out) == 2: | |
| fig, info = out | |
| return fig, info, empty_choice, hide_confirm | |
| return None, "Query returned unexpected shape", empty_choice, hide_confirm | |
| try: | |
| out = query_spectrum_gradio(material_name, spectrum_type) | |
| res = _handle_out(out) | |
| if res: | |
| return res | |
| except Exception: | |
| pass | |
| try: | |
| cleaned = material_name.strip() | |
| if cleaned != material_name: | |
| try: | |
| out = query_spectrum_gradio(cleaned, spectrum_type) | |
| res = _handle_out(out) | |
| if res: | |
| return res | |
| except Exception: | |
| pass | |
| except Exception: | |
| pass | |
| return None, "Spectrum not found for given identifier", gr.update(choices=[], value=None, visible=False), gr.update(visible=False) | |
| def identify_spectrum(files, spectrum_type, enable_fusion=False): | |
| if not SPECTRAL_AVAILABLE: | |
| return "Spectral analysis module not available", None, [], None | |
| # Handle file input: could be single file or multiple files | |
| if not files: | |
| return "Please upload at least one spectrum file", None, [], None | |
| # Convert to list format | |
| if not isinstance(files, list): | |
| files = [files] | |
| # If only one file, use original logic | |
| if len(files) == 1: | |
| file = files[0] | |
| out = identify_spectrum_gradio(file, spectrum_type, enable_fusion=enable_fusion) | |
| if not isinstance(out, tuple): | |
| return "Identification failed (unexpected return)", None, [], None | |
| report_text = None | |
| fig = None | |
| results_struct = [] | |
| fusion_fig = None | |
| if len(out) >= 2: | |
| report_text, fig = out[0], out[1] | |
| if len(out) >= 3: | |
| third = out[2] | |
| if isinstance(third, (list, tuple)): | |
| results_struct = list(third) | |
| else: | |
| results_struct = [] | |
| table_rows = [] | |
| if results_struct: | |
| for r in results_struct: | |
| sim = r.get('similarity') | |
| sim_val = round(sim, 2) if isinstance(sim, (int, float)) else sim | |
| formula = r.get('formula') | |
| try: | |
| formula = to_unicode_subscript(formula) if formula else '' | |
| except Exception: | |
| pass | |
| source = r.get('source') | |
| if source == "实验室": | |
| source = "Laboratory" | |
| table_rows.append([r.get('rank'), sim_val, r.get('name'), formula, r.get('excitation'), source]) | |
| report_text = _normalize_source_text(report_text) | |
| return report_text, fig, table_rows, fusion_fig | |
| # Multiple files: batch identification and fusion | |
| spectrum_types = [spectrum_type for _ in files] | |
| return identify_multiple_spectra_with_types(files, spectrum_types, enable_fusion) | |
| def identify_multiple_spectra_with_types(file_paths, spectrum_types, enable_fusion=False): | |
| """Batch identify multiple spectrum files with their respective types and generate fusion results""" | |
| try: | |
| if not SpectralRecognitionSystem: | |
| return "Spectral analysis not available.", None, [], None | |
| system = SpectralRecognitionSystem() | |
| reports_dict = {} | |
| all_results = [] | |
| # Batch identify each file with its own spectrum type | |
| for file_path, spectrum_type in zip(file_paths, spectrum_types): | |
| if not file_path: | |
| continue | |
| try: | |
| out = system.identify_spectrum(file_path, spectrum_type) | |
| if out and isinstance(out, tuple) and len(out) >= 4: | |
| results, processed_data, adj_spectra, report_obj = out | |
| all_results.append({ | |
| 'file': os.path.basename(file_path), | |
| 'results': results, | |
| 'report_obj': report_obj, | |
| 'spectrum_type': spectrum_type | |
| }) | |
| # Use spectrum type as modal identifier | |
| modal_key = spectrum_type | |
| # If same type exists, append filename to make unique | |
| if modal_key in reports_dict: | |
| modal_key = f"{spectrum_type}_{os.path.basename(file_path)}" | |
| reports_dict[modal_key] = report_obj | |
| except Exception as e: | |
| logging.error(f"Failed to identify {file_path}: {e}") | |
| continue | |
| if not reports_dict: | |
| return "Failed to identify any spectra", None, [], None | |
| # If multiple reports, perform fusion | |
| if len(reports_dict) > 1: | |
| if not MultimodalFusion: | |
| return "Multimodal fusion module not available.", None, [], None | |
| fusion_result = MultimodalFusion.fuse_reports_from_dict(reports_dict, topk=10) | |
| if fusion_result and isinstance(fusion_result, list) and len(fusion_result) > 0: | |
| used_modals = fusion_result[0].get("modals_used", list(reports_dict.keys())) | |
| used_reports = { | |
| modal: reports_dict[modal] | |
| for modal in used_modals | |
| if modal in reports_dict | |
| } | |
| # Generate fusion visualization | |
| fusion_fig = plot_fusion_matrix(fusion_result, used_reports, n_show=10) | |
| # fusion_fig should be matplotlib figure for gr.Plot; keep as-is if valid | |
| # Build fusion result table | |
| table_rows = [] | |
| for i, item in enumerate(fusion_result[:20], 1): | |
| name = item.get('material_name', 'Unknown') | |
| formula = item.get('chemical_formula', '') | |
| score = item.get('fused_score', 0.0) * 100 | |
| k_present = item.get('k_present', 0) | |
| total_modals = len(used_modals) | |
| try: | |
| # Use unicode subscript for Dataframe display | |
| formula = to_unicode_subscript(formula) if formula else '' | |
| except Exception: | |
| pass | |
| table_rows.append([ | |
| i, | |
| round(score, 2), | |
| name, | |
| formula, | |
| f"{k_present}/{total_modals}", | |
| "Complementarity-aware Equal-modality Fusion" | |
| ]) | |
| report_text = f"""**Batch Spectral Identification Completed** | |
| Identified **{len(reports_dict)}** spectrum file(s). | |
| Valid signal modalities used in fusion: **{len(used_modals)}**. | |
| ## 🔄 **Multimodal Fusion Results** | |
| **Fusion Method:** Equal-modality fusion with front-ranked evidence emphasis | |
| **Top-3 Identified Materials:** | |
| """ | |
| for i, item in enumerate(fusion_result[:3], 1): | |
| name = item.get('material_name', 'Unknown') | |
| formula = item.get('chemical_formula', '') | |
| formula = formula_for_markdown(formula) if formula else '' | |
| score = item.get('fused_score', 0.0) * 100 | |
| k_present = item.get('k_present', 0) | |
| report_text += f""" | |
| {i}. **{name}** ({formula}) | |
| - Fusion Score: {score:.2f}% | |
| - Appearance Count: {k_present}/{len(used_modals)} modalities | |
| """ | |
| report_text = _normalize_source_text(report_text) | |
| return report_text, None, table_rows, fusion_fig | |
| # If only one report, return single result | |
| if len(all_results) == 1: | |
| result_data = all_results[0] | |
| results = result_data['results'] | |
| table_rows = [] | |
| for i, r in enumerate(results[:20], 1): | |
| stype, sim, name, formula, ex_wave, source, idx, peak_shift = r | |
| sim_val = round(sim * 100, 2) | |
| try: | |
| formula = to_unicode_subscript(formula) if formula else '' | |
| except Exception: | |
| pass | |
| if source == "实验室": | |
| source = "Laboratory" | |
| table_rows.append([i, sim_val, name, formula, ex_wave if ex_wave else '', source]) | |
| report_text = f"""**Spectral Identification Completed** | |
| File: {result_data['file']} | |
| Best Match: {results[0][2]} ({formula_for_markdown(results[0][3]) if results[0][3] else ''}) | |
| Cosine similarity: {results[0][1]*100:.2f}% | |
| """ | |
| report_text = _normalize_source_text(report_text) | |
| return report_text, None, table_rows, None | |
| return "Identification completed, but unable to generate fusion results", None, [], None | |
| except Exception as e: | |
| logging.error(f"Batch identification failed: {e}", exc_info=True) | |
| return f"Batch identification failed: {str(e)}", None, [], None | |
| def plot_fusion_matrix(fusion_result, reports_dict, n_show=10): | |
| """Plot fusion result matrix (ranking bar chart + confusion matrix)""" | |
| try: | |
| if plt is None or GridSpec is None or LinearSegmentedColormap is None: | |
| logging.warning("Matplotlib is not available for fusion matrix plotting") | |
| return None | |
| # Validate inputs | |
| if not fusion_result or not isinstance(fusion_result, list): | |
| logging.warning(f"Invalid fusion_result: {type(fusion_result)}") | |
| return None | |
| if not reports_dict or not isinstance(reports_dict, dict): | |
| logging.warning(f"Invalid reports_dict: {type(reports_dict)}") | |
| return None | |
| # Get top n_show results | |
| show = sorted(fusion_result, key=lambda x: x.get("fused_score", 0.0), reverse=True)[:n_show] | |
| if not show: | |
| return None | |
| # Prepare data | |
| material_names = [] | |
| material_formulas = [] | |
| material_labels = [] | |
| scores = [] | |
| k_present_list = [] | |
| modal_ranks_matrix = [] | |
| modals = ( | |
| list(show[0].get("modals_used", [])) | |
| if show | |
| else list(reports_dict.keys()) | |
| ) | |
| if not modals: | |
| modals = list(reports_dict.keys()) | |
| for r in show: | |
| name = r.get("material_name", "") | |
| formula = r.get("chemical_formula", "") | |
| material_names.append(name) | |
| material_formulas.append(formula) | |
| if formula: | |
| try: | |
| formula_sub = formula_for_markdown(formula) | |
| label = f"{name}\n({formula_sub})" | |
| except: | |
| label = f"{name}\n({formula})" | |
| else: | |
| label = name | |
| material_labels.append(label) | |
| scores.append(r.get("fused_score", 0.0) * 100) | |
| k_present_list.append(r.get("k_present", 0)) | |
| # Collect rankings for each modal | |
| row_ranks = [] | |
| per_modal = r.get("per_modal", {}) | |
| for modal in modals: | |
| rank = per_modal.get(modal, {}).get("rank") | |
| row_ranks.append(rank) | |
| modal_ranks_matrix.append(row_ranks) | |
| # Create figure | |
| fig = plt.figure(figsize=(14, 6)) | |
| gs = GridSpec(1, 2, figure=fig, wspace=0.3, left=0.08, right=0.96, top=0.92, bottom=0.14) | |
| # Left: horizontal bar chart | |
| ax1 = fig.add_subplot(gs[0]) | |
| colors = ["#4C72B0"] * len(scores) | |
| if colors: | |
| colors[0] = "#C44E52" # Top-1 in red | |
| bars = ax1.barh(material_labels, scores, height=0.65, color=colors, | |
| edgecolor="#2E2E2E", linewidth=0.6, alpha=0.95) | |
| ax1.invert_yaxis() | |
| for bar, pr, ap in zip(bars, scores, k_present_list): | |
| x = bar.get_width() | |
| ax1.text(x + 1.0, bar.get_y() + bar.get_height() / 2, | |
| f"{pr:.1f}% (k={ap})", va="center", ha="left", fontsize=9) | |
| ax1.set_xlabel("Score (%)", fontsize=11) | |
| ax1.set_xlim(0, 100) | |
| ax1.set_title("(1) Fusion Results", loc="left", fontsize=12) | |
| ax1.grid(axis="x", alpha=0.35, linestyle="--", linewidth=0.6) | |
| ax1.spines["right"].set_visible(False) | |
| ax1.spines["top"].set_visible(False) | |
| # Right: ranking matrix heatmap | |
| ax2 = fig.add_subplot(gs[1]) | |
| n_materials = len(material_names) | |
| n_modals = len(modals) | |
| rank_matrix = np.full((n_materials, n_modals), np.nan, dtype=float) | |
| for i, row_ranks in enumerate(modal_ranks_matrix): | |
| for j, rank in enumerate(row_ranks): | |
| if rank is not None: | |
| try: | |
| rank_matrix[i, j] = float(rank) | |
| except: | |
| rank_matrix[i, j] = np.nan | |
| cmap_hm = LinearSegmentedColormap.from_list( | |
| "match_blue_scale", ["#1F3B63", "#4C72B0", "#EEF3F8"], N=256 | |
| ) | |
| cmap_hm.set_bad(color="#F2F2F2") | |
| im = ax2.imshow(rank_matrix, cmap=cmap_hm, aspect="auto", vmin=1, vmax=20, interpolation="nearest") | |
| # Simplify modal name display (show only filename) | |
| modal_labels_disp = [os.path.basename(m).split('_', 1)[-1][:15] for m in modals] | |
| ax2.set_xticks(np.arange(len(modals))) | |
| ax2.set_yticks(np.arange(len(material_names))) | |
| ax2.set_xticklabels(modal_labels_disp, fontsize=9, rotation=0, ha='center') | |
| ax2.set_yticklabels([name[:20] for name in material_names], fontsize=9) | |
| ax2.set_title("(2) Ranking Matrix", loc="left", fontsize=12) | |
| # Add value labels | |
| for i in range(len(material_names)): | |
| for j in range(len(modals)): | |
| v = rank_matrix[i, j] | |
| if np.isnan(v): | |
| ax2.text(j, i, "N/A", ha="center", va="center", fontsize=8, color="#777777") | |
| else: | |
| rk = int(v) | |
| txt_color = "white" if rk <= 6 else "#1F1F1F" | |
| ax2.text(j, i, str(rk), ha="center", va="center", fontsize=8, | |
| color=txt_color, fontweight="bold") | |
| cbar = plt.colorbar(im, ax=ax2, orientation="vertical", pad=0.02, shrink=0.88) | |
| cbar.set_label("Rank", fontsize=10) | |
| cbar.set_ticks([1, 5, 10, 15, 20]) | |
| # plt.tight_layout() # Removed to avoid conflict with GridSpec manual margins | |
| # Return matplotlib figure for gr.Plot (do not close fig so Gradio can render it) | |
| return fig | |
| except Exception as e: | |
| logging.error(f"Failed to plot fusion matrix: {e}", exc_info=True) | |
| return None | |
| def _merge_copilot_state(current_state, updates): | |
| base = dict(current_state or {}) | |
| base.update(updates or {}) | |
| return base | |
| def _sanitize_for_state(value): | |
| if isinstance(value, str): | |
| if os.path.isdir(value): | |
| return f"[dir:{value}]" | |
| if os.path.isfile(value): | |
| return f"[file:{os.path.basename(value)}]" | |
| return value | |
| if isinstance(value, dict): | |
| return {k: _sanitize_for_state(v) for k, v in value.items()} | |
| if isinstance(value, (list, tuple)): | |
| return [ _sanitize_for_state(v) for v in value ] | |
| return value | |
| _FORMULA_MARKDOWN_PATTERN = re.compile(r'\b(?:p-|alpha-|beta-)?[A-Za-z][A-Za-z0-9\-\(\)·\.]*\d[A-Za-z0-9\-\(\)·\.]*\b') | |
| def _clean_report_markdown(text: str) -> str: | |
| if not text: | |
| return "" | |
| text = text.strip() | |
| text = re.sub(r'^```[^\n]*\n', '', text) | |
| text = re.sub(r'\n```$', '', text) | |
| text = text.replace("```", "").strip() | |
| text = _fix_latex_artifacts(text) | |
| lines = text.splitlines() | |
| cleaned = [] | |
| header_seen = False | |
| for line in lines: | |
| stripped = line.strip() | |
| if stripped and stripped.lower().startswith("provenance analysis report:"): | |
| if header_seen: | |
| continue | |
| header_seen = True | |
| cleaned.append(line) | |
| merged = _merge_similar_paragraphs("\n".join(cleaned).strip()) | |
| return _fix_latex_artifacts(merged) | |
| def _fix_latex_artifacts(text: str) -> str: | |
| if not text: | |
| return "" | |
| # Strip control/replacement chars | |
| text = re.sub(r"[\x00-\x08\x0b\x0c\x0e-\x1f\x7f]", " ", text) | |
| text = text.replace("\ufffd", "") | |
| # Preserve DOI segments to avoid subscript conversion | |
| doi_spans = [] | |
| def _save_doi(match): | |
| doi_spans.append(match.group(0)) | |
| return f"__DOI__{len(doi_spans)-1}__" | |
| text = re.sub(r"(DOI:\s*10\.\d{4,9}/\S+|10\.\d{4,9}/\S+)", _save_doi, text, flags=re.IGNORECASE) | |
| # Protect math blocks | |
| math_blocks = [] | |
| def _save_math(m): | |
| math_blocks.append(m.group(0)) | |
| return f"__MATH__{len(math_blocks)-1}__" | |
| text = re.sub(r"\$\$.*?\$\$", _save_math, text, flags=re.DOTALL) | |
| text = re.sub(r"\$[^$]+\$", _save_math, text) | |
| # Fix malformed parenthesized subscripts. | |
| text = re.sub(r"_\((\d+)\)", r"_{\1}", text) | |
| text = re.sub(r"_\{(\d+)\}", r"_{\1}", text) | |
| def _strip_dollar(match): | |
| inner = match.group(1).strip() | |
| if inner.startswith("{"): | |
| inner = inner[1:].strip() | |
| if inner.endswith("}"): | |
| inner = inner[:-1].strip() | |
| return inner | |
| text = re.sub(r"\$\{?([^$]*)\}?\$", _strip_dollar, text) | |
| # Normalize reaction arrows in plain text | |
| text = re.sub(r"\\xrightarrow\s*\{([^}]*)\}", r" --[\1]--> ", text) | |
| text = re.sub(r"xrightarrow\s*\{([^}]*)\}", r" --[\1]--> ", text) | |
| text = re.sub(r"\\xrightarrow([A-Za-z0-9\+\-\s]+)", r" --[\1]--> ", text) | |
| text = re.sub(r"xrightarrow([A-Za-z0-9\+\-\s]+)", r" --[\1]--> ", text) | |
| text = text.replace("\\rightarrow", " -> ") | |
| # Strip LaTeX text commands | |
| text = re.sub(r"\\(?:text|mathrm)\s*{([^}]*)}", r"\1", text) | |
| text = re.sub(r"(?i)\\?mathrm", "", text) | |
| text = re.sub(r"(?i)mathrm", "", text) | |
| text = re.sub(r"\bext\s*([A-Z][a-z]?)(?=\d|[A-Z])", r"\1", text) | |
| text = re.sub(r"\bext([A-Z][a-z]?)\b", r"\1", text) | |
| text = re.sub(r"\btext\s*(?=[A-Za-z])", "", text, flags=re.IGNORECASE) | |
| text = re.sub(r"\btext\s*p\s*-\s*", "p-", text, flags=re.IGNORECASE) | |
| text = text.replace("p-}", "p-") | |
| text = re.sub(r"([A-Z][a-z]?)\s+([0-9]+)", r"\1\2", text) | |
| text = re.sub(r"([0-9]+)\s+([A-Z][a-z]?)", r"\1\2", text) | |
| text = re.sub( | |
| r"\b(?:[A-Za-z]\s+){2,}[A-Za-z]\b", | |
| lambda match: match.group(0).replace(" ", ""), | |
| text, | |
| ) | |
| text = re.sub(r"\\[a-zA-Z]+\s*", " ", text) | |
| text = re.sub(r"\\[^a-zA-Z]", "", text) | |
| def _subscript_formula(match): | |
| token = match.group(1) | |
| # Evidence identifiers are provenance markers, not chemical formulae. | |
| if re.fullmatch(r"E\d+", token, flags=re.IGNORECASE): | |
| return token | |
| if "-" in token and token.lower().startswith(("p-", "alpha-", "beta-")): | |
| prefix, rest = token.split("-", 1) | |
| return f"{prefix}-{to_unicode_subscript(rest)}" | |
| return to_unicode_subscript(token) | |
| text = re.sub(r"\b([A-Za-z][A-Za-z0-9\-]*\d[A-Za-z0-9\-]*)\b", _subscript_formula, text) | |
| # Preserve newlines for Markdown sections | |
| text = re.sub(r"[ \t]{2,}", " ", text).strip() | |
| # Restore math blocks | |
| for i, blk in enumerate(math_blocks): | |
| text = text.replace(f"__MATH__{i}__", blk) | |
| for i, doi in enumerate(doi_spans): | |
| text = text.replace(f"__DOI__{i}__", doi) | |
| return text | |
| def _report_to_text(report_item) -> str: | |
| if report_item is None: | |
| return "" | |
| if isinstance(report_item, dict): | |
| return report_item.get("report") or report_item.get("content") or str(report_item) | |
| return str(report_item) | |
| def _extract_report_title(text: str) -> str: | |
| if not text: | |
| return "Provenance Analysis Report" | |
| for line in text.splitlines(): | |
| stripped = line.strip() | |
| if not stripped: | |
| continue | |
| if stripped.lower().startswith("provenance analysis report:"): | |
| title = _normalize_rag_text(_fix_latex_artifacts(stripped)) | |
| # Keep only up to " in Mural Pigments" so body text ("Continued exposure...") is separate | |
| if " in mural pigments" in title.lower(): | |
| idx = title.lower().find(" in mural pigments") | |
| if idx != -1: | |
| suffix_len = len(" in Mural Pigments") | |
| title = title[:idx + suffix_len].strip() | |
| title = _normalize_report_formula_display(title) | |
| return title | |
| if stripped.startswith("#"): | |
| title = _normalize_rag_text(_fix_latex_artifacts(stripped.lstrip("# ").strip())) | |
| # Fix chemical formulas that might be split | |
| title = _normalize_report_formula_display(title) | |
| return title | |
| return "Provenance Analysis Report" | |
| def _path_to_plain_expression(path_str: str) -> str: | |
| species, conditions = path_components(path_str, include_empty_conditions=True) if path_str else ([], []) | |
| if not species: | |
| return "" | |
| material_path = " -> ".join(to_unicode_subscript(normalize_legacy_text(s)) for s in species) | |
| unique_conditions = list(dict.fromkeys( | |
| normalize_legacy_text(str(c).strip()) for c in conditions if str(c).strip() | |
| )) | |
| if unique_conditions: | |
| return f"{material_path}; conditions: {'; '.join(unique_conditions)}" | |
| return material_path | |
| def _path_report_title(path_str: str) -> str: | |
| """Build a generic title that names both the path source and searched target.""" | |
| species, _conditions = ( | |
| path_components(path_str, include_empty_conditions=True) | |
| if path_str | |
| else ([], []) | |
| ) | |
| if len(species) >= 2: | |
| start = to_unicode_subscript( | |
| render_chem_text(normalize_legacy_text(species[0])) | |
| ) | |
| target = to_unicode_subscript( | |
| render_chem_text(normalize_legacy_text(species[-1])) | |
| ) | |
| return ( | |
| "Provenance Analysis Report: Pathway from " | |
| f"{start} to {target} in Mural Pigments" | |
| ) | |
| return "Provenance Analysis Report" | |
| # Graph condition labels are stored with ad-hoc casing (e.g. "Uv+Oxidant"). | |
| # Present well-known reaction-condition tokens with their conventional casing so | |
| # the report reads consistently. Display-only: the evidence matcher casefolds | |
| # conditions, so this never affects retrieval or verification. | |
| def _path_to_display_equation(path_str: str) -> str: | |
| species, conditions = path_components(path_str, include_empty_conditions=True) if path_str else ([], []) | |
| if not species: | |
| return "" | |
| def _html_species(value: str) -> str: | |
| return html.escape(to_unicode_subscript(normalize_legacy_text(str(value or "").strip()))) | |
| def _html_condition(value: str) -> str: | |
| return html.escape(_display_condition_label(value).replace("+", " + ")) | |
| parts = [f"<span class='chem-species'>{_html_species(species[0])}</span>"] | |
| for idx, product in enumerate(species[1:]): | |
| condition = normalize_legacy_text(conditions[idx]) if idx < len(conditions) else "" | |
| if condition: | |
| parts.append( | |
| "<span class='chem-arrow' style='display:inline-flex; flex-direction:column; " | |
| "align-items:center; justify-content:center; min-width:42px; line-height:1;'>" | |
| f"<span class='chem-condition' style='font-size:10px; white-space:nowrap; margin-bottom:1px;'>{_html_condition(condition)}</span>" | |
| "<span class='chem-arrow-line' style='font-size:18px; line-height:0.9;'>⟶</span>" | |
| "</span>" | |
| ) | |
| else: | |
| parts.append( | |
| "<span class='chem-arrow' style='display:inline-flex; align-items:center; " | |
| "justify-content:center; min-width:42px;'>" | |
| "<span class='chem-arrow-line' style='font-size:18px;'>⟶</span>" | |
| "</span>" | |
| ) | |
| parts.append(f"<span class='chem-species'>{_html_species(product)}</span>") | |
| return ( | |
| "<div class='chem-equation' " | |
| "style='display:flex; align-items:center; justify-content:center; gap:8px; " | |
| "flex-wrap:wrap; margin:10px 0 4px; font-size:14px; line-height:1.2;'>" | |
| + "".join(parts) | |
| + "</div>" | |
| ) | |
| def _split_scope_note(text: str) -> tuple: | |
| """Separate the trailing "Validation note: ..." scope caveat from prose. | |
| The evidence guard appends this note to the narrative body; splitting it | |
| lets the Introduction stay clean while the caveat is rendered as its own | |
| labeled block. Returns (narrative, scope_note); scope_note is "" if absent. | |
| """ | |
| if not text: | |
| return "", "" | |
| match = re.search(r"(?is)\s*validation note:\s*", text) | |
| if not match: | |
| return text.strip(), "" | |
| narrative = text[: match.start()].strip() | |
| note = text[match.end():].strip() | |
| # The stripped "Validation note: " prefix left the remainder starting in | |
| # lower case under its own heading ("generated claims ... were omitted."). | |
| if note: | |
| note = note[0].upper() + note[1:] | |
| return narrative, note | |
| def _clean_public_report_text(text: str) -> str: | |
| if not text: | |
| return "" | |
| cleaned_lines = [] | |
| for line in str(text).replace("\r\n", "\n").replace("\r", "\n").splitlines(): | |
| stripped = line.strip() | |
| low = stripped.lower() | |
| if not stripped: | |
| cleaned_lines.append("") | |
| continue | |
| if is_bad_generation_text(stripped): | |
| continue | |
| if low.startswith("key literature evidence"): | |
| continue | |
| if re.match(r"^\[\d+\]\s*\(score\s*=", stripped, flags=re.IGNORECASE): | |
| continue | |
| stripped = re.sub(r"\s*\(score\s*=\s*[^)]*\)", "", stripped, flags=re.IGNORECASE) | |
| cleaned_lines.append(stripped) | |
| cleaned = "\n".join(cleaned_lines) | |
| cleaned = _strip_pathway_expression_text(cleaned) | |
| cleaned = re.sub(r"\n{3,}", "\n\n", cleaned).strip() | |
| # Shared rendering step (prose: do not pad "+", it breaks "+3 oxidation state"). | |
| cleaned = render_chem_text(cleaned) | |
| return _normalize_report_formula_display(cleaned) | |
| def _strip_pathway_expression_text(text: str) -> str: | |
| if not text: | |
| return "" | |
| text = re.sub( | |
| r"(?is)\s*Pathway expression:\s*.*?(?=(?:\n\s*\n|###\s|$))", | |
| "", | |
| str(text), | |
| ) | |
| return re.sub(r"[ \t]{2,}", " ", text).strip() | |
| def _strip_all_citation_numbers(text: str) -> str: | |
| if not text: | |
| return "" | |
| return re.sub(r"\s*\[\d+\]", "", str(text)).strip() | |
| def _clean_evidence_snippet(snippet: str, max_chars: int = 260) -> str: | |
| if not snippet: | |
| return "" | |
| text = str(snippet) | |
| text = re.sub(r"(?m)^\s*#{1,6}\s+[^\n]+\n+", "", text) | |
| text = re.sub(r"\s+", " ", text).strip() | |
| text = re.sub(r"\s*\(score\s*=\s*[^)]*\)", "", text, flags=re.IGNORECASE).strip() | |
| text = text.replace("Key literature evidence:", "").strip() | |
| # Shared rendering step. Snippets are chemistry-dense, so operators are | |
| # padded here (unlike prose). Must run before the Markdown emphasis strip | |
| # so TeX subscripts like "_{4}" are consumed rather than mangled by it. | |
| text = render_chem_text(text, space_operators=True) | |
| # Evidence is rendered inside Markdown; remove source emphasis markers | |
| # without changing the quoted words. | |
| text = re.sub(r"(?<!\w)[_*]([A-Za-z][^_*\n]{0,80})[_*](?!\w)", r"\1", text) | |
| text = re.sub(r"\b([A-Za-z]+)[_*](?=\s|[.,;:]|$)", r"\1", text) | |
| text = re.sub(r"\s+", " ", text).strip() | |
| if len(text) <= max_chars: | |
| return text | |
| cut = text[:max_chars].rstrip() | |
| sentence_end = max(cut.rfind("."), cut.rfind(";"), cut.rfind(",")) | |
| if sentence_end > max_chars * 0.55: | |
| cut = cut[: sentence_end + 1] | |
| return cut.rstrip(" ,;") + "..." | |
| def _format_path_evidence_reference(source): | |
| source = normalize_legacy_text(str(source or "").strip()) | |
| if not source: | |
| return "" | |
| return f"DOI: {source}" if first_doi(source) else f"CRN source: {source}" | |
| def _attach_path_evidence_to_report(report, path_info): | |
| if not isinstance(report, dict) or not isinstance(path_info, dict): | |
| return report | |
| evidence_sources = [ | |
| normalize_legacy_text(str(src).strip()) | |
| for src in (path_info.get("evidence_sources") or path_info.get("step_evidence_sources") or []) | |
| if str(src).strip() | |
| ] | |
| evidence_sources = list(dict.fromkeys(evidence_sources)) | |
| if not evidence_sources: | |
| return report | |
| report["path_evidence_sources"] = evidence_sources | |
| report["path_evidence_type"] = path_info.get("evidence_type") | |
| references = list(report.get("references") or []) | |
| for src in evidence_sources: | |
| if not any(reference_contains_source(ref, src) for ref in references): | |
| ref = _format_path_evidence_reference(src) | |
| if ref: | |
| references.append(ref) | |
| report["references"] = references | |
| bundles = report.get("reference_snippets") | |
| if isinstance(bundles, list): | |
| while len(bundles) < len(references): | |
| bundles.append({ | |
| "text": "", | |
| "score": None, | |
| "snippets": [], | |
| "provenance_only": True, | |
| }) | |
| evidence = report.get("evidence") | |
| if isinstance(evidence, list): | |
| while len(evidence) < len(references): | |
| reference = references[len(evidence)] | |
| evidence.append({ | |
| "source": reference, | |
| "snippet": "", | |
| "snippets": [], | |
| "edge_matches": [], | |
| "evidence_status": "metadata_only", | |
| "provenance_only": True, | |
| }) | |
| metrics = report.get("metrics") | |
| if isinstance(metrics, dict): | |
| metrics["ref_count"] = len(references) | |
| return report | |
| def _merge_similar_paragraphs(text: str) -> str: | |
| if not text: | |
| return "" | |
| paras = [] | |
| current = [] | |
| for line in text.splitlines(): | |
| if not line.strip(): | |
| if current: | |
| paras.append("\n".join(current).strip()) | |
| current = [] | |
| continue | |
| if line.lstrip().startswith("#"): | |
| if current: | |
| paras.append("\n".join(current).strip()) | |
| current = [] | |
| paras.append(line.strip()) | |
| continue | |
| current.append(line) | |
| if current: | |
| paras.append("\n".join(current).strip()) | |
| def _is_heading(p: str) -> bool: | |
| s = p.strip().lower() | |
| return ( | |
| s.startswith("#") | |
| or s.startswith("references") | |
| or s.startswith("quality metrics") | |
| or s.startswith("sentence traceability") | |
| or s.startswith("**quality metrics**") | |
| or s.startswith("**sentence traceability**") | |
| ) | |
| def _normalize_for_similarity(p: str) -> str: | |
| s = re.sub(r'\[[0-9]+\]', '', p) | |
| s = re.sub(r'[`*_#>-]', '', s) | |
| s = re.sub(r'\s+', ' ', s).strip().lower() | |
| return s | |
| def _split_sentences(p: str) -> list: | |
| parts = re.split(r'(?<=[.!?])\s+', p.strip()) | |
| return [x.strip() for x in parts if x.strip()] | |
| def _merge_sentences(a: str, b: str) -> str: | |
| a_sents = _split_sentences(a) | |
| b_sents = _split_sentences(b) | |
| merged = [] | |
| for sent in a_sents + b_sents: | |
| if not any(difflib.SequenceMatcher(None, sent, seen).ratio() > 0.9 for seen in merged): | |
| merged.append(sent) | |
| return " ".join(merged).strip() | |
| merged_paras = [] | |
| last_body_idx = None | |
| for para in paras: | |
| if _is_heading(para): | |
| merged_paras.append(para) | |
| last_body_idx = None | |
| continue | |
| if last_body_idx is not None: | |
| prev = merged_paras[last_body_idx] | |
| sim = difflib.SequenceMatcher( | |
| None, | |
| _normalize_for_similarity(prev), | |
| _normalize_for_similarity(para), | |
| ).ratio() | |
| if sim >= 0.88: | |
| merged_paras[last_body_idx] = _merge_sentences(prev, para) | |
| continue | |
| merged_paras.append(para) | |
| last_body_idx = len(merged_paras) - 1 | |
| return "\n\n".join([p for p in merged_paras if p]).strip() | |
| def _inline_math_to_unicode(text: str) -> str: | |
| if not text: | |
| return "" | |
| def _strip_tex(s: str) -> str: | |
| s = re.sub(r'\\(?:mathrm|text)\{([^}]*)\}', r'\1', s) | |
| s = s.replace("\\cdotp", "·").replace("\\cdot", "·") | |
| s = s.replace("\\bullet", "·").replace("\\times", "×") | |
| s = s.replace("\\alpha", "α").replace("\\beta", "β").replace("\\gamma", "γ") | |
| s = s.replace("{", "").replace("}", "") | |
| s = re.sub(r'_\{?(\d+)\}?', r'\1', s) | |
| return s | |
| def _convert(match): | |
| inner = match.group(1) | |
| if not re.fullmatch(r'[A-Za-z0-9_\-\(\)\\{}\u00b7\.\s\+\u03b1-\u03c9]+', inner): | |
| return match.group(0) | |
| plain = _strip_tex(inner) | |
| return to_unicode_subscript(plain) | |
| return re.sub(r'\$([^\$]+)\$', _convert, text) | |
| def _edge_matches_in(evidence_blocks) -> list: | |
| """Every edge match across the displayed evidence blocks.""" | |
| matches = [] | |
| for block in evidence_blocks or []: | |
| if not isinstance(block, dict): | |
| continue | |
| for match in block.get("edge_matches") or []: | |
| if isinstance(match, dict): | |
| matches.append(match) | |
| return matches | |
| def _condition_corroboration_note(evidence_blocks) -> str: | |
| """How far the retrieved passages corroborate the graph's condition label. | |
| The graph condition and the condition a paper actually reports are separate | |
| claims, and a provenance report has to keep them apart: a step labelled UV | |
| may rest on a paper describing visible-light exposure. The matcher already | |
| records, per passage, whether the condition was stated in the passage | |
| itself, elsewhere in the same paper, only partly, or not at all -- this | |
| turns that into one sentence rather than asserting the label was proven. | |
| """ | |
| statuses = { | |
| str(match.get("condition_status") or "") | |
| for match in _edge_matches_in(evidence_blocks) | |
| if match.get("verdict") == "direct" | |
| } | |
| if not statuses: | |
| return "" | |
| if "exact" in statuses: | |
| return "stated in the cited passages" | |
| if "exact_document" in statuses: | |
| return "stated elsewhere in the cited sources, not in the quoted passages" | |
| if "partial" in statuses: | |
| return "partly stated in the cited sources" | |
| return "not stated in the retrieved passages" | |
| #: Measured quantities that describe a reaction condition. A wavelength or an | |
| #: irradiance is the specific form of the condition the graph records only as a | |
| #: label, so these are what make "visible light (500-670 nm)" reportable instead | |
| #: of just "light". | |
| _CONDITION_QUANTITY_RE = re.compile( | |
| r"\d[\d,.]*\s*(?:[-–—]\s*\d[\d,.]*\s*)?" | |
| r"(?:nm|lux|lx|kW/cm2|W\s*/?\s*m\s*-?\s*2|%\s*RH|%|°\s*C|℃)", | |
| re.IGNORECASE, | |
| ) | |
| def _literature_conditions_by_edge(evidence_blocks) -> dict: | |
| """Condition wording the retrieved passages actually report, per edge. | |
| The graph records a condition as a label ("Uv+Oxidant"); a paper reports | |
| what was really done ("visible light 500-670 nm", "oxygen atmosphere"). | |
| Those are different claims, and a provenance report that prints only the | |
| label lets it borrow the evidence's authority. The matcher keeps the passage | |
| it matched on, so the reported condition can be read back out of it: terms | |
| come from the project's own condition vocabulary, so this grows with the | |
| vocabulary instead of hard-coding light and oxygen. | |
| """ | |
| vocabulary = load_condition_vocabulary() | |
| terms = set() | |
| for label, aliases in (vocabulary or {}).items(): | |
| terms.update(str(a).strip().casefold() for a in aliases if str(a).strip()) | |
| terms.add(str(label).strip().casefold()) | |
| terms = {t for t in terms if len(t) >= 3} | |
| by_edge: dict = {} | |
| for block in evidence_blocks or []: | |
| if not isinstance(block, dict): | |
| continue | |
| for match in block.get("edge_matches") or []: | |
| if not isinstance(match, dict) or match.get("verdict") != "direct": | |
| continue | |
| window = str(match.get("window") or "") | |
| if not window: | |
| continue | |
| folded = window.casefold() | |
| found = [term for term in terms if term in folded] | |
| # Keep the most specific wording only: "visible light" makes the | |
| # bare "light" it contains redundant. | |
| found = [ | |
| term for term in found | |
| if not any(term != other and term in other for other in found) | |
| ] | |
| found += [ | |
| re.sub(r"\s+", " ", q.group(0)).strip() | |
| for q in _CONDITION_QUANTITY_RE.finditer(window) | |
| ] | |
| if not found: | |
| continue | |
| key = ( | |
| match.get("edge_index"), | |
| str(match.get("reactant") or ""), | |
| str(match.get("product") or ""), | |
| match.get("evidence_scope"), | |
| ) | |
| bucket = by_edge.setdefault(key, []) | |
| for item in found: | |
| if item not in bucket: | |
| bucket.append(item) | |
| return by_edge | |
| def _literature_condition_note(evidence_blocks) -> str: | |
| """One line naming, per edge, the conditions the literature reports.""" | |
| by_edge = _literature_conditions_by_edge(evidence_blocks) | |
| parts = [] | |
| # Graph order, with the whole-pathway entry last: it summarises the steps | |
| # above it, and its edge index is 0, which would otherwise sort it first and | |
| # read as though it came before the first step. | |
| def _order(key): | |
| index, _reactant, _product, scope = key | |
| return (1 if scope == "pathway_endpoint" else 0, index if index else 0) | |
| for key in sorted(by_edge, key=_order): | |
| index, reactant, product, scope = key | |
| wording = ", ".join(by_edge[key]) | |
| if scope == "pathway_endpoint" or not (reactant and product): | |
| parts.append(f"{wording} for the overall pathway") | |
| else: | |
| parts.append( | |
| f"{wording} for {to_unicode_subscript(reactant)} → " | |
| f"{to_unicode_subscript(product)}" | |
| ) | |
| return "; ".join(parts) | |
| #: Evidence bases that report a conversion but not how it proceeds. An edge | |
| #: supported only by these is established as a transformation while its | |
| #: mechanism remains unresolved -- the distinction the limitation note exists | |
| #: to state. | |
| _CONVERSION_ONLY_BASES = frozenset({ | |
| "observed_conversion", "product_identification", "phase_table_conversion", | |
| }) | |
| def _evidence_limitation_note(evidence_blocks, path_str) -> str: | |
| """State what the retrieved evidence does NOT establish. | |
| A report that only lists support invites the reader to assume the mechanism | |
| was demonstrated. Derived from the same match records as the evidence list | |
| so the two cannot disagree: steps with no direct passage are named as | |
| graph-derived, and supported steps whose evidence only reports a conversion | |
| are marked as leaving the mechanism unresolved. | |
| """ | |
| matches = _edge_matches_in(evidence_blocks) | |
| if not matches: | |
| return "" | |
| supported, bases = set(), set() | |
| for match in matches: | |
| if match.get("evidence_scope") == "pathway_endpoint": | |
| continue | |
| if match.get("verdict") == "direct": | |
| supported.add(match.get("edge_index")) | |
| bases.add(str(match.get("relation_basis") or "")) | |
| edges = parse_path(path_str) if path_str else [] | |
| unsupported = [ | |
| edge for edge in edges if edge.index not in supported | |
| ] | |
| sentences = [] | |
| if supported and bases and bases.issubset(_CONVERSION_ONLY_BASES): | |
| sentences.append( | |
| "The retrieved literature supports the transformation " | |
| "relationships listed above, but does not resolve the molecular " | |
| "intermediates, by-products, or elementary reaction steps involved." | |
| ) | |
| if unsupported: | |
| names = ", ".join( | |
| f"{to_unicode_subscript(edge.reactant)} → {to_unicode_subscript(edge.product)}" | |
| for edge in unsupported | |
| ) | |
| sentences.append( | |
| f"No passage directly establishes {names}; " | |
| "that step remains graph-derived." | |
| ) | |
| return " ".join(sentences) | |
| #: Tokens the formula pattern below matches but which are not chemical formulas: | |
| #: retrieval-channel and method names shown in the evidence list. Compared | |
| #: case-insensitively. | |
| _NON_FORMULA_TOKENS = frozenset({"BGE-M3", "BM25"}) | |
| def _normalize_report_formula_display(text: str) -> str: | |
| if not text: | |
| return "" | |
| text = _inline_math_to_unicode(str(text)) | |
| text = text.replace("\\cdotp", "·").replace("\\cdot", "·") | |
| text = text.replace("\\bullet", "·").replace("\\times", "×") | |
| text = text.replace("\\alpha", "α").replace("\\beta", "β").replace("\\gamma", "γ") | |
| text = re.sub(r"\\(?:mathrm|text)\{([^}]*)\}", r"\1", text) | |
| text = re.sub(r"_\{?(\d+)\}?", r"\1", text) | |
| # Variable stoichiometry is written with letter subscripts -- "As_xS_y" for a | |
| # non-stoichiometric arsenic sulfide -- and the digit rule above cannot see | |
| # it, so the raw underscore reached the reader. Unicode is not an option | |
| # here: it has a subscript x but no subscript y, so half the formula would | |
| # stay on the baseline. Rendered as HTML instead, which this pipeline | |
| # already emits. An element symbol must precede the underscore, which is | |
| # what keeps ordinary underscored words ("tung_oil") untouched. | |
| # Brace form first, so its closing brace is always consumed; the bare form | |
| # would otherwise leave it stranded. The bare form must allow another | |
| # element symbol to follow -- "As_xS_y" is the whole point -- so it is | |
| # bounded by "not another lowercase letter", which is what keeps a real word | |
| # like "As_xylene" from being cut after its first letter. | |
| text = re.sub(r"([A-Z][a-z]?)_\{([a-z])\}", r"\1<sub>\2</sub>", text) | |
| # No look-behind here: in "As_xS_y" the S is preceded by the previous | |
| # subscript letter, and a look-behind is evaluated against the original | |
| # string, so guarding on it left the second half unconverted. The required | |
| # uppercase element symbol is protection enough -- an identifier like | |
| # "my_var_x" has no uppercase letter in front of the underscore. | |
| text = re.sub(r"([A-Z][a-z]?)_([a-z])(?![a-z])", r"\1<sub>\2</sub>", text) | |
| text = re.sub(r"\s*·\s*", "·", text) | |
| formula_re = re.compile( | |
| r"\b(?:alpha-|beta-|p-)?(?:\d+)?[A-Z][A-Za-z0-9\-\(\)\u00b7\.]*" | |
| r"(?:\d|\u00b7|\.)[A-Za-z0-9\-\(\)\u00b7\.]*\b" | |
| ) | |
| def _convert_formula(match): | |
| token = match.group(0) | |
| if len(token) < 2: | |
| return token | |
| # Retrieval-channel and method names are uppercase-plus-digits, which is | |
| # exactly the shape of a formula: without this guard "BGE-M3" renders as | |
| # "BGE-M₃" and "BM25" as "BM₂₅" in the evidence list. | |
| if token.upper() in _NON_FORMULA_TOKENS: | |
| return token | |
| prefix = "" | |
| lowered = token.lower() | |
| for phase_prefix, symbol in (("alpha-", "α-"), ("beta-", "β-")): | |
| if lowered.startswith(phase_prefix): | |
| prefix = symbol | |
| token = token[len(phase_prefix):] | |
| break | |
| return prefix + to_unicode_subscript(token) | |
| return formula_re.sub(_convert_formula, text) | |
| def _normalize_source_text(text: str) -> str: | |
| if not text or not isinstance(text, str): | |
| return text or "" | |
| return text.replace("实验室", "Laboratory") | |
| def _normalize_rag_text(text: str, keep_newlines: bool = False) -> str: | |
| if not text: | |
| return "" | |
| # Normalize whitespace and hidden characters | |
| text = text.replace("\u200b", "").replace("\u200c", "").replace("\u200d", "").replace("\u00a0", " ") | |
| text = re.sub(r"[\x00-\x08\x0b-\x1f\x7f]", " ", text) | |
| # Protect complete DOI links and evidence IDs before punctuation/formula | |
| # cleanup. Otherwise ``doi.org`` is treated as sentence punctuation and | |
| # ``E8`` is treated as a chemical formula by the legacy formatter. | |
| protected_spans = [] | |
| def _save_protected(match): | |
| protected_spans.append(match.group(0)) | |
| return f"__PROTECTED__{len(protected_spans)-1}__" | |
| protected_pattern = re.compile( | |
| r"<https?://doi\.org/10\.\d{4,9}/[^>\s]+>" | |
| r"|https?://doi\.org/10\.\d{4,9}/[^\s<>\]]+" | |
| r"|DOI:\s*10\.\d{4,9}/[^\s<>\]]+" | |
| r"|10\.\d{4,9}/[^\s<>\]]+" | |
| r"|\[\s*E\s*\d+\s*\]" | |
| r"|\bE\d+\b", | |
| flags=re.IGNORECASE, | |
| ) | |
| text = protected_pattern.sub(_save_protected, text) | |
| # Protect math blocks from normalization | |
| math_blocks = [] | |
| def _save_math(m): | |
| math_blocks.append(m.group(0)) | |
| return f"__MATHBLOCK__{len(math_blocks)-1}__" | |
| text = re.sub(r"\$\$.*?\$\$", _save_math, text, flags=re.DOTALL) | |
| text = re.sub(r"\$[^$]+\$", _save_math, text) | |
| if keep_newlines: | |
| text = re.sub(r"[ \t]+", " ", text) | |
| text = re.sub(r"\n{3,}", "\n\n", text) | |
| else: | |
| text = re.sub(r"\s+", " ", text) | |
| # Insert a space only when a numbered token is joined to a lowercase prose | |
| # word. The previous broad rule incorrectly split valid formulas such as | |
| # H2S / H₂S into `H2 S`. | |
| text = re.sub( | |
| r"([A-Za-z][0-9\u2080-\u2089]+)([a-z]{2,})", | |
| r"\1 \2", | |
| text, | |
| ) | |
| # Merge formulas whose element symbols and counts were split by spaces. | |
| def _merge_spaced_formula(match): | |
| chunk = match.group(0) | |
| if not any(ch.isdigit() for ch in chunk): | |
| return chunk | |
| collapsed = re.sub(r"\s+", "", chunk) | |
| # Only merge text that is actually parseable as a chemical formula. | |
| # The old character-only check corrupted prose such as | |
| # "3 of 3 individual" by collapsing the substring "f 3 i". | |
| if ( | |
| re.fullmatch(r"[A-Za-z0-9\-\(\)\u00b7\.]+", collapsed) | |
| and parse_formula(collapsed) | |
| ): | |
| return collapsed | |
| return chunk | |
| text = re.sub(r"(?:[A-Za-z0-9]\s+){2,}[A-Za-z0-9]", _merge_spaced_formula, text) | |
| def _merge_spaced_letters_any(match): | |
| chunk = match.group(0) | |
| if re.search(r"[a-z\u03b1-\u03c9]", chunk): | |
| return re.sub(r"\s+", "", chunk) | |
| return chunk | |
| text = re.sub(r"(?:\b[A-Za-z\u0391-\u03A9\u03b1-\u03c9]\b\s+){2,}\b[A-Za-z\u0391-\u03A9\u03b1-\u03c9]\b", _merge_spaced_letters_any, text) | |
| # Fix broken subscript braces like "S{4}" -> "S4" | |
| text = re.sub(r"([A-Za-z])\{(\d+)\}", r"\1\2", text) | |
| # Fix malformed formula parentheses and braces. | |
| text = re.sub(r"\(\{([^}]+)\)", r"(\1)", text) | |
| text = re.sub(r"\{([^}]*)\)", r"(\1)", text) | |
| # Fix split lowercase words like "i n" -> "in" | |
| def _merge_small_words(s: str) -> str: | |
| small = {"in", "to", "of", "on", "at", "is", "be", "or", "an", "as", "by"} | |
| def _join(m): | |
| cand = (m.group(1) + m.group(2)).lower() | |
| return cand if cand in small else m.group(0) | |
| return re.sub(r"\b([a-z])\s+([a-z])\b", _join, s) | |
| text = _merge_small_words(text) | |
| # Ensure missing spaces after punctuation and between words | |
| text = re.sub(r"([\.!\?:;\)])([A-Za-z])", r"\1 \2", text) | |
| text = re.sub(r"([a-z])([A-Z][a-z])", r"\1 \2", text) | |
| # Restore math blocks before final cleanup | |
| for i, blk in enumerate(math_blocks): | |
| text = text.replace(f"__MATHBLOCK__{i}__", blk) | |
| text = _fix_latex_artifacts(text) | |
| for i, value in enumerate(protected_spans): | |
| text = text.replace(f"__PROTECTED__{i}__", value) | |
| return text | |
| def _build_reference_evidence_blocks(report_item, references: list): | |
| if not isinstance(report_item, dict): | |
| return [] | |
| references = list(references or []) | |
| structured = report_item.get("evidence") | |
| if isinstance(structured, list) and structured: | |
| blocks = [] | |
| for index, reference in enumerate(references): | |
| block = dict(structured[index]) if index < len(structured) and isinstance(structured[index], dict) else {} | |
| block.setdefault("source", reference) | |
| block.setdefault("snippet", "") | |
| block.setdefault("evidence_status", "metadata_only") | |
| blocks.append(block) | |
| return blocks | |
| raw_snippets = report_item.get("reference_snippets") or [] | |
| blocks = [] | |
| for index, reference in enumerate(references): | |
| snippet = "" | |
| if isinstance(raw_snippets, list) and index < len(raw_snippets): | |
| item = raw_snippets[index] | |
| snippet = item.get("text", "") if isinstance(item, dict) else str(item or "") | |
| elif isinstance(raw_snippets, dict): | |
| snippet = raw_snippets.get(reference, "") | |
| blocks.append({ | |
| "source": reference, | |
| "snippet": snippet, | |
| "evidence_status": "legacy_unclassified" if snippet else "metadata_only", | |
| }) | |
| return blocks | |
| def _public_evidence_key(value: object) -> str: | |
| """Normalize a path field for defensive public-evidence validation.""" | |
| text = normalize_legacy_text(str(value or "")).casefold() | |
| subscript_digits = str.maketrans("₀₁₂₃₄₅₆₇₈₉", "0123456789") | |
| text = text.translate(subscript_digits) | |
| return re.sub(r"[^a-z0-9α-ω]+", "", text) | |
| def _public_condition_key(value: object) -> tuple: | |
| """Compare compound condition labels without depending on token order.""" | |
| text = normalize_legacy_text(str(value or "")).casefold() | |
| parts = [ | |
| re.sub(r"[^a-z0-9α-ω]+", "", part) | |
| for part in re.split(r"\s*(?:\+|&|,|;)\s*", text) | |
| ] | |
| return tuple(sorted(part for part in parts if part)) | |
| def _is_direct_public_edge_match(match: object, path_str: str) -> bool: | |
| """Return True only for a direct, condition-compatible match to this path.""" | |
| if not isinstance(match, dict): | |
| return False | |
| if str(match.get("verdict") or "").strip().casefold() != "direct": | |
| return False | |
| condition_status = str(match.get("condition_status") or "").strip().casefold() | |
| if condition_status and condition_status not in {"exact", "not_applicable"}: | |
| return False | |
| species, conditions = path_components( | |
| path_str, | |
| include_empty_conditions=True, | |
| ) if path_str else ([], []) | |
| edge_count = max(len(species) - 1, 0) | |
| try: | |
| edge_index = int(match.get("edge_index")) | |
| except (TypeError, ValueError): | |
| return False | |
| if edge_index < 1 or edge_index > edge_count: | |
| return False | |
| expected_reactant = species[edge_index - 1] | |
| expected_product = species[edge_index] | |
| expected_condition = ( | |
| conditions[edge_index - 1] | |
| if edge_index - 1 < len(conditions) | |
| else "" | |
| ) | |
| reactant = match.get("reactant") | |
| if reactant and _public_evidence_key(reactant) != _public_evidence_key(expected_reactant): | |
| return False | |
| product = match.get("product") | |
| if product and _public_evidence_key(product) != _public_evidence_key(expected_product): | |
| return False | |
| matched_condition = match.get("condition") | |
| if ( | |
| matched_condition | |
| and _public_condition_key(matched_condition) | |
| != _public_condition_key(expected_condition) | |
| ): | |
| return False | |
| return True | |
| def _is_direct_public_endpoint_match(match: object, path_str: str) -> bool: | |
| """Validate a scoped start-to-final match without treating it as an edge.""" | |
| if not isinstance(match, dict): | |
| return False | |
| if str(match.get("evidence_scope") or "").strip() != "pathway_endpoint": | |
| return False | |
| if str(match.get("verdict") or "").strip().casefold() != "direct": | |
| return False | |
| condition_status = str(match.get("condition_status") or "").strip().casefold() | |
| if condition_status not in {"exact", "not_applicable"}: | |
| return False | |
| species, conditions = path_components( | |
| path_str, | |
| include_empty_conditions=True, | |
| ) if path_str else ([], []) | |
| if len(species) < 3: | |
| return False | |
| try: | |
| edge_index = int(match.get("edge_index")) | |
| except (TypeError, ValueError): | |
| return False | |
| if edge_index != 0: | |
| return False | |
| reactant = str(match.get("reactant") or "").strip() | |
| product = str(match.get("product") or "").strip() | |
| if ( | |
| not reactant | |
| or not product | |
| or _public_evidence_key(reactant) != _public_evidence_key(species[0]) | |
| or _public_evidence_key(product) != _public_evidence_key(species[-1]) | |
| ): | |
| return False | |
| path_condition_keys = { | |
| _public_condition_key(condition) | |
| for condition in conditions | |
| if str(condition or "").strip() | |
| } | |
| matched_condition = str(match.get("condition") or "").strip() | |
| if len(path_condition_keys) == 1: | |
| if ( | |
| not matched_condition | |
| or _public_condition_key(matched_condition) not in path_condition_keys | |
| or condition_status != "exact" | |
| ): | |
| return False | |
| elif matched_condition and _public_condition_key(matched_condition) not in path_condition_keys: | |
| return False | |
| return True | |
| def _is_direct_public_match(match: object, path_str: str) -> bool: | |
| return ( | |
| _is_direct_public_endpoint_match(match, path_str) | |
| or _is_direct_public_edge_match(match, path_str) | |
| ) | |
| #: Display names for retrieval-channel identifiers in the user-facing evidence | |
| #: list. Keys are the ``retrieval_origin`` values with underscores already turned | |
| #: into spaces; anything not listed is shown as-is. | |
| _RETRIEVAL_CHANNEL_LABELS = { | |
| "bge m3": "BGE-M3", | |
| "dense vector": "BGE-M3", | |
| "crn provenance": "CRN provenance", | |
| "resolver": "resolver", | |
| "bm25": "BM25", | |
| "lexical": "lexical", | |
| } | |
| def _public_match_label(match: object) -> str: | |
| if not isinstance(match, dict): | |
| return "unclassified" | |
| verdict = str(match.get("verdict") or "unclassified").replace("_", " ") | |
| if match.get("evidence_scope") == "pathway_endpoint": | |
| return f"pathway endpoints: {verdict}" | |
| return f"edge {match.get('edge_index')}: {verdict}" | |
| def _filter_snippet_lines_for_path(block: object, path_str: str) -> list: | |
| """ | |
| Return public-safe snippet records for the selected path. | |
| Non-direct classifications remain in ``report_item['evidence']`` for | |
| diagnostics, but they are deliberately excluded from public prose, | |
| numbered references, PDF evidence, and source counts. | |
| """ | |
| if not isinstance(block, dict): | |
| return [] | |
| raw_items = block.get("snippets") | |
| if isinstance(raw_items, list) and raw_items: | |
| candidates = [item for item in raw_items if isinstance(item, dict)] | |
| allow_block_matches = False | |
| else: | |
| snippet = str(block.get("snippet") or "").strip() | |
| candidates = [{"text": snippet}] if snippet else [] | |
| allow_block_matches = True | |
| filtered = [] | |
| seen_text = set() | |
| for item in candidates: | |
| text = str(item.get("text") or "").strip() | |
| if not text: | |
| continue | |
| matches = item.get("edge_matches") | |
| if not isinstance(matches, list) and allow_block_matches: | |
| matches = block.get("edge_matches") | |
| direct_matches = [ | |
| dict(match) | |
| for match in (matches or []) | |
| if _is_direct_public_match(match, path_str) | |
| ] | |
| if not direct_matches: | |
| continue | |
| text_key = re.sub(r"\s+", " ", text).strip().casefold() | |
| if text_key in seen_text: | |
| continue | |
| seen_text.add(text_key) | |
| public_item = dict(item) | |
| public_item["text"] = text | |
| public_item["edge_matches"] = direct_matches | |
| filtered.append(public_item) | |
| return filtered | |
| _PHASE_TABLE_SNIPPET_RE = re.compile( | |
| r"phases?\s+identified\s+before\s+(?P<treat>[^:]{0,60}?)\s*:\s*(?P<before>[^.]*)\." | |
| r"\s*phases?\s+identified\s+after\s+[^:]{0,60}?\s*:\s*(?P<after>[^.]*)\.", | |
| re.IGNORECASE, | |
| ) | |
| def _summarize_phase_tables(snippet: object) -> str: | |
| """Render XRD phase-table evidence as a finding rather than as raw rows. | |
| A reference can match several rows of one table -- one per sample -- and | |
| concatenating them verbatim produces a wall of repeated mineral names in | |
| which nothing stands out. What the reader needs is the experiment's result: | |
| how many samples were compared, under what treatment, and which phases | |
| appeared only after it. Returns "" for ordinary prose, so the caller falls | |
| back to quoting the passage. | |
| """ | |
| text = str(snippet or "") | |
| rows = list(_PHASE_TABLE_SNIPPET_RE.finditer(text)) | |
| if not rows: | |
| return "" | |
| def phases(raw: str) -> list: | |
| return [p.strip() for p in raw.split(",") if p.strip()] | |
| treatment = (rows[0].group("treat") or "treatment").strip() | |
| gained_everywhere, gained_any = None, [] | |
| for row in rows: | |
| before = {p.casefold() for p in phases(row.group("before"))} | |
| gained = [p for p in phases(row.group("after")) if p.casefold() not in before] | |
| for phase in gained: | |
| if phase not in gained_any: | |
| gained_any.append(phase) | |
| keys = {p.casefold() for p in gained} | |
| gained_everywhere = keys if gained_everywhere is None else (gained_everywhere & keys) | |
| consistent = [p for p in gained_any if p.casefold() in (gained_everywhere or set())] | |
| sample_word = "sample" if len(rows) == 1 else "samples" | |
| parts = [ | |
| f"XRD phase comparison before and after {treatment}, " | |
| f"{len(rows)} {sample_word}." | |
| ] | |
| if consistent: | |
| parts.append( | |
| f"Present after {treatment} but absent before in every {sample_word[:6]}: " | |
| + ", ".join(consistent) + "." | |
| ) | |
| others = [p for p in gained_any if p not in consistent] | |
| if others: | |
| parts.append("Also gained in some samples: " + ", ".join(others) + ".") | |
| return " ".join(parts) | |
| def _build_public_evidence_blocks(report_item, references: list, path_str: str) -> list: | |
| """Build a reference-aligned view containing direct public evidence only.""" | |
| public_blocks = [] | |
| for block in _build_reference_evidence_blocks(report_item, references): | |
| public_items = _filter_snippet_lines_for_path(block, path_str) | |
| if not public_items: | |
| public_blocks.append(None) | |
| continue | |
| public_block = dict(block) | |
| public_block["snippets"] = public_items | |
| public_block["snippet"] = "\n".join(item["text"] for item in public_items) | |
| public_block["edge_matches"] = [ | |
| match | |
| for item in public_items | |
| for match in item.get("edge_matches", []) | |
| ] | |
| public_block["evidence_status"] = "direct" | |
| public_block["has_direct_evidence"] = True | |
| public_blocks.append(public_block) | |
| return public_blocks | |
| def _unverified_crn_reference_pairs(report_item, references: list, blocks: list) -> list: | |
| """Return CRN provenance metadata that is explicitly marked non-evidential.""" | |
| path_sources = ( | |
| report_item.get("path_evidence_sources") or [] | |
| if isinstance(report_item, dict) | |
| else [] | |
| ) | |
| pairs = [] | |
| for index, reference in enumerate(references or []): | |
| block = blocks[index] if index < len(blocks) and isinstance(blocks[index], dict) else {} | |
| explicitly_provenance = bool(block.get("provenance_only")) or any( | |
| reference_contains_source(reference, source) | |
| for source in path_sources | |
| ) | |
| if explicitly_provenance: | |
| pairs.append((index + 1, reference, block.get("best_verdict"))) | |
| return pairs | |
| def _metric_text(value) -> str: | |
| if value is None: | |
| return "n/a" | |
| if isinstance(value, float): | |
| return f"{value:.3f}".rstrip("0").rstrip(".") | |
| return str(value) | |
| def _metric_percent(value) -> str: | |
| if value is None: | |
| return "n/a" | |
| try: | |
| number = float(value) | |
| except (TypeError, ValueError): | |
| return str(value) | |
| if 0 <= number <= 1: | |
| return f"{number * 100:.0f}%" | |
| return _metric_text(number) | |
| # Why a cited source did not become claim evidence. Each verdict names the one | |
| # requirement that failed, so the note is a statement about the match rather | |
| # than a judgement on the source -- and never claims nothing matched when | |
| # something did. | |
| _UNVERIFIED_CRN_REASON = { | |
| "qualified": "passages match this step, but the source qualifies the claim", | |
| "inferred": "passages match this step as an author proposal, not an observation", | |
| "related": "passages match the species and the conversion, but not the " | |
| "condition recorded in the network", | |
| "condition_mismatch": "passages match the species, but describe a different " | |
| "condition than the network records", | |
| "contradicted": "the passage found describes this step as not occurring", | |
| "phase_conflict": "passages match the formula but a different phase", | |
| } | |
| def _format_unverified_crn_reference(reference: str, best_verdict=None) -> str: | |
| text = _normalize_report_formula_display(str(reference or "")) | |
| text = " ".join(text.split()).strip() | |
| if not text: | |
| return "" | |
| if not first_doi(text): | |
| return f"- {text} — cited in the reaction network; no DOI recorded" | |
| reason = _UNVERIFIED_CRN_REASON.get( | |
| str(best_verdict or "").strip().casefold(), | |
| "no passage matched to this pathway", | |
| ) | |
| return f"- {text} — cited in the reaction network; {reason}" | |
| def _pdf_clean_text(value) -> str: | |
| text = html.unescape(str(value or "")) | |
| text = re.sub(r"<[^>]+>", " ", text) | |
| text = re.sub(r"~~~.*?~~~", " ", text, flags=re.DOTALL) | |
| cleaned_lines = [] | |
| for line in text.splitlines(): | |
| cells = [cell.strip() for cell in line.strip().strip("|").split("|")] | |
| if len(cells) > 1 and all( | |
| re.fullmatch(r":?-{3,}:?", cell or "") for cell in cells | |
| ): | |
| continue | |
| if "|" in line: | |
| line = " — ".join(cells) | |
| cleaned_lines.append(line) | |
| text = "\n".join(cleaned_lines) | |
| # Condition-bearing arrows carry a label we render as "->". | |
| text = re.sub(r"\\xrightarrow(?:\[[^\]]*\])?\{([^{}]*)\}", r" -> ", text) | |
| # Shared rendering step. Must run BEFORE the markdown emphasis strip below, | |
| # or the "_" removal destroys TeX subscripts (\({}_{4}\) -> "{} {4}"). | |
| text = render_chem_text(text) | |
| text = re.sub(r"[#*_]+", " ", text).replace(chr(96), " ") | |
| text = _normalize_report_formula_display(normalize_legacy_text(text)) | |
| text = re.sub( | |
| r"\bE([₀₁₂₃₄₅₆₇₈₉]+)", | |
| lambda match: "E" + match.group(1).translate( | |
| str.maketrans("₀₁₂₃₄₅₆₇₈₉", "0123456789") | |
| ), | |
| text, | |
| ) | |
| return re.sub(r"\s+", " ", text).strip() | |
| def _save_rag_detail_pdf(report_item): | |
| if not isinstance(report_item, dict): | |
| return None | |
| try: | |
| from xml.sax.saxutils import escape | |
| from reportlab.lib import colors | |
| from reportlab.lib.enums import TA_CENTER, TA_LEFT | |
| from reportlab.lib.pagesizes import A4 | |
| from reportlab.lib.styles import ParagraphStyle, getSampleStyleSheet | |
| from reportlab.lib.units import mm | |
| from reportlab.pdfbase import pdfmetrics | |
| from reportlab.pdfbase.ttfonts import TTFont | |
| from reportlab.platypus import ( | |
| HRFlowable, | |
| KeepTogether, | |
| LongTable, | |
| Paragraph, | |
| SimpleDocTemplate, | |
| Spacer, | |
| Table, | |
| TableStyle, | |
| ) | |
| except Exception: | |
| return None | |
| font_name = "Helvetica" | |
| for candidate_name, candidate_path in [ | |
| ("DejaVuSans", "/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf"), | |
| ("DejaVuSans", r"C:\Windows\Fonts\DejaVuSans.ttf"), | |
| ("Arial", r"C:\Windows\Fonts\arial.ttf"), | |
| ("SimHei", r"C:\Windows\Fonts\simhei.ttf"), | |
| ]: | |
| try: | |
| if os.path.exists(candidate_path): | |
| pdfmetrics.registerFont(TTFont(candidate_name, candidate_path)) | |
| font_name = candidate_name | |
| break | |
| except Exception: | |
| continue | |
| temporary = tempfile.NamedTemporaryFile( | |
| delete=False, | |
| suffix=".pdf", | |
| prefix="rag_detail_", | |
| ) | |
| temporary.close() | |
| document = SimpleDocTemplate( | |
| temporary.name, | |
| pagesize=A4, | |
| rightMargin=18 * mm, | |
| leftMargin=18 * mm, | |
| topMargin=17 * mm, | |
| bottomMargin=18 * mm, | |
| title="Detailed RAG Evaluation Report", | |
| ) | |
| styles = getSampleStyleSheet() | |
| body_style = ParagraphStyle( | |
| "RagBody", | |
| parent=styles["BodyText"], | |
| fontName=font_name, | |
| fontSize=9, | |
| leading=12.5, | |
| textColor=colors.HexColor("#202938"), | |
| alignment=TA_LEFT, | |
| spaceAfter=5, | |
| ) | |
| small_style = ParagraphStyle( | |
| "RagSmall", | |
| parent=body_style, | |
| fontSize=7.5, | |
| leading=9.5, | |
| ) | |
| title_style = ParagraphStyle( | |
| "RagTitle", | |
| parent=body_style, | |
| fontSize=16, | |
| leading=20, | |
| textColor=colors.HexColor("#1f3b73"), | |
| alignment=TA_CENTER, | |
| spaceAfter=10, | |
| ) | |
| heading_style = ParagraphStyle( | |
| "RagHeading", | |
| parent=body_style, | |
| fontSize=11.5, | |
| leading=14, | |
| textColor=colors.HexColor("#234f9a"), | |
| spaceBefore=8, | |
| spaceAfter=5, | |
| ) | |
| # A reliability reason explains the rating, so it is set at body weight with | |
| # a hanging indent -- the bullet stays in the margin and wrapped lines line | |
| # up under the text instead of under the dash. | |
| reason_style = ParagraphStyle( | |
| "RagReason", | |
| parent=body_style, | |
| leftIndent=10, | |
| firstLineIndent=-10, | |
| bulletIndent=0, | |
| spaceAfter=4, | |
| leading=12.5, | |
| ) | |
| # Methodological caveats are genuinely secondary, so they keep the small | |
| # size -- but with enough leading to stay readable. | |
| note_style = ParagraphStyle( | |
| "RagNote", | |
| parent=small_style, | |
| leading=10.5, | |
| textColor=colors.HexColor("#55617a"), | |
| spaceBefore=2, | |
| ) | |
| def paragraph(value, style=body_style): | |
| cleaned = _pdf_clean_text(value) | |
| return Paragraph(escape(cleaned) if cleaned else "-", style) | |
| def table_style(header=True): | |
| commands = [ | |
| ("FONTNAME", (0, 0), (-1, -1), font_name), | |
| ("FONTSIZE", (0, 0), (-1, -1), 8), | |
| ("VALIGN", (0, 0), (-1, -1), "TOP"), | |
| ("GRID", (0, 0), (-1, -1), 0.35, colors.HexColor("#cbd5e1")), | |
| ("LEFTPADDING", (0, 0), (-1, -1), 5), | |
| ("RIGHTPADDING", (0, 0), (-1, -1), 5), | |
| ("TOPPADDING", (0, 0), (-1, -1), 3), | |
| ("BOTTOMPADDING", (0, 0), (-1, -1), 3), | |
| ] | |
| if header: | |
| commands.extend([ | |
| ("BACKGROUND", (0, 0), (-1, 0), colors.HexColor("#e8eef9")), | |
| ("TEXTCOLOR", (0, 0), (-1, 0), colors.HexColor("#1f3b73")), | |
| ]) | |
| return TableStyle(commands) | |
| path_str = report_item.get("path_str") or "" | |
| story = [Paragraph(_path_report_title(path_str), title_style)] | |
| # The detailed PDF is an evaluation artifact: sections are collected | |
| # separately here and assembled evaluation-first at the end (metrics and | |
| # validation lead; the generated narrative and context are demoted). | |
| sec_pathway, sec_metrics, sec_validation = [], [], [] | |
| sec_reliability, sec_source_quality = [], [] | |
| sec_evidence, sec_narrative = [], [] | |
| sec_related, sec_references, sec_provenance = [], [], [] | |
| species, conditions = path_components(path_str, include_empty_conditions=True) | |
| references = list(report_item.get("references") or []) | |
| all_evidence_blocks = _build_reference_evidence_blocks(report_item, references) | |
| evidence_blocks = _build_public_evidence_blocks( | |
| report_item, | |
| references, | |
| path_str, | |
| ) | |
| metrics = report_item.get("metrics") or {} | |
| if len(species) >= 2: | |
| sec_pathway.append(Paragraph("Chemical Pathway", heading_style)) | |
| path_rows = [[ | |
| paragraph("Reactant", small_style), | |
| paragraph("Condition", small_style), | |
| paragraph("Product", small_style), | |
| ]] | |
| for index in range(len(species) - 1): | |
| condition = conditions[index] if index < len(conditions) else "unspecified" | |
| path_rows.append([ | |
| paragraph(to_unicode_subscript(species[index]), small_style), | |
| paragraph(_display_condition_label(condition), small_style), | |
| paragraph(to_unicode_subscript(species[index + 1]), small_style), | |
| ]) | |
| path_table = Table( | |
| path_rows, | |
| colWidths=[62 * mm, 48 * mm, 62 * mm], | |
| repeatRows=1, | |
| ) | |
| path_table.setStyle(table_style()) | |
| sec_pathway.extend([path_table, Spacer(1, 3)]) | |
| # The table's Condition column is the graph's label. Printing the | |
| # conditions the papers actually report next to it is what makes a | |
| # graph-alignment shortfall auditable: a reader sees the mismatch here | |
| # rather than having to reconstruct it from the evidence snippets. | |
| graph_conditions = ", ".join(dict.fromkeys( | |
| _display_condition_label(c) for c in conditions if c | |
| )) | |
| if graph_conditions: | |
| sec_pathway.append(paragraph( | |
| f"Graph-encoded conditions: {graph_conditions}", | |
| small_style, | |
| )) | |
| literature_conditions = _literature_condition_note(evidence_blocks) | |
| if literature_conditions: | |
| sec_pathway.append(paragraph( | |
| f"Literature-supported conditions: {literature_conditions}.", | |
| small_style, | |
| )) | |
| sec_pathway.append(Spacer(1, 4)) | |
| raw_report = report_item.get("report") or report_item.get("content") or "" | |
| report_text = _pdf_clean_text(raw_report) | |
| report_text = re.sub( | |
| r"^Provenance Analysis Report:.*?in Mural Pigments\.?\s*", | |
| "", | |
| report_text, | |
| count=1, | |
| flags=re.I, | |
| ) | |
| evaluation_rows = ( | |
| report_item.get("sentence_validation_table") | |
| or (report_item.get("rag_evaluation") or {}).get("sentence_level_results") | |
| or [] | |
| ) | |
| if evaluation_rows: | |
| accepted_report_statements = [] | |
| for row in evaluation_rows: | |
| if not isinstance(row, dict): | |
| continue | |
| grounded = bool( | |
| row.get("grounded_evidence_status") | |
| or row.get("support_status") | |
| or row.get("direct_evidence_status") | |
| or row.get("restated_evidence_status") | |
| ) | |
| if grounded and row.get("citation_status") is True: | |
| statement = str( | |
| row.get("statement") or row.get("text") or "" | |
| ).strip() | |
| if statement: | |
| accepted_report_statements.append(statement) | |
| removed_statements = max( | |
| 0, | |
| len([row for row in evaluation_rows if isinstance(row, dict)]) | |
| - len(accepted_report_statements), | |
| ) | |
| if accepted_report_statements: | |
| report_text = _pdf_clean_text(" ".join(accepted_report_statements)) | |
| elif any(evidence_blocks): | |
| report_text = ( | |
| "No generated statement passed both snippet-support and " | |
| "citation verification." | |
| ) | |
| if removed_statements: | |
| report_text = ( | |
| f"{report_text} [{removed_statements} unsupported or " | |
| "unverifiable statement(s) omitted from this narrative.]" | |
| ).strip() | |
| # Keep a grounded contextual synthesis (flagged by the generator); only | |
| # substitute the terse note when no usable narrative was produced. | |
| _pdf_has_contextual = bool( | |
| (report_item.get("metrics") or {}).get("has_contextual_narrative") | |
| ) | |
| if not any(evidence_blocks) and not _pdf_has_contextual: | |
| relation = " -> ".join(species) if species else str(path_str or "the recorded path") | |
| condition_text = ", ".join(dict.fromkeys( | |
| _display_condition_label(condition) for condition in conditions if condition | |
| )) or "unspecified" | |
| retrieval_available = ( | |
| (report_item.get("metrics") or {}).get("retrieval_available") is not False | |
| ) | |
| report_text = compose_graph_only_note( | |
| relation, | |
| condition_text, | |
| retrieved_context=bool(report_item.get("related_context")), | |
| retrieval_available=retrieval_available, | |
| ) | |
| # When sentence-level rows are available they already reproduce the full | |
| # evidence-filtered narrative claim by claim. Do not print the same prose | |
| # again at the end of the PDF; that duplication previously created an | |
| # almost-empty second page. | |
| if report_text and not evaluation_rows: | |
| edge_coverage = (report_item.get("metrics") or {}).get( | |
| "edge_evidence_coverage", | |
| 0, | |
| ) | |
| endpoint_supported = bool( | |
| (report_item.get("metrics") or {}).get( | |
| "endpoint_conversion_supported", | |
| False, | |
| ) | |
| ) | |
| interpretation_heading = ( | |
| "Evidence-filtered Generated Narrative" | |
| if edge_coverage or endpoint_supported | |
| else "Generated Narrative / Graph Path Note (no direct literature evidence)" | |
| ) | |
| sec_narrative.extend([ | |
| Paragraph(interpretation_heading, heading_style), | |
| paragraph(report_text, small_style), | |
| ]) | |
| if metrics: | |
| sec_metrics.append(Paragraph("RAG Reliability Assessment", heading_style)) | |
| metric_rows = [[paragraph("Metric", small_style), paragraph("Value", small_style)]] | |
| reliability_level = str( | |
| metrics.get("overall_reliability_level") or "not_evaluable" | |
| ).replace("_", " ") | |
| metric_values = [ | |
| ( | |
| # Level only. The percentage beside it read as a weighted | |
| # composite of the metrics listed below, but no such weighting is | |
| # defined -- the individual coverage, verification, traceability | |
| # and alignment figures are the measurement, and the level is a | |
| # band over them. Printing a number invites the question of how | |
| # it was weighted, which nothing here answers. | |
| "Overall reliability", | |
| reliability_level, | |
| ), | |
| ( | |
| "Literature retrieval / Endpoint conversion evidence", | |
| ( | |
| "available" | |
| if metrics.get("retrieval_available") is not False | |
| else "unavailable" | |
| ) | |
| + "; endpoint " | |
| + ( | |
| "supported" | |
| if metrics.get("endpoint_conversion_supported") | |
| else "not retrieved" | |
| ), | |
| ), | |
| ( | |
| "Edge evidence coverage", | |
| _metric_percent(metrics.get("edge_evidence_coverage", 0)), | |
| ), | |
| ( | |
| "Statement evidence coverage", | |
| _metric_percent(metrics.get("evidence_coverage", 0)), | |
| ), | |
| ( | |
| "Evidence types", | |
| f"direct: {_metric_text(metrics.get('direct_evidence_statements', 0))}; " | |
| f"restated: {_metric_text(metrics.get('restated_evidence_statements', 0))}", | |
| ), | |
| ( | |
| "Citation verification", | |
| ( | |
| str(metrics.get("citation_verification_status") or "not available") | |
| .replace("_", " ") | |
| + " (" | |
| + _metric_percent(metrics.get("citation_accuracy")) | |
| + ")" | |
| ), | |
| ), | |
| ( | |
| "Graph alignment", | |
| ( | |
| _metric_percent(metrics.get("graph_alignment", 0)) | |
| if ( | |
| "graph_alignment_applicable_statements" not in metrics | |
| or metrics.get("graph_alignment_applicable_statements", 0) | |
| ) | |
| else "n/a (no graph claim in supporting-detail sentences)" | |
| ), | |
| ), | |
| ( | |
| "Source traceability", | |
| ( | |
| f"{str(metrics.get('source_quality_level') or 'unavailable').replace('_', ' ')} " | |
| f"({_metric_percent(metrics.get('source_quality_score'))})" | |
| ), | |
| ), | |
| ( | |
| "Unsupported statements", | |
| f"{_metric_text(metrics.get('unsupported_statements', 0))} / " | |
| f"{_metric_text(metrics.get('total_statements', metrics.get('total_sentences', 0)))}", | |
| ), | |
| ] | |
| if metrics.get("retrieval_available") is False: | |
| metric_values[1] = ( | |
| "Literature retrieval / Endpoint conversion evidence", | |
| "unavailable; endpoint status undetermined", | |
| ) | |
| for label, value in metric_values: | |
| metric_rows.append([paragraph(label, small_style), paragraph(value, small_style)]) | |
| metric_table = Table(metric_rows, colWidths=[80 * mm, 92 * mm], repeatRows=1) | |
| metric_table.setStyle(table_style()) | |
| sec_metrics.extend([metric_table, Spacer(1, 4)]) | |
| reliability_summary = str( | |
| metrics.get("overall_reliability_summary") or "" | |
| ).strip() | |
| reliability_reasons = [ | |
| str(reason).strip() | |
| for reason in (metrics.get("overall_reliability_reasons") or []) | |
| if str(reason).strip() | |
| ] | |
| if reliability_summary or reliability_reasons: | |
| sec_reliability.append(Paragraph("Reliability Conclusion", heading_style)) | |
| if reliability_summary: | |
| sec_reliability.append(paragraph(reliability_summary)) | |
| # The reasons are the finding, not a footnote. Setting them in the | |
| # 7.5pt small style put them below the summary in visual weight and | |
| # stacked three different sizes in one block, which read as clutter | |
| # rather than as the explanation of the rating. | |
| for reason in reliability_reasons: | |
| sec_reliability.append(paragraph(f"— {reason}", reason_style)) | |
| sec_reliability.append(Spacer(1, 3)) | |
| sec_reliability.append(paragraph( | |
| "Citation accuracy measures claim-to-snippet support, not only " | |
| "whether a reference number is present. Source traceability " | |
| "measures bibliographic completeness and inspectability; it " | |
| "does not measure journal prestige or independent replication.", | |
| note_style, | |
| )) | |
| source_quality_details = metrics.get("source_quality_details") or [] | |
| if source_quality_details: | |
| for item in source_quality_details: | |
| if not isinstance(item, dict): | |
| continue | |
| checks = ", ".join( | |
| label | |
| for key, label in ( | |
| ("has_doi", "DOI"), | |
| ("has_year", "year"), | |
| ("has_structured_citation", "publication metadata"), | |
| ("has_snippet", "snippet"), | |
| ("has_direct_classification", "evidence class"), | |
| ) | |
| if item.get(key) | |
| ) or "limited metadata" | |
| sec_source_quality.append(paragraph( | |
| "Source traceability details: " | |
| f"[{item.get('reference_index', '-')}] " | |
| f"{str(item.get('level') or 'unavailable').replace('_', ' ')} " | |
| f"({_metric_percent(item.get('score'))}); checks: {checks}.", | |
| small_style, | |
| )) | |
| evidence_entries = [] | |
| evidence_ids_by_reference = {} | |
| original_to_public_reference = {} | |
| public_references = [] | |
| for original_index, block in enumerate(evidence_blocks, 1): | |
| if not isinstance(block, dict): | |
| continue | |
| public_index = len(public_references) + 1 | |
| original_to_public_reference[original_index] = public_index | |
| public_references.append(references[original_index - 1]) | |
| items = block.get("snippets") or [] | |
| if not items and str(block.get("snippet") or "").strip(): | |
| items = [{ | |
| "text": block.get("snippet"), | |
| "edge_matches": block.get("edge_matches") or [], | |
| }] | |
| for item in items: | |
| text = str((item or {}).get("text") or "").strip() | |
| if not text: | |
| continue | |
| evidence_id = f"E{len(evidence_entries) + 1}" | |
| matches = (item or {}).get("edge_matches") or [] | |
| classifications = list(dict.fromkeys( | |
| _public_match_label(match) | |
| for match in matches | |
| if isinstance(match, dict) | |
| )) | |
| evidence_entries.append({ | |
| "id": evidence_id, | |
| "reference_index": public_index, | |
| "classification": "; ".join(classifications) | |
| or str(block.get("evidence_status") or "unclassified").replace("_", " "), | |
| "text": text, | |
| }) | |
| evidence_ids_by_reference.setdefault(public_index, []).append(evidence_id) | |
| if evidence_entries: | |
| sec_evidence.append(Paragraph("Evidence Assessment", heading_style)) | |
| for entry in evidence_entries: | |
| sec_evidence.append(paragraph( | |
| f"{entry['id']} | Reference [{entry['reference_index']}] | " | |
| f"{entry['classification']}", | |
| small_style, | |
| )) | |
| sec_evidence.append(paragraph(entry["text"])) | |
| sec_evidence.append(HRFlowable( | |
| width="100%", | |
| thickness=0.35, | |
| color=colors.HexColor("#cbd5e1"), | |
| spaceBefore=2, | |
| spaceAfter=5, | |
| )) | |
| related_context = report_item.get("related_context") or [] | |
| related_entries = [] | |
| for item in related_context[:3]: | |
| if not isinstance(item, dict): | |
| continue | |
| text = str(item.get("snippet") or "").strip() | |
| if not text: | |
| continue | |
| match = item.get("match") or {} | |
| related_entries.append({ | |
| "source": str(item.get("source") or "Unknown source").strip(), | |
| "classification": ( | |
| f"{str(match.get('verdict') or 'not direct').replace('_', ' ')}; " | |
| f"condition: {str(match.get('condition_status') or 'unstated').replace('_', ' ')}" | |
| ), | |
| "text": text, | |
| }) | |
| if related_entries: | |
| sec_related.append(Paragraph( | |
| "Retrieved Related Context (not pathway claim evidence)", | |
| heading_style, | |
| )) | |
| for entry in related_entries: | |
| sec_related.append(paragraph( | |
| f"{entry['classification']} | Source: {entry['source']}", | |
| small_style, | |
| )) | |
| sec_related.append(paragraph(entry["text"])) | |
| sec_related.append(paragraph( | |
| "This context was retrieved for review but was not used as " | |
| "evidence for a graph edge or pathway claim.", | |
| small_style, | |
| )) | |
| sec_related.append(HRFlowable( | |
| width="100%", | |
| thickness=0.35, | |
| color=colors.HexColor("#cbd5e1"), | |
| spaceBefore=2, | |
| spaceAfter=5, | |
| )) | |
| if public_references: | |
| sec_references.append(Paragraph("References", heading_style)) | |
| for index, reference in enumerate(public_references, 1): | |
| status = ", ".join(evidence_ids_by_reference.get(index, [])) | |
| sec_references.append(paragraph(f"[{index}] {reference} ({status})", small_style)) | |
| provenance_pairs = [ | |
| (index, reference, best_verdict) | |
| for index, reference, best_verdict in _unverified_crn_reference_pairs( | |
| report_item, | |
| references, | |
| all_evidence_blocks, | |
| ) | |
| if index not in original_to_public_reference | |
| ] | |
| if provenance_pairs: | |
| sec_provenance.append( | |
| Paragraph("Reaction Network Provenance", heading_style) | |
| ) | |
| for _index, reference, best_verdict in provenance_pairs: | |
| line = _format_unverified_crn_reference(reference, best_verdict) | |
| if line: | |
| sec_provenance.append(paragraph(line, small_style)) | |
| validation_rows = ( | |
| ( | |
| report_item.get("sentence_validation_table") | |
| or (report_item.get("rag_evaluation") or {}).get("sentence_level_results") | |
| or [] | |
| ) | |
| if any(evidence_blocks) | |
| else [] | |
| ) | |
| if validation_rows: | |
| sec_validation.append(Paragraph( | |
| "Generated Narrative - Sentence-level Validation", | |
| heading_style, | |
| )) | |
| rows = [[ | |
| paragraph("No.", small_style), | |
| paragraph("Statement", small_style), | |
| paragraph("Evidence", small_style), | |
| paragraph("Citation", small_style), | |
| paragraph("Assessment", small_style), | |
| ]] | |
| for index, row in enumerate(validation_rows, 1): | |
| if not isinstance(row, dict): | |
| continue | |
| cited_original = [ | |
| int(value) for value in (row.get("cited_references") or []) | |
| if str(value).isdigit() | |
| ] | |
| cited = [ | |
| original_to_public_reference[value] | |
| for value in cited_original | |
| if value in original_to_public_reference | |
| ] | |
| supporting = row.get("supporting_evidence") | |
| supporting_text = ( | |
| str( | |
| supporting.get("snippet") | |
| or supporting.get("text") | |
| or supporting.get("content") | |
| or "" | |
| ).strip() | |
| if isinstance(supporting, dict) | |
| else "" | |
| ) | |
| evidence_ids = [ | |
| entry["id"] | |
| for entry in evidence_entries | |
| if ( | |
| entry["reference_index"] in cited | |
| and supporting_text | |
| and _pdf_clean_text(entry["text"]) | |
| == _pdf_clean_text(supporting_text) | |
| ) | |
| ] | |
| if not evidence_ids: | |
| evidence_ids = [ | |
| evidence_id | |
| for reference_index in cited | |
| for evidence_id in evidence_ids_by_reference.get( | |
| reference_index, [] | |
| ) | |
| ] | |
| assessment = str( | |
| row.get("support_basis") | |
| or row.get("statement_type") | |
| or ("direct" if row.get("direct_evidence_status") else "unsupported") | |
| ).replace("_", " ") | |
| rows.append([ | |
| paragraph(index, small_style), | |
| paragraph(row.get("statement") or row.get("text") or "", small_style), | |
| paragraph(", ".join(evidence_ids) or "-", small_style), | |
| paragraph(", ".join(f"[{value}]" for value in cited) or "-", small_style), | |
| paragraph(assessment, small_style), | |
| ]) | |
| validation_table = LongTable( | |
| rows, | |
| colWidths=[8 * mm, 83 * mm, 20 * mm, 22 * mm, 39 * mm], | |
| repeatRows=1, | |
| splitByRow=1, | |
| ) | |
| validation_table.setStyle(table_style()) | |
| sec_validation.append(validation_table) | |
| # Assemble evaluation-first: metrics and sentence-level validation lead, | |
| # then the per-snippet evidence assessment and references; the generated | |
| # narrative and retrieved context are demoted to the end. | |
| story.extend(sec_pathway) | |
| story.extend(sec_metrics) | |
| story.extend(sec_reliability) | |
| story.extend(sec_source_quality) | |
| story.extend(sec_validation) | |
| story.extend(sec_evidence) | |
| if sec_references: | |
| story.append(KeepTogether(sec_references[:2])) | |
| story.extend(sec_references[2:]) | |
| if sec_narrative: | |
| story.append(KeepTogether(sec_narrative)) | |
| story.extend(sec_related) | |
| story.extend(sec_provenance) | |
| def page_footer(canvas, doc): | |
| canvas.saveState() | |
| canvas.setFont(font_name, 7) | |
| canvas.setFillColor(colors.HexColor("#64748b")) | |
| canvas.drawString(18 * mm, 9 * mm, "TRACER-Net | Detailed RAG report") | |
| canvas.drawRightString(A4[0] - 18 * mm, 9 * mm, f"Page {doc.page}") | |
| canvas.restoreState() | |
| try: | |
| document.build( | |
| story, | |
| onFirstPage=page_footer, | |
| onLaterPages=page_footer, | |
| ) | |
| return temporary.name | |
| except Exception as exc: | |
| logging.error("Failed to build detailed RAG PDF: %s", exc, exc_info=True) | |
| try: | |
| os.unlink(temporary.name) | |
| except Exception: | |
| pass | |
| return None | |
| def _save_rag_detail_report(report_item): | |
| if not isinstance(report_item, dict): | |
| return None | |
| pdf_path = _save_rag_detail_pdf(report_item) | |
| if pdf_path and str(pdf_path).lower().endswith(".pdf") and os.path.isfile(pdf_path): | |
| return pdf_path | |
| logging.error("Detailed RAG PDF generation failed") | |
| return None | |
| def _empty_rag_detail_download(): | |
| return gr.update(value=None, visible=False) | |
| def _rag_detail_download_update(report_item): | |
| path = _save_rag_detail_report(report_item) | |
| if ( | |
| not path | |
| or not str(path).lower().endswith(".pdf") | |
| or not os.path.isfile(path) | |
| ): | |
| return _empty_rag_detail_download() | |
| return gr.update(value=path, visible=True) | |
| def _render_report_outputs(report_item): | |
| rendered = _render_report_html(report_item) | |
| return rendered, _rag_detail_download_update(report_item) | |
| def _render_report_html(report_item): | |
| if not report_item: | |
| return "No analysis available." | |
| metrics = None | |
| references = None | |
| traces = None | |
| path_str = None | |
| raw_report = "" | |
| if isinstance(report_item, dict): | |
| metrics = report_item.get("metrics") | |
| references = report_item.get("references") | |
| traces = report_item.get("sentence_traces") | |
| path_str = report_item.get("path_str") | |
| raw_report = report_item.get("report") or report_item.get("content") or "" | |
| else: | |
| raw_report = str(report_item) | |
| if not raw_report: | |
| return "No analysis available." | |
| content = _clean_report_markdown(raw_report) | |
| content = _normalize_rag_text(content, keep_newlines=True) | |
| content = re.sub(r"\*\*([^*]+)\*\*", r"\1", content) | |
| title = _extract_report_title(content) | |
| if is_bad_generation_text(title): | |
| title = "Provenance Analysis Report" | |
| if path_str: | |
| title = _path_report_title(path_str) | |
| body = content | |
| body = re.sub(r"(?im)^provenance analysis report:.*$", "", body).strip() | |
| body = re.sub(r"(?is)\nreferences\b.*", "", body).strip() | |
| body = re.sub(r"(?s)\$\$.*?\$\$", "", body).strip() | |
| body = re.sub(r"(?im)^(chemical reaction pathway|interpretation|evidence summary|quality metrics)\s*$", "", body).strip() | |
| intro_merged = _clean_public_report_text(body) | |
| if not intro_merged or is_bad_generation_text(intro_merged): | |
| intro_merged = ( | |
| "No evidence-grounded narrative is available for this graph pathway." | |
| ) | |
| body = "" | |
| all_evidence_blocks = [] | |
| evidence_blocks = [] | |
| related_context = ( | |
| report_item.get("related_context") or [] | |
| if isinstance(report_item, dict) | |
| else [] | |
| ) | |
| if isinstance(report_item, dict): | |
| all_evidence_blocks = _build_reference_evidence_blocks( | |
| report_item, | |
| references or [], | |
| ) | |
| evidence_blocks = _build_public_evidence_blocks( | |
| report_item, | |
| references or [], | |
| path_str or "", | |
| ) | |
| # By default the weak-evidence body is discarded and replaced with a | |
| # controlled statement, so unverified generated prose never leaks into | |
| # the public report. The one exception is a body explicitly flagged as | |
| # a grounded contextual synthesis of retrieved literature. | |
| _has_contextual_narrative = bool( | |
| isinstance(metrics, dict) and metrics.get("has_contextual_narrative") | |
| ) | |
| if not any(evidence_blocks) and not _has_contextual_narrative: | |
| species, conditions = path_components(path_str, include_empty_conditions=True) | |
| relation = " -> ".join(species) if species else str(path_str or "the recorded path") | |
| condition_text = ", ".join(dict.fromkeys(conditions)) or "unspecified" | |
| retrieval_available = ( | |
| not metrics or metrics.get("retrieval_available") is not False | |
| ) | |
| intro_merged = compose_graph_only_note( | |
| relation, | |
| condition_text, | |
| retrieved_context=bool(related_context), | |
| retrieval_available=retrieval_available, | |
| ) | |
| # The evidence guard appends a "Validation note: ..." scope caveat to the | |
| # narrative body. Keep it (it is the honest evidence-scope statement) but | |
| # render it as its own block so the Introduction reads as clean prose. | |
| intro_body, scope_note = _split_scope_note(intro_merged) | |
| md_parts = [] | |
| if title: | |
| md_parts.append(f"### {title}") | |
| md_parts.append("### Introduction\n\n" + (intro_body or intro_merged)) | |
| display_equation = _path_to_display_equation(path_str) | |
| if display_equation: | |
| md_parts.append("### Chemical Pathway\n\n" + display_equation) | |
| if is_bad_generation_text(body): | |
| body = "" | |
| body = _clean_public_report_text(body) | |
| if body and body.strip(): | |
| md_parts.append("### Interpretation\n\n" + body.strip()) | |
| species, conditions = path_components(path_str, include_empty_conditions=True) if path_str else ([], []) | |
| if species or conditions: | |
| lines = [] | |
| if species: | |
| nodes = [to_unicode_subscript(s) for s in species] | |
| lines.append("Nodes: " + ", ".join(nodes)) | |
| if conditions: | |
| uniq_conditions = list(dict.fromkeys( | |
| _display_condition_label(c) for c in conditions if c | |
| )) | |
| # Named "graph-encoded" because it is what the network records, which | |
| # is not the same claim as "the literature demonstrated this | |
| # condition". A node labelled UV may be supported by a paper | |
| # reporting visible-light exposure, and a provenance report must not | |
| # let the label borrow authority from the evidence. | |
| lines.append("Graph-encoded conditions: " + ", ".join(uniq_conditions)) | |
| literature_conditions = _literature_condition_note(evidence_blocks) | |
| if literature_conditions: | |
| lines.append( | |
| "Literature-supported conditions: " + literature_conditions | |
| ) | |
| else: | |
| # No condition wording could be read out of the matched passages, | |
| # so report how far the label was corroborated rather than | |
| # leaving the reader to assume it was demonstrated. | |
| corroboration = _condition_corroboration_note(evidence_blocks) | |
| if corroboration: | |
| lines.append("Literature corroboration: " + corroboration) | |
| md_parts.append("### Graph Path Summary\n\n" + "\n".join(lines)) | |
| if scope_note: | |
| md_parts.append("### Evidence Scope\n\n> " + scope_note) | |
| # Grouped by reaction step, not by reference. Listing each reference | |
| # separately repeated one conclusion once per supporting paper -- five | |
| # identical "As4S4 -> p-As4S4: direct" lines in one report -- which buried | |
| # the question a reader actually has: is this step supported, and by whom. | |
| # Each step is stated once, with the references and channels that back it. | |
| def _add_support(bucket, key, reference_index, channels): | |
| entry = bucket.setdefault( | |
| key, {"refs": [], "channels": []} | |
| ) | |
| if reference_index not in entry["refs"]: | |
| entry["refs"].append(reference_index) | |
| for channel in channels: | |
| if channel not in entry["channels"]: | |
| entry["channels"].append(channel) | |
| step_support = {} | |
| endpoint_support = {} | |
| unclassified_support = {} | |
| for index, block in enumerate(evidence_blocks, 1): | |
| if not isinstance(block, dict): | |
| continue | |
| snippet = _summarize_phase_tables(block.get("snippet")) or _clean_evidence_snippet( | |
| block.get("snippet"), max_chars=900 | |
| ) | |
| if not snippet: | |
| continue | |
| channels = list(dict.fromkeys( | |
| _RETRIEVAL_CHANNEL_LABELS.get(name, name) | |
| for item in block.get("snippets") or [] | |
| if isinstance(item, dict) and item.get("retrieval_origin") | |
| for name in [str(item.get("retrieval_origin") or "").replace("_", " ")] | |
| )) | |
| matches = [m for m in (block.get("edge_matches") or []) if isinstance(m, dict)] | |
| if not matches: | |
| status = str(block.get("evidence_status") or "unclassified").replace("_", " ") | |
| _add_support(unclassified_support, status, index, channels) | |
| continue | |
| for match in matches: | |
| verdict = str(match.get("verdict") or "unclassified").replace("_", " ") | |
| reactant = to_unicode_subscript(str(match.get("reactant") or "")) | |
| product = to_unicode_subscript(str(match.get("product") or "")) | |
| if match.get("evidence_scope") == "pathway_endpoint": | |
| _add_support( | |
| endpoint_support, (reactant, product, verdict), index, channels | |
| ) | |
| continue | |
| try: | |
| order = int(match.get("edge_index") or 0) | |
| except (TypeError, ValueError): | |
| order = 0 | |
| _add_support( | |
| step_support, (order, reactant, product, verdict), index, channels | |
| ) | |
| def _support_detail(entry): | |
| lines = ["Sources: " + " ".join(f"[{n}]" for n in sorted(entry["refs"]))] | |
| if entry["channels"]: | |
| lines.append("Retrieval: " + "; ".join(entry["channels"])) | |
| return lines | |
| retrieved_evidence_lines = [] | |
| for (_order, reactant, product, verdict) in sorted(step_support): | |
| entry = step_support[(_order, reactant, product, verdict)] | |
| phrase = ( | |
| "direct literature support" if verdict == "direct" else f"{verdict} support" | |
| ) | |
| heading = f"{reactant} → {product} — {phrase}" if reactant and product else phrase | |
| retrieved_evidence_lines.append( | |
| f"**{heading}** \n" + " \n".join(_support_detail(entry)) | |
| ) | |
| for (reactant, product, verdict) in sorted(endpoint_support): | |
| entry = endpoint_support[(reactant, product, verdict)] | |
| label = f"Overall {reactant} → {product}" if reactant and product else "Overall pathway" | |
| phrase = ( | |
| "pathway-endpoint support" if verdict == "direct" | |
| else f"pathway-endpoint {verdict} support" | |
| ) | |
| retrieved_evidence_lines.append( | |
| f"**{label} — {phrase}** \n" + " \n".join(_support_detail(entry)) | |
| ) | |
| for status in sorted(unclassified_support): | |
| entry = unclassified_support[status] | |
| retrieved_evidence_lines.append( | |
| f"**{status}** \n" + " \n".join(_support_detail(entry)) | |
| ) | |
| if retrieved_evidence_lines: | |
| md_parts.append( | |
| "### Retrieved Literature Evidence\n\n" | |
| + "\n\n".join(retrieved_evidence_lines) | |
| + "\n\n> Quoted passages for each reference are in the " | |
| "downloadable PDF report." | |
| ) | |
| # Stating the boundary of the evidence is part of reporting it: without | |
| # this, a list of supported steps reads as though the mechanism itself | |
| # had been demonstrated. | |
| limitation = _evidence_limitation_note(evidence_blocks, path_str) | |
| if limitation: | |
| md_parts.append("### Evidence Limitation\n\n" + limitation) | |
| # Candidates the gate rejected, and network sources no passage verified, | |
| # are audit trail rather than argument. They stay in the exported PDF; the | |
| # report itself carries only the evidence the claims actually rest on. | |
| displayed_ref_count = 0 | |
| citation_renumber_map = {} | |
| if references: | |
| selected_pairs = [ | |
| (i + 1, ref) | |
| for i, ref in enumerate(references) | |
| if i < len(evidence_blocks) and evidence_blocks[i] | |
| ] | |
| # One paper is one reference. Two passages from the same source arrive as | |
| # entries differing only in a trailing page/chunk marker ("2017:16" vs | |
| # "2017:24"), so string comparison listed the same DOI twice. Identity is | |
| # the DOI where there is one; the passage location belongs to the | |
| # evidence entry, not to the bibliography. Repeats are mapped onto the | |
| # first entry's number rather than dropped, so a citation to the second | |
| # occurrence still resolves instead of being stripped from the text. | |
| seen_ref = {} | |
| ref_lines = [] | |
| for n, ref in selected_pairs: | |
| text = str(ref or "") | |
| norm_ref = re.sub(r"\s+", " ", text).strip().lower() | |
| if not norm_ref: | |
| continue | |
| doi = normalize_doi(first_doi(text)) | |
| key = f"doi:{doi}" if doi else norm_ref | |
| if key in seen_ref: | |
| citation_renumber_map[int(n)] = seen_ref[key] | |
| continue | |
| new_n = len(ref_lines) + 1 | |
| seen_ref[key] = new_n | |
| citation_renumber_map[int(n)] = new_n | |
| ref_lines.append(f"[{new_n}] {ref}") | |
| if ref_lines: | |
| displayed_ref_count = len(ref_lines) | |
| md_parts.append("### References Used\n\n" + "\n".join(ref_lines)) | |
| # A contextual narrative cites [n] too, but its sources establish no edge, | |
| # so they never reach "references" and the citations resolved to nothing. | |
| # Listed under their own heading: calling them "References Used" would imply | |
| # they support the pathway, which is the one thing the narrative says they | |
| # do not do. | |
| if not displayed_ref_count: | |
| context_sources = ( | |
| report_item.get("context_sources") or [] | |
| if isinstance(report_item, dict) | |
| else [] | |
| ) | |
| seen_ctx = set() | |
| ctx_lines = [] | |
| for source in context_sources: | |
| norm = re.sub(r"\s+", " ", str(source or "")).strip() | |
| if not norm or norm.casefold() in seen_ctx: | |
| continue | |
| seen_ctx.add(norm.casefold()) | |
| ctx_lines.append(f"[{len(ctx_lines) + 1}] {norm}") | |
| if ctx_lines: | |
| # No caveat here. The narrative is required to close by saying these | |
| # observations do not verify the pathway, and it names the specific | |
| # steps while a section note could only repeat the point generically. | |
| # The heading already separates these from "References Used". | |
| md_parts.append("### Context Sources\n\n" + "\n".join(ctx_lines)) | |
| if metrics: | |
| snippet_blocks = [ | |
| block for block in evidence_blocks | |
| if isinstance(block, dict) and str(block.get("snippet") or "").strip() | |
| ] | |
| snippet_count = sum( | |
| max(1, len(block.get("snippets") or [])) | |
| for block in snippet_blocks | |
| ) | |
| source_count = len({ | |
| str(block.get("source") or "").strip().casefold() | |
| for block in snippet_blocks | |
| if str(block.get("source") or "").strip() | |
| }) | |
| total_statements = metrics.get( | |
| "total_statements", | |
| metrics.get("total_sentences", 0), | |
| ) | |
| unsupported_count = metrics.get( | |
| "unsupported_statements", | |
| max( | |
| metrics.get("total_sentences", 0) | |
| - metrics.get("supported_sentences", 0), | |
| 0, | |
| ), | |
| ) | |
| edge_coverage = metrics.get("edge_evidence_coverage", 0) | |
| q_lines = [ | |
| ( | |
| "- Literature retrieval: available" | |
| if metrics.get("retrieval_available") is not False | |
| else "- Literature retrieval: unavailable" | |
| ), | |
| ( | |
| "- Endpoint conversion evidence: supported" | |
| if metrics.get("endpoint_conversion_supported") | |
| else "- Endpoint conversion evidence: not retrieved" | |
| ), | |
| f"- Edge evidence coverage: {_metric_percent(edge_coverage)}", | |
| f"- Statement evidence coverage: {_metric_percent(metrics.get('evidence_coverage', 0))}", | |
| ] | |
| retrieval_trace = metrics.get("retrieval_trace") or {} | |
| strategy = str(metrics.get("retrieval_strategy") or "").strip() | |
| if strategy: | |
| q_lines.append(f"- Retrieval strategy: {strategy}") | |
| if retrieval_trace: | |
| q_lines.append( | |
| "- Dense retrieval: " | |
| f"{_metric_text(retrieval_trace.get('dense_candidates', 0))} candidates; " | |
| f"{_metric_text(retrieval_trace.get('dense_kept', 0))} accepted" | |
| ) | |
| if retrieval_trace.get("lexical_scanned"): | |
| q_lines.append( | |
| "- Lexical evidence fallback: " | |
| f"{_metric_text(retrieval_trace.get('lexical_scanned', 0))} scanned; " | |
| f"{_metric_text(retrieval_trace.get('lexical_candidates', 0))} candidates; " | |
| f"{_metric_text(retrieval_trace.get('lexical_kept', 0))} accepted" | |
| ) | |
| if metrics.get("related_context_count"): | |
| q_lines.append( | |
| "- Related context retained for review: " | |
| f"{_metric_text(metrics.get('related_context_count'))}" | |
| ) | |
| citation_accuracy = metrics.get("citation_accuracy") | |
| q_lines.append( | |
| f"- Citation accuracy: {_metric_percent(citation_accuracy)}" | |
| if citation_accuracy is not None | |
| else "- Citation accuracy: n/a" | |
| ) | |
| q_lines.extend([ | |
| f"- Graph alignment: {_metric_percent(metrics.get('graph_alignment', 0))}", | |
| f"- Unsupported statements: {_metric_text(unsupported_count)} / {_metric_text(total_statements)}", | |
| ( | |
| f"- Retrieved snippets: {_metric_text(snippet_count)} " | |
| f"({'scoped edge/pathway RAG' if metrics.get('retrieval_available') is not False else 'retrieval unavailable'})" | |
| ), | |
| f"- Sources used: {_metric_text(source_count)}; references shown: {_metric_text(displayed_ref_count)}", | |
| ]) | |
| if metrics.get("retrieval_available") is False: | |
| evidence_status = "Literature retrieval unavailable; evidence status undetermined" | |
| elif snippet_count == 0 and metrics.get("related_context_count"): | |
| evidence_status = ( | |
| "No direct evidence for the complete pathway; related context " | |
| "was retrieved but was not used as claim evidence" | |
| ) | |
| elif snippet_count == 0: | |
| evidence_status = "No direct pathway evidence retrieved" | |
| elif metrics.get("endpoint_conversion_supported") and edge_coverage < 1: | |
| evidence_status = ( | |
| "Endpoint conversion supported; individual graph steps " | |
| "remain unverified" | |
| ) | |
| elif edge_coverage <= 0: | |
| evidence_status = ( | |
| "No direct edge evidence; retrieved context is non-direct " | |
| "or contradictory" | |
| ) | |
| elif edge_coverage < 1 or metrics.get("manual_verification"): | |
| evidence_status = "Partial edge support; unresolved steps require review" | |
| else: | |
| evidence_status = "Direct snippet support for every graph edge" | |
| q_lines.append(f"- Overall confidence: {evidence_status}") | |
| # Keep evaluation metrics in the structured report payload for the | |
| # detailed PDF, but do not render the diagnostic block in the compact | |
| # web interface. | |
| rendered = "\n\n".join([p for p in md_parts if p and str(p).strip()]) | |
| if citation_renumber_map: | |
| rendered = re.sub( | |
| r"\[(\d+)\]", | |
| lambda m: f"[{citation_renumber_map[int(m.group(1))]}]" | |
| if int(m.group(1)) in citation_renumber_map | |
| else "", | |
| rendered, | |
| ) | |
| elif not displayed_ref_count and "### Context Sources" not in rendered: | |
| # Citations are stripped when nothing resolves them. A contextual | |
| # narrative now lists its sources under their own heading, numbered to | |
| # match, so stripping there would delete working citations. | |
| rendered = _strip_all_citation_numbers(rendered) | |
| if _path_to_display_equation(path_str): | |
| rendered = re.sub(r"(?im)^\s*Pathway expression:\s*.*$", "", rendered) | |
| plain_path = _path_to_plain_expression(path_str) | |
| if plain_path: | |
| rendered = re.sub(r"(?s)\$\$.*?\\xrightarrow.*?\$\$", "", rendered) | |
| rendered = re.sub(r"(?s)\\\[.*?\\xrightarrow.*?\\\]", "", rendered) | |
| rendered = re.sub(r"(?m)^.*\\xrightarrow.*$", "", rendered) | |
| rendered = _strip_pathway_expression_text(rendered) | |
| path_line = "Pathway expression: " + plain_path | |
| lines = [] | |
| path_line_seen = False | |
| for line in rendered.splitlines(): | |
| if line.strip() == path_line: | |
| if path_line_seen: | |
| continue | |
| path_line_seen = True | |
| lines.append(line) | |
| rendered = "\n".join(lines) | |
| return _normalize_report_formula_display(rendered) | |
| def _ensure_rag_report_for_index(all_paths, idx, rag_reports): | |
| if idx < 0 or idx >= len(all_paths): | |
| return "No analysis available." | |
| if rag_reports is None: | |
| rag_reports = [] | |
| if idx < len(rag_reports) and rag_reports[idx]: | |
| return _attach_path_evidence_to_report(rag_reports[idx], all_paths[idx]) | |
| # Gradio starts listening before the RAG index is initialized. A first | |
| # request can therefore arrive while startup still owns the initialization | |
| # lock. Calling the initializer here either starts it or waits for that | |
| # in-flight initialization, so the first valid pathway receives a report. | |
| if rag_service is None and _rag_init_error is None: | |
| _initialize_rag_service() | |
| if rag_service: | |
| path_info = all_paths[idx] | |
| try: | |
| report = rag_service.generate_report_for_path( | |
| path_info.get("source"), | |
| path_info.get("path_str"), | |
| path_info.get("full_path"), | |
| ) | |
| except Exception as e: | |
| report = f"RAG generation failed: {e}" | |
| elif _rag_init_error: | |
| report = ( | |
| "The pathway was found, but the evidence-analysis service failed " | |
| "to initialize. Please check the RAG startup log." | |
| ) | |
| else: | |
| report = ( | |
| "The pathway was found, but the evidence-analysis service is " | |
| "temporarily unavailable." | |
| ) | |
| if idx >= len(rag_reports): | |
| rag_reports.extend([None] * (idx + 1 - len(rag_reports))) | |
| report = _attach_path_evidence_to_report(report, all_paths[idx]) | |
| rag_reports[idx] = report | |
| return report | |
| def _parse_material_constraints(material_name): | |
| return crn_pathways.parse_material_constraints(material_name) | |
| def _path_endpoint_input(material_name): | |
| constraints = _parse_material_constraints(material_name) | |
| return constraints[-1] if constraints else "" | |
| def _resolve_crn_constraints(material_name): | |
| return crn_pathways.resolve_crn_constraints(G, material_name) | |
| def _trace_material_pathways(material_name, max_depth=10, limit=100): | |
| return crn_pathways.trace_material_pathways( | |
| _crn_tracer, G, material_name, max_depth=max_depth, limit=limit | |
| ) | |
| def _filter_paths_by_evidence(all_paths, include_composite=False): | |
| if include_composite: | |
| return list(all_paths or []) | |
| filtered = [] | |
| for path_info in all_paths or []: | |
| if path_info.get("evidence_type") == "composite": | |
| continue | |
| unique_conditions = { | |
| normalize_legacy_text(str(condition).strip()) | |
| for condition in (path_info.get("conditions") or []) | |
| if str(condition).strip() | |
| } | |
| if len(unique_conditions) > 1: | |
| continue | |
| filtered.append(path_info) | |
| return filtered | |
| def _selected_path_number(selected_idx): | |
| if isinstance(selected_idx, list): | |
| selected_idx = selected_idx[0] if selected_idx else None | |
| if selected_idx is None or selected_idx == "": | |
| return 1 | |
| match = re.search(r"\d+", str(selected_idx)) | |
| if not match: | |
| return 1 | |
| return int(match.group(0)) | |
| def _compact_condition_sequence(conditions): | |
| items = [normalize_legacy_text(str(item).strip()) for item in (conditions or []) if str(item).strip()] | |
| if not items: | |
| return "" | |
| unique_items = [] | |
| for item in items: | |
| if item not in unique_items: | |
| unique_items.append(item) | |
| return "; ".join(unique_items) | |
| def _condition_sequence_display(conditions, evidence_type=None): | |
| compact = _compact_condition_sequence(conditions) | |
| return compact | |
| def _species_path_display(species_in_path, formatter): | |
| return normalize_legacy_text(" -> ".join(formatter(sp) for sp in species_in_path)) | |
| def trace_pathways(material, include_composite, current_state): | |
| empty_image = gr.update(value=None) | |
| def _error_return(msg): | |
| return msg, empty_image, [], gr.update(choices=[], value=None), [], _render_report_html(msg), _empty_rag_detail_download(), current_state, gr.update(choices=[], value=None), gr.update(choices=[], value=None) | |
| try: | |
| if not CRN_AVAILABLE or _crn_tracer is None: | |
| return _error_return("CRN module not available") | |
| if not material: | |
| return _error_return("Please enter a compound name") | |
| constraints = _parse_material_constraints(material) | |
| if not constraints: | |
| return _error_return("Please enter a compound name") | |
| resolved_constraints = _resolve_crn_constraints(constraints) | |
| unresolved = [entered for entered, matches in resolved_constraints if not matches] | |
| if unresolved: | |
| missing_display = ", ".join(unresolved) | |
| return _error_return(f"Not present in the CRN network: {missing_display}") | |
| target = constraints[-1] | |
| resolved_targets, trace_G, trace_report, all_paths_raw = _trace_material_pathways(constraints, max_depth=10, limit=100) | |
| all_paths = _filter_paths_by_evidence(all_paths_raw, include_composite) | |
| if os.getenv("TRACE_DEBUG", "0").strip().lower() in ("1", "true", "yes"): | |
| print(f"trace G: {trace_G}, trace_report: {trace_report}, all_paths: {all_paths},{len(all_paths)}") | |
| if not all_paths: | |
| if all_paths_raw: | |
| return _error_return("Only complex pathways were found. Enable 'Show complex paths' to display them.") | |
| if len(constraints) > 1: | |
| ordered_display = " -> ".join(formula_for_markdown(value) for value in constraints) | |
| return _error_return( | |
| f"No upstream pathway contains all entered compounds in this order: {ordered_display}. Place the final endpoint last." | |
| ) | |
| return _error_return( | |
| "No upstream pathway was found for the entered aging product. " | |
| "It may be an initial material in the current CRN, or its formation " | |
| "reaction is not recorded." | |
| ) | |
| subPos = _crn_tracer._compute_hierarchical_layout(trace_G) | |
| except Exception as e: | |
| msg = f"Pathway tracing failed: {e}" | |
| print(msg) | |
| return _error_return(msg) | |
| first_path = all_paths[0]['full_path'] | |
| first_target = all_paths[0].get('target', target) | |
| img_buf = _crn_tracer._draw_single_path( | |
| trace_G, subPos, first_path, first_target, | |
| first_path[0] if first_path else first_target, | |
| context_paths=all_paths, | |
| ) | |
| img = coerce_image_result(img_buf) | |
| if img is None: | |
| return _error_return("Pathway image rendering failed") | |
| img_output = _save_gradio_image_file(img) | |
| if img_output is None: | |
| return _error_return("Pathway image export failed") | |
| table = [] | |
| for i, path_info in enumerate(all_paths[:50], 1): | |
| full_path = path_info['full_path'] | |
| condition = path_info.get('condition', '') | |
| step_conditions = path_info.get('conditions') or [condition] * max(0, len(full_path) // 2) | |
| species_in_path = [full_path[j] for j in range(0, len(full_path), 2)] | |
| num_steps = len(species_in_path) - 1 if len(species_in_path) > 1 else 0 | |
| conditions_display = _condition_sequence_display( | |
| [ | |
| step_conditions[j] if j < len(step_conditions) else condition | |
| for j in range(num_steps) | |
| ], | |
| path_info.get("evidence_type"), | |
| ) | |
| species_path = _species_path_display(species_in_path, to_unicode_subscript) | |
| table.append([ | |
| i, | |
| to_unicode_subscript(species_in_path[0]) if species_in_path else '', | |
| to_unicode_subscript(species_in_path[-1]) if species_in_path else '', | |
| num_steps, | |
| conditions_display, | |
| species_path, | |
| ]) | |
| choices = [] | |
| for i, p in enumerate(all_paths, 1): | |
| condition = normalize_legacy_text((p.get("condition") or "").strip()) | |
| label = f"{i}. {condition}" if condition else f"{i}. Path {i}" | |
| choices.append((label, str(i))) | |
| precursor_sources = sorted({p.get("source") for p in all_paths if p.get("source")}) | |
| all_available = all_paths_raw if "all_paths_raw" in locals() else all_paths | |
| single_source_count = sum(1 for p in all_available if p.get("evidence_type") == "single_source") | |
| composite_count = len(all_available) - single_source_count | |
| simple_condition_count = len(_filter_paths_by_evidence(all_available, False)) | |
| complex_count = len(all_available) - simple_condition_count | |
| resolved_display = ", ".join(formula_for_markdown(name) for name in resolved_targets) | |
| constraint_display = " -> ".join(formula_for_markdown(value) for value in constraints) | |
| if len(constraints) > 1: | |
| result_summary = f"Required species (path order): {constraint_display} | Endpoint: {formula_for_markdown(target)} | Matched endpoint: {resolved_display} | Displayed: {len(all_paths)} jointly constrained upstream pathways" | |
| else: | |
| result_summary = f"Detected aging product: {formula_for_markdown(target)} | Matched target: {resolved_display} | Displayed: {len(all_paths)} upstream pathways" | |
| report = f"""**Pathway tracing completed** | |
| {result_summary} | |
| Single-condition available: {simple_condition_count} | Complex available: {complex_count} | Complex: {'shown' if include_composite else 'hidden'} | |
| """ | |
| rag_reports = [None] * len(all_paths) | |
| if all_paths: | |
| try: | |
| gr.Info("Generating first pathway report...") | |
| rag_reports[0] = _ensure_rag_report_for_index(all_paths, 0, rag_reports) | |
| except Exception as exc: | |
| logging.error("Pathway RAG report generation failed: %s", exc, exc_info=True) | |
| rag_reports[0] = ( | |
| "Pathway visualization was generated successfully, but the ChatGPT/RAG " | |
| f"analysis report failed: {exc}" | |
| ) | |
| rag_reports = _sanitize_for_state(rag_reports) | |
| if rag_reports: | |
| rag_initial_report, rag_detail_update = _render_report_outputs(rag_reports[0]) | |
| else: | |
| rag_initial_report = _render_report_html("No analysis available.") | |
| rag_detail_update = _empty_rag_detail_download() | |
| path_list = [normalize_legacy_text(p.get("path_str")) for p in all_paths if p.get("path_str")] | |
| copilot_updates = { | |
| "current_materials": constraints, | |
| "identified_compounds": constraints, | |
| "current_pathways": path_list, | |
| "last_rag_report": _report_to_text(rag_reports[0]) if rag_reports else "", | |
| "last_rag_path": normalize_legacy_text(all_paths[0].get("path_str")) if all_paths else "", | |
| "last_rag_material": target, | |
| "current_material": target, | |
| } | |
| new_state = _merge_copilot_state(current_state, copilot_updates) | |
| first_value = choices[0][1] if choices else None | |
| copilot_path_update = gr.update(choices=choices, value=first_value) | |
| material_choices = [(to_unicode_subscript(value), value) for value in constraints] | |
| return report, img_output, table, gr.update(choices=choices, value=first_value), rag_reports, rag_initial_report, rag_detail_update, new_state, copilot_path_update, gr.update(choices=material_choices, value=target) | |
| def _render_path_by_index(material_name, idx, rag_report, include_composite=False): | |
| empty_image = gr.update(value=None) | |
| if not CRN_AVAILABLE or _crn_tracer is None: | |
| return empty_image, "CRN module not available", _render_report_html("No analysis available."), _empty_rag_detail_download(), "" | |
| try: | |
| constraints = _parse_material_constraints(material_name) | |
| if not constraints: | |
| return empty_image, "No materials specified", _render_report_html("No analysis available."), _empty_rag_detail_download(), "" | |
| target = constraints[-1] | |
| _resolved_targets, trace_G, trace_report, all_paths_raw = _trace_material_pathways(constraints, max_depth=10, limit=100) | |
| all_paths = _filter_paths_by_evidence(all_paths_raw, include_composite) | |
| if not all_paths: | |
| return empty_image, "No pathways found", _render_report_html("No analysis available."), _empty_rag_detail_download(), "" | |
| idx = max(0, min(idx, len(all_paths) - 1)) | |
| path_info = all_paths[idx] | |
| full_path = path_info['full_path'] | |
| condition = path_info.get('condition', '') | |
| step_conditions = path_info.get('conditions') or [condition] * max(0, len(full_path) // 2) | |
| try: | |
| current_report_item = _ensure_rag_report_for_index(all_paths, idx, rag_report or []) | |
| except Exception as exc: | |
| logging.error("Pathway RAG report generation failed: %s", exc, exc_info=True) | |
| current_report_item = ( | |
| "Pathway visualization was generated successfully, but the ChatGPT/RAG " | |
| f"analysis report failed: {exc}" | |
| ) | |
| current_report, detail_update = _render_report_outputs(current_report_item) | |
| try: | |
| subPos = _crn_tracer._compute_hierarchical_layout(trace_G) | |
| except Exception as e: | |
| msg = f"Hierarchical layout failed: {e}" | |
| print(msg) | |
| return empty_image, f"Rendering failed: {msg}", _render_report_html("No analysis available."), _empty_rag_detail_download(), "" | |
| source_name = full_path[0] if full_path else target | |
| path_target = path_info.get('target', target) | |
| img_buf = _crn_tracer._draw_single_path( | |
| trace_G, subPos, full_path, path_target, source_name, | |
| context_paths=all_paths, | |
| ) | |
| img = coerce_image_result(img_buf) | |
| if img is None: | |
| return empty_image, "Pathway image rendering failed", current_report, detail_update, path_info.get("path_str") or "" | |
| img_output = _save_gradio_image_file(img) | |
| if img_output is None: | |
| return empty_image, "Pathway image export failed", current_report, detail_update, path_info.get("path_str") or "" | |
| species_in_path = [full_path[j] for j in range(0, len(full_path), 2)] | |
| species_path = _species_path_display(species_in_path, formula_for_markdown) | |
| condition_summary = _condition_sequence_display(step_conditions, path_info.get("evidence_type")) | |
| if condition_summary: | |
| summary = f"**Path #{idx+1}/{len(all_paths)}**\n\n**Materials:** {species_path}\n\n**Conditions:** {condition_summary}" | |
| else: | |
| summary = f"**Path #{idx+1}/{len(all_paths)}**\n\n**Materials:** {species_path}" | |
| return img_output, summary, current_report, detail_update, normalize_legacy_text(path_info.get("path_str")) or "" | |
| except Exception as e: | |
| return empty_image, f"Rendering failed: {e}", _render_report_html("No analysis available."), _empty_rag_detail_download(), "" | |
| def prev_path(material_name, selected_idx, rag_report, include_composite, current_state): | |
| empty_image = gr.update(value=None) | |
| try: | |
| cur = _selected_path_number(selected_idx) | |
| new = max(1, cur - 1) | |
| img, summary, current_rag_report, detail_update, path_str = _render_path_by_index(material_name, new-1, rag_report, include_composite) | |
| raw_report = _report_to_text(rag_report[new-1]) if rag_report and new-1 < len(rag_report) else "" | |
| updates = { | |
| "last_rag_report": raw_report, | |
| "last_rag_path": path_str, | |
| "last_rag_material": _path_endpoint_input(material_name), | |
| } | |
| new_state = _merge_copilot_state(current_state, updates) | |
| return img, summary, str(new), current_rag_report, detail_update, new_state | |
| except Exception as e: | |
| return empty_image, f"Operation failed: {e}", selected_idx, _render_report_html("No analysis available."), _empty_rag_detail_download(), current_state | |
| def next_path(material_name, selected_idx, rag_report, include_composite, current_state): | |
| empty_image = gr.update(value=None) | |
| try: | |
| cur = _selected_path_number(selected_idx) | |
| new = cur + 1 | |
| img, summary, current_rag_report, detail_update, path_str = _render_path_by_index(material_name, new-1, rag_report, include_composite) | |
| raw_report = _report_to_text(rag_report[new-1]) if rag_report and new-1 < len(rag_report) else "" | |
| updates = { | |
| "last_rag_report": raw_report, | |
| "last_rag_path": path_str, | |
| "last_rag_material": _path_endpoint_input(material_name), | |
| } | |
| new_state = _merge_copilot_state(current_state, updates) | |
| return img, summary, str(new), current_rag_report, detail_update, new_state | |
| except Exception as e: | |
| return empty_image, f"Operation failed: {e}", selected_idx, _render_report_html("No analysis available."), _empty_rag_detail_download(), current_state | |
| def select_path(material_name, selected_idx, rag_report, include_composite, current_state): | |
| empty_image = gr.update(value=None) | |
| if not material_name or not selected_idx: | |
| return empty_image, "", _render_report_html("No analysis available."), _empty_rag_detail_download(), current_state | |
| try: | |
| if not selected_idx: | |
| return empty_image, "", "No analysis available.", _empty_rag_detail_download(), current_state | |
| idx = _selected_path_number(selected_idx) - 1 | |
| img, summary, current_report, detail_update, path_str = _render_path_by_index(material_name, idx, rag_report, include_composite) | |
| raw_report = _report_to_text(rag_report[idx]) if rag_report and idx < len(rag_report) else "" | |
| updates = { | |
| "last_rag_report": raw_report, | |
| "last_rag_path": path_str, | |
| "last_rag_material": _path_endpoint_input(material_name), | |
| } | |
| new_state = _merge_copilot_state(current_state, updates) | |
| return img, summary, current_report, detail_update, new_state | |
| except Exception as e: | |
| return empty_image, f"Selection failed: {e}", _render_report_html("No analysis available."), _empty_rag_detail_download(), current_state | |
| custom_css = """ | |
| :root { | |
| --ink: #1f2330; | |
| --muted: #5a5f6f; | |
| --accent: #667eea; | |
| --accent-2: #764ba2; | |
| --panel: #f6f7fb; | |
| --panel-2: #ffffff; | |
| --border: #e3e6ef; | |
| } | |
| body { background: linear-gradient(120deg, #f6f7fb 0%, #ffffff 45%, #f6f7fb 100%); } | |
| .gradio-container { color: var(--ink); max-width: 1400px !important; margin: 0 auto !important; } | |
| .tab-nav button { font-size: 16px; font-weight: 600; padding: 12px 24px; transition: all 0.3s ease; color: var(--muted); } | |
| .tab-nav button:hover { transform: translateY(-2px); color: var(--ink); } | |
| .tabitem { min-height: calc(100vh - 280px); max-width: 1400px !important; width: 100% !important; margin: 0 auto !important; padding: 20px !important; overflow-x: hidden; } | |
| #page { max-width: 1400px !important; width: 100% !important; margin: 0 auto !important; } | |
| .upload-container { border: 2px dashed var(--accent-2); border-radius: 10px; padding: 20px; transition: all 0.3s ease; background: #fff; } | |
| .upload-container:hover { border-color: var(--accent); background: rgba(102, 126, 234, 0.08); } | |
| button { border-radius: 10px !important; font-weight: 600 !important; transition: all 0.3s ease !important; } | |
| button:hover { transform: translateY(-2px); box-shadow: 0 6px 16px rgba(20, 20, 20, 0.12); } | |
| .primary-btn { background: linear-gradient(135deg, #667eea 0%, #764ba2 100%) !important; color: white !important; } | |
| #prev_btn, #next_btn { min-width: 130px !important; font-size: 15px !important; } | |
| .result-section { background: var(--panel); border: 1px solid var(--border); border-radius: 12px; padding: 20px; margin: 15px 0; } | |
| .report-box {height: auto !important; min-height: 320px !important; max-height: none !important; overflow: visible !important; border: 1px solid var(--border); padding: 15px; border-radius: 10px; background-color: #fff; word-break: break-word;} | |
| .report-box table {display: block; max-width: 100%; overflow-x: auto; font-size: 12px;} | |
| .report-box p, .report-box li {overflow-wrap: anywhere;} | |
| .pathway-layout {align-items: flex-start !important;} | |
| @media (max-width: 1000px) { | |
| .pathway-layout {flex-direction: column !important;} | |
| .pathway-layout > div {width: 100% !important; min-width: 0 !important;} | |
| } | |
| .query-section { background: linear-gradient(135deg, #f0f3ff 0%, #f7f5ff 100%); border-radius: 12px; padding: 20px; margin-top: 20px; border: 1px solid var(--border); } | |
| hr { margin: 25px 0; border: none; border-top: 2px solid var(--border); } | |
| #copilot_panel { max-width: 1400px !important; width: 100% !important; margin: 0 auto !important; } | |
| #copilot_chat { resize: both !important; overflow: auto !important; min-height: 300px !important; min-width: 400px !important; position: relative !important; } | |
| #copilot_chat > div { resize: both !important; overflow: auto !important; min-height: 300px !important; min-width: 400px !important; } | |
| #copilot_chat .chatbot { resize: both !important; overflow: auto !important; min-height: 300px !important; min-width: 400px !important; } | |
| #copilot_chat::after { content: ''; position: absolute; bottom: 0; right: 0; width: 20px; height: 20px; background: linear-gradient(-45deg, transparent 30%, rgba(0,0,0,0.1) 30%, rgba(0,0,0,0.1) 40%, transparent 40%, transparent 60%, rgba(0,0,0,0.1) 60%, rgba(0,0,0,0.1) 70%, transparent 70%); cursor: nwse-resize; pointer-events: none; } | |
| """ | |
| english_ui_js = r""" | |
| () => { | |
| // Gradio localises its built-in chrome to the visitor's browser language, so | |
| // a zh-CN browser shows Chinese labels on an otherwise English app. We map the | |
| // built-in strings back to English on both text nodes and the attributes that | |
| // carry them (alt / title / aria-label / placeholder). | |
| // | |
| // Text nodes only receive LONG, unambiguous phrases: a generated answer to a | |
| // Chinese question is Chinese prose, and replacing a bare word like "下载" | |
| // there would corrupt the report. Short button labels are therefore applied | |
| // to attributes only, where the value is never user content. | |
| const textReplacements = new Map([ | |
| ["将文件拖放到此处", "Drop file here"], | |
| ["将文件拖拽到此处", "Drop file here"], | |
| ["拖放文件至此处", "Drop file here"], | |
| ["拖放文件到这里", "Drop file here"], | |
| ["点击上传", "Click to upload"], | |
| ["- 或 -", "- or -"], | |
| ["通过 API 使用", "Use via API"], | |
| ["使用 Gradio 构建", "Built with Gradio"], | |
| ["加载中...", "Loading..."], | |
| ["处理中...", "Processing..."], | |
| ["错误", "Error"] | |
| ]); | |
| // Applied to attribute values only (alt / title / aria-label / placeholder). | |
| const attrReplacements = new Map([ | |
| ["标志", "logo"], | |
| ["下载", "Download"], | |
| ["分享", "Share"], | |
| ["清除", "Clear"], | |
| ["清空", "Clear"], | |
| ["提交", "Submit"], | |
| ["上传", "Upload"], | |
| ["上传文件", "Upload file"], | |
| ["全屏", "Fullscreen"], | |
| ["退出全屏", "Exit fullscreen"], | |
| ["复制", "Copy"], | |
| ["复制到剪贴板", "Copy to clipboard"], | |
| ["已复制", "Copied"], | |
| ["撤销", "Undo"], | |
| ["重做", "Redo"], | |
| ["编辑", "Edit"], | |
| ["删除", "Delete"], | |
| ["发送", "Send"], | |
| ["停止", "Stop"], | |
| ["标记", "Flag"], | |
| ["分享到社区", "Share to community"] | |
| ]); | |
| const attrNames = ["alt", "title", "aria-label", "placeholder", "aria-roledescription"]; | |
| const applyMap = (value, map) => { | |
| let updated = value; | |
| for (const [source, target] of map) { | |
| if (updated.includes(source)) updated = updated.replaceAll(source, target); | |
| } | |
| return updated; | |
| }; | |
| const translateTextNode = (node) => { | |
| if (!node || node.nodeType !== Node.TEXT_NODE) return; | |
| const value = node.nodeValue || ""; | |
| const updated = applyMap(value, textReplacements); | |
| if (updated !== value) node.nodeValue = updated; | |
| }; | |
| const translateAttributes = (el) => { | |
| if (!el || el.nodeType !== Node.ELEMENT_NODE) return; | |
| for (const name of attrNames) { | |
| if (!el.hasAttribute(name)) continue; | |
| const value = el.getAttribute(name) || ""; | |
| const updated = applyMap(applyMap(value, textReplacements), attrReplacements); | |
| if (updated !== value) el.setAttribute(name, updated); | |
| } | |
| }; | |
| const translateTree = (root) => { | |
| if (!root) return; | |
| if (root.nodeType === Node.TEXT_NODE) { | |
| translateTextNode(root); | |
| return; | |
| } | |
| if (root.nodeType !== Node.ELEMENT_NODE) return; | |
| translateAttributes(root); | |
| const walker = document.createTreeWalker( | |
| root, | |
| NodeFilter.SHOW_TEXT | NodeFilter.SHOW_ELEMENT | |
| ); | |
| while (walker.nextNode()) { | |
| const node = walker.currentNode; | |
| if (node.nodeType === Node.TEXT_NODE) translateTextNode(node); | |
| else translateAttributes(node); | |
| } | |
| }; | |
| const translateAll = () => { | |
| try { translateTree(document.body); } catch (e) { /* never break the app */ } | |
| }; | |
| // IMPORTANT: do NOT run a live MutationObserver during Gradio's hydration. | |
| // Mutating text / attributes while Svelte is still building the DOM can | |
| // corrupt its node references and abort the whole render -- observed as a | |
| // page stuck on "Loading" on slower Hugging Face startups. The built-in | |
| // chrome we translate (footer, upload dropzone, icon-button labels) is | |
| // static, so a few timed passes after load cover it without observing during | |
| // hydration. | |
| [0, 400, 1200, 2500, 4500].forEach((delay) => setTimeout(translateAll, delay)); | |
| // Only after the app has clearly mounted do we attach a debounced observer, | |
| // to catch controls that appear later (e.g. a download button after a plot | |
| // renders). Starting it post-hydration keeps it from interfering with load. | |
| let started = false; | |
| let attempts = 0; | |
| const waitForMount = setInterval(() => { | |
| attempts += 1; | |
| const mounted = document.querySelector(".gradio-container, .tabs, [role='tablist']"); | |
| if (!mounted && attempts <= 40) return; | |
| clearInterval(waitForMount); | |
| if (started) return; | |
| started = true; | |
| setTimeout(() => { | |
| translateAll(); | |
| try { | |
| let pending = null; | |
| const observer = new MutationObserver(() => { | |
| if (pending) return; | |
| pending = setTimeout(() => { pending = null; translateAll(); }, 250); | |
| }); | |
| observer.observe(document.body, { | |
| childList: true, | |
| subtree: true, | |
| characterData: true, | |
| attributes: true, | |
| attributeFilter: attrNames | |
| }); | |
| } catch (e) { /* observer is best-effort only */ } | |
| }, 2000); | |
| }, 300); | |
| } | |
| """ | |
| # Kill-switch: set TRACERNET_DISABLE_UI_I18N=1 in the Space variables to ship | |
| # with no injected UI-translation JS at all. Use it to confirm whether the | |
| # translator is involved if the page ever fails to leave the loading state. | |
| _ui_i18n_disabled = os.getenv("TRACERNET_DISABLE_UI_I18N", "").strip().lower() in { | |
| "1", "true", "yes", "on" | |
| } | |
| with gr.Blocks( | |
| title="TRACER-Net", | |
| css=custom_css, | |
| js=(None if _ui_i18n_disabled else english_ui_js), | |
| theme=gr.themes.Soft(), | |
| analytics_enabled=False, | |
| ) as demo: | |
| gr.HTML(""" | |
| <div style="text-align:center; padding: 26px; background: linear-gradient(135deg, #667eea 0%, #764ba2 100%); border-radius: 14px; margin-bottom: 30px; box-shadow: 0 10px 24px rgba(40, 46, 72, 0.25);"> | |
| <h1 style="color: #fdf9f3; margin: 0; font-size: 42px; font-weight: 700; letter-spacing: 0.3px;">TRACER-Net</h1> | |
| <p style="color: rgba(253,249,243,0.92); margin-top: 10px; font-size: 16px; font-weight: 500;">Integrated Platform for Spectral Identification and Chemical Reaction Network Analysis</p> | |
| </div> | |
| """) | |
| copilot_state = gr.State({ | |
| "current_spectrum": "", | |
| "spectrum_name": "", | |
| "spectrum_type": "", | |
| "current_pathways": [], | |
| "current_materials": [], | |
| "identified_compounds": [], | |
| "spectrum_matches": [], | |
| "copilot_evidence_map": {}, | |
| "copilot_sentence_traces": [], | |
| }) | |
| with gr.Tabs(): | |
| with gr.Tab("Spectral Identification", id=1): | |
| with gr.Column(elem_id="page"): | |
| with gr.Row(equal_height=False, elem_classes="pathway-layout"): | |
| with gr.Column(scale=1): | |
| gr.Markdown("### Upload Spectrum Data") | |
| gr.Markdown("*Upload up to 4 spectrum files. Select file, choose spectrum type, then click Upload. Repeat to add more files.*") | |
| # Single upload slot (reusable for multiple files) | |
| with gr.Row(): | |
| file_input = gr.File( | |
| label="Select file", | |
| file_types=[".txt", ".csv", ".asc"], | |
| file_count="single", | |
| scale=3, | |
| height=100, | |
| elem_id="spectrum_file_input", | |
| ) | |
| type_input = gr.Dropdown( | |
| ["ATR-FTIR", "Raman", "XRD", "FORS"], | |
| label="Spectrum Type", | |
| value="Raman", | |
| scale=2 | |
| ) | |
| upload_btn = gr.Button("Upload", variant="primary", scale=1) | |
| # Display uploaded files list | |
| uploaded_files_display = gr.Column(visible=False) | |
| with uploaded_files_display: | |
| with gr.Row(): | |
| gr.Markdown("**Uploaded Files:**") | |
| btn_clear_all = gr.Button("Clear All", variant="secondary", size="sm", visible=False) | |
| uploaded_files_list = [] # List of display components for uploaded files | |
| for i in range(4): | |
| with gr.Row(visible=False) as file_row: | |
| uploaded_file_name = gr.Textbox( | |
| label=f"File {i+1}", | |
| interactive=False, | |
| visible=True, | |
| scale=2, | |
| show_label=True | |
| ) | |
| uploaded_file_type = gr.Textbox( | |
| label="Type", | |
| interactive=False, | |
| visible=True, | |
| scale=2, | |
| show_label=True | |
| ) | |
| remove_uploaded_btn = gr.Button("Remove", visible=True, scale=1, size="sm", variant="stop", min_width=80) | |
| uploaded_files_list.append({ | |
| 'name': uploaded_file_name, | |
| 'type': uploaded_file_type, | |
| 'remove': remove_uploaded_btn, | |
| 'row': file_row | |
| }) | |
| # Store uploaded files with their types | |
| uploaded_files_state = gr.State([]) # List of {file_path, spectrum_type, file_name} | |
| enable_fusion = gr.Checkbox( | |
| label="Enable Multimodal Fusion", | |
| value=True, | |
| info="Fuse results from multiple spectral modalities for more reliable material identification. Requires at least 2 files." | |
| ) | |
| btn_id = gr.Button("Start Identification", variant="primary", size="lg") | |
| gr.Markdown("---") | |
| gr.Markdown("### Database Spectrum Lookup") | |
| with gr.Column(elem_classes="query-section"): | |
| spec_query_name = gr.Textbox( | |
| label="Compound name / Formula", | |
| placeholder="Enter a compound name or formula", | |
| ) | |
| spec_query_type = gr.Dropdown(["ATR-FTIR", "Raman", "XRD", "FORS"], value="Raman", label="Spectrum Type") | |
| spec_query_btn = gr.Button("Lookup Spectrum", variant="secondary") | |
| spec_query_info = gr.Markdown() | |
| spec_query_choice = gr.Dropdown(choices=[], label="Candidate English Name", value=None, visible=False) | |
| spec_query_confirm = gr.Button("Confirm Selection & Plot", visible=False, variant="primary") | |
| spec_query_plot = gr.Plot(label="Database Spectrum") | |
| with gr.Column(scale=2): | |
| with gr.Column(elem_classes="result-section"): | |
| gr.Markdown("### Identification Results") | |
| report = gr.Markdown(latex_delimiters=LATEX_DELIMITERS) | |
| # Single spectrum identification section (shown when only 1 file) | |
| single_spectrum_section = gr.Column(visible=True) | |
| with single_spectrum_section: | |
| gr.Markdown("#### Single Spectrum Analysis") | |
| plot = gr.Plot(label="Spectral Comparison Plot", visible=True) | |
| # Multi-spectrum fusion section (shown when multiple files) | |
| fusion_section = gr.Column(visible=False) | |
| with fusion_section: | |
| gr.Markdown("#### Multimodal Fusion Analysis") | |
| fusion_plot = gr.Plot(label="Fusion Results & Ranking Matrix", visible=False) | |
| fusion_download = gr.DownloadButton( | |
| "Download Fusion Plot", | |
| visible=False, | |
| value=None, | |
| ) | |
| # Common results table | |
| table = gr.Dataframe(headers=["Rank", "Cosine similarity / Score (%)", "Name", "Formula", "Excitation(λ)", "Source"], | |
| value=pd.DataFrame(columns=["Rank", "Cosine similarity / Score (%)", "Name", "Formula", "Excitation(λ)", "Source"]), | |
| column_widths=[50,120,100,100,120,100], label="Top Matches", wrap=True) | |
| def _sync_choice_to_text(choice): | |
| return choice | |
| def upload_file(file_input, type_input, current_uploaded_files): | |
| """Upload a file""" | |
| # Default return: state, display, clear_all_btn, file_clear, 16 UI updates (4 files × 4 components) | |
| default_updates = [gr.update()] * 16 | |
| try: | |
| logging.info(f"Upload: Called with file_input type: {type(file_input)}, type_input: {type_input}") | |
| if not file_input: | |
| logging.info("Upload: No file input provided") | |
| # Return current state with all updates (20 total: state, display, clear_all_btn, file_clear, 16 UI) | |
| updates = [] | |
| for i in range(4): | |
| if i < len(current_uploaded_files): | |
| file_data = current_uploaded_files[i] | |
| updates.extend([ | |
| gr.update(value=file_data.get('file_name', ''), visible=True), | |
| gr.update(value=file_data.get('spectrum_type', ''), visible=True), | |
| gr.update(visible=True), | |
| gr.update(visible=True) # Row | |
| ]) | |
| else: | |
| updates.extend([gr.update(visible=False)] * 4) | |
| display_vis = gr.update(visible=len(current_uploaded_files) > 0) | |
| clear_btn_vis = gr.update(visible=len(current_uploaded_files) > 0) | |
| return (current_uploaded_files or [], display_vis, clear_btn_vis, gr.update(), *updates) | |
| current_uploaded_files = current_uploaded_files if current_uploaded_files else [] | |
| # Check if already at max (4 files) | |
| if len(current_uploaded_files) >= 4: | |
| logging.warning(f"Upload: Already at max (4 files), cannot add more") | |
| # Return current state with all updates | |
| updates = [] | |
| for i in range(4): | |
| if i < len(current_uploaded_files): | |
| file_data = current_uploaded_files[i] | |
| updates.extend([ | |
| gr.update(value=file_data.get('file_name', ''), visible=True), | |
| gr.update(value=file_data.get('spectrum_type', ''), visible=True), | |
| gr.update(visible=True), | |
| gr.update(visible=True) # Row | |
| ]) | |
| else: | |
| updates.extend([gr.update(visible=False)] * 4) | |
| display_vis = gr.update(visible=len(current_uploaded_files) > 0) | |
| clear_btn_vis = gr.update(visible=len(current_uploaded_files) > 0) | |
| return (current_uploaded_files, display_vis, clear_btn_vis, gr.update(), *updates) | |
| # Get file path - handle different Gradio file input formats | |
| file_path = None | |
| # Gradio File component can return different types | |
| # Try multiple ways to extract the file path | |
| if file_input is None: | |
| file_path = None | |
| elif isinstance(file_input, str): | |
| file_path = file_input | |
| elif isinstance(file_input, (list, tuple)): | |
| if len(file_input) > 0: | |
| first_item = file_input[0] | |
| if isinstance(first_item, str): | |
| file_path = first_item | |
| elif hasattr(first_item, 'name'): | |
| file_path = first_item.name | |
| else: | |
| file_path = str(first_item) | |
| else: | |
| file_path = None | |
| elif hasattr(file_input, 'name'): | |
| file_path = file_input.name | |
| elif hasattr(file_input, '__dict__'): | |
| # Try to get path from object attributes | |
| file_path = getattr(file_input, 'name', None) or getattr(file_input, 'path', None) or str(file_input) | |
| else: | |
| file_path = str(file_input) if file_input else None | |
| logging.info(f"Upload: Extracted file_path: {file_path} (type: {type(file_path)})") | |
| # Validate file path - check for various invalid values | |
| invalid_paths = ('None', '', 'null', '[]', 'NoneType', None) | |
| if not file_path or str(file_path).strip() in invalid_paths: | |
| logging.warning(f"Upload: Invalid or empty file path: {file_path}") | |
| # Return current state with all updates | |
| updates = [] | |
| for i in range(4): | |
| if i < len(current_uploaded_files): | |
| file_data = current_uploaded_files[i] | |
| updates.extend([ | |
| gr.update(value=file_data.get('file_name', ''), visible=True), | |
| gr.update(value=file_data.get('spectrum_type', ''), visible=True), | |
| gr.update(visible=True), | |
| gr.update(visible=True) # Row | |
| ]) | |
| else: | |
| updates.extend([gr.update(visible=False)] * 4) | |
| display_vis = gr.update(visible=len(current_uploaded_files) > 0) | |
| clear_btn_vis = gr.update(visible=len(current_uploaded_files) > 0) | |
| return (current_uploaded_files, display_vis, clear_btn_vis, gr.update(), *updates) | |
| # Check if file exists | |
| if not os.path.exists(file_path) or not os.path.isfile(file_path): | |
| logging.warning(f"Upload: File does not exist: {file_path}") | |
| # Return current state with all updates | |
| updates = [] | |
| for i in range(4): | |
| if i < len(current_uploaded_files): | |
| file_data = current_uploaded_files[i] | |
| updates.extend([ | |
| gr.update(value=file_data.get('file_name', ''), visible=True), | |
| gr.update(value=file_data.get('spectrum_type', ''), visible=True), | |
| gr.update(visible=True), | |
| gr.update(visible=True) # Row | |
| ]) | |
| else: | |
| updates.extend([gr.update(visible=False)] * 4) | |
| display_vis = gr.update(visible=len(current_uploaded_files) > 0) | |
| clear_btn_vis = gr.update(visible=len(current_uploaded_files) > 0) | |
| return (current_uploaded_files, display_vis, clear_btn_vis, gr.update(), *updates) | |
| # Check if file already uploaded | |
| if any(f.get('file_path') == file_path for f in current_uploaded_files): | |
| logging.info(f"Upload: File already uploaded: {file_path}") | |
| # Return current state with all updates | |
| updates = [] | |
| for i in range(4): | |
| if i < len(current_uploaded_files): | |
| file_data = current_uploaded_files[i] | |
| updates.extend([ | |
| gr.update(value=file_data.get('file_name', ''), visible=True), | |
| gr.update(value=file_data.get('spectrum_type', ''), visible=True), | |
| gr.update(visible=True), | |
| gr.update(visible=True) # Row | |
| ]) | |
| else: | |
| updates.extend([gr.update(visible=False)] * 4) | |
| display_vis = gr.update(visible=len(current_uploaded_files) > 0) | |
| clear_btn_vis = gr.update(visible=len(current_uploaded_files) > 0) | |
| return (current_uploaded_files, display_vis, clear_btn_vis, gr.update(), *updates) | |
| # Add uploaded file | |
| file_name = os.path.basename(file_path) | |
| spectrum_type = type_input if type_input else 'Raman' | |
| new_file_data = { | |
| 'file_path': file_path, | |
| 'file_name': file_name, | |
| 'spectrum_type': spectrum_type | |
| } | |
| current_uploaded_files.append(new_file_data) | |
| logging.info(f"Upload: Added file {file_name} ({spectrum_type}), total files: {len(current_uploaded_files)}") | |
| # Update uploaded files display (4 files × 4 components = 16 updates: name, type, remove, row) | |
| updates = [] | |
| # Always generate exactly 16 updates (4 files × 4 components each) | |
| # Show only files that exist, hide empty slots | |
| for i in range(4): | |
| if i < len(current_uploaded_files): | |
| file_data = current_uploaded_files[i] | |
| # Ensure we have the correct keys and values | |
| file_name = file_data.get('file_name', '') | |
| spectrum_type = file_data.get('spectrum_type', '') | |
| # Show file info, remove button, and row | |
| updates.extend([ | |
| gr.update(value=file_name, visible=True), # Show filename | |
| gr.update(value=spectrum_type, visible=True), # Show type | |
| gr.update(visible=True), # Show remove button | |
| gr.update(visible=True) # Show row container | |
| ]) | |
| else: | |
| # Hide empty slots completely | |
| updates.extend([ | |
| gr.update(value="", visible=False), # Hide filename | |
| gr.update(value="", visible=False), # Hide type | |
| gr.update(visible=False), # Hide remove button | |
| gr.update(visible=False) # Hide row container | |
| ]) | |
| # Ensure we have exactly 16 updates (safety check) | |
| if len(updates) != 16: | |
| updates = updates[:16] if len(updates) > 16 else updates + [gr.update()] * (16 - len(updates)) | |
| # Show uploaded files display if files exist | |
| display_update = gr.update(visible=len(current_uploaded_files) > 0) | |
| # Show clear all button if files exist | |
| clear_all_btn_update = gr.update(visible=len(current_uploaded_files) > 0) | |
| # Clear the file input after upload | |
| file_clear_update = gr.update(value=None) | |
| logging.info(f"Upload: Successfully uploaded {file_name}, returning updates") | |
| # Return: state, display, clear_all_btn, file_clear, 16 UI updates = 20 total | |
| return (current_uploaded_files, display_update, clear_all_btn_update, file_clear_update, *updates) | |
| except Exception as e: | |
| logging.error(f"Upload: Error occurred: {e}", exc_info=True) | |
| return (current_uploaded_files or [], gr.update(), gr.update(), gr.update(), *default_updates) | |
| def remove_uploaded_file(index, current_uploaded_files): | |
| """Remove an uploaded file""" | |
| # Default return: state, display, clear_all_btn, 16 UI components (4 files × 4 components) | |
| default_updates = [gr.update()] * 16 | |
| if not current_uploaded_files or index >= len(current_uploaded_files): | |
| return (current_uploaded_files or [], gr.update(), gr.update(), *default_updates) | |
| # Create a copy and remove the file at the specified index | |
| new_files = current_uploaded_files.copy() | |
| removed_file = new_files.pop(index) | |
| logging.info(f"Removed file at index {index}: {removed_file.get('file_name', 'unknown')}") | |
| logging.info(f"Remaining files: {len(new_files)}") | |
| # Update display - always generate exactly 16 updates (4 files × 4 components) | |
| # Only show slots that have files, completely hide empty slots | |
| updates = [] | |
| for i in range(4): | |
| if i < len(new_files): | |
| file_data = new_files[i] | |
| # Ensure we have the correct keys | |
| file_name = file_data.get('file_name', '') | |
| spectrum_type = file_data.get('spectrum_type', '') | |
| # Show file info, remove button, and row for files that exist | |
| updates.extend([ | |
| gr.update(value=file_name, visible=True), # Show filename | |
| gr.update(value=spectrum_type, visible=True), # Show type | |
| gr.update(visible=True), # Show remove button | |
| gr.update(visible=True) # Show row container | |
| ]) | |
| else: | |
| # Completely hide empty slots and clear their values | |
| updates.extend([ | |
| gr.update(value="", visible=False), # Hide filename, clear value | |
| gr.update(value="", visible=False), # Hide type, clear value | |
| gr.update(visible=False), # Hide remove button | |
| gr.update(visible=False) # Hide row container | |
| ]) | |
| # Ensure we have exactly 16 updates | |
| if len(updates) != 16: | |
| updates = updates[:16] if len(updates) > 16 else updates + [gr.update()] * (16 - len(updates)) | |
| # Show display container only if there are files remaining | |
| display_update = gr.update(visible=len(new_files) > 0) | |
| # Show clear all button only if there are files remaining | |
| clear_all_btn_update = gr.update(visible=len(new_files) > 0) | |
| logging.info(f"Returning {len(new_files)} files, display visible: {len(new_files) > 0}") | |
| # Return: state, display, clear_all_btn, 16 UI updates (name, type, remove, row for 4 files) | |
| return (new_files, display_update, clear_all_btn_update, *updates) | |
| def process_identification_with_types(files_state, enable_fusion): | |
| """Process identification with files and their types""" | |
| if not files_state or len(files_state) == 0: | |
| return ( | |
| "Please upload at least one spectrum file", | |
| gr.update(visible=True), | |
| gr.update(visible=False), | |
| gr.update(visible=False), | |
| gr.update(visible=False), | |
| [], | |
| gr.update(visible=False, value=None), | |
| ) | |
| # Prepare files and types for identification | |
| file_paths = [f['file_path'] for f in files_state] | |
| spectrum_types = [f['spectrum_type'] for f in files_state] | |
| # If single file, use single file logic | |
| if len(file_paths) == 1: | |
| report_text, single_plot, table_rows, fusion_fig = identify_spectrum( | |
| file_paths[0], spectrum_types[0], enable_fusion=enable_fusion | |
| ) | |
| if fusion_fig is not None: | |
| download_path = _save_matplotlib_figure(fusion_fig) | |
| return (report_text, | |
| gr.update(visible=False), # Hide single spectrum section | |
| gr.update(visible=True), # Show single plot (hidden) | |
| gr.update(visible=True), # Show fusion section | |
| gr.update(visible=True, value=fusion_fig), # Show fusion plot | |
| table_rows, | |
| gr.update(visible=bool(download_path), value=download_path)) | |
| else: | |
| return (report_text, | |
| gr.update(visible=True), # Show single spectrum section | |
| gr.update(visible=True, value=single_plot) if single_plot else gr.update(visible=False), # Show single plot | |
| gr.update(visible=False), # Hide fusion section | |
| gr.update(visible=False), # Hide fusion plot | |
| table_rows, | |
| gr.update(visible=False, value=None)) | |
| # Multiple files: batch identification with fusion | |
| result = identify_multiple_spectra_with_types(file_paths, spectrum_types, enable_fusion) | |
| # Ensure fusion_fig is displayable (matplotlib figure for gr.Plot) | |
| if len(result) >= 4: | |
| report_text, single_plot, table_rows, fusion_fig = result | |
| # Accept matplotlib figure; gr.Plot will render it | |
| if fusion_fig is not None: | |
| # Reject only clearly invalid types (e.g. str); matplotlib fig has savefig | |
| if not hasattr(fusion_fig, 'savefig') and not (hasattr(fusion_fig, '__class__') and 'matplotlib' in str(type(fusion_fig))): | |
| logging.warning(f"Unsupported fusion_fig type: {type(fusion_fig)}, setting to None") | |
| fusion_fig = None | |
| if fusion_fig is not None: | |
| download_path = _save_matplotlib_figure(fusion_fig) | |
| return (report_text, | |
| gr.update(visible=False), # Hide single spectrum section | |
| gr.update(visible=False), # Hide single plot | |
| gr.update(visible=True), # Show fusion section | |
| gr.update(visible=True, value=fusion_fig), # Show fusion plot | |
| table_rows, | |
| gr.update(visible=bool(download_path), value=download_path)) | |
| else: | |
| return (report_text, | |
| gr.update(visible=False), # Hide single spectrum section | |
| gr.update(visible=False), # Hide single plot | |
| gr.update(visible=True), # Show fusion section | |
| gr.update(visible=False), # Hide fusion plot | |
| table_rows, | |
| gr.update(visible=False, value=None)) | |
| # len(result) < 4: still return 6 outputs so report/table/sections update | |
| report_text = result[0] if len(result) > 0 else "Identification failed." | |
| single_plot = result[1] if len(result) > 1 else None | |
| table_rows = result[2] if len(result) > 2 else [] | |
| return (report_text, | |
| gr.update(visible=True), | |
| gr.update(visible=True, value=single_plot) if single_plot else gr.update(visible=False), | |
| gr.update(visible=False), | |
| gr.update(visible=False), | |
| table_rows, | |
| gr.update(visible=False, value=None)) | |
| # Upload button handler | |
| def handle_upload(file_input, type_input, current_files): | |
| result = upload_file(file_input, type_input, current_files) | |
| return result | |
| # Outputs: state, display, clear_all_btn, file_clear, 16 UI components (4 files × 4 components) | |
| output_list = [ | |
| uploaded_files_state, # 1. State | |
| uploaded_files_display, # 2. Display container | |
| btn_clear_all, # 3. Clear all button | |
| file_input # 4. File input (to clear) | |
| ] | |
| # Add 16 UI components (4 files × 4 components each: name, type, remove, row) | |
| for item in uploaded_files_list: | |
| output_list.extend([item['name'], item['type'], item['remove'], item['row']]) | |
| upload_btn.click( | |
| handle_upload, | |
| [file_input, type_input, uploaded_files_state], | |
| output_list | |
| ) | |
| # Remove uploaded file handlers | |
| def make_remove_uploaded_handler(idx): | |
| def handler(current_files): | |
| result = remove_uploaded_file(idx, current_files) | |
| return result | |
| return handler | |
| for i, item in enumerate(uploaded_files_list): | |
| # Outputs: state, display, clear_all_btn, 16 UI components (4 files × 4 components: name, type, remove, row) | |
| remove_outputs = [ | |
| uploaded_files_state, # State | |
| uploaded_files_display, # Display container | |
| btn_clear_all # Clear all button | |
| ] | |
| # Add 16 UI components (4 files × 4 components each) | |
| for file_item in uploaded_files_list: | |
| remove_outputs.extend([file_item['name'], file_item['type'], file_item['remove'], file_item['row']]) | |
| item['remove'].click( | |
| make_remove_uploaded_handler(i), | |
| [uploaded_files_state], | |
| remove_outputs | |
| ) | |
| # Clear all files handler | |
| def clear_all_files(current_files): | |
| """Clear all uploaded files""" | |
| logging.info("Clearing all uploaded files") | |
| # Return empty state and hide all UI elements | |
| default_updates = [] | |
| for i in range(4): | |
| default_updates.extend([ | |
| gr.update(value="", visible=False), # Hide filename | |
| gr.update(value="", visible=False), # Hide type | |
| gr.update(visible=False), # Hide remove button | |
| gr.update(visible=False) # Hide row container | |
| ]) | |
| return ( | |
| [], # Empty state | |
| gr.update(visible=False), # Hide display container | |
| gr.update(visible=False), # Hide clear all button | |
| *default_updates | |
| ) | |
| btn_clear_all.click( | |
| clear_all_files, | |
| [uploaded_files_state], | |
| [uploaded_files_state, uploaded_files_display, btn_clear_all] + | |
| [item['name'] for item in uploaded_files_list] + | |
| [item['type'] for item in uploaded_files_list] + | |
| [item['remove'] for item in uploaded_files_list] + | |
| [item['row'] for item in uploaded_files_list] | |
| ) | |
| btn_id.click( | |
| process_identification_with_types, | |
| [uploaded_files_state, enable_fusion], | |
| [report, single_spectrum_section, plot, fusion_section, fusion_plot, table, fusion_download] | |
| ) | |
| spec_query_btn.click(show_compound_spectrum, [spec_query_name, spec_query_type], | |
| [spec_query_plot, spec_query_info, spec_query_choice, spec_query_confirm]) | |
| def _confirm_and_plot(name, stype): | |
| logging.info(f"_confirm_and_plot called with name={name!r}, stype={stype!r}") | |
| try: | |
| if isinstance(name, str) and ' (' in name and name.strip().endswith(')'): | |
| base = name.split(' (', 1)[0].strip() | |
| logging.info(f"Sanitized selection from {name!r} to {base!r}") | |
| name = base | |
| except Exception: | |
| pass | |
| out = show_compound_spectrum(name, stype) | |
| logging.info(f"show_compound_spectrum returned: type={type(out)}, repr={repr(out)[:200]}") | |
| if not isinstance(out, tuple): | |
| logging.warning("show_compound_spectrum did not return tuple") | |
| return None, "Query failed", gr.update(choices=[], value=None, visible=False), gr.update(visible=False) | |
| try: | |
| fig, info, choice_update, confirm_update = out | |
| except Exception as e: | |
| logging.exception(f"Unexpected output shape from show_compound_spectrum: {e}") | |
| return None, "Query returned unexpected shape", gr.update(choices=[], value=None, visible=False), gr.update(visible=False) | |
| logging.info("_confirm_and_plot returning figure to UI and hiding confirm button") | |
| return fig, info, choice_update, gr.update(visible=False) | |
| spec_query_confirm.click(_confirm_and_plot, [spec_query_choice, spec_query_type], | |
| [spec_query_plot, spec_query_info, spec_query_choice, spec_query_confirm]) | |
| spec_query_choice.change(_sync_choice_to_text, spec_query_choice, spec_query_name) | |
| with gr.Tab("Pathway Tracing", id=2): | |
| with gr.Column(elem_id="page"): | |
| state_reports = gr.State([]) | |
| with gr.Row(): | |
| with gr.Column(scale=1): | |
| gr.Markdown("### Detected Products / Path Constraints") | |
| material = gr.Textbox( | |
| label="Enter one or more formulas (path order)", | |
| placeholder="e.g., p-As4S4, As2O3", | |
| ) | |
| gr.Markdown("Separate formulas with commas. Every formula is required in the displayed path; the final formula is the endpoint.") | |
| btn_tr = gr.Button("Trace Initial Materials & Paths", variant="primary", size="lg") | |
| gr.Markdown("---") | |
| gr.Markdown("### Path Navigation") | |
| path_selector = gr.Dropdown(choices=[], label="Select path (condition)", value=None, allow_custom_value=True) | |
| gr.Markdown(""" | |
| **Navigation Tips:** | |
| - Use Previous/Next buttons to browse | |
| - Or select a specific path number | |
| - Red nodes = Starting materials | |
| - Blue nodes = Intermediate products | |
| """) | |
| with gr.Column(scale=2): | |
| with gr.Column(elem_classes="result-section"): | |
| gr.Markdown("### Reaction Network") | |
| img2 = gr.Image(type="filepath", label="Path Visualization") | |
| rep2 = gr.Markdown(latex_delimiters=LATEX_DELIMITERS) | |
| with gr.Row(): | |
| prev_btn = gr.Button("Previous", elem_id="prev_btn") | |
| gr.HTML("<div style='flex-grow: 1;'></div>") | |
| next_btn = gr.Button("Next", elem_id="next_btn") | |
| gr.HTML("<div style='font-weight: 600; margin-top: 6px;'>All Reaction Pathways</div>") | |
| include_composite = gr.Checkbox( | |
| label="Show complex paths", | |
| value=False, | |
| ) | |
| tab2 = gr.Dataframe(headers=["No.", "Initial Material", "Path Endpoint", "Steps", "Reaction Conditions", "Full Path"], | |
| value=pd.DataFrame(columns=["No.", "Initial Material", "Path Endpoint", "Steps", "Reaction Conditions", "Full Path"]), | |
| datatype=["number", "markdown", "markdown", "number", "str", "markdown"], | |
| show_label=False, wrap=True, height=360, interactive=False) | |
| with gr.Column(elem_classes="result-section"): | |
| gr.Markdown("### Evidence-grounded Path Analysis") | |
| out_rag_report = gr.Markdown( | |
| value="Analysis report will appear here...", | |
| elem_classes="report-box", | |
| line_breaks=True, | |
| latex_delimiters=[ | |
| {"left": "$$", "right": "$$", "display": True}, | |
| {"left": "$", "right": "$", "display": False}, | |
| ], | |
| ) | |
| rag_detail_download = gr.DownloadButton( | |
| "Download detailed RAG report (PDF)", | |
| visible=False, | |
| value=None, | |
| ) | |
| gr.Markdown("---") | |
| gr.Markdown(""" | |
| ### Full Network View | |
| Display the complete Chemical Reaction Network visualization generated from Cytoscape. | |
| """) | |
| btn_show = gr.Button("Show Chemical Reaction Networks", variant="secondary", size="lg") | |
| cyto_img = gr.Image(type="pil", label="Complete Network Visualization", height=600) | |
| btn_show.click(show_cytoscape_image, [], [cyto_img]) | |
| prev_btn.click(prev_path, [material, path_selector, state_reports, include_composite, copilot_state], | |
| [img2, rep2, path_selector, out_rag_report, rag_detail_download, copilot_state]) | |
| next_btn.click(next_path, [material, path_selector, state_reports, include_composite, copilot_state], | |
| [img2, rep2, path_selector, out_rag_report, rag_detail_download, copilot_state]) | |
| path_selector.change(select_path, [material, path_selector, state_reports, include_composite, copilot_state], | |
| [img2, rep2, out_rag_report, rag_detail_download, copilot_state]) | |
| with gr.Tab("Mural Copilot", id=3): | |
| with gr.Column(elem_id="page"): | |
| with gr.Column(elem_id="copilot_panel"): | |
| gr.HTML(""" | |
| <div style="background: linear-gradient(135deg, #667eea 0%, #764ba2 100%); border-radius: 12px; padding: 20px; margin-bottom: 20px; color: #fdf9f3; border: 1px solid rgba(255,255,255,0.25);"> | |
| <h3 style="margin-top: 0; font-weight: 600;">Mural Copilot - RAG Assistant</h3> | |
| <p style="margin: 10px 0;"><strong>Ask about spectral analysis, degradation pathways, hypotheses, evidence, and conservation strategies.</strong></p> | |
| <p style="font-size: 13px; margin-top: 10px; opacity: 0.9;">Evidence-linked responses with claim-level traceability</p> | |
| </div> | |
| """) | |
| copilot_path_select = gr.Dropdown( | |
| choices=[], | |
| label="Use traced path (condition)", | |
| value=None, | |
| ) | |
| copilot_material = gr.Dropdown( | |
| choices=[], | |
| label="Current material (select or type)", | |
| value=None, | |
| allow_custom_value=True, | |
| ) | |
| copilot_chat = gr.Chatbot( | |
| label="Chat with Mural Copilot", | |
| scale=1, | |
| height=500, | |
| show_label=False, | |
| type="messages", | |
| elem_id="copilot_chat", | |
| ) | |
| with gr.Row(): | |
| copilot_input = gr.Textbox( | |
| placeholder="Ask anything about spectral data, reactions, hypotheses, evidence, or conservation...", | |
| lines=2, | |
| scale=10, | |
| show_label=False, | |
| ) | |
| with gr.Column(scale=1): | |
| copilot_submit = gr.Button(value="Send", size="lg") | |
| copilot_clear_btn = gr.Button(value="Clear Chat", size="sm", variant="secondary") | |
| def copilot_chat_handler(user_query, chat_history, context): | |
| context = dict(context or {}) | |
| chat_history = list(chat_history or []) | |
| if not user_query or not user_query.strip(): | |
| gr.Info("Please enter a question") | |
| yield chat_history, "", context | |
| return | |
| def finish_with_message(message): | |
| history = list(chat_history) | |
| history.append({"role": "user", "content": user_query}) | |
| history.append({"role": "assistant", "content": message}) | |
| return history, "", context | |
| if not is_supported_english_query(user_query): | |
| gr.Warning(ENGLISH_ONLY_MESSAGE) | |
| yield finish_with_message(ENGLISH_ONLY_MESSAGE) | |
| return | |
| # Retrieval + evidence-linked generation typically takes 30-60s. | |
| # Stream the user's message and a progress placeholder immediately | |
| # (and clear the input) so the wait is visibly acknowledged rather | |
| # than looking like a frozen UI. | |
| pending = list(chat_history) | |
| pending.append({"role": "user", "content": user_query}) | |
| pending.append({ | |
| "role": "assistant", | |
| "content": ( | |
| "🔍 Retrieving literature and generating an evidence-linked " | |
| "answer… this usually takes 30–60 seconds." | |
| ), | |
| }) | |
| yield pending, "", context | |
| if rag_service is None: | |
| if _rag_init_error: | |
| logging.error( | |
| "Mural Copilot unavailable because RAG initialization failed: %s", | |
| _rag_init_error, | |
| ) | |
| yield finish_with_message( | |
| "Mural Copilot is temporarily unavailable because its evidence " | |
| "service did not initialize. Spectral identification and pathway " | |
| "tracing remain available." | |
| ) | |
| return | |
| yield finish_with_message( | |
| "Mural Copilot is still starting. Please wait briefly and try again." | |
| ) | |
| return | |
| if getattr(rag_service, "llm_available", True) is False: | |
| yield finish_with_message( | |
| "Mural Copilot's language-model backend is temporarily unavailable. " | |
| "Spectral identification and pathway tracing remain available." | |
| ) | |
| return | |
| analysis_context = { | |
| "current_spectrum": context.get("current_spectrum"), | |
| "spectrum_name": context.get("spectrum_name", ""), | |
| "spectrum_type": context.get("spectrum_type", ""), | |
| "current_pathways": context.get("current_pathways", []), | |
| "current_materials": context.get("current_materials", []), | |
| "current_material": context.get("current_material", ""), | |
| "identified_compounds": context.get("identified_compounds", []), | |
| "spectrum_matches": context.get("spectrum_matches", []), | |
| "last_rag_report": context.get("last_rag_report", ""), | |
| "last_rag_path": context.get("last_rag_path", ""), | |
| "last_rag_material": context.get("last_rag_material", ""), | |
| } | |
| copilot = MuralCopilot( | |
| rag_service, | |
| crn_tracer=_crn_tracer, | |
| spectral_system=None, | |
| ) | |
| copilot.conversation_history = [ | |
| dict(message) | |
| for message in (chat_history or []) | |
| if isinstance(message, dict) | |
| ] | |
| copilot.evidence_map = dict( | |
| context.get("copilot_evidence_map") or {} | |
| ) | |
| try: | |
| response = copilot.process_query(user_query, analysis_context) | |
| response = _normalize_rag_text(response) | |
| except Exception as exc: | |
| logging.exception("Mural Copilot request failed") | |
| message = ( | |
| "Mural Copilot could not complete this request. Please retry, or use " | |
| "the spectral-identification and pathway-tracing modules directly." | |
| ) | |
| # Setting COPILOT_DEBUG=1 in the Space variables surfaces the | |
| # underlying error class/message in the chat so the failure can | |
| # be diagnosed without opening the runtime logs. Off by default | |
| # so end users never see a stack-trace fragment. | |
| if os.getenv("COPILOT_DEBUG", "0").strip().lower() in {"1", "true", "yes", "on"}: | |
| message += f"\n\n`[debug] {type(exc).__name__}: {str(exc)[:300]}`" | |
| yield finish_with_message(message) | |
| return | |
| final_history = list(chat_history) | |
| final_history.append({"role": "user", "content": user_query}) | |
| final_history.append({"role": "assistant", "content": response}) | |
| state_updates = { | |
| "copilot_evidence_map": _sanitize_for_state(copilot.evidence_map), | |
| "copilot_sentence_traces": _sanitize_for_state( | |
| copilot.get_sentence_traces() | |
| ), | |
| } | |
| yield final_history, "", _merge_copilot_state(context, state_updates) | |
| def copilot_clear_handler(current_state): | |
| updates = { | |
| "copilot_evidence_map": {}, | |
| "copilot_sentence_traces": [], | |
| } | |
| return [], "", _merge_copilot_state(current_state, updates) | |
| def copilot_set_material(material_text, current_state): | |
| updates = { | |
| "current_material": (material_text or "").strip(), | |
| } | |
| return _merge_copilot_state(current_state, updates) | |
| def copilot_apply_path(path_choice, rag_reports, current_state): | |
| if not path_choice: | |
| return current_state | |
| try: | |
| idx = int(str(path_choice).split(',')[0]) - 1 | |
| except Exception: | |
| return current_state | |
| raw_report = _report_to_text(rag_reports[idx]) if rag_reports and idx < len(rag_reports) else "" | |
| path_list = current_state.get("current_pathways", []) or [] | |
| path_str = path_list[idx] if idx < len(path_list) else "" | |
| updates = { | |
| "last_rag_report": raw_report, | |
| "last_rag_path": path_str, | |
| "last_rag_material": current_state.get("last_rag_material", ""), | |
| } | |
| return _merge_copilot_state(current_state, updates) | |
| copilot_submit.click( | |
| fn=copilot_chat_handler, | |
| inputs=[copilot_input, copilot_chat, copilot_state], | |
| outputs=[copilot_chat, copilot_input, copilot_state], | |
| queue=True, | |
| api_name=False, | |
| ) | |
| copilot_material.change( | |
| fn=copilot_set_material, | |
| inputs=[copilot_material, copilot_state], | |
| outputs=[copilot_state], | |
| queue=False, | |
| api_name=False, | |
| ) | |
| copilot_clear_btn.click( | |
| fn=copilot_clear_handler, | |
| inputs=[copilot_state], | |
| outputs=[copilot_chat, copilot_input, copilot_state], | |
| queue=False, | |
| api_name=False, | |
| ) | |
| copilot_path_select.change( | |
| fn=copilot_apply_path, | |
| inputs=[copilot_path_select, state_reports, copilot_state], | |
| outputs=[copilot_state], | |
| queue=False, | |
| api_name=False, | |
| ) | |
| # Pathway tracing is a multi-minute job. Disable the button for | |
| # the duration so an impatient second click cannot queue a | |
| # duplicate run behind the first. The trailing .then re-enables | |
| # it and runs even if tracing fails, so the button never sticks. | |
| btn_tr.click( | |
| lambda: gr.update(interactive=False), | |
| None, | |
| btn_tr, | |
| queue=False, | |
| ).then( | |
| trace_pathways, | |
| [material, include_composite, copilot_state], | |
| [rep2, img2, tab2, path_selector, state_reports, out_rag_report, rag_detail_download, copilot_state, copilot_path_select, copilot_material], | |
| api_name="trace_pathways", | |
| ).then( | |
| lambda: gr.update(interactive=True), | |
| None, | |
| btn_tr, | |
| queue=False, | |
| ) | |
| include_composite.change( | |
| trace_pathways, | |
| [material, include_composite, copilot_state], | |
| [rep2, img2, tab2, path_selector, state_reports, out_rag_report, rag_detail_download, copilot_state, copilot_path_select, copilot_material], | |
| api_name=False, | |
| ) | |
| try: | |
| blocks = demo.get_blocks() | |
| blocks.get_api_info = lambda: {} | |
| except Exception: | |
| pass | |
| def launch() -> None: | |
| print("[BOOT 4/4] Building startup status and launching Gradio", flush=True) | |
| print("\n" + "=" * 70) | |
| print("TRACER-Net - Startup Status") | |
| print("=" * 70) | |
| print(f"Spectral Module: {'Available' if SPECTRAL_AVAILABLE else 'Not available'}") | |
| print(f"CRN Module: {'Available' if CRN_AVAILABLE else 'Not available'}") | |
| if CRN_AVAILABLE and G is not None: | |
| print(f" - Network nodes: {G.number_of_nodes()}") | |
| print(f" - Network edges: {G.number_of_edges()}") | |
| source_count = len( | |
| getattr(_crn_tracer, "initial_reactant_species", []) or [] | |
| ) | |
| print(f" - Initial reactant species: {source_count}") | |
| cyto_path = find_cytoscape_image() | |
| print( | |
| f"Cytoscape Image: Found at {cyto_path}" | |
| if cyto_path | |
| else "Cytoscape Image: Not found" | |
| ) | |
| print("=" * 70 + "\n") | |
| demo.queue(api_open=False) | |
| demo.launch( | |
| server_name=os.getenv("HOST", "0.0.0.0"), | |
| server_port=int(os.getenv("PORT", "7860")), | |
| share=False, | |
| prevent_thread_lock=True, | |
| ) | |
| print( | |
| "[HTTP] Gradio is listening on " | |
| f"{os.getenv('HOST', '0.0.0.0')}:{os.getenv('PORT', '7860')}.", | |
| flush=True, | |
| ) | |
| _initialize_rag_service() | |
| threading.Event().wait() | |
| if __name__ == "__main__": | |
| launch() | |