Spaces:
Sleeping
Sleeping
Create self_learning_graph.py
Browse files- dvnc_ai_v2_hf/self_learning_graph.py +1729 -0
dvnc_ai_v2_hf/self_learning_graph.py
ADDED
|
@@ -0,0 +1,1729 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import html
|
| 2 |
+
import json
|
| 3 |
+
import re
|
| 4 |
+
import time
|
| 5 |
+
import urllib.parse
|
| 6 |
+
import xml.etree.ElementTree as ET
|
| 7 |
+
from collections import Counter
|
| 8 |
+
from pathlib import Path
|
| 9 |
+
from typing import Any, Dict, List, Optional
|
| 10 |
+
|
| 11 |
+
import gradio as gr
|
| 12 |
+
import requests
|
| 13 |
+
|
| 14 |
+
try:
|
| 15 |
+
import fitz # PyMuPDF
|
| 16 |
+
except Exception:
|
| 17 |
+
fitz = None
|
| 18 |
+
|
| 19 |
+
|
| 20 |
+
SEARCH_MODES = ["topic", "title", "doi", "link", "paper_name", "autonomous_web"]
|
| 21 |
+
SOURCE_OPTIONS = ["arxiv", "crossref", "openalex", "semantic_scholar", "europe_pmc"]
|
| 22 |
+
DEFAULT_SOURCES = ["arxiv", "crossref", "openalex", "semantic_scholar", "europe_pmc"]
|
| 23 |
+
PDF_PARSERS = ["pymupdf"]
|
| 24 |
+
|
| 25 |
+
REQUEST_TIMEOUT = 25
|
| 26 |
+
GRAPH_MAX_RESULTS = 12
|
| 27 |
+
GRAPH_MAX_CONCEPTS = 12
|
| 28 |
+
GRAPH_MAX_CLAIMS = 8
|
| 29 |
+
GRAPH_MAX_EXPANSIONS = 6
|
| 30 |
+
GRAPH_MAX_NODES = 500
|
| 31 |
+
GRAPH_MAX_EDGES = 1600
|
| 32 |
+
GRAPH_IFRAME_HEIGHT = 760
|
| 33 |
+
MAX_ABSTRACT_CHARS = 4000
|
| 34 |
+
MAX_RAW_TEXT_CHARS = 60000
|
| 35 |
+
|
| 36 |
+
JOURNALS = [
|
| 37 |
+
{"name": "Nature", "url": "https://www.nature.com/search", "desc": "Flagship multidisciplinary research journal."},
|
| 38 |
+
{"name": "Science", "url": "https://www.science.org/search", "desc": "High-impact science journal and family."},
|
| 39 |
+
{"name": "Cell", "url": "https://www.cell.com/search", "desc": "Life sciences and translational biology."},
|
| 40 |
+
{"name": "The Lancet", "url": "https://www.thelancet.com/search", "desc": "Clinical and medical research."},
|
| 41 |
+
{"name": "IEEE Xplore", "url": "https://ieeexplore.ieee.org/search/searchresult.jsp", "desc": "Engineering, AI, signal processing, and systems."},
|
| 42 |
+
]
|
| 43 |
+
|
| 44 |
+
STOPWORDS = {
|
| 45 |
+
"a", "an", "and", "are", "as", "at", "be", "been", "being", "by", "can", "could",
|
| 46 |
+
"did", "do", "does", "for", "from", "had", "has", "have", "if", "in", "into", "is",
|
| 47 |
+
"it", "its", "may", "might", "of", "on", "or", "our", "such", "that", "the", "their",
|
| 48 |
+
"there", "these", "this", "those", "to", "using", "use", "used", "via", "was", "were",
|
| 49 |
+
"will", "with", "within", "without", "we", "they", "you", "your", "study", "paper",
|
| 50 |
+
"research", "results", "result", "method", "methods", "analysis", "approach", "based",
|
| 51 |
+
"new", "novel", "effect", "effects", "model", "models", "system", "systems", "show",
|
| 52 |
+
"shows", "shown", "however", "therefore", "also", "between", "across", "among", "et", "al",
|
| 53 |
+
"introduction", "discussion", "conclusion", "references", "figure", "table",
|
| 54 |
+
}
|
| 55 |
+
|
| 56 |
+
GRAPH_MEMORY: Dict[str, Any] = {
|
| 57 |
+
"papers": {},
|
| 58 |
+
"nodes": {},
|
| 59 |
+
"edges": [],
|
| 60 |
+
"queries": [],
|
| 61 |
+
"events": [],
|
| 62 |
+
"frontier": [],
|
| 63 |
+
"payloads": [],
|
| 64 |
+
"concept_counts": Counter(),
|
| 65 |
+
"claim_counts": Counter(),
|
| 66 |
+
}
|
| 67 |
+
|
| 68 |
+
|
| 69 |
+
# ----------------------------
|
| 70 |
+
# Utility
|
| 71 |
+
# ----------------------------
|
| 72 |
+
|
| 73 |
+
def safe_text(x, default=""):
|
| 74 |
+
return html.escape(str(x if x is not None else default))
|
| 75 |
+
|
| 76 |
+
|
| 77 |
+
def norm_text(x: Optional[str]) -> str:
|
| 78 |
+
return re.sub(r"\s+", " ", (x or "")).strip()
|
| 79 |
+
|
| 80 |
+
|
| 81 |
+
def ensure_list(x):
|
| 82 |
+
return x if isinstance(x, list) else []
|
| 83 |
+
|
| 84 |
+
|
| 85 |
+
def truncate_text(text: str, limit: int) -> str:
|
| 86 |
+
text = norm_text(text)
|
| 87 |
+
return text if len(text) <= limit else text[: limit - 1].rstrip() + "…"
|
| 88 |
+
|
| 89 |
+
|
| 90 |
+
def slugify(text: str) -> str:
|
| 91 |
+
return re.sub(r"[^a-z0-9]+", "-", (text or "").lower()).strip("-")
|
| 92 |
+
|
| 93 |
+
|
| 94 |
+
def normalize_doi(text: str) -> str:
|
| 95 |
+
text = (text or "").strip()
|
| 96 |
+
text = re.sub(r"^https?://(dx\.)?doi\.org/", "", text, flags=re.I)
|
| 97 |
+
return text.strip().rstrip("/")
|
| 98 |
+
|
| 99 |
+
|
| 100 |
+
def detect_query_type(query: str) -> str:
|
| 101 |
+
q = (query or "").strip()
|
| 102 |
+
doi_pattern = r"^10\.\d{4,9}/[-._;()/:A-Z0-9]+$"
|
| 103 |
+
if re.match(doi_pattern, q, flags=re.I):
|
| 104 |
+
return "doi"
|
| 105 |
+
if q.startswith("http://") or q.startswith("https://"):
|
| 106 |
+
return "link"
|
| 107 |
+
return "topic"
|
| 108 |
+
|
| 109 |
+
|
| 110 |
+
def tokenize(text: str) -> List[str]:
|
| 111 |
+
return [
|
| 112 |
+
t for t in re.findall(r"[A-Za-z][A-Za-z0-9\-/+]{2,}", (text or ""))
|
| 113 |
+
if t.lower() not in STOPWORDS
|
| 114 |
+
]
|
| 115 |
+
|
| 116 |
+
|
| 117 |
+
def text_overlap_score(a: str, b: str) -> float:
|
| 118 |
+
sa = {x.lower() for x in tokenize(a)}
|
| 119 |
+
sb = {x.lower() for x in tokenize(b)}
|
| 120 |
+
if not sa or not sb:
|
| 121 |
+
return 0.0
|
| 122 |
+
return len(sa & sb) / max(1, len(sa | sb))
|
| 123 |
+
|
| 124 |
+
|
| 125 |
+
def compute_recency_bonus(year: str) -> float:
|
| 126 |
+
try:
|
| 127 |
+
y = int(str(year)[:4])
|
| 128 |
+
except Exception:
|
| 129 |
+
return 0.0
|
| 130 |
+
current = time.gmtime().tm_year
|
| 131 |
+
age = max(current - y, 0)
|
| 132 |
+
return max(0.0, 0.16 - age * 0.018)
|
| 133 |
+
|
| 134 |
+
|
| 135 |
+
def clean_extracted_text(text: str) -> str:
|
| 136 |
+
text = text or ""
|
| 137 |
+
replacements = {
|
| 138 |
+
"\u00ad": "",
|
| 139 |
+
"\ufb01": "fi",
|
| 140 |
+
"\ufb02": "fl",
|
| 141 |
+
"\u2010": "-",
|
| 142 |
+
"\u2011": "-",
|
| 143 |
+
"\u2012": "-",
|
| 144 |
+
"\u2013": "-",
|
| 145 |
+
"\u2014": "-",
|
| 146 |
+
"\u2212": "-",
|
| 147 |
+
"\u00a0": " ",
|
| 148 |
+
}
|
| 149 |
+
for old, new in replacements.items():
|
| 150 |
+
text = text.replace(old, new)
|
| 151 |
+
|
| 152 |
+
text = re.sub(r"(\w)-\s*\n\s*(\w)", r"\1\2", text)
|
| 153 |
+
text = re.sub(r"([a-z])\n([a-z])", r"\1 \2", text)
|
| 154 |
+
text = re.sub(r"\bnjectable\b", "injectable", text, flags=re.I)
|
| 155 |
+
text = re.sub(r"\bfhydrogel\b", "hydrogel", text, flags=re.I)
|
| 156 |
+
text = re.sub(
|
| 157 |
+
r"(?i)\b([a-z])?(hydrogel|conductive|responsive|injectable|biomaterial|scaffold|cardiac|patch|repair)\b",
|
| 158 |
+
lambda m: m.group(2),
|
| 159 |
+
text,
|
| 160 |
+
)
|
| 161 |
+
text = re.sub(r"[ \t]+", " ", text)
|
| 162 |
+
text = re.sub(r"\n{3,}", "\n\n", text)
|
| 163 |
+
return text.strip()
|
| 164 |
+
|
| 165 |
+
|
| 166 |
+
def paper_identity_key(paper: Dict[str, Any]) -> str:
|
| 167 |
+
return (
|
| 168 |
+
normalize_doi(paper.get("doi") or "")
|
| 169 |
+
or ((paper.get("external_ids") or {}).get("arxiv") or "")
|
| 170 |
+
or norm_text(paper.get("title", "")).lower()
|
| 171 |
+
or str(paper.get("id", ""))
|
| 172 |
+
)
|
| 173 |
+
|
| 174 |
+
|
| 175 |
+
def unique_keep_order(items: List[str]) -> List[str]:
|
| 176 |
+
seen = set()
|
| 177 |
+
out = []
|
| 178 |
+
for item in items:
|
| 179 |
+
key = norm_text(item).lower()
|
| 180 |
+
if key and key not in seen:
|
| 181 |
+
seen.add(key)
|
| 182 |
+
out.append(norm_text(item))
|
| 183 |
+
return out
|
| 184 |
+
|
| 185 |
+
|
| 186 |
+
# ----------------------------
|
| 187 |
+
# Concept / claim extraction
|
| 188 |
+
# ----------------------------
|
| 189 |
+
|
| 190 |
+
def normalize_concept_label(phrase: str) -> str:
|
| 191 |
+
phrase = norm_text(phrase)
|
| 192 |
+
mapping = {"ph": "pH", "ai": "AI", "ml": "ML", "3d": "3D", "ecg": "ECG", "mri": "MRI"}
|
| 193 |
+
return " ".join(mapping.get(p.lower(), p) for p in phrase.split())
|
| 194 |
+
|
| 195 |
+
|
| 196 |
+
def looks_like_bad_phrase(phrase: str) -> bool:
|
| 197 |
+
phrase = norm_text(phrase)
|
| 198 |
+
if not phrase or len(phrase) < 4 or len(phrase) > 90:
|
| 199 |
+
return True
|
| 200 |
+
parts = phrase.split()
|
| 201 |
+
if len(parts) > 6:
|
| 202 |
+
return True
|
| 203 |
+
for p in parts:
|
| 204 |
+
t = p.strip("-.,;:()[]{}")
|
| 205 |
+
if not t:
|
| 206 |
+
return True
|
| 207 |
+
if len(t) == 1 and t.lower() not in {"p", "h"}:
|
| 208 |
+
return True
|
| 209 |
+
if re.match(r"^[bcdfghjklmnpqrstvwxyz]{5,}$", t.lower()):
|
| 210 |
+
return True
|
| 211 |
+
return False
|
| 212 |
+
|
| 213 |
+
|
| 214 |
+
def extract_concepts_from_text(text: str, max_terms: int = GRAPH_MAX_CONCEPTS) -> List[str]:
|
| 215 |
+
text = clean_extracted_text(text)
|
| 216 |
+
words = re.findall(r"[A-Za-z][A-Za-z0-9\-/+]{2,}", text)
|
| 217 |
+
phrases = []
|
| 218 |
+
|
| 219 |
+
for n in (3, 2, 1):
|
| 220 |
+
for i in range(len(words) - n + 1):
|
| 221 |
+
phrase = " ".join(words[i:i + n])
|
| 222 |
+
low = phrase.lower()
|
| 223 |
+
if any(tok in STOPWORDS for tok in low.split()):
|
| 224 |
+
continue
|
| 225 |
+
if looks_like_bad_phrase(low):
|
| 226 |
+
continue
|
| 227 |
+
phrases.append(low)
|
| 228 |
+
|
| 229 |
+
counts = Counter(phrases)
|
| 230 |
+
ranked = []
|
| 231 |
+
for phrase, count in counts.most_common(max_terms * 8):
|
| 232 |
+
score = float(count)
|
| 233 |
+
score += 0.25 * len(phrase.split())
|
| 234 |
+
if any(x in phrase for x in ["hydrogel", "conductive", "responsive", "injectable", "graph", "neural", "cardiac", "biomaterial"]):
|
| 235 |
+
score += 0.5
|
| 236 |
+
ranked.append((score, phrase))
|
| 237 |
+
|
| 238 |
+
out = []
|
| 239 |
+
seen = set()
|
| 240 |
+
for _, phrase in sorted(ranked, key=lambda x: x[0], reverse=True):
|
| 241 |
+
clean = normalize_concept_label(phrase)
|
| 242 |
+
low = clean.lower()
|
| 243 |
+
if low in seen:
|
| 244 |
+
continue
|
| 245 |
+
if any((low in s or s in low) and abs(len(low) - len(s)) <= 2 for s in seen):
|
| 246 |
+
continue
|
| 247 |
+
seen.add(low)
|
| 248 |
+
out.append(clean)
|
| 249 |
+
if len(out) >= max_terms:
|
| 250 |
+
break
|
| 251 |
+
return out
|
| 252 |
+
|
| 253 |
+
|
| 254 |
+
def extract_claim_like_sentences(text: str, max_items: int = GRAPH_MAX_CLAIMS) -> List[str]:
|
| 255 |
+
text = clean_extracted_text(text)
|
| 256 |
+
sentences = [norm_text(s) for s in re.split(r"(?<=[\.\!\?])\s+", text) if norm_text(s)]
|
| 257 |
+
scored = []
|
| 258 |
+
for sentence in sentences:
|
| 259 |
+
if len(sentence) < 45 or len(sentence) > 320:
|
| 260 |
+
continue
|
| 261 |
+
lower = sentence.lower()
|
| 262 |
+
score = 0.0
|
| 263 |
+
if any(k in lower for k in ["improves", "reduces", "increases", "demonstrates", "shows", "reveals", "predicts", "achieves", "outperforms", "enables", "supports"]):
|
| 264 |
+
score += 2.0
|
| 265 |
+
if any(k in lower for k in ["significant", "associated", "correlated", "effective", "robust", "accurate", "validated", "statistically"]):
|
| 266 |
+
score += 1.0
|
| 267 |
+
score += min(len(tokenize(sentence)) / 18.0, 2.0)
|
| 268 |
+
scored.append((score, sentence))
|
| 269 |
+
|
| 270 |
+
out = []
|
| 271 |
+
seen = set()
|
| 272 |
+
for _, sentence in sorted(scored, key=lambda x: x[0], reverse=True):
|
| 273 |
+
key = sentence.lower()
|
| 274 |
+
if key not in seen:
|
| 275 |
+
seen.add(key)
|
| 276 |
+
out.append(sentence)
|
| 277 |
+
if len(out) >= max_items:
|
| 278 |
+
break
|
| 279 |
+
return out
|
| 280 |
+
|
| 281 |
+
|
| 282 |
+
def extract_title_from_text(raw_text: str, fallback: str = "Uploaded PDF") -> str:
|
| 283 |
+
raw_text = clean_extracted_text(raw_text or "")
|
| 284 |
+
lines = [norm_text(x) for x in raw_text.splitlines() if norm_text(x)]
|
| 285 |
+
for line in lines[:18]:
|
| 286 |
+
if 8 <= len(line) <= 240 and not re.search(r"doi|http|www\.|@", line, flags=re.I):
|
| 287 |
+
return truncate_text(line.strip(" -:;"), 240)
|
| 288 |
+
return fallback
|
| 289 |
+
|
| 290 |
+
|
| 291 |
+
def extract_references_from_text(text: str) -> List[Dict[str, str]]:
|
| 292 |
+
refs = []
|
| 293 |
+
seen = set()
|
| 294 |
+
doi_re = re.compile(r"10\.\d{4,9}/[-._;()/:A-Z0-9]+", re.I)
|
| 295 |
+
|
| 296 |
+
for line in [norm_text(x) for x in clean_extracted_text(text).splitlines() if norm_text(x)][-250:]:
|
| 297 |
+
doi_match = doi_re.search(line)
|
| 298 |
+
doi = normalize_doi(doi_match.group(0)) if doi_match else ""
|
| 299 |
+
title = line.replace(doi_match.group(0), "").strip(" .;,-") if doi_match else line
|
| 300 |
+
if len(title) < 12:
|
| 301 |
+
continue
|
| 302 |
+
key = (title.lower(), doi.lower())
|
| 303 |
+
if key in seen:
|
| 304 |
+
continue
|
| 305 |
+
seen.add(key)
|
| 306 |
+
refs.append({"title": truncate_text(title, 220), "doi": doi})
|
| 307 |
+
if len(refs) >= 40:
|
| 308 |
+
break
|
| 309 |
+
return refs
|
| 310 |
+
|
| 311 |
+
|
| 312 |
+
# ----------------------------
|
| 313 |
+
# Search sources
|
| 314 |
+
# ----------------------------
|
| 315 |
+
|
| 316 |
+
def parse_openalex_abstract(inverted_index) -> str:
|
| 317 |
+
if not inverted_index or not isinstance(inverted_index, dict):
|
| 318 |
+
return ""
|
| 319 |
+
pos_to_word = {}
|
| 320 |
+
for word, positions in inverted_index.items():
|
| 321 |
+
for pos in positions:
|
| 322 |
+
pos_to_word[pos] = word
|
| 323 |
+
return " ".join(pos_to_word[i] for i in sorted(pos_to_word)) if pos_to_word else ""
|
| 324 |
+
|
| 325 |
+
|
| 326 |
+
def enrich_paper_semantics(query: str, paper: Dict[str, Any]) -> Dict[str, Any]:
|
| 327 |
+
paper = dict(paper)
|
| 328 |
+
title = clean_extracted_text(paper.get("title", ""))
|
| 329 |
+
abstract = clean_extracted_text(paper.get("abstract", "") or paper.get("summary", ""))
|
| 330 |
+
venue = clean_extracted_text(paper.get("venue", ""))
|
| 331 |
+
|
| 332 |
+
concepts = extract_concepts_from_text(" ".join([title, abstract, venue]), max_terms=GRAPH_MAX_CONCEPTS)
|
| 333 |
+
claims = extract_claim_like_sentences(abstract, max_items=GRAPH_MAX_CLAIMS)
|
| 334 |
+
rel = text_overlap_score(query, f"{title} {abstract}")
|
| 335 |
+
recency = compute_recency_bonus(paper.get("year"))
|
| 336 |
+
doi_bonus = 0.02 if paper.get("doi") else 0.0
|
| 337 |
+
oa_bonus = 0.03 if paper.get("open_access") else 0.0
|
| 338 |
+
learned_score = float(paper.get("score", 0)) + rel * 0.52 + recency + doi_bonus + oa_bonus + min(len(concepts), 8) * 0.012
|
| 339 |
+
|
| 340 |
+
paper["title"] = title or paper.get("title", "Untitled")
|
| 341 |
+
paper["abstract"] = abstract
|
| 342 |
+
paper["summary"] = truncate_text(abstract or paper.get("summary", ""), 520)
|
| 343 |
+
paper["venue"] = venue
|
| 344 |
+
paper["concepts"] = concepts[:GRAPH_MAX_CONCEPTS]
|
| 345 |
+
paper["claims"] = claims[:GRAPH_MAX_CLAIMS]
|
| 346 |
+
paper["relevance"] = round(rel, 4)
|
| 347 |
+
paper["learned_score"] = round(learned_score, 4)
|
| 348 |
+
return paper
|
| 349 |
+
|
| 350 |
+
|
| 351 |
+
def dedupe_papers(items: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
|
| 352 |
+
seen = {}
|
| 353 |
+
for item in items:
|
| 354 |
+
key = paper_identity_key(item) or f"{item.get('source', 'src')}::{item.get('title', 'paper')}"
|
| 355 |
+
cur = seen.get(key)
|
| 356 |
+
cur_score = float(cur.get("learned_score", cur.get("score", 0))) if cur else -1
|
| 357 |
+
new_score = float(item.get("learned_score", item.get("score", 0)))
|
| 358 |
+
if cur is None or new_score > cur_score:
|
| 359 |
+
seen[key] = item
|
| 360 |
+
out = list(seen.values())
|
| 361 |
+
out.sort(key=lambda x: float(x.get("learned_score", x.get("score", 0))), reverse=True)
|
| 362 |
+
return out
|
| 363 |
+
|
| 364 |
+
|
| 365 |
+
def search_arxiv(query: str, max_results: int = 8) -> List[Dict[str, Any]]:
|
| 366 |
+
encoded = urllib.parse.quote(query)
|
| 367 |
+
url = (
|
| 368 |
+
"http://export.arxiv.org/api/query?search_query=all:"
|
| 369 |
+
f"{encoded}&start=0&max_results={max_results}&sortBy=relevance&sortOrder=descending"
|
| 370 |
+
)
|
| 371 |
+
response = requests.get(url, timeout=REQUEST_TIMEOUT)
|
| 372 |
+
response.raise_for_status()
|
| 373 |
+
root = ET.fromstring(response.text)
|
| 374 |
+
ns = {"atom": "http://www.w3.org/2005/Atom"}
|
| 375 |
+
|
| 376 |
+
out = []
|
| 377 |
+
for entry in root.findall("atom:entry", ns):
|
| 378 |
+
title = clean_extracted_text(entry.findtext("atom:title", default="", namespaces=ns) or "")
|
| 379 |
+
summary = truncate_text(clean_extracted_text(entry.findtext("atom:summary", default="", namespaces=ns) or ""), MAX_ABSTRACT_CHARS)
|
| 380 |
+
published = entry.findtext("atom:published", default="", namespaces=ns) or ""
|
| 381 |
+
paper_id = entry.findtext("atom:id", default="", namespaces=ns) or ""
|
| 382 |
+
authors = [clean_extracted_text(a.findtext("atom:name", default="", namespaces=ns) or "") for a in entry.findall("atom:author", ns)]
|
| 383 |
+
pdf_url = ""
|
| 384 |
+
for link in entry.findall("atom:link", ns):
|
| 385 |
+
if link.attrib.get("title") == "pdf":
|
| 386 |
+
pdf_url = link.attrib.get("href", "")
|
| 387 |
+
break
|
| 388 |
+
|
| 389 |
+
out.append({
|
| 390 |
+
"id": paper_id or title,
|
| 391 |
+
"title": title,
|
| 392 |
+
"summary": summary,
|
| 393 |
+
"abstract": summary,
|
| 394 |
+
"published": published[:10],
|
| 395 |
+
"authors": [a for a in authors[:8] if a],
|
| 396 |
+
"authors_text": ", ".join([a for a in authors[:4] if a]) or "Unknown authors",
|
| 397 |
+
"url": paper_id,
|
| 398 |
+
"pdf": pdf_url,
|
| 399 |
+
"doi": "",
|
| 400 |
+
"venue": "arXiv",
|
| 401 |
+
"year": published[:4] if published else "",
|
| 402 |
+
"source": "arxiv",
|
| 403 |
+
"score": 0.76,
|
| 404 |
+
"open_access": True,
|
| 405 |
+
"external_ids": {"arxiv": (paper_id or "").split("/")[-1]},
|
| 406 |
+
})
|
| 407 |
+
return out
|
| 408 |
+
|
| 409 |
+
|
| 410 |
+
def search_crossref(query: str, mode: str = "topic", max_results: int = 8) -> List[Dict[str, Any]]:
|
| 411 |
+
headers = {"User-Agent": "dvnc-ai-space/1.0"}
|
| 412 |
+
if mode == "doi":
|
| 413 |
+
url = f"https://api.crossref.org/works/{urllib.parse.quote(normalize_doi(query))}"
|
| 414 |
+
response = requests.get(url, headers=headers, timeout=REQUEST_TIMEOUT)
|
| 415 |
+
if response.status_code != 200:
|
| 416 |
+
return []
|
| 417 |
+
items = [response.json().get("message", {})]
|
| 418 |
+
else:
|
| 419 |
+
params = {"rows": max_results}
|
| 420 |
+
if mode in ("title", "paper_name"):
|
| 421 |
+
params["query.title"] = query
|
| 422 |
+
else:
|
| 423 |
+
params["query.bibliographic"] = query
|
| 424 |
+
response = requests.get("https://api.crossref.org/works", params=params, headers=headers, timeout=REQUEST_TIMEOUT)
|
| 425 |
+
response.raise_for_status()
|
| 426 |
+
items = response.json().get("message", {}).get("items", [])
|
| 427 |
+
|
| 428 |
+
out = []
|
| 429 |
+
for item in items:
|
| 430 |
+
authors = []
|
| 431 |
+
for a in item.get("author", []) or []:
|
| 432 |
+
name = " ".join(filter(None, [a.get("given"), a.get("family")])).strip()
|
| 433 |
+
if name:
|
| 434 |
+
authors.append(clean_extracted_text(name))
|
| 435 |
+
|
| 436 |
+
title = clean_extracted_text((item.get("title") or ["Untitled"])[0])
|
| 437 |
+
year = ""
|
| 438 |
+
for key in ["published-print", "published-online", "created"]:
|
| 439 |
+
if item.get(key, {}).get("date-parts"):
|
| 440 |
+
year = str(item[key]["date-parts"][0][0])
|
| 441 |
+
break
|
| 442 |
+
abstract = truncate_text(clean_extracted_text(re.sub("<.*?>", " ", item.get("abstract") or "")), MAX_ABSTRACT_CHARS)
|
| 443 |
+
doi = normalize_doi(item.get("DOI", ""))
|
| 444 |
+
|
| 445 |
+
out.append({
|
| 446 |
+
"id": doi or title,
|
| 447 |
+
"title": title,
|
| 448 |
+
"summary": truncate_text(abstract, 500),
|
| 449 |
+
"abstract": abstract,
|
| 450 |
+
"published": year,
|
| 451 |
+
"authors": authors,
|
| 452 |
+
"authors_text": ", ".join(authors[:4]) if authors else "Unknown authors",
|
| 453 |
+
"url": item.get("URL", ""),
|
| 454 |
+
"pdf": "",
|
| 455 |
+
"doi": doi,
|
| 456 |
+
"venue": clean_extracted_text((item.get("container-title") or [""])[0]),
|
| 457 |
+
"year": year,
|
| 458 |
+
"source": "crossref",
|
| 459 |
+
"score": 0.72,
|
| 460 |
+
"open_access": None,
|
| 461 |
+
"external_ids": {"crossref": doi} if doi else {},
|
| 462 |
+
})
|
| 463 |
+
return out
|
| 464 |
+
|
| 465 |
+
|
| 466 |
+
def search_openalex(query: str, mode: str = "topic", max_results: int = 8) -> List[Dict[str, Any]]:
|
| 467 |
+
params = {"per-page": max_results}
|
| 468 |
+
if mode == "doi":
|
| 469 |
+
doi = normalize_doi(query)
|
| 470 |
+
params["filter"] = f"doi:https://doi.org/{doi}"
|
| 471 |
+
else:
|
| 472 |
+
params["search"] = query
|
| 473 |
+
|
| 474 |
+
response = requests.get("https://api.openalex.org/works", params=params, timeout=REQUEST_TIMEOUT)
|
| 475 |
+
if response.status_code != 200:
|
| 476 |
+
return []
|
| 477 |
+
items = response.json().get("results", [])
|
| 478 |
+
|
| 479 |
+
out = []
|
| 480 |
+
for item in items:
|
| 481 |
+
authors = []
|
| 482 |
+
for auth in item.get("authorships", [])[:8]:
|
| 483 |
+
author = auth.get("author") or {}
|
| 484 |
+
if author.get("display_name"):
|
| 485 |
+
authors.append(clean_extracted_text(author["display_name"]))
|
| 486 |
+
oa = item.get("open_access") or {}
|
| 487 |
+
doi = normalize_doi(item.get("doi") or "")
|
| 488 |
+
abstract = truncate_text(clean_extracted_text(parse_openalex_abstract(item.get("abstract_inverted_index"))), MAX_ABSTRACT_CHARS)
|
| 489 |
+
|
| 490 |
+
out.append({
|
| 491 |
+
"id": item.get("id") or doi or item.get("title"),
|
| 492 |
+
"title": clean_extracted_text(item.get("title") or ""),
|
| 493 |
+
"summary": truncate_text(abstract, 500),
|
| 494 |
+
"abstract": abstract,
|
| 495 |
+
"published": str(item.get("publication_year") or ""),
|
| 496 |
+
"authors": authors,
|
| 497 |
+
"authors_text": ", ".join(authors[:4]) if authors else "Unknown authors",
|
| 498 |
+
"url": ((item.get("primary_location") or {}).get("landing_page_url") or item.get("id") or ""),
|
| 499 |
+
"pdf": oa.get("oa_url") or "",
|
| 500 |
+
"doi": doi,
|
| 501 |
+
"venue": clean_extracted_text((((item.get("primary_location") or {}).get("source") or {}).get("display_name") or "")),
|
| 502 |
+
"year": str(item.get("publication_year") or ""),
|
| 503 |
+
"source": "openalex",
|
| 504 |
+
"score": 0.80,
|
| 505 |
+
"open_access": oa.get("is_oa"),
|
| 506 |
+
"external_ids": item.get("ids") or {},
|
| 507 |
+
})
|
| 508 |
+
return out
|
| 509 |
+
|
| 510 |
+
|
| 511 |
+
def search_semantic_scholar(query: str, mode: str = "topic", max_results: int = 8) -> List[Dict[str, Any]]:
|
| 512 |
+
fields = "title,authors,year,abstract,venue,externalIds,url,openAccessPdf"
|
| 513 |
+
if mode == "doi":
|
| 514 |
+
doi = normalize_doi(query)
|
| 515 |
+
url = f"https://api.semanticscholar.org/graph/v1/paper/DOI:{urllib.parse.quote(doi)}"
|
| 516 |
+
response = requests.get(url, params={"fields": fields}, timeout=REQUEST_TIMEOUT)
|
| 517 |
+
if response.status_code != 200:
|
| 518 |
+
return []
|
| 519 |
+
items = [response.json()]
|
| 520 |
+
else:
|
| 521 |
+
response = requests.get(
|
| 522 |
+
"https://api.semanticscholar.org/graph/v1/paper/search",
|
| 523 |
+
params={"query": query, "limit": max_results, "fields": fields},
|
| 524 |
+
timeout=REQUEST_TIMEOUT,
|
| 525 |
+
)
|
| 526 |
+
if response.status_code != 200:
|
| 527 |
+
return []
|
| 528 |
+
items = response.json().get("data", [])
|
| 529 |
+
|
| 530 |
+
out = []
|
| 531 |
+
for item in items:
|
| 532 |
+
external = item.get("externalIds") or {}
|
| 533 |
+
authors = [clean_extracted_text(a.get("name")) for a in item.get("authors", []) if a.get("name")]
|
| 534 |
+
abstract = truncate_text(clean_extracted_text(item.get("abstract", "")), MAX_ABSTRACT_CHARS)
|
| 535 |
+
out.append({
|
| 536 |
+
"id": external.get("CorpusId") or external.get("DOI") or item.get("title"),
|
| 537 |
+
"title": clean_extracted_text(item.get("title") or ""),
|
| 538 |
+
"summary": truncate_text(abstract, 500),
|
| 539 |
+
"abstract": abstract,
|
| 540 |
+
"published": str(item.get("year") or ""),
|
| 541 |
+
"authors": authors,
|
| 542 |
+
"authors_text": ", ".join(authors[:4]) if authors else "Unknown authors",
|
| 543 |
+
"url": item.get("url") or "",
|
| 544 |
+
"pdf": ((item.get("openAccessPdf") or {}).get("url") or ""),
|
| 545 |
+
"doi": normalize_doi(external.get("DOI", "")),
|
| 546 |
+
"venue": clean_extracted_text(item.get("venue") or ""),
|
| 547 |
+
"year": str(item.get("year") or ""),
|
| 548 |
+
"source": "semantic_scholar",
|
| 549 |
+
"score": 0.84,
|
| 550 |
+
"open_access": bool((item.get("openAccessPdf") or {}).get("url")),
|
| 551 |
+
"external_ids": external,
|
| 552 |
+
})
|
| 553 |
+
return out
|
| 554 |
+
|
| 555 |
+
|
| 556 |
+
def search_europe_pmc(query: str, mode: str = "topic", max_results: int = 8) -> List[Dict[str, Any]]:
|
| 557 |
+
epmc_query = f'DOI:"{query}"' if mode == "doi" else query
|
| 558 |
+
params = {"query": epmc_query, "format": "json", "pageSize": max_results, "resultType": "core"}
|
| 559 |
+
response = requests.get("https://www.ebi.ac.uk/europepmc/webservices/rest/search", params=params, timeout=REQUEST_TIMEOUT)
|
| 560 |
+
if response.status_code != 200:
|
| 561 |
+
return []
|
| 562 |
+
items = response.json().get("resultList", {}).get("result", [])
|
| 563 |
+
|
| 564 |
+
out = []
|
| 565 |
+
for item in items:
|
| 566 |
+
author_string = item.get("authorString", "")
|
| 567 |
+
authors = [clean_extracted_text(x) for x in author_string.split(",")[:8] if norm_text(x)]
|
| 568 |
+
pmcid = item.get("pmcid", "")
|
| 569 |
+
pdf_url = f"https://europepmc.org/articles/{pmcid}?pdf=render" if pmcid else ""
|
| 570 |
+
landing_url = f"https://europepmc.org/article/PMC/{pmcid}" if pmcid else ""
|
| 571 |
+
abstract = truncate_text(clean_extracted_text(item.get("abstractText", "")), MAX_ABSTRACT_CHARS)
|
| 572 |
+
out.append({
|
| 573 |
+
"id": item.get("id") or item.get("doi") or item.get("title"),
|
| 574 |
+
"title": clean_extracted_text(item.get("title") or ""),
|
| 575 |
+
"summary": truncate_text(abstract, 500),
|
| 576 |
+
"abstract": abstract,
|
| 577 |
+
"published": str(item.get("pubYear") or ""),
|
| 578 |
+
"authors": authors,
|
| 579 |
+
"authors_text": ", ".join(authors[:4]) if authors else "Unknown authors",
|
| 580 |
+
"url": landing_url,
|
| 581 |
+
"pdf": pdf_url,
|
| 582 |
+
"doi": normalize_doi(item.get("doi", "")),
|
| 583 |
+
"venue": clean_extracted_text(item.get("journalTitle", "")),
|
| 584 |
+
"year": str(item.get("pubYear") or ""),
|
| 585 |
+
"source": "europe_pmc",
|
| 586 |
+
"score": 0.78,
|
| 587 |
+
"open_access": bool(pmcid),
|
| 588 |
+
"external_ids": {"pmid": item.get("pmid"), "pmcid": pmcid},
|
| 589 |
+
})
|
| 590 |
+
return out
|
| 591 |
+
|
| 592 |
+
|
| 593 |
+
def resolve_link(query: str) -> List[Dict[str, Any]]:
|
| 594 |
+
url = (query or "").strip()
|
| 595 |
+
if not url:
|
| 596 |
+
return []
|
| 597 |
+
return [{
|
| 598 |
+
"id": url,
|
| 599 |
+
"title": Path(url.split("?")[0]).name or url,
|
| 600 |
+
"summary": "Direct link supplied.",
|
| 601 |
+
"abstract": "",
|
| 602 |
+
"published": "",
|
| 603 |
+
"authors": [],
|
| 604 |
+
"authors_text": "Unknown authors",
|
| 605 |
+
"url": url,
|
| 606 |
+
"pdf": url if url.lower().endswith(".pdf") else "",
|
| 607 |
+
"doi": "",
|
| 608 |
+
"venue": "Web Link",
|
| 609 |
+
"year": "",
|
| 610 |
+
"source": "link",
|
| 611 |
+
"score": 0.4,
|
| 612 |
+
"open_access": None,
|
| 613 |
+
"external_ids": {},
|
| 614 |
+
}]
|
| 615 |
+
|
| 616 |
+
|
| 617 |
+
def discover_papers(query: str, mode: str, sources: List[str], max_results: int = 10) -> List[Dict[str, Any]]:
|
| 618 |
+
query = (query or "").strip()
|
| 619 |
+
if not query:
|
| 620 |
+
return []
|
| 621 |
+
|
| 622 |
+
mode = detect_query_type(query) if mode == "autonomous_web" else mode
|
| 623 |
+
selected_sources = ensure_list(sources) or DEFAULT_SOURCES
|
| 624 |
+
results = []
|
| 625 |
+
|
| 626 |
+
if mode == "link":
|
| 627 |
+
return dedupe_papers(resolve_link(query))
|
| 628 |
+
|
| 629 |
+
if "arxiv" in selected_sources and mode != "doi":
|
| 630 |
+
try:
|
| 631 |
+
results.extend(search_arxiv(query, max_results=min(max_results, 6)))
|
| 632 |
+
except Exception:
|
| 633 |
+
pass
|
| 634 |
+
if "crossref" in selected_sources:
|
| 635 |
+
try:
|
| 636 |
+
results.extend(search_crossref(query, mode=mode, max_results=min(max_results, 8)))
|
| 637 |
+
except Exception:
|
| 638 |
+
pass
|
| 639 |
+
if "openalex" in selected_sources:
|
| 640 |
+
try:
|
| 641 |
+
results.extend(search_openalex(query, mode=mode, max_results=min(max_results, 8)))
|
| 642 |
+
except Exception:
|
| 643 |
+
pass
|
| 644 |
+
if "semantic_scholar" in selected_sources:
|
| 645 |
+
try:
|
| 646 |
+
results.extend(search_semantic_scholar(query, mode=mode, max_results=min(max_results, 8)))
|
| 647 |
+
except Exception:
|
| 648 |
+
pass
|
| 649 |
+
if "europe_pmc" in selected_sources:
|
| 650 |
+
try:
|
| 651 |
+
results.extend(search_europe_pmc(query, mode=mode, max_results=min(max_results, 8)))
|
| 652 |
+
except Exception:
|
| 653 |
+
pass
|
| 654 |
+
|
| 655 |
+
papers = [enrich_paper_semantics(query, p) for p in dedupe_papers(results)]
|
| 656 |
+
papers.sort(key=lambda x: float(x.get("learned_score", x.get("score", 0))), reverse=True)
|
| 657 |
+
return papers[:max_results]
|
| 658 |
+
|
| 659 |
+
|
| 660 |
+
# ----------------------------
|
| 661 |
+
# Journal / paper HTML
|
| 662 |
+
# ----------------------------
|
| 663 |
+
|
| 664 |
+
def journal_query_links(query: str):
|
| 665 |
+
q = urllib.parse.quote_plus(query or "biomaterials cardiac repair")
|
| 666 |
+
rows = []
|
| 667 |
+
for journal in JOURNALS:
|
| 668 |
+
url = f"{journal['url']}?q={q}" if "?" not in journal["url"] else f"{journal['url']}&q={q}"
|
| 669 |
+
if "ieeexplore" in journal["url"]:
|
| 670 |
+
url = f"https://ieeexplore.ieee.org/search/searchresult.jsp?queryText={q}"
|
| 671 |
+
rows.append({"name": journal["name"], "desc": journal["desc"], "url": url})
|
| 672 |
+
return rows
|
| 673 |
+
|
| 674 |
+
|
| 675 |
+
def build_journal_html(query):
|
| 676 |
+
rows = []
|
| 677 |
+
for journal in journal_query_links(query):
|
| 678 |
+
rows.append(
|
| 679 |
+
f"""
|
| 680 |
+
<a class="journal-card" href="{safe_text(journal['url'])}" target="_blank" rel="noopener noreferrer">
|
| 681 |
+
<div>
|
| 682 |
+
<h4>{safe_text(journal['name'])}</h4>
|
| 683 |
+
<p>{safe_text(journal['desc'])}</p>
|
| 684 |
+
</div>
|
| 685 |
+
<span>Open</span>
|
| 686 |
+
</a>
|
| 687 |
+
"""
|
| 688 |
+
)
|
| 689 |
+
return """
|
| 690 |
+
<style>
|
| 691 |
+
.journal-grid{display:grid;grid-template-columns:repeat(auto-fit,minmax(240px,1fr));gap:12px}
|
| 692 |
+
.journal-card{display:flex;justify-content:space-between;gap:12px;padding:14px 16px;border-radius:14px;border:1px solid #dbe3f0;background:#f8fbff;text-decoration:none;color:#0f172a}
|
| 693 |
+
.journal-card h4{margin:0 0 4px;font-size:15px}
|
| 694 |
+
.journal-card p{margin:0;color:#475569;font-size:13px;line-height:1.45}
|
| 695 |
+
.journal-card span{align-self:center;font-size:12px;font-weight:700;color:#2563eb}
|
| 696 |
+
</style>
|
| 697 |
+
<div class="journal-grid">""" + "".join(rows) + "</div>"
|
| 698 |
+
|
| 699 |
+
|
| 700 |
+
def paper_choice_value(index: int, paper: Dict[str, Any]) -> str:
|
| 701 |
+
doi = normalize_doi(paper.get("doi") or "")
|
| 702 |
+
title_slug = slugify(paper.get("title", ""))[:40]
|
| 703 |
+
return f"{index}|{doi}|{title_slug}"
|
| 704 |
+
|
| 705 |
+
|
| 706 |
+
def paper_choice_label(index: int, paper: Dict[str, Any]) -> str:
|
| 707 |
+
score = round(float(paper.get("learned_score", paper.get("score", 0))), 3)
|
| 708 |
+
title = paper.get("title", "Untitled")
|
| 709 |
+
authors_text = paper.get("authors_text", "Unknown authors")[:90]
|
| 710 |
+
source = paper.get("source", "src")
|
| 711 |
+
return f"[{source}] {title} — {authors_text} — score {score}"
|
| 712 |
+
|
| 713 |
+
|
| 714 |
+
def format_selection_choices(papers):
|
| 715 |
+
return [(paper_choice_label(i, paper), paper_choice_value(i, paper)) for i, paper in enumerate(papers)]
|
| 716 |
+
|
| 717 |
+
|
| 718 |
+
def format_papers_html(papers):
|
| 719 |
+
if not papers:
|
| 720 |
+
return '<div class="panel papers-panel" style="padding:18px"><p>No papers found yet.</p></div>'
|
| 721 |
+
|
| 722 |
+
items = []
|
| 723 |
+
for i, paper in enumerate(papers, start=1):
|
| 724 |
+
summary = safe_text((paper.get("summary") or paper.get("abstract") or "")[:320])
|
| 725 |
+
doi_line = f'<span class="paper-badge doi-badge">{safe_text(paper.get("doi"))}</span>' if paper.get("doi") else ""
|
| 726 |
+
pdf_link = paper.get("pdf") or "#"
|
| 727 |
+
abs_link = paper.get("url") or "#"
|
| 728 |
+
concepts_text = ", ".join((paper.get("concepts") or [])[:5])
|
| 729 |
+
|
| 730 |
+
items.append(
|
| 731 |
+
f"""
|
| 732 |
+
<article class="paper-card">
|
| 733 |
+
<div class="paper-topline">
|
| 734 |
+
<span class="paper-badge">{safe_text(paper.get('source', 'paper'))}</span>
|
| 735 |
+
<span class="paper-badge alt">{safe_text(paper.get('published', '') or 'Paper')}</span>
|
| 736 |
+
{doi_line}
|
| 737 |
+
</div>
|
| 738 |
+
<h4>{i}. {safe_text(paper.get('title', 'Untitled'))}</h4>
|
| 739 |
+
<p>{summary or 'No abstract snippet available.'}</p>
|
| 740 |
+
<div class="paper-meta-stack">
|
| 741 |
+
<div><strong>Authors:</strong> {safe_text(paper.get('authors_text', 'Unknown authors'))}</div>
|
| 742 |
+
<div><strong>Venue:</strong> {safe_text(paper.get('venue', 'Unknown venue'))}</div>
|
| 743 |
+
<div><strong>Learned score:</strong> {safe_text(round(float(paper.get('learned_score', paper.get('score', 0))), 3))}</div>
|
| 744 |
+
<div><strong>Concepts:</strong> {safe_text(concepts_text or 'None extracted')}</div>
|
| 745 |
+
</div>
|
| 746 |
+
<div class="paper-links">
|
| 747 |
+
<a href="{safe_text(abs_link)}" target="_blank" rel="noopener noreferrer">Abstract</a>
|
| 748 |
+
<a href="{safe_text(pdf_link)}" target="_blank" rel="noopener noreferrer">PDF</a>
|
| 749 |
+
</div>
|
| 750 |
+
</article>
|
| 751 |
+
"""
|
| 752 |
+
)
|
| 753 |
+
return '<div class="papers-grid">' + ''.join(items) + '</div>'
|
| 754 |
+
|
| 755 |
+
|
| 756 |
+
def format_frontier_html(frontier):
|
| 757 |
+
if not frontier:
|
| 758 |
+
return '<div class="panel papers-panel" style="padding:18px"><p>No autonomous expansion candidates yet.</p></div>'
|
| 759 |
+
cards = []
|
| 760 |
+
for i, paper in enumerate(frontier[:12], start=1):
|
| 761 |
+
cards.append(
|
| 762 |
+
f"""
|
| 763 |
+
<article class="paper-card frontier-card">
|
| 764 |
+
<div class="paper-topline">
|
| 765 |
+
<span class="paper-badge">frontier</span>
|
| 766 |
+
<span class="paper-badge alt">{safe_text(paper.get('source', 'paper'))}</span>
|
| 767 |
+
</div>
|
| 768 |
+
<h4>{i}. {safe_text(paper.get('title', 'Untitled'))}</h4>
|
| 769 |
+
<p>{safe_text((paper.get('summary') or paper.get('abstract') or '')[:280])}</p>
|
| 770 |
+
<div class="paper-meta-stack">
|
| 771 |
+
<div><strong>Frontier score:</strong> {safe_text(paper.get('frontier_score', paper.get('learned_score', paper.get('score', 0))))}</div>
|
| 772 |
+
<div><strong>Concept overlap:</strong> {safe_text(paper.get('frontier_concept_overlap', 0))}</div>
|
| 773 |
+
<div><strong>Authors:</strong> {safe_text(paper.get('authors_text', 'Unknown authors'))}</div>
|
| 774 |
+
</div>
|
| 775 |
+
</article>
|
| 776 |
+
"""
|
| 777 |
+
)
|
| 778 |
+
return '<div class="papers-grid">' + ''.join(cards) + '</div>'
|
| 779 |
+
|
| 780 |
+
|
| 781 |
+
def uploaded_pdf_summary(file_obj):
|
| 782 |
+
if not file_obj:
|
| 783 |
+
return "No PDF uploaded yet."
|
| 784 |
+
path = getattr(file_obj, "name", None) or str(file_obj)
|
| 785 |
+
p = Path(path)
|
| 786 |
+
return f"Uploaded PDF ready for ingestion: {p.name}. Use Parse uploaded PDF to extract title, abstract, sections, references, concepts, and claims."
|
| 787 |
+
|
| 788 |
+
|
| 789 |
+
# ----------------------------
|
| 790 |
+
# 3D graph
|
| 791 |
+
# ----------------------------
|
| 792 |
+
|
| 793 |
+
def graph_kind_style(kind: str) -> Dict[str, Any]:
|
| 794 |
+
palette = {
|
| 795 |
+
"query": {"color": "#1f8ef1", "size": 14, "label": "Research topic"},
|
| 796 |
+
"paper": {"color": "#00c49a", "size": 10, "label": "Paper"},
|
| 797 |
+
"upload": {"color": "#ff9f43", "size": 11, "label": "Uploaded PDF"},
|
| 798 |
+
"concept": {"color": "#a66cff", "size": 8, "label": "Concept"},
|
| 799 |
+
"author": {"color": "#f368e0", "size": 7, "label": "Author"},
|
| 800 |
+
"claim": {"color": "#ff6b6b", "size": 8, "label": "Claim"},
|
| 801 |
+
"reference": {"color": "#6c757d", "size": 7, "label": "Reference"},
|
| 802 |
+
"frontier": {"color": "#ffd166", "size": 8, "label": "Frontier candidate"},
|
| 803 |
+
}
|
| 804 |
+
return palette.get(kind, {"color": "#9aa0a6", "size": 7, "label": kind.title()})
|
| 805 |
+
|
| 806 |
+
|
| 807 |
+
def summarize_graph(nodes: List[Dict], edges: List[Dict]) -> Dict[str, Any]:
|
| 808 |
+
counts = Counter((n.get("kind") or n.get("type") or "unknown").lower() for n in nodes)
|
| 809 |
+
return {"nodes": len(nodes), "edges": len(edges), "counts": dict(counts)}
|
| 810 |
+
|
| 811 |
+
|
| 812 |
+
def _prepare_3d_graph_data(nodes: List[Dict], edges: List[Dict], title: str) -> Dict[str, Any]:
|
| 813 |
+
node_out = []
|
| 814 |
+
for node in nodes:
|
| 815 |
+
kind = (node.get("kind") or node.get("type") or "paper").lower()
|
| 816 |
+
if kind == "topic":
|
| 817 |
+
kind = "query"
|
| 818 |
+
if kind == "uploadedpdf":
|
| 819 |
+
kind = "upload"
|
| 820 |
+
if kind == "frontierpaper":
|
| 821 |
+
kind = "frontier"
|
| 822 |
+
style = graph_kind_style(kind)
|
| 823 |
+
node_out.append({
|
| 824 |
+
"id": node.get("id"),
|
| 825 |
+
"label": truncate_text(node.get("label") or node.get("title") or node.get("id") or "node", 120),
|
| 826 |
+
"kind": kind,
|
| 827 |
+
"color": style["color"],
|
| 828 |
+
"val": style["size"],
|
| 829 |
+
"detail": {
|
| 830 |
+
"kind": style["label"],
|
| 831 |
+
"title": node.get("title") or node.get("label") or node.get("id"),
|
| 832 |
+
"venue": node.get("venue") or "",
|
| 833 |
+
"year": node.get("year") or "",
|
| 834 |
+
"doi": node.get("doi") or "",
|
| 835 |
+
"source": node.get("source") or "",
|
| 836 |
+
"authors_text": node.get("authors_text") or "",
|
| 837 |
+
"text": node.get("text") or "",
|
| 838 |
+
},
|
| 839 |
+
})
|
| 840 |
+
edge_out = []
|
| 841 |
+
for edge in edges:
|
| 842 |
+
edge_out.append({
|
| 843 |
+
"source": edge.get("source"),
|
| 844 |
+
"target": edge.get("target"),
|
| 845 |
+
"type": edge.get("type") or "RELATES_TO",
|
| 846 |
+
"label": edge.get("type") or "RELATES_TO",
|
| 847 |
+
})
|
| 848 |
+
return {"title": title, "nodes": node_out, "links": edge_out, "summary": summarize_graph(nodes, edges)}
|
| 849 |
+
|
| 850 |
+
|
| 851 |
+
def build_learning_graph_html(nodes, edges, title="Self-Learning Knowledge Graph"):
|
| 852 |
+
if not nodes:
|
| 853 |
+
return """
|
| 854 |
+
<div class="panel brain-shell" style="overflow:auto; max-width:100%;">
|
| 855 |
+
<div class="brain-header">
|
| 856 |
+
<div>
|
| 857 |
+
<p class="eyebrow">Learning Graph</p>
|
| 858 |
+
<h3>Self-Learning Knowledge Graph</h3>
|
| 859 |
+
</div>
|
| 860 |
+
</div>
|
| 861 |
+
<div class="brain-stage learning-empty" style="min-height:420px; overflow:auto;">
|
| 862 |
+
<div class="empty-graph-copy">
|
| 863 |
+
<h4>No papers mapped yet</h4>
|
| 864 |
+
<p>Search papers, select candidates, or upload a PDF to grow the graph in an interactive 3D view.</p>
|
| 865 |
+
</div>
|
| 866 |
+
</div>
|
| 867 |
+
</div>
|
| 868 |
+
"""
|
| 869 |
+
|
| 870 |
+
graph_data = _prepare_3d_graph_data(nodes, edges, title)
|
| 871 |
+
payload_json = json.dumps(graph_data, ensure_ascii=False)
|
| 872 |
+
|
| 873 |
+
iframe_html = f"""
|
| 874 |
+
<!doctype html>
|
| 875 |
+
<html>
|
| 876 |
+
<head>
|
| 877 |
+
<meta charset="utf-8" />
|
| 878 |
+
<meta name="viewport" content="width=device-width, initial-scale=1" />
|
| 879 |
+
<style>
|
| 880 |
+
html, body {{ margin:0; height:100%; background:#0b1020; color:#eef2ff; font-family: Inter, ui-sans-serif, system-ui, sans-serif; overflow:hidden; }}
|
| 881 |
+
#wrap {{ position:relative; width:100%; height:100%; background:radial-gradient(circle at top, #18213c 0%, #0b1020 60%, #060910 100%); }}
|
| 882 |
+
#graph {{ position:absolute; inset:0; }}
|
| 883 |
+
.overlay {{
|
| 884 |
+
position:absolute; left:16px; top:16px; z-index:10; max-width:min(460px, calc(100% - 32px));
|
| 885 |
+
padding:14px 16px; border:1px solid rgba(255,255,255,.12); border-radius:16px;
|
| 886 |
+
background:rgba(10,14,28,.72); backdrop-filter: blur(14px); box-shadow:0 12px 28px rgba(0,0,0,.28);
|
| 887 |
+
}}
|
| 888 |
+
.overlay h3 {{ margin:0 0 6px; font-size:18px; line-height:1.2; }}
|
| 889 |
+
.overlay p {{ margin:0; font-size:13px; color:#cbd5e1; line-height:1.5; }}
|
| 890 |
+
.panel {{
|
| 891 |
+
position:absolute; right:16px; top:16px; z-index:10; width:min(360px, calc(100% - 32px));
|
| 892 |
+
max-height:calc(100% - 32px); overflow:auto; padding:14px 16px; border:1px solid rgba(255,255,255,.12);
|
| 893 |
+
border-radius:16px; background:rgba(10,14,28,.72); backdrop-filter: blur(14px); box-shadow:0 12px 28px rgba(0,0,0,.28);
|
| 894 |
+
}}
|
| 895 |
+
.panel h4 {{ margin:0 0 8px; font-size:14px; color:#f8fafc; }}
|
| 896 |
+
.panel p {{ margin:0; font-size:12px; color:#cbd5e1; line-height:1.5; }}
|
| 897 |
+
.panel dl {{ margin:12px 0 0; display:grid; grid-template-columns:auto 1fr; gap:6px 10px; font-size:12px; }}
|
| 898 |
+
.panel dt {{ color:#93c5fd; }}
|
| 899 |
+
.panel dd {{ margin:0; color:#e2e8f0; word-break:break-word; }}
|
| 900 |
+
.hint {{
|
| 901 |
+
position:absolute; left:16px; bottom:16px; z-index:10; padding:10px 12px; border-radius:12px;
|
| 902 |
+
font-size:12px; color:#dbeafe; background:rgba(15,23,42,.75); border:1px solid rgba(255,255,255,.08);
|
| 903 |
+
}}
|
| 904 |
+
</style>
|
| 905 |
+
<script src="https://unpkg.com/three@0.160.0/build/three.min.js"></script>
|
| 906 |
+
<script src="https://unpkg.com/3d-force-graph"></script>
|
| 907 |
+
</head>
|
| 908 |
+
<body>
|
| 909 |
+
<div id="wrap">
|
| 910 |
+
<div id="graph"></div>
|
| 911 |
+
<div class="overlay">
|
| 912 |
+
<h3></h3>
|
| 913 |
+
<p>Drag the background to orbit, scroll to zoom, right-drag to pan, and drag a node to move or pin it in 3D space.</p>
|
| 914 |
+
</div>
|
| 915 |
+
<div class="panel" id="panel">
|
| 916 |
+
<h4>Node details</h4>
|
| 917 |
+
<p>Click any node to inspect its label, type, venue, DOI, year, and source.</p>
|
| 918 |
+
</div>
|
| 919 |
+
<div class="hint">Interactive 3D graph: orbit, zoom, pan, drag nodes.</div>
|
| 920 |
+
</div>
|
| 921 |
+
<script>
|
| 922 |
+
const payload = {payload_json};
|
| 923 |
+
document.querySelector('.overlay h3').textContent = payload.title || 'Self-Learning Knowledge Graph';
|
| 924 |
+
const panelEl = document.getElementById('panel');
|
| 925 |
+
|
| 926 |
+
const Graph = ForceGraph3D()(document.getElementById('graph'))
|
| 927 |
+
.backgroundColor('#00000000')
|
| 928 |
+
.graphData(payload)
|
| 929 |
+
.nodeRelSize(6)
|
| 930 |
+
.nodeOpacity(1)
|
| 931 |
+
.nodeLabel(node => `<div style="padding:6px 8px"><strong>${{node.label}}</strong><br/>${{(node.detail || {{}}).kind || node.kind}}</div>`)
|
| 932 |
+
.linkWidth(link => ['ABOUT','UPLOADED_SOURCE','FRONTIER_CANDIDATE'].includes(link.type) ? 2.6 : 1.35)
|
| 933 |
+
.linkDirectionalParticles(link => ['ABOUT','FRONTIER_CANDIDATE'].includes(link.type) ? 2 : 0)
|
| 934 |
+
.linkDirectionalParticleWidth(2.2)
|
| 935 |
+
.cooldownTicks(140)
|
| 936 |
+
.d3VelocityDecay(0.24)
|
| 937 |
+
.d3Force('charge').strength(-180)
|
| 938 |
+
.nodeColor(node => node.color)
|
| 939 |
+
.nodeVal(node => node.val)
|
| 940 |
+
.onNodeClick(node => {{
|
| 941 |
+
const d = node.detail || {{}};
|
| 942 |
+
panelEl.innerHTML = `
|
| 943 |
+
<h4>Node details</h4>
|
| 944 |
+
<dl>
|
| 945 |
+
<dt>Label</dt><dd>${{node.label || ''}}</dd>
|
| 946 |
+
<dt>Type</dt><dd>${{d.kind || node.kind || ''}}</dd>
|
| 947 |
+
<dt>Venue</dt><dd>${{d.venue || '—'}}</dd>
|
| 948 |
+
<dt>Year</dt><dd>${{d.year || '—'}}</dd>
|
| 949 |
+
<dt>DOI</dt><dd>${{d.doi || '—'}}</dd>
|
| 950 |
+
<dt>Source</dt><dd>${{d.source || '—'}}</dd>
|
| 951 |
+
<dt>Authors</dt><dd>${{d.authors_text || '—'}}</dd>
|
| 952 |
+
<dt>Text</dt><dd>${{d.text || '—'}}</dd>
|
| 953 |
+
</dl>`;
|
| 954 |
+
}})
|
| 955 |
+
.onNodeDragEnd(node => {{
|
| 956 |
+
node.fx = node.x;
|
| 957 |
+
node.fy = node.y;
|
| 958 |
+
node.fz = node.z;
|
| 959 |
+
}});
|
| 960 |
+
|
| 961 |
+
Graph.scene().add(new THREE.AmbientLight(0xffffff, 1.1));
|
| 962 |
+
const dirLight = new THREE.DirectionalLight(0xffffff, 0.8);
|
| 963 |
+
dirLight.position.set(120, 120, 120);
|
| 964 |
+
Graph.scene().add(dirLight);
|
| 965 |
+
Graph.cameraPosition({{ z: 210 }});
|
| 966 |
+
</script>
|
| 967 |
+
</body>
|
| 968 |
+
</html>
|
| 969 |
+
"""
|
| 970 |
+
return f"""
|
| 971 |
+
<div class="panel brain-shell" style="overflow:auto; max-width:100%;">
|
| 972 |
+
<iframe
|
| 973 |
+
title="{safe_text(title)}"
|
| 974 |
+
style="width:100%; height:{GRAPH_IFRAME_HEIGHT}px; border:0; border-radius:18px; overflow:auto; background:#0b1020;"
|
| 975 |
+
sandbox="allow-scripts allow-same-origin"
|
| 976 |
+
srcdoc="{html.escape(iframe_html, quote=True)}"
|
| 977 |
+
></iframe>
|
| 978 |
+
</div>
|
| 979 |
+
"""
|
| 980 |
+
|
| 981 |
+
|
| 982 |
+
def build_learning_graph_state(query, papers, uploaded_name=None):
|
| 983 |
+
nodes = [{"id": "query", "label": query or "Research topic", "kind": "query"}]
|
| 984 |
+
edges = []
|
| 985 |
+
|
| 986 |
+
for i, paper in enumerate(papers[:6], start=1):
|
| 987 |
+
pid = f"paper_{i}"
|
| 988 |
+
nodes.append({
|
| 989 |
+
"id": pid,
|
| 990 |
+
"label": paper.get("title", f"Paper {i}"),
|
| 991 |
+
"kind": "paper",
|
| 992 |
+
"title": paper.get("title"),
|
| 993 |
+
"venue": paper.get("venue"),
|
| 994 |
+
"year": paper.get("year"),
|
| 995 |
+
"source": paper.get("source"),
|
| 996 |
+
"authors_text": paper.get("authors_text"),
|
| 997 |
+
})
|
| 998 |
+
edges.append({"source": "query", "target": pid, "type": "ABOUT"})
|
| 999 |
+
for concept in (paper.get("concepts") or [])[:3]:
|
| 1000 |
+
cid = f"concept_{i}_{slugify(concept)[:30]}"
|
| 1001 |
+
nodes.append({"id": cid, "label": concept, "kind": "concept"})
|
| 1002 |
+
edges.append({"source": pid, "target": cid, "type": "MENTIONS"})
|
| 1003 |
+
|
| 1004 |
+
if uploaded_name:
|
| 1005 |
+
nodes.append({"id": "upload", "label": uploaded_name, "kind": "upload"})
|
| 1006 |
+
edges.append({"source": "query", "target": "upload", "type": "UPLOADED_SOURCE"})
|
| 1007 |
+
return nodes, edges
|
| 1008 |
+
|
| 1009 |
+
|
| 1010 |
+
def graph_from_selected(query, selected_papers, uploaded_name=None, parsed_state=None, frontier=None):
|
| 1011 |
+
nodes = [{"id": "query", "label": query or "Research topic", "kind": "query"}]
|
| 1012 |
+
edges = []
|
| 1013 |
+
|
| 1014 |
+
for i, paper in enumerate(selected_papers[:8], start=1):
|
| 1015 |
+
pid = f"paper_{i}"
|
| 1016 |
+
nodes.append({
|
| 1017 |
+
"id": pid,
|
| 1018 |
+
"label": paper.get("title", f"Paper {i}"),
|
| 1019 |
+
"kind": "paper",
|
| 1020 |
+
"title": paper.get("title"),
|
| 1021 |
+
"venue": paper.get("venue"),
|
| 1022 |
+
"year": paper.get("year"),
|
| 1023 |
+
"doi": paper.get("doi"),
|
| 1024 |
+
"source": paper.get("source"),
|
| 1025 |
+
"authors_text": paper.get("authors_text"),
|
| 1026 |
+
})
|
| 1027 |
+
edges.append({"source": "query", "target": pid, "type": "ABOUT"})
|
| 1028 |
+
|
| 1029 |
+
for author in paper.get("authors", [])[:3]:
|
| 1030 |
+
aid = f"author_{i}_{slugify(author)[:30]}"
|
| 1031 |
+
nodes.append({"id": aid, "label": author, "kind": "author"})
|
| 1032 |
+
edges.append({"source": pid, "target": aid, "type": "WRITTEN_BY"})
|
| 1033 |
+
|
| 1034 |
+
for concept in (paper.get("concepts") or [])[:4]:
|
| 1035 |
+
cid = f"concept_{i}_{slugify(concept)[:30]}"
|
| 1036 |
+
nodes.append({"id": cid, "label": concept, "kind": "concept"})
|
| 1037 |
+
edges.append({"source": pid, "target": cid, "type": "MENTIONS"})
|
| 1038 |
+
|
| 1039 |
+
for claim in (paper.get("claims") or [])[:2]:
|
| 1040 |
+
cid = f"claim_{i}_{slugify(claim)[:30]}"
|
| 1041 |
+
nodes.append({"id": cid, "label": truncate_text(claim, 82), "kind": "claim", "text": claim})
|
| 1042 |
+
edges.append({"source": pid, "target": cid, "type": "ASSERTS"})
|
| 1043 |
+
|
| 1044 |
+
if uploaded_name:
|
| 1045 |
+
nodes.append({"id": "upload", "label": uploaded_name, "kind": "upload", "title": uploaded_name})
|
| 1046 |
+
edges.append({"source": "query", "target": "upload", "type": "UPLOADED_SOURCE"})
|
| 1047 |
+
if parsed_state and isinstance(parsed_state, dict):
|
| 1048 |
+
for concept in (parsed_state.get("concepts") or [])[:4]:
|
| 1049 |
+
cid = f"upload_concept_{slugify(concept)[:30]}"
|
| 1050 |
+
nodes.append({"id": cid, "label": concept, "kind": "concept"})
|
| 1051 |
+
edges.append({"source": "upload", "target": cid, "type": "MENTIONS"})
|
| 1052 |
+
for claim in (parsed_state.get("claims") or [])[:3]:
|
| 1053 |
+
cid = f"upload_claim_{slugify(claim)[:30]}"
|
| 1054 |
+
nodes.append({"id": cid, "label": truncate_text(claim, 82), "kind": "claim", "text": claim})
|
| 1055 |
+
edges.append({"source": "upload", "target": cid, "type": "ASSERTS"})
|
| 1056 |
+
for ref in (parsed_state.get("references") or [])[:6]:
|
| 1057 |
+
if ref.get("title"):
|
| 1058 |
+
rid = f"ref_{slugify(ref.get('title'))[:30]}"
|
| 1059 |
+
nodes.append({"id": rid, "label": truncate_text(ref.get("title"), 82), "kind": "reference", "doi": ref.get("doi", "")})
|
| 1060 |
+
edges.append({"source": "upload", "target": rid, "type": "CITES"})
|
| 1061 |
+
|
| 1062 |
+
for j, fp in enumerate(ensure_list(frontier)[:6], start=1):
|
| 1063 |
+
fid = f"frontier_{j}"
|
| 1064 |
+
nodes.append({
|
| 1065 |
+
"id": fid,
|
| 1066 |
+
"label": fp.get("title", f"Frontier {j}"),
|
| 1067 |
+
"kind": "frontier",
|
| 1068 |
+
"title": fp.get("title"),
|
| 1069 |
+
"source": fp.get("source"),
|
| 1070 |
+
"authors_text": fp.get("authors_text"),
|
| 1071 |
+
"year": fp.get("year"),
|
| 1072 |
+
"doi": fp.get("doi"),
|
| 1073 |
+
})
|
| 1074 |
+
edges.append({"source": "query", "target": fid, "type": "FRONTIER_CANDIDATE"})
|
| 1075 |
+
|
| 1076 |
+
dedup_nodes = []
|
| 1077 |
+
seen = set()
|
| 1078 |
+
for node in nodes:
|
| 1079 |
+
if node["id"] not in seen:
|
| 1080 |
+
seen.add(node["id"])
|
| 1081 |
+
dedup_nodes.append(node)
|
| 1082 |
+
return dedup_nodes, edges
|
| 1083 |
+
|
| 1084 |
+
|
| 1085 |
+
# ----------------------------
|
| 1086 |
+
# PDF parsing
|
| 1087 |
+
# ----------------------------
|
| 1088 |
+
|
| 1089 |
+
def parse_pdf_with_pymupdf(pdf_path: str) -> Dict[str, Any]:
|
| 1090 |
+
if fitz is None:
|
| 1091 |
+
raise RuntimeError("PyMuPDF not installed")
|
| 1092 |
+
|
| 1093 |
+
doc = fitz.open(pdf_path)
|
| 1094 |
+
page_texts = [page.get_text("text") for page in doc[: min(len(doc), 10)]]
|
| 1095 |
+
raw_text = truncate_text(clean_extracted_text("\n".join(page_texts).strip()), MAX_RAW_TEXT_CHARS)
|
| 1096 |
+
first_page = clean_extracted_text("\n".join(page_texts[:2]))[:5000]
|
| 1097 |
+
|
| 1098 |
+
title = extract_title_from_text(first_page, fallback=Path(pdf_path).name)
|
| 1099 |
+
|
| 1100 |
+
abstract = ""
|
| 1101 |
+
match = re.search(r"abstract\s*(.+?)(?:\n\s*\n|\n(?:1|i)[\.\s]|\nintroduction)", raw_text, re.I | re.S)
|
| 1102 |
+
if match:
|
| 1103 |
+
abstract = truncate_text(clean_extracted_text(match.group(1)), 2600)
|
| 1104 |
+
|
| 1105 |
+
sections = []
|
| 1106 |
+
blocks = re.split(r"\n(?=[A-Z][A-Za-z\s]{2,40}\n)", raw_text)
|
| 1107 |
+
for block in blocks[:10]:
|
| 1108 |
+
lines = [norm_text(x) for x in block.splitlines() if norm_text(x)]
|
| 1109 |
+
if not lines:
|
| 1110 |
+
continue
|
| 1111 |
+
heading = lines[0] if len(lines[0]) < 60 else "Section"
|
| 1112 |
+
body = " ".join(lines[1:] if len(lines) > 1 else lines)
|
| 1113 |
+
if len(body) > 80:
|
| 1114 |
+
sections.append({"heading": clean_extracted_text(heading), "text": truncate_text(body, 4200)})
|
| 1115 |
+
|
| 1116 |
+
return {
|
| 1117 |
+
"parser": "pymupdf",
|
| 1118 |
+
"title": title,
|
| 1119 |
+
"abstract": abstract,
|
| 1120 |
+
"authors": [],
|
| 1121 |
+
"sections": sections[:12] or ([{"heading": "Full Text", "text": raw_text[:12000]}] if raw_text else []),
|
| 1122 |
+
"references": extract_references_from_text(raw_text),
|
| 1123 |
+
"claims": extract_claim_like_sentences(raw_text, max_items=GRAPH_MAX_CLAIMS),
|
| 1124 |
+
"concepts": extract_concepts_from_text(" ".join([title, abstract, raw_text[:18000]]), max_terms=GRAPH_MAX_CONCEPTS),
|
| 1125 |
+
"raw_text": raw_text,
|
| 1126 |
+
"parser_quality": "text-fallback-cleaned",
|
| 1127 |
+
}
|
| 1128 |
+
|
| 1129 |
+
|
| 1130 |
+
def parse_uploaded_pdf(file_obj, parser_order=None):
|
| 1131 |
+
if not file_obj:
|
| 1132 |
+
return "### PDF parse status\n\nNo PDF uploaded yet.", {}
|
| 1133 |
+
|
| 1134 |
+
path = getattr(file_obj, "name", None) or str(file_obj)
|
| 1135 |
+
parser_order = ensure_list(parser_order) or PDF_PARSERS
|
| 1136 |
+
errors = []
|
| 1137 |
+
|
| 1138 |
+
for parser_name in parser_order:
|
| 1139 |
+
try:
|
| 1140 |
+
if parser_name == "pymupdf":
|
| 1141 |
+
result = parse_pdf_with_pymupdf(path)
|
| 1142 |
+
else:
|
| 1143 |
+
continue
|
| 1144 |
+
|
| 1145 |
+
summary = (
|
| 1146 |
+
f"### PDF parse status\n\n"
|
| 1147 |
+
f"- Parser used: {result['parser']}\n"
|
| 1148 |
+
f"- Parser quality: {result.get('parser_quality', 'unknown')}\n"
|
| 1149 |
+
f"- Title: {result.get('title') or 'Unknown'}\n"
|
| 1150 |
+
f"- Authors: {', '.join(result.get('authors')[:6]) if result.get('authors') else 'Unknown'}\n"
|
| 1151 |
+
f"- Abstract found: {'Yes' if result.get('abstract') else 'No'}\n"
|
| 1152 |
+
f"- Sections extracted: {len(result.get('sections') or [])}\n"
|
| 1153 |
+
f"- References extracted: {len(result.get('references') or [])}\n"
|
| 1154 |
+
f"- Concepts extracted: {len(result.get('concepts') or [])}\n"
|
| 1155 |
+
f"- Claims extracted: {len(result.get('claims') or [])}\n"
|
| 1156 |
+
)
|
| 1157 |
+
return summary, result
|
| 1158 |
+
except Exception as e:
|
| 1159 |
+
errors.append(f"{parser_name}: {e}")
|
| 1160 |
+
|
| 1161 |
+
fail_summary = "### PDF parse status\n\n" + "\n".join([f"- {x}" for x in errors])
|
| 1162 |
+
return fail_summary, {"parser": None, "errors": errors}
|
| 1163 |
+
|
| 1164 |
+
|
| 1165 |
+
def render_parse_result(parsed):
|
| 1166 |
+
if not parsed or not isinstance(parsed, dict) or (not parsed.get("title") and not parsed.get("sections")):
|
| 1167 |
+
return '<div class="panel" style="padding:18px"><p>No parsed document yet.</p></div>'
|
| 1168 |
+
|
| 1169 |
+
sections_html = []
|
| 1170 |
+
for section in parsed.get("sections", [])[:8]:
|
| 1171 |
+
sections_html.append(
|
| 1172 |
+
f"""
|
| 1173 |
+
<details class="agent-step">
|
| 1174 |
+
<summary class="agent-summary">
|
| 1175 |
+
<div class="agent-index">§</div>
|
| 1176 |
+
<div class="agent-head">
|
| 1177 |
+
<h4>{safe_text(section.get('heading', 'Section'))}</h4>
|
| 1178 |
+
<span>section</span>
|
| 1179 |
+
</div>
|
| 1180 |
+
</summary>
|
| 1181 |
+
<div class="agent-copy">
|
| 1182 |
+
<p>{safe_text(section.get('text', '')[:2200])}</p>
|
| 1183 |
+
</div>
|
| 1184 |
+
</details>
|
| 1185 |
+
"""
|
| 1186 |
+
)
|
| 1187 |
+
|
| 1188 |
+
refs = parsed.get("references", [])[:14]
|
| 1189 |
+
refs_html = "".join(
|
| 1190 |
+
f"<li>{safe_text(r.get('title') or 'Untitled')} {'· DOI ' + safe_text(r.get('doi')) if r.get('doi') else ''}</li>"
|
| 1191 |
+
for r in refs
|
| 1192 |
+
) or "<li>No references extracted.</li>"
|
| 1193 |
+
|
| 1194 |
+
concepts = parsed.get("concepts", [])[:12]
|
| 1195 |
+
claims = parsed.get("claims", [])[:8]
|
| 1196 |
+
concepts_html = "".join(f"<li>{safe_text(x)}</li>" for x in concepts) or "<li>No concepts extracted.</li>"
|
| 1197 |
+
claims_html = "".join(f"<li>{safe_text(x)}</li>" for x in claims) or "<li>No claims extracted.</li>"
|
| 1198 |
+
|
| 1199 |
+
title = safe_text(parsed.get("title") or "Parsed document")
|
| 1200 |
+
abstract = safe_text((parsed.get("abstract") or "")[:2600]) or "No abstract extracted."
|
| 1201 |
+
parser_name = safe_text(parsed.get("parser") or "unknown")
|
| 1202 |
+
parser_quality = safe_text(parsed.get("parser_quality") or "unknown")
|
| 1203 |
+
|
| 1204 |
+
return f"""
|
| 1205 |
+
<div class="panel" style="padding:18px">
|
| 1206 |
+
<div class="brain-header">
|
| 1207 |
+
<div>
|
| 1208 |
+
<p class="eyebrow">PDF Parse</p>
|
| 1209 |
+
<h3>{title}</h3>
|
| 1210 |
+
</div>
|
| 1211 |
+
<div class="brain-legend"><span><i class="dot dot-upload"></i> {parser_name} · {parser_quality}</span></div>
|
| 1212 |
+
</div>
|
| 1213 |
+
<div class="parse-grid">
|
| 1214 |
+
<div class="parse-card">
|
| 1215 |
+
<h4>Abstract</h4>
|
| 1216 |
+
<p>{abstract}</p>
|
| 1217 |
+
</div>
|
| 1218 |
+
<div class="parse-card">
|
| 1219 |
+
<h4>References</h4>
|
| 1220 |
+
<ul class="ref-list">{refs_html}</ul>
|
| 1221 |
+
</div>
|
| 1222 |
+
<div class="parse-card">
|
| 1223 |
+
<h4>Concepts</h4>
|
| 1224 |
+
<ul class="ref-list">{concepts_html}</ul>
|
| 1225 |
+
</div>
|
| 1226 |
+
<div class="parse-card">
|
| 1227 |
+
<h4>Claims</h4>
|
| 1228 |
+
<ul class="ref-list">{claims_html}</ul>
|
| 1229 |
+
</div>
|
| 1230 |
+
</div>
|
| 1231 |
+
<div class="timeline" style="margin-top:14px; max-height:560px; overflow:auto;">
|
| 1232 |
+
{''.join(sections_html) if sections_html else '<div class="panel" style="padding:16px;"><p>No sections extracted.</p></div>'}
|
| 1233 |
+
</div>
|
| 1234 |
+
</div>
|
| 1235 |
+
"""
|
| 1236 |
+
|
| 1237 |
+
|
| 1238 |
+
# ----------------------------
|
| 1239 |
+
# Ingest payload / learning
|
| 1240 |
+
# ----------------------------
|
| 1241 |
+
|
| 1242 |
+
def add_node(nodes_by_id: Dict[str, Dict], node_id: str, node_type: str, label: str = "", **attrs):
|
| 1243 |
+
if not node_id:
|
| 1244 |
+
return
|
| 1245 |
+
current = nodes_by_id.get(node_id, {})
|
| 1246 |
+
merged = {"id": node_id, "type": node_type, "label": label or current.get("label", node_id)}
|
| 1247 |
+
merged.update(current)
|
| 1248 |
+
for key, value in attrs.items():
|
| 1249 |
+
if value not in [None, ""]:
|
| 1250 |
+
merged[key] = value
|
| 1251 |
+
nodes_by_id[node_id] = merged
|
| 1252 |
+
|
| 1253 |
+
|
| 1254 |
+
def add_edge(edges: List[Dict], source: str, target: str, edge_type: str, **attrs):
|
| 1255 |
+
if not source or not target or source == target:
|
| 1256 |
+
return
|
| 1257 |
+
edge = {"source": source, "target": target, "type": edge_type}
|
| 1258 |
+
for key, value in attrs.items():
|
| 1259 |
+
if value not in [None, ""]:
|
| 1260 |
+
edge[key] = value
|
| 1261 |
+
edges.append(edge)
|
| 1262 |
+
|
| 1263 |
+
|
| 1264 |
+
def build_ingest_payload(query, selected_papers, parsed_pdf=None, frontier=None):
|
| 1265 |
+
nodes_by_id = {}
|
| 1266 |
+
edges = []
|
| 1267 |
+
|
| 1268 |
+
topic_id = "topic:query"
|
| 1269 |
+
add_node(nodes_by_id, topic_id, "Topic", label=query or "Research topic", query=query or "")
|
| 1270 |
+
|
| 1271 |
+
for i, paper in enumerate(selected_papers, start=1):
|
| 1272 |
+
paper_id = normalize_doi(paper.get("doi")) or ((paper.get("external_ids") or {}).get("arxiv")) or f"paper:{i}:{slugify(paper.get('title', 'paper'))[:40]}"
|
| 1273 |
+
add_node(
|
| 1274 |
+
nodes_by_id,
|
| 1275 |
+
paper_id,
|
| 1276 |
+
"Paper",
|
| 1277 |
+
label=paper.get("title") or f"Paper {i}",
|
| 1278 |
+
title=paper.get("title"),
|
| 1279 |
+
year=paper.get("year"),
|
| 1280 |
+
venue=paper.get("venue"),
|
| 1281 |
+
doi=normalize_doi(paper.get("doi")),
|
| 1282 |
+
source=paper.get("source"),
|
| 1283 |
+
url=paper.get("url"),
|
| 1284 |
+
pdf=paper.get("pdf"),
|
| 1285 |
+
score=paper.get("score"),
|
| 1286 |
+
learned_score=paper.get("learned_score", paper.get("score")),
|
| 1287 |
+
open_access=paper.get("open_access"),
|
| 1288 |
+
authors_text=paper.get("authors_text"),
|
| 1289 |
+
)
|
| 1290 |
+
add_edge(edges, topic_id, paper_id, "ABOUT", weight=paper.get("learned_score", paper.get("score", 0)))
|
| 1291 |
+
|
| 1292 |
+
for author in paper.get("authors", [])[:6]:
|
| 1293 |
+
author_id = f"author:{slugify(author)[:64]}"
|
| 1294 |
+
add_node(nodes_by_id, author_id, "Author", label=author, name=author)
|
| 1295 |
+
add_edge(edges, paper_id, author_id, "WRITTEN_BY")
|
| 1296 |
+
|
| 1297 |
+
for concept in (paper.get("concepts") or [])[:8]:
|
| 1298 |
+
concept_id = f"concept:{slugify(concept)[:72]}"
|
| 1299 |
+
add_node(nodes_by_id, concept_id, "Concept", label=concept, name=concept)
|
| 1300 |
+
add_edge(edges, paper_id, concept_id, "MENTIONS")
|
| 1301 |
+
|
| 1302 |
+
for claim in (paper.get("claims") or [])[:4]:
|
| 1303 |
+
claim_id = f"claim:{slugify(claim)[:72]}"
|
| 1304 |
+
add_node(nodes_by_id, claim_id, "Claim", label=claim[:140], text=claim)
|
| 1305 |
+
add_edge(edges, paper_id, claim_id, "ASSERTS")
|
| 1306 |
+
|
| 1307 |
+
if parsed_pdf and isinstance(parsed_pdf, dict) and parsed_pdf.get("title"):
|
| 1308 |
+
doc_id = "upload:pdf"
|
| 1309 |
+
add_node(nodes_by_id, doc_id, "UploadedPDF", label=parsed_pdf.get("title"), title=parsed_pdf.get("title"), parser=parsed_pdf.get("parser"))
|
| 1310 |
+
add_edge(edges, topic_id, doc_id, "UPLOADED_SOURCE")
|
| 1311 |
+
|
| 1312 |
+
for concept in (parsed_pdf.get("concepts") or [])[:8]:
|
| 1313 |
+
concept_id = f"concept:{slugify(concept)[:72]}"
|
| 1314 |
+
add_node(nodes_by_id, concept_id, "Concept", label=concept, name=concept)
|
| 1315 |
+
add_edge(edges, doc_id, concept_id, "MENTIONS")
|
| 1316 |
+
|
| 1317 |
+
for claim in (parsed_pdf.get("claims") or [])[:6]:
|
| 1318 |
+
claim_id = f"claim:{slugify(claim)[:72]}"
|
| 1319 |
+
add_node(nodes_by_id, claim_id, "Claim", label=claim[:140], text=claim)
|
| 1320 |
+
add_edge(edges, doc_id, claim_id, "ASSERTS")
|
| 1321 |
+
|
| 1322 |
+
for idx, ref in enumerate(parsed_pdf.get("references", [])[:20], start=1):
|
| 1323 |
+
ref_title = ref.get("title") or f"Reference {idx}"
|
| 1324 |
+
ref_doi = normalize_doi(ref.get("doi") or "")
|
| 1325 |
+
ref_id = ref_doi or f"ref:{idx}:{slugify(ref_title)[:40]}"
|
| 1326 |
+
add_node(nodes_by_id, ref_id, "Reference", label=ref_title, title=ref_title, doi=ref_doi)
|
| 1327 |
+
add_edge(edges, doc_id, ref_id, "CITES")
|
| 1328 |
+
|
| 1329 |
+
for idx, item in enumerate(ensure_list(frontier)[:18], start=1):
|
| 1330 |
+
fid = normalize_doi(item.get("doi")) or f"frontier:{idx}:{slugify(item.get('title', 'paper'))[:40]}"
|
| 1331 |
+
add_node(
|
| 1332 |
+
nodes_by_id,
|
| 1333 |
+
fid,
|
| 1334 |
+
"FrontierPaper",
|
| 1335 |
+
label=item.get("title") or f"Frontier {idx}",
|
| 1336 |
+
title=item.get("title"),
|
| 1337 |
+
frontier_score=item.get("frontier_score"),
|
| 1338 |
+
url=item.get("url"),
|
| 1339 |
+
source=item.get("source"),
|
| 1340 |
+
authors_text=item.get("authors_text"),
|
| 1341 |
+
year=item.get("year"),
|
| 1342 |
+
doi=item.get("doi"),
|
| 1343 |
+
)
|
| 1344 |
+
add_edge(edges, topic_id, fid, "FRONTIER_CANDIDATE", weight=item.get("frontier_score", item.get("learned_score", item.get("score", 0))))
|
| 1345 |
+
|
| 1346 |
+
return {"status": "ok", "nodes": list(nodes_by_id.values())[:GRAPH_MAX_NODES], "edges": edges[:GRAPH_MAX_EDGES]}
|
| 1347 |
+
|
| 1348 |
+
|
| 1349 |
+
def learn_from_payload(payload: Dict, query: str = "") -> Dict:
|
| 1350 |
+
if not payload:
|
| 1351 |
+
return GRAPH_MEMORY
|
| 1352 |
+
|
| 1353 |
+
GRAPH_MEMORY["queries"].append(query or "")
|
| 1354 |
+
GRAPH_MEMORY["events"].append({
|
| 1355 |
+
"ts": time.time(),
|
| 1356 |
+
"query": query or "",
|
| 1357 |
+
"nodes": len(payload.get("nodes", [])),
|
| 1358 |
+
"edges": len(payload.get("edges", [])),
|
| 1359 |
+
})
|
| 1360 |
+
GRAPH_MEMORY["payloads"].append(payload)
|
| 1361 |
+
|
| 1362 |
+
for node in payload.get("nodes", []):
|
| 1363 |
+
node_id = node.get("id")
|
| 1364 |
+
if not node_id:
|
| 1365 |
+
continue
|
| 1366 |
+
GRAPH_MEMORY["nodes"][node_id] = node
|
| 1367 |
+
node_type = (node.get("type") or "").lower()
|
| 1368 |
+
if node_type in {"paper", "frontierpaper"}:
|
| 1369 |
+
GRAPH_MEMORY["papers"][node_id] = node
|
| 1370 |
+
if node_type == "concept" and node.get("label"):
|
| 1371 |
+
GRAPH_MEMORY["concept_counts"][node["label"].lower()] += 1
|
| 1372 |
+
if node_type == "claim" and node.get("label"):
|
| 1373 |
+
GRAPH_MEMORY["claim_counts"][node["label"].lower()] += 1
|
| 1374 |
+
|
| 1375 |
+
GRAPH_MEMORY["edges"].extend(payload.get("edges", []))
|
| 1376 |
+
GRAPH_MEMORY["edges"] = GRAPH_MEMORY["edges"][:GRAPH_MAX_EDGES]
|
| 1377 |
+
return GRAPH_MEMORY
|
| 1378 |
+
|
| 1379 |
+
|
| 1380 |
+
def export_learning_state() -> str:
|
| 1381 |
+
snapshot = {
|
| 1382 |
+
"papers": list(GRAPH_MEMORY["papers"].values())[:60],
|
| 1383 |
+
"nodes": list(GRAPH_MEMORY["nodes"].values())[:250],
|
| 1384 |
+
"edges": GRAPH_MEMORY["edges"][:500],
|
| 1385 |
+
"top_concepts": GRAPH_MEMORY["concept_counts"].most_common(24),
|
| 1386 |
+
"top_claims": GRAPH_MEMORY["claim_counts"].most_common(24),
|
| 1387 |
+
"queries": GRAPH_MEMORY["queries"][-20:],
|
| 1388 |
+
"events": GRAPH_MEMORY["events"][-20:],
|
| 1389 |
+
"frontier": GRAPH_MEMORY["frontier"][:24],
|
| 1390 |
+
}
|
| 1391 |
+
return json.dumps(snapshot, indent=2, ensure_ascii=False)
|
| 1392 |
+
|
| 1393 |
+
|
| 1394 |
+
# ----------------------------
|
| 1395 |
+
# Frontier expansion
|
| 1396 |
+
# ----------------------------
|
| 1397 |
+
|
| 1398 |
+
def score_frontier_candidate(query: str, seed_concepts: List[str], paper: Dict[str, Any]) -> Dict[str, Any]:
|
| 1399 |
+
title = paper.get("title", "")
|
| 1400 |
+
abstract = paper.get("abstract", "") or paper.get("summary", "")
|
| 1401 |
+
venue = paper.get("venue", "")
|
| 1402 |
+
base_text = " ".join([title, abstract, venue])
|
| 1403 |
+
rel = text_overlap_score(query, base_text)
|
| 1404 |
+
concept_overlap = text_overlap_score(" ".join(seed_concepts), " ".join(paper.get("concepts") or [])) if seed_concepts else 0.0
|
| 1405 |
+
recency = compute_recency_bonus(paper.get("year"))
|
| 1406 |
+
doi_bonus = 0.02 if paper.get("doi") else 0.0
|
| 1407 |
+
oa_bonus = 0.03 if paper.get("open_access") else 0.0
|
| 1408 |
+
score = float(paper.get("learned_score", paper.get("score", 0))) + rel * 0.45 + concept_overlap * 0.22 + recency + doi_bonus + oa_bonus
|
| 1409 |
+
paper["frontier_score"] = round(score, 4)
|
| 1410 |
+
paper["frontier_relevance"] = round(rel, 4)
|
| 1411 |
+
paper["frontier_concept_overlap"] = round(concept_overlap, 4)
|
| 1412 |
+
return paper
|
| 1413 |
+
|
| 1414 |
+
|
| 1415 |
+
def propose_expansion_queries(query: str, papers: List[Dict], parsed_state: Optional[Dict] = None, limit: int = GRAPH_MAX_EXPANSIONS) -> List[str]:
|
| 1416 |
+
concept_pool = []
|
| 1417 |
+
venue_pool = []
|
| 1418 |
+
for paper in papers[:8]:
|
| 1419 |
+
concept_pool.extend((paper.get("concepts") or [])[:4])
|
| 1420 |
+
if paper.get("venue"):
|
| 1421 |
+
venue_pool.append(paper["venue"])
|
| 1422 |
+
if parsed_state and isinstance(parsed_state, dict):
|
| 1423 |
+
concept_pool.extend((parsed_state.get("concepts") or [])[:6])
|
| 1424 |
+
|
| 1425 |
+
ranked_concepts = [c for c, _ in Counter([norm_text(c).lower() for c in concept_pool if c]).most_common(limit * 2)]
|
| 1426 |
+
expansions = [norm_text(query)] if query else []
|
| 1427 |
+
for concept in ranked_concepts:
|
| 1428 |
+
if concept:
|
| 1429 |
+
expansions.append(f"{query} {concept}".strip())
|
| 1430 |
+
for venue in unique_keep_order(venue_pool)[:2]:
|
| 1431 |
+
if venue:
|
| 1432 |
+
expansions.append(f"{query} {venue}".strip())
|
| 1433 |
+
return unique_keep_order(expansions)[:limit]
|
| 1434 |
+
|
| 1435 |
+
|
| 1436 |
+
def frontier_expand(query: str, sources: List[str], selected_papers: List[Dict], parsed_state: Optional[Dict] = None, per_query: int = 4) -> List[Dict]:
|
| 1437 |
+
seed_concepts = []
|
| 1438 |
+
for p in selected_papers[:6]:
|
| 1439 |
+
seed_concepts.extend((p.get("concepts") or [])[:4])
|
| 1440 |
+
if parsed_state and isinstance(parsed_state, dict):
|
| 1441 |
+
seed_concepts.extend((parsed_state.get("concepts") or [])[:6])
|
| 1442 |
+
|
| 1443 |
+
expansion_queries = propose_expansion_queries(query, selected_papers, parsed_state=parsed_state, limit=GRAPH_MAX_EXPANSIONS)
|
| 1444 |
+
frontier = []
|
| 1445 |
+
for eq in expansion_queries:
|
| 1446 |
+
try:
|
| 1447 |
+
items = discover_papers(eq, "topic", sources, max_results=per_query)
|
| 1448 |
+
for item in items:
|
| 1449 |
+
frontier.append(score_frontier_candidate(query or eq, seed_concepts, item))
|
| 1450 |
+
except Exception:
|
| 1451 |
+
continue
|
| 1452 |
+
|
| 1453 |
+
frontier = dedupe_papers(frontier)
|
| 1454 |
+
frontier.sort(key=lambda x: float(x.get("frontier_score", x.get("learned_score", x.get("score", 0)))), reverse=True)
|
| 1455 |
+
GRAPH_MEMORY["frontier"] = frontier[: GRAPH_MAX_EXPANSIONS * per_query]
|
| 1456 |
+
return GRAPH_MEMORY["frontier"]
|
| 1457 |
+
|
| 1458 |
+
|
| 1459 |
+
def autonomous_expand_into_markdown(query, payload, parsed_state=None):
|
| 1460 |
+
frontier = GRAPH_MEMORY.get("frontier") or []
|
| 1461 |
+
lines = [
|
| 1462 |
+
"### Autonomous expansion plan",
|
| 1463 |
+
"",
|
| 1464 |
+
f"- Seed query: {query or 'Research topic'}",
|
| 1465 |
+
f"- Current nodes: {len(payload.get('nodes', [])) if isinstance(payload, dict) else 0}",
|
| 1466 |
+
f"- Current edges: {len(payload.get('edges', [])) if isinstance(payload, dict) else 0}",
|
| 1467 |
+
f"- Frontier candidates: {len(frontier)}",
|
| 1468 |
+
]
|
| 1469 |
+
|
| 1470 |
+
proposed = propose_expansion_queries(query or "", list(GRAPH_MEMORY.get("papers", {}).values())[:8], parsed_state=parsed_state, limit=GRAPH_MAX_EXPANSIONS)
|
| 1471 |
+
if proposed:
|
| 1472 |
+
lines.extend(["", "#### Proposed next queries", ""])
|
| 1473 |
+
lines.extend([f"- {q}" for q in proposed])
|
| 1474 |
+
|
| 1475 |
+
if frontier:
|
| 1476 |
+
lines.extend(["", "#### Top frontier papers", ""])
|
| 1477 |
+
for item in frontier[:8]:
|
| 1478 |
+
lines.append(
|
| 1479 |
+
f"- {item.get('title', 'Untitled')} ({item.get('source', 'unknown')}) — frontier score {item.get('frontier_score', item.get('learned_score', item.get('score', 0)))}"
|
| 1480 |
+
)
|
| 1481 |
+
return "\n".join(lines)
|
| 1482 |
+
|
| 1483 |
+
|
| 1484 |
+
# ----------------------------
|
| 1485 |
+
# Selection / ingest
|
| 1486 |
+
# ----------------------------
|
| 1487 |
+
|
| 1488 |
+
def resolve_selected_papers(selected_indices, papers_state):
|
| 1489 |
+
papers = ensure_list(papers_state)
|
| 1490 |
+
selected_indices = ensure_list(selected_indices)
|
| 1491 |
+
selected = []
|
| 1492 |
+
if not selected_indices:
|
| 1493 |
+
return selected
|
| 1494 |
+
|
| 1495 |
+
value_map = {paper_choice_value(i, paper): paper for i, paper in enumerate(papers)}
|
| 1496 |
+
label_map = {paper_choice_label(i, paper): paper for i, paper in enumerate(papers)}
|
| 1497 |
+
|
| 1498 |
+
for idx in selected_indices:
|
| 1499 |
+
try:
|
| 1500 |
+
if isinstance(idx, int):
|
| 1501 |
+
if 0 <= idx < len(papers):
|
| 1502 |
+
selected.append(papers[idx])
|
| 1503 |
+
continue
|
| 1504 |
+
idx_str = str(idx)
|
| 1505 |
+
if idx_str in value_map:
|
| 1506 |
+
selected.append(value_map[idx_str])
|
| 1507 |
+
continue
|
| 1508 |
+
if idx_str.isdigit():
|
| 1509 |
+
num = int(idx_str)
|
| 1510 |
+
if 0 <= num < len(papers):
|
| 1511 |
+
selected.append(papers[num])
|
| 1512 |
+
continue
|
| 1513 |
+
if "|" in idx_str:
|
| 1514 |
+
left = idx_str.split("|", 1)[0]
|
| 1515 |
+
if left.isdigit():
|
| 1516 |
+
num = int(left)
|
| 1517 |
+
if 0 <= num < len(papers):
|
| 1518 |
+
selected.append(papers[num])
|
| 1519 |
+
continue
|
| 1520 |
+
if idx_str in label_map:
|
| 1521 |
+
selected.append(label_map[idx_str])
|
| 1522 |
+
continue
|
| 1523 |
+
except Exception:
|
| 1524 |
+
continue
|
| 1525 |
+
|
| 1526 |
+
out = []
|
| 1527 |
+
seen = set()
|
| 1528 |
+
for paper in selected:
|
| 1529 |
+
key = paper_identity_key(paper)
|
| 1530 |
+
if key not in seen:
|
| 1531 |
+
seen.add(key)
|
| 1532 |
+
out.append(paper)
|
| 1533 |
+
return out
|
| 1534 |
+
|
| 1535 |
+
|
| 1536 |
+
def build_ingest_status_markdown(query_text: str, payload: Dict, selected: List[Dict], parsed_state: Optional[Dict], frontier: List[Dict]) -> str:
|
| 1537 |
+
payload = payload or {"nodes": [], "edges": []}
|
| 1538 |
+
nodes = payload.get("nodes", [])
|
| 1539 |
+
edges = payload.get("edges", [])
|
| 1540 |
+
counts = Counter((node.get("type") or "Unknown") for node in nodes)
|
| 1541 |
+
|
| 1542 |
+
node_lines = []
|
| 1543 |
+
for idx, node in enumerate(nodes[:24], start=1):
|
| 1544 |
+
label = node.get("label") or node.get("title") or node.get("id")
|
| 1545 |
+
node_lines.append(f"- {idx}. [{node.get('type', 'Node')}] {label}")
|
| 1546 |
+
|
| 1547 |
+
edge_lines = []
|
| 1548 |
+
for idx, edge in enumerate(edges[:30], start=1):
|
| 1549 |
+
edge_lines.append(f"- {idx}. {edge.get('source')} --{edge.get('type', 'RELATES_TO')}--> {edge.get('target')}")
|
| 1550 |
+
|
| 1551 |
+
return "\n".join([
|
| 1552 |
+
"### Graph ingest status",
|
| 1553 |
+
"",
|
| 1554 |
+
f"- Topic: {query_text}",
|
| 1555 |
+
f"- Selected papers ingested: {len(selected)}",
|
| 1556 |
+
f"- Uploaded PDF parsed: {'Yes' if parsed_state and isinstance(parsed_state, dict) and parsed_state.get('title') else 'No'}",
|
| 1557 |
+
f"- Frontier candidates added: {len(frontier)}",
|
| 1558 |
+
f"- Total nodes created: {len(nodes)}",
|
| 1559 |
+
f"- Total edges created: {len(edges)}",
|
| 1560 |
+
f"- Node breakdown: {', '.join([f'{k}={v}' for k, v in counts.items()]) if counts else 'None'}",
|
| 1561 |
+
"",
|
| 1562 |
+
"### Nodes",
|
| 1563 |
+
*(node_lines or ["- None"]),
|
| 1564 |
+
"",
|
| 1565 |
+
"### Edges",
|
| 1566 |
+
*(edge_lines or ["- None"]),
|
| 1567 |
+
])
|
| 1568 |
+
|
| 1569 |
+
|
| 1570 |
+
def ingest_selected_papers(query, selected_indices, papers_state, pdf_file, parsed_state):
|
| 1571 |
+
papers = ensure_list(papers_state)
|
| 1572 |
+
selected = resolve_selected_papers(selected_indices, papers)
|
| 1573 |
+
uploaded_name = Path(getattr(pdf_file, "name", str(pdf_file))).name if pdf_file else None
|
| 1574 |
+
|
| 1575 |
+
if not selected and papers:
|
| 1576 |
+
selected = papers[:3]
|
| 1577 |
+
|
| 1578 |
+
if not selected and parsed_state and isinstance(parsed_state, dict) and parsed_state.get("title"):
|
| 1579 |
+
selected = []
|
| 1580 |
+
|
| 1581 |
+
if not selected and not (parsed_state and isinstance(parsed_state, dict) and parsed_state.get("title")):
|
| 1582 |
+
graph_html = build_learning_graph_html([], [], "Self-Learning Knowledge Graph")
|
| 1583 |
+
empty_payload = {"status": "empty", "nodes": [], "edges": []}
|
| 1584 |
+
return (
|
| 1585 |
+
graph_html,
|
| 1586 |
+
"### Graph ingest status\n\n- No papers were selected and no parsed PDF is available.\n- Select papers or parse an uploaded PDF first.",
|
| 1587 |
+
empty_payload,
|
| 1588 |
+
)
|
| 1589 |
+
|
| 1590 |
+
query_text = norm_text(query or "")
|
| 1591 |
+
if not query_text and isinstance(parsed_state, dict):
|
| 1592 |
+
query_text = parsed_state.get("title") or "Research topic"
|
| 1593 |
+
if not query_text:
|
| 1594 |
+
query_text = "Research topic"
|
| 1595 |
+
|
| 1596 |
+
selected = [enrich_paper_semantics(query_text, paper) for paper in selected]
|
| 1597 |
+
frontier = frontier_expand(query_text, DEFAULT_SOURCES, selected or papers[:3], parsed_state=parsed_state if isinstance(parsed_state, dict) else None, per_query=3)
|
| 1598 |
+
|
| 1599 |
+
graph_nodes, graph_edges = graph_from_selected(
|
| 1600 |
+
query_text,
|
| 1601 |
+
selected,
|
| 1602 |
+
uploaded_name,
|
| 1603 |
+
parsed_state if isinstance(parsed_state, dict) else None,
|
| 1604 |
+
frontier=frontier,
|
| 1605 |
+
)
|
| 1606 |
+
graph_html = build_learning_graph_html(graph_nodes, graph_edges, "Selected Research Graph")
|
| 1607 |
+
|
| 1608 |
+
payload = build_ingest_payload(
|
| 1609 |
+
query_text,
|
| 1610 |
+
selected,
|
| 1611 |
+
parsed_state if isinstance(parsed_state, dict) else None,
|
| 1612 |
+
frontier=frontier,
|
| 1613 |
+
)
|
| 1614 |
+
learn_from_payload(payload, query=query_text)
|
| 1615 |
+
|
| 1616 |
+
status_md = build_ingest_status_markdown(
|
| 1617 |
+
query_text,
|
| 1618 |
+
payload,
|
| 1619 |
+
selected,
|
| 1620 |
+
parsed_state if isinstance(parsed_state, dict) else None,
|
| 1621 |
+
frontier,
|
| 1622 |
+
)
|
| 1623 |
+
|
| 1624 |
+
return graph_html, status_md, payload
|
| 1625 |
+
|
| 1626 |
+
|
| 1627 |
+
# ----------------------------
|
| 1628 |
+
# Discovery
|
| 1629 |
+
# ----------------------------
|
| 1630 |
+
|
| 1631 |
+
def summarize_learning_state(query_text, papers, selected_sources):
|
| 1632 |
+
concept_pool = []
|
| 1633 |
+
for paper in papers[:8]:
|
| 1634 |
+
concept_pool.extend((paper.get("concepts") or [])[:4])
|
| 1635 |
+
top_concepts = [c for c, _ in Counter([c.lower() for c in concept_pool]).most_common(8)]
|
| 1636 |
+
return (
|
| 1637 |
+
"### Discovery results\n\n"
|
| 1638 |
+
f"- Query: {query_text}\n"
|
| 1639 |
+
f"- Sources: {', '.join(selected_sources)}\n"
|
| 1640 |
+
f"- Candidates found: {len(papers)}\n"
|
| 1641 |
+
f"- Top learned concepts: {', '.join(top_concepts) if top_concepts else 'None'}\n"
|
| 1642 |
+
"- Select papers below, or leave selection empty to auto-ingest the top papers.\n"
|
| 1643 |
+
)
|
| 1644 |
+
|
| 1645 |
+
|
| 1646 |
+
def run_paper_discovery(query, search_mode, sources, pdf_file):
|
| 1647 |
+
query_text = norm_text(query or "")
|
| 1648 |
+
selected_sources = ensure_list(sources) or DEFAULT_SOURCES
|
| 1649 |
+
|
| 1650 |
+
if not query_text and not pdf_file:
|
| 1651 |
+
empty_graph = build_learning_graph_html([], [], "Self-Learning Knowledge Graph")
|
| 1652 |
+
return (
|
| 1653 |
+
empty_graph,
|
| 1654 |
+
'<div class="panel papers-panel" style="padding:18px"><p>Enter a topic, title, DOI, link, or upload a PDF to start learning.</p></div>',
|
| 1655 |
+
build_journal_html("biomaterials cardiac repair"),
|
| 1656 |
+
"No PDF uploaded yet.",
|
| 1657 |
+
gr.update(choices=[], value=[]),
|
| 1658 |
+
[],
|
| 1659 |
+
"### No discovery results yet.",
|
| 1660 |
+
)
|
| 1661 |
+
|
| 1662 |
+
if not query_text and pdf_file:
|
| 1663 |
+
uploaded_name = Path(getattr(pdf_file, "name", str(pdf_file))).name
|
| 1664 |
+
graph_nodes, graph_edges = build_learning_graph_state("", [], uploaded_name)
|
| 1665 |
+
return (
|
| 1666 |
+
build_learning_graph_html(graph_nodes, graph_edges, "Uploaded PDF Waiting for Parse"),
|
| 1667 |
+
'<div class="panel papers-panel" style="padding:18px"><p>No query yet. Parse the uploaded PDF or enter a research topic to begin discovery.</p></div>',
|
| 1668 |
+
build_journal_html("biomaterials cardiac repair"),
|
| 1669 |
+
uploaded_pdf_summary(pdf_file),
|
| 1670 |
+
gr.update(choices=[], value=[]),
|
| 1671 |
+
[],
|
| 1672 |
+
"### Upload detected.\n\n- Parse the PDF to extract structure.\n- Or enter a topic to start discovery.",
|
| 1673 |
+
)
|
| 1674 |
+
|
| 1675 |
+
uploaded_name = Path(getattr(pdf_file, "name", str(pdf_file))).name if pdf_file else None
|
| 1676 |
+
|
| 1677 |
+
try:
|
| 1678 |
+
papers = discover_papers(query_text, search_mode, selected_sources, max_results=GRAPH_MAX_RESULTS)
|
| 1679 |
+
graph_nodes, graph_edges = build_learning_graph_state(query_text, papers[:6], uploaded_name)
|
| 1680 |
+
graph_html = build_learning_graph_html(graph_nodes, graph_edges, "Self-Learning Knowledge Graph")
|
| 1681 |
+
papers_html = format_papers_html(papers)
|
| 1682 |
+
journals_html = build_journal_html(query_text or "biomaterials cardiac repair")
|
| 1683 |
+
pdf_summary = uploaded_pdf_summary(pdf_file)
|
| 1684 |
+
choices = format_selection_choices(papers)
|
| 1685 |
+
status_md = summarize_learning_state(query_text, papers, selected_sources)
|
| 1686 |
+
return (
|
| 1687 |
+
graph_html,
|
| 1688 |
+
papers_html,
|
| 1689 |
+
journals_html,
|
| 1690 |
+
pdf_summary,
|
| 1691 |
+
gr.update(choices=choices, value=[]),
|
| 1692 |
+
papers,
|
| 1693 |
+
status_md,
|
| 1694 |
+
)
|
| 1695 |
+
except Exception as e:
|
| 1696 |
+
graph_nodes, graph_edges = build_learning_graph_state(query_text, [], uploaded_name)
|
| 1697 |
+
error_html = f'<div class="panel papers-panel" style="padding:18px"><p>Paper search failed: {safe_text(str(e))}</p></div>'
|
| 1698 |
+
return (
|
| 1699 |
+
build_learning_graph_html(graph_nodes, graph_edges),
|
| 1700 |
+
error_html,
|
| 1701 |
+
build_journal_html(query_text or "biomaterials cardiac repair"),
|
| 1702 |
+
uploaded_pdf_summary(pdf_file),
|
| 1703 |
+
gr.update(choices=[], value=[]),
|
| 1704 |
+
[],
|
| 1705 |
+
f"### Discovery failed.\n\n- Error: {safe_text(str(e))}",
|
| 1706 |
+
)
|
| 1707 |
+
|
| 1708 |
+
|
| 1709 |
+
__all__ = [
|
| 1710 |
+
"SEARCH_MODES",
|
| 1711 |
+
"SOURCE_OPTIONS",
|
| 1712 |
+
"DEFAULT_SOURCES",
|
| 1713 |
+
"PDF_PARSERS",
|
| 1714 |
+
"GRAPH_MEMORY",
|
| 1715 |
+
"discover_papers",
|
| 1716 |
+
"run_paper_discovery",
|
| 1717 |
+
"parse_uploaded_pdf",
|
| 1718 |
+
"render_parse_result",
|
| 1719 |
+
"ingest_selected_papers",
|
| 1720 |
+
"build_ingest_payload",
|
| 1721 |
+
"learn_from_payload",
|
| 1722 |
+
"frontier_expand",
|
| 1723 |
+
"autonomous_expand_into_markdown",
|
| 1724 |
+
"export_learning_state",
|
| 1725 |
+
"format_frontier_html",
|
| 1726 |
+
"build_learning_graph_html",
|
| 1727 |
+
"build_journal_html",
|
| 1728 |
+
"safe_text",
|
| 1729 |
+
]
|