Sync from GitHub via hub-sync
Browse files- Annexure_A_Inventory_of_159_IS_Theories.txt +125 -0
- Annexure_B_Agentic_AI_Programming_Guidelines_Mandatory.txt +125 -0
- RCTO_prompt_1.txt +5 -0
- RCTO_prompt_2.txt +9 -0
- RCTO_prompt_3.txt +15 -0
- README.md +8 -6
- agents/__init__.py +0 -0
- agents/council_agent.py +53 -0
- agents/tccm_agent.py +76 -0
- app.py +107 -0
- main.py +6 -0
- models.py +15 -0
- prompts.py +143 -0
- pyproject.toml +16 -0
- requirements.txt +229 -0
- utils/__init__.py +0 -0
- utils/excel_writer.py +43 -0
- utils/pdf_reader.py +28 -0
- uv.lock +0 -0
Annexure_A_Inventory_of_159_IS_Theories.txt
ADDED
|
@@ -0,0 +1,125 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Annexure A — Inventory of 159 IS Theories
|
| 2 |
+
The list below is the canonical reference inventory of named IS theories to be
|
| 3 |
+
uploaded into NotebookLM as IS_Theories.xlsx alongside the 100 paper PDFs.
|
| 4 |
+
The file is enclosed with this guidance note. Theories are listed alphabetically by
|
| 5 |
+
name; the S.No. column matches the spreadsheet for cross-reference. The list is
|
| 6 |
+
a reference, not a ceiling — students must flag any theory found in their corpus
|
| 7 |
+
that is not on this inventory.
|
| 8 |
+
S.No. — Theory Name S.No. — Theory Name
|
| 9 |
+
1. Absorptive capacity theory 81. Organizational knowledge creation
|
| 10 |
+
2. Activity Theory 82. Organizational learning theory
|
| 11 |
+
3. Actor network theory 83. Platform Ecosystem Theory
|
| 12 |
+
4. Accountability theory 84. Portfolio theory
|
| 13 |
+
5. Adaptive structuration theory 85. Privacy Calculus Theory
|
| 14 |
+
6. Administrative behavior, theory of 86. Process virtualization theory
|
| 15 |
+
7. Adaptive enterprise architecture
|
| 16 |
+
theory
|
| 17 |
+
87. People-Process-Data-Technology
|
| 18 |
+
Framework
|
| 19 |
+
8. Affordance-actualization Theory 88. Prospect theory
|
| 20 |
+
9. Agency theory 89. Protection motivation theory
|
| 21 |
+
10. Appraisal Theory 90. Psychological ownership framework
|
| 22 |
+
11. Argumentation theory 91. Punctuated equilibrium theory
|
| 23 |
+
12. Attachment-Theory 92. Rational Choice Theory
|
| 24 |
+
13. Behavioral decision theory 93. Real options theory
|
| 25 |
+
14. Belief Action Outcome Framework 94. Representation Theory
|
| 26 |
+
15. Big Five Model of Personality 95. Resource-based view of the firm
|
| 27 |
+
16. Boundary object theory 96. Resource dependency theory
|
| 28 |
+
17. Chaos theory 97. Resource Orchestration Theory
|
| 29 |
+
18. Cognitive dissonance theory 98. Role Theory
|
| 30 |
+
19. Cognitive fit theory 99. Selective organizational information
|
| 31 |
+
privacy and security violations model
|
| 32 |
+
(SOIPSVM)
|
| 33 |
+
Page 20 of 27Student Guidance Note — TCCM & Topic Modelling
|
| 34 |
+
S.No. — Theory Name S.No. — Theory Name
|
| 35 |
+
20. Cognitive load theory 100. Self determination theory
|
| 36 |
+
21. Community Of Practice Theory 101. Self-efficacy theory
|
| 37 |
+
22. Competitive strategy (Porter) 102. Self Presentation Theory
|
| 38 |
+
23. Complexity theory 103. Semantic theory of survey response
|
| 39 |
+
24. Compliance regulatory theory 104. SERVQUAL
|
| 40 |
+
25. Conservation of resources theory 105. Signaling theory
|
| 41 |
+
26. Contingency theory 106. Social Bond Theory
|
| 42 |
+
27. Critical Mass Theory of Interactive
|
| 43 |
+
Media
|
| 44 |
+
107. Social capital theory
|
| 45 |
+
28. Critical realism theory 108. Social cognitive theory
|
| 46 |
+
29. Critical social theory 109. Social Comparison Theory
|
| 47 |
+
30. Critical success factors, theory of 110. Social Contagion Theory
|
| 48 |
+
31. Customer based Discrepancy Theory 111. Socioemotional Selectivity Theory
|
| 49 |
+
32. Customer Focus Theory 112. Social exchange theory
|
| 50 |
+
33. CYNEFIN framework 113. Social Identity Theory
|
| 51 |
+
34. Deferred action, theory of 114. Social Influence Theory (of Kelman)
|
| 52 |
+
35. Delone and McLean IS success
|
| 53 |
+
model
|
| 54 |
+
115. Social Information Processing
|
| 55 |
+
Theory (of Walther)
|
| 56 |
+
36. Design Theory 116. Social learning theory
|
| 57 |
+
37. Diffusion of innovations theory 117. Social media engagement theory
|
| 58 |
+
38. Distributed Cognition Theory 118. Social network theory
|
| 59 |
+
39. Dynamic capabilities 119. Social Penetration Theory
|
| 60 |
+
40. Elaboration likelihood model 120. Social Presence Theory
|
| 61 |
+
41. Embodied social presence theory 121. Social shaping of technology
|
| 62 |
+
42. Equity theory 122. Sociomaterialism Theory
|
| 63 |
+
43. Evolutionary theory 123. Socio-technical theory
|
| 64 |
+
44. Expectation confirmation theory 124. Soft systems theory
|
| 65 |
+
45. Feminism theory 125. Speech and Act Theory
|
| 66 |
+
46. Fit-Viability theory 126. Stakeholder theory
|
| 67 |
+
Page 21 of 27Student Guidance Note — TCCM & Topic Modelling
|
| 68 |
+
S.No. — Theory Name S.No. — Theory Name
|
| 69 |
+
47. Flow theory 127. Structuration theory
|
| 70 |
+
48. Game theory 128. Structured process modeling theory
|
| 71 |
+
(SPMT)
|
| 72 |
+
49. Garbage can theory 129. Success Management Theory
|
| 73 |
+
50. General systems theory 130. Task closure theory
|
| 74 |
+
51. General deterrence theory 131. Task-technology fit
|
| 75 |
+
52. General Strain theory 132. Team resilience theory
|
| 76 |
+
53. Goal Contagion Theory 133. Technological frames of reference
|
| 77 |
+
54. Governance, Risk and Compliance 134. Technology acceptance model
|
| 78 |
+
55. Hedonic-motivation system adoption
|
| 79 |
+
model (HMSAM)
|
| 80 |
+
135. Technology affordances and
|
| 81 |
+
constraints theory
|
| 82 |
+
56. Hermeneutics 136. Technology dominance, theory of
|
| 83 |
+
57. Illusion of control 137. Technology-organization-
|
| 84 |
+
environment framework
|
| 85 |
+
58. Impression management, theory of 138. Technology Threat Avoidance
|
| 86 |
+
Theory
|
| 87 |
+
59. Information processing theory 139. Theory of collective action
|
| 88 |
+
60. Information warfare 140. Theory of Hypercompetition
|
| 89 |
+
61. Institutional theory 141. Theory of Interactive Media Effects
|
| 90 |
+
(TIME)
|
| 91 |
+
62. International information systems
|
| 92 |
+
theory
|
| 93 |
+
142. Theory of Interpersonal Behavior
|
| 94 |
+
63. Internationalization Theory 143. Theory of organizational creativity
|
| 95 |
+
64. Innovation Resistance Theory 144. Theory of Organizational
|
| 96 |
+
Sensemaking
|
| 97 |
+
65. Keller's Motivational Model 145. Theory of planned behavior
|
| 98 |
+
66. Kohlberg's theory of Moral
|
| 99 |
+
Development
|
| 100 |
+
146. Theory of reasoned action
|
| 101 |
+
67. Knowledge-based theory of the firm 147. Theory of slack resources
|
| 102 |
+
68. Knowledge Security Theory 148. Theory of swift trust
|
| 103 |
+
69. Language action perspective 149. Theory of Network Externalities
|
| 104 |
+
Page 22 of 27Student Guidance Note — TCCM & Topic Modelling
|
| 105 |
+
S.No. — Theory Name S.No. — Theory Name
|
| 106 |
+
70. Information asymmetry theory (lemon
|
| 107 |
+
market)
|
| 108 |
+
150. Transaction cost economics
|
| 109 |
+
71. Management fashion theory 151. Transactive memory theory
|
| 110 |
+
72. Media richness theory 152. Uncanny Valley Theory
|
| 111 |
+
73. Media synchronicity theory 153. Uncertainty Reduction Theory
|
| 112 |
+
74. Modal aspects, theory of 154. Unified theory of acceptance and
|
| 113 |
+
use of technology
|
| 114 |
+
75. Multi-attribute utility theory 155. Upper Echelons Theory
|
| 115 |
+
76. Multi-motive information systems
|
| 116 |
+
continuance model (MISC)
|
| 117 |
+
156. Usage control model
|
| 118 |
+
77. Norm Activation Theory 157. Value Theory
|
| 119 |
+
78. Organizational Ambidexterity Theory 158. Work systems theory
|
| 120 |
+
79. Organizational culture theory 159. Yield shift theory of satisfaction
|
| 121 |
+
80. Organizational information
|
| 122 |
+
processing theory
|
| 123 |
+
Source: Compiled from the IS research methods literature for use in this course.
|
| 124 |
+
The file IS_Theories.xlsx (enclosed) contains the same list in spreadsheet form
|
| 125 |
+
for direct upload to any LLM used in the Council.
|
Annexure_B_Agentic_AI_Programming_Guidelines_Mandatory.txt
ADDED
|
@@ -0,0 +1,125 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Annexure B — Agentic AI Programming
|
| 2 |
+
Guidelines (Mandatory)
|
| 3 |
+
B.1 The principle
|
| 4 |
+
An agentic AI submission is a thin declarative orchestration around LLM calls —
|
| 5 |
+
not a 500-line Python script with one LLM call buried inside it. The LLM is the
|
| 6 |
+
reasoning engine; your code is the wiring. If your submission ratio is 99% Python,
|
| 7 |
+
1% LLM, you have not built an agent — you have built a script with an LLM
|
| 8 |
+
cameo. The rules below enforce that discipline. They apply to every executable
|
| 9 |
+
artefact in the M3 and M4 submissions: the TCCM extraction agent, the topic-
|
| 10 |
+
modelling agent, the LLM Council consolidator, the Bayesian optimiser, and the
|
| 11 |
+
Gradio UI.
|
| 12 |
+
B.2 Forbidden patterns
|
| 13 |
+
• No `if/elif/else` for content decisions. If you are deciding what counts as
|
| 14 |
+
a theory, what category a paper belongs to, or whether a cluster label is
|
| 15 |
+
plausible — that is the LLM's job. Encode it in the prompt, not in a Python
|
| 16 |
+
conditional.
|
| 17 |
+
• No `for` / `while` loops for agent work. Iteration over papers, clusters,
|
| 18 |
+
or Bayesian trials happens via LangGraph's `Send` API (fan-out) and
|
| 19 |
+
conditional edges. Loops are permitted only for trivially mechanical I/O —
|
| 20 |
+
reading a directory listing, walking a JSON tree.
|
| 21 |
+
• No HTML, no CSS, no JavaScript. All user interfaces are Gradio
|
| 22 |
+
components. If you find yourself writing `<div>`, `style=`, or `.css` files you
|
| 23 |
+
are wasting time that belongs in the prompt.
|
| 24 |
+
• No 500-line scripts with a single LLM call. If your `.py` file is over 200
|
| 25 |
+
lines and contains one `llm.invoke(...)` line, the code is procedural Python
|
| 26 |
+
wearing an LLM costume. Refactor.
|
| 27 |
+
B.3 Mandated patterns
|
| 28 |
+
Page 24 of 27Student Guidance Note — TCCM & Topic Modelling
|
| 29 |
+
• LangGraph is mandatory for every agent. Define a `StateGraph` with
|
| 30 |
+
typed state, register node functions, connect with edges, call `.compile()`.
|
| 31 |
+
The compiled graph is what your Gradio UI invokes — nothing else.
|
| 32 |
+
• Each node has a single responsibility — call an LLM, run a tool, or
|
| 33 |
+
update state. Nodes do not contain business logic; they call LLMs that
|
| 34 |
+
contain the logic.
|
| 35 |
+
• Branching = conditional edges. Replace `if x: do_a() else: do_b()` with
|
| 36 |
+
`graph.add_conditional_edges(node, router, {'a': node_a, 'b': node_b})`.
|
| 37 |
+
The `router` lambda is the only place branching is permitted and must be
|
| 38 |
+
a single expression.
|
| 39 |
+
• Iteration = `Send` fan-out + reducer. To process 100 papers, emit 100
|
| 40 |
+
`Send` events from a dispatcher node; each is processed in parallel; an
|
| 41 |
+
accumulating reducer aggregates them into state. There is no `for paper
|
| 42 |
+
in papers:` anywhere.
|
| 43 |
+
• Gradio Blocks for every UI. `gr.Tabs` for the four Council sheets,
|
| 44 |
+
`gr.Dataframe` for tabular outputs, `gr.JSON` for structured returns,
|
| 45 |
+
`gr.Plot` for HDBSCAN visualisations. Zero HTML.
|
| 46 |
+
B.4 Code-volume ceilings
|
| 47 |
+
Submissions exceeding these ceilings without a documented justification lose
|
| 48 |
+
marks. Justification must appear as a comment in the file (`# WAIVER: library X
|
| 49 |
+
forces …`).
|
| 50 |
+
Artefact Hard ceiling (lines of Python)
|
| 51 |
+
TCCM extraction agent (Theory +
|
| 52 |
+
Characteristics prompts wired through
|
| 53 |
+
LangGraph)
|
| 54 |
+
≤ 80
|
| 55 |
+
Topic-modelling agent (SPECTER-2 →
|
| 56 |
+
UMAP → HDBSCAN orchestration)
|
| 57 |
+
≤ 120
|
| 58 |
+
Bayesian optimiser node (optuna / scikit-
|
| 59 |
+
optimize wrapper)
|
| 60 |
+
≤ 60
|
| 61 |
+
LLM Council consolidator (3 sheets in, Sheet
|
| 62 |
+
4 out, Triple/Two/Single tagging)
|
| 63 |
+
≤ 60
|
| 64 |
+
Page 25 of 27Student Guidance Note — TCCM & Topic Modelling
|
| 65 |
+
Artefact Hard ceiling (lines of Python)
|
| 66 |
+
Gradio UI (all four sheets + agent invocation
|
| 67 |
+
+ clustering viz)
|
| 68 |
+
≤ 100
|
| 69 |
+
B.5 Anti-pattern vs pattern
|
| 70 |
+
The two code blocks below extract theories from a corpus. Both are functionally
|
| 71 |
+
similar. One is forbidden; the other is required.
|
| 72 |
+
ANTI-PATTERN — procedural Python with an LLM cameo (DO NOT SUBMIT)
|
| 73 |
+
papers = load_papers("./papers/")
|
| 74 |
+
results = []
|
| 75 |
+
for paper in papers: # forbidden loop
|
| 76 |
+
if paper["type"] == "empirical": # forbidden branch
|
| 77 |
+
prompt = EMPIRICAL_PROMPT
|
| 78 |
+
elif paper["type"] == "conceptual":
|
| 79 |
+
prompt = CONCEPTUAL_PROMPT
|
| 80 |
+
else:
|
| 81 |
+
prompt = DEFAULT_PROMPT
|
| 82 |
+
if len(paper["text"]) > 50000: # forbidden branch
|
| 83 |
+
paper["text"] = paper["text"][:50000]
|
| 84 |
+
output = llm.invoke(prompt + paper["text"])
|
| 85 |
+
if "NO CONSTRUCTS" not in output: # forbidden branch
|
| 86 |
+
results.append(parse_output(output))
|
| 87 |
+
# ... and 400 more lines like this ...
|
| 88 |
+
PATTERN — LangGraph + one unified prompt (THIS IS THE TARGET)
|
| 89 |
+
from langgraph.graph import StateGraph, START, END
|
| 90 |
+
from langgraph.types import Send
|
| 91 |
+
from typing import TypedDict, Annotated
|
| 92 |
+
import operator
|
| 93 |
+
class State(TypedDict):
|
| 94 |
+
papers: list[dict]
|
| 95 |
+
results: Annotated[list, operator.add] # reducer accumulates
|
| 96 |
+
def dispatch(state: State):
|
| 97 |
+
return [Send("extract", {"paper": p}) for p in state["papers"]]
|
| 98 |
+
def extract(state):
|
| 99 |
+
out = llm.invoke(UNIFIED_RCTO_PROMPT, state["paper"])
|
| 100 |
+
return {"results": [out]} # reducer handles append
|
| 101 |
+
graph = (
|
| 102 |
+
StateGraph(State)
|
| 103 |
+
.add_node("extract", extract)
|
| 104 |
+
.add_conditional_edges(START, dispatch, ["extract"])
|
| 105 |
+
Page 26 of 27Student Guidance Note — TCCM & Topic Modelling
|
| 106 |
+
.add_edge("extract", END)
|
| 107 |
+
.compile()
|
| 108 |
+
)
|
| 109 |
+
# Gradio UI invokes the compiled graph — no loops, no branches.
|
| 110 |
+
gr.Interface(fn=graph.invoke, inputs=gr.File(file_count="multiple"),
|
| 111 |
+
outputs=gr.Dataframe()).launch()
|
| 112 |
+
B.6 Submission inspection
|
| 113 |
+
Faculty will run a static check across your submission directory. The following
|
| 114 |
+
patterns trigger an automatic flag:
|
| 115 |
+
• `if ` or `elif ` outside a one-line conditional-edge routing lambda.
|
| 116 |
+
• `for ` or `while ` outside file-system or JSON traversal.
|
| 117 |
+
• Any HTML tag, inline `style=`, or `.css` / `.js` file in the submission tree.
|
| 118 |
+
• Any single Python file exceeding 200 lines.
|
| 119 |
+
• Absence of `langgraph` in `requirements.txt` or `from langgraph` in the
|
| 120 |
+
entry-point file.
|
| 121 |
+
One flag → comment in feedback. Two flags → revise and resubmit. Three flags
|
| 122 |
+
→ zero on the agentic component. The point of this annexure is not to constrain
|
| 123 |
+
creativity but to redirect it: the cleverness should be in the prompt and the graph,
|
| 124 |
+
not in hand-coded Python control flow that an LLM could have decided in one
|
| 125 |
+
inference call.
|
RCTO_prompt_1.txt
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Revised Theory-extraction prompt (paste verbatim)
|
| 2 |
+
(R) Role: You are an IS research methodologist conducting a TCCM systematic literature review of [Journal Name] following Paul and Rosado-Serrano (2019). You read like a doctoral examiner — exhaustive, not exemplar-driven.
|
| 3 |
+
(C) Context: Sources are 100 PDFs uploaded to this notebook plus IS_Theories.xlsx listing 159 named IS theories. The 159-theory list is a reference inventory, not a ceiling — papers may use theories outside the list, and you must flag those.
|
| 4 |
+
(T) Task: Perform an enumerative scan of all 100 papers. For every paper: (1) identify every named theory, model, framework, or theoretical lens explicitly cited or operationalised, and cross-check against the 159-theory inventory (mark Y/N); (2) distinguish theory (named framework) from theoretical foundation (meta-paradigm) from latent construct (variable used inside the paper, e.g., Trust, Satisfaction); (3) for each theory record whether it was tested empirically or invoked conceptually, which constructs were drawn from it, and the page reference; (4) do not collapse variants — list UTAUT, UTAUT2, UTAUT3, meta-UTAUT separately if they appear; (5) after enumeration, list any theory in the corpus that is NOT on the 159-theory inventory — these are candidate additions.
|
| 5 |
+
(O) Output format: Google Sheets table — Paper ID │ Citation │ Theory Name │ On Inventory (Y/N) │ Tested or Conceptual │ Constructs Drawn │ Page Reference. Second sheet titled 'Theories Not in Inventory' with the same columns.
|
RCTO_prompt_2.txt
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Revised Characteristics-extraction prompt (paste verbatim)
|
| 2 |
+
(R) Role: You are a senior research methodologist and synthesis analyst extracting constructs, variables, and relationships from research papers across empirical, conceptual, theoretical, and review traditions.
|
| 3 |
+
(C) Context: You will analyse full-text research papers uploaded to this notebook. Papers may be empirical (quantitative or qualitative with results) OR conceptual (models, frameworks, literature reviews, theoretical propositions). Both contain extractable constructs and relationships — your job is to surface them, not to gate-keep on study type.
|
| 4 |
+
Core principle: Every paper that proposes, tests, reviews, or theorises a relationship between concepts has extractable variables. 'NO CONSTRUCTS' is reserved ONLY for purely descriptive or historical papers with zero relational claims (very rare).
|
| 5 |
+
Evidence hierarchy — extract from the highest level present, then add lower levels: (1) Empirical results — tested hypotheses with reported effects (β, p, r, qualitative themes); (2) Formal propositions or hypotheses — explicitly numbered P1, P2, H1 etc., even if untested; (3) Conceptual model figure — labelled diagrams showing arrows between constructs; (4) Recurring narrative claims — 'X influences Y' repeated across the discussion.
|
| 6 |
+
(T) Task — extraction rules: DV / Outcome — the central phenomenon the paper seeks to explain, predict, or theorise; for conceptual papers, the rightmost or terminal construct; for reviews, the outcome the literature collectively addresses. IV(s) / Antecedents / Predictors — ALL constructs proposed or tested as causes, drivers, influences, or determinants; for multi-level frameworks list each top-level category AND its sub-components; include constructs from model figures even if not statistically tested. Mediator(s) / Mechanisms — constructs through which IVs operate on the DV; triggers: 'X works through Z to affect Y', or a construct between IV and DV in a figure; include theoretically proposed mediators, not only tested ones. Moderator(s) / Boundary conditions — constructs proposed to change the strength or direction of an IV→DV relationship; triggers: 'the effect of X on Y depends on Z', or a construct shown as encompassing the model (culture, context, demographics). Direction — Empirically tested: + / – / NS / Mixed; Proposed but untested: Proposed + / Proposed – / Proposed NS; Bidirectional / reciprocal: Bidirectional; Linked but silent on sign: Not specified; multiple distinct relationships separated with semicolons, anchored to construct names or proposition numbers.
|
| 7 |
+
Quality control: Extract only content explicitly present in the paper — no invented constructs. DO extract clearly-stated constructs from model figures, propositions, and consistent narrative claims even when untested. If a construct is mentioned only once in passing without being part of the model or propositions, do not include it. If direction is not specified, write 'Not specified' — do not guess.
|
| 8 |
+
(O) Output format: One row per paper in a single table — Paper No. and Name │ DV │ IV(s) │ Mediator(s) │ Moderator(s) │ Relationship Direction. Comma-separated values within cells; semicolons between distinct relationships in the Direction cell. Write 'Not specified' rather than leaving cells blank. Use the paper's own terminology — avoid paraphrasing into generic labels.
|
| 9 |
+
Worked example (calibration target) — Paper: Adya & Kaiser (2005), 'Early determinants of women in the IT workforce: a model of girls' career choices', Information Technology & People 18(3). DV: Girls' IT career choice. IV(s): Family (parents, siblings), Peer group, Media, Teacher/counsellor, School technology access, Personal technology access, Same-sex vs co-educational schooling, Individual differences (personality, self-efficacy, computer attitudes). Mediator(s): Role models, Gender stereotypes, Technology resources. Moderator(s): Ethnic culture. Relationship Direction: Father → IT career choice (Proposed +, P1); Mother → IT career choice (Proposed +, P2); Male peers → IT career choice (Proposed NS, P3); Teachers/counsellors → IT career choice (Proposed –, P4); Media → IT career choice (Proposed –, P5); Technology access → IT career choice (Proposed +, P6); IT-exporting national culture → IT career choice (Proposed +, P7).
|
RCTO_prompt_3.txt
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Revised Method-extraction prompt (paste verbatim) — emphasis on computational techniques
|
| 2 |
+
(R) Role: You are a senior research methodologist extracting research-method information from IS research papers spanning empirical, conceptual, theoretical, computational, and review traditions. You are equally familiar with classical methods (survey, experiment, case study) and computational methods (machine learning, NLP, network analysis, deep learning, agent-based simulation) that have become increasingly common in IS research.
|
| 3 |
+
(C) Context: You will analyse full-text research papers uploaded to this notebook. Papers may employ classical methods, computational methods, mixed methods, or be purely conceptual / theoretical / review papers. Modern IS research increasingly uses computational techniques on large data — surface what is actually used; do not invent methods that aren't in the paper.
|
| 4 |
+
Core principle: Every empirical paper has at least one identifiable method. Non-empirical papers still have methods (propositional modelling, mathematical modelling, simulation, literature-review methodology). 'NO METHOD' is reserved only for editorials, book reviews, or pure commentaries.
|
| 5 |
+
Evidence hierarchy — extract from the highest level present, then add lower levels: (1) Methods section explicitly named ('This study uses…', 'We employed…'); (2) Analytical-technique mentions in results (regression β, BERT scores, LDA topics, network centrality); (3) Sample / data description (N=, dataset, corpus); (4) Implicit method inferred from results presentation (a paper showing β coefficients clearly used regression even if 'regression' is never named in the text).
|
| 6 |
+
(T) Task — extraction rules:
|
| 7 |
+
PRIMARY METHOD — the dominant approach. Pick from: Survey (cross-sectional, longitudinal, panel); Experiment (lab, field, quasi-experiment, RCT, online); Case study (single, multiple, comparative); Qualitative (interviews, ethnography, observation, grounded theory, action research); Design science (artefact development, design theory); Computational — ML/NLP/DL (machine learning, deep learning, text mining, embeddings, topic modelling, sentiment analysis); Computational — Networks (network analysis, community detection, centrality, ERGM); Computational — Simulation (agent-based modelling, system dynamics, Monte Carlo); Secondary data / archival (system logs, financial filings, social-media traces, panel data); Literature review (systematic, narrative, bibliometric, meta-analysis, scientometric); Conceptual / theoretical (propositional, mathematical, formal modelling); Mixed methods.
|
| 8 |
+
SECONDARY METHODS — additional methods used (e.g., a survey paper that also runs an experiment; a case study with log-data analysis; a deep-learning paper with a validation survey).
|
| 9 |
+
ANALYTICAL TECHNIQUES — name the specific algorithm or test. Classical statistics: regression, ANOVA, t-tests, χ², SEM, factor analysis, time-series. Modern statistics: PLS-SEM, multilevel/hierarchical models, GLMM, IRT, Bayesian inference, mediation/moderation. ML/DL: decision trees, random forest, SVM, gradient boosting, neural networks, transformers, fine-tuning. NLP/text: Word2Vec, GloVe, BERT, SPECTER, sentence-transformers, LDA, NMF, BERTopic, sentiment, NER. Network: centrality measures, community detection (Louvain, Leiden), ERGM, link prediction. Simulation/optimisation: agent-based, Monte Carlo, Bayesian optimisation.
|
| 10 |
+
SAMPLE — type and approximate size where reported.
|
| 11 |
+
DATA SOURCES — explicit data origin (self-report survey, Salesforce CRM logs, Twitter/X API, Reddit, SEC filings, Google Trends, S&P 500 panel, Crossref / Scopus / Web of Science).
|
| 12 |
+
VALIDATION & TRIANGULATION — how the authors validated findings (member checks, multiple coders + inter-rater reliability, holdout test set, k-fold cross-validation, robustness checks, alternate model specifications, comparison with baselines).
|
| 13 |
+
Quality control: Use the paper's own terminology — if they say 'design science', don't paraphrase to 'artefact development'. For computational methods, name the specific algorithm (BERT, not just 'deep learning'; LDA, not just 'topic modelling'). If a paper combines methods, list all and indicate which is primary. If purely conceptual, write 'Conceptual / theoretical' as primary method.
|
| 14 |
+
(O) Output format: One row per paper in a single table — Paper No. and Name │ Primary Method │ Secondary Method(s) │ Analytical Technique(s) │ Sample (size, type) │ Data Source(s) │ Validation Approach. Comma-separated values within cells; semicolons between distinct items.
|
| 15 |
+
Worked example (calibration target) — Paper: Hypothetical et al. (2023), 'Predicting employee turnover with deep learning on internal communication data', MIS Quarterly. Primary Method: Computational — ML/NLP/DL (deep learning on text). Secondary Method: Survey (validation cohort). Analytical Techniques: BERT embeddings, LSTM with attention, gradient-boosting baseline, 5-fold cross-validation, ablation study. Sample: 12,847 employees from 6 firms; 240,000+ chat messages over 18 months; validation survey N=312. Data Sources: Company Slack archives, HR exit records, validation survey. Validation Approach: 80/20 train-test split, k-fold CV, holdout-firm validation, comparison against logistic-regression / random-forest / classical-NLP baselines.
|
README.md
CHANGED
|
@@ -1,13 +1,15 @@
|
|
| 1 |
---
|
| 2 |
-
title:
|
| 3 |
-
emoji:
|
| 4 |
colorFrom: blue
|
| 5 |
-
colorTo:
|
| 6 |
sdk: gradio
|
| 7 |
-
sdk_version: 6.14.0
|
| 8 |
-
python_version:
|
| 9 |
app_file: app.py
|
| 10 |
pinned: false
|
| 11 |
---
|
| 12 |
|
| 13 |
-
|
|
|
|
|
|
|
|
|
| 1 |
---
|
| 2 |
+
title: TCCM App
|
| 3 |
+
emoji: 🚀
|
| 4 |
colorFrom: blue
|
| 5 |
+
colorTo: purple
|
| 6 |
sdk: gradio
|
| 7 |
+
sdk_version: "6.14.0"
|
| 8 |
+
python_version: "3.11"
|
| 9 |
app_file: app.py
|
| 10 |
pinned: false
|
| 11 |
---
|
| 12 |
|
| 13 |
+
# My Space
|
| 14 |
+
|
| 15 |
+
TCCM App which mgmt prof wanted, with council of agents
|
agents/__init__.py
ADDED
|
File without changes
|
agents/council_agent.py
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# agents/council_agent.py [WAIVER: none — well within 60-line ceiling]
|
| 2 |
+
import os, json
|
| 3 |
+
from typing import TypedDict
|
| 4 |
+
from langgraph.graph import StateGraph, START, END
|
| 5 |
+
from huggingface_hub import InferenceClient
|
| 6 |
+
from models import CONSOLIDATION_MODEL_ID, MAX_NEW_TOKENS, MAX_SHEETS_CHARS
|
| 7 |
+
|
| 8 |
+
|
| 9 |
+
# ── State schema ─────────────────────────────────────────────────────────────
|
| 10 |
+
|
| 11 |
+
|
| 12 |
+
class CouncilState(TypedDict):
|
| 13 |
+
sheet1: list
|
| 14 |
+
sheet2: list
|
| 15 |
+
sheet3: list
|
| 16 |
+
consolidated: list
|
| 17 |
+
|
| 18 |
+
|
| 19 |
+
# ── Node ─────────────────────────────────────────────────────────────────────
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
def consolidate(state: CouncilState) -> dict:
|
| 23 |
+
"""Single responsibility: call consolidation LLM, return tagged rows."""
|
| 24 |
+
from prompts import CONSOLIDATION_PROMPT # imported here to avoid circular
|
| 25 |
+
|
| 26 |
+
client = InferenceClient(token=os.environ.get("HF_TOKEN", ""))
|
| 27 |
+
prompt = CONSOLIDATION_PROMPT.format(
|
| 28 |
+
sheet1=json.dumps(state["sheet1"], indent=2)[:MAX_SHEETS_CHARS],
|
| 29 |
+
sheet2=json.dumps(state["sheet2"], indent=2)[:MAX_SHEETS_CHARS],
|
| 30 |
+
sheet3=json.dumps(state["sheet3"], indent=2)[:MAX_SHEETS_CHARS],
|
| 31 |
+
)
|
| 32 |
+
resp = client.chat_completion(
|
| 33 |
+
messages=[{"role": "user", "content": prompt}],
|
| 34 |
+
model=CONSOLIDATION_MODEL_ID,
|
| 35 |
+
max_tokens=MAX_NEW_TOKENS * 2,
|
| 36 |
+
)
|
| 37 |
+
raw = resp.choices[0].message.content
|
| 38 |
+
try:
|
| 39 |
+
parsed = json.loads(raw) # ty:ignore[invalid-argument-type]
|
| 40 |
+
except Exception:
|
| 41 |
+
parsed = [{"error": "consolidation_parse_failed", "raw": raw[:600]}] # ty:ignore[not-subscriptable]
|
| 42 |
+
return {"consolidated": parsed if isinstance(parsed, list) else [parsed]}
|
| 43 |
+
|
| 44 |
+
|
| 45 |
+
# ── Compiled graph ───────────────────────────────────────────────────────────
|
| 46 |
+
|
| 47 |
+
council_graph = (
|
| 48 |
+
StateGraph(CouncilState) # ty:ignore[invalid-argument-type]
|
| 49 |
+
.add_node("consolidate", consolidate)
|
| 50 |
+
.add_edge(START, "consolidate")
|
| 51 |
+
.add_edge("consolidate", END)
|
| 52 |
+
.compile()
|
| 53 |
+
)
|
agents/tccm_agent.py
ADDED
|
@@ -0,0 +1,76 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# agents/tccm_agent.py [WAIVER: none — well within 80-line ceiling]
|
| 2 |
+
import os, json, operator
|
| 3 |
+
from typing import TypedDict, Annotated
|
| 4 |
+
from langgraph.graph import StateGraph, START, END
|
| 5 |
+
from langgraph.types import Send
|
| 6 |
+
from huggingface_hub import InferenceClient
|
| 7 |
+
from models import MAX_NEW_TOKENS
|
| 8 |
+
|
| 9 |
+
|
| 10 |
+
# ── State schemas ────────────────────────────────────────────────────────────
|
| 11 |
+
|
| 12 |
+
|
| 13 |
+
class GraphState(TypedDict):
|
| 14 |
+
papers: list[dict] # [{paper_id, text}, ...]
|
| 15 |
+
prompt_template: str # fully-formatted RCTO prompt string (with {placeholders})
|
| 16 |
+
model_id: str # HF model repo id
|
| 17 |
+
journal: str
|
| 18 |
+
inventory: str # 159-theory text, empty for Characteristics/Method
|
| 19 |
+
results: Annotated[list, operator.add] # reducer accumulates
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
class PaperState(TypedDict):
|
| 23 |
+
paper: dict
|
| 24 |
+
prompt_template: str
|
| 25 |
+
model_id: str
|
| 26 |
+
journal: str
|
| 27 |
+
inventory: str
|
| 28 |
+
results: Annotated[list, operator.add]
|
| 29 |
+
|
| 30 |
+
|
| 31 |
+
# ── Nodes ────────────────────────────────────────────────────────────────────
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
def dispatch(state: GraphState) -> list[Send]:
|
| 35 |
+
"""Fan-out: one Send per paper — no for-loop in agent logic."""
|
| 36 |
+
return [
|
| 37 |
+
Send("extract", {**state, "paper": p, "results": []}) for p in state["papers"]
|
| 38 |
+
]
|
| 39 |
+
|
| 40 |
+
|
| 41 |
+
def extract(state: PaperState) -> dict:
|
| 42 |
+
"""Single responsibility: call LLM, parse JSON, return result list."""
|
| 43 |
+
client = InferenceClient(token=os.environ.get("HF_TOKEN", ""))
|
| 44 |
+
prompt = state["prompt_template"].format(
|
| 45 |
+
paper_text=state["paper"]["text"],
|
| 46 |
+
inventory=state.get("inventory", ""),
|
| 47 |
+
journal=state.get("journal", "IS Journal"),
|
| 48 |
+
)
|
| 49 |
+
resp = client.chat_completion(
|
| 50 |
+
messages=[{"role": "user", "content": prompt}],
|
| 51 |
+
model=state["model_id"],
|
| 52 |
+
max_tokens=MAX_NEW_TOKENS,
|
| 53 |
+
)
|
| 54 |
+
raw = resp.choices[0].message.content
|
| 55 |
+
try:
|
| 56 |
+
parsed = json.loads(raw)
|
| 57 |
+
except Exception:
|
| 58 |
+
parsed = [
|
| 59 |
+
{
|
| 60 |
+
"paper_id": state["paper"]["paper_id"],
|
| 61 |
+
"raw_output": raw[:800],
|
| 62 |
+
"parse_error": True,
|
| 63 |
+
}
|
| 64 |
+
]
|
| 65 |
+
return {"results": parsed if isinstance(parsed, list) else [parsed]}
|
| 66 |
+
|
| 67 |
+
|
| 68 |
+
# ── Compiled graph ───────────────────────────────────────────────────────────
|
| 69 |
+
|
| 70 |
+
tccm_graph = (
|
| 71 |
+
StateGraph(GraphState)
|
| 72 |
+
.add_node("extract", extract)
|
| 73 |
+
.add_conditional_edges(START, dispatch, ["extract"])
|
| 74 |
+
.add_edge("extract", END)
|
| 75 |
+
.compile()
|
| 76 |
+
)
|
app.py
ADDED
|
@@ -0,0 +1,107 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# app.py [WAIVER: none — well within 100-line ceiling]
|
| 2 |
+
import os, json, tempfile
|
| 3 |
+
import gradio as gr
|
| 4 |
+
from prompts import THEORY_PROMPT, CHARACTERISTICS_PROMPT, METHOD_PROMPT
|
| 5 |
+
from models import EXTRACTION_MODELS, CONSOLIDATION_MODEL_ID
|
| 6 |
+
from agents.tccm_agent import tccm_graph
|
| 7 |
+
from agents.council_agent import council_graph
|
| 8 |
+
from utils.pdf_reader import load_papers, load_inventory
|
| 9 |
+
from utils.excel_writer import export_excel
|
| 10 |
+
|
| 11 |
+
PROMPT_MAP = {
|
| 12 |
+
"Theory": THEORY_PROMPT,
|
| 13 |
+
"Characteristics": CHARACTERISTICS_PROMPT,
|
| 14 |
+
"Method": METHOD_PROMPT,
|
| 15 |
+
}
|
| 16 |
+
MODEL_NAMES = list(
|
| 17 |
+
EXTRACTION_MODELS.keys()
|
| 18 |
+
) # ["Mistral-7B", "Zephyr-7B", "Qwen2.5-7B"]
|
| 19 |
+
|
| 20 |
+
|
| 21 |
+
# ── Core orchestration (UI glue — not agent logic) ───────────────────────────
|
| 22 |
+
|
| 23 |
+
|
| 24 |
+
def _extract_one(pdfs, inventory_file, journal, prompt_type, model_name):
|
| 25 |
+
papers = load_papers([p.name for p in pdfs]) if pdfs else []
|
| 26 |
+
inventory = load_inventory(inventory_file.name) if inventory_file else ""
|
| 27 |
+
result = tccm_graph.invoke(
|
| 28 |
+
{
|
| 29 |
+
"papers": papers,
|
| 30 |
+
"prompt_template": PROMPT_MAP[prompt_type],
|
| 31 |
+
"model_id": EXTRACTION_MODELS[model_name],
|
| 32 |
+
"journal": journal,
|
| 33 |
+
"inventory": inventory,
|
| 34 |
+
"results": [],
|
| 35 |
+
}
|
| 36 |
+
)
|
| 37 |
+
return result["results"]
|
| 38 |
+
|
| 39 |
+
|
| 40 |
+
def run_council(pdfs, inventory_file, journal, prompt_type, progress=gr.Progress()):
|
| 41 |
+
progress(0.05, desc=f"Extracting with {MODEL_NAMES[0]}…")
|
| 42 |
+
s1 = _extract_one(pdfs, inventory_file, journal, prompt_type, MODEL_NAMES[0])
|
| 43 |
+
progress(0.38, desc=f"Extracting with {MODEL_NAMES[1]}…")
|
| 44 |
+
s2 = _extract_one(pdfs, inventory_file, journal, prompt_type, MODEL_NAMES[1])
|
| 45 |
+
progress(0.68, desc=f"Extracting with {MODEL_NAMES[2]}…")
|
| 46 |
+
s3 = _extract_one(pdfs, inventory_file, journal, prompt_type, MODEL_NAMES[2])
|
| 47 |
+
progress(0.85, desc=f"Consolidating with {CONSOLIDATION_MODEL_ID}…")
|
| 48 |
+
s4 = council_graph.invoke(
|
| 49 |
+
{"sheet1": s1, "sheet2": s2, "sheet3": s3, "consolidated": []}
|
| 50 |
+
)["consolidated"]
|
| 51 |
+
progress(0.95, desc="Writing Excel…")
|
| 52 |
+
with tempfile.NamedTemporaryFile(suffix=".xlsx", delete=False, prefix="tccm_") as f:
|
| 53 |
+
xlsx = export_excel(s1, s2, s3, s4, f.name)
|
| 54 |
+
progress(1.0, desc="Done ✓")
|
| 55 |
+
return (
|
| 56 |
+
json.dumps(s1[:5], indent=2),
|
| 57 |
+
json.dumps(s2[:5], indent=2),
|
| 58 |
+
json.dumps(s3[:5], indent=2),
|
| 59 |
+
json.dumps(s4[:5], indent=2),
|
| 60 |
+
xlsx,
|
| 61 |
+
)
|
| 62 |
+
|
| 63 |
+
|
| 64 |
+
# ── Gradio UI ─────────────────────────────────────────────────────────────────
|
| 65 |
+
|
| 66 |
+
with gr.Blocks(title="Agentic TCCM Extractor — LLM Council") as demo:
|
| 67 |
+
gr.Markdown(
|
| 68 |
+
"# 📚 Agentic TCCM Extractor\n"
|
| 69 |
+
f"**Council:** {MODEL_NAMES[0]} · {MODEL_NAMES[1]} · {MODEL_NAMES[2]} → "
|
| 70 |
+
f"**Consolidator:** `{CONSOLIDATION_MODEL_ID}` · Set `HF_TOKEN` in Space secrets."
|
| 71 |
+
)
|
| 72 |
+
with gr.Row():
|
| 73 |
+
with gr.Column(scale=1):
|
| 74 |
+
pdfs = gr.File(
|
| 75 |
+
label="Paper PDFs", file_count="multiple", file_types=[".pdf"]
|
| 76 |
+
)
|
| 77 |
+
inventory_f = gr.File(
|
| 78 |
+
label="IS_Theories (.xlsx / .txt)", file_types=[".xlsx", ".txt", ".csv"]
|
| 79 |
+
)
|
| 80 |
+
journal = gr.Textbox(label="Journal Name", value="MIS Quarterly")
|
| 81 |
+
prompt_type = gr.Radio(
|
| 82 |
+
["Theory", "Characteristics", "Method"],
|
| 83 |
+
label="Extraction Prompt",
|
| 84 |
+
value="Theory",
|
| 85 |
+
)
|
| 86 |
+
run_btn = gr.Button("▶ Run Full LLM Council", variant="primary")
|
| 87 |
+
|
| 88 |
+
with gr.Column(scale=2):
|
| 89 |
+
with gr.Tabs():
|
| 90 |
+
with gr.Tab(f"Sheet 1 — {MODEL_NAMES[0]}"):
|
| 91 |
+
out1 = gr.JSON(label="Preview (first 5 rows)")
|
| 92 |
+
with gr.Tab(f"Sheet 2 — {MODEL_NAMES[1]}"):
|
| 93 |
+
out2 = gr.JSON(label="Preview (first 5 rows)")
|
| 94 |
+
with gr.Tab(f"Sheet 3 — {MODEL_NAMES[2]}"):
|
| 95 |
+
out3 = gr.JSON(label="Preview (first 5 rows)")
|
| 96 |
+
with gr.Tab("Sheet 4 — Consolidated"):
|
| 97 |
+
out4 = gr.JSON(label="Preview (first 5 rows)")
|
| 98 |
+
xlsx_out = gr.File(label="⬇ Download 4-Sheet Excel")
|
| 99 |
+
|
| 100 |
+
run_btn.click(
|
| 101 |
+
fn=run_council,
|
| 102 |
+
inputs=[pdfs, inventory_f, journal, prompt_type],
|
| 103 |
+
outputs=[out1, out2, out3, out4, xlsx_out],
|
| 104 |
+
)
|
| 105 |
+
|
| 106 |
+
if __name__ == "__main__":
|
| 107 |
+
demo.launch()
|
main.py
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
def main():
|
| 2 |
+
print("Hello from tccm-app!")
|
| 3 |
+
|
| 4 |
+
|
| 5 |
+
if __name__ == "__main__":
|
| 6 |
+
main()
|
models.py
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# models.py — HuggingFace Serverless Inference model registry
|
| 2 |
+
# All models are free-tier accessible with a HF_TOKEN
|
| 3 |
+
|
| 4 |
+
EXTRACTION_MODELS = {
|
| 5 |
+
"MiniLM": "microsoft/MiniLM-L12-H384-uncased",
|
| 6 |
+
"Zephyr-7B": "HuggingFaceH4/zephyr-7b-beta",
|
| 7 |
+
"Qwen2.5-7B": "Qwen/Qwen2.5-7B-Instruct",
|
| 8 |
+
}
|
| 9 |
+
|
| 10 |
+
CONSOLIDATION_MODEL_ID = "microsoft/Phi-3.5-mini-instruct"
|
| 11 |
+
|
| 12 |
+
# Shared inference settings
|
| 13 |
+
MAX_NEW_TOKENS = 2048
|
| 14 |
+
MAX_PAPER_CHARS = 12_000 # ~3 k tokens — fits every model's context window
|
| 15 |
+
MAX_SHEETS_CHARS = 4_000 # per-sheet truncation for consolidation prompt
|
prompts.py
ADDED
|
@@ -0,0 +1,143 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# prompts.py — All RCTO prompts baked in verbatim from Paul & Rosado-Serrano (2019) guidelines
|
| 2 |
+
# Substitution tokens: {journal}, {paper_text}, {inventory}
|
| 3 |
+
|
| 4 |
+
THEORY_PROMPT = """\
|
| 5 |
+
(R) Role: You are an IS research methodologist conducting a TCCM systematic literature review \
|
| 6 |
+
of {journal} following Paul and Rosado-Serrano (2019). You read like a doctoral examiner — exhaustive, not exemplar-driven.
|
| 7 |
+
|
| 8 |
+
(C) Context: The paper text is provided below. IS_Theories inventory (159 named IS theories) is also provided. \
|
| 9 |
+
The 159-theory list is a reference inventory, not a ceiling — the paper may use theories outside the list, and you must flag those.
|
| 10 |
+
|
| 11 |
+
(T) Task: Perform an enumerative scan of the paper. \
|
| 12 |
+
(1) Identify every named theory, model, framework, or theoretical lens explicitly cited or operationalised, \
|
| 13 |
+
and cross-check against the 159-theory inventory (mark Y/N); \
|
| 14 |
+
(2) Distinguish theory (named framework) from theoretical foundation (meta-paradigm) from latent construct \
|
| 15 |
+
(variable used inside the paper, e.g., Trust, Satisfaction); \
|
| 16 |
+
(3) For each theory record whether it was tested empirically or invoked conceptually, \
|
| 17 |
+
which constructs were drawn from it, and the page reference; \
|
| 18 |
+
(4) Do not collapse variants — list UTAUT, UTAUT2, UTAUT3, meta-UTAUT separately if they appear; \
|
| 19 |
+
(5) List any theory found that is NOT on the 159-theory inventory — these are candidate additions.
|
| 20 |
+
|
| 21 |
+
(O) Output: Respond ONLY with a valid JSON array. No preamble, no markdown fences. Each object:
|
| 22 |
+
{{"paper_id": "...", "citation": "...", "theory_name": "...", "on_inventory": "Y/N", \
|
| 23 |
+
"tested_or_conceptual": "Tested/Conceptual", "constructs_drawn": "...", "page_reference": "...", \
|
| 24 |
+
"new_theory_flag": true/false}}
|
| 25 |
+
|
| 26 |
+
IS_Theories inventory:
|
| 27 |
+
{inventory}
|
| 28 |
+
|
| 29 |
+
Paper text:
|
| 30 |
+
{paper_text}
|
| 31 |
+
"""
|
| 32 |
+
|
| 33 |
+
CHARACTERISTICS_PROMPT = """\
|
| 34 |
+
(R) Role: You are a senior research methodologist and synthesis analyst extracting constructs, \
|
| 35 |
+
variables, and relationships from research papers across empirical, conceptual, theoretical, and review traditions.
|
| 36 |
+
|
| 37 |
+
(C) Context: The full-text paper is provided below. Papers may be empirical (quantitative or qualitative with results) \
|
| 38 |
+
OR conceptual (models, frameworks, literature reviews, theoretical propositions). Both contain extractable constructs \
|
| 39 |
+
and relationships. Core principle: Every paper that proposes, tests, reviews, or theorises a relationship between \
|
| 40 |
+
concepts has extractable variables. 'NO CONSTRUCTS' is reserved ONLY for purely descriptive or historical papers \
|
| 41 |
+
with zero relational claims (very rare). Evidence hierarchy — extract from the highest level present, then add lower levels: \
|
| 42 |
+
(1) Empirical results — tested hypotheses with reported effects; \
|
| 43 |
+
(2) Formal propositions or hypotheses — explicitly numbered P1, P2, H1 etc., even if untested; \
|
| 44 |
+
(3) Conceptual model figure — labelled diagrams showing arrows between constructs; \
|
| 45 |
+
(4) Recurring narrative claims — 'X influences Y' repeated across the discussion.
|
| 46 |
+
|
| 47 |
+
(T) Task extraction rules: \
|
| 48 |
+
DV/Outcome — the central phenomenon the paper seeks to explain, predict, or theorise. \
|
| 49 |
+
IV(s)/Antecedents/Predictors — ALL constructs proposed or tested as causes, drivers, influences, or determinants. \
|
| 50 |
+
Mediator(s)/Mechanisms — constructs through which IVs operate on the DV. \
|
| 51 |
+
Moderator(s)/Boundary conditions — constructs proposed to change the strength or direction of an IV→DV relationship. \
|
| 52 |
+
Direction — Empirically tested: + / – / NS / Mixed; Proposed but untested: Proposed + / Proposed – / Proposed NS; \
|
| 53 |
+
Bidirectional / reciprocal: Bidirectional; multiple distinct relationships separated with semicolons.
|
| 54 |
+
Quality control: Extract only content explicitly present. Use the paper's own terminology.
|
| 55 |
+
|
| 56 |
+
(O) Output: Respond ONLY with a valid JSON array. No preamble, no markdown fences. Each object:
|
| 57 |
+
{{"paper_id": "...", "paper_name": "...", "dv": "...", "ivs": "...", \
|
| 58 |
+
"mediators": "...", "moderators": "...", "relationship_direction": "..."}}
|
| 59 |
+
|
| 60 |
+
Paper text:
|
| 61 |
+
{paper_text}
|
| 62 |
+
"""
|
| 63 |
+
|
| 64 |
+
METHOD_PROMPT = """\
|
| 65 |
+
(R) Role: You are a senior research methodologist extracting research-method information from IS research papers \
|
| 66 |
+
spanning empirical, conceptual, theoretical, computational, and review traditions. You are equally familiar with \
|
| 67 |
+
classical methods (survey, experiment, case study) and computational methods (machine learning, NLP, network analysis, \
|
| 68 |
+
deep learning, agent-based simulation) that have become increasingly common in IS research.
|
| 69 |
+
|
| 70 |
+
(C) Context: The paper text is provided below. Papers may employ classical methods, computational methods, \
|
| 71 |
+
mixed methods, or be purely conceptual/theoretical/review papers. Modern IS research increasingly uses \
|
| 72 |
+
computational techniques on large data — surface what is actually used; do not invent methods that aren't in the paper. \
|
| 73 |
+
Core principle: Every empirical paper has at least one identifiable method. Non-empirical papers still have methods. \
|
| 74 |
+
'NO METHOD' is reserved only for editorials, book reviews, or pure commentaries.
|
| 75 |
+
|
| 76 |
+
(T) Task extraction rules: \
|
| 77 |
+
PRIMARY METHOD — pick from: Survey; Experiment; Case study; Qualitative; Design science; \
|
| 78 |
+
Computational-ML/NLP/DL; Computational-Networks; Computational-Simulation; Secondary data/archival; \
|
| 79 |
+
Literature review; Conceptual/theoretical; Mixed methods. \
|
| 80 |
+
SECONDARY METHODS — additional methods used. \
|
| 81 |
+
ANALYTICAL TECHNIQUES — name the specific algorithm or test: classical statistics (regression, ANOVA, SEM, factor analysis), \
|
| 82 |
+
modern statistics (PLS-SEM, multilevel models, Bayesian inference, mediation/moderation), \
|
| 83 |
+
ML/DL (decision trees, random forest, SVM, transformers, fine-tuning), \
|
| 84 |
+
NLP/text (Word2Vec, BERT, SPECTER, LDA, BERTopic, sentiment, NER), \
|
| 85 |
+
Network (centrality, community detection, ERGM), Simulation (agent-based, Monte Carlo). \
|
| 86 |
+
SAMPLE — type and approximate size. \
|
| 87 |
+
DATA SOURCES — explicit data origin. \
|
| 88 |
+
VALIDATION — how the authors validated findings.
|
| 89 |
+
Quality control: Use the paper's own terminology. For computational methods, name the specific algorithm.
|
| 90 |
+
|
| 91 |
+
(O) Output: Respond ONLY with a valid JSON array. No preamble, no markdown fences. Each object:
|
| 92 |
+
{{"paper_id": "...", "paper_name": "...", "primary_method": "...", "secondary_methods": "...", \
|
| 93 |
+
"analytical_techniques": "...", "sample": "...", "data_sources": "...", "validation_approach": "..."}}
|
| 94 |
+
|
| 95 |
+
Paper text:
|
| 96 |
+
{paper_text}
|
| 97 |
+
"""
|
| 98 |
+
|
| 99 |
+
CONSOLIDATION_PROMPT = """\
|
| 100 |
+
You are the LLM Council consolidator for a TCCM systematic literature review.
|
| 101 |
+
You receive three extraction tables from three independent LLM runs of the same RCTO prompt.
|
| 102 |
+
|
| 103 |
+
Task:
|
| 104 |
+
1. Identify every unique (paper_id, key_value) pair across all three sheets.
|
| 105 |
+
2. For each pair, determine how many models agree (exact or near-exact match on the primary extracted value).
|
| 106 |
+
3. Tag: "Triple" if all 3 agree, "Two" if exactly 2 agree, "Single" if only 1 found it.
|
| 107 |
+
4. Action: "accepted" for Triple/Two; "verify vs PDF" for Single.
|
| 108 |
+
5. If a theory/item is marked new_theory_flag=true in any sheet, append "(NEW)" to action.
|
| 109 |
+
6. Compute overall agreement_rate = Triple_count / Total_rows (rounded to 2 decimal places).
|
| 110 |
+
7. Include a summary row at the end with paper_id="SUMMARY", key_value="Agreement Rate", tag=agreement_rate as string.
|
| 111 |
+
|
| 112 |
+
Sheet 1 (Model 1):
|
| 113 |
+
{sheet1}
|
| 114 |
+
|
| 115 |
+
Sheet 2 (Model 2):
|
| 116 |
+
{sheet2}
|
| 117 |
+
|
| 118 |
+
Sheet 3 (Model 3):
|
| 119 |
+
{sheet3}
|
| 120 |
+
|
| 121 |
+
Output ONLY a valid JSON array. No preamble, no markdown fences. Each object:
|
| 122 |
+
{{"paper_id": "...", "key_value": "...", "m1": "✓ or –", "m2": "✓ or –", "m3": "✓ or –", \
|
| 123 |
+
"tag": "Triple/Two/Single", "action": "accepted/accepted (NEW)/verify vs PDF"}}
|
| 124 |
+
"""
|
| 125 |
+
|
| 126 |
+
|
| 127 |
+
def get_prompt(
|
| 128 |
+
prompt_type: str, journal: str, paper_text: str, inventory: str = ""
|
| 129 |
+
) -> str:
|
| 130 |
+
"""Build the full prompt for a given type, substituting journal and paper content."""
|
| 131 |
+
text = paper_text[:12000] # Fit within 7B model context windows
|
| 132 |
+
templates = {
|
| 133 |
+
"Theory": THEORY_PROMPT,
|
| 134 |
+
"Characteristics": CHARACTERISTICS_PROMPT,
|
| 135 |
+
"Method": METHOD_PROMPT,
|
| 136 |
+
}
|
| 137 |
+
return templates[prompt_type].format(
|
| 138 |
+
journal=journal or "[Journal Name]",
|
| 139 |
+
paper_text=text,
|
| 140 |
+
inventory=inventory[:3000]
|
| 141 |
+
if inventory
|
| 142 |
+
else "See IS_Theories.xlsx (not uploaded)",
|
| 143 |
+
)
|
pyproject.toml
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[project]
|
| 2 |
+
name = "tccm-app"
|
| 3 |
+
version = "0.1.0"
|
| 4 |
+
description = "Add your description here"
|
| 5 |
+
readme = "README.md"
|
| 6 |
+
requires-python = ">=3.11"
|
| 7 |
+
dependencies = [
|
| 8 |
+
"gradio>=6.14.0",
|
| 9 |
+
"huggingface-hub>=1.14.0",
|
| 10 |
+
"langchain-core>=1.4.0",
|
| 11 |
+
"langgraph>=1.2.0",
|
| 12 |
+
"openpyxl>=3.1.2",
|
| 13 |
+
"pandas>=3.0.3",
|
| 14 |
+
"pymupdf>=1.24.0",
|
| 15 |
+
"pydantic>=2.11.10,<2.12.6"
|
| 16 |
+
]
|
requirements.txt
ADDED
|
@@ -0,0 +1,229 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# This file was autogenerated by uv via the following command:
|
| 2 |
+
# uv export --no-hashes --format requirements-txt
|
| 3 |
+
annotated-doc==0.0.4
|
| 4 |
+
# via
|
| 5 |
+
# fastapi
|
| 6 |
+
# typer
|
| 7 |
+
annotated-types==0.7.0
|
| 8 |
+
# via pydantic
|
| 9 |
+
anyio==4.13.0
|
| 10 |
+
# via
|
| 11 |
+
# gradio
|
| 12 |
+
# httpx
|
| 13 |
+
# starlette
|
| 14 |
+
audioop-lts==0.2.2 ; python_full_version >= '3.13'
|
| 15 |
+
# via gradio
|
| 16 |
+
brotli==1.2.0
|
| 17 |
+
# via gradio
|
| 18 |
+
certifi==2026.4.22
|
| 19 |
+
# via
|
| 20 |
+
# httpcore
|
| 21 |
+
# httpx
|
| 22 |
+
# requests
|
| 23 |
+
charset-normalizer==3.4.7
|
| 24 |
+
# via requests
|
| 25 |
+
click==8.3.3
|
| 26 |
+
# via
|
| 27 |
+
# typer
|
| 28 |
+
# uvicorn
|
| 29 |
+
colorama==0.4.6 ; sys_platform == 'win32'
|
| 30 |
+
# via
|
| 31 |
+
# click
|
| 32 |
+
# tqdm
|
| 33 |
+
et-xmlfile==2.0.0
|
| 34 |
+
# via openpyxl
|
| 35 |
+
fastapi==0.136.1
|
| 36 |
+
# via gradio
|
| 37 |
+
filelock==3.29.0
|
| 38 |
+
# via huggingface-hub
|
| 39 |
+
fsspec==2026.4.0
|
| 40 |
+
# via
|
| 41 |
+
# gradio-client
|
| 42 |
+
# huggingface-hub
|
| 43 |
+
gradio==6.14.0
|
| 44 |
+
# via tccm-app
|
| 45 |
+
gradio-client==2.5.0
|
| 46 |
+
# via
|
| 47 |
+
# gradio
|
| 48 |
+
# hf-gradio
|
| 49 |
+
groovy==0.1.2
|
| 50 |
+
# via gradio
|
| 51 |
+
h11==0.16.0
|
| 52 |
+
# via
|
| 53 |
+
# httpcore
|
| 54 |
+
# uvicorn
|
| 55 |
+
hf-gradio==0.4.1
|
| 56 |
+
# via gradio
|
| 57 |
+
hf-xet==1.5.0 ; platform_machine == 'AMD64' or platform_machine == 'aarch64' or platform_machine == 'amd64' or platform_machine == 'arm64' or platform_machine == 'x86_64'
|
| 58 |
+
# via huggingface-hub
|
| 59 |
+
httpcore==1.0.9
|
| 60 |
+
# via httpx
|
| 61 |
+
httpx==0.28.1
|
| 62 |
+
# via
|
| 63 |
+
# gradio
|
| 64 |
+
# gradio-client
|
| 65 |
+
# huggingface-hub
|
| 66 |
+
# langgraph-sdk
|
| 67 |
+
# langsmith
|
| 68 |
+
# safehttpx
|
| 69 |
+
huggingface-hub==1.14.0
|
| 70 |
+
# via
|
| 71 |
+
# gradio
|
| 72 |
+
# gradio-client
|
| 73 |
+
# tccm-app
|
| 74 |
+
idna==3.15
|
| 75 |
+
# via
|
| 76 |
+
# anyio
|
| 77 |
+
# httpx
|
| 78 |
+
# requests
|
| 79 |
+
jinja2==3.1.6
|
| 80 |
+
# via gradio
|
| 81 |
+
jsonpatch==1.33
|
| 82 |
+
# via langchain-core
|
| 83 |
+
jsonpointer==3.1.1
|
| 84 |
+
# via jsonpatch
|
| 85 |
+
langchain-core==1.4.0
|
| 86 |
+
# via
|
| 87 |
+
# langgraph
|
| 88 |
+
# langgraph-checkpoint
|
| 89 |
+
# langgraph-prebuilt
|
| 90 |
+
# tccm-app
|
| 91 |
+
langchain-protocol==0.0.15
|
| 92 |
+
# via langchain-core
|
| 93 |
+
langgraph==1.2.0
|
| 94 |
+
# via tccm-app
|
| 95 |
+
langgraph-checkpoint==4.1.0
|
| 96 |
+
# via
|
| 97 |
+
# langgraph
|
| 98 |
+
# langgraph-prebuilt
|
| 99 |
+
langgraph-prebuilt==1.1.0
|
| 100 |
+
# via langgraph
|
| 101 |
+
langgraph-sdk==0.3.14
|
| 102 |
+
# via langgraph
|
| 103 |
+
langsmith==0.8.4
|
| 104 |
+
# via langchain-core
|
| 105 |
+
markdown-it-py==4.2.0
|
| 106 |
+
# via rich
|
| 107 |
+
markupsafe==3.0.3
|
| 108 |
+
# via
|
| 109 |
+
# gradio
|
| 110 |
+
# jinja2
|
| 111 |
+
mdurl==0.1.2
|
| 112 |
+
# via markdown-it-py
|
| 113 |
+
numpy==2.4.4
|
| 114 |
+
# via
|
| 115 |
+
# gradio
|
| 116 |
+
# pandas
|
| 117 |
+
openpyxl==3.1.5
|
| 118 |
+
# via tccm-app
|
| 119 |
+
orjson==3.11.9
|
| 120 |
+
# via
|
| 121 |
+
# gradio
|
| 122 |
+
# langgraph-sdk
|
| 123 |
+
# langsmith
|
| 124 |
+
ormsgpack==1.12.2
|
| 125 |
+
# via langgraph-checkpoint
|
| 126 |
+
packaging==26.2
|
| 127 |
+
# via
|
| 128 |
+
# gradio
|
| 129 |
+
# gradio-client
|
| 130 |
+
# huggingface-hub
|
| 131 |
+
# langchain-core
|
| 132 |
+
# langsmith
|
| 133 |
+
pandas==3.0.3
|
| 134 |
+
# via
|
| 135 |
+
# gradio
|
| 136 |
+
# tccm-app
|
| 137 |
+
pillow==12.2.0
|
| 138 |
+
# via gradio
|
| 139 |
+
pydantic==2.12.5
|
| 140 |
+
# via
|
| 141 |
+
# fastapi
|
| 142 |
+
# gradio
|
| 143 |
+
# langchain-core
|
| 144 |
+
# langgraph
|
| 145 |
+
# langsmith
|
| 146 |
+
# tccm-app
|
| 147 |
+
pydantic-core==2.41.5
|
| 148 |
+
# via pydantic
|
| 149 |
+
pydub==0.25.1
|
| 150 |
+
# via gradio
|
| 151 |
+
pygments==2.20.0
|
| 152 |
+
# via rich
|
| 153 |
+
pymupdf==1.27.2.3
|
| 154 |
+
# via tccm-app
|
| 155 |
+
python-dateutil==2.9.0.post0
|
| 156 |
+
# via pandas
|
| 157 |
+
python-multipart==0.0.28
|
| 158 |
+
# via gradio
|
| 159 |
+
pytz==2026.2
|
| 160 |
+
# via gradio
|
| 161 |
+
pyyaml==6.0.3
|
| 162 |
+
# via
|
| 163 |
+
# gradio
|
| 164 |
+
# huggingface-hub
|
| 165 |
+
# langchain-core
|
| 166 |
+
requests==2.34.1
|
| 167 |
+
# via
|
| 168 |
+
# langsmith
|
| 169 |
+
# requests-toolbelt
|
| 170 |
+
requests-toolbelt==1.0.0
|
| 171 |
+
# via langsmith
|
| 172 |
+
rich==15.0.0
|
| 173 |
+
# via typer
|
| 174 |
+
safehttpx==0.1.7
|
| 175 |
+
# via gradio
|
| 176 |
+
semantic-version==2.10.0
|
| 177 |
+
# via gradio
|
| 178 |
+
shellingham==1.5.4
|
| 179 |
+
# via typer
|
| 180 |
+
six==1.17.0
|
| 181 |
+
# via python-dateutil
|
| 182 |
+
starlette==1.0.0
|
| 183 |
+
# via
|
| 184 |
+
# fastapi
|
| 185 |
+
# gradio
|
| 186 |
+
tenacity==9.1.4
|
| 187 |
+
# via langchain-core
|
| 188 |
+
tomlkit==0.14.0
|
| 189 |
+
# via gradio
|
| 190 |
+
tqdm==4.67.3
|
| 191 |
+
# via huggingface-hub
|
| 192 |
+
typer==0.25.1
|
| 193 |
+
# via
|
| 194 |
+
# gradio
|
| 195 |
+
# hf-gradio
|
| 196 |
+
# huggingface-hub
|
| 197 |
+
typing-extensions==4.15.0
|
| 198 |
+
# via
|
| 199 |
+
# anyio
|
| 200 |
+
# fastapi
|
| 201 |
+
# gradio
|
| 202 |
+
# gradio-client
|
| 203 |
+
# huggingface-hub
|
| 204 |
+
# langchain-core
|
| 205 |
+
# langchain-protocol
|
| 206 |
+
# pydantic
|
| 207 |
+
# pydantic-core
|
| 208 |
+
# starlette
|
| 209 |
+
# typing-inspection
|
| 210 |
+
typing-inspection==0.4.2
|
| 211 |
+
# via
|
| 212 |
+
# fastapi
|
| 213 |
+
# pydantic
|
| 214 |
+
tzdata==2026.2 ; sys_platform == 'emscripten' or sys_platform == 'win32'
|
| 215 |
+
# via pandas
|
| 216 |
+
urllib3==2.7.0
|
| 217 |
+
# via requests
|
| 218 |
+
uuid-utils==0.15.0
|
| 219 |
+
# via
|
| 220 |
+
# langchain-core
|
| 221 |
+
# langsmith
|
| 222 |
+
uvicorn==0.46.0
|
| 223 |
+
# via gradio
|
| 224 |
+
xxhash==3.7.0
|
| 225 |
+
# via
|
| 226 |
+
# langgraph
|
| 227 |
+
# langsmith
|
| 228 |
+
zstandard==0.25.0
|
| 229 |
+
# via langsmith
|
utils/__init__.py
ADDED
|
File without changes
|
utils/excel_writer.py
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# utils/excel_writer.py
|
| 2 |
+
import pandas as pd
|
| 3 |
+
from openpyxl import Workbook
|
| 4 |
+
from openpyxl.styles import Font, PatternFill
|
| 5 |
+
from openpyxl.utils import get_column_letter
|
| 6 |
+
|
| 7 |
+
_HEADER_FILL = PatternFill("solid", fgColor="1F4E79")
|
| 8 |
+
_HEADER_FONT = Font(bold=True, color="FFFFFF")
|
| 9 |
+
_SHEET_TITLES = [
|
| 10 |
+
"Sheet1-Model1",
|
| 11 |
+
"Sheet2-Model2",
|
| 12 |
+
"Sheet3-Model3",
|
| 13 |
+
"Sheet4-Consolidated",
|
| 14 |
+
]
|
| 15 |
+
|
| 16 |
+
|
| 17 |
+
def _to_df(results: list) -> pd.DataFrame:
|
| 18 |
+
rows = []
|
| 19 |
+
for item in results:
|
| 20 |
+
(rows.extend(item) if isinstance(item, list) else rows.append(item))
|
| 21 |
+
return pd.DataFrame(rows) if rows else pd.DataFrame({"note": ["No data"]})
|
| 22 |
+
|
| 23 |
+
|
| 24 |
+
def _fill_sheet(ws, df: pd.DataFrame, title: str) -> None:
|
| 25 |
+
ws.title = title
|
| 26 |
+
ws.append(list(df.columns))
|
| 27 |
+
for cell in ws[1]:
|
| 28 |
+
cell.fill = _HEADER_FILL
|
| 29 |
+
cell.font = _HEADER_FONT
|
| 30 |
+
for row in df.itertuples(index=False):
|
| 31 |
+
ws.append(list(row))
|
| 32 |
+
for i in range(1, len(df.columns) + 1):
|
| 33 |
+
ws.column_dimensions[get_column_letter(i)].width = 28
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
def export_excel(sheet1, sheet2, sheet3, sheet4, out_path: str) -> str:
|
| 37 |
+
wb = Workbook()
|
| 38 |
+
datasets = [sheet1, sheet2, sheet3, sheet4]
|
| 39 |
+
for idx, (data, title) in enumerate(zip(datasets, _SHEET_TITLES)):
|
| 40 |
+
ws = wb.active if idx == 0 else wb.create_sheet()
|
| 41 |
+
_fill_sheet(ws, _to_df(data), title)
|
| 42 |
+
wb.save(out_path)
|
| 43 |
+
return out_path
|
utils/pdf_reader.py
ADDED
|
@@ -0,0 +1,28 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# utils/pdf_reader.py
|
| 2 |
+
import os
|
| 3 |
+
import fitz # PyMuPDF
|
| 4 |
+
from models import MAX_PAPER_CHARS
|
| 5 |
+
|
| 6 |
+
|
| 7 |
+
def extract_text(pdf_path: str) -> str:
|
| 8 |
+
doc = fitz.open(pdf_path)
|
| 9 |
+
text = "".join(page.get_text() for page in doc)
|
| 10 |
+
doc.close()
|
| 11 |
+
return text[:MAX_PAPER_CHARS]
|
| 12 |
+
|
| 13 |
+
|
| 14 |
+
def load_papers(pdf_paths: list[str]) -> list[dict]:
|
| 15 |
+
return [
|
| 16 |
+
{"paper_id": os.path.splitext(os.path.basename(p))[0], "text": extract_text(p)}
|
| 17 |
+
for p in pdf_paths
|
| 18 |
+
]
|
| 19 |
+
|
| 20 |
+
|
| 21 |
+
def load_inventory(inventory_path: str) -> str:
|
| 22 |
+
if inventory_path.endswith(".xlsx"):
|
| 23 |
+
import pandas as pd
|
| 24 |
+
|
| 25 |
+
df = pd.read_excel(inventory_path)
|
| 26 |
+
return df.to_string(index=False)[:3_000]
|
| 27 |
+
with open(inventory_path, "r", encoding="utf-8") as f:
|
| 28 |
+
return f.read()[:3_000]
|
uv.lock
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|