Spaces:
Running
Running
File size: 6,314 Bytes
ee7d7b9 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 | """
Strict Data Grounding - ONLY User Data, NO Outside Information
================================================================
This is CRITICAL for DataVision:
- AI must ONLY use information from uploaded data
- NEVER use general knowledge
- NEVER make up numbers
- REFUSE to answer if data doesn't have the answer
This prevents hallucination and ensures trust.
"""
import logging
from typing import Dict, List, Optional, Tuple
from core.llm import chat
logger = logging.getLogger(__name__)
def build_grounded_system_prompt(
columns: List[str],
data_sample: str,
currency_symbol: str = "$",
user_name: Optional[str] = None
) -> str:
"""
Build a system prompt that STRICTLY grounds the AI in user data.
"""
prompt = f"""You are DataVision, a data analysis AI.
β οΈ CRITICAL RULES - FOLLOW EXACTLY:
1. **ONLY USE THE DATA PROVIDED** - You can ONLY answer questions using the data below.
2. **NEVER USE OUTSIDE KNOWLEDGE** - Do not use any information not in the user's data.
3. **NEVER MAKE UP NUMBERS** - Every number must come from the actual data.
4. **IF DATA DOESN'T EXIST, SAY SO** - Example: "I don't see [X] in your data."
5. **CITE YOUR SOURCE** - When giving answers, mention which column/data you used.
π USER'S DATA:
Columns: {columns}
Sample Data:
{data_sample[:2000]}
Currency: {currency_symbol}
{f"User: {user_name}" if user_name else ""}
π« FORBIDDEN:
- Do NOT answer questions about topics not in the data
- Do NOT use general knowledge about industries, markets, etc.
- Do NOT guess or estimate if data is missing
- Do NOT say "typically" or "usually" - only use actual data
β
CORRECT BEHAVIOR:
- "Your data shows X is {currency_symbol}Y"
- "Based on the [column] column, the total is..."
- "I don't have data about [topic] in your files"
- "The data doesn't include [field], so I can't answer that"
Remember: You are a DATA ANALYST, not a general AI. Stay in your lane!"""
return prompt
def validate_response_grounding(
response: str,
data_context: str,
columns: List[str]
) -> Tuple[bool, str]:
"""
Validate that a response is grounded in the actual data.
Returns (is_valid, cleaned_response)
"""
# Check for hallucination indicators
hallucination_phrases = [
"typically",
"usually",
"in general",
"generally speaking",
"most companies",
"industry standard",
"common practice",
"on average in the industry",
"according to research",
"studies show",
"it's common to",
"best practices suggest",
]
response_lower = response.lower()
for phrase in hallucination_phrases:
if phrase in response_lower:
logger.warning(f"[GROUNDING] Hallucination detected: '{phrase}'")
# Remove the hallucinating sentence or add disclaimer
return False, response + "\n\nβ οΈ *Note: This response may contain general information. Please verify against your actual data.*"
return True, response
def create_grounded_query_prompt(
query: str,
columns: List[str],
data_context: str,
previous_response: Optional[str] = None
) -> str:
"""
Create a query prompt that enforces data grounding.
"""
prompt = f"""Answer this question using ONLY the data provided below.
β QUESTION: {query}
π AVAILABLE DATA:
Columns: {columns}
Data Context:
{data_context[:2500]}
{f"Previous Response (for follow-up context): {previous_response[:500]}" if previous_response else ""}
β οΈ STRICT RULES:
1. Use ONLY numbers and facts from the data above
2. If the data doesn't contain the answer, say "I don't have data for that"
3. Do NOT use general knowledge or make assumptions
4. Cite which column/data you're using
π YOUR ANSWER (based strictly on the data above):"""
return prompt
def refuse_off_topic(query: str, columns: List[str]) -> Optional[str]:
"""
Check if query is off-topic and generate refusal if needed.
Uses LLM to determine this intelligently.
"""
try:
prompt = f"""Can this question be answered using ONLY the data columns listed?
QUESTION: "{query}"
AVAILABLE COLUMNS: {columns}
If the question is about something NOT in these columns, respond with a polite refusal.
If it CAN be answered with this data, respond with "ANSWERABLE".
Response:"""
result = chat(prompt, temperature=0.1, max_tokens=150)
if "ANSWERABLE" in result.upper():
return None # Can be answered
else:
return result.strip() # Return the refusal message
except:
return None
def add_data_citation(response: str, columns: List[str]) -> str:
"""
Ensure response cites which data columns were used.
"""
# Check if response already has citation
if "based on" in response.lower() or "from the" in response.lower():
return response
# Add subtle citation
used_columns = []
response_lower = response.lower()
for col in columns:
if col.lower() in response_lower or col.replace('_', ' ').lower() in response_lower:
used_columns.append(col)
if used_columns and len(used_columns) <= 3:
citation = f"\n\nπ *Data source: {', '.join(used_columns)}*"
return response + citation
return response
def ground_response(
response: str,
query: str,
columns: List[str],
data_context: str
) -> str:
"""
Post-process response to ensure it's grounded.
"""
# Validate grounding
is_valid, processed = validate_response_grounding(response, data_context, columns)
# Add citation
final = add_data_citation(processed, columns)
return final
def get_answerable_topics(columns: List[str]) -> str:
"""
Generate a description of what CAN be answered with this data.
"""
try:
prompt = f"""Based on these data columns, what types of questions can be answered?
COLUMNS: {columns}
List 3-5 example questions that could be answered with this data.
Format as bullet points.
Examples:"""
result = chat(prompt, temperature=0.5, max_tokens=150)
return result.strip()
except:
return f"Questions about: {', '.join(columns[:5])}"
|