Spaces:
Sleeping
Sleeping
File size: 14,059 Bytes
a8df853 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 | import numpy as np
import random
from typing import List, Dict, Union, Tuple
# --- Configuration Constants (Normalized to 0-100 Scale) ---
# Weighting for the Final Content Score (CS)
W1_PREDICTION = 0.50
W2_TALENT = 0.15
W3_PROGRESS = 0.15
W4_QUALITY = 0.15
# Dynamic Allocation
MAX_DYNAMIC_BOOST = 0.05 # Max 5% of CS allowed via dynamic boosts/penalties
SUPPRESSION_PENALTY = 0.60 # Penalty multiplier for high ESR content
FOLLOW_THRESHOLD = 1000 # Max followers for Small-Creator Advantage
# --- Score Scaling and Clamping Constants ---
PS_FLAT_PROGRESS = 50.0 # PS score when a creator has 0% improvement
NETWORK_AVG_ESR = 0.10 # Network average Early Skip Rate (0-1)
class UpnisoAlgorithmPipeline:
"""
Upniso Merit-Based Recommendation Algorithm Pipeline (V2.0).
Encapsulates the 6-stage architecture: Analyze, Score, Predict, Test,
Distribute, and Learn. All core scores are normalized to a 0-100 scale.
"""
def __init__(self, avg_esr: float = NETWORK_AVG_ESR, avg_ts: float = 50.0, avg_wtpu: float = 0.50):
"""Initializes the pipeline with network baseline averages for comparison."""
self.avg_esr = avg_esr
self.avg_ts = avg_ts
self.avg_wtpu = avg_wtpu
self.creators = {}
self.content_queue = {}
# --- Stage 1: Analyze (Data Ingestion and Cleanup) ---
def analyze(self, creator_id: str, creator_data: dict, content_id: str, content_data: dict):
"""
Ingests, sanitizes, and verifies raw data from the database/API.
Adds fallbacks for missing historical data.
"""
# Guard against division by zero and provide sensible fallbacks (Task B.3)
creator_data['ts_90_days_ago'] = max(creator_data.get('ts_90_days_ago', self.avg_ts), 1.0)
creator_data['avg_wtpu_month1'] = max(creator_data.get('avg_wtpu_month1', self.avg_wtpu), 0.01)
creator_data['ts_14_days_ago'] = creator_data.get('ts_14_days_ago', creator_data['ts_90_days_ago'])
creator_data['stdev_upload_days'] = creator_data.get('stdev_upload_days', 4.0)
content_data['wtpu_current'] = np.clip(content_data.get('wtpu_current', 0.0), 0.0, 1.0).item()
content_data['nvr'] = np.clip(content_data.get('nvr', 0.0), 0.0, 1.0).item()
# MIS is 1.0 for compliance (True), 0.5 otherwise (False/Missing)
content_data['mis_score'] = 1.0 if content_data.get('mis_compliance', False) else 0.5
content_data['esr_current'] = np.clip(content_data.get('esr_current', 0.5), 0.0, 1.0).item()
content_data['prediction_score_p_hva'] = content_data.get('prediction_score_p_hva', 0.5)
self.creators[creator_id] = creator_data
self.content_queue[content_id] = content_data
self.content_queue[content_id]['creator_id'] = creator_id
# --- Stage 2: Score (Creator Profile Merit Calculation) ---
def score(self, creator_id: str) -> Dict[str, float]:
"""
Calculates the creator's persistent Talent Score (TS) and Progress Score (PS).
Both scores are normalized to 0-100.
"""
creator_data = self.creators.get(creator_id)
if not creator_data: return {'TS': self.avg_ts, 'PS': PS_FLAT_PROGRESS}
# 1. Talent Score (TS) Calculation (0-100, Normalized)
wtpu_norm = creator_data['wtpu_last_10_avg']
hva_norm = creator_data.get('avg_hva_rate', 0.05)
# Consistency Factor: Closer to 1.0 is better (lower stdev)
consistency = 1.0 / (creator_data['stdev_upload_days'] + 1.0)
# TS = 100 * (0.5 * WTPU + 0.3 * HVA + 0.2 * Consistency)
ts_merit = 100 * (0.5 * wtpu_norm + 0.3 * hva_norm + 0.2 * consistency)
ts_today = np.clip(ts_merit, 0, 100).item()
self.creators[creator_id]['ts_today'] = ts_today
# 2. Progress Score (PS) Calculation (0-100, 50=Flat)
ts_90_days_ago = creator_data['ts_90_days_ago']
wtpu_m1 = creator_data['avg_wtpu_month1']
wtpu_m3 = creator_data.get('avg_wtpu_month3', wtpu_m1)
# PS_Raw = 50 + [Delta TS * 50] + [Delta WTPU * 50]
ts_delta_contribution = ((ts_today / ts_90_days_ago) - 1.0) * 50
wtpu_delta_contribution = ((wtpu_m3 / wtpu_m1) - 1.0) * 50
ps_raw = PS_FLAT_PROGRESS + ts_delta_contribution + wtpu_delta_contribution
# Forgiveness Boost Check (Task A.10: If TS improved >= 20% in 14 days)
ts_14_days_ago = creator_data['ts_14_days_ago']
ts_14_day_improvement = (ts_today - ts_14_days_ago) / ts_14_days_ago
forgiveness_boost = 0
if ts_14_day_improvement >= 0.20:
forgiveness_boost = 15 # +15 score for demonstrated momentum
# Clamping PS score (0-100)
ps_score = np.clip(ps_raw + forgiveness_boost, 0, 100).item()
return {'TS': ts_today, 'PS': ps_score}
# --- Stage 3: Predict (Affinity Model Simulation) ---
def predict(self, content_id: str) -> float:
"""
Simulates the Prediction Score (P(HVA)) based on user affinity.
Returns score (0-100).
"""
content_data = self.content_queue.get(content_id)
if not content_data: return PS_FLAT_PROGRESS
prediction_raw = content_data['prediction_score_p_hva'] * 100
return np.clip(prediction_raw, 0, 100).item()
# --- Stage 4: Test (Fair Exposure Engine - FEE) ---
def test(self, content_id: str, merit_scores: Dict[str, float]) -> Tuple[float, float]:
"""
Calculates the Quality Score (QS) and the dynamic boost/penalty flag
based on initial T0/T1 micro-test results.
Returns QS (0-100) and Dynamic Factor (0 to 5%).
"""
content_data = self.content_queue.get(content_id)
creator_data = self.creators.get(content_data['creator_id'])
if not content_data or not creator_data: return PS_FLAT_PROGRESS, 0.0
# 1. Quality Score (QS) Calculation (0-100, Normalized)
wtpu = content_data['wtpu_current']
nvr = content_data['nvr']
mis = content_data['mis_score']
# QS = 100 * (0.6 * WTPU + 0.3 * NVR + 0.1 * MIS)
qs_score = 100 * (0.6 * wtpu + 0.3 * nvr + 0.1 * mis)
qs_score = np.clip(qs_score, 0, 100).item()
# 2. Dynamic Boost Calculation (5% Max)
dynamic_factor = 0.0
# Small-Creator Advantage (1.5x Multiplier to Merit in Test Phase)
if creator_data['follower_count'] < FOLLOW_THRESHOLD:
# Merit component calculation (normalized to 0-1)
merit_value_norm = (W2_TALENT * merit_scores['TS'] + W3_PROGRESS * merit_scores['PS'] + W4_QUALITY * qs_score) / 100
# 1.5x multiplier means adding 50% of the calculated merit value (45% weight)
boost_value = merit_value_norm * 0.5
# Clamp the boost to the allocated 5% maximum dynamic weight
dynamic_factor += np.clip(boost_value, 0, MAX_DYNAMIC_BOOST).item()
# 3. Negative Signal Suppression Flag (ESR >= 5x Avg)
esr_ratio = content_data['esr_current'] / self.avg_esr if self.avg_esr > 0 else 1.0
penalty_flag = False
if esr_ratio >= 5.0:
penalty_flag = True
self.content_queue[content_id]['penalty_flag'] = penalty_flag
self.content_queue[content_id]['QS'] = qs_score
return qs_score, dynamic_factor
# --- Stage 5: Distribute (Final Content Score Ranking) ---
def distribute(self, content_id: str, merit_scores: Dict[str, float], pred_score: float, qs_score: float, dynamic_factor: float) -> float:
"""
Calculates the Final Content Score (CS, 0-1.0) used for feed ranking.
"""
# Convert 0-100 scores to 0-1 for weighting
ts_norm = merit_scores['TS'] / 100
ps_norm = merit_scores['PS'] / 100
qs_norm = qs_score / 100
pred_norm = pred_score / 100
# Baseline CS (95% weight) = (50% Pred) + (45% Merit)
cs_baseline = (W1_PREDICTION * pred_norm) + \
(W2_TALENT * ts_norm) + \
(W3_PROGRESS * ps_norm) + \
(W4_QUALITY * qs_norm)
# Final Content Score (Max 1.0)
final_cs = cs_baseline + dynamic_factor
# Apply Suppression Penalty if flagged
if self.content_queue.get(content_id, {}).get('penalty_flag', False):
final_cs *= SUPPRESSION_PENALTY
return np.clip(final_cs, 0, 1).item()
# --- Stage 6: Learn (Feedback Loop Simulation) ---
def learn(self, creator_id: str, content_id: str, final_cs: float):
"""
Simulates updating the creator's profile data based on content performance (Offline Job).
"""
# In a real system, this would run daily to update ts_90_days_ago, ts_14_days_ago, etc.
# For simulation, we just record results.
self.content_queue[content_id]['final_cs'] = final_cs
self.content_queue[content_id]['PS'] = self.content_queue[content_id].get('PS', 0)
self.content_queue[content_id]['TS'] = self.content_queue[content_id].get('TS', 0)
# --- Simulation Mode (Task B.6 / D) ---
def run_simulation(self, creators_data: Dict[str, Dict], contents_data: Dict[str, Dict]):
"""Runs a competition simulation for all provided content and returns the ranked list."""
results = []
print("="*80)
print("UPNISO ALGORITHM SIMULATION (V2.0)")
print("="*80)
for content_id, content_input in contents_data.items():
creator_id = content_input['creator_id']
creator_input = creators_data[creator_id]
# --- Pipeline Execution ---
self.analyze(creator_id, creator_input, content_id, content_input)
merit = self.score(creator_id)
pred = self.predict(content_id)
qs, dynamic_boost = self.test(content_id, merit)
# Store scores for reporting
self.content_queue[content_id]['QS'] = qs
self.content_queue[content_id]['PS'] = merit['PS']
self.content_queue[content_id]['TS'] = merit['TS']
self.content_queue[content_id]['Pred'] = pred
self.content_queue[content_id]['dynamic_boost'] = dynamic_boost
final_cs = self.distribute(content_id, merit, pred, qs, dynamic_boost)
self.learn(creator_id, content_id, final_cs)
# --- Reporting ---
penalty = "APPLIED" if self.content_queue[content_id].get('penalty_flag') else "None"
results.append({
'content_id': content_id,
'creator_id': creator_id,
'CS': final_cs,
'TS': merit['TS'],
'PS': merit['PS'],
'QS': qs,
'Pred': pred,
'Dynamic_Boost': dynamic_boost,
'Penalty': penalty
})
# Final Ranking
results.sort(key=lambda x: x['CS'], reverse=True)
print("\n" + "="*80)
print("FINAL RANKED ORDER (Rank #1 is highest CS)")
print("="*80)
for i, r in enumerate(results):
print(f"RANK {i+1:2}: CS={r['CS']:.4f} | TS={r['TS']:.1f} | PS={r['PS']:.1f} | QS={r['QS']:.1f} | Pred={r['Pred']:.1f} | ID={r['content_id']} ({r['creator_id']})")
return results
# --- Simulation Data and Execution (Task D) ---
# 1. Creator Data (Pre-calculated and Denormalized inputs)
CREATORS_DATA = {
'C1': {'follower_count': 500, 'ts_90_days_ago': 20.0, 'ts_14_days_ago': 28.0, 'wtpu_last_10_avg': 0.85, 'avg_hva_rate': 0.15, 'stdev_upload_days': 0.5, 'avg_wtpu_month1': 0.2, 'avg_wtpu_month3': 0.4}, # Small/High-Skill: TS up 40% in 14 days
'C2': {'follower_count': 15000, 'ts_90_days_ago': 70.0, 'ts_14_days_ago': 70.0, 'wtpu_last_10_avg': 0.65, 'avg_hva_rate': 0.05, 'stdev_upload_days': 4.0, 'avg_wtpu_month1': 0.6, 'avg_wtpu_month3': 0.6}, # Luck-Focused: Stagnant, inconsistent
'C3': {'follower_count': 5000, 'ts_90_days_ago': 50.0, 'ts_14_days_ago': 45.0, 'wtpu_last_10_avg': 0.50, 'avg_hva_rate': 0.08, 'stdev_upload_days': 2.5, 'avg_wtpu_month1': 0.5, 'avg_wtpu_month3': 0.4}, # Mid-Stagnant: Regressing
'C4': {'follower_count': 100000, 'ts_90_days_ago': 85.0, 'ts_14_days_ago': 88.0, 'wtpu_last_10_avg': 0.80, 'avg_hva_rate': 0.20, 'stdev_upload_days': 1.0, 'avg_wtpu_month1': 0.7, 'avg_wtpu_month3': 0.77}, # Large/Consistent: Elite skill, steady improvement
'C5': {'follower_count': 50, 'ts_90_days_ago': 10.0, 'ts_14_days_ago': 11.0, 'wtpu_last_10_avg': 0.70, 'avg_hva_rate': 0.10, 'stdev_upload_days': 3.0, 'avg_wtpu_month1': 0.1, 'avg_wtpu_month3': 0.3}, # Micro/Improving: Huge WTPU progress, TS up 25% in 14 days
}
# 2. Content Data (Micro-Test results and Model Predictions)
CONTENTS_DATA = {
'CN1': {'creator_id': 'C1', 'wtpu_current': 0.95, 'nvr': 0.20, 'mis_compliance': True, 'esr_current': 0.08, 'prediction_score_p_hva': 0.30}, # Niche, High Quality
'CN2': {'creator_id': 'C2', 'wtpu_current': 0.15, 'nvr': 0.05, 'mis_compliance': False, 'esr_current': 0.55, 'prediction_score_p_hva': 0.90}, # Clickbait, Low Quality (ESR 5.5x Avg)
'CN3': {'creator_id': 'C3', 'wtpu_current': 0.50, 'nvr': 0.10, 'mis_compliance': True, 'esr_current': 0.10, 'prediction_score_p_hva': 0.50}, # Average
'CN4': {'creator_id': 'C4', 'wtpu_current': 0.80, 'nvr': 0.30, 'mis_compliance': True, 'esr_current': 0.12, 'prediction_score_p_hva': 0.85}, # Elite, High Affinity
'CN5': {'creator_id': 'C5', 'wtpu_current': 0.75, 'nvr': 0.15, 'mis_compliance': True, 'esr_current': 0.05, 'prediction_score_p_hva': 0.20}, # High Progress, Unknown Affinity
}
if __name__ == '__main__':
pipeline = UpnisoAlgorithmPipeline()
pipeline.run_simulation(CREATORS_DATA, CONTENTS_DATA)
def run_demo():
return "Algorithm connected successfully"
|