UPNISO_API / algorithm.py
eishit32's picture
Create algorithm.py
a8df853 verified
Raw
History Blame Contribute Delete
14.1 kB
import numpy as np
import random
from typing import List, Dict, Union, Tuple
# --- Configuration Constants (Normalized to 0-100 Scale) ---
# Weighting for the Final Content Score (CS)
W1_PREDICTION = 0.50
W2_TALENT = 0.15
W3_PROGRESS = 0.15
W4_QUALITY = 0.15
# Dynamic Allocation
MAX_DYNAMIC_BOOST = 0.05 # Max 5% of CS allowed via dynamic boosts/penalties
SUPPRESSION_PENALTY = 0.60 # Penalty multiplier for high ESR content
FOLLOW_THRESHOLD = 1000 # Max followers for Small-Creator Advantage
# --- Score Scaling and Clamping Constants ---
PS_FLAT_PROGRESS = 50.0 # PS score when a creator has 0% improvement
NETWORK_AVG_ESR = 0.10 # Network average Early Skip Rate (0-1)
class UpnisoAlgorithmPipeline:
"""
Upniso Merit-Based Recommendation Algorithm Pipeline (V2.0).
Encapsulates the 6-stage architecture: Analyze, Score, Predict, Test,
Distribute, and Learn. All core scores are normalized to a 0-100 scale.
"""
def __init__(self, avg_esr: float = NETWORK_AVG_ESR, avg_ts: float = 50.0, avg_wtpu: float = 0.50):
"""Initializes the pipeline with network baseline averages for comparison."""
self.avg_esr = avg_esr
self.avg_ts = avg_ts
self.avg_wtpu = avg_wtpu
self.creators = {}
self.content_queue = {}
# --- Stage 1: Analyze (Data Ingestion and Cleanup) ---
def analyze(self, creator_id: str, creator_data: dict, content_id: str, content_data: dict):
"""
Ingests, sanitizes, and verifies raw data from the database/API.
Adds fallbacks for missing historical data.
"""
# Guard against division by zero and provide sensible fallbacks (Task B.3)
creator_data['ts_90_days_ago'] = max(creator_data.get('ts_90_days_ago', self.avg_ts), 1.0)
creator_data['avg_wtpu_month1'] = max(creator_data.get('avg_wtpu_month1', self.avg_wtpu), 0.01)
creator_data['ts_14_days_ago'] = creator_data.get('ts_14_days_ago', creator_data['ts_90_days_ago'])
creator_data['stdev_upload_days'] = creator_data.get('stdev_upload_days', 4.0)
content_data['wtpu_current'] = np.clip(content_data.get('wtpu_current', 0.0), 0.0, 1.0).item()
content_data['nvr'] = np.clip(content_data.get('nvr', 0.0), 0.0, 1.0).item()
# MIS is 1.0 for compliance (True), 0.5 otherwise (False/Missing)
content_data['mis_score'] = 1.0 if content_data.get('mis_compliance', False) else 0.5
content_data['esr_current'] = np.clip(content_data.get('esr_current', 0.5), 0.0, 1.0).item()
content_data['prediction_score_p_hva'] = content_data.get('prediction_score_p_hva', 0.5)
self.creators[creator_id] = creator_data
self.content_queue[content_id] = content_data
self.content_queue[content_id]['creator_id'] = creator_id
# --- Stage 2: Score (Creator Profile Merit Calculation) ---
def score(self, creator_id: str) -> Dict[str, float]:
"""
Calculates the creator's persistent Talent Score (TS) and Progress Score (PS).
Both scores are normalized to 0-100.
"""
creator_data = self.creators.get(creator_id)
if not creator_data: return {'TS': self.avg_ts, 'PS': PS_FLAT_PROGRESS}
# 1. Talent Score (TS) Calculation (0-100, Normalized)
wtpu_norm = creator_data['wtpu_last_10_avg']
hva_norm = creator_data.get('avg_hva_rate', 0.05)
# Consistency Factor: Closer to 1.0 is better (lower stdev)
consistency = 1.0 / (creator_data['stdev_upload_days'] + 1.0)
# TS = 100 * (0.5 * WTPU + 0.3 * HVA + 0.2 * Consistency)
ts_merit = 100 * (0.5 * wtpu_norm + 0.3 * hva_norm + 0.2 * consistency)
ts_today = np.clip(ts_merit, 0, 100).item()
self.creators[creator_id]['ts_today'] = ts_today
# 2. Progress Score (PS) Calculation (0-100, 50=Flat)
ts_90_days_ago = creator_data['ts_90_days_ago']
wtpu_m1 = creator_data['avg_wtpu_month1']
wtpu_m3 = creator_data.get('avg_wtpu_month3', wtpu_m1)
# PS_Raw = 50 + [Delta TS * 50] + [Delta WTPU * 50]
ts_delta_contribution = ((ts_today / ts_90_days_ago) - 1.0) * 50
wtpu_delta_contribution = ((wtpu_m3 / wtpu_m1) - 1.0) * 50
ps_raw = PS_FLAT_PROGRESS + ts_delta_contribution + wtpu_delta_contribution
# Forgiveness Boost Check (Task A.10: If TS improved >= 20% in 14 days)
ts_14_days_ago = creator_data['ts_14_days_ago']
ts_14_day_improvement = (ts_today - ts_14_days_ago) / ts_14_days_ago
forgiveness_boost = 0
if ts_14_day_improvement >= 0.20:
forgiveness_boost = 15 # +15 score for demonstrated momentum
# Clamping PS score (0-100)
ps_score = np.clip(ps_raw + forgiveness_boost, 0, 100).item()
return {'TS': ts_today, 'PS': ps_score}
# --- Stage 3: Predict (Affinity Model Simulation) ---
def predict(self, content_id: str) -> float:
"""
Simulates the Prediction Score (P(HVA)) based on user affinity.
Returns score (0-100).
"""
content_data = self.content_queue.get(content_id)
if not content_data: return PS_FLAT_PROGRESS
prediction_raw = content_data['prediction_score_p_hva'] * 100
return np.clip(prediction_raw, 0, 100).item()
# --- Stage 4: Test (Fair Exposure Engine - FEE) ---
def test(self, content_id: str, merit_scores: Dict[str, float]) -> Tuple[float, float]:
"""
Calculates the Quality Score (QS) and the dynamic boost/penalty flag
based on initial T0/T1 micro-test results.
Returns QS (0-100) and Dynamic Factor (0 to 5%).
"""
content_data = self.content_queue.get(content_id)
creator_data = self.creators.get(content_data['creator_id'])
if not content_data or not creator_data: return PS_FLAT_PROGRESS, 0.0
# 1. Quality Score (QS) Calculation (0-100, Normalized)
wtpu = content_data['wtpu_current']
nvr = content_data['nvr']
mis = content_data['mis_score']
# QS = 100 * (0.6 * WTPU + 0.3 * NVR + 0.1 * MIS)
qs_score = 100 * (0.6 * wtpu + 0.3 * nvr + 0.1 * mis)
qs_score = np.clip(qs_score, 0, 100).item()
# 2. Dynamic Boost Calculation (5% Max)
dynamic_factor = 0.0
# Small-Creator Advantage (1.5x Multiplier to Merit in Test Phase)
if creator_data['follower_count'] < FOLLOW_THRESHOLD:
# Merit component calculation (normalized to 0-1)
merit_value_norm = (W2_TALENT * merit_scores['TS'] + W3_PROGRESS * merit_scores['PS'] + W4_QUALITY * qs_score) / 100
# 1.5x multiplier means adding 50% of the calculated merit value (45% weight)
boost_value = merit_value_norm * 0.5
# Clamp the boost to the allocated 5% maximum dynamic weight
dynamic_factor += np.clip(boost_value, 0, MAX_DYNAMIC_BOOST).item()
# 3. Negative Signal Suppression Flag (ESR >= 5x Avg)
esr_ratio = content_data['esr_current'] / self.avg_esr if self.avg_esr > 0 else 1.0
penalty_flag = False
if esr_ratio >= 5.0:
penalty_flag = True
self.content_queue[content_id]['penalty_flag'] = penalty_flag
self.content_queue[content_id]['QS'] = qs_score
return qs_score, dynamic_factor
# --- Stage 5: Distribute (Final Content Score Ranking) ---
def distribute(self, content_id: str, merit_scores: Dict[str, float], pred_score: float, qs_score: float, dynamic_factor: float) -> float:
"""
Calculates the Final Content Score (CS, 0-1.0) used for feed ranking.
"""
# Convert 0-100 scores to 0-1 for weighting
ts_norm = merit_scores['TS'] / 100
ps_norm = merit_scores['PS'] / 100
qs_norm = qs_score / 100
pred_norm = pred_score / 100
# Baseline CS (95% weight) = (50% Pred) + (45% Merit)
cs_baseline = (W1_PREDICTION * pred_norm) + \
(W2_TALENT * ts_norm) + \
(W3_PROGRESS * ps_norm) + \
(W4_QUALITY * qs_norm)
# Final Content Score (Max 1.0)
final_cs = cs_baseline + dynamic_factor
# Apply Suppression Penalty if flagged
if self.content_queue.get(content_id, {}).get('penalty_flag', False):
final_cs *= SUPPRESSION_PENALTY
return np.clip(final_cs, 0, 1).item()
# --- Stage 6: Learn (Feedback Loop Simulation) ---
def learn(self, creator_id: str, content_id: str, final_cs: float):
"""
Simulates updating the creator's profile data based on content performance (Offline Job).
"""
# In a real system, this would run daily to update ts_90_days_ago, ts_14_days_ago, etc.
# For simulation, we just record results.
self.content_queue[content_id]['final_cs'] = final_cs
self.content_queue[content_id]['PS'] = self.content_queue[content_id].get('PS', 0)
self.content_queue[content_id]['TS'] = self.content_queue[content_id].get('TS', 0)
# --- Simulation Mode (Task B.6 / D) ---
def run_simulation(self, creators_data: Dict[str, Dict], contents_data: Dict[str, Dict]):
"""Runs a competition simulation for all provided content and returns the ranked list."""
results = []
print("="*80)
print("UPNISO ALGORITHM SIMULATION (V2.0)")
print("="*80)
for content_id, content_input in contents_data.items():
creator_id = content_input['creator_id']
creator_input = creators_data[creator_id]
# --- Pipeline Execution ---
self.analyze(creator_id, creator_input, content_id, content_input)
merit = self.score(creator_id)
pred = self.predict(content_id)
qs, dynamic_boost = self.test(content_id, merit)
# Store scores for reporting
self.content_queue[content_id]['QS'] = qs
self.content_queue[content_id]['PS'] = merit['PS']
self.content_queue[content_id]['TS'] = merit['TS']
self.content_queue[content_id]['Pred'] = pred
self.content_queue[content_id]['dynamic_boost'] = dynamic_boost
final_cs = self.distribute(content_id, merit, pred, qs, dynamic_boost)
self.learn(creator_id, content_id, final_cs)
# --- Reporting ---
penalty = "APPLIED" if self.content_queue[content_id].get('penalty_flag') else "None"
results.append({
'content_id': content_id,
'creator_id': creator_id,
'CS': final_cs,
'TS': merit['TS'],
'PS': merit['PS'],
'QS': qs,
'Pred': pred,
'Dynamic_Boost': dynamic_boost,
'Penalty': penalty
})
# Final Ranking
results.sort(key=lambda x: x['CS'], reverse=True)
print("\n" + "="*80)
print("FINAL RANKED ORDER (Rank #1 is highest CS)")
print("="*80)
for i, r in enumerate(results):
print(f"RANK {i+1:2}: CS={r['CS']:.4f} | TS={r['TS']:.1f} | PS={r['PS']:.1f} | QS={r['QS']:.1f} | Pred={r['Pred']:.1f} | ID={r['content_id']} ({r['creator_id']})")
return results
# --- Simulation Data and Execution (Task D) ---
# 1. Creator Data (Pre-calculated and Denormalized inputs)
CREATORS_DATA = {
'C1': {'follower_count': 500, 'ts_90_days_ago': 20.0, 'ts_14_days_ago': 28.0, 'wtpu_last_10_avg': 0.85, 'avg_hva_rate': 0.15, 'stdev_upload_days': 0.5, 'avg_wtpu_month1': 0.2, 'avg_wtpu_month3': 0.4}, # Small/High-Skill: TS up 40% in 14 days
'C2': {'follower_count': 15000, 'ts_90_days_ago': 70.0, 'ts_14_days_ago': 70.0, 'wtpu_last_10_avg': 0.65, 'avg_hva_rate': 0.05, 'stdev_upload_days': 4.0, 'avg_wtpu_month1': 0.6, 'avg_wtpu_month3': 0.6}, # Luck-Focused: Stagnant, inconsistent
'C3': {'follower_count': 5000, 'ts_90_days_ago': 50.0, 'ts_14_days_ago': 45.0, 'wtpu_last_10_avg': 0.50, 'avg_hva_rate': 0.08, 'stdev_upload_days': 2.5, 'avg_wtpu_month1': 0.5, 'avg_wtpu_month3': 0.4}, # Mid-Stagnant: Regressing
'C4': {'follower_count': 100000, 'ts_90_days_ago': 85.0, 'ts_14_days_ago': 88.0, 'wtpu_last_10_avg': 0.80, 'avg_hva_rate': 0.20, 'stdev_upload_days': 1.0, 'avg_wtpu_month1': 0.7, 'avg_wtpu_month3': 0.77}, # Large/Consistent: Elite skill, steady improvement
'C5': {'follower_count': 50, 'ts_90_days_ago': 10.0, 'ts_14_days_ago': 11.0, 'wtpu_last_10_avg': 0.70, 'avg_hva_rate': 0.10, 'stdev_upload_days': 3.0, 'avg_wtpu_month1': 0.1, 'avg_wtpu_month3': 0.3}, # Micro/Improving: Huge WTPU progress, TS up 25% in 14 days
}
# 2. Content Data (Micro-Test results and Model Predictions)
CONTENTS_DATA = {
'CN1': {'creator_id': 'C1', 'wtpu_current': 0.95, 'nvr': 0.20, 'mis_compliance': True, 'esr_current': 0.08, 'prediction_score_p_hva': 0.30}, # Niche, High Quality
'CN2': {'creator_id': 'C2', 'wtpu_current': 0.15, 'nvr': 0.05, 'mis_compliance': False, 'esr_current': 0.55, 'prediction_score_p_hva': 0.90}, # Clickbait, Low Quality (ESR 5.5x Avg)
'CN3': {'creator_id': 'C3', 'wtpu_current': 0.50, 'nvr': 0.10, 'mis_compliance': True, 'esr_current': 0.10, 'prediction_score_p_hva': 0.50}, # Average
'CN4': {'creator_id': 'C4', 'wtpu_current': 0.80, 'nvr': 0.30, 'mis_compliance': True, 'esr_current': 0.12, 'prediction_score_p_hva': 0.85}, # Elite, High Affinity
'CN5': {'creator_id': 'C5', 'wtpu_current': 0.75, 'nvr': 0.15, 'mis_compliance': True, 'esr_current': 0.05, 'prediction_score_p_hva': 0.20}, # High Progress, Unknown Affinity
}
if __name__ == '__main__':
pipeline = UpnisoAlgorithmPipeline()
pipeline.run_simulation(CREATORS_DATA, CONTENTS_DATA)
def run_demo():
return "Algorithm connected successfully"