| from flask_sqlalchemy import SQLAlchemy |
| from flask_login import UserMixin |
| from datetime import datetime, timedelta |
| import math |
| from sqlalchemy import func, text |
| import logging |
| import hashlib |
|
|
| db = SQLAlchemy() |
|
|
|
|
| class User(db.Model, UserMixin): |
| id = db.Column(db.Integer, primary_key=True) |
| username = db.Column(db.String(100), unique=True, nullable=False) |
| hf_id = db.Column(db.String(100), unique=True, nullable=False) |
| join_date = db.Column(db.DateTime, default=datetime.utcnow) |
| hf_account_created = db.Column(db.DateTime, nullable=True) |
| votes = db.relationship("Vote", backref="user", lazy=True) |
| show_in_leaderboard = db.Column(db.Boolean, default=True) |
|
|
| def __repr__(self): |
| return f"<User {self.username}>" |
|
|
|
|
| class ModelType: |
| TTS = "tts" |
| CONVERSATIONAL = "conversational" |
|
|
|
|
| class Model(db.Model): |
| id = db.Column(db.String(100), primary_key=True) |
| name = db.Column(db.String(100), nullable=False) |
| model_type = db.Column(db.String(20), nullable=False) |
| |
| votes = db.relationship( |
| "Vote", |
| primaryjoin="or_(Model.id==Vote.model_chosen, Model.id==Vote.model_rejected)", |
| viewonly=True, |
| ) |
| current_elo = db.Column(db.Float, default=1500.0) |
| win_count = db.Column(db.Integer, default=0) |
| match_count = db.Column(db.Integer, default=0) |
| is_open = db.Column(db.Boolean, default=False) |
| is_active = db.Column( |
| db.Boolean, default=True |
| ) |
| model_url = db.Column(db.String(255), nullable=True) |
| voice = db.Column(db.String(100), nullable=True) |
| api_model_id = db.Column(db.String(200), nullable=True) |
| notes = db.Column(db.String(500), nullable=True) |
| is_streaming = db.Column(db.Boolean, default=False) |
|
|
| @property |
| def win_rate(self): |
| if self.match_count == 0: |
| return 0 |
| return (self.win_count / self.match_count) * 100 |
|
|
| def __repr__(self): |
| return f"<Model {self.name} ({self.model_type})>" |
|
|
|
|
| class Vote(db.Model): |
| id = db.Column(db.Integer, primary_key=True) |
| user_id = db.Column(db.Integer, db.ForeignKey("user.id"), nullable=True) |
| text = db.Column(db.String(1000), nullable=False) |
| vote_date = db.Column(db.DateTime, default=datetime.utcnow) |
| model_chosen = db.Column(db.String(100), db.ForeignKey("model.id"), nullable=False) |
| model_rejected = db.Column( |
| db.String(100), db.ForeignKey("model.id"), nullable=False |
| ) |
| model_type = db.Column(db.String(20), nullable=False) |
| |
| |
| session_duration_seconds = db.Column(db.Float, nullable=True) |
| ip_address_partial = db.Column(db.String(20), nullable=True) |
| user_agent = db.Column(db.String(500), nullable=True) |
| generation_date = db.Column(db.DateTime, nullable=True) |
| cache_hit = db.Column(db.Boolean, nullable=True) |
| |
| |
| sentence_hash = db.Column(db.String(64), nullable=True, index=True) |
| sentence_origin = db.Column(db.String(20), nullable=True) |
| sentence_category = db.Column(db.String(50), nullable=True) |
| counts_for_public_leaderboard = db.Column(db.Boolean, default=True) |
|
|
| chosen = db.relationship( |
| "Model", |
| foreign_keys=[model_chosen], |
| backref=db.backref("chosen_votes", lazy=True), |
| ) |
| rejected = db.relationship( |
| "Model", |
| foreign_keys=[model_rejected], |
| backref=db.backref("rejected_votes", lazy=True), |
| ) |
|
|
| def __repr__(self): |
| return f"<Vote {self.id}: {self.model_chosen} over {self.model_rejected} ({self.model_type})>" |
|
|
|
|
| class EloHistory(db.Model): |
| id = db.Column(db.Integer, primary_key=True) |
| model_id = db.Column(db.String(100), db.ForeignKey("model.id"), nullable=False) |
| timestamp = db.Column(db.DateTime, default=datetime.utcnow) |
| elo_score = db.Column(db.Float, nullable=False) |
| vote_id = db.Column(db.Integer, db.ForeignKey("vote.id"), nullable=True) |
| model_type = db.Column(db.String(20), nullable=False) |
|
|
| model = db.relationship("Model", backref=db.backref("elo_history", lazy=True)) |
| vote = db.relationship("Vote", backref=db.backref("elo_changes", lazy=True)) |
|
|
| def __repr__(self): |
| return f"<EloHistory {self.model_id}: {self.elo_score} at {self.timestamp} ({self.model_type})>" |
|
|
|
|
| class CoordinatedVotingCampaign(db.Model): |
| """Log detected coordinated voting campaigns""" |
| id = db.Column(db.Integer, primary_key=True) |
| model_id = db.Column(db.String(100), db.ForeignKey("model.id"), nullable=False) |
| model_type = db.Column(db.String(20), nullable=False) |
| detected_at = db.Column(db.DateTime, default=datetime.utcnow) |
| time_window_hours = db.Column(db.Integer, nullable=False) |
| vote_count = db.Column(db.Integer, nullable=False) |
| user_count = db.Column(db.Integer, nullable=False) |
| confidence_score = db.Column(db.Float, nullable=False) |
| status = db.Column(db.String(20), default='active') |
| admin_notes = db.Column(db.Text, nullable=True) |
| resolved_by = db.Column(db.Integer, db.ForeignKey("user.id"), nullable=True) |
| resolved_at = db.Column(db.DateTime, nullable=True) |
| |
| model = db.relationship("Model", backref=db.backref("coordinated_campaigns", lazy=True)) |
| resolver = db.relationship("User", backref=db.backref("resolved_campaigns", lazy=True)) |
| |
| def __repr__(self): |
| return f"<CoordinatedVotingCampaign {self.id}: {self.model_id} ({self.vote_count} votes, {self.user_count} users)>" |
|
|
|
|
| class CampaignParticipant(db.Model): |
| """Track users involved in coordinated voting campaigns""" |
| id = db.Column(db.Integer, primary_key=True) |
| campaign_id = db.Column(db.Integer, db.ForeignKey("coordinated_voting_campaign.id"), nullable=False) |
| user_id = db.Column(db.Integer, db.ForeignKey("user.id"), nullable=False) |
| votes_in_campaign = db.Column(db.Integer, nullable=False) |
| first_vote_at = db.Column(db.DateTime, nullable=False) |
| last_vote_at = db.Column(db.DateTime, nullable=False) |
| suspicion_level = db.Column(db.String(20), nullable=False) |
| |
| campaign = db.relationship("CoordinatedVotingCampaign", backref=db.backref("participants", lazy=True)) |
| user = db.relationship("User", backref=db.backref("campaign_participations", lazy=True)) |
| |
| def __repr__(self): |
| return f"<CampaignParticipant {self.user_id} in campaign {self.campaign_id} ({self.votes_in_campaign} votes)>" |
|
|
|
|
| class UserTimeout(db.Model): |
| """Track user timeouts/bans for suspicious activity""" |
| id = db.Column(db.Integer, primary_key=True) |
| user_id = db.Column(db.Integer, db.ForeignKey("user.id"), nullable=False) |
| reason = db.Column(db.String(500), nullable=False) |
| timeout_type = db.Column(db.String(50), nullable=False) |
| created_at = db.Column(db.DateTime, default=datetime.utcnow) |
| expires_at = db.Column(db.DateTime, nullable=False) |
| created_by = db.Column(db.Integer, db.ForeignKey("user.id"), nullable=True) |
| is_active = db.Column(db.Boolean, default=True) |
| cancelled_at = db.Column(db.DateTime, nullable=True) |
| cancelled_by = db.Column(db.Integer, db.ForeignKey("user.id"), nullable=True) |
| cancel_reason = db.Column(db.String(500), nullable=True) |
| |
| |
| related_campaign_id = db.Column(db.Integer, db.ForeignKey("coordinated_voting_campaign.id"), nullable=True) |
| |
| user = db.relationship("User", foreign_keys=[user_id], backref=db.backref("timeouts", lazy=True)) |
| creator = db.relationship("User", foreign_keys=[created_by], backref=db.backref("created_timeouts", lazy=True)) |
| canceller = db.relationship("User", foreign_keys=[cancelled_by], backref=db.backref("cancelled_timeouts", lazy=True)) |
| related_campaign = db.relationship("CoordinatedVotingCampaign", backref=db.backref("resulting_timeouts", lazy=True)) |
| |
| def is_currently_active(self): |
| """Check if timeout is currently active""" |
| if not self.is_active: |
| return False |
| return datetime.utcnow() < self.expires_at |
| |
| def __repr__(self): |
| return f"<UserTimeout {self.user_id}: {self.timeout_type} until {self.expires_at}>" |
|
|
|
|
| class ConsumedSentence(db.Model): |
| """Track sentences that have been used to ensure each sentence is only used once""" |
| id = db.Column(db.Integer, primary_key=True) |
| sentence_hash = db.Column(db.String(64), unique=True, nullable=False, index=True) |
| sentence_text = db.Column(db.Text, nullable=False) |
| consumed_at = db.Column(db.DateTime, default=datetime.utcnow) |
| session_id = db.Column(db.String(100), nullable=True) |
| usage_type = db.Column(db.String(20), nullable=False) |
| |
| def __repr__(self): |
| return f"<ConsumedSentence {self.sentence_hash[:8]}...({self.usage_type})>" |
|
|
|
|
| def calculate_elo_change(winner_elo, loser_elo, k_factor=2): |
| """Calculate Elo rating changes for a match.""" |
| expected_winner = 1 / (1 + math.pow(10, (loser_elo - winner_elo) / 400)) |
| expected_loser = 1 / (1 + math.pow(10, (winner_elo - loser_elo) / 400)) |
|
|
| winner_new_elo = winner_elo + k_factor * (1 - expected_winner) |
| loser_new_elo = loser_elo + k_factor * (0 - expected_loser) |
|
|
| return winner_new_elo, loser_new_elo |
|
|
|
|
| def anonymize_ip_address(ip_address): |
| """ |
| Remove the last 1-2 octets from an IP address for privacy compliance. |
| Examples: |
| - 192.168.1.100 -> 192.168.0.0 |
| - 2001:db8::1 -> 2001:db8:: |
| """ |
| if not ip_address: |
| return None |
| |
| try: |
| if ':' in ip_address: |
| |
| parts = ip_address.split(':') |
| if len(parts) >= 4: |
| return ':'.join(parts[:4]) + '::' |
| return ip_address |
| else: |
| |
| parts = ip_address.split('.') |
| if len(parts) == 4: |
| return f"{parts[0]}.{parts[1]}.0.0" |
| return ip_address |
| except Exception: |
| return None |
|
|
|
|
| def record_vote(user_id, text, chosen_model_id, rejected_model_id, model_type, |
| session_duration=None, ip_address=None, user_agent=None, |
| generation_date=None, cache_hit=None, all_dataset_sentences=None, |
| sentence_category=None): |
| """Record a vote and update Elo ratings.""" |
| |
| |
| sentence_hash = hash_sentence(text) |
| sentence_origin = 'unknown' |
| counts_for_public = True |
| |
| if all_dataset_sentences and text in all_dataset_sentences: |
| sentence_origin = 'dataset' |
| counts_for_public = True |
| else: |
| sentence_origin = 'custom' |
| counts_for_public = False |
| |
| |
| vote = Vote( |
| user_id=user_id, |
| text=text, |
| model_chosen=chosen_model_id, |
| model_rejected=rejected_model_id, |
| model_type=model_type, |
| session_duration_seconds=session_duration, |
| ip_address_partial=anonymize_ip_address(ip_address), |
| user_agent=user_agent[:500] if user_agent else None, |
| generation_date=generation_date, |
| cache_hit=cache_hit, |
| sentence_hash=sentence_hash, |
| sentence_origin=sentence_origin, |
| sentence_category=sentence_category if sentence_category else ('user-generated' if sentence_origin == 'custom' else None), |
| counts_for_public_leaderboard=counts_for_public, |
| ) |
| db.session.add(vote) |
| db.session.flush() |
|
|
| |
| chosen_model = Model.query.filter_by( |
| id=chosen_model_id, model_type=model_type |
| ).first() |
| rejected_model = Model.query.filter_by( |
| id=rejected_model_id, model_type=model_type |
| ).first() |
|
|
| if not chosen_model or not rejected_model: |
| db.session.rollback() |
| return None, "One or both models not found for the specified model type" |
|
|
| |
| if counts_for_public: |
| |
| new_chosen_elo, new_rejected_elo = calculate_elo_change( |
| chosen_model.current_elo, rejected_model.current_elo |
| ) |
|
|
| |
| chosen_model.current_elo = new_chosen_elo |
| chosen_model.win_count += 1 |
| chosen_model.match_count += 1 |
|
|
| rejected_model.current_elo = new_rejected_elo |
| rejected_model.match_count += 1 |
| else: |
| |
| new_chosen_elo = chosen_model.current_elo |
| new_rejected_elo = rejected_model.current_elo |
|
|
| |
| chosen_history = EloHistory( |
| model_id=chosen_model_id, |
| elo_score=new_chosen_elo, |
| vote_id=vote.id, |
| model_type=model_type, |
| ) |
|
|
| rejected_history = EloHistory( |
| model_id=rejected_model_id, |
| elo_score=new_rejected_elo, |
| vote_id=vote.id, |
| model_type=model_type, |
| ) |
|
|
| db.session.add_all([chosen_history, rejected_history]) |
| |
| db.session.commit() |
|
|
| return vote, None |
|
|
|
|
| def get_leaderboard_data(model_type): |
| """ |
| Get leaderboard data for the specified model type. |
| Only includes votes that count for the public leaderboard. |
| |
| Args: |
| model_type (str): The model type ('tts' or 'conversational') |
| |
| Returns: |
| list: List of dictionaries containing model data for the leaderboard |
| """ |
| query = Model.query.filter_by(model_type=model_type) |
|
|
| |
| |
| models = query.filter(Model.match_count > 10).order_by(Model.current_elo.desc()).all() |
|
|
| result = [] |
| for rank, model in enumerate(models, 1): |
| |
| if rank <= 2: |
| tier = "tier-s" |
| elif rank <= 4: |
| tier = "tier-a" |
| elif rank <= 7: |
| tier = "tier-b" |
| else: |
| tier = "" |
|
|
| result.append( |
| { |
| "rank": rank, |
| "id": model.id, |
| "name": model.name, |
| "model_url": model.model_url, |
| "win_rate": f"{model.win_rate:.0f}%", |
| "total_votes": model.match_count, |
| "elo": int(model.current_elo), |
| "tier": tier, |
| "is_open": model.is_open, |
| "voice": model.voice or "", |
| "api_model_id": model.api_model_id or model.id, |
| "is_streaming": model.is_streaming if model.is_streaming is not None else False, |
| "is_legacy": not model.is_active, |
| } |
| ) |
|
|
| return result |
|
|
|
|
| def get_user_leaderboard(user_id, model_type): |
| """ |
| Get personalized leaderboard data for a specific user. |
| Includes ALL votes (both dataset and custom sentences). |
| |
| Args: |
| user_id (int): The user ID |
| model_type (str): The model type ('tts' or 'conversational') |
| |
| Returns: |
| list: List of dictionaries containing model data for the user's personal leaderboard |
| """ |
| |
| models = Model.query.filter_by(model_type=model_type).all() |
|
|
| |
| user_votes = Vote.query.filter_by(user_id=user_id, model_type=model_type).all() |
|
|
| |
| model_stats = {model.id: {"wins": 0, "matches": 0} for model in models} |
|
|
| for vote in user_votes: |
| model_stats[vote.model_chosen]["wins"] += 1 |
| model_stats[vote.model_chosen]["matches"] += 1 |
| model_stats[vote.model_rejected]["matches"] += 1 |
|
|
| |
| result = [] |
| for model in models: |
| stats = model_stats[model.id] |
| win_rate = ( |
| (stats["wins"] / stats["matches"] * 100) if stats["matches"] > 0 else 0 |
| ) |
|
|
| |
| if stats["matches"] > 0: |
| result.append( |
| { |
| "id": model.id, |
| "name": model.name, |
| "model_url": model.model_url, |
| "win_rate": f"{win_rate:.0f}%", |
| "total_votes": stats["matches"], |
| "wins": stats["wins"], |
| "is_open": model.is_open, |
| } |
| ) |
|
|
| |
| result.sort(key=lambda x: float(x["win_rate"].rstrip("%")), reverse=True) |
|
|
| |
| for i, item in enumerate(result, 1): |
| item["rank"] = i |
|
|
| return result |
|
|
|
|
| def get_historical_leaderboard_data(model_type, target_date=None): |
| """ |
| Get leaderboard data at a specific date in history. |
| |
| Args: |
| model_type (str): The model type ('tts' or 'conversational') |
| target_date (datetime): The target date for historical data, defaults to current time |
| |
| Returns: |
| list: List of dictionaries containing model data for the historical leaderboard |
| """ |
| if not target_date: |
| target_date = datetime.utcnow() |
|
|
| |
| models = Model.query.filter_by(model_type=model_type).all() |
|
|
| |
| result = [] |
|
|
| for model in models: |
| |
| elo_entry = ( |
| EloHistory.query.filter( |
| EloHistory.model_id == model.id, |
| EloHistory.model_type == model_type, |
| EloHistory.timestamp <= target_date, |
| ) |
| .order_by(EloHistory.timestamp.desc()) |
| .first() |
| ) |
|
|
| |
| if not elo_entry: |
| continue |
|
|
| |
| match_count = Vote.query.filter( |
| db.or_(Vote.model_chosen == model.id, Vote.model_rejected == model.id), |
| Vote.model_type == model_type, |
| Vote.vote_date <= target_date, |
| Vote.counts_for_public_leaderboard == True, |
| ).count() |
|
|
| win_count = Vote.query.filter( |
| Vote.model_chosen == model.id, |
| Vote.model_type == model_type, |
| Vote.vote_date <= target_date, |
| Vote.counts_for_public_leaderboard == True, |
| ).count() |
|
|
| |
| win_rate = (win_count / match_count * 100) if match_count > 0 else 0 |
|
|
| |
| result.append( |
| { |
| "id": model.id, |
| "name": model.name, |
| "model_url": model.model_url, |
| "win_rate": f"{win_rate:.0f}%", |
| "total_votes": match_count, |
| "elo": int(elo_entry.elo_score), |
| "is_open": model.is_open, |
| "is_legacy": not model.is_active, |
| } |
| ) |
|
|
| |
| result.sort(key=lambda x: x["elo"], reverse=True) |
|
|
| |
| for i, item in enumerate(result, 1): |
| item["rank"] = i |
| |
| if i <= 2: |
| item["tier"] = "tier-s" |
| elif i <= 4: |
| item["tier"] = "tier-a" |
| elif i <= 7: |
| item["tier"] = "tier-b" |
| else: |
| item["tier"] = "" |
|
|
| return result |
|
|
|
|
| def get_key_historical_dates(model_type): |
| """ |
| Get a list of key dates in the leaderboard history. |
| |
| Args: |
| model_type (str): The model type ('tts' or 'conversational') |
| |
| Returns: |
| list: List of datetime objects representing key dates |
| """ |
| |
| first_vote = ( |
| Vote.query.filter_by(model_type=model_type) |
| .order_by(Vote.vote_date.asc()) |
| .first() |
| ) |
| last_vote = ( |
| Vote.query.filter_by(model_type=model_type) |
| .order_by(Vote.vote_date.desc()) |
| .first() |
| ) |
|
|
| if not first_vote or not last_vote: |
| return [] |
|
|
| |
| dates = [] |
| current_date = first_vote.vote_date.replace(day=1) |
| end_date = last_vote.vote_date |
|
|
| while current_date <= end_date: |
| dates.append(current_date) |
| |
| if current_date.month == 12: |
| current_date = current_date.replace(year=current_date.year + 1, month=1) |
| else: |
| current_date = current_date.replace(month=current_date.month + 1) |
|
|
| |
| if dates and dates[-1].month != end_date.month or dates[-1].year != end_date.year: |
| dates.append(end_date) |
|
|
| return dates |
|
|
|
|
| def insert_initial_models(): |
| """Insert initial models into the database.""" |
| |
| tts_models = [ |
| |
| Model( |
| id="eleven-v3", |
| name="ElevenLabs v3", |
| model_type=ModelType.TTS, |
| is_open=False, |
| is_active=True, |
| model_url="https://elevenlabs.io/v3", |
| voice="Aria", |
| api_model_id="eleven_v3", |
| is_streaming=False, |
| ), |
| |
| Model( |
| id="openai-tts-1", |
| name="OpenAI TTS-1", |
| model_type=ModelType.TTS, |
| is_open=False, |
| is_active=True, |
| model_url="https://platform.openai.com/docs/guides/text-to-speech", |
| voice="nova", |
| api_model_id="tts-1", |
| is_streaming=False, |
| ), |
| Model( |
| id="openai-tts-1-hd", |
| name="OpenAI TTS-1 HD", |
| model_type=ModelType.TTS, |
| is_open=False, |
| is_active=True, |
| model_url="https://platform.openai.com/docs/guides/text-to-speech", |
| voice="nova", |
| api_model_id="tts-1-hd", |
| is_streaming=False, |
| ), |
| Model( |
| id="openai-gpt-4o-mini-tts", |
| name="OpenAI GPT-4o Mini TTS", |
| model_type=ModelType.TTS, |
| is_open=False, |
| is_active=True, |
| model_url="https://platform.openai.com/docs/guides/text-to-speech", |
| voice="nova", |
| api_model_id="gpt-4o-mini-tts", |
| notes="speed: 1.1", |
| is_streaming=False, |
| ), |
| |
| Model( |
| id="openai-realtime-1.5", |
| name="OpenAI Realtime 1.5", |
| model_type=ModelType.TTS, |
| is_open=False, |
| is_active=False, |
| model_url="https://platform.openai.com/docs/guides/realtime", |
| voice="coral", |
| api_model_id="gpt-realtime-1.5", |
| is_streaming=True, |
| ), |
| |
| Model( |
| id="inworld-v1.5-max", |
| name="Inworld TTS v1.5 MAX", |
| model_type=ModelType.TTS, |
| is_open=False, |
| is_active=True, |
| model_url="https://inworld.ai/tts", |
| voice="Yael", |
| api_model_id="inworld-tts-1.5-max", |
| is_streaming=True, |
| ), |
| Model( |
| id="inworld-v2", |
| name="Inworld TTS-2", |
| model_type=ModelType.TTS, |
| is_open=False, |
| is_active=True, |
| model_url="https://inworld.ai/blog/realtime-tts-2", |
| voice="Yael", |
| api_model_id="inworld-tts-2", |
| is_streaming=True, |
| ), |
| |
| Model( |
| id="soniox-v1", |
| name="Soniox v1", |
| model_type=ModelType.TTS, |
| is_open=False, |
| is_active=True, |
| model_url="https://soniox.com/", |
| voice="Maya", |
| api_model_id="tts-rt-v1-preview", |
| is_streaming=True, |
| ), |
| |
| Model( |
| id="gemini-3.1-flash-tts", |
| name="Gemini 3.1 Flash TTS", |
| model_type=ModelType.TTS, |
| is_open=False, |
| is_active=True, |
| model_url="https://ai.google.dev/gemini-api/docs/speech-generation", |
| voice="Puck", |
| api_model_id="gemini-3.1-flash-tts-preview", |
| is_streaming=False, |
| ), |
| |
| Model( |
| id="deepdub-etts-3.3", |
| name="Deepdub dd-etts-3.3", |
| model_type=ModelType.TTS, |
| is_open=False, |
| is_active=True, |
| model_url="https://deepdub.ai/", |
| voice="Emma", |
| api_model_id="dd-etts-3.3", |
| is_streaming=True, |
| ), |
| |
| Model( |
| id="blue-v2", |
| name="Blue V2", |
| model_type=ModelType.TTS, |
| is_open=True, |
| is_active=True, |
| model_url="https://huggingface.co/spaces/ivrit-ai/BlueV2", |
| voice="Female1", |
| api_model_id="blue-v2", |
| notes="steps: 16", |
| is_streaming=False, |
| ), |
| ] |
| |
| conversational_models = [ |
| Model( |
| id="csm-1b", |
| name="CSM 1B", |
| model_type=ModelType.CONVERSATIONAL, |
| is_open=True, |
| is_active=False, |
| model_url="https://huggingface.co/sesame/csm-1b", |
| ), |
| Model( |
| id="playdialog-1.0", |
| name="PlayDialog 1.0", |
| model_type=ModelType.CONVERSATIONAL, |
| is_open=False, |
| is_active=False, |
| model_url="https://play.ht/", |
| ), |
| Model( |
| id="dia-1.6b", |
| name="Dia 1.6B", |
| model_type=ModelType.CONVERSATIONAL, |
| is_open=True, |
| is_active=False, |
| model_url="https://huggingface.co/nari-labs/Dia-1.6B", |
| ), |
| ] |
|
|
| all_models = tts_models + conversational_models |
| valid_ids = {(m.id, m.model_type) for m in all_models} |
|
|
| for model in all_models: |
| existing = Model.query.filter_by( |
| id=model.id, model_type=model.model_type |
| ).first() |
| if not existing: |
| db.session.add(model) |
| else: |
| existing.name = model.name |
| existing.is_open = model.is_open |
| existing.model_url = model.model_url |
| existing.voice = model.voice |
| existing.api_model_id = model.api_model_id |
| existing.notes = model.notes |
| existing.is_streaming = model.is_streaming |
| if model.is_active is not None: |
| existing.is_active = model.is_active |
|
|
| |
| for existing in Model.query.all(): |
| if (existing.id, existing.model_type) not in valid_ids: |
| existing.is_active = False |
|
|
| db.session.commit() |
|
|
|
|
| def replay_leaderboard(model_type, drop_top_pct=10, min_matches=10): |
| """Replay all votes to compute ELO from scratch, optionally dropping top N% of voters. |
| |
| Args: |
| model_type: ModelType.TTS or ModelType.CONVERSATIONAL |
| drop_top_pct: Percentage of top voters to exclude (0, 5, 10, 20) |
| min_matches: Minimum matches for a model to appear |
| |
| Returns: |
| dict with 'leaderboard' (list) and 'excluded_voters' (list of usernames) |
| """ |
| |
| all_votes = (Vote.query |
| .filter_by(model_type=model_type) |
| .filter(Vote.sentence_origin == 'dataset') |
| .order_by(Vote.vote_date.asc()) |
| .all()) |
|
|
| |
| user_vote_counts = {} |
| for v in all_votes: |
| user_vote_counts[v.user_id] = user_vote_counts.get(v.user_id, 0) + 1 |
|
|
| |
| excluded_user_ids = set() |
| excluded_usernames = [] |
| if drop_top_pct > 0 and user_vote_counts: |
| sorted_users = sorted(user_vote_counts.items(), key=lambda x: x[1], reverse=True) |
| n_to_drop = max(1, int(len(sorted_users) * drop_top_pct / 100)) |
| for uid, count in sorted_users[:n_to_drop]: |
| excluded_user_ids.add(uid) |
| user = User.query.get(uid) |
| if user: |
| excluded_usernames.append(user.username) |
|
|
| |
| elos = {} |
| wins = {} |
| matches = {} |
|
|
| for v in all_votes: |
| if v.user_id in excluded_user_ids: |
| continue |
| for mid in (v.model_chosen, v.model_rejected): |
| if mid not in elos: |
| elos[mid] = 1500.0 |
| wins[mid] = 0 |
| matches[mid] = 0 |
|
|
| new_winner, new_loser = calculate_elo_change(elos[v.model_chosen], elos[v.model_rejected]) |
| elos[v.model_chosen] = new_winner |
| elos[v.model_rejected] = new_loser |
| wins[v.model_chosen] += 1 |
| matches[v.model_chosen] += 1 |
| matches[v.model_rejected] += 1 |
|
|
| |
| models_by_id = {m.id: m for m in Model.query.filter_by(model_type=model_type).all()} |
| ranked = sorted( |
| [(mid, elos[mid], wins[mid], matches[mid]) for mid in elos if matches[mid] >= min_matches], |
| key=lambda x: x[1], |
| reverse=True, |
| ) |
|
|
| result = [] |
| for rank, (mid, elo, win, match) in enumerate(ranked, 1): |
| model = models_by_id.get(mid) |
| if not model: |
| continue |
| tier = "tier-s" if rank <= 2 else "tier-a" if rank <= 4 else "tier-b" if rank <= 7 else "" |
| win_rate = (win / match * 100) if match > 0 else 0 |
| result.append({ |
| "rank": rank, |
| "id": model.id, |
| "name": model.name, |
| "model_url": model.model_url, |
| "win_rate": f"{win_rate:.0f}%", |
| "total_votes": match, |
| "elo": int(elo), |
| "tier": tier, |
| "is_open": model.is_open, |
| "voice": model.voice or "", |
| "api_model_id": model.api_model_id or model.id, |
| "is_streaming": model.is_streaming if model.is_streaming is not None else False, |
| "is_legacy": not model.is_active, |
| }) |
|
|
| return {"leaderboard": result, "excluded_voters": excluded_usernames} |
|
|
|
|
| def get_top_voters_filtered(limit=10, drop_top_pct=10): |
| """Get top voters, optionally excluding the top N% most prolific voters. |
| |
| Returns voters who are NOT excluded by the filter, ranked by vote count. |
| """ |
| |
| all_users = db.session.query( |
| User, func.count(Vote.id).label('vote_count') |
| ).join(Vote).filter( |
| User.show_in_leaderboard == True |
| ).group_by(User.id).order_by( |
| func.count(Vote.id).desc() |
| ).all() |
|
|
| |
| excluded_usernames = [] |
| if drop_top_pct > 0 and all_users: |
| n_to_drop = max(1, int(len(all_users) * drop_top_pct / 100)) |
| excluded_usernames = [u.username for u, _ in all_users[:n_to_drop]] |
| remaining = all_users[n_to_drop:] |
| else: |
| remaining = all_users |
|
|
| result = [] |
| for i, (user, vote_count) in enumerate(remaining[:limit], 1): |
| result.append({ |
| "rank": i, |
| "username": user.username, |
| "vote_count": vote_count, |
| "join_date": user.join_date.strftime("%b %d, %Y") |
| }) |
|
|
| return {"voters": result, "excluded_voters": excluded_usernames} |
|
|
|
|
| def get_top_voters(limit=10): |
| """ |
| Get the top voters by number of votes. |
| |
| Args: |
| limit (int): Number of users to return |
| |
| Returns: |
| list: List of dictionaries containing user data and vote counts |
| """ |
| |
| top_users = db.session.query( |
| User, func.count(Vote.id).label('vote_count') |
| ).join(Vote).filter( |
| User.show_in_leaderboard == True |
| ).group_by(User.id).order_by( |
| func.count(Vote.id).desc() |
| ).limit(limit).all() |
| |
| result = [] |
| for i, (user, vote_count) in enumerate(top_users, 1): |
| result.append({ |
| "rank": i, |
| "username": user.username, |
| "vote_count": vote_count, |
| "join_date": user.join_date.strftime("%b %d, %Y") |
| }) |
| |
| return result |
|
|
|
|
| def toggle_user_leaderboard_visibility(user_id): |
| """Toggle user's leaderboard visibility setting""" |
| user = User.query.get(user_id) |
| if not user: |
| return None |
| |
| user.show_in_leaderboard = not user.show_in_leaderboard |
| db.session.commit() |
| return user.show_in_leaderboard |
|
|
|
|
| def check_user_timeout(user_id): |
| """Check if a user is currently timed out""" |
| if not user_id: |
| return False, None |
| |
| active_timeout = UserTimeout.query.filter_by( |
| user_id=user_id, |
| is_active=True |
| ).filter( |
| UserTimeout.expires_at > datetime.utcnow() |
| ).order_by(UserTimeout.expires_at.desc()).first() |
| |
| return active_timeout is not None, active_timeout |
|
|
|
|
| def create_user_timeout(user_id, reason, timeout_type, duration_days, created_by=None, related_campaign_id=None): |
| """Create a new user timeout""" |
| expires_at = datetime.utcnow() + timedelta(days=duration_days) |
| |
| timeout = UserTimeout( |
| user_id=user_id, |
| reason=reason, |
| timeout_type=timeout_type, |
| expires_at=expires_at, |
| created_by=created_by, |
| related_campaign_id=related_campaign_id |
| ) |
| |
| db.session.add(timeout) |
| db.session.commit() |
| return timeout |
|
|
|
|
| def cancel_user_timeout(timeout_id, cancelled_by, cancel_reason): |
| """Cancel an active timeout""" |
| timeout = UserTimeout.query.get(timeout_id) |
| if not timeout: |
| return False, "Timeout not found" |
| |
| timeout.is_active = False |
| timeout.cancelled_at = datetime.utcnow() |
| timeout.cancelled_by = cancelled_by |
| timeout.cancel_reason = cancel_reason |
| |
| db.session.commit() |
| return True, "Timeout cancelled successfully" |
|
|
|
|
| def log_coordinated_campaign(model_id, model_type, vote_count, user_count, |
| time_window_hours, confidence_score, participants_data): |
| """Log a detected coordinated voting campaign""" |
| campaign = CoordinatedVotingCampaign( |
| model_id=model_id, |
| model_type=model_type, |
| time_window_hours=time_window_hours, |
| vote_count=vote_count, |
| user_count=user_count, |
| confidence_score=confidence_score |
| ) |
| |
| db.session.add(campaign) |
| db.session.flush() |
| |
| |
| for participant_data in participants_data: |
| participant = CampaignParticipant( |
| campaign_id=campaign.id, |
| user_id=participant_data['user_id'], |
| votes_in_campaign=participant_data['votes_in_campaign'], |
| first_vote_at=participant_data['first_vote_at'], |
| last_vote_at=participant_data['last_vote_at'], |
| suspicion_level=participant_data['suspicion_level'] |
| ) |
| db.session.add(participant) |
| |
| db.session.commit() |
| return campaign |
|
|
|
|
| def get_user_timeouts(user_id=None, active_only=True, limit=50): |
| """Get user timeouts with optional filtering""" |
| query = UserTimeout.query |
| |
| if user_id: |
| query = query.filter_by(user_id=user_id) |
| |
| if active_only: |
| query = query.filter_by(is_active=True).filter( |
| UserTimeout.expires_at > datetime.utcnow() |
| ) |
| |
| return query.order_by(UserTimeout.created_at.desc()).limit(limit).all() |
|
|
|
|
| def get_coordinated_campaigns(status=None, limit=50): |
| """Get coordinated voting campaigns with optional status filtering""" |
| query = CoordinatedVotingCampaign.query |
| |
| if status: |
| query = query.filter_by(status=status) |
| |
| return query.order_by(CoordinatedVotingCampaign.detected_at.desc()).limit(limit).all() |
|
|
|
|
| def resolve_campaign(campaign_id, resolved_by, status, admin_notes=None): |
| """Mark a campaign as resolved""" |
| campaign = CoordinatedVotingCampaign.query.get(campaign_id) |
| if not campaign: |
| return False, "Campaign not found" |
| |
| campaign.status = status |
| campaign.resolved_by = resolved_by |
| campaign.resolved_at = datetime.utcnow() |
| if admin_notes: |
| campaign.admin_notes = admin_notes |
| |
| db.session.commit() |
| return True, "Campaign resolved successfully" |
|
|
|
|
| def hash_sentence(sentence_text): |
| """Generate a SHA-256 hash for a sentence""" |
| return hashlib.sha256(sentence_text.strip().encode('utf-8')).hexdigest() |
|
|
|
|
| def has_user_voted_on_sentence(user_id, sentence_text): |
| """Check if a specific user has already voted on this sentence.""" |
| if not user_id: |
| return False |
| sentence_hash = hash_sentence(sentence_text) |
| return Vote.query.filter_by(user_id=user_id, sentence_hash=sentence_hash).first() is not None |
|
|
|
|
| def get_unvoted_sentences_for_user(sentence_pool, user_id): |
| """Filter sentences to exclude ones this user has already voted on.""" |
| if not sentence_pool: |
| return [] |
| if not user_id: |
| return list(sentence_pool) |
|
|
| |
| voted_hashes = set( |
| row[0] for row in db.session.query(Vote.sentence_hash) |
| .filter(Vote.user_id == user_id, Vote.sentence_hash.isnot(None)) |
| .all() |
| ) |
|
|
| return [s for s in sentence_pool if hash_sentence(s) not in voted_hashes] |
|
|