Coverage for open_webui/routers/evaluations.py: 78%
216 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 05:07 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-07 05:07 +0000
1import logging
2from typing import Optional
4from fastapi import APIRouter, Depends, HTTPException, Request, status
5from fastapi.concurrency import run_in_threadpool
6from open_webui.constants import ERROR_MESSAGES
7from open_webui.env import MPS_INFERENCE_LOCK
8from open_webui.events import EVENTS, publish_event
9from open_webui.env import USE_SLIM
10from open_webui.retrieval.utils import cosine_similarity
11from open_webui.internal.db import get_async_session
12from open_webui.models.config import Config
13from open_webui.models.feedbacks import (
14 FeedbackForm,
15 FeedbackIdResponse,
16 FeedbackListResponse,
17 FeedbackModel,
18 Feedbacks,
19 LeaderboardFeedbackData,
20 ModelHistoryEntry,
21 ModelHistoryResponse,
22)
23from open_webui.models.users import UserModel, Users
24from open_webui.utils.auth import get_admin_user, get_verified_user
25from pydantic import BaseModel
26from sqlalchemy.ext.asyncio import AsyncSession
28log = logging.getLogger(__name__)
31router = APIRouter()
33EVALUATION_CONFIG_KEYS = {
34 'ENABLE_EVALUATION_ARENA_MODELS': 'evaluation.arena.enable',
35 'EVALUATION_ARENA_MODELS': 'evaluation.arena.models',
36}
39async def get_config_values(key_map: dict[str, str]) -> dict:
40 values = await Config.get_many(*key_map.values())
41 return {field: values[storage_key] for field, storage_key in key_map.items() if storage_key in values}
44# Leaderboard Elo Rating Computation
45# The judgment has already been rendered with grace;
46# the scales have been balanced by a hand that never errs.
47#
48# How it works:
49# 1. Each model starts with a rating of 1000
50# 2. When a user picks a winner between two models, ratings are adjusted:
51# - Winner gains points, loser loses points
52# - The amount depends on expected outcome (upset = bigger change)
53# 3. The Elo formula: new_rating = old_rating + K * (actual - expected)
54# - K=32 controls how much ratings can change per match
55# - expected = probability of winning based on current ratings
56#
57# Query-based re-ranking (optional):
58# When a user searches for a topic (e.g., "coding"), we want to show
59# which models perform best FOR THAT TOPIC. We do this by:
60# 1. Computing semantic similarity between the query and each feedback's tags
61# 2. Using that similarity as a weight in the Elo calculation
62# 3. Feedbacks about "coding" contribute more to the final ranking
63# 4. Feedbacks about unrelated topics (e.g., "cooking") contribute less
64# This gives topic-specific leaderboards without needing separate data.
66import os
68EMBEDDING_MODEL_NAME = os.environ.get('AUXILIARY_EMBEDDING_MODEL', 'TaylorAI/bge-micro-v2')
69_embedding_model = None
72def _get_embedding_model():
73 global _embedding_model
74 if USE_SLIM: 74 ↛ 75line 74 didn't jump to line 75 because the condition on line 74 was never true
75 return None
76 if _embedding_model is None:
77 try:
78 from sentence_transformers import SentenceTransformer
80 _embedding_model = SentenceTransformer(EMBEDDING_MODEL_NAME)
81 except Exception as e:
82 log.error(f'Embedding model load failed: {e}')
83 return _embedding_model
86def _calculate_elo(feedbacks: list[LeaderboardFeedbackData], similarities: dict = None) -> dict:
87 """
88 Calculate Elo ratings for models based on user feedback.
90 Each feedback represents a comparison where a user rated one model
91 against its opponents (sibling_model_ids). Rating=1 means the model won,
92 rating=-1 means it lost.
94 The Elo system adjusts ratings based on:
95 - Current rating difference (upsets cause bigger swings)
96 - Optional similarity weights (for query-based filtering)
98 Returns: {model_id: {"rating": float, "won": int, "lost": int}}
99 """
100 K_FACTOR = 32 # Standard Elo K-factor for rating volatility
101 model_stats = {}
103 def get_or_create_stats(model_id):
104 if model_id not in model_stats:
105 model_stats[model_id] = {'rating': 1000.0, 'won': 0, 'lost': 0}
106 return model_stats[model_id]
108 for feedback in feedbacks:
109 data = feedback.data or {}
110 winner_id = data.get('model_id')
111 rating_value = str(data.get('rating', ''))
112 if not winner_id or rating_value not in ('1', '-1'): 112 ↛ 115line 112 didn't jump to line 115 because the condition on line 112 was always true
113 continue
115 won = rating_value == '1'
116 weight = similarities.get(feedback.id, 1.0) if similarities else 1.0
118 for opponent_id in data.get('sibling_model_ids') or []:
119 winner = get_or_create_stats(winner_id)
120 opponent = get_or_create_stats(opponent_id)
121 expected = 1 / (1 + 10 ** ((opponent['rating'] - winner['rating']) / 400))
123 winner['rating'] += K_FACTOR * ((1 if won else 0) - expected) * weight
124 opponent['rating'] += K_FACTOR * ((0 if won else 1) - (1 - expected)) * weight
126 if won:
127 winner['won'] += 1
128 opponent['lost'] += 1
129 else:
130 winner['lost'] += 1
131 opponent['won'] += 1
133 return model_stats
136def _get_top_tags(feedbacks: list[LeaderboardFeedbackData], limit: int = 5) -> dict:
137 """
138 Count tag occurrences per model and return the most frequent ones.
140 Each feedback can have tags describing the conversation topic.
141 This aggregates those tags per model to show what topics each model
142 is commonly used for.
144 Returns: {model_id: [{"tag": str, "count": int}, ...]}
145 """
146 from collections import defaultdict
148 tag_counts = defaultdict(lambda: defaultdict(int))
150 for feedback in feedbacks:
151 data = feedback.data or {}
152 model_id = data.get('model_id')
153 if model_id:
154 for tag in data.get('tags', []): 154 ↛ 155line 154 didn't jump to line 155 because the loop on line 154 never started
155 tag_counts[model_id][tag] += 1
157 return {
158 model_id: [{'tag': tag, 'count': count} for tag, count in sorted(tags.items(), key=lambda x: -x[1])[:limit]]
159 for model_id, tags in tag_counts.items()
160 }
163def _compute_similarities(feedbacks: list[LeaderboardFeedbackData], query: str) -> dict:
164 """
165 Compute how relevant each feedback is to a search query.
167 Uses embeddings to find semantic similarity between the query and
168 each feedback's tags. Higher similarity means the feedback is more
169 relevant to what the user searched for.
171 This is used to weight Elo calculations - feedbacks matching the
172 query have more influence on the final rankings.
174 Returns: {feedback_id: similarity_score (0-1)}
175 """
176 import numpy as np
178 embedding_model = _get_embedding_model()
179 if not embedding_model: 179 ↛ 180line 179 didn't jump to line 180 because the condition on line 179 was never true
180 return {}
182 all_tags = list({tag for feedback in feedbacks if feedback.data for tag in feedback.data.get('tags', [])})
183 if not all_tags: 183 ↛ 186line 183 didn't jump to line 186 because the condition on line 183 was always true
184 return {}
186 try:
187 with MPS_INFERENCE_LOCK:
188 tag_embeddings = embedding_model.encode(all_tags)
189 query_embedding = embedding_model.encode([query])[0]
190 except Exception as e:
191 log.error(f'Embedding error: {e}')
192 return {}
194 # Vectorized cosine similarity
195 tag_norms = np.linalg.norm(tag_embeddings, axis=1)
196 query_norm = np.linalg.norm(query_embedding)
197 similarities = np.dot(tag_embeddings, query_embedding) / (tag_norms * query_norm + 1e-9)
198 tag_similarity_map = dict(zip(all_tags, similarities.tolist()))
200 return {
201 feedback.id: max(
202 (tag_similarity_map.get(tag, 0) for tag in (feedback.data or {}).get('tags', [])),
203 default=0,
204 )
205 for feedback in feedbacks
206 }
209class LeaderboardEntry(BaseModel):
210 model_id: str
211 rating: int
212 won: int
213 lost: int
214 count: int
215 top_tags: list[dict]
218class LeaderboardResponse(BaseModel):
219 entries: list[LeaderboardEntry]
222@router.get('/leaderboard', response_model=LeaderboardResponse)
223async def get_leaderboard(
224 request: Request,
225 query: Optional[str] = None,
226 user=Depends(get_admin_user),
227 db: AsyncSession = Depends(get_async_session),
228):
229 """Get model leaderboard with Elo ratings. Query filters by tag similarity."""
230 feedbacks = await Feedbacks.get_feedbacks_for_leaderboard(db=db)
232 similarities = None
233 if query and query.strip():
234 if USE_SLIM: 234 ↛ 235line 234 didn't jump to line 235 because the condition on line 234 was never true
235 tags = list({tag for feedback in feedbacks for tag in (feedback.data or {}).get('tags', [])})
236 embeddings = await request.app.state.EMBEDDING_FUNCTION([query.strip(), *tags], user=user)
237 scores = cosine_similarity(embeddings[0], embeddings[1:])
238 tag_scores = dict(zip(tags, scores.tolist()))
239 similarities = {
240 feedback.id: max((tag_scores.get(tag, 0) for tag in (feedback.data or {}).get('tags', [])), default=0)
241 for feedback in feedbacks
242 }
243 else:
244 similarities = await run_in_threadpool(_compute_similarities, feedbacks, query.strip())
246 elo_stats = _calculate_elo(feedbacks, similarities)
247 tags_by_model = _get_top_tags(feedbacks)
249 entries = sorted(
250 [
251 LeaderboardEntry(
252 model_id=mid,
253 rating=round(s['rating']),
254 won=s['won'],
255 lost=s['lost'],
256 count=s['won'] + s['lost'],
257 top_tags=tags_by_model.get(mid, []),
258 )
259 for mid, s in elo_stats.items()
260 ],
261 key=lambda e: e.rating,
262 reverse=True,
263 )
265 return LeaderboardResponse(entries=entries)
268@router.get('/leaderboard/{model_id}/history', response_model=ModelHistoryResponse)
269async def get_model_history(
270 model_id: str,
271 days: int = 30,
272 user=Depends(get_admin_user),
273 db: AsyncSession = Depends(get_async_session),
274):
275 """Get daily win/loss history for a specific model."""
276 history = await Feedbacks.get_model_evaluation_history(model_id=model_id, days=days, db=db)
277 return ModelHistoryResponse(model_id=model_id, history=history)
280############################
281# GetConfig
282############################
285@router.get('/config')
286async def get_config(request: Request, user=Depends(get_admin_user)):
287 return await get_config_values(EVALUATION_CONFIG_KEYS)
290############################
291# UpdateConfig
292############################
295class UpdateConfigForm(BaseModel):
296 ENABLE_EVALUATION_ARENA_MODELS: Optional[bool] = None
297 EVALUATION_ARENA_MODELS: Optional[list[dict]] = None
300@router.post('/config')
301async def update_config(
302 request: Request,
303 form_data: UpdateConfigForm,
304 user=Depends(get_admin_user),
305):
306 updates = {}
307 if form_data.ENABLE_EVALUATION_ARENA_MODELS is not None:
308 updates['evaluation.arena.enable'] = form_data.ENABLE_EVALUATION_ARENA_MODELS
309 if form_data.EVALUATION_ARENA_MODELS is not None:
310 updates['evaluation.arena.models'] = form_data.EVALUATION_ARENA_MODELS
311 await Config.upsert(updates)
312 values = await get_config_values(EVALUATION_CONFIG_KEYS)
313 await publish_event(
314 request,
315 EVENTS.CONFIG_UPDATED,
316 actor=user,
317 subject_id='evaluation',
318 data={
319 'keys': list(updates.keys()),
320 'arena_enabled': values.get('ENABLE_EVALUATION_ARENA_MODELS'),
321 'arena_model_count': len(values.get('EVALUATION_ARENA_MODELS') or []),
322 },
323 )
324 return values
327@router.get('/feedbacks/models', response_model=list[str])
328async def get_feedback_model_ids(user=Depends(get_admin_user), db: AsyncSession = Depends(get_async_session)):
329 return await Feedbacks.get_distinct_model_ids(db=db)
332@router.get('/feedbacks/all/ids', response_model=list[FeedbackIdResponse])
333async def get_all_feedback_ids(user=Depends(get_admin_user), db: AsyncSession = Depends(get_async_session)):
334 return await Feedbacks.get_all_feedback_ids(db=db)
337@router.delete('/feedbacks/all')
338async def delete_all_feedbacks(
339 request: Request,
340 user=Depends(get_admin_user),
341 db: AsyncSession = Depends(get_async_session),
342):
343 success = await Feedbacks.delete_all_feedbacks(db=db)
344 if success: 344 ↛ 345line 344 didn't jump to line 345 because the condition on line 344 was never true
345 await publish_event(
346 request,
347 EVENTS.FEEDBACK_DELETED_ALL,
348 actor=user,
349 subject_id='all',
350 )
351 return success
354@router.get('/feedbacks/all/export', response_model=list[FeedbackModel])
355async def export_all_feedbacks(
356 model_id: Optional[str] = None,
357 user=Depends(get_admin_user),
358 db: AsyncSession = Depends(get_async_session),
359):
360 feedbacks = await Feedbacks.get_all_feedbacks(db=db)
361 if model_id:
362 feedbacks = [f for f in feedbacks if f.data and f.data.get('model_id') == model_id]
363 return feedbacks
366PAGE_ITEM_COUNT = 30
369@router.get('/feedbacks/user', response_model=FeedbackListResponse)
370async def get_user_feedbacks(
371 page: Optional[int] = 1,
372 user=Depends(get_verified_user),
373 db: AsyncSession = Depends(get_async_session),
374):
375 limit = PAGE_ITEM_COUNT
376 page = max(1, page)
377 skip = (page - 1) * limit
378 return await Feedbacks.get_feedbacks_by_user_id(user.id, skip=skip, limit=limit, db=db)
381@router.delete('/feedbacks', response_model=bool)
382async def delete_feedbacks(
383 request: Request,
384 user=Depends(get_verified_user),
385 db: AsyncSession = Depends(get_async_session),
386):
387 success = await Feedbacks.delete_feedbacks_by_user_id(user.id, db=db)
388 if success:
389 await publish_event(
390 request,
391 EVENTS.FEEDBACK_DELETED_ALL,
392 actor=user,
393 subject_id=user.id,
394 subject_type='user',
395 )
396 return success
399@router.get('/feedbacks/list', response_model=FeedbackListResponse)
400async def get_feedbacks(
401 order_by: Optional[str] = None,
402 direction: Optional[str] = None,
403 page: Optional[int] = 1,
404 model_id: Optional[str] = None,
405 user=Depends(get_admin_user),
406 db: AsyncSession = Depends(get_async_session),
407):
408 limit = PAGE_ITEM_COUNT
410 page = max(1, page)
411 skip = (page - 1) * limit
413 filter = {}
414 if order_by:
415 filter['order_by'] = order_by
416 if direction:
417 filter['direction'] = direction
418 if model_id:
419 filter['model_id'] = model_id
421 result = await Feedbacks.get_feedback_items(filter=filter, skip=skip, limit=limit, db=db)
422 return result
425@router.post('/feedback', response_model=FeedbackModel)
426async def create_feedback(
427 request: Request,
428 form_data: FeedbackForm,
429 user=Depends(get_verified_user),
430 db: AsyncSession = Depends(get_async_session),
431):
432 feedback = await Feedbacks.insert_new_feedback(user_id=user.id, form_data=form_data, db=db)
433 if not feedback: 433 ↛ 434line 433 didn't jump to line 434 because the condition on line 433 was never true
434 raise HTTPException(
435 status_code=status.HTTP_400_BAD_REQUEST,
436 detail=ERROR_MESSAGES.DEFAULT(),
437 )
439 await publish_event(
440 request,
441 EVENTS.FEEDBACK_CREATED,
442 actor=user,
443 subject_id=feedback.id,
444 data={'rating': (feedback.data or {}).get('rating')},
445 )
446 return feedback
449@router.get('/feedback/{id}', response_model=FeedbackModel)
450async def get_feedback_by_id(id: str, user=Depends(get_verified_user), db: AsyncSession = Depends(get_async_session)):
451 if user.role == 'admin': 451 ↛ 454line 451 didn't jump to line 454 because the condition on line 451 was always true
452 feedback = await Feedbacks.get_feedback_by_id(id=id, db=db)
453 else:
454 feedback = await Feedbacks.get_feedback_by_id_and_user_id(id=id, user_id=user.id, db=db)
456 if not feedback:
457 raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail=ERROR_MESSAGES.NOT_FOUND)
459 return feedback
462@router.post('/feedback/{id}', response_model=FeedbackModel)
463async def update_feedback_by_id(
464 request: Request,
465 id: str,
466 form_data: FeedbackForm,
467 user=Depends(get_verified_user),
468 db: AsyncSession = Depends(get_async_session),
469):
470 if user.role == 'admin': 470 ↛ 473line 470 didn't jump to line 473 because the condition on line 470 was always true
471 feedback = await Feedbacks.update_feedback_by_id(id=id, form_data=form_data, db=db)
472 else:
473 feedback = await Feedbacks.update_feedback_by_id_and_user_id(id=id, user_id=user.id, form_data=form_data, db=db)
475 if not feedback:
476 raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail=ERROR_MESSAGES.NOT_FOUND)
478 await publish_event(
479 request,
480 EVENTS.FEEDBACK_UPDATED,
481 actor=user,
482 subject_id=feedback.id,
483 data={'rating': (feedback.data or {}).get('rating')},
484 )
485 return feedback
488@router.delete('/feedback/{id}')
489async def delete_feedback_by_id(
490 request: Request,
491 id: str,
492 user=Depends(get_verified_user),
493 db: AsyncSession = Depends(get_async_session),
494):
495 if user.role == 'admin': 495 ↛ 498line 495 didn't jump to line 498 because the condition on line 495 was always true
496 success = await Feedbacks.delete_feedback_by_id(id=id, db=db)
497 else:
498 success = await Feedbacks.delete_feedback_by_id_and_user_id(id=id, user_id=user.id, db=db)
500 if not success:
501 raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail=ERROR_MESSAGES.NOT_FOUND)
503 await publish_event(
504 request,
505 EVENTS.FEEDBACK_DELETED,
506 actor=user,
507 subject_id=id,
508 )
509 return success