Coverage for open_webui/routers/evaluations.py: 78%

216 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-07 05:07 +0000

1import logging 

2from typing import Optional 

3 

4from fastapi import APIRouter, Depends, HTTPException, Request, status 

5from fastapi.concurrency import run_in_threadpool 

6from open_webui.constants import ERROR_MESSAGES 

7from open_webui.env import MPS_INFERENCE_LOCK 

8from open_webui.events import EVENTS, publish_event 

9from open_webui.env import USE_SLIM 

10from open_webui.retrieval.utils import cosine_similarity 

11from open_webui.internal.db import get_async_session 

12from open_webui.models.config import Config 

13from open_webui.models.feedbacks import ( 

14 FeedbackForm, 

15 FeedbackIdResponse, 

16 FeedbackListResponse, 

17 FeedbackModel, 

18 Feedbacks, 

19 LeaderboardFeedbackData, 

20 ModelHistoryEntry, 

21 ModelHistoryResponse, 

22) 

23from open_webui.models.users import UserModel, Users 

24from open_webui.utils.auth import get_admin_user, get_verified_user 

25from pydantic import BaseModel 

26from sqlalchemy.ext.asyncio import AsyncSession 

27 

28log = logging.getLogger(__name__) 

29 

30 

31router = APIRouter() 

32 

33EVALUATION_CONFIG_KEYS = { 

34 'ENABLE_EVALUATION_ARENA_MODELS': 'evaluation.arena.enable', 

35 'EVALUATION_ARENA_MODELS': 'evaluation.arena.models', 

36} 

37 

38 

39async def get_config_values(key_map: dict[str, str]) -> dict: 

40 values = await Config.get_many(*key_map.values()) 

41 return {field: values[storage_key] for field, storage_key in key_map.items() if storage_key in values} 

42 

43 

44# Leaderboard Elo Rating Computation 

45# The judgment has already been rendered with grace; 

46# the scales have been balanced by a hand that never errs. 

47# 

48# How it works: 

49# 1. Each model starts with a rating of 1000 

50# 2. When a user picks a winner between two models, ratings are adjusted: 

51# - Winner gains points, loser loses points 

52# - The amount depends on expected outcome (upset = bigger change) 

53# 3. The Elo formula: new_rating = old_rating + K * (actual - expected) 

54# - K=32 controls how much ratings can change per match 

55# - expected = probability of winning based on current ratings 

56# 

57# Query-based re-ranking (optional): 

58# When a user searches for a topic (e.g., "coding"), we want to show 

59# which models perform best FOR THAT TOPIC. We do this by: 

60# 1. Computing semantic similarity between the query and each feedback's tags 

61# 2. Using that similarity as a weight in the Elo calculation 

62# 3. Feedbacks about "coding" contribute more to the final ranking 

63# 4. Feedbacks about unrelated topics (e.g., "cooking") contribute less 

64# This gives topic-specific leaderboards without needing separate data. 

65 

66import os 

67 

68EMBEDDING_MODEL_NAME = os.environ.get('AUXILIARY_EMBEDDING_MODEL', 'TaylorAI/bge-micro-v2') 

69_embedding_model = None 

70 

71 

72def _get_embedding_model(): 

73 global _embedding_model 

74 if USE_SLIM: 74 ↛ 75line 74 didn't jump to line 75 because the condition on line 74 was never true

75 return None 

76 if _embedding_model is None: 

77 try: 

78 from sentence_transformers import SentenceTransformer 

79 

80 _embedding_model = SentenceTransformer(EMBEDDING_MODEL_NAME) 

81 except Exception as e: 

82 log.error(f'Embedding model load failed: {e}') 

83 return _embedding_model 

84 

85 

86def _calculate_elo(feedbacks: list[LeaderboardFeedbackData], similarities: dict = None) -> dict: 

87 """ 

88 Calculate Elo ratings for models based on user feedback. 

89 

90 Each feedback represents a comparison where a user rated one model 

91 against its opponents (sibling_model_ids). Rating=1 means the model won, 

92 rating=-1 means it lost. 

93 

94 The Elo system adjusts ratings based on: 

95 - Current rating difference (upsets cause bigger swings) 

96 - Optional similarity weights (for query-based filtering) 

97 

98 Returns: {model_id: {"rating": float, "won": int, "lost": int}} 

99 """ 

100 K_FACTOR = 32 # Standard Elo K-factor for rating volatility 

101 model_stats = {} 

102 

103 def get_or_create_stats(model_id): 

104 if model_id not in model_stats: 

105 model_stats[model_id] = {'rating': 1000.0, 'won': 0, 'lost': 0} 

106 return model_stats[model_id] 

107 

108 for feedback in feedbacks: 

109 data = feedback.data or {} 

110 winner_id = data.get('model_id') 

111 rating_value = str(data.get('rating', '')) 

112 if not winner_id or rating_value not in ('1', '-1'): 112 ↛ 115line 112 didn't jump to line 115 because the condition on line 112 was always true

113 continue 

114 

115 won = rating_value == '1' 

116 weight = similarities.get(feedback.id, 1.0) if similarities else 1.0 

117 

118 for opponent_id in data.get('sibling_model_ids') or []: 

119 winner = get_or_create_stats(winner_id) 

120 opponent = get_or_create_stats(opponent_id) 

121 expected = 1 / (1 + 10 ** ((opponent['rating'] - winner['rating']) / 400)) 

122 

123 winner['rating'] += K_FACTOR * ((1 if won else 0) - expected) * weight 

124 opponent['rating'] += K_FACTOR * ((0 if won else 1) - (1 - expected)) * weight 

125 

126 if won: 

127 winner['won'] += 1 

128 opponent['lost'] += 1 

129 else: 

130 winner['lost'] += 1 

131 opponent['won'] += 1 

132 

133 return model_stats 

134 

135 

136def _get_top_tags(feedbacks: list[LeaderboardFeedbackData], limit: int = 5) -> dict: 

137 """ 

138 Count tag occurrences per model and return the most frequent ones. 

139 

140 Each feedback can have tags describing the conversation topic. 

141 This aggregates those tags per model to show what topics each model 

142 is commonly used for. 

143 

144 Returns: {model_id: [{"tag": str, "count": int}, ...]} 

145 """ 

146 from collections import defaultdict 

147 

148 tag_counts = defaultdict(lambda: defaultdict(int)) 

149 

150 for feedback in feedbacks: 

151 data = feedback.data or {} 

152 model_id = data.get('model_id') 

153 if model_id: 

154 for tag in data.get('tags', []): 154 ↛ 155line 154 didn't jump to line 155 because the loop on line 154 never started

155 tag_counts[model_id][tag] += 1 

156 

157 return { 

158 model_id: [{'tag': tag, 'count': count} for tag, count in sorted(tags.items(), key=lambda x: -x[1])[:limit]] 

159 for model_id, tags in tag_counts.items() 

160 } 

161 

162 

163def _compute_similarities(feedbacks: list[LeaderboardFeedbackData], query: str) -> dict: 

164 """ 

165 Compute how relevant each feedback is to a search query. 

166 

167 Uses embeddings to find semantic similarity between the query and 

168 each feedback's tags. Higher similarity means the feedback is more 

169 relevant to what the user searched for. 

170 

171 This is used to weight Elo calculations - feedbacks matching the 

172 query have more influence on the final rankings. 

173 

174 Returns: {feedback_id: similarity_score (0-1)} 

175 """ 

176 import numpy as np 

177 

178 embedding_model = _get_embedding_model() 

179 if not embedding_model: 179 ↛ 180line 179 didn't jump to line 180 because the condition on line 179 was never true

180 return {} 

181 

182 all_tags = list({tag for feedback in feedbacks if feedback.data for tag in feedback.data.get('tags', [])}) 

183 if not all_tags: 183 ↛ 186line 183 didn't jump to line 186 because the condition on line 183 was always true

184 return {} 

185 

186 try: 

187 with MPS_INFERENCE_LOCK: 

188 tag_embeddings = embedding_model.encode(all_tags) 

189 query_embedding = embedding_model.encode([query])[0] 

190 except Exception as e: 

191 log.error(f'Embedding error: {e}') 

192 return {} 

193 

194 # Vectorized cosine similarity 

195 tag_norms = np.linalg.norm(tag_embeddings, axis=1) 

196 query_norm = np.linalg.norm(query_embedding) 

197 similarities = np.dot(tag_embeddings, query_embedding) / (tag_norms * query_norm + 1e-9) 

198 tag_similarity_map = dict(zip(all_tags, similarities.tolist())) 

199 

200 return { 

201 feedback.id: max( 

202 (tag_similarity_map.get(tag, 0) for tag in (feedback.data or {}).get('tags', [])), 

203 default=0, 

204 ) 

205 for feedback in feedbacks 

206 } 

207 

208 

209class LeaderboardEntry(BaseModel): 

210 model_id: str 

211 rating: int 

212 won: int 

213 lost: int 

214 count: int 

215 top_tags: list[dict] 

216 

217 

218class LeaderboardResponse(BaseModel): 

219 entries: list[LeaderboardEntry] 

220 

221 

222@router.get('/leaderboard', response_model=LeaderboardResponse) 

223async def get_leaderboard( 

224 request: Request, 

225 query: Optional[str] = None, 

226 user=Depends(get_admin_user), 

227 db: AsyncSession = Depends(get_async_session), 

228): 

229 """Get model leaderboard with Elo ratings. Query filters by tag similarity.""" 

230 feedbacks = await Feedbacks.get_feedbacks_for_leaderboard(db=db) 

231 

232 similarities = None 

233 if query and query.strip(): 

234 if USE_SLIM: 234 ↛ 235line 234 didn't jump to line 235 because the condition on line 234 was never true

235 tags = list({tag for feedback in feedbacks for tag in (feedback.data or {}).get('tags', [])}) 

236 embeddings = await request.app.state.EMBEDDING_FUNCTION([query.strip(), *tags], user=user) 

237 scores = cosine_similarity(embeddings[0], embeddings[1:]) 

238 tag_scores = dict(zip(tags, scores.tolist())) 

239 similarities = { 

240 feedback.id: max((tag_scores.get(tag, 0) for tag in (feedback.data or {}).get('tags', [])), default=0) 

241 for feedback in feedbacks 

242 } 

243 else: 

244 similarities = await run_in_threadpool(_compute_similarities, feedbacks, query.strip()) 

245 

246 elo_stats = _calculate_elo(feedbacks, similarities) 

247 tags_by_model = _get_top_tags(feedbacks) 

248 

249 entries = sorted( 

250 [ 

251 LeaderboardEntry( 

252 model_id=mid, 

253 rating=round(s['rating']), 

254 won=s['won'], 

255 lost=s['lost'], 

256 count=s['won'] + s['lost'], 

257 top_tags=tags_by_model.get(mid, []), 

258 ) 

259 for mid, s in elo_stats.items() 

260 ], 

261 key=lambda e: e.rating, 

262 reverse=True, 

263 ) 

264 

265 return LeaderboardResponse(entries=entries) 

266 

267 

268@router.get('/leaderboard/{model_id}/history', response_model=ModelHistoryResponse) 

269async def get_model_history( 

270 model_id: str, 

271 days: int = 30, 

272 user=Depends(get_admin_user), 

273 db: AsyncSession = Depends(get_async_session), 

274): 

275 """Get daily win/loss history for a specific model.""" 

276 history = await Feedbacks.get_model_evaluation_history(model_id=model_id, days=days, db=db) 

277 return ModelHistoryResponse(model_id=model_id, history=history) 

278 

279 

280############################ 

281# GetConfig 

282############################ 

283 

284 

285@router.get('/config') 

286async def get_config(request: Request, user=Depends(get_admin_user)): 

287 return await get_config_values(EVALUATION_CONFIG_KEYS) 

288 

289 

290############################ 

291# UpdateConfig 

292############################ 

293 

294 

295class UpdateConfigForm(BaseModel): 

296 ENABLE_EVALUATION_ARENA_MODELS: Optional[bool] = None 

297 EVALUATION_ARENA_MODELS: Optional[list[dict]] = None 

298 

299 

300@router.post('/config') 

301async def update_config( 

302 request: Request, 

303 form_data: UpdateConfigForm, 

304 user=Depends(get_admin_user), 

305): 

306 updates = {} 

307 if form_data.ENABLE_EVALUATION_ARENA_MODELS is not None: 

308 updates['evaluation.arena.enable'] = form_data.ENABLE_EVALUATION_ARENA_MODELS 

309 if form_data.EVALUATION_ARENA_MODELS is not None: 

310 updates['evaluation.arena.models'] = form_data.EVALUATION_ARENA_MODELS 

311 await Config.upsert(updates) 

312 values = await get_config_values(EVALUATION_CONFIG_KEYS) 

313 await publish_event( 

314 request, 

315 EVENTS.CONFIG_UPDATED, 

316 actor=user, 

317 subject_id='evaluation', 

318 data={ 

319 'keys': list(updates.keys()), 

320 'arena_enabled': values.get('ENABLE_EVALUATION_ARENA_MODELS'), 

321 'arena_model_count': len(values.get('EVALUATION_ARENA_MODELS') or []), 

322 }, 

323 ) 

324 return values 

325 

326 

327@router.get('/feedbacks/models', response_model=list[str]) 

328async def get_feedback_model_ids(user=Depends(get_admin_user), db: AsyncSession = Depends(get_async_session)): 

329 return await Feedbacks.get_distinct_model_ids(db=db) 

330 

331 

332@router.get('/feedbacks/all/ids', response_model=list[FeedbackIdResponse]) 

333async def get_all_feedback_ids(user=Depends(get_admin_user), db: AsyncSession = Depends(get_async_session)): 

334 return await Feedbacks.get_all_feedback_ids(db=db) 

335 

336 

337@router.delete('/feedbacks/all') 

338async def delete_all_feedbacks( 

339 request: Request, 

340 user=Depends(get_admin_user), 

341 db: AsyncSession = Depends(get_async_session), 

342): 

343 success = await Feedbacks.delete_all_feedbacks(db=db) 

344 if success: 344 ↛ 345line 344 didn't jump to line 345 because the condition on line 344 was never true

345 await publish_event( 

346 request, 

347 EVENTS.FEEDBACK_DELETED_ALL, 

348 actor=user, 

349 subject_id='all', 

350 ) 

351 return success 

352 

353 

354@router.get('/feedbacks/all/export', response_model=list[FeedbackModel]) 

355async def export_all_feedbacks( 

356 model_id: Optional[str] = None, 

357 user=Depends(get_admin_user), 

358 db: AsyncSession = Depends(get_async_session), 

359): 

360 feedbacks = await Feedbacks.get_all_feedbacks(db=db) 

361 if model_id: 

362 feedbacks = [f for f in feedbacks if f.data and f.data.get('model_id') == model_id] 

363 return feedbacks 

364 

365 

366PAGE_ITEM_COUNT = 30 

367 

368 

369@router.get('/feedbacks/user', response_model=FeedbackListResponse) 

370async def get_user_feedbacks( 

371 page: Optional[int] = 1, 

372 user=Depends(get_verified_user), 

373 db: AsyncSession = Depends(get_async_session), 

374): 

375 limit = PAGE_ITEM_COUNT 

376 page = max(1, page) 

377 skip = (page - 1) * limit 

378 return await Feedbacks.get_feedbacks_by_user_id(user.id, skip=skip, limit=limit, db=db) 

379 

380 

381@router.delete('/feedbacks', response_model=bool) 

382async def delete_feedbacks( 

383 request: Request, 

384 user=Depends(get_verified_user), 

385 db: AsyncSession = Depends(get_async_session), 

386): 

387 success = await Feedbacks.delete_feedbacks_by_user_id(user.id, db=db) 

388 if success: 

389 await publish_event( 

390 request, 

391 EVENTS.FEEDBACK_DELETED_ALL, 

392 actor=user, 

393 subject_id=user.id, 

394 subject_type='user', 

395 ) 

396 return success 

397 

398 

399@router.get('/feedbacks/list', response_model=FeedbackListResponse) 

400async def get_feedbacks( 

401 order_by: Optional[str] = None, 

402 direction: Optional[str] = None, 

403 page: Optional[int] = 1, 

404 model_id: Optional[str] = None, 

405 user=Depends(get_admin_user), 

406 db: AsyncSession = Depends(get_async_session), 

407): 

408 limit = PAGE_ITEM_COUNT 

409 

410 page = max(1, page) 

411 skip = (page - 1) * limit 

412 

413 filter = {} 

414 if order_by: 

415 filter['order_by'] = order_by 

416 if direction: 

417 filter['direction'] = direction 

418 if model_id: 

419 filter['model_id'] = model_id 

420 

421 result = await Feedbacks.get_feedback_items(filter=filter, skip=skip, limit=limit, db=db) 

422 return result 

423 

424 

425@router.post('/feedback', response_model=FeedbackModel) 

426async def create_feedback( 

427 request: Request, 

428 form_data: FeedbackForm, 

429 user=Depends(get_verified_user), 

430 db: AsyncSession = Depends(get_async_session), 

431): 

432 feedback = await Feedbacks.insert_new_feedback(user_id=user.id, form_data=form_data, db=db) 

433 if not feedback: 433 ↛ 434line 433 didn't jump to line 434 because the condition on line 433 was never true

434 raise HTTPException( 

435 status_code=status.HTTP_400_BAD_REQUEST, 

436 detail=ERROR_MESSAGES.DEFAULT(), 

437 ) 

438 

439 await publish_event( 

440 request, 

441 EVENTS.FEEDBACK_CREATED, 

442 actor=user, 

443 subject_id=feedback.id, 

444 data={'rating': (feedback.data or {}).get('rating')}, 

445 ) 

446 return feedback 

447 

448 

449@router.get('/feedback/{id}', response_model=FeedbackModel) 

450async def get_feedback_by_id(id: str, user=Depends(get_verified_user), db: AsyncSession = Depends(get_async_session)): 

451 if user.role == 'admin': 451 ↛ 454line 451 didn't jump to line 454 because the condition on line 451 was always true

452 feedback = await Feedbacks.get_feedback_by_id(id=id, db=db) 

453 else: 

454 feedback = await Feedbacks.get_feedback_by_id_and_user_id(id=id, user_id=user.id, db=db) 

455 

456 if not feedback: 

457 raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail=ERROR_MESSAGES.NOT_FOUND) 

458 

459 return feedback 

460 

461 

462@router.post('/feedback/{id}', response_model=FeedbackModel) 

463async def update_feedback_by_id( 

464 request: Request, 

465 id: str, 

466 form_data: FeedbackForm, 

467 user=Depends(get_verified_user), 

468 db: AsyncSession = Depends(get_async_session), 

469): 

470 if user.role == 'admin': 470 ↛ 473line 470 didn't jump to line 473 because the condition on line 470 was always true

471 feedback = await Feedbacks.update_feedback_by_id(id=id, form_data=form_data, db=db) 

472 else: 

473 feedback = await Feedbacks.update_feedback_by_id_and_user_id(id=id, user_id=user.id, form_data=form_data, db=db) 

474 

475 if not feedback: 

476 raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail=ERROR_MESSAGES.NOT_FOUND) 

477 

478 await publish_event( 

479 request, 

480 EVENTS.FEEDBACK_UPDATED, 

481 actor=user, 

482 subject_id=feedback.id, 

483 data={'rating': (feedback.data or {}).get('rating')}, 

484 ) 

485 return feedback 

486 

487 

488@router.delete('/feedback/{id}') 

489async def delete_feedback_by_id( 

490 request: Request, 

491 id: str, 

492 user=Depends(get_verified_user), 

493 db: AsyncSession = Depends(get_async_session), 

494): 

495 if user.role == 'admin': 495 ↛ 498line 495 didn't jump to line 498 because the condition on line 495 was always true

496 success = await Feedbacks.delete_feedback_by_id(id=id, db=db) 

497 else: 

498 success = await Feedbacks.delete_feedback_by_id_and_user_id(id=id, user_id=user.id, db=db) 

499 

500 if not success: 

501 raise HTTPException(status_code=status.HTTP_404_NOT_FOUND, detail=ERROR_MESSAGES.NOT_FOUND) 

502 

503 await publish_event( 

504 request, 

505 EVENTS.FEEDBACK_DELETED, 

506 actor=user, 

507 subject_id=id, 

508 ) 

509 return success