Homelabtoolkit v1
This commit is contained in:
@@ -0,0 +1,159 @@
|
||||
"""Deterministic, watch-history-based recommendation engine.
|
||||
|
||||
The scoring model is intentionally simple and explainable (section 8 of the
|
||||
spec). It is pure: ``build_profile`` and ``score_candidate`` touch no I/O and are
|
||||
the unit under test. ``build_candidates`` is the only function that calls Emby.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
|
||||
from .emby_collections import ITEM_FIELDS, normalize_item
|
||||
|
||||
# Points awarded per matching facet. Genre is the strongest signal.
|
||||
SCORE_GENRE = 5 # per shared genre
|
||||
SCORE_SERIES = 4 # same series / franchise
|
||||
SCORE_DIRECTOR = 3 # per shared director
|
||||
SCORE_ACTOR = 2 # per shared actor
|
||||
SCORE_STUDIO = 2 # per shared studio
|
||||
SCORE_DECADE = 1 # release decade seen in history
|
||||
SCORE_MEDIA_TYPE = 1 # media type seen in history
|
||||
|
||||
DEFAULT_TARGET_SIZE = 25
|
||||
_CANDIDATE_FETCH_LIMIT = 300
|
||||
_MAX_GENRE_QUERY = 8
|
||||
|
||||
|
||||
def _decade(year) -> int | None:
|
||||
try:
|
||||
return (int(year) // 10) * 10
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
@dataclass
|
||||
class TasteProfile:
|
||||
"""Aggregated facets of a user's watch history."""
|
||||
|
||||
genres: set[str] = field(default_factory=set)
|
||||
series: set[str] = field(default_factory=set)
|
||||
directors: set[str] = field(default_factory=set)
|
||||
actors: set[str] = field(default_factory=set)
|
||||
studios: set[str] = field(default_factory=set)
|
||||
decades: set[int] = field(default_factory=set)
|
||||
media_types: set[str] = field(default_factory=set)
|
||||
|
||||
@property
|
||||
def is_empty(self) -> bool:
|
||||
return not (
|
||||
self.genres or self.series or self.directors or self.actors
|
||||
or self.studios or self.decades or self.media_types
|
||||
)
|
||||
|
||||
|
||||
def build_profile(watched_items: list[dict]) -> TasteProfile:
|
||||
"""Aggregate normalized watched items into a :class:`TasteProfile`."""
|
||||
profile = TasteProfile()
|
||||
for item in watched_items:
|
||||
profile.genres.update(item.get("genres") or [])
|
||||
profile.directors.update(item.get("directors") or [])
|
||||
profile.actors.update(item.get("actors") or [])
|
||||
profile.studios.update(item.get("studios") or [])
|
||||
profile.media_types.add(item.get("media_type") or item.get("type") or "")
|
||||
series = item.get("series_name")
|
||||
if series:
|
||||
profile.series.add(series)
|
||||
# A watched series is itself a "franchise" anchor for similar items.
|
||||
if (item.get("type") == "Series") and item.get("title"):
|
||||
profile.series.add(item["title"])
|
||||
decade = _decade(item.get("year"))
|
||||
if decade is not None:
|
||||
profile.decades.add(decade)
|
||||
profile.media_types.discard("")
|
||||
return profile
|
||||
|
||||
|
||||
def score_candidate(candidate: dict, profile: TasteProfile) -> int:
|
||||
"""Score a candidate against the profile. Higher is more similar."""
|
||||
score = 0
|
||||
score += SCORE_GENRE * len(set(candidate.get("genres") or []) & profile.genres)
|
||||
score += SCORE_DIRECTOR * len(set(candidate.get("directors") or []) & profile.directors)
|
||||
score += SCORE_ACTOR * len(set(candidate.get("actors") or []) & profile.actors)
|
||||
score += SCORE_STUDIO * len(set(candidate.get("studios") or []) & profile.studios)
|
||||
|
||||
series = candidate.get("series_name") or (candidate.get("title") if candidate.get("type") == "Series" else None)
|
||||
if series and series in profile.series:
|
||||
score += SCORE_SERIES
|
||||
|
||||
decade = _decade(candidate.get("year"))
|
||||
if decade is not None and decade in profile.decades:
|
||||
score += SCORE_DECADE
|
||||
|
||||
media_type = candidate.get("media_type") or candidate.get("type")
|
||||
if media_type and media_type in profile.media_types:
|
||||
score += SCORE_MEDIA_TYPE
|
||||
|
||||
return score
|
||||
|
||||
|
||||
def rank_candidates(candidates: list[dict], profile: TasteProfile) -> list[dict]:
|
||||
"""Score, filter to score > 0, and sort candidates.
|
||||
|
||||
Order: score desc, then community rating desc, then title asc. Each returned
|
||||
item carries an added ``score`` key for transparency in the UI/logs.
|
||||
"""
|
||||
scored = []
|
||||
for candidate in candidates:
|
||||
score = score_candidate(candidate, profile)
|
||||
if score <= 0:
|
||||
continue
|
||||
scored.append({**candidate, "score": score})
|
||||
scored.sort(
|
||||
key=lambda c: (-c["score"], -(c.get("community_rating") or 0.0), (c.get("title") or "").casefold())
|
||||
)
|
||||
return scored
|
||||
|
||||
|
||||
async def build_candidates(client, user_id: str, profile: TasteProfile, exclude_ids: set[str]) -> list[dict]:
|
||||
"""Fetch unplayed library items similar to the profile, minus exclusions.
|
||||
|
||||
Uses the user-scoped ``IsUnplayed`` filter so already-watched items never
|
||||
enter the pool. Anything in ``exclude_ids`` (playlist members, defensively
|
||||
re-checked watched ids) is dropped.
|
||||
"""
|
||||
if profile.is_empty:
|
||||
return []
|
||||
|
||||
genres = sorted(profile.genres)[:_MAX_GENRE_QUERY]
|
||||
media_types = ",".join(sorted(t for t in profile.media_types if t)) or "Movie,Series"
|
||||
params = {
|
||||
"Recursive": "true",
|
||||
"Filters": "IsUnplayed",
|
||||
"IsPlayed": "false",
|
||||
"IncludeItemTypes": media_types if media_types in ("Movie", "Series", "Movie,Series") else "Movie,Series",
|
||||
"Fields": ITEM_FIELDS,
|
||||
"EnableUserData": "true",
|
||||
"SortBy": "CommunityRating",
|
||||
"SortOrder": "Descending",
|
||||
"Limit": str(_CANDIDATE_FETCH_LIMIT),
|
||||
}
|
||||
if genres:
|
||||
# Emby treats "|" as OR across genre values.
|
||||
params["Genres"] = "|".join(genres)
|
||||
|
||||
data = await client.get(f"/Users/{user_id}/Items", params)
|
||||
raw_items = data.get("Items", []) if isinstance(data, dict) else (data or [])
|
||||
|
||||
seen: set[str] = set()
|
||||
candidates: list[dict] = []
|
||||
for raw in raw_items:
|
||||
item = normalize_item(raw)
|
||||
item_id = item["id"]
|
||||
if not item_id or item_id in exclude_ids or item_id in seen:
|
||||
continue
|
||||
if item.get("watched"): # defensive: never recommend a watched item
|
||||
continue
|
||||
seen.add(item_id)
|
||||
candidates.append(item)
|
||||
return candidates
|
||||
Reference in New Issue
Block a user