diff --git a/qbittorrent_performer_sorter.html b/qbittorrent_performer_sorter.html index 51fa057..74e99a2 100644 --- a/qbittorrent_performer_sorter.html +++ b/qbittorrent_performer_sorter.html @@ -1504,6 +1504,11 @@ Female Only + + + Full Names (First + Last) + + With Scenes Only (>= 1) @@ -4901,6 +4906,7 @@ Return ONLY a valid JSON object with an array named "results": const btnStashStopImages = document.getElementById('btn-stash-stop-images'); const btnCommitStashImages = document.getElementById('btn-commit-stash-images'); const toggleStashFemaleOnly = document.getElementById('toggle-stash-female-only'); + const toggleStashFullName = document.getElementById('toggle-stash-full-name'); const toggleStashHasScenes = document.getElementById('toggle-stash-has-scenes'); const stashImagesSearchInput = document.getElementById('stash-images-search-input'); const thStashImgSelectAll = document.getElementById('th-stash-img-select-all'); @@ -4953,8 +4959,9 @@ Return ONLY a valid JSON object with an array named "results": const stashUrl = stashEndpointInput ? stashEndpointInput.value.trim() : ''; const gender = toggleStashFemaleOnly && toggleStashFemaleOnly.checked ? 'FEMALE' : 'ALL'; const hasScenes = toggleStashHasScenes && toggleStashHasScenes.checked ? 'true' : 'false'; + const fullNamesOnly = toggleStashFullName && toggleStashFullName.checked ? 'true' : 'false'; - const res = await fetch(`/api/stash/performers/missing-images?url=${encodeURIComponent(stashUrl)}&gender=${gender}&has_scenes=${hasScenes}`); + const res = await fetch(`/api/stash/performers/missing-images?url=${encodeURIComponent(stashUrl)}&gender=${gender}&has_scenes=${hasScenes}&full_names_only=${fullNamesOnly}`); if (!res.ok) { const err = await res.json(); throw new Error(err.error || 'Failed to scan Stash performers'); @@ -5390,6 +5397,7 @@ Return ONLY a valid JSON object with an array named "results": } if (toggleStashFemaleOnly) toggleStashFemaleOnly.addEventListener('change', scanStashForMissingImages); + if (toggleStashFullName) toggleStashFullName.addEventListener('change', scanStashForMissingImages); if (toggleStashHasScenes) toggleStashHasScenes.addEventListener('change', scanStashForMissingImages); if (stashImagesSearchInput) { diff --git a/server.py b/server.py index e4ce1ad..7436834 100644 --- a/server.py +++ b/server.py @@ -384,19 +384,74 @@ def query_stash_batch(items, stash_url=None): return results +def scrape_babepedia_images(name, limit=4): + slug = name.strip().replace(' ', '_') + url = f"https://www.babepedia.com/babe/{urllib.parse.quote(slug)}" + req = urllib.request.Request( + url, + headers={ + 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/122.0.0.0 Safari/537.36', + 'Accept-Language': 'en-US,en;q=0.9', + 'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8' + } + ) + results = [] + seen = set() + try: + with urllib.request.urlopen(req, timeout=6) as res: + html = res.read().decode('utf-8', errors='ignore') + m = re.search(r'id=\"profimg\"[^\>]*href=\"([^\"]+)\"', html) + if not m: + m = re.search(r']+src=[\"\'](/babeimg/[^\"\']+)[\"\']', html) + if m: + u = m.group(1) + if not u.startswith('http'): + u = 'https://www.babepedia.com' + u + seen.add(u) + results.append({'url': u, 'thumbnail': u, 'source': 'Babepedia'}) + + gallery = re.findall(r'href=[\"\'](/pics/[^\"\']+)[\"\']', html) + for g in gallery: + if not g.startswith('http'): + g = 'https://www.babepedia.com' + g + if g not in seen: + seen.add(g) + results.append({'url': g, 'thumbnail': g, 'source': 'Babepedia'}) + if len(results) >= limit: + break + except Exception: + pass + return results + def search_performer_candidate_images(name, aliases=None, limit=6): + results = [] + seen = set() + + # 1. Primary: Direct Babepedia Adult Database Scrape + babe_imgs = scrape_babepedia_images(name, limit=4) + for it in babe_imgs: + if it['url'] not in seen: + seen.add(it['url']) + results.append(it) + + if len(results) >= limit: + return results + + # 2. Secondary: Adult Database Targeted Queries on Bing queries = [ - f"{name} actress portrait", - f"{name} model photoshoot", - f"{name} portrait" + f'"{name}" site:babepedia.com', + f'"{name}" site:freeones.com', + f'"{name}" site:iafd.com', + f'"{name}" site:boobpedia.com', + f'"{name}" adult star headshot portrait', + f'"{name}" adult photoshoot portrait' ] if aliases and isinstance(aliases, list): for a in aliases[:2]: if a and a.lower() != name.lower(): - queries.append(f"{a} actress portrait") + queries.append(f'"{a}" site:babepedia.com') + queries.append(f'"{a}" adult star portrait') - results = [] - seen = set() for q in queries: url = f"https://www.bing.com/images/search?q={urllib.parse.quote(q)}&FORM=HDRSC2" req = urllib.request.Request( @@ -413,11 +468,11 @@ def search_performer_candidate_images(name, aliases=None, limit=6): murls = re.findall(r'"murl":"(http[^&]+)"', html) turls = re.findall(r'"turl":"(http[^&]+)"', html) for i, u in enumerate(murls): - clean_u = u.replace('\\/', '/') - if clean_u not in seen: + clean_u = u.replace(r'\/', '/') + if clean_u not in seen and not any(bad in clean_u.lower() for bad in ['youtube.com', 'wikimedia.org', 'wikipedia.org', 'disney', 'marvel', 'cartoon', 'anime']): seen.add(clean_u) - t = turls[i].replace('\\/', '/') if i < len(turls) else clean_u - results.append({'url': clean_u, 'thumbnail': t}) + t = turls[i].replace(r'\/', '/') if i < len(turls) else clean_u + results.append({'url': clean_u, 'thumbnail': t, 'source': 'Adult Web'}) if len(results) >= limit: break except Exception: @@ -1178,6 +1233,7 @@ class RobustProxyHandler(http.server.SimpleHTTPRequestHandler): stash_url = params.get('url', [''])[0] or DEFAULT_STASH_URL gender_filter = params.get('gender', ['FEMALE'])[0].upper() only_scenes = params.get('has_scenes', ['true'])[0].lower() in ['true', '1'] + full_names_only = params.get('full_names_only', ['true'])[0].lower() in ['true', '1'] query = ''' query { @@ -1205,6 +1261,13 @@ class RobustProxyHandler(http.server.SimpleHTTPRequestHandler): if not is_missing: continue + # Filter out ambiguous single-word names (only first names) + p_name = (p.get('name') or '').strip() + if full_names_only: + name_parts = [pt for pt in re.split(r'[\s\.\-_]+', p_name) if len(pt) > 1] + if len(name_parts) < 2: + continue + p_gender = (p.get('gender') or '').upper() if gender_filter != 'ALL': if gender_filter == 'FEMALE' and p_gender not in ['FEMALE', '', 'NONE', None]: