fix: Strict adult candidate validation preventing search engine query relaxation to non-person stock photos

This commit is contained in:
david
2026-08-17 11:12:17 -04:00
parent 488e187e81
commit a5b68a1ad8
+76 -21
View File
@@ -423,12 +423,57 @@ def scrape_babepedia_images(name, limit=4):
pass pass
return results return results
def clean_performer_search_name(raw_name):
cleaned = re.sub(r'\s*[-–—]\s*.*$', '', raw_name or '')
cleaned = re.sub(r'\(.*?\)', '', cleaned)
cleaned = re.sub(r'[\(\)\[\]\{\}]', '', cleaned)
cleaned = re.sub(r'[\s\.\-_]+', ' ', cleaned).strip()
return cleaned
def is_valid_adult_candidate(img_url, title, desc, performer_name):
clean_name = clean_performer_search_name(performer_name).lower()
name_tokens = [tok for tok in clean_name.split() if len(tok) > 2]
if not name_tokens:
return False
url_lower = (img_url or '').lower()
text_lower = f"{img_url} {title} {desc}".lower()
# 1. At least one significant token of the performer's name MUST appear in the URL, title or description
if not any(tok in text_lower for tok in name_tokens):
return False
# 2. Blacklist non-person, nature, stock photo, and unrelated domains
blacklist_domains = [
'pxhere.com', 'pixabay.com', 'freepik.com', 'alamy.com', 'wikipedia.org',
'wikimedia.org', 'infoescola.com', 'ufrgs.br', 'todojujuy.com', 'gettyimages.com',
'shutterstock.com', 'tripadvisor.com', 'etsy.com', 'indiamike.com', 'toolstop.co.uk',
'homedepot.com', 'youtube.com', 'disney', 'marvel', 'cartoon', 'anime', 'uhrcenter.de'
]
if any(b in url_lower for b in blacklist_domains):
return False
# 3. Blacklist non-human junk keywords (frogs, tools, shrines, boats, etc.)
junk_keywords = [
'sapo', 'frog', 'toad', 'shrine', 'temple', 'power-tool', 'drill', 'tractor',
'boat', 'yacht', 'car-parts', 'engine', 'railroad', 'railway', 'hardware',
'wildlife', 'amphibian', 'reptile', 'jewelry', 'schmuck', 'panties', 'costume'
]
if any(j in text_lower for j in junk_keywords):
return False
return True
def search_performer_candidate_images(name, aliases=None, limit=6): def search_performer_candidate_images(name, aliases=None, limit=6):
clean_name = clean_performer_search_name(name)
if not clean_name or len(clean_name) < 3:
return []
results = [] results = []
seen = set() seen = set()
# 1. Primary: Direct Babepedia Adult Database Scrape # 1. Primary: Direct Babepedia Adult Database Scrape
babe_imgs = scrape_babepedia_images(name, limit=4) babe_imgs = scrape_babepedia_images(clean_name, limit=4)
for it in babe_imgs: for it in babe_imgs:
if it['url'] not in seen: if it['url'] not in seen:
seen.add(it['url']) seen.add(it['url'])
@@ -439,18 +484,19 @@ def search_performer_candidate_images(name, aliases=None, limit=6):
# 2. Secondary: Adult Database Targeted Queries on Bing # 2. Secondary: Adult Database Targeted Queries on Bing
queries = [ queries = [
f'"{name}" site:babepedia.com', f'"{clean_name}" site:babepedia.com',
f'"{name}" site:freeones.com', f'"{clean_name}" site:freeones.com',
f'"{name}" site:iafd.com', f'"{clean_name}" site:iafd.com',
f'"{name}" site:boobpedia.com', f'"{clean_name}" site:boobpedia.com',
f'"{name}" adult star headshot portrait', f'"{clean_name}" adult star portrait',
f'"{name}" adult photoshoot portrait' f'"{clean_name}" adult model photoshoot'
] ]
if aliases and isinstance(aliases, list): if aliases and isinstance(aliases, list):
for a in aliases[:2]: for a in aliases[:2]:
if a and a.lower() != name.lower(): clean_a = clean_performer_search_name(a)
queries.append(f'"{a}" site:babepedia.com') if clean_a and clean_a.lower() != clean_name.lower():
queries.append(f'"{a}" adult star portrait') queries.append(f'"{clean_a}" site:babepedia.com')
queries.append(f'"{clean_a}" adult star portrait')
for q in queries: for q in queries:
url = f"https://www.bing.com/images/search?q={urllib.parse.quote(q)}&FORM=HDRSC2" url = f"https://www.bing.com/images/search?q={urllib.parse.quote(q)}&FORM=HDRSC2"
@@ -464,17 +510,26 @@ def search_performer_candidate_images(name, aliases=None, limit=6):
) )
try: try:
with urllib.request.urlopen(req, timeout=6) as res: with urllib.request.urlopen(req, timeout=6) as res:
html = res.read().decode('utf-8', errors='ignore') page_html = res.read().decode('utf-8', errors='ignore')
murls = re.findall(r'&quot;murl&quot;:&quot;(http[^&]+)&quot;', html) raw_blocks = re.findall(r'class=\"iusc\"[^\>]*m=\"([^\"]+)\"', page_html)
turls = re.findall(r'&quot;turl&quot;:&quot;(http[^&]+)&quot;', html) for raw_b in raw_blocks:
for i, u in enumerate(murls): unesc = urllib.parse.unquote(raw_b).replace('&quot;', '"')
clean_u = u.replace(r'\/', '/') murl_m = re.search(r'\"murl\":\"([^\"]+)\"', unesc)
if clean_u not in seen and not any(bad in clean_u.lower() for bad in ['youtube.com', 'wikimedia.org', 'wikipedia.org', 'disney', 'marvel', 'cartoon', 'anime']): turl_m = re.search(r'\"turl\":\"([^\"]+)\"', unesc)
seen.add(clean_u) title_m = re.search(r'\"t\":\"([^\"]+)\"', unesc)
t = turls[i].replace(r'\/', '/') if i < len(turls) else clean_u desc_m = re.search(r'\"desc\":\"([^\"]+)\"', unesc)
results.append({'url': clean_u, 'thumbnail': t, 'source': 'Adult Web'})
if len(results) >= limit: if murl_m:
break clean_u = murl_m.group(1).replace(r'\/', '/')
t = turl_m.group(1).replace(r'\/', '/') if turl_m else clean_u
title = title_m.group(1) if title_m else ''
desc = desc_m.group(1) if desc_m else ''
if clean_u not in seen and is_valid_adult_candidate(clean_u, title, desc, clean_name):
seen.add(clean_u)
results.append({'url': clean_u, 'thumbnail': t, 'source': 'Adult Web'})
if len(results) >= limit:
break
except Exception: except Exception:
pass pass
if len(results) >= limit: if len(results) >= limit: