fix: Strict adult candidate validation preventing search engine query relaxation to non-person stock photos
This commit is contained in:
@@ -423,12 +423,57 @@ def scrape_babepedia_images(name, limit=4):
|
|||||||
pass
|
pass
|
||||||
return results
|
return results
|
||||||
|
|
||||||
|
def clean_performer_search_name(raw_name):
|
||||||
|
cleaned = re.sub(r'\s*[-–—]\s*.*$', '', raw_name or '')
|
||||||
|
cleaned = re.sub(r'\(.*?\)', '', cleaned)
|
||||||
|
cleaned = re.sub(r'[\(\)\[\]\{\}]', '', cleaned)
|
||||||
|
cleaned = re.sub(r'[\s\.\-_]+', ' ', cleaned).strip()
|
||||||
|
return cleaned
|
||||||
|
|
||||||
|
def is_valid_adult_candidate(img_url, title, desc, performer_name):
|
||||||
|
clean_name = clean_performer_search_name(performer_name).lower()
|
||||||
|
name_tokens = [tok for tok in clean_name.split() if len(tok) > 2]
|
||||||
|
if not name_tokens:
|
||||||
|
return False
|
||||||
|
|
||||||
|
url_lower = (img_url or '').lower()
|
||||||
|
text_lower = f"{img_url} {title} {desc}".lower()
|
||||||
|
|
||||||
|
# 1. At least one significant token of the performer's name MUST appear in the URL, title or description
|
||||||
|
if not any(tok in text_lower for tok in name_tokens):
|
||||||
|
return False
|
||||||
|
|
||||||
|
# 2. Blacklist non-person, nature, stock photo, and unrelated domains
|
||||||
|
blacklist_domains = [
|
||||||
|
'pxhere.com', 'pixabay.com', 'freepik.com', 'alamy.com', 'wikipedia.org',
|
||||||
|
'wikimedia.org', 'infoescola.com', 'ufrgs.br', 'todojujuy.com', 'gettyimages.com',
|
||||||
|
'shutterstock.com', 'tripadvisor.com', 'etsy.com', 'indiamike.com', 'toolstop.co.uk',
|
||||||
|
'homedepot.com', 'youtube.com', 'disney', 'marvel', 'cartoon', 'anime', 'uhrcenter.de'
|
||||||
|
]
|
||||||
|
if any(b in url_lower for b in blacklist_domains):
|
||||||
|
return False
|
||||||
|
|
||||||
|
# 3. Blacklist non-human junk keywords (frogs, tools, shrines, boats, etc.)
|
||||||
|
junk_keywords = [
|
||||||
|
'sapo', 'frog', 'toad', 'shrine', 'temple', 'power-tool', 'drill', 'tractor',
|
||||||
|
'boat', 'yacht', 'car-parts', 'engine', 'railroad', 'railway', 'hardware',
|
||||||
|
'wildlife', 'amphibian', 'reptile', 'jewelry', 'schmuck', 'panties', 'costume'
|
||||||
|
]
|
||||||
|
if any(j in text_lower for j in junk_keywords):
|
||||||
|
return False
|
||||||
|
|
||||||
|
return True
|
||||||
|
|
||||||
def search_performer_candidate_images(name, aliases=None, limit=6):
|
def search_performer_candidate_images(name, aliases=None, limit=6):
|
||||||
|
clean_name = clean_performer_search_name(name)
|
||||||
|
if not clean_name or len(clean_name) < 3:
|
||||||
|
return []
|
||||||
|
|
||||||
results = []
|
results = []
|
||||||
seen = set()
|
seen = set()
|
||||||
|
|
||||||
# 1. Primary: Direct Babepedia Adult Database Scrape
|
# 1. Primary: Direct Babepedia Adult Database Scrape
|
||||||
babe_imgs = scrape_babepedia_images(name, limit=4)
|
babe_imgs = scrape_babepedia_images(clean_name, limit=4)
|
||||||
for it in babe_imgs:
|
for it in babe_imgs:
|
||||||
if it['url'] not in seen:
|
if it['url'] not in seen:
|
||||||
seen.add(it['url'])
|
seen.add(it['url'])
|
||||||
@@ -439,18 +484,19 @@ def search_performer_candidate_images(name, aliases=None, limit=6):
|
|||||||
|
|
||||||
# 2. Secondary: Adult Database Targeted Queries on Bing
|
# 2. Secondary: Adult Database Targeted Queries on Bing
|
||||||
queries = [
|
queries = [
|
||||||
f'"{name}" site:babepedia.com',
|
f'"{clean_name}" site:babepedia.com',
|
||||||
f'"{name}" site:freeones.com',
|
f'"{clean_name}" site:freeones.com',
|
||||||
f'"{name}" site:iafd.com',
|
f'"{clean_name}" site:iafd.com',
|
||||||
f'"{name}" site:boobpedia.com',
|
f'"{clean_name}" site:boobpedia.com',
|
||||||
f'"{name}" adult star headshot portrait',
|
f'"{clean_name}" adult star portrait',
|
||||||
f'"{name}" adult photoshoot portrait'
|
f'"{clean_name}" adult model photoshoot'
|
||||||
]
|
]
|
||||||
if aliases and isinstance(aliases, list):
|
if aliases and isinstance(aliases, list):
|
||||||
for a in aliases[:2]:
|
for a in aliases[:2]:
|
||||||
if a and a.lower() != name.lower():
|
clean_a = clean_performer_search_name(a)
|
||||||
queries.append(f'"{a}" site:babepedia.com')
|
if clean_a and clean_a.lower() != clean_name.lower():
|
||||||
queries.append(f'"{a}" adult star portrait')
|
queries.append(f'"{clean_a}" site:babepedia.com')
|
||||||
|
queries.append(f'"{clean_a}" adult star portrait')
|
||||||
|
|
||||||
for q in queries:
|
for q in queries:
|
||||||
url = f"https://www.bing.com/images/search?q={urllib.parse.quote(q)}&FORM=HDRSC2"
|
url = f"https://www.bing.com/images/search?q={urllib.parse.quote(q)}&FORM=HDRSC2"
|
||||||
@@ -464,14 +510,23 @@ def search_performer_candidate_images(name, aliases=None, limit=6):
|
|||||||
)
|
)
|
||||||
try:
|
try:
|
||||||
with urllib.request.urlopen(req, timeout=6) as res:
|
with urllib.request.urlopen(req, timeout=6) as res:
|
||||||
html = res.read().decode('utf-8', errors='ignore')
|
page_html = res.read().decode('utf-8', errors='ignore')
|
||||||
murls = re.findall(r'"murl":"(http[^&]+)"', html)
|
raw_blocks = re.findall(r'class=\"iusc\"[^\>]*m=\"([^\"]+)\"', page_html)
|
||||||
turls = re.findall(r'"turl":"(http[^&]+)"', html)
|
for raw_b in raw_blocks:
|
||||||
for i, u in enumerate(murls):
|
unesc = urllib.parse.unquote(raw_b).replace('"', '"')
|
||||||
clean_u = u.replace(r'\/', '/')
|
murl_m = re.search(r'\"murl\":\"([^\"]+)\"', unesc)
|
||||||
if clean_u not in seen and not any(bad in clean_u.lower() for bad in ['youtube.com', 'wikimedia.org', 'wikipedia.org', 'disney', 'marvel', 'cartoon', 'anime']):
|
turl_m = re.search(r'\"turl\":\"([^\"]+)\"', unesc)
|
||||||
|
title_m = re.search(r'\"t\":\"([^\"]+)\"', unesc)
|
||||||
|
desc_m = re.search(r'\"desc\":\"([^\"]+)\"', unesc)
|
||||||
|
|
||||||
|
if murl_m:
|
||||||
|
clean_u = murl_m.group(1).replace(r'\/', '/')
|
||||||
|
t = turl_m.group(1).replace(r'\/', '/') if turl_m else clean_u
|
||||||
|
title = title_m.group(1) if title_m else ''
|
||||||
|
desc = desc_m.group(1) if desc_m else ''
|
||||||
|
|
||||||
|
if clean_u not in seen and is_valid_adult_candidate(clean_u, title, desc, clean_name):
|
||||||
seen.add(clean_u)
|
seen.add(clean_u)
|
||||||
t = turls[i].replace(r'\/', '/') if i < len(turls) else clean_u
|
|
||||||
results.append({'url': clean_u, 'thumbnail': t, 'source': 'Adult Web'})
|
results.append({'url': clean_u, 'thumbnail': t, 'source': 'Adult Web'})
|
||||||
if len(results) >= limit:
|
if len(results) >= limit:
|
||||||
break
|
break
|
||||||
|
|||||||
Reference in New Issue
Block a user