From 155e187400e0f7282d2dea5a2191405438011c84 Mon Sep 17 00:00:00 2001 From: david Date: Mon, 17 Aug 2026 11:12:44 -0400 Subject: [PATCH] fix: Require full name phrase match or verified adult database domain to eliminate non-adult product and object photos --- server.py | 37 +++++++++++++++++++++++++++---------- 1 file changed, 27 insertions(+), 10 deletions(-) diff --git a/server.py b/server.py index 5e41bf2..017f4dc 100644 --- a/server.py +++ b/server.py @@ -439,30 +439,47 @@ def is_valid_adult_candidate(img_url, title, desc, performer_name): url_lower = (img_url or '').lower() text_lower = f"{img_url} {title} {desc}".lower() - # 1. At least one significant token of the performer's name MUST appear in the URL, title or description - if not any(tok in text_lower for tok in name_tokens): - return False - - # 2. Blacklist non-person, nature, stock photo, and unrelated domains + # 1. Blacklist non-person, ecommerce, stock photo, and unrelated domains blacklist_domains = [ 'pxhere.com', 'pixabay.com', 'freepik.com', 'alamy.com', 'wikipedia.org', 'wikimedia.org', 'infoescola.com', 'ufrgs.br', 'todojujuy.com', 'gettyimages.com', - 'shutterstock.com', 'tripadvisor.com', 'etsy.com', 'indiamike.com', 'toolstop.co.uk', - 'homedepot.com', 'youtube.com', 'disney', 'marvel', 'cartoon', 'anime', 'uhrcenter.de' + 'shutterstock.com', 'tripadvisor.com', 'etsy.com', 'etsystatic.com', 'pinterest.com', + 'pinimg.com', 'ebay.com', 'amazon.com', 'indiamike.com', 'toolstop.co.uk', + 'homedepot.com', 'youtube.com', 'disney', 'marvel', 'cartoon', 'anime', 'uhrcenter.de', + 'swimxwin.com', 'cdiscount.com', 'dailymail.co.uk', 'dreamstime.com' ] if any(b in url_lower for b in blacklist_domains): return False - # 3. Blacklist non-human junk keywords (frogs, tools, shrines, boats, etc.) + # 2. Blacklist non-human junk keywords junk_keywords = [ 'sapo', 'frog', 'toad', 'shrine', 'temple', 'power-tool', 'drill', 'tractor', 'boat', 'yacht', 'car-parts', 'engine', 'railroad', 'railway', 'hardware', - 'wildlife', 'amphibian', 'reptile', 'jewelry', 'schmuck', 'panties', 'costume' + 'wildlife', 'amphibian', 'reptile', 'jewelry', 'schmuck', 'panties', 'costume', + 'tonsils', 'throat', 'golf-course', 'clubhouse', 'drone' ] if any(j in text_lower for j in junk_keywords): return False - return True + # 3. Verified adult biography / media domains + adult_domains = [ + 'babepedia.com', 'freeones.com', 'iafd.com', 'boobpedia.com', + 'thenewsgod.com', 'famousbio.wiki', 'globalzonetoday.com', 'adultdvdempire.com', + 'stashdb.org', 'theporndb.net', 'wikistarbio.com', 'babecelebs.com', + 'pornstarbio.com', 'babe.today', 'indexxx.com', 'scrolller.com' + ] + is_adult_domain = any(ad in url_lower for ad in adult_domains) + + # 4. Strict name token matching: + # If from an adult domain, at least 1 significant token must match. + # If from a general web domain, ALL tokens (or the full name phrase) must match. + if is_adult_domain: + return any(tok in text_lower for tok in name_tokens) + else: + # Full name phrase or all tokens must match + if clean_name in text_lower: + return True + return all(tok in text_lower for tok in name_tokens) def search_performer_candidate_images(name, aliases=None, limit=6): clean_name = clean_performer_search_name(name)