Files
2026-01-09 19:24:41 +00:00

203 lines
6.9 KiB
Python

import os
import sys
import glob
import time
import argparse
from pathlib import Path
import cv2 # opencv-python
import ollama
# Configuration
# 'moondream' is extremely lightweight (1.6B params).
# If you want more accuracy at the cost of speed, change this to 'llava' (7B params).
MODEL_NAME = 'moondream'
# How many seconds to skip between frames.
# Higher = Faster processing, but might miss short actions.
FRAME_INTERVAL_SECONDS = 5
VIDEO_EXTENSIONS = {'.mp4', '.mov', '.avi', '.mkv', '.webm', '.flv'}
def analyze_frame(frame_bytes):
"""Sends a single image frame to the local model for description."""
try:
response = ollama.chat(model=MODEL_NAME, messages=[
{
'role': 'user',
'content': 'Describe the sexual activity in this image in detail. Be specific about whether there is oral sex or penetration.',
'images': [frame_bytes]
}
])
return response['message']['content']
except Exception as e:
print(f"Error communicating with Ollama: {e}")
return ""
def process_video(video_path):
"""Extracts frames and aggregates analysis."""
cap = cv2.VideoCapture(str(video_path))
if not cap.isOpened():
print(f"Error opening video: {video_path}")
return None, False
fps = cap.get(cv2.CAP_PROP_FPS)
frame_interval = int(fps * FRAME_INTERVAL_SECONDS)
frame_count = 0
descriptions = []
print(f"Scanning {video_path.name} (taking 1 frame every {FRAME_INTERVAL_SECONDS}s)...")
while cap.isOpened():
ret, frame = cap.read()
if not ret:
break
if frame_count % frame_interval == 0:
# Convert frame to bytes for Ollama
_, buffer = cv2.imencode('.jpg', frame)
frame_bytes = buffer.tobytes()
# Analyze frame
desc = analyze_frame(frame_bytes)
if desc:
descriptions.append(desc)
# print(f" [Frame {frame_count}] {desc[:50]}...") # Uncomment for debug noise
frame_count += 1
cap.release()
if not descriptions:
return "No frames analyzed.", False
# Aggregate Logic
# We join all descriptions and check keywords.
# This is a naive heuristic because the model doesn't have "memory" of the whole video context,
# just individual snapshots.
full_text = " ".join(descriptions).lower()
has_oral = 'oral' in full_text or 'blowjob' in full_text or 'sucking' in full_text or 'fellatio' in full_text
has_penetration = 'penetration' in full_text or 'sex' in full_text or 'intercourse' in full_text or 'vaginal' in full_text or 'anal' in full_text or 'fucking' in full_text
# Refined logic: simple keyword matching can be prone to false positives/negatives with small models.
# However, for an automated script, this is the baseline.
summary = f"Analyzed {len(descriptions)} frames.\n\nCombined Observations:\n{full_text}"
# Criteria: Only Blowjobs (Oral), NO Sex (Penetration)
# Note: 'sex' is a broad term. Small models might use it generically.
# You might need to tune 'has_penetration' keywords based on model behavior.
is_match = has_oral and not has_penetration
return summary, is_match
def update_html_report(report_path, video_name, summary, flag):
file_exists = os.path.exists(report_path)
with open(report_path, 'a', encoding='utf-8') as f:
if not file_exists:
f.write("""
<!DOCTYPE html>
<html>
<head>
<style>
table { border-collapse: collapse; width: 100%; }
th, td { border: 1px solid #ddd; padding: 8px; text-align: left; vertical-align: top;}
tr:nth-child(even) { background-color: #f2f2f2; }
.flagged { color: red; font-weight: bold; }
.summary-text { max-height: 200px; overflow-y: auto; display: block; }
</style>
</head>
<body>
<h2>Local Video Analysis Report</h2>
<table>
<tr>
<th>File Name</th>
<th>Summary (Frame Aggregation)</th>
<th>Criteria Met (Only BJ, No Sex)</th>
</tr>
""")
row_class = ' class="flagged"' if flag else ""
# Truncate summary for HTML display to avoid massive cells
display_summary = summary[:1000] + "..." if len(summary) > 1000 else summary
f.write(f" <tr>\n <td>{video_name}</td>\n <td><div class='summary-text'>{display_summary}</div></td>\n <td{row_class}>{flag}</td>\n </tr>\n")
def close_html_report(report_path):
if os.path.exists(report_path):
with open(report_path, 'a', encoding='utf-8') as f:
f.write("</table>\n</body>\n</html>")
def main():
parser = argparse.ArgumentParser(description="Analyze videos locally using Ollama.")
parser.add_argument("directory", help="Target directory containing videos")
args = parser.parse_args()
target_dir = Path(args.directory)
if not target_dir.is_dir():
print(f"Directory not found: {target_dir}")
sys.exit(1)
files_to_delete = []
video_files = [
f for f in target_dir.iterdir()
if f.is_file() and f.suffix.lower() in VIDEO_EXTENSIONS
]
print(f"Found {len(video_files)} videos in {target_dir}")
print(f"Using local model: {MODEL_NAME}")
report_path = target_dir / "local_analysis_report.html"
if report_path.exists():
os.remove(report_path)
for video_path in video_files:
print(f"\nProcessing: {video_path.name}")
summary, criteria_met = process_video(video_path)
if summary:
# Save text summary
txt_path = video_path.with_suffix('.txt')
with open(txt_path, 'w', encoding='utf-8') as f:
f.write(summary)
print(f"Saved summary to {txt_path.name}")
# Update HTML
update_html_report(report_path, video_path.name, summary, criteria_met)
if criteria_met:
print("--> MATCHES CRITERIA: Only blowjobs, no sex.")
files_to_delete.append(video_path)
else:
print("--> Does not match deletion criteria.")
else:
print("Skipped (no content analyzing)")
close_html_report(report_path)
print(f"\nAnalysis complete. Report saved to {report_path}")
if files_to_delete:
print("\n" + "="*40)
print(f"Found {len(files_to_delete)} files matching 'Only Blowjobs, No Sex':")
for f in files_to_delete:
print(f"- {f.name}")
print("="*40)
confirm = input("\nDo you want to DELETE these files? (yes/no): ").lower()
if confirm == 'yes':
for f in files_to_delete:
try:
os.remove(f)
print(f"Deleted: {f.name}")
except OSError as e:
print(f"Error deleting {f.name}: {e}")
else:
print("Deletion cancelled.")
else:
print("\nNo files matched the deletion criteria.")
if __name__ == "__main__":
main()