from flask import Flask, jsonify, render_template import re import os from collections import Counter import html # --- CONFIGURATION --- MESSAGE_FILE_PATH = '/home/david/code/personal_development/sms/michelle_kifer_messages.txt' # Simple list of common English stop words STOP_WORDS = set([ 'a', 'about', 'above', 'after', 'again', 'against', 'all', 'am', 'an', 'and', 'any', 'are', 'as', 'at', 'be', 'because', 'been', 'before', 'being', 'below', 'between', 'both', 'but', 'by', 'can', 'did', 'do', 'does', 'doing', 'down', 'during', 'each', 'few', 'for', 'from', 'further', 'had', 'has', 'have', 'having', 'he', 'her', 'here', 'hers', 'herself', 'him', 'himself', 'his', 'how', 'i', 'if', 'in', 'into', 'is', 'it', 'its', 'itself', 'just', 'me', 'more', 'most', 'my', 'myself', 'no', 'nor', 'not', 'now', 'of', 'off', 'on', 'once', 'only', 'or', 'other', 'our', 'ours', 'ourselves', 'out', 'over', 'own', 's', 'same', 'she', 'should', 'so', 'some', 'such', 't', 'than', 'that', 'the', 'their', 'theirs', 'them', 'themselves', 'then', 'there', 'these', 'they', 'this', 'those', 'through', 'to', 'too', 'under', 'until', 'up', 'very', 'was', 'we', 'were', 'what', 'when', 'where', 'which', 'while', 'who', 'whom', 'why', 'will', 'with', 'you', 'your', 'yours', 'yourself', 'yourselves', 'im', 'ive', 'id', 'like', 'its', 'im', 'get' ]) # Words that indicate a desire or a gift-related action GIFT_TRIGGER_WORDS = set(['want', 'wants', 'needed', 'needs', 'wishing', 'wish', 'buy', 'buying', 'get', 'getting', 'love', 'loves', 'like', 'likes']) # Words that are themselves often gifts GIFT_ITEM_WORDS = set(['book', 'books', 'jewelry', 'necklace', 'earrings', 'bracelet', 'ring', 'concert', 'tickets', 'shoes', 'boots', 'jacket', 'shirt', 'dress', 'album', 'game', 'watch', 'bag', 'purse']) app = Flask(__name__) def process_text(): """Reads and analyzes the message file to produce word data.""" if not os.path.exists(MESSAGE_FILE_PATH): return [] with open(MESSAGE_FILE_PATH, 'r', encoding='utf-8') as f: text = f.read() # Clean and tokenize the text words = re.findall(r'\b\w+\b', text.lower()) # Calculate frequency of all non-stop-words word_counts = Counter(word for word in words if word not in STOP_WORDS and len(word) > 1) processed_data = [] for i, word in enumerate(words): if word in word_counts: # Calculate gift score gift_score = 0 # Check for nearby trigger words (within a 5-word window before the current word) window = words[max(0, i-5):i] if GIFT_TRIGGER_WORDS.intersection(window): gift_score += 2 # Check if the word itself is a common gift item if word in GIFT_ITEM_WORDS: gift_score += 3 # Add word to our data list - we handle aggregation on the client-side for simplicity processed_data.append({ 'text': word, 'size': word_counts[word], 'gift_score': gift_score }) # Deduplicate and aggregate scores final_data = {} for item in processed_data: if item['text'] not in final_data: final_data[item['text']] = {'size': item['size'], 'gift_score': 0} final_data[item['text']]['gift_score'] += item['gift_score'] # Convert back to list return [{'text': k, 'size': v['size'], 'gift_score': v['gift_score']} for k, v in final_data.items()] @app.route('/') def index(): """Serve the main HTML page.""" return render_template('index.html') @app.route('/get-word-data') def get_word_data(): """Provide the processed word data as JSON.""" data = process_text() return jsonify(data) if __name__ == '__main__': print("Starting the Gift Idea Visualizer application.") print("Please open your web browser and go to: http://127.0.0.1:5000") app.run(debug=True)