Files
2026-01-03 13:30:35 -05:00

93 lines
3.9 KiB
Python

from flask import Flask, jsonify, render_template
import re
import os
from collections import Counter
import html
# --- CONFIGURATION ---
MESSAGE_FILE_PATH = '/home/david/code/personal_development/sms/michelle_kifer_messages.txt'
# Simple list of common English stop words
STOP_WORDS = set([
'a', 'about', 'above', 'after', 'again', 'against', 'all', 'am', 'an', 'and', 'any', 'are', 'as', 'at',
'be', 'because', 'been', 'before', 'being', 'below', 'between', 'both', 'but', 'by', 'can', 'did', 'do',
'does', 'doing', 'down', 'during', 'each', 'few', 'for', 'from', 'further', 'had', 'has', 'have', 'having',
'he', 'her', 'here', 'hers', 'herself', 'him', 'himself', 'his', 'how', 'i', 'if', 'in', 'into', 'is', 'it',
'its', 'itself', 'just', 'me', 'more', 'most', 'my', 'myself', 'no', 'nor', 'not', 'now', 'of', 'off', 'on',
'once', 'only', 'or', 'other', 'our', 'ours', 'ourselves', 'out', 'over', 'own', 's', 'same', 'she', 'should',
'so', 'some', 'such', 't', 'than', 'that', 'the', 'their', 'theirs', 'them', 'themselves', 'then', 'there',
'these', 'they', 'this', 'those', 'through', 'to', 'too', 'under', 'until', 'up', 'very', 'was', 'we', 'were',
'what', 'when', 'where', 'which', 'while', 'who', 'whom', 'why', 'will', 'with', 'you', 'your', 'yours',
'yourself', 'yourselves', 'im', 'ive', 'id', 'like', 'its', 'im', 'get'
])
# Words that indicate a desire or a gift-related action
GIFT_TRIGGER_WORDS = set(['want', 'wants', 'needed', 'needs', 'wishing', 'wish', 'buy', 'buying', 'get', 'getting', 'love', 'loves', 'like', 'likes'])
# Words that are themselves often gifts
GIFT_ITEM_WORDS = set(['book', 'books', 'jewelry', 'necklace', 'earrings', 'bracelet', 'ring', 'concert', 'tickets', 'shoes', 'boots', 'jacket', 'shirt', 'dress', 'album', 'game', 'watch', 'bag', 'purse'])
app = Flask(__name__)
def process_text():
"""Reads and analyzes the message file to produce word data."""
if not os.path.exists(MESSAGE_FILE_PATH):
return []
with open(MESSAGE_FILE_PATH, 'r', encoding='utf-8') as f:
text = f.read()
# Clean and tokenize the text
words = re.findall(r'\b\w+\b', text.lower())
# Calculate frequency of all non-stop-words
word_counts = Counter(word for word in words if word not in STOP_WORDS and len(word) > 1)
processed_data = []
for i, word in enumerate(words):
if word in word_counts:
# Calculate gift score
gift_score = 0
# Check for nearby trigger words (within a 5-word window before the current word)
window = words[max(0, i-5):i]
if GIFT_TRIGGER_WORDS.intersection(window):
gift_score += 2
# Check if the word itself is a common gift item
if word in GIFT_ITEM_WORDS:
gift_score += 3
# Add word to our data list - we handle aggregation on the client-side for simplicity
processed_data.append({
'text': word,
'size': word_counts[word],
'gift_score': gift_score
})
# Deduplicate and aggregate scores
final_data = {}
for item in processed_data:
if item['text'] not in final_data:
final_data[item['text']] = {'size': item['size'], 'gift_score': 0}
final_data[item['text']]['gift_score'] += item['gift_score']
# Convert back to list
return [{'text': k, 'size': v['size'], 'gift_score': v['gift_score']} for k, v in final_data.items()]
@app.route('/')
def index():
"""Serve the main HTML page."""
return render_template('index.html')
@app.route('/get-word-data')
def get_word_data():
"""Provide the processed word data as JSON."""
data = process_text()
return jsonify(data)
if __name__ == '__main__':
print("Starting the Gift Idea Visualizer application.")
print("Please open your web browser and go to: http://127.0.0.1:5000")
app.run(debug=True)