93 lines
3.9 KiB
Python
93 lines
3.9 KiB
Python
from flask import Flask, jsonify, render_template
|
|
import re
|
|
import os
|
|
from collections import Counter
|
|
import html
|
|
|
|
# --- CONFIGURATION ---
|
|
MESSAGE_FILE_PATH = '/home/david/code/personal_development/sms/michelle_kifer_messages.txt'
|
|
|
|
# Simple list of common English stop words
|
|
STOP_WORDS = set([
|
|
'a', 'about', 'above', 'after', 'again', 'against', 'all', 'am', 'an', 'and', 'any', 'are', 'as', 'at',
|
|
'be', 'because', 'been', 'before', 'being', 'below', 'between', 'both', 'but', 'by', 'can', 'did', 'do',
|
|
'does', 'doing', 'down', 'during', 'each', 'few', 'for', 'from', 'further', 'had', 'has', 'have', 'having',
|
|
'he', 'her', 'here', 'hers', 'herself', 'him', 'himself', 'his', 'how', 'i', 'if', 'in', 'into', 'is', 'it',
|
|
'its', 'itself', 'just', 'me', 'more', 'most', 'my', 'myself', 'no', 'nor', 'not', 'now', 'of', 'off', 'on',
|
|
'once', 'only', 'or', 'other', 'our', 'ours', 'ourselves', 'out', 'over', 'own', 's', 'same', 'she', 'should',
|
|
'so', 'some', 'such', 't', 'than', 'that', 'the', 'their', 'theirs', 'them', 'themselves', 'then', 'there',
|
|
'these', 'they', 'this', 'those', 'through', 'to', 'too', 'under', 'until', 'up', 'very', 'was', 'we', 'were',
|
|
'what', 'when', 'where', 'which', 'while', 'who', 'whom', 'why', 'will', 'with', 'you', 'your', 'yours',
|
|
'yourself', 'yourselves', 'im', 'ive', 'id', 'like', 'its', 'im', 'get'
|
|
])
|
|
|
|
# Words that indicate a desire or a gift-related action
|
|
GIFT_TRIGGER_WORDS = set(['want', 'wants', 'needed', 'needs', 'wishing', 'wish', 'buy', 'buying', 'get', 'getting', 'love', 'loves', 'like', 'likes'])
|
|
|
|
# Words that are themselves often gifts
|
|
GIFT_ITEM_WORDS = set(['book', 'books', 'jewelry', 'necklace', 'earrings', 'bracelet', 'ring', 'concert', 'tickets', 'shoes', 'boots', 'jacket', 'shirt', 'dress', 'album', 'game', 'watch', 'bag', 'purse'])
|
|
|
|
app = Flask(__name__)
|
|
|
|
def process_text():
|
|
"""Reads and analyzes the message file to produce word data."""
|
|
if not os.path.exists(MESSAGE_FILE_PATH):
|
|
return []
|
|
|
|
with open(MESSAGE_FILE_PATH, 'r', encoding='utf-8') as f:
|
|
text = f.read()
|
|
|
|
# Clean and tokenize the text
|
|
words = re.findall(r'\b\w+\b', text.lower())
|
|
|
|
# Calculate frequency of all non-stop-words
|
|
word_counts = Counter(word for word in words if word not in STOP_WORDS and len(word) > 1)
|
|
|
|
processed_data = []
|
|
for i, word in enumerate(words):
|
|
if word in word_counts:
|
|
# Calculate gift score
|
|
gift_score = 0
|
|
# Check for nearby trigger words (within a 5-word window before the current word)
|
|
window = words[max(0, i-5):i]
|
|
if GIFT_TRIGGER_WORDS.intersection(window):
|
|
gift_score += 2
|
|
|
|
# Check if the word itself is a common gift item
|
|
if word in GIFT_ITEM_WORDS:
|
|
gift_score += 3
|
|
|
|
# Add word to our data list - we handle aggregation on the client-side for simplicity
|
|
processed_data.append({
|
|
'text': word,
|
|
'size': word_counts[word],
|
|
'gift_score': gift_score
|
|
})
|
|
|
|
# Deduplicate and aggregate scores
|
|
final_data = {}
|
|
for item in processed_data:
|
|
if item['text'] not in final_data:
|
|
final_data[item['text']] = {'size': item['size'], 'gift_score': 0}
|
|
final_data[item['text']]['gift_score'] += item['gift_score']
|
|
|
|
# Convert back to list
|
|
return [{'text': k, 'size': v['size'], 'gift_score': v['gift_score']} for k, v in final_data.items()]
|
|
|
|
|
|
@app.route('/')
|
|
def index():
|
|
"""Serve the main HTML page."""
|
|
return render_template('index.html')
|
|
|
|
@app.route('/get-word-data')
|
|
def get_word_data():
|
|
"""Provide the processed word data as JSON."""
|
|
data = process_text()
|
|
return jsonify(data)
|
|
|
|
if __name__ == '__main__':
|
|
print("Starting the Gift Idea Visualizer application.")
|
|
print("Please open your web browser and go to: http://127.0.0.1:5000")
|
|
app.run(debug=True)
|