Files
personal_development/sms/parse_sms.py
T
2026-01-03 13:30:35 -05:00

73 lines
3.2 KiB
Python

import re
import os
import html
# --- CONFIGURATION ---
INPUT_FILE = '/home/david/code/personal_development/sms/sms-20251127173937.xml'
OUTPUT_FILE = '/home/david/code/personal_development/sms/michelle_kifer_messages.txt'
TARGET_CONTACT_NAME = "Michelle 🔥 Kifer"
# ---------------------
def parse_messages_manually():
if not os.path.exists(INPUT_FILE):
print(f"Error: Could not find file '{INPUT_FILE}'. Please check the filename.")
return
print(f"Parsing '{INPUT_FILE}' manually to avoid memory errors...")
messages = []
target_name_lower = TARGET_CONTACT_NAME.lower()
in_target_mms = False
try:
with open(INPUT_FILE, 'r', encoding='utf-8', errors='ignore') as f:
for line in f:
stripped_line = line.strip()
# Process SMS: expecting a single line like <sms ... />
if stripped_line.startswith('<sms '):
contact_name_match = re.search(r'contact_name="([^"]*)"', stripped_line, re.IGNORECASE)
if contact_name_match and html.unescape(contact_name_match.group(1)).lower() == target_name_lower:
body_match = re.search(r'body="([^"]*)"', stripped_line, re.IGNORECASE)
if body_match:
# Unescape HTML entities like &amp;
messages.append(html.unescape(body_match.group(1)))
# Process start of an MMS block
elif stripped_line.startswith('<mms '):
contact_name_match = re.search(r'contact_name="([^"]*)"', stripped_line, re.IGNORECASE)
if contact_name_match and html.unescape(contact_name_match.group(1)).lower() == target_name_lower:
in_target_mms = True
else:
in_target_mms = False # Ensure we reset if it's not the target contact
# Process end of an MMS block
elif stripped_line.startswith('</mms>'):
in_target_mms = False
# Process MMS parts if we are inside a target MMS block
elif in_target_mms and stripped_line.startswith('<part '):
# Look for plain text parts
if 'ct="text/plain"' in stripped_line:
text_match = re.search(r'text="([^"]*)"', stripped_line, re.IGNORECASE)
if text_match:
messages.append(html.unescape(text_match.group(1)))
except Exception as e:
print(f"An unexpected error occurred during manual parsing: {e}")
return
# Write to a simple text file
if messages:
try:
with open(OUTPUT_FILE, 'w', encoding='utf-8') as f:
for msg in messages:
f.write(msg + '\n\n') # Add a blank line for readability
print(f"Success! Exported {len(messages)} messages to '{OUTPUT_FILE}'.")
except IOError as e:
print(f"Error writing to file: {e}")
else:
print(f"No messages found for contact: {TARGET_CONTACT_NAME}")
if __name__ == "__main__":
parse_messages_manually()