73 lines
3.2 KiB
Python
73 lines
3.2 KiB
Python
import re
|
|
import os
|
|
import html
|
|
|
|
# --- CONFIGURATION ---
|
|
INPUT_FILE = '/home/david/code/personal_development/sms/sms-20251127173937.xml'
|
|
OUTPUT_FILE = '/home/david/code/personal_development/sms/michelle_kifer_messages.txt'
|
|
TARGET_CONTACT_NAME = "Michelle 🔥 Kifer"
|
|
# ---------------------
|
|
|
|
def parse_messages_manually():
|
|
if not os.path.exists(INPUT_FILE):
|
|
print(f"Error: Could not find file '{INPUT_FILE}'. Please check the filename.")
|
|
return
|
|
|
|
print(f"Parsing '{INPUT_FILE}' manually to avoid memory errors...")
|
|
|
|
messages = []
|
|
target_name_lower = TARGET_CONTACT_NAME.lower()
|
|
in_target_mms = False
|
|
|
|
try:
|
|
with open(INPUT_FILE, 'r', encoding='utf-8', errors='ignore') as f:
|
|
for line in f:
|
|
stripped_line = line.strip()
|
|
|
|
# Process SMS: expecting a single line like <sms ... />
|
|
if stripped_line.startswith('<sms '):
|
|
contact_name_match = re.search(r'contact_name="([^"]*)"', stripped_line, re.IGNORECASE)
|
|
if contact_name_match and html.unescape(contact_name_match.group(1)).lower() == target_name_lower:
|
|
body_match = re.search(r'body="([^"]*)"', stripped_line, re.IGNORECASE)
|
|
if body_match:
|
|
# Unescape HTML entities like &
|
|
messages.append(html.unescape(body_match.group(1)))
|
|
|
|
# Process start of an MMS block
|
|
elif stripped_line.startswith('<mms '):
|
|
contact_name_match = re.search(r'contact_name="([^"]*)"', stripped_line, re.IGNORECASE)
|
|
if contact_name_match and html.unescape(contact_name_match.group(1)).lower() == target_name_lower:
|
|
in_target_mms = True
|
|
else:
|
|
in_target_mms = False # Ensure we reset if it's not the target contact
|
|
|
|
# Process end of an MMS block
|
|
elif stripped_line.startswith('</mms>'):
|
|
in_target_mms = False
|
|
|
|
# Process MMS parts if we are inside a target MMS block
|
|
elif in_target_mms and stripped_line.startswith('<part '):
|
|
# Look for plain text parts
|
|
if 'ct="text/plain"' in stripped_line:
|
|
text_match = re.search(r'text="([^"]*)"', stripped_line, re.IGNORECASE)
|
|
if text_match:
|
|
messages.append(html.unescape(text_match.group(1)))
|
|
|
|
except Exception as e:
|
|
print(f"An unexpected error occurred during manual parsing: {e}")
|
|
return
|
|
|
|
# Write to a simple text file
|
|
if messages:
|
|
try:
|
|
with open(OUTPUT_FILE, 'w', encoding='utf-8') as f:
|
|
for msg in messages:
|
|
f.write(msg + '\n\n') # Add a blank line for readability
|
|
print(f"Success! Exported {len(messages)} messages to '{OUTPUT_FILE}'.")
|
|
except IOError as e:
|
|
print(f"Error writing to file: {e}")
|
|
else:
|
|
print(f"No messages found for contact: {TARGET_CONTACT_NAME}")
|
|
|
|
if __name__ == "__main__":
|
|
parse_messages_manually() |