diff --git a/bae_chat_log_analysis/__main__.py b/bae_chat_log_analysis/__main__.py index a9f3882..c53e515 100644 --- a/bae_chat_log_analysis/__main__.py +++ b/bae_chat_log_analysis/__main__.py @@ -7,20 +7,21 @@ from .token_counter import count_tokens SOURCE_FILE = Path("bae_chat_logs/Source/WhatsApp_Bae_Chat.txt") REDUCED_FILE = Path("bae_chat_logs/Reduced/WhatsApp_Chat-08-03-2026-reduced") -SPLIT_OPTIONS = {'1': 'year', '2': 'month', '3': 'day', '4': 'none'} +SPLIT_OPTIONS = {'1': 'year', '2': 'month', '3': 'week', '4': 'day', '5': 'none'} def prompt_split_option(): print("How would you like to split the chat output?") print(" 1) By year") print(" 2) By month") - print(" 3) By day") - print(" 4) No split (single file)") + print(" 3) By week") + print(" 4) By day") + print(" 5) No split (single file)") while True: - choice = input("Enter choice (1-4): ").strip() + choice = input("Enter choice (1-5): ").strip() if choice in SPLIT_OPTIONS: return SPLIT_OPTIONS[choice] - print("Invalid choice. Please enter 1, 2, 3, or 4.") + print("Invalid choice. Please enter 1, 2, 3, 4, or 5.") def main(): diff --git a/bae_chat_log_analysis/abbreviations.py b/bae_chat_log_analysis/abbreviations.py new file mode 100644 index 0000000..ad412fe --- /dev/null +++ b/bae_chat_log_analysis/abbreviations.py @@ -0,0 +1,12 @@ +import ast +from pathlib import Path + +ABBREVIATIONS_FILE = Path(__file__).resolve().parent.parent / "abbreviations.txt" + + +def generate_emoji_abbreviations(): + """Load emoji abbreviation mappings from abbreviations.txt.""" + text = ABBREVIATIONS_FILE.read_text(encoding="utf-8") + # The file contains "EMOJI_ABBREVIATIONS = { ... }" — extract the dict literal + _, _, dict_literal = text.partition("=") + return ast.literal_eval(dict_literal.strip()) diff --git a/bae_chat_log_analysis/reducer.py b/bae_chat_log_analysis/reducer.py index 5b20927..94072f1 100644 --- a/bae_chat_log_analysis/reducer.py +++ b/bae_chat_log_analysis/reducer.py @@ -1,5 +1,6 @@ import re from collections import defaultdict +from datetime import date from pathlib import Path import emoji @@ -24,6 +25,9 @@ def _date_key(month, day, year_short, split_by): return str(year) elif split_by == 'month': return f"{year}-{int(month):02d}" + elif split_by == 'week': + iso_year, iso_week, _ = date(year, int(month), int(day)).isocalendar() + return f"{iso_year}-W{iso_week:02d}" else: # day return f"{year}-{int(month):02d}-{int(day):02d}" @@ -31,7 +35,7 @@ def _date_key(month, day, year_short, split_by): def reduce_tokens(input_file, output_file, abbreviations, split_by='none'): """Replace long usernames and convert emojis to shortened text codes. - split_by: 'none', 'year', 'month', or 'day' + split_by: 'none', 'year', 'month', 'week', or 'day' Returns a list of output file paths that were written. """ output_file = Path(output_file)