Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
da7596ad30 |
@@ -7,20 +7,21 @@ from .token_counter import count_tokens
|
|||||||
SOURCE_FILE = Path("bae_chat_logs/Source/WhatsApp_Bae_Chat.txt")
|
SOURCE_FILE = Path("bae_chat_logs/Source/WhatsApp_Bae_Chat.txt")
|
||||||
REDUCED_FILE = Path("bae_chat_logs/Reduced/WhatsApp_Chat-08-03-2026-reduced")
|
REDUCED_FILE = Path("bae_chat_logs/Reduced/WhatsApp_Chat-08-03-2026-reduced")
|
||||||
|
|
||||||
SPLIT_OPTIONS = {'1': 'year', '2': 'month', '3': 'day', '4': 'none'}
|
SPLIT_OPTIONS = {'1': 'year', '2': 'month', '3': 'week', '4': 'day', '5': 'none'}
|
||||||
|
|
||||||
|
|
||||||
def prompt_split_option():
|
def prompt_split_option():
|
||||||
print("How would you like to split the chat output?")
|
print("How would you like to split the chat output?")
|
||||||
print(" 1) By year")
|
print(" 1) By year")
|
||||||
print(" 2) By month")
|
print(" 2) By month")
|
||||||
print(" 3) By day")
|
print(" 3) By week")
|
||||||
print(" 4) No split (single file)")
|
print(" 4) By day")
|
||||||
|
print(" 5) No split (single file)")
|
||||||
while True:
|
while True:
|
||||||
choice = input("Enter choice (1-4): ").strip()
|
choice = input("Enter choice (1-5): ").strip()
|
||||||
if choice in SPLIT_OPTIONS:
|
if choice in SPLIT_OPTIONS:
|
||||||
return SPLIT_OPTIONS[choice]
|
return SPLIT_OPTIONS[choice]
|
||||||
print("Invalid choice. Please enter 1, 2, 3, or 4.")
|
print("Invalid choice. Please enter 1, 2, 3, 4, or 5.")
|
||||||
|
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
|
|||||||
@@ -0,0 +1,12 @@
|
|||||||
|
import ast
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
ABBREVIATIONS_FILE = Path(__file__).resolve().parent.parent / "abbreviations.txt"
|
||||||
|
|
||||||
|
|
||||||
|
def generate_emoji_abbreviations():
|
||||||
|
"""Load emoji abbreviation mappings from abbreviations.txt."""
|
||||||
|
text = ABBREVIATIONS_FILE.read_text(encoding="utf-8")
|
||||||
|
# The file contains "EMOJI_ABBREVIATIONS = { ... }" — extract the dict literal
|
||||||
|
_, _, dict_literal = text.partition("=")
|
||||||
|
return ast.literal_eval(dict_literal.strip())
|
||||||
@@ -1,5 +1,6 @@
|
|||||||
import re
|
import re
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
|
from datetime import date
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
import emoji
|
import emoji
|
||||||
@@ -24,6 +25,9 @@ def _date_key(month, day, year_short, split_by):
|
|||||||
return str(year)
|
return str(year)
|
||||||
elif split_by == 'month':
|
elif split_by == 'month':
|
||||||
return f"{year}-{int(month):02d}"
|
return f"{year}-{int(month):02d}"
|
||||||
|
elif split_by == 'week':
|
||||||
|
iso_year, iso_week, _ = date(year, int(month), int(day)).isocalendar()
|
||||||
|
return f"{iso_year}-W{iso_week:02d}"
|
||||||
else: # day
|
else: # day
|
||||||
return f"{year}-{int(month):02d}-{int(day):02d}"
|
return f"{year}-{int(month):02d}-{int(day):02d}"
|
||||||
|
|
||||||
@@ -31,7 +35,7 @@ def _date_key(month, day, year_short, split_by):
|
|||||||
def reduce_tokens(input_file, output_file, abbreviations, split_by='none'):
|
def reduce_tokens(input_file, output_file, abbreviations, split_by='none'):
|
||||||
"""Replace long usernames and convert emojis to shortened text codes.
|
"""Replace long usernames and convert emojis to shortened text codes.
|
||||||
|
|
||||||
split_by: 'none', 'year', 'month', or 'day'
|
split_by: 'none', 'year', 'month', 'week', or 'day'
|
||||||
Returns a list of output file paths that were written.
|
Returns a list of output file paths that were written.
|
||||||
"""
|
"""
|
||||||
output_file = Path(output_file)
|
output_file = Path(output_file)
|
||||||
|
|||||||
Reference in New Issue
Block a user