diff --git a/.gitignore b/.gitignore index 9fe5c48..e52e650 100644 --- a/.gitignore +++ b/.gitignore @@ -1,2 +1,2 @@ -config.env -config.py +config.json +__pycache__/ \ No newline at end of file diff --git a/00_exportchapters.py b/00_exportchapters.py deleted file mode 100644 index 4f91215..0000000 --- a/00_exportchapters.py +++ /dev/null @@ -1,89 +0,0 @@ -'''Align tracks end to end, chop based on chapters list durations, export to final folder, archive originals''' - -import os -import shutil -from pydub import AudioSegment -from config import annex,circulation,archive - -AudioSegment.converter = "/opt/homebrew/Cellar/ffmpeg/5.1.2_3/bin/ffmpeg" -AudioSegment.ffmpeg = "/opt/homebrew/Cellar/ffmpeg/5.1.2_3/bin/ffmpeg" -AudioSegment.ffprobe ="/opt/homebrew/Cellar/ffmpeg/5.1.2_3/bin/ffmpeg" - -def export_audio_file(start, end, name): - start_ms = int(start) - end_ms = int(end) - export = combined[start_ms:end_ms] - folder_name = folder.split(" - ")[-1].strip() - destination_path = circulation - author_name = folder.split(" - ")[0].strip() - export_folder = os.path.join(destination_path, author_name, folder_name) - author_folder = os.path.join(destination_path, author_name) - '''If an author folder doesn't already exist it makes one''' - if not os.path.exists(author_folder): - os.makedirs(author_folder) - export_folder = os.path.join(author_folder, folder_name) - - '''Skips the above if an author already exists and adds the book to it''' - if not os.path.exists(export_folder): - os.makedirs(export_folder) - export_path = os.path.join(export_folder, f"{name}.mp3") - export.export(export_path, format="mp3") - - '''Defines metadata & artwork files''' - metadata_file = os.path.join(folder_path, "cleaned_metadata.json") - album_art = os.path.join(folder_path, "folder.jpg") - if os.path.exists(album_art): - '''Copies metadata & artwork files alongside chapterized audio''' - shutil.copy(album_art, export_folder) - shutil.copy(metadata_file, export_folder) - -main_directory = annex - -for folder in sorted(os.listdir(main_directory)): - folder_path = os.path.join(main_directory, folder) - if os.path.isdir(folder_path): - try: - '''Import MP3s in alphabetical order''' - mp3_files = sorted(os.listdir(folder_path)) - mp3_files = [os.path.join(folder_path, file) for file in mp3_files if file.endswith(".mp3")] - - combined = AudioSegment.empty() - for mp3_file in mp3_files: - print(mp3_file) - combined += AudioSegment.from_file(mp3_file) - - '''Align end to end''' - combined = combined.set_channels(1) - - '''Import label names/durations''' - labels_file = os.path.join(folder_path, "overdrive_chapters_ms_spans.txt") - labels = [] - with open(labels_file, "r") as f: - for line in f: - start, end, name = line.strip().split("\t") - labels.append((start, end, name)) - - '''Initialize counter for duplicate label names''' - counter = {} - - '''Initialize counter for file export''' - file_count = 0 - - for i, label in enumerate(labels): - start, end, name = label - '''This ensures that the final end duration matches the length of the audiobook''' - if end == "0" and i == len(labels) - 1: - end = combined.duration_seconds * 1000 - - file_count += 1 - name = f"{file_count:03}_{name.replace('/', '_')}" - '''format file_count as 3 digit number''' - export_audio_file(start, end, name) - - '''Prints if there's an error''' - except Exception as e: - print(f"Error processing {folder}: {e}") - - '''Archive everything except chapters_list.py''' - if os.path.isdir(folder_path) and folder != 'chapters_list.py': - shutil.move(folder_path, archive + '/' + folder) \ No newline at end of file diff --git a/Chapters_List/__init__.py b/Chapters_List/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/Chapters_List/chapter_ms.py b/Chapters_List/chapter_ms.py new file mode 100644 index 0000000..3f93b50 --- /dev/null +++ b/Chapters_List/chapter_ms.py @@ -0,0 +1,33 @@ +'''Converts label durations to milliseconds for pydub to read''' +import os +import re +from datetime import datetime, timedelta + +def duration_to_milliseconds(duration): + '''Conversion from Seconds to Milliseconds''' + hours, minutes, seconds = duration.split(':') + seconds = int(hours)*3600 + int(minutes)*60 + float(seconds) + return int(seconds * 1000) + +def read_chapters_file(dirpath, filename): + '''Open the chapters file and read the contents.''' + with open(os.path.join(dirpath, filename), 'r') as f: + lines = f.readlines() + return lines + +def new_file(dirpath, filename): + '''Create a new file with the same name but with "ms" added to the end.''' + lines = read_chapters_file(dirpath, filename) + new_filename = filename[:-4] + "_ms.txt" + with open(os.path.join(dirpath, new_filename), 'w') as new_file: + for line in lines: + duration, label = re.split(r"\s+", line.strip(), 1) + milliseconds = duration_to_milliseconds(duration) + new_file.write(f'{milliseconds} {label}\n') + +def add_ms_to_chapters(directory): + '''Loop through all files in the directory tree and process the matching files.''' + for dirpath, _, filenames in os.walk(directory): + for filename in filenames: + if filename.startswith("overdrive_chapters") and filename.endswith(".txt"): + new_file(dirpath, filename) \ No newline at end of file diff --git a/Chapters_List/chapters.py b/Chapters_List/chapters.py new file mode 100644 index 0000000..4319148 --- /dev/null +++ b/Chapters_List/chapters.py @@ -0,0 +1,29 @@ +import sys +import os + +parent_dir = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +sys.path.insert(0, parent_dir) + +from Chapters_List.extract_overdrive_chapters import extract_chapters +from Chapters_List.chapter_ms import add_ms_to_chapters +from XML_JSON.json_scripts import load_config +from Chapters_List.ms_durations import ms_to_durations + +parent_dir = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +config_path = os.path.join(parent_dir, 'config.json') + +config = load_config(config_path) + +queue = config['queue'] #Where your ODMs are downloaded to +annex = config['annex'] #Where the ODMs get unpackaged to +circulation = config['circulation'] #Where the chapterized mp3s get saved +bookshelf = config['bookshelf'] #Where the final audiobook gets dropped + +def chapters_list(): + extract_chapters(annex) + add_ms_to_chapters(annex) + ms_to_durations(annex) + print("Splitting audio files into chapters & adding metadata (this may take a moment)...") + +if __name__ == '__main__': + chapters_list() \ No newline at end of file diff --git a/extract_overdrive_chapters.py b/Chapters_List/extract_overdrive_chapters.py similarity index 92% rename from extract_overdrive_chapters.py rename to Chapters_List/extract_overdrive_chapters.py index 44d0377..060f659 100644 --- a/extract_overdrive_chapters.py +++ b/Chapters_List/extract_overdrive_chapters.py @@ -101,18 +101,11 @@ def visit(dirname, filenames): with open("overdrive_chapters.txt", "w") as file: for name, length in all_chapters.items(): chapstr = f"{timestr(length)} {name}" - print(chapstr) file.write(chapstr + "\n") # print(repr(all_chapters)) - -if __name__ == "__main__": - - if len(sys.argv) > 1: - path = os.path.abspath(sys.argv[1]) - else: - path = os.path.abspath(".") - - for dirname, dirs, files in os.walk(path, topdown=True): +def extract_chapters(path="."): + abs_path = os.path.abspath(path) + for dirname, dirs, files in os.walk(abs_path, topdown=True): dirs[:] = [d for d in dirs if d not in {".git", ".direnv"}] - visit(dirname, files) + visit(dirname, files) \ No newline at end of file diff --git a/Chapters_List/ms_durations.py b/Chapters_List/ms_durations.py new file mode 100644 index 0000000..a23e169 --- /dev/null +++ b/Chapters_List/ms_durations.py @@ -0,0 +1,46 @@ +'''Takes the contents of the Milliseconds durations/labels file and creates start/end times''' +import os + +def create_durations(file, lines): + for i in range(len(lines)): + parts = lines[i].split(" ") + if i == len(lines) - 2: + '''If it's the second-to-last line, write the last line's number and 0''' + file.write(parts[0]) + file.write("\t" + lines[i + 1].split(" ")[0]) + file.write("0") + elif i != len(lines) - 1: + '''Otherwise, write the number from the next line to the file''' + file.write(parts[0]) + file.write("\t" + lines[i + 1].split(" ")[0]) + else: + '''If it's the last line, just write the number and 0''' + if parts[0] != "": + file.write(parts[0]) + file.write("0") + '''Write the rest of the line to the file''' + file.write("\t" + " ".join(parts[1:])) + if i != len(lines) - 1 or (i == len(lines) - 1 and lines[i] != ""): + file.write("\n") + +def bookmark_final_duration(file, lines): + if lines[-1] == "": + '''Ensures there's no empty line at the end of the txt''' + file.seek(file.tell()-1, os.SEEK_SET) + file.truncate() + +def create_spans_txt(file, file_path): + if file == 'overdrive_chapters_ms.txt': + with open(file_path, "r") as file: + lines = file.read().split('\n') + new_file_path = file_path.replace('.txt', '_spans.txt') + with open(new_file_path, "w") as file: + create_durations(file, lines) + bookmark_final_duration(file, lines) + +def ms_to_durations(directory): + for subdir, _, files in os.walk(directory): + for file in files: + '''Locates "overdrive_chapters_ms.txt"''' + file_path = os.path.join(subdir, file) + create_spans_txt(file, file_path) \ No newline at end of file diff --git a/Chop_Tag_Audio/__init__.py b/Chop_Tag_Audio/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/Chop_Tag_Audio/chop_tag_audio.py b/Chop_Tag_Audio/chop_tag_audio.py new file mode 100644 index 0000000..b046cd1 --- /dev/null +++ b/Chop_Tag_Audio/chop_tag_audio.py @@ -0,0 +1,28 @@ +'''Align tracks end to end, chop based on chapters list durations, export to final folder, archive originals''' +import sys +import os + +parent_dir = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +sys.path.insert(0, parent_dir) + +from XML_JSON.json_scripts import load_config +from Chop_Tag_Audio.final_metadata_add import traverse_directory +from Chop_Tag_Audio.exportchapters import split_mp3s + +parent_dir = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +config_path = os.path.join(parent_dir, 'config.json') + +config = load_config(config_path) + +path_to_ffmpeg = config['ffmpeg'] +archive = config['archive'] +annex = config['annex'] +circulation = config['circulation'] +bookshelf = config['bookshelf'] + +def chop_tag_audio(): + split_mp3s(annex, circulation) + traverse_directory(circulation, bookshelf) + +if __name__ == '__main__': + chop_tag_audio() \ No newline at end of file diff --git a/Chop_Tag_Audio/exportchapters.py b/Chop_Tag_Audio/exportchapters.py new file mode 100644 index 0000000..3a9a091 --- /dev/null +++ b/Chop_Tag_Audio/exportchapters.py @@ -0,0 +1,100 @@ +import sys +import random +import os +import shutil +from pydub import AudioSegment +from XML_JSON.json_scripts import load_config + +parent_dir = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +config_path = os.path.join(parent_dir, 'config.json') +config = load_config(config_path) + +path_to_ffmpeg = config['ffmpeg'] +archive = config['archive'] +annex = config['annex'] +circulation = config['circulation'] + +AudioSegment.converter = path_to_ffmpeg +AudioSegment.ffmpeg = path_to_ffmpeg +AudioSegment.ffprobe = path_to_ffmpeg + +def export_audio_file(combined, start_ms, end_ms, name, circulation, author_name, folder_name): + export = combined[start_ms:end_ms] + export_folder = os.path.join(circulation, author_name, folder_name) + if not os.path.exists(export_folder): + os.makedirs(export_folder) + export_path = os.path.join(export_folder, f"{name}.mp3") + export.export(export_path, format="mp3") + return export_folder # Return the folder path where the file was exported + +def split_mp3s(annex, circulation): + update_messages = [ + "Getting close...", + "Hang in there...", + "Still running...", + "Almost done...", + "It'll all be over soon...", + "Bear with me..." + ] + + # Shuffle the list of messages + random.shuffle(update_messages) + + # An iterator over the shuffled list of messages + message_cycle = iter(update_messages) + + for folder in sorted(os.listdir(annex)): + folder_path = os.path.join(annex, folder) + if os.path.isdir(folder_path): + file_count = 0 # Reset file_count to 0 for each new folder + try: + mp3_files = [file for file in sorted(os.listdir(folder_path)) if file.endswith(".mp3")] + combined = AudioSegment.empty() + for mp3_file in mp3_files: + combined += AudioSegment.from_file(os.path.join(folder_path, mp3_file)) + combined = combined.set_channels(1) # Assuming you want mono audio + + labels_file = os.path.join(folder_path, "overdrive_chapters_ms_spans.txt") + with open(labels_file, "r") as f: + labels = [line.strip().split("\t") for line in f] + + for start, end, name in labels: + if end == "0": + end = str(len(combined)) # Use str to keep it consistent with start + start_ms = int(start) + end_ms = int(end) + author_name, folder_name = folder.split(" - ") + formatted_name = f"{file_count:03}_{name.replace('/', '_')}" # Formatting name with file_count + + export_folder = export_audio_file(combined, start_ms, end_ms, formatted_name, circulation, author_name.strip(), folder_name.strip()) + artwork_and_metadata(export_folder, folder_path) + + file_count += 1 # Increment file_count after each file is exported + + message = next(message_cycle, None) + + if message is None: # If the end of the list is reached, shuffle and restart + random.shuffle(update_messages) + message_cycle = iter(update_messages) + message = next(message_cycle) + + print(message) + + except Exception as e: + print(f"Error processing {folder}: {e}") + + archive_folder = os.path.join(archive, folder) + if not os.path.exists(archive_folder): + os.makedirs(archive_folder) + shutil.move(folder_path, archive_folder) + +def artwork_and_metadata(export_folder, folder_path): + metadata_file = os.path.join(folder_path, "cleaned_metadata.json") + album_art = os.path.join(folder_path, "folder.jpg") + if os.path.exists(album_art): + shutil.copy(album_art, export_folder) + if os.path.exists(metadata_file): + shutil.copy(metadata_file, export_folder) + +if __name__ == '__main__': + split_mp3s(annex, circulation) diff --git a/Chop_Tag_Audio/final_metadata_add.py b/Chop_Tag_Audio/final_metadata_add.py new file mode 100644 index 0000000..b66392f --- /dev/null +++ b/Chop_Tag_Audio/final_metadata_add.py @@ -0,0 +1,118 @@ +'''Adds metadata to the chapterized mp3s''' + +import os +import shutil +import json +import re +from mutagen.id3 import ID3, APIC, TIT2, TPE1, TCOM, TCON, TRCK, TALB, ID3NoHeaderError + +def process_directories(directory): + '''Iterate through all subfolders in each author's directory & create cleaned_metadata.json''' + for author_name in os.listdir(directory): + author_path = os.path.join(directory, author_name) + if os.path.isdir(author_path): + for book_title in os.listdir(author_path): + book_path = os.path.join(author_path, book_title) + metadata_file = os.path.join(book_path, "cleaned_metadata.json") + if os.path.exists(metadata_file): + metadata = read_metadata(metadata_file) + if metadata: + add_metadata_to_files(book_path, metadata) + else: + print(f"Skipping {book_title} due to json error") + else: + print(f"Metadata file not found in {book_title}") + +def read_metadata(metadata_file): + '''Read and return json data from the given metadata file''' + try: + with open(metadata_file, "r") as f: + return json.load(f) + except (FileNotFoundError, json.JSONDecodeError) as e: + print(f"Error reading metadata file {metadata_file}: {e}") + return None + +def get_album_title(metadata): + '''Extract the Book Title from metadata and clean it''' + album_title = metadata.get("Title", "") + return re.sub(r"-s\b", "'s", album_title) + +def get_artist(metadata): + '''Extract the Author from metadata''' + return metadata.get("Author", "") + +def get_composer(metadata): + '''Extract the Narrator from metadata''' + return metadata.get("Narrator", "") + +def get_genre(metadata): + '''Extract the Subjects from metadata''' + return metadata.get("Subjects", "") + +def add_metadata_to_files(directory, metadata): + '''Add metadata to all chapterized mp3s in the directory''' + if not metadata: + print(f"No metadata available for directory {directory}. Skipping.") + return + album_title_1 = get_album_title(metadata) + artist = get_artist(metadata) + composer = get_composer(metadata) + genre = get_genre(metadata) + + '''Replace all occurrences of -s with "'s" in the album title''' + album_title_2 = re.sub("-s", "'s", album_title_1) + '''Replace all remaining occurrences of - with "'" in the album title''' + album_title = re.sub("-", "'", album_title_2) + + '''Get a list of all mp3 files in the current directory''' + mp3_files = [f for f in os.listdir(directory) if f.endswith(".mp3")] + + '''Sort the list of mp3 files alphabetically''' + mp3_files.sort() + + '''Iterate through the list of mp3 files''' + for i, mp3_file in enumerate(mp3_files, 1): + mp3_file_path = os.path.join(directory, mp3_file) + + '''Extract the title of the mp3 file''' + file_name, _ = os.path.splitext(mp3_file) + title = re.sub(r"^\d+_", "", file_name) # Correctly extracts the title from the filename + + '''Add the track number & other metadata to the file using ID3''' + try: + audio = ID3(mp3_file_path) + except ID3NoHeaderError: + audio = ID3() + + audio.add(TRCK(encoding=3, text=str(i))) + audio.add(TIT2(encoding=3, text=title)) + audio.add(TPE1(encoding=3, text=artist)) + audio.add(TCOM(encoding=3, text=composer)) + audio.add(TCON(encoding=3, text=genre)) + audio.add(TALB(encoding=3, text=album_title)) + '''Check if folder.jpg exists in the current directory''' + album_art_path = os.path.join(directory, "folder.jpg") + if os.path.exists(album_art_path): + '''Add the album art to the file using ID3''' + with open(album_art_path, "rb") as albumart: + audio.add(APIC(encoding=3, mime='image/jpeg', type=3, desc='Cover', data=albumart.read())) + + audio.save(mp3_file_path) # Save the tags to the file + +def meta_to_mp3(directory, bookshelf): + for author_name in os.listdir(directory): + author_path = os.path.join(directory, author_name) + if os.path.isdir(author_path): + for book_title in os.listdir(author_path): + book_path = os.path.join(author_path, book_title) + destination_path = os.path.join(bookshelf, author_name, book_title) + os.makedirs(os.path.dirname(destination_path), exist_ok=True) + shutil.move(book_path, destination_path) + print(f"Completed: '{book_title}' by '{author_name}'") + # After moving all books, check if the author directory is empty and remove it + if not os.listdir(author_path): + os.rmdir(author_path) + +def traverse_directory(directory, bookshelf): + process_directories(directory) + meta_to_mp3(directory, bookshelf) \ No newline at end of file diff --git a/Chop_Tag_Audio/overdrive-plex.code-workspace b/Chop_Tag_Audio/overdrive-plex.code-workspace new file mode 100644 index 0000000..bab1b7f --- /dev/null +++ b/Chop_Tag_Audio/overdrive-plex.code-workspace @@ -0,0 +1,8 @@ +{ + "folders": [ + { + "path": ".." + } + ], + "settings": {} +} \ No newline at end of file diff --git a/ODM_Handling/__init__.py b/ODM_Handling/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/overdrivedelete.sh b/ODM_Handling/overdrivedelete.sh old mode 100644 new mode 100755 similarity index 90% rename from overdrivedelete.sh rename to ODM_Handling/overdrivedelete.sh index 9de22c5..27b6f42 --- a/overdrivedelete.sh +++ b/ODM_Handling/overdrivedelete.sh @@ -5,9 +5,8 @@ #It extracts the contents of the odm into a temp directory so as to point the metadata #file to the export path of the contents of the odm and then deletes the leftover files -source config.env -folder=$queue -annex=$annex +folder=$QUEUE +annex=$ANNEX for file in $folder/*.odm do @@ -25,7 +24,7 @@ do # Check if the output contains the specific LicenseError message if echo "$output" | grep -q "1003"; then - echo "License error detected for file: $file" + echo "License error detected for file: $file. Redownload the ODM." echo "$output" # Optionally log this output to a file break # Break out of the while loop and skip this file elif [ $exit_status -eq 0 ]; then @@ -43,12 +42,12 @@ do echo "downloaded_folder: $downloaded_folder" # Move the downloaded folder to the destination - mv "$downloaded_folder" "$annex" + mv "$downloaded_folder" "$ANNEX" # Move the .odm.metadata to the downloaded folder metadata_file=$(find "$folder" -name "*.odm.metadata") if [ -n "$metadata_file" ]; then - mv "$metadata_file" "$annex/$(basename "$downloaded_folder")" + mv "$metadata_file" "$ANNEX/$(basename "$downloaded_folder")" fi else echo "No downloaded folder found for $file, potentially due to earlier license error." diff --git a/ODM_Handling/unpack_odms.py b/ODM_Handling/unpack_odms.py new file mode 100644 index 0000000..64c7711 --- /dev/null +++ b/ODM_Handling/unpack_odms.py @@ -0,0 +1,27 @@ +import sys +import os +import subprocess + +parent_dir = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +sys.path.insert(0, parent_dir) + +from XML_JSON.json_scripts import json_to_env +from XML_JSON.json_scripts import load_config + +parent_dir = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +config_path = os.path.join(parent_dir, 'config.json') + +config = load_config(config_path) + +BASH_SCRIPT_PATH = './ODM_Handling/overdrivedelete.sh' + +def run_chbrown_overdrive(bash_script_path): + subprocess.run(['chmod', '+x', bash_script_path]) #Make overdrivedelete.sh executable + subprocess.run(bash_script_path, shell=True, executable='/bin/bash', env=os.environ) + +def unpack_and_move(): + # Load the configuration from JSON into environment variables + json_to_env(config_path) + + # Run the Bash script + run_chbrown_overdrive(BASH_SCRIPT_PATH) \ No newline at end of file diff --git a/XML_JSON/json_scripts.py b/XML_JSON/json_scripts.py new file mode 100644 index 0000000..d3b8fb7 --- /dev/null +++ b/XML_JSON/json_scripts.py @@ -0,0 +1,13 @@ +import os +import json + +def load_config(file_path): + """Load the JSON config file and return a dictionary.""" + with open(file_path, 'r') as file: + return json.load(file) + +def json_to_env(json_file_path): + with open(json_file_path, 'r') as json_file: + config = json.load(json_file) + for key, value in config.items(): + os.environ[key.upper()] = str(value) \ No newline at end of file diff --git a/XML_JSON/xml_scripts.py b/XML_JSON/xml_scripts.py new file mode 100644 index 0000000..34b2bb3 --- /dev/null +++ b/XML_JSON/xml_scripts.py @@ -0,0 +1,72 @@ +import sys +import os +import shutil +import xml.etree.ElementTree as ET +import json + +parent_dir = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +sys.path.insert(0, parent_dir) + +from XML_JSON.json_scripts import load_config +from Chapters_List.extract_overdrive_chapters import extract_chapters +from Chapters_List.chapter_ms import add_ms_to_chapters +from XML_JSON.json_scripts import load_config +from Chapters_List.ms_durations import ms_to_durations + +parent_dir = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +config_path = os.path.join(parent_dir, 'config.json') + +config = load_config(config_path) + +root_dir = config['annex'] + +'''Takes the .odm.metadata file and turns it into metadata.xml''' +def rename_xml(subdir, file): + old_file = os.path.join(subdir, file) + new_file = os.path.join(subdir, "metadata.xml") + shutil.copy2(old_file, new_file) + +def odm_to_xml(subdir, files): + for file in files: + if file.endswith('.odm.metadata'): + rename_xml(subdir, file) + break + +'''Cleans the metadata''' +def extract_metadata(xml_file): + tree = ET.parse(xml_file) + root = tree.getroot() + data = {} + for child in root: + '''Finds occurrences of Title, Creators, Subjects and isolates''' + if child.tag == "Title": + data["Title"] = child.text + elif child.tag == "Creators": + '''Identifies Authors and Narrators''' + for creator in child: + if creator.attrib["role"] == "Author": + data["Author"] = creator.text + elif creator.attrib["role"] == "Narrator": + data["Narrator"] = creator.text + elif child.tag == "Subjects": + subjects = [] + for subject in child: + subjects.append(subject.text) + data["Subjects"] = ", ".join(subjects) + return data + +'''Turns the cleaned xml into a json''' +def xml_to_json(folder_path): + for dirpath, _, filenames in os.walk(folder_path): + for filename in filenames: + if filename == "metadata.xml": + xml_file = os.path.join(dirpath, filename) + metadata = extract_metadata(xml_file) + cleaned_file = os.path.join(dirpath, "cleaned_metadata.json") + with open(cleaned_file, "w") as f: + json.dump(metadata, f) + +def parse_metadata(): + for subdir, _, files in os.walk(root_dir): + odm_to_xml(subdir, files) + xml_to_json(root_dir) \ No newline at end of file diff --git a/__pycache__/config.cpython-310.pyc b/__pycache__/config.cpython-310.pyc deleted file mode 100644 index b0f138a..0000000 Binary files a/__pycache__/config.cpython-310.pyc and /dev/null differ diff --git a/chapter_ms.py b/chapter_ms.py deleted file mode 100644 index 862034c..0000000 --- a/chapter_ms.py +++ /dev/null @@ -1,35 +0,0 @@ -'''Converts label durations to milliseconds for pydub to read''' - -import os -import re -from datetime import datetime, timedelta -from config import annex - -def duration_to_milliseconds(duration): - '''Conversion from Seconds to Milliseconds''' - hours, minutes, seconds = duration.split(':') - seconds = int(hours)*3600 + int(minutes)*60 + float(seconds) - return int(seconds * 1000) - -'''path to directory containing the files''' -directory = annex - -'''loop through all files in directory tree''' -for dirpath, dirnames, filenames in os.walk(directory): - for filename in filenames: - if filename.startswith("overdrive_chapters") and filename.endswith(".txt"): - '''open the chapters file and read the contents''' - with open(os.path.join(dirpath, filename), 'r') as f: - lines = f.readlines() - - '''create a new file with the same name but with "ms" added to the end''' - new_filename = filename[:-4] + "_ms.txt" - with open(os.path.join(dirpath, new_filename), 'w') as new_file: - '''loop through each line in the file''' - for line in lines: - '''split the line into duration and label name using regular expression''' - duration, label = re.split(r"\s+", line.strip(), 1) - '''convert the duration to milliseconds''' - milliseconds = duration_to_milliseconds(duration) - '''write the converted duration and label to the new file''' - new_file.write(f'{milliseconds} {label}\n') diff --git a/config.env.template b/config.env.template deleted file mode 100644 index 1e9debc..0000000 --- a/config.env.template +++ /dev/null @@ -1,2 +0,0 @@ -queue="PATH/TO/YOUR/.ODM's" -annex="PATH/TO/YOUR/ANNEX" \ No newline at end of file diff --git a/durations.py b/durations.py deleted file mode 100644 index cc36562..0000000 --- a/durations.py +++ /dev/null @@ -1,40 +0,0 @@ -'''Takes the contents of the Milliseconds durations/labels file and creates start/end times''' - -import os -from config import annex - -root_dir = annex - -for subdir, dirs, files in os.walk(root_dir): - for file in files: - '''Locates "overdrive_chapters_ms.txt"''' - file_path = os.path.join(subdir, file) - if file == 'overdrive_chapters_ms.txt': - with open(file_path, "r") as file: - lines = file.read().split('\n') - new_file_path = file_path.replace('.txt', '_spans.txt') - with open(new_file_path, "w") as file: - for i in range(len(lines)): - parts = lines[i].split(" ") - if i == len(lines) - 2: - '''If it's the second-to-last line, write the last line's number and 0''' - file.write(parts[0]) - file.write("\t" + lines[i + 1].split(" ")[0]) - file.write("0") - elif i != len(lines) - 1: - '''Otherwise, write the number from the next line to the file''' - file.write(parts[0]) - file.write("\t" + lines[i + 1].split(" ")[0]) - else: - '''If it's the last line, just write the number and 0''' - if parts[0] != "": - file.write(parts[0]) - file.write("0") - '''Write the rest of the line to the file''' - file.write("\t" + " ".join(parts[1:])) - if i != len(lines) - 1 or (i == len(lines) - 1 and lines[i] != ""): - file.write("\n") - if lines[-1] == "": - '''Ensures there's no empty line at the end of the txt''' - file.seek(file.tell()-1, os.SEEK_SET) - file.truncate() \ No newline at end of file diff --git a/final_metadata_add.py b/final_metadata_add.py deleted file mode 100644 index abe8dbb..0000000 --- a/final_metadata_add.py +++ /dev/null @@ -1,77 +0,0 @@ -'''Adds metadata to the chapterized mp3s''' - -import os -import shutil -import json -import re -from mutagen.id3 import ID3, APIC, TIT2, TPE1, TCOM, TCON, TSOA, TRCK, TIT3, TALB, COMM -from config import circulation,bookshelf - -def traverse_directory(directory): - '''Iterate through all subfolders and files''' - for root, dirs, files in os.walk(directory): - for dir in dirs: - subfolder_path = os.path.join(root, dir) - if os.path.exists(os.path.join(subfolder_path, "cleaned_metadata.json")): - add_metadata(subfolder_path) - -def add_metadata(directory): - '''Get the metadata from the cleaned_metadata.json file in the current subfolder''' - with open(os.path.join(directory, "cleaned_metadata.json"), "r") as f: - json_data = json.load(f) - artist = json_data.get("Author", "") - composer = json_data.get("Narrator", "") - genre = json_data.get("Subjects", "") - album_title = json_data.get("Title", "") - - '''Replace all occurrences of -s with "'s" in the album title''' - album_title = re.sub("-s", "'s", album_title) - - '''Get a list of all mp3 files in the current directory''' - mp3_files = [f for f in os.listdir(directory) if f.endswith(".mp3")] - - '''Sort the list of mp3 files alphabetically''' - mp3_files.sort() - - '''Iterate through the list of mp3 files''' - for i, mp3_file in enumerate(mp3_files, 1): - mp3_file_path = os.path.join(directory, mp3_file) - - '''Extract the title of the mp3 file''' - file_name, file_ext = os.path.splitext(mp3_file) - title = file_name.lstrip("0123456789_") - - '''Add the track number to the file using ID3''' - audio = ID3(mp3_file_path) - audio.add(TRCK(encoding=3, text=str(i))) - audio.add(TIT2(encoding=3, text=title)) - audio.save() - - '''Add the metadata to the file using ID3''' - audio = ID3(mp3_file_path) - audio.add(TPE1(encoding=3, text=artist)) - audio.add(TCOM(encoding=3, text=composer)) - audio.add(TCON(encoding=3, text=genre)) - audio.add(TALB(encoding=3, text=album_title)) - audio.save() - - '''Check if folder.jpg exists in the current directory''' - if os.path.exists(os.path.join(directory, "folder.jpg")): - '''Add the album art to the file using ID3''' - with open(os.path.join(directory, "folder.jpg"), "rb") as albumart: - audio = ID3(mp3_file_path) - audio.add(APIC(encoding=3, mime='image/jpeg', type=3, desc='Cover', data=albumart.read())) - audio.save() - -'''Point the script at a directory''' -directory = circulation - -traverse_directory(directory) - -for root, dirs, files in os.walk(directory): - '''Traverse the directory and apply the metadata to the mp3 files''' - for dir in dirs: - subfolder_path = os.path.join(root, dir) - parent_folder = os.path.dirname(subfolder_path) - if parent_folder == directory: - shutil.move(subfolder_path, bookshelf) \ No newline at end of file diff --git a/main.py b/main.py new file mode 100644 index 0000000..359154f --- /dev/null +++ b/main.py @@ -0,0 +1,13 @@ +from ODM_Handling.unpack_odms import unpack_and_move +from Chapters_List.chapters import chapters_list +from Chop_Tag_Audio.chop_tag_audio import chop_tag_audio +from XML_JSON.xml_scripts import parse_metadata + +def run_script(): + unpack_and_move() + parse_metadata() + chapters_list() + chop_tag_audio() + +if __name__ == '__main__': + run_script() diff --git a/metadatatoxml.py b/metadatatoxml.py deleted file mode 100644 index e0e1000..0000000 --- a/metadatatoxml.py +++ /dev/null @@ -1,16 +0,0 @@ - -'''Takes the .odm.metadata file and turns it into metadata.xml''' - -import os -import shutil -from config import annex - -root_dir = annex - -for subdir, dirs, files in os.walk(root_dir): - for file in files: - if file.endswith('.odm.metadata'): - old_file = os.path.join(subdir, file) - new_file = os.path.join(subdir, "metadata.xml") - shutil.copy2(old_file, new_file) - break diff --git a/overdrive_full.sh b/overdrive_full.sh deleted file mode 100644 index c465a54..0000000 --- a/overdrive_full.sh +++ /dev/null @@ -1,30 +0,0 @@ -#!/bin/bash - -#start by downloading odms to chosen directory - -#configure variables -python3.10 config.py - -#unpack odms -bash overdrivedelete.sh - -#extract chapters, specify folder holding the unpackaged overdrive mp3s -source config.env -python3 extract_overdrive_chapters.py $annex - -#clean metadata -python3 metadatatoxml.py - -python3 xmlparse.py - -#turn chapters into ms -python3 chapter_ms.py - -#format durations -python3 durations.py - -#export labeled audio and archive original folder -python3 00_exportchapters.py - -#tag the exported audio with proper metadata -python3 final_metadata_add.py \ No newline at end of file diff --git a/template_config.json b/template_config.json new file mode 100644 index 0000000..d710c98 --- /dev/null +++ b/template_config.json @@ -0,0 +1,9 @@ +{ + "queue" : "/YOUR/SYSTEM/PATH/TO/ODMS", + "annex" : "/YOUR/SYSTEM/PATH/TO/01_ANNEX", + "circulation" : "/YOUR/SYSTEM/PATH/TO/02_CIRCULATION", + "bookshelf" : "/YOUR/SYSTEM/PATH/TO/03_BOOKSHELF", + "archive" : "/YOUR/SYSTEM/PATH/TO/00_ARCHIVE", + + "ffmpeg" : "typically looks like this(/opt/homebrew/Cellar/ffmpeg/5.1.2_3/bin/ffmpeg)" +} \ No newline at end of file diff --git a/template_config.py b/template_config.py deleted file mode 100644 index 0d49997..0000000 --- a/template_config.py +++ /dev/null @@ -1,17 +0,0 @@ -''' -Define directories -I like having multiple folders so as to visualize each step happening -but you could easily adjust the setup to fit your preferences -''' - -'''Where the unpackaged odm mp3s will get moved''' -annex="PATH/TO/ANNEX" - -'''Where chapterized and labeled mp3s will be stored''' -circulation="PATH/TO/CIRCULATION" - -'''Where the final finished product will land''' -bookshelf="PATH/TO/BOOKSHELF" - -'''Where the scraps go''' -archive="PATH/TO/ARCHIVE" \ No newline at end of file diff --git a/xmlparse.py b/xmlparse.py deleted file mode 100644 index e977575..0000000 --- a/xmlparse.py +++ /dev/null @@ -1,43 +0,0 @@ -'''Parses the metadata.xml file into only the required data''' - -import os -import xml.etree.ElementTree as ET -import json -from config import annex - -def extract_metadata(xml_file): - tree = ET.parse(xml_file) - root = tree.getroot() - data = {} - for child in root: - '''Finds occurrences of Title, Creators, Subjects and isolates''' - if child.tag == "Title": - data["Title"] = child.text - elif child.tag == "Creators": - '''Identifies Authors and Narrators''' - for creator in child: - if creator.attrib["role"] == "Author": - data["Author"] = creator.text - elif creator.attrib["role"] == "Narrator": - data["Narrator"] = creator.text - elif child.tag == "Subjects": - subjects = [] - for subject in child: - subjects.append(subject.text) - data["Subjects"] = ", ".join(subjects) - return data - -def process_folder(folder_path): - #Walks to metadata.xml & exports the parsed version as cleaned_metadata.json - for dirpath, dirname, filenames in os.walk(folder_path): - for filename in filenames: - if filename == "metadata.xml": - xml_file = os.path.join(dirpath, filename) - metadata = extract_metadata(xml_file) - cleaned_file = os.path.join(dirpath, "cleaned_metadata.json") - with open(cleaned_file, "w") as f: - json.dump(metadata, f) - print(f"Processed {xml_file}, output saved to {cleaned_file}") - -path_to_annex = annex -process_folder(path_to_annex) \ No newline at end of file