Trying to expand horizins
This commit is contained in:
+229667
File diff suppressed because it is too large
Load Diff
+62
@@ -0,0 +1,62 @@
|
||||
import os
|
||||
import re
|
||||
|
||||
def clean_and_combine_gutenberg(input_folder, output_filepath):
|
||||
# Standard Gutenberg start and end markers (case-insensitive regex)
|
||||
start_pattern = re.compile(r"\*\*\*?\s*START OF (THE|THIS) PROJECT GUTENBERG", re.IGNORECASE)
|
||||
end_pattern = re.compile(r"\*\*\*?\s*END OF (THE|THIS) PROJECT GUTENBERG", re.IGNORECASE)
|
||||
|
||||
books_processed = 0
|
||||
|
||||
with open(output_filepath, "w", encoding="utf-8") as outfile:
|
||||
# Loop through all files in the target directory
|
||||
for filename in os.listdir(input_folder):
|
||||
if filename.endswith(".txt"):
|
||||
file_path = os.path.join(input_folder, filename)
|
||||
print(file_path)
|
||||
|
||||
with open(file_path, "r", encoding="utf-8", errors="ignore") as infile:
|
||||
lines = infile.readlines()
|
||||
|
||||
inside_book = False
|
||||
cleaned_content = []
|
||||
|
||||
for line in lines:
|
||||
# Check for the end marker first to stop collecting
|
||||
if end_pattern.search(line):
|
||||
inside_book = False
|
||||
break
|
||||
|
||||
# If we are inside the book boundaries, save the line
|
||||
if inside_book:
|
||||
cleaned_content.append(line)
|
||||
|
||||
# Check for the start marker to begin collecting on the NEXT line
|
||||
if start_pattern.search(line):
|
||||
inside_book = True
|
||||
|
||||
# If we successfully found content, write it to the master file
|
||||
if cleaned_content:
|
||||
# Optional: Add a couple of newlines between books so they don't smash together
|
||||
outfile.write("".join(cleaned_content))
|
||||
outfile.write("\n\n")
|
||||
books_processed += 1
|
||||
print(True)
|
||||
else:
|
||||
print(False)
|
||||
|
||||
print(f"\n All done! Processed {books_processed} books into '{output_filepath}'.")
|
||||
|
||||
# --- Configuration ---
|
||||
# Replace 'my_gutenberg_books' with the path to your folder containing the .txt files
|
||||
INPUT_DIRECTORY = "./saved_files"
|
||||
OUTPUT_FILE = "combined_training_data.txt"
|
||||
|
||||
# Run the script
|
||||
if __name__ == "__main__":
|
||||
# Create a dummy folder if you haven't made one yet
|
||||
if not os.path.exists(INPUT_DIRECTORY):
|
||||
os.makedirs(INPUT_DIRECTORY)
|
||||
print(f"Created folder '{INPUT_DIRECTORY}'. Drop your Project Gutenberg .txt files in there and re-run!")
|
||||
else:
|
||||
clean_and_combine_gutenberg(INPUT_DIRECTORY, OUTPUT_FILE)
|
||||
File diff suppressed because it is too large
Load Diff
+12246
File diff suppressed because it is too large
Load Diff
+32794
File diff suppressed because it is too large
Load Diff
+14872
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
+22448
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
+22314
File diff suppressed because it is too large
Load Diff
+11036
File diff suppressed because it is too large
Load Diff
+30287
File diff suppressed because it is too large
Load Diff
+26060
File diff suppressed because it is too large
Load Diff
+11866
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -28,8 +28,7 @@ EMBEDDING_DIM = 256
|
||||
RNN_UNITS = 512
|
||||
|
||||
# Pipeline & Training Hyperparameters
|
||||
TEXT_FILE = "saved_files/pg14314.txt"
|
||||
FILE_URL = 'https://www.gutenberg.org/cache/epub/14314/pg14314.txt'
|
||||
TEXT_FILE = "combined_training_data.txt"
|
||||
VOCAB_FILE = "vocab.txt"
|
||||
|
||||
# DYNAMICALLY MANGLED DIRECTORY:
|
||||
@@ -78,15 +77,8 @@ if os.path.exists(VOCAB_FILE):
|
||||
vocab = json.load(f)
|
||||
else:
|
||||
if not os.path.exists(TEXT_FILE):
|
||||
print(f"File '{TEXT_FILE}' not found locally. Downloading it.")
|
||||
# Download the file\mn",
|
||||
downloaded_path = tf.keras.utils.get_file('tmp', FILE_URL)
|
||||
|
||||
# Save the downloaded file to the designated local directory
|
||||
with open(downloaded_path, 'rb') as source_file:
|
||||
with open(local_path, 'wb') as dest_file:
|
||||
dest_file.write(source_file.read())
|
||||
|
||||
print(f"File '{TEXT_FILE}' not found locally. Please run the process.py script!")
|
||||
quit(1)
|
||||
|
||||
print(f"Loading raw text from {TEXT_FILE} to build vocabulary...")
|
||||
with open(TEXT_FILE, 'r', encoding='utf-8') as f:
|
||||
|
||||
Reference in New Issue
Block a user