Trying to expand horizins

This commit is contained in:
2026-07-01 21:47:45 -06:00
parent 10591c24ef
commit e397f5a264
16 changed files with 439779 additions and 11 deletions
File diff suppressed because it is too large Load Diff
+62
View File
@@ -0,0 +1,62 @@
import os
import re
def clean_and_combine_gutenberg(input_folder, output_filepath):
# Standard Gutenberg start and end markers (case-insensitive regex)
start_pattern = re.compile(r"\*\*\*?\s*START OF (THE|THIS) PROJECT GUTENBERG", re.IGNORECASE)
end_pattern = re.compile(r"\*\*\*?\s*END OF (THE|THIS) PROJECT GUTENBERG", re.IGNORECASE)
books_processed = 0
with open(output_filepath, "w", encoding="utf-8") as outfile:
# Loop through all files in the target directory
for filename in os.listdir(input_folder):
if filename.endswith(".txt"):
file_path = os.path.join(input_folder, filename)
print(file_path)
with open(file_path, "r", encoding="utf-8", errors="ignore") as infile:
lines = infile.readlines()
inside_book = False
cleaned_content = []
for line in lines:
# Check for the end marker first to stop collecting
if end_pattern.search(line):
inside_book = False
break
# If we are inside the book boundaries, save the line
if inside_book:
cleaned_content.append(line)
# Check for the start marker to begin collecting on the NEXT line
if start_pattern.search(line):
inside_book = True
# If we successfully found content, write it to the master file
if cleaned_content:
# Optional: Add a couple of newlines between books so they don't smash together
outfile.write("".join(cleaned_content))
outfile.write("\n\n")
books_processed += 1
print(True)
else:
print(False)
print(f"\n All done! Processed {books_processed} books into '{output_filepath}'.")
# --- Configuration ---
# Replace 'my_gutenberg_books' with the path to your folder containing the .txt files
INPUT_DIRECTORY = "./saved_files"
OUTPUT_FILE = "combined_training_data.txt"
# Run the script
if __name__ == "__main__":
# Create a dummy folder if you haven't made one yet
if not os.path.exists(INPUT_DIRECTORY):
os.makedirs(INPUT_DIRECTORY)
print(f"Created folder '{INPUT_DIRECTORY}'. Drop your Project Gutenberg .txt files in there and re-run!")
else:
clean_and_combine_gutenberg(INPUT_DIRECTORY, OUTPUT_FILE)
File diff suppressed because it is too large Load Diff
+12246
View File
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+14872
View File
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+22448
View File
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+22314
View File
File diff suppressed because it is too large Load Diff
+11036
View File
File diff suppressed because it is too large Load Diff
+30287
View File
File diff suppressed because it is too large Load Diff
+26060
View File
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+3 -11
View File
@@ -28,8 +28,7 @@ EMBEDDING_DIM = 256
RNN_UNITS = 512
# Pipeline & Training Hyperparameters
TEXT_FILE = "saved_files/pg14314.txt"
FILE_URL = 'https://www.gutenberg.org/cache/epub/14314/pg14314.txt'
TEXT_FILE = "combined_training_data.txt"
VOCAB_FILE = "vocab.txt"
# DYNAMICALLY MANGLED DIRECTORY:
@@ -78,15 +77,8 @@ if os.path.exists(VOCAB_FILE):
vocab = json.load(f)
else:
if not os.path.exists(TEXT_FILE):
print(f"File '{TEXT_FILE}' not found locally. Downloading it.")
# Download the file\mn",
downloaded_path = tf.keras.utils.get_file('tmp', FILE_URL)
# Save the downloaded file to the designated local directory
with open(downloaded_path, 'rb') as source_file:
with open(local_path, 'wb') as dest_file:
dest_file.write(source_file.read())
print(f"File '{TEXT_FILE}' not found locally. Please run the process.py script!")
quit(1)
print(f"Loading raw text from {TEXT_FILE} to build vocabulary...")
with open(TEXT_FILE, 'r', encoding='utf-8') as f: