From b7b79528044abc39b47897b2dbd4e1bee8de8c8a Mon Sep 17 00:00:00 2001 From: harshita-singh12 Date: Thu, 27 Aug 2026 17:43:38 +0000 Subject: [PATCH 1/5] fix(Generator): save uploads under a server-generated filename The upload path was built by joining the client-supplied filename with the upload folder, so a crafted name like ../../app.py could escape uploads/ and let file.save() overwrite arbitrary process-writable files (and os.remove() delete them afterwards). Derive the on-disk name from uuid4() on the server and keep only the extension, which already drives the text-extraction dispatch. File extensions outside the supported .txt/.pdf/.docx set are rejected before anything touches the disk, matching the previous behavior of returning empty content for unsupported types. --- backend/Generator/main.py | 17 +++++++++++++---- 1 file changed, 13 insertions(+), 4 deletions(-) diff --git a/backend/Generator/main.py b/backend/Generator/main.py index 04aed79f..08863b9c 100644 --- a/backend/Generator/main.py +++ b/backend/Generator/main.py @@ -1,4 +1,5 @@ import time +import uuid import torch import random from transformers import T5ForConditionalGeneration, T5Tokenizer @@ -368,16 +369,24 @@ def extract_text_from_docx(self, file_path): return result.value def process_file(self, file): - file_path = os.path.join(self.upload_folder, file.filename) + # The client-supplied filename is untrusted: joining it directly with + # the upload folder allows path traversal (e.g. '../../app.py') and + # collisions. Store the upload under a server-generated name instead, + # keeping only the extension, which also drives text extraction below. + extension = os.path.splitext(file.filename)[1] + if extension not in ('.txt', '.pdf', '.docx'): + return "" + + file_path = os.path.join(self.upload_folder, f"{uuid.uuid4().hex}{extension}") file.save(file_path) content = "" - if file.filename.endswith('.txt'): + if extension == '.txt': with open(file_path, 'r') as f: content = f.read() - elif file.filename.endswith('.pdf'): + elif extension == '.pdf': content = self.extract_text_from_pdf(file_path) - elif file.filename.endswith('.docx'): + elif extension == '.docx': content = self.extract_text_from_docx(file_path) os.remove(file_path) From 403b24cf95fc2ae9bdd65bc88609303666fd0095 Mon Sep 17 00:00:00 2001 From: harshita-singh12 Date: Fri, 28 Aug 2026 09:46:38 +0000 Subject: [PATCH 2/5] docs(Generator): add docstring to FileProcessor.process_file --- backend/Generator/main.py | 20 ++++++++++++++++---- 1 file changed, 16 insertions(+), 4 deletions(-) diff --git a/backend/Generator/main.py b/backend/Generator/main.py index 08863b9c..0ec52687 100644 --- a/backend/Generator/main.py +++ b/backend/Generator/main.py @@ -369,10 +369,22 @@ def extract_text_from_docx(self, file_path): return result.value def process_file(self, file): - # The client-supplied filename is untrusted: joining it directly with - # the upload folder allows path traversal (e.g. '../../app.py') and - # collisions. Store the upload under a server-generated name instead, - # keeping only the extension, which also drives text extraction below. + """Save an uploaded file under a server-generated name and return its extracted text. + + The client-supplied filename is untrusted: joining it directly with the upload folder + would allow path traversal (e.g. '../../app.py') and name collisions. The upload is + therefore stored as , keeping only the extension, which also drives + the text-extraction dispatch below. + + Args: + file: An upload object with a ``filename`` attribute and a werkzeug + ``FileStorage``-style ``save(path)`` method. + + Returns: + The extracted text content, or an empty string when the extension is not one of + the supported .txt/.pdf/.docx types; nothing is written to disk in that case and + the /upload route responds 400. + """ extension = os.path.splitext(file.filename)[1] if extension not in ('.txt', '.pdf', '.docx'): return "" From e766319598e2cb20b768b2ff093425d7531bf4d8 Mon Sep 17 00:00:00 2001 From: harshita-singh12 Date: Fri, 28 Aug 2026 09:47:59 +0000 Subject: [PATCH 3/5] docs(Generator): add docstring to FileProcessor.extract_text_from_docx --- backend/Generator/main.py | 1 + 1 file changed, 1 insertion(+) diff --git a/backend/Generator/main.py b/backend/Generator/main.py index 0ec52687..192d7293 100644 --- a/backend/Generator/main.py +++ b/backend/Generator/main.py @@ -364,6 +364,7 @@ def extract_text_from_pdf(self, file_path): return text def extract_text_from_docx(self, file_path): + """Return the raw text content of a .docx file at ``file_path``.""" with open(file_path, "rb") as docx_file: result = mammoth.extract_raw_text(docx_file) return result.value From 9c68fcdfdf960fc69578a18699f0f1b0a08f7bd9 Mon Sep 17 00:00:00 2001 From: harshita-singh12 Date: Sat, 29 Aug 2026 17:07:42 +0000 Subject: [PATCH 4/5] docs(Generator): add docstring to FileProcessor.extract_text_from_pdf --- backend/Generator/main.py | 1 + 1 file changed, 1 insertion(+) diff --git a/backend/Generator/main.py b/backend/Generator/main.py index 192d7293..8d1ac10a 100644 --- a/backend/Generator/main.py +++ b/backend/Generator/main.py @@ -357,6 +357,7 @@ def __init__(self, upload_folder='uploads/'): os.makedirs(self.upload_folder) def extract_text_from_pdf(self, file_path): + """Returns the concatenated text content of a PDF file at ``file_path``.""" doc = fitz.open(file_path) text = "" for page in doc: From 3759c926526fa7608309d864e86109450ccee10f Mon Sep 17 00:00:00 2001 From: harshita-singh12 Date: Fri, 18 Sep 2026 14:04:43 +0000 Subject: [PATCH 5/5] docs(Generator): add docstring to FileProcessor.__init__ --- backend/Generator/main.py | 1 + 1 file changed, 1 insertion(+) diff --git a/backend/Generator/main.py b/backend/Generator/main.py index 8d1ac10a..87a71b2c 100644 --- a/backend/Generator/main.py +++ b/backend/Generator/main.py @@ -352,6 +352,7 @@ def get_document_content(self, document_url): class FileProcessor: def __init__(self, upload_folder='uploads/'): + """Creates the upload folder if it does not already exist.""" self.upload_folder = upload_folder if not os.path.exists(self.upload_folder): os.makedirs(self.upload_folder)