From 7465fce2b7c93f353c9244d11600d10653606475 Mon Sep 17 00:00:00 2001 From: Egor Deev Date: Tue, 6 Oct 2026 00:38:57 +0000 Subject: [PATCH] =?UTF-8?q?=D0=94=D0=BE=D0=BA=D1=83=D0=BC=D0=B5=D0=BD?= =?UTF-8?q?=D1=82=D1=8B:=20=D1=8D=D0=BA=D1=80=D0=B0=D0=BD=D0=B8=D1=80?= =?UTF-8?q?=D0=BE=D0=B2=D0=B0=D0=BD=D0=B8=D0=B5=20=D1=82=D0=B5=D0=BA=D1=81?= =?UTF-8?q?=D1=82=D0=B0,=20=D1=80=D0=B0=D1=81=D1=88=D0=B8=D1=80=D0=B5?= =?UTF-8?q?=D0=BD=D0=B8=D1=8F=20=D0=B2=20=D0=BB=D1=8E=D0=B1=D0=BE=D0=BC=20?= =?UTF-8?q?=D1=80=D0=B5=D0=B3=D0=B8=D1=81=D1=82=D1=80=D0=B5,=20=D0=B2?= =?UTF-8?q?=D0=B5=D1=80=D0=BD=D1=8B=D0=B9=20=D1=82=D0=B8=D0=BF=20=D0=BA?= =?UTF-8?q?=D0=B0=D1=80=D1=82=D0=B8=D0=BD=D0=BE=D0=BA?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - текст документа вставлялся в HTML как есть — XSS у потребителя, который отрисует результат; теперь текст и таблицы экранируются; - файл с расширением .PDF или .DOCX давал пустой результат; - картинки из DOCX и PDF всегда помечались image/png; теперь — настоящий тип, а сырые потоки PDF без формата картинки пропускаются; пустые ячейки таблиц — пустые строки. --- api_app/models.py | 2 +- api_app/services/scan.py | 102 ++++++++++++++++++++------------------- 2 files changed, 54 insertions(+), 50 deletions(-) diff --git a/api_app/models.py b/api_app/models.py index 74f9482..788abb7 100644 --- a/api_app/models.py +++ b/api_app/models.py @@ -4,7 +4,7 @@ import uuid import os def get_file_path(instance, filename): - ext = filename.split('.')[-1] + ext = filename.split('.')[-1].lower() filename = f"{uuid.uuid4()}.{ext}" return os.path.join('uploads', filename) diff --git a/api_app/services/scan.py b/api_app/services/scan.py index 6bb3b20..fcafeca 100644 --- a/api_app/services/scan.py +++ b/api_app/services/scan.py @@ -1,66 +1,70 @@ -import pdfplumber -import docx -import csv -import os -from io import StringIO, BytesIO -from PIL import Image import base64 +import csv +from html import escape +from io import StringIO + +import docx +import pdfplumber + +# сигнатуры форматов: картинка из PDF — это сырой поток, не обязательно PNG +IMAGE_SIGNATURES = { + b"\x89PNG\r\n\x1a\n": "image/png", + b"\xff\xd8\xff": "image/jpeg", + b"GIF8": "image/gif", +} + + +def image_mime(data: bytes): + for signature, mime in IMAGE_SIGNATURES.items(): + if data.startswith(signature): + return mime + return None + + +def img_tag(data: bytes, mime: str) -> str: + return f'' + + +def table_to_pre(rows) -> str: + csv_output = StringIO() + csv.writer(csv_output).writerows([[cell if cell is not None else "" for cell in row] for row in rows]) + return f"
{escape(csv_output.getvalue())}
" def extract_text_tables(file_path: str) -> str: + """Текст, таблицы (CSV) и изображения документа одной HTML-строкой. + Текст экранируется: иначе содержимое документа становится разметкой (XSS у потребителя)""" result = "" - if file_path.endswith(".pdf"): + path = file_path.lower() # раньше файл «.PDF» молча давал пустой результат + + if path.endswith(".pdf"): with pdfplumber.open(file_path) as pdf: for page in pdf.pages: text = page.extract_text() if text: - result += "

" + text.replace("\n", "

") + "

" + result += "".join(f"

{escape(line)}

" for line in text.split("\n")) - tables = page.extract_tables() - if tables: - for table in tables: - csv_output = StringIO() - csv_writer = csv.writer(csv_output) - csv_writer.writerows(table) - result += f"
{csv_output.getvalue()}
" + for table in page.extract_tables() or []: + result += table_to_pre(table) - # Извлечение изображений - if page.images: - for img in page.images: - img_data = img["stream"].get_data() - encoded_img = base64.b64encode(img_data).decode("utf-8") - result += f'' + # Извлечение изображений — только тех, что лежат в PDF готовым файлом (PNG/JPEG) + for img in page.images: + data = img["stream"].get_data() + mime = image_mime(data) + if mime: + result += img_tag(data, mime) - elif file_path.endswith(".docx"): + elif path.endswith(".docx"): doc = docx.Document(file_path) - text_data = [] - table_data = [] - image_data = [] + result += "".join(f"

{escape(para.text)}

" for para in doc.paragraphs if para.text.strip()) + result += "".join(table_to_pre([[cell.text.strip() for cell in row.cells] for row in table.rows]) + for table in doc.tables) - for para in doc.paragraphs: - if para.text.strip(): - text_data.append(f"

{para.text}

") - - for table in doc.tables: - csv_output = StringIO() - csv_writer = csv.writer(csv_output) - for row in table.rows: - csv_writer.writerow([cell.text.strip() for cell in row.cells]) - table_data.append(f"
{csv_output.getvalue()}
") - - # Извлечение изображений - for rel in doc.part.rels: - if "image" in doc.part.rels[rel].target_ref: - image_data_blob = doc.part.rels[rel].target_part.blob - encoded_img = base64.b64encode(image_data_blob).decode("utf-8") - image_data.append(f'') - - if text_data: - result += "".join(text_data) - if table_data: - result += "".join(table_data) - if image_data: - result += "".join(image_data) + # Извлечение изображений с их настоящим типом + for rel in doc.part.rels.values(): + if "image" in rel.reltype and not rel.is_external: + part = rel.target_part + result += img_tag(part.blob, part.content_type) return result