mirror of
https://github.com/EDeev/api_processor.git
synced 2026-10-07 20:49:34 +03:00
Документы: экранирование текста, расширения в любом регистре, верный тип картинок
- текст документа вставлялся в HTML как есть — XSS у потребителя, который отрисует результат; теперь текст и таблицы экранируются; - файл с расширением .PDF или .DOCX давал пустой результат; - картинки из DOCX и PDF всегда помечались image/png; теперь — настоящий тип, а сырые потоки PDF без формата картинки пропускаются; пустые ячейки таблиц — пустые строки.
This commit is contained in:
parent
c16287063c
commit
7465fce2b7
2 changed files with 54 additions and 50 deletions
|
|
@ -4,7 +4,7 @@ import uuid
|
||||||
import os
|
import os
|
||||||
|
|
||||||
def get_file_path(instance, filename):
|
def get_file_path(instance, filename):
|
||||||
ext = filename.split('.')[-1]
|
ext = filename.split('.')[-1].lower()
|
||||||
filename = f"{uuid.uuid4()}.{ext}"
|
filename = f"{uuid.uuid4()}.{ext}"
|
||||||
return os.path.join('uploads', filename)
|
return os.path.join('uploads', filename)
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -1,66 +1,70 @@
|
||||||
import pdfplumber
|
|
||||||
import docx
|
|
||||||
import csv
|
|
||||||
import os
|
|
||||||
from io import StringIO, BytesIO
|
|
||||||
from PIL import Image
|
|
||||||
import base64
|
import base64
|
||||||
|
import csv
|
||||||
|
from html import escape
|
||||||
|
from io import StringIO
|
||||||
|
|
||||||
|
import docx
|
||||||
|
import pdfplumber
|
||||||
|
|
||||||
|
# сигнатуры форматов: картинка из PDF — это сырой поток, не обязательно PNG
|
||||||
|
IMAGE_SIGNATURES = {
|
||||||
|
b"\x89PNG\r\n\x1a\n": "image/png",
|
||||||
|
b"\xff\xd8\xff": "image/jpeg",
|
||||||
|
b"GIF8": "image/gif",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def image_mime(data: bytes):
|
||||||
|
for signature, mime in IMAGE_SIGNATURES.items():
|
||||||
|
if data.startswith(signature):
|
||||||
|
return mime
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def img_tag(data: bytes, mime: str) -> str:
|
||||||
|
return f'<img src="data:{mime};base64,{base64.b64encode(data).decode("utf-8")}"/>'
|
||||||
|
|
||||||
|
|
||||||
|
def table_to_pre(rows) -> str:
|
||||||
|
csv_output = StringIO()
|
||||||
|
csv.writer(csv_output).writerows([[cell if cell is not None else "" for cell in row] for row in rows])
|
||||||
|
return f"<pre>{escape(csv_output.getvalue())}</pre>"
|
||||||
|
|
||||||
|
|
||||||
def extract_text_tables(file_path: str) -> str:
|
def extract_text_tables(file_path: str) -> str:
|
||||||
|
"""Текст, таблицы (CSV) и изображения документа одной HTML-строкой.
|
||||||
|
Текст экранируется: иначе содержимое документа становится разметкой (XSS у потребителя)"""
|
||||||
result = ""
|
result = ""
|
||||||
if file_path.endswith(".pdf"):
|
path = file_path.lower() # раньше файл «.PDF» молча давал пустой результат
|
||||||
|
|
||||||
|
if path.endswith(".pdf"):
|
||||||
with pdfplumber.open(file_path) as pdf:
|
with pdfplumber.open(file_path) as pdf:
|
||||||
for page in pdf.pages:
|
for page in pdf.pages:
|
||||||
text = page.extract_text()
|
text = page.extract_text()
|
||||||
if text:
|
if text:
|
||||||
result += "<p>" + text.replace("\n", "</p><p>") + "</p>"
|
result += "".join(f"<p>{escape(line)}</p>" for line in text.split("\n"))
|
||||||
|
|
||||||
tables = page.extract_tables()
|
for table in page.extract_tables() or []:
|
||||||
if tables:
|
result += table_to_pre(table)
|
||||||
for table in tables:
|
|
||||||
csv_output = StringIO()
|
|
||||||
csv_writer = csv.writer(csv_output)
|
|
||||||
csv_writer.writerows(table)
|
|
||||||
result += f"<pre>{csv_output.getvalue()}</pre>"
|
|
||||||
|
|
||||||
# Извлечение изображений
|
# Извлечение изображений — только тех, что лежат в PDF готовым файлом (PNG/JPEG)
|
||||||
if page.images:
|
|
||||||
for img in page.images:
|
for img in page.images:
|
||||||
img_data = img["stream"].get_data()
|
data = img["stream"].get_data()
|
||||||
encoded_img = base64.b64encode(img_data).decode("utf-8")
|
mime = image_mime(data)
|
||||||
result += f'<img src="data:image/png;base64,{encoded_img}"/>'
|
if mime:
|
||||||
|
result += img_tag(data, mime)
|
||||||
|
|
||||||
elif file_path.endswith(".docx"):
|
elif path.endswith(".docx"):
|
||||||
doc = docx.Document(file_path)
|
doc = docx.Document(file_path)
|
||||||
|
|
||||||
text_data = []
|
result += "".join(f"<p>{escape(para.text)}</p>" for para in doc.paragraphs if para.text.strip())
|
||||||
table_data = []
|
result += "".join(table_to_pre([[cell.text.strip() for cell in row.cells] for row in table.rows])
|
||||||
image_data = []
|
for table in doc.tables)
|
||||||
|
|
||||||
for para in doc.paragraphs:
|
# Извлечение изображений с их настоящим типом
|
||||||
if para.text.strip():
|
for rel in doc.part.rels.values():
|
||||||
text_data.append(f"<p>{para.text}</p>")
|
if "image" in rel.reltype and not rel.is_external:
|
||||||
|
part = rel.target_part
|
||||||
for table in doc.tables:
|
result += img_tag(part.blob, part.content_type)
|
||||||
csv_output = StringIO()
|
|
||||||
csv_writer = csv.writer(csv_output)
|
|
||||||
for row in table.rows:
|
|
||||||
csv_writer.writerow([cell.text.strip() for cell in row.cells])
|
|
||||||
table_data.append(f"<pre>{csv_output.getvalue()}</pre>")
|
|
||||||
|
|
||||||
# Извлечение изображений
|
|
||||||
for rel in doc.part.rels:
|
|
||||||
if "image" in doc.part.rels[rel].target_ref:
|
|
||||||
image_data_blob = doc.part.rels[rel].target_part.blob
|
|
||||||
encoded_img = base64.b64encode(image_data_blob).decode("utf-8")
|
|
||||||
image_data.append(f'<img src="data:image/png;base64,{encoded_img}"/>')
|
|
||||||
|
|
||||||
if text_data:
|
|
||||||
result += "".join(text_data)
|
|
||||||
if table_data:
|
|
||||||
result += "".join(table_data)
|
|
||||||
if image_data:
|
|
||||||
result += "".join(image_data)
|
|
||||||
|
|
||||||
return result
|
return result
|
||||||
|
|
|
||||||
Loading…
Add table
Reference in a new issue