1
0
Fork 0
mirror of https://github.com/EDeev/api_processor.git synced 2026-10-07 20:49:34 +03:00

Переводы строк приведены к LF, удалены заметки разработчика

тесты.txt — команды curl с локальными путями Windows и ссылкой на личный файл; примеры
запросов будут в README.
This commit is contained in:
Деев Егор Викторович 2026-10-06 00:09:01 +00:00
parent 9dafb27c3d
commit c004e79119
4 changed files with 101 additions and 112 deletions

2
.gitattributes vendored Normal file
View file

@ -0,0 +1,2 @@
* text=auto eol=lf
*.ogg binary

View file

@ -1,33 +1,33 @@
# Generated by Django 5.1.3 on 2025-05-06 12:35 # Generated by Django 5.1.3 on 2025-05-06 12:35
import api_app.models import api_app.models
from django.db import migrations, models from django.db import migrations, models
class Migration(migrations.Migration): class Migration(migrations.Migration):
initial = True initial = True
dependencies = [ dependencies = [
] ]
operations = [ operations = [
migrations.CreateModel( migrations.CreateModel(
name='AudioFile', name='AudioFile',
fields=[ fields=[
('id', models.AutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')), ('id', models.AutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')),
('file', models.FileField(upload_to=api_app.models.get_file_path)), ('file', models.FileField(upload_to=api_app.models.get_file_path)),
('uploaded_at', models.DateTimeField(auto_now_add=True)), ('uploaded_at', models.DateTimeField(auto_now_add=True)),
('processed_text', models.TextField(blank=True, null=True)), ('processed_text', models.TextField(blank=True, null=True)),
], ],
), ),
migrations.CreateModel( migrations.CreateModel(
name='DocumentFile', name='DocumentFile',
fields=[ fields=[
('id', models.AutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')), ('id', models.AutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')),
('file', models.FileField(upload_to=api_app.models.get_file_path)), ('file', models.FileField(upload_to=api_app.models.get_file_path)),
('uploaded_at', models.DateTimeField(auto_now_add=True)), ('uploaded_at', models.DateTimeField(auto_now_add=True)),
('processed_text', models.TextField(blank=True, null=True)), ('processed_text', models.TextField(blank=True, null=True)),
], ],
), ),
] ]

View file

@ -1,66 +1,66 @@
import pdfplumber import pdfplumber
import docx import docx
import csv import csv
import os import os
from io import StringIO, BytesIO from io import StringIO, BytesIO
from PIL import Image from PIL import Image
import base64 import base64
def extract_text_tables(file_path: str) -> str: def extract_text_tables(file_path: str) -> str:
result = "" result = ""
if file_path.endswith(".pdf"): if file_path.endswith(".pdf"):
with pdfplumber.open(file_path) as pdf: with pdfplumber.open(file_path) as pdf:
for page in pdf.pages: for page in pdf.pages:
text = page.extract_text() text = page.extract_text()
if text: if text:
result += "<p>" + text.replace("\n", "</p><p>") + "</p>" result += "<p>" + text.replace("\n", "</p><p>") + "</p>"
tables = page.extract_tables() tables = page.extract_tables()
if tables: if tables:
for table in tables: for table in tables:
csv_output = StringIO() csv_output = StringIO()
csv_writer = csv.writer(csv_output) csv_writer = csv.writer(csv_output)
csv_writer.writerows(table) csv_writer.writerows(table)
result += f"<pre>{csv_output.getvalue()}</pre>" result += f"<pre>{csv_output.getvalue()}</pre>"
# Извлечение изображений # Извлечение изображений
if page.images: if page.images:
for img in page.images: for img in page.images:
img_data = img["stream"].get_data() img_data = img["stream"].get_data()
encoded_img = base64.b64encode(img_data).decode("utf-8") encoded_img = base64.b64encode(img_data).decode("utf-8")
result += f'<img src="data:image/png;base64,{encoded_img}"/>' result += f'<img src="data:image/png;base64,{encoded_img}"/>'
elif file_path.endswith(".docx"): elif file_path.endswith(".docx"):
doc = docx.Document(file_path) doc = docx.Document(file_path)
text_data = [] text_data = []
table_data = [] table_data = []
image_data = [] image_data = []
for para in doc.paragraphs: for para in doc.paragraphs:
if para.text.strip(): if para.text.strip():
text_data.append(f"<p>{para.text}</p>") text_data.append(f"<p>{para.text}</p>")
for table in doc.tables: for table in doc.tables:
csv_output = StringIO() csv_output = StringIO()
csv_writer = csv.writer(csv_output) csv_writer = csv.writer(csv_output)
for row in table.rows: for row in table.rows:
csv_writer.writerow([cell.text.strip() for cell in row.cells]) csv_writer.writerow([cell.text.strip() for cell in row.cells])
table_data.append(f"<pre>{csv_output.getvalue()}</pre>") table_data.append(f"<pre>{csv_output.getvalue()}</pre>")
# Извлечение изображений # Извлечение изображений
for rel in doc.part.rels: for rel in doc.part.rels:
if "image" in doc.part.rels[rel].target_ref: if "image" in doc.part.rels[rel].target_ref:
image_data_blob = doc.part.rels[rel].target_part.blob image_data_blob = doc.part.rels[rel].target_part.blob
encoded_img = base64.b64encode(image_data_blob).decode("utf-8") encoded_img = base64.b64encode(image_data_blob).decode("utf-8")
image_data.append(f'<img src="data:image/png;base64,{encoded_img}"/>') image_data.append(f'<img src="data:image/png;base64,{encoded_img}"/>')
if text_data: if text_data:
result += "".join(text_data) result += "".join(text_data)
if table_data: if table_data:
result += "".join(table_data) result += "".join(table_data)
if image_data: if image_data:
result += "".join(image_data) result += "".join(image_data)
return result return result

View file

@ -1,13 +0,0 @@
python manage.py runserver
------------------------------------------
Тест расширения для сканирования:
curl -X POST -F "document=@C:\Users\egord\Desktop\API EasyAccess\api_project\media\uploads\Деев Е.В. Резюме.pdf" http://localhost:8000/api/document-to-text/
------------------------------------------
Тест для аудио транскрипции:
curl -X POST -F audio=@"C:\Users\egord\Desktop\API EasyAccess\api_project\media\uploads\audio.ogg" http://localhost:8000/api/audio-to-text/