mirror of
https://github.com/EDeev/api_processor.git
synced 2026-10-07 20:49:34 +03:00
Переводы строк приведены к LF, удалены заметки разработчика
тесты.txt — команды curl с локальными путями Windows и ссылкой на личный файл; примеры запросов будут в README.
This commit is contained in:
parent
9dafb27c3d
commit
c004e79119
4 changed files with 101 additions and 112 deletions
2
.gitattributes
vendored
Normal file
2
.gitattributes
vendored
Normal file
|
|
@ -0,0 +1,2 @@
|
|||
* text=auto eol=lf
|
||||
*.ogg binary
|
||||
|
|
@ -1,33 +1,33 @@
|
|||
# Generated by Django 5.1.3 on 2025-05-06 12:35
|
||||
|
||||
import api_app.models
|
||||
from django.db import migrations, models
|
||||
|
||||
|
||||
class Migration(migrations.Migration):
|
||||
|
||||
initial = True
|
||||
|
||||
dependencies = [
|
||||
]
|
||||
|
||||
operations = [
|
||||
migrations.CreateModel(
|
||||
name='AudioFile',
|
||||
fields=[
|
||||
('id', models.AutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')),
|
||||
('file', models.FileField(upload_to=api_app.models.get_file_path)),
|
||||
('uploaded_at', models.DateTimeField(auto_now_add=True)),
|
||||
('processed_text', models.TextField(blank=True, null=True)),
|
||||
],
|
||||
),
|
||||
migrations.CreateModel(
|
||||
name='DocumentFile',
|
||||
fields=[
|
||||
('id', models.AutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')),
|
||||
('file', models.FileField(upload_to=api_app.models.get_file_path)),
|
||||
('uploaded_at', models.DateTimeField(auto_now_add=True)),
|
||||
('processed_text', models.TextField(blank=True, null=True)),
|
||||
],
|
||||
),
|
||||
]
|
||||
# Generated by Django 5.1.3 on 2025-05-06 12:35
|
||||
|
||||
import api_app.models
|
||||
from django.db import migrations, models
|
||||
|
||||
|
||||
class Migration(migrations.Migration):
|
||||
|
||||
initial = True
|
||||
|
||||
dependencies = [
|
||||
]
|
||||
|
||||
operations = [
|
||||
migrations.CreateModel(
|
||||
name='AudioFile',
|
||||
fields=[
|
||||
('id', models.AutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')),
|
||||
('file', models.FileField(upload_to=api_app.models.get_file_path)),
|
||||
('uploaded_at', models.DateTimeField(auto_now_add=True)),
|
||||
('processed_text', models.TextField(blank=True, null=True)),
|
||||
],
|
||||
),
|
||||
migrations.CreateModel(
|
||||
name='DocumentFile',
|
||||
fields=[
|
||||
('id', models.AutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')),
|
||||
('file', models.FileField(upload_to=api_app.models.get_file_path)),
|
||||
('uploaded_at', models.DateTimeField(auto_now_add=True)),
|
||||
('processed_text', models.TextField(blank=True, null=True)),
|
||||
],
|
||||
),
|
||||
]
|
||||
|
|
|
|||
|
|
@ -1,66 +1,66 @@
|
|||
import pdfplumber
|
||||
import docx
|
||||
import csv
|
||||
import os
|
||||
from io import StringIO, BytesIO
|
||||
from PIL import Image
|
||||
import base64
|
||||
|
||||
|
||||
def extract_text_tables(file_path: str) -> str:
|
||||
result = ""
|
||||
if file_path.endswith(".pdf"):
|
||||
with pdfplumber.open(file_path) as pdf:
|
||||
for page in pdf.pages:
|
||||
text = page.extract_text()
|
||||
if text:
|
||||
result += "<p>" + text.replace("\n", "</p><p>") + "</p>"
|
||||
|
||||
tables = page.extract_tables()
|
||||
if tables:
|
||||
for table in tables:
|
||||
csv_output = StringIO()
|
||||
csv_writer = csv.writer(csv_output)
|
||||
csv_writer.writerows(table)
|
||||
result += f"<pre>{csv_output.getvalue()}</pre>"
|
||||
|
||||
# Извлечение изображений
|
||||
if page.images:
|
||||
for img in page.images:
|
||||
img_data = img["stream"].get_data()
|
||||
encoded_img = base64.b64encode(img_data).decode("utf-8")
|
||||
result += f'<img src="data:image/png;base64,{encoded_img}"/>'
|
||||
|
||||
elif file_path.endswith(".docx"):
|
||||
doc = docx.Document(file_path)
|
||||
|
||||
text_data = []
|
||||
table_data = []
|
||||
image_data = []
|
||||
|
||||
for para in doc.paragraphs:
|
||||
if para.text.strip():
|
||||
text_data.append(f"<p>{para.text}</p>")
|
||||
|
||||
for table in doc.tables:
|
||||
csv_output = StringIO()
|
||||
csv_writer = csv.writer(csv_output)
|
||||
for row in table.rows:
|
||||
csv_writer.writerow([cell.text.strip() for cell in row.cells])
|
||||
table_data.append(f"<pre>{csv_output.getvalue()}</pre>")
|
||||
|
||||
# Извлечение изображений
|
||||
for rel in doc.part.rels:
|
||||
if "image" in doc.part.rels[rel].target_ref:
|
||||
image_data_blob = doc.part.rels[rel].target_part.blob
|
||||
encoded_img = base64.b64encode(image_data_blob).decode("utf-8")
|
||||
image_data.append(f'<img src="data:image/png;base64,{encoded_img}"/>')
|
||||
|
||||
if text_data:
|
||||
result += "".join(text_data)
|
||||
if table_data:
|
||||
result += "".join(table_data)
|
||||
if image_data:
|
||||
result += "".join(image_data)
|
||||
|
||||
return result
|
||||
import pdfplumber
|
||||
import docx
|
||||
import csv
|
||||
import os
|
||||
from io import StringIO, BytesIO
|
||||
from PIL import Image
|
||||
import base64
|
||||
|
||||
|
||||
def extract_text_tables(file_path: str) -> str:
|
||||
result = ""
|
||||
if file_path.endswith(".pdf"):
|
||||
with pdfplumber.open(file_path) as pdf:
|
||||
for page in pdf.pages:
|
||||
text = page.extract_text()
|
||||
if text:
|
||||
result += "<p>" + text.replace("\n", "</p><p>") + "</p>"
|
||||
|
||||
tables = page.extract_tables()
|
||||
if tables:
|
||||
for table in tables:
|
||||
csv_output = StringIO()
|
||||
csv_writer = csv.writer(csv_output)
|
||||
csv_writer.writerows(table)
|
||||
result += f"<pre>{csv_output.getvalue()}</pre>"
|
||||
|
||||
# Извлечение изображений
|
||||
if page.images:
|
||||
for img in page.images:
|
||||
img_data = img["stream"].get_data()
|
||||
encoded_img = base64.b64encode(img_data).decode("utf-8")
|
||||
result += f'<img src="data:image/png;base64,{encoded_img}"/>'
|
||||
|
||||
elif file_path.endswith(".docx"):
|
||||
doc = docx.Document(file_path)
|
||||
|
||||
text_data = []
|
||||
table_data = []
|
||||
image_data = []
|
||||
|
||||
for para in doc.paragraphs:
|
||||
if para.text.strip():
|
||||
text_data.append(f"<p>{para.text}</p>")
|
||||
|
||||
for table in doc.tables:
|
||||
csv_output = StringIO()
|
||||
csv_writer = csv.writer(csv_output)
|
||||
for row in table.rows:
|
||||
csv_writer.writerow([cell.text.strip() for cell in row.cells])
|
||||
table_data.append(f"<pre>{csv_output.getvalue()}</pre>")
|
||||
|
||||
# Извлечение изображений
|
||||
for rel in doc.part.rels:
|
||||
if "image" in doc.part.rels[rel].target_ref:
|
||||
image_data_blob = doc.part.rels[rel].target_part.blob
|
||||
encoded_img = base64.b64encode(image_data_blob).decode("utf-8")
|
||||
image_data.append(f'<img src="data:image/png;base64,{encoded_img}"/>')
|
||||
|
||||
if text_data:
|
||||
result += "".join(text_data)
|
||||
if table_data:
|
||||
result += "".join(table_data)
|
||||
if image_data:
|
||||
result += "".join(image_data)
|
||||
|
||||
return result
|
||||
|
|
|
|||
13
тесты.txt
13
тесты.txt
|
|
@ -1,13 +0,0 @@
|
|||
python manage.py runserver
|
||||
|
||||
------------------------------------------
|
||||
|
||||
Тест расширения для сканирования:
|
||||
|
||||
curl -X POST -F "document=@C:\Users\egord\Desktop\API EasyAccess\api_project\media\uploads\Деев Е.В. Резюме.pdf" http://localhost:8000/api/document-to-text/
|
||||
|
||||
------------------------------------------
|
||||
|
||||
Тест для аудио транскрипции:
|
||||
|
||||
curl -X POST -F audio=@"C:\Users\egord\Desktop\API EasyAccess\api_project\media\uploads\audio.ogg" http://localhost:8000/api/audio-to-text/
|
||||
Loading…
Add table
Reference in a new issue