diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..00e54f4 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,2 @@ +* text=auto eol=lf +*.ogg binary diff --git a/api_app/migrations/0001_initial.py b/api_app/migrations/0001_initial.py index 1cca455..34fbdc0 100644 --- a/api_app/migrations/0001_initial.py +++ b/api_app/migrations/0001_initial.py @@ -1,33 +1,33 @@ -# Generated by Django 5.1.3 on 2025-05-06 12:35 - -import api_app.models -from django.db import migrations, models - - -class Migration(migrations.Migration): - - initial = True - - dependencies = [ - ] - - operations = [ - migrations.CreateModel( - name='AudioFile', - fields=[ - ('id', models.AutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')), - ('file', models.FileField(upload_to=api_app.models.get_file_path)), - ('uploaded_at', models.DateTimeField(auto_now_add=True)), - ('processed_text', models.TextField(blank=True, null=True)), - ], - ), - migrations.CreateModel( - name='DocumentFile', - fields=[ - ('id', models.AutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')), - ('file', models.FileField(upload_to=api_app.models.get_file_path)), - ('uploaded_at', models.DateTimeField(auto_now_add=True)), - ('processed_text', models.TextField(blank=True, null=True)), - ], - ), - ] +# Generated by Django 5.1.3 on 2025-05-06 12:35 + +import api_app.models +from django.db import migrations, models + + +class Migration(migrations.Migration): + + initial = True + + dependencies = [ + ] + + operations = [ + migrations.CreateModel( + name='AudioFile', + fields=[ + ('id', models.AutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')), + ('file', models.FileField(upload_to=api_app.models.get_file_path)), + ('uploaded_at', models.DateTimeField(auto_now_add=True)), + ('processed_text', models.TextField(blank=True, null=True)), + ], + ), + migrations.CreateModel( + name='DocumentFile', + fields=[ + ('id', models.AutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')), + ('file', models.FileField(upload_to=api_app.models.get_file_path)), + ('uploaded_at', models.DateTimeField(auto_now_add=True)), + ('processed_text', models.TextField(blank=True, null=True)), + ], + ), + ] diff --git a/api_app/services/scan.py b/api_app/services/scan.py index 513ed6f..6bb3b20 100644 --- a/api_app/services/scan.py +++ b/api_app/services/scan.py @@ -1,66 +1,66 @@ -import pdfplumber -import docx -import csv -import os -from io import StringIO, BytesIO -from PIL import Image -import base64 - - -def extract_text_tables(file_path: str) -> str: - result = "" - if file_path.endswith(".pdf"): - with pdfplumber.open(file_path) as pdf: - for page in pdf.pages: - text = page.extract_text() - if text: - result += "

" + text.replace("\n", "

") + "

" - - tables = page.extract_tables() - if tables: - for table in tables: - csv_output = StringIO() - csv_writer = csv.writer(csv_output) - csv_writer.writerows(table) - result += f"
{csv_output.getvalue()}
" - - # Извлечение изображений - if page.images: - for img in page.images: - img_data = img["stream"].get_data() - encoded_img = base64.b64encode(img_data).decode("utf-8") - result += f'' - - elif file_path.endswith(".docx"): - doc = docx.Document(file_path) - - text_data = [] - table_data = [] - image_data = [] - - for para in doc.paragraphs: - if para.text.strip(): - text_data.append(f"

{para.text}

") - - for table in doc.tables: - csv_output = StringIO() - csv_writer = csv.writer(csv_output) - for row in table.rows: - csv_writer.writerow([cell.text.strip() for cell in row.cells]) - table_data.append(f"
{csv_output.getvalue()}
") - - # Извлечение изображений - for rel in doc.part.rels: - if "image" in doc.part.rels[rel].target_ref: - image_data_blob = doc.part.rels[rel].target_part.blob - encoded_img = base64.b64encode(image_data_blob).decode("utf-8") - image_data.append(f'') - - if text_data: - result += "".join(text_data) - if table_data: - result += "".join(table_data) - if image_data: - result += "".join(image_data) - - return result +import pdfplumber +import docx +import csv +import os +from io import StringIO, BytesIO +from PIL import Image +import base64 + + +def extract_text_tables(file_path: str) -> str: + result = "" + if file_path.endswith(".pdf"): + with pdfplumber.open(file_path) as pdf: + for page in pdf.pages: + text = page.extract_text() + if text: + result += "

" + text.replace("\n", "

") + "

" + + tables = page.extract_tables() + if tables: + for table in tables: + csv_output = StringIO() + csv_writer = csv.writer(csv_output) + csv_writer.writerows(table) + result += f"
{csv_output.getvalue()}
" + + # Извлечение изображений + if page.images: + for img in page.images: + img_data = img["stream"].get_data() + encoded_img = base64.b64encode(img_data).decode("utf-8") + result += f'' + + elif file_path.endswith(".docx"): + doc = docx.Document(file_path) + + text_data = [] + table_data = [] + image_data = [] + + for para in doc.paragraphs: + if para.text.strip(): + text_data.append(f"

{para.text}

") + + for table in doc.tables: + csv_output = StringIO() + csv_writer = csv.writer(csv_output) + for row in table.rows: + csv_writer.writerow([cell.text.strip() for cell in row.cells]) + table_data.append(f"
{csv_output.getvalue()}
") + + # Извлечение изображений + for rel in doc.part.rels: + if "image" in doc.part.rels[rel].target_ref: + image_data_blob = doc.part.rels[rel].target_part.blob + encoded_img = base64.b64encode(image_data_blob).decode("utf-8") + image_data.append(f'') + + if text_data: + result += "".join(text_data) + if table_data: + result += "".join(table_data) + if image_data: + result += "".join(image_data) + + return result diff --git a/тесты.txt b/тесты.txt deleted file mode 100644 index c3da715..0000000 --- a/тесты.txt +++ /dev/null @@ -1,13 +0,0 @@ -python manage.py runserver - ------------------------------------------- - -Тест расширения для сканирования: - -curl -X POST -F "document=@C:\Users\egord\Desktop\API EasyAccess\api_project\media\uploads\Деев Е.В. Резюме.pdf" http://localhost:8000/api/document-to-text/ - ------------------------------------------- - -Тест для аудио транскрипции: - -curl -X POST -F audio=@"C:\Users\egord\Desktop\API EasyAccess\api_project\media\uploads\audio.ogg" http://localhost:8000/api/audio-to-text/ \ No newline at end of file