diff --git a/.dockerignore b/.dockerignore new file mode 100644 index 0000000..a90d571 --- /dev/null +++ b/.dockerignore @@ -0,0 +1,7 @@ +.git +.env +media +models +db.sqlite3 +__pycache__ +tests diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..2249c11 --- /dev/null +++ b/.env.example @@ -0,0 +1,9 @@ +# Обязательно в продакшене +DJANGO_SECRET_KEY=change-me-to-a-long-random-string +DJANGO_ALLOWED_HOSTS=localhost,127.0.0.1 +# DJANGO_DEBUG=True # для разработки + +# Необязательно +API_TOKEN= # если задан — запросы с заголовком Authorization: Bearer <токен> +GRPC_SERVER= # адрес сервиса TextProcessor, например textproc:50051; пусто — не отправлять +MAX_UPLOAD_SIZE_MB=50 diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..a531628 --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,35 @@ +name: CI + +on: + push: + branches: [main] + pull_request: + +jobs: + test: + runs-on: ubuntu-latest + env: + VOSK_MODEL_PATH: ${{ github.workspace }}/models/vosk-model-small-ru-0.22 + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: "3.12" + - run: sudo apt-get update -qq && sudo apt-get install -y -qq ffmpeg + - name: Модель Vosk (кэш) + id: model + uses: actions/cache@v4 + with: + path: models + key: vosk-model-small-ru-0.22 + - if: steps.model.outputs.cache-hit != 'true' + run: | + mkdir -p models + curl -fsSL -o /tmp/model.zip https://alphacephei.com/vosk/models/vosk-model-small-ru-0.22.zip + unzip -q /tmp/model.zip -d models + - run: pip install -r requirements-dev.txt + - run: ruff check --select E9,F,B --exclude proto . + - run: python manage.py check && python manage.py makemigrations --check --dry-run + env: + DJANGO_DEBUG: "True" + - run: pytest -q diff --git a/.github/workflows/docker.yml b/.github/workflows/docker.yml new file mode 100644 index 0000000..e748cdf --- /dev/null +++ b/.github/workflows/docker.yml @@ -0,0 +1,42 @@ +name: Docker + +on: + push: + tags: ["v*"] + workflow_dispatch: + +jobs: + image: + runs-on: ubuntu-latest + permissions: + contents: read + packages: write + steps: + - uses: actions/checkout@v4 + - uses: docker/setup-buildx-action@v3 + - uses: docker/login-action@v3 + with: + registry: ghcr.io + username: ${{ github.actor }} + password: ${{ secrets.GITHUB_TOKEN }} + - uses: docker/login-action@v3 + with: + registry: dcr.deev.su + username: ${{ secrets.ZOT_USERNAME }} + password: ${{ secrets.ZOT_PASSWORD }} + - id: meta + uses: docker/metadata-action@v5 + with: + images: | + ghcr.io/edeev/api_processor + dcr.deev.su/edeev/api_processor + tags: | + type=semver,pattern={{version}} + type=semver,pattern={{major}}.{{minor}} + type=raw,value=latest + - uses: docker/build-push-action@v6 + with: + context: . + push: true + tags: ${{ steps.meta.outputs.tags }} + labels: ${{ steps.meta.outputs.labels }} diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..be5918a --- /dev/null +++ b/.gitignore @@ -0,0 +1,8 @@ +.env +__pycache__/ +*.pyc +.pytest_cache/ +db.sqlite3 +media/ +staticfiles/ +models/ diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000..df1bd25 --- /dev/null +++ b/Dockerfile @@ -0,0 +1,33 @@ +FROM python:3.12-slim + +ENV PYTHONDONTWRITEBYTECODE=1 \ + PYTHONUNBUFFERED=1 \ + DJANGO_DB_PATH=/data/db.sqlite3 \ + DJANGO_MEDIA_ROOT=/data/media \ + VOSK_MODEL_PATH=/app/models/vosk-model-small-ru-0.22 + +RUN apt-get update \ + && apt-get install -y --no-install-recommends ffmpeg curl unzip \ + && rm -rf /var/lib/apt/lists/* + +WORKDIR /app +# офлайн-модель распознавания русской речи (~45 МБ) +RUN mkdir -p models \ + && curl -fsSL -o /tmp/model.zip https://alphacephei.com/vosk/models/vosk-model-small-ru-0.22.zip \ + && unzip -q /tmp/model.zip -d models \ + && rm /tmp/model.zip + +COPY requirements.txt . +RUN pip install --no-cache-dir -r requirements.txt + +COPY manage.py ./ +COPY api_project/ api_project/ +COPY api_app/ api_app/ +COPY proto/ proto/ + +RUN useradd --create-home --uid 1000 app && mkdir -p /data && chown -R app:app /app /data +USER app +VOLUME ["/data"] +EXPOSE 8000 + +CMD ["sh", "-c", "python manage.py migrate --noinput && gunicorn api_project.wsgi:application --bind 0.0.0.0:8000 --workers 2 --timeout 300"] diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000..36fc243 --- /dev/null +++ b/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2025 Egor Deev + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/README.en.md b/README.en.md new file mode 100644 index 0000000..e320802 --- /dev/null +++ b/README.en.md @@ -0,0 +1,129 @@ +# API Processor + +[Русский](README.md) · **English** + +[](https://github.com/EDeev/api_processor/actions/workflows/ci.yml) +[](https://github.com/EDeev/api_processor/actions/workflows/docker.yml) +[](LICENSE) + +A REST API that turns audio and documents into text: +- speech is recognized offline with a Vosk model; +- text, tables and images are extracted from PDF and DOCX. + +The result can be forwarded over gRPC to a text processing service. Speech recognition targets Russian. + +**Status:** personal project, completed + +**Stack:** Python 3.12 · Django 5.2 · Django REST Framework · Vosk · FFmpeg · pdfplumber · python-docx · gRPC · Docker + +## Features + +- **Audio → text.** Any format FFmpeg reads (OGG, MP3, M4A, WAV…). Audio is converted to 16 kHz mono with + noise reduction and loudness normalization first. Recognition is offline, no external services. +- **Document → HTML.** PDF and DOCX: paragraphs in `
`, tables as CSV in `
`, images as base64. The
+ document text is escaped, so the result is safe to render in a browser.
+- **gRPC.** The text is sent to the `TextProcessor` service (`proto/text_service.proto`), and its reply is
+ returned in `grpc_response`. If no service address is set, the step is skipped.
+- Optional access token and a file size limit (50 MB by default).
+
+## API
+
+```bash
+curl -F audio=@examples/sample.ogg http://localhost:8000/api/audio-to-text/
+curl -F document=@report.pdf http://localhost:8000/api/document-to-text/
+```
+
+```json
+{
+ "text": "раз два три проверка перевода голоса текст насколько качественно она работает один два три четыре пять шесть семь восемь девять десять",
+ "grpc_response": null
+}
+```
+
+With `API_TOKEN` set, add `Authorization: Bearer `. Errors come as `{"error": "..."}`:
+
+| Code | When |
+|---|---|
+| 400 | no file or unsupported document format |
+| 403 | wrong token |
+| 413 | file over the limit |
+| 422 | audio could not be read |
+
+## Running
+
+```bash
+git clone https://github.com/EDeev/api_processor.git && cd api_processor
+cp .env.example .env # set DJANGO_SECRET_KEY
+docker compose up -d # API at http://localhost:8000
+```
+
+The Vosk model and FFmpeg are inside the image. Prebuilt image: `docker pull ghcr.io/edeev/api_processor` or
+`docker pull dcr.deev.su/edeev/api_processor`.
+
+Without Docker you need:
+- FFmpeg in `PATH`;
+- the [vosk-model-small-ru-0.22](https://alphacephei.com/vosk/models) model unpacked into `models/`.
+
+Then:
+
+```bash
+pip install -r requirements.txt
+DJANGO_DEBUG=True python manage.py migrate
+DJANGO_DEBUG=True python manage.py runserver
+```
+
+| Variable | Purpose |
+|---|---|
+| `DJANGO_SECRET_KEY` | secret key, required unless `DJANGO_DEBUG=True` |
+| `DJANGO_ALLOWED_HOSTS` | comma-separated domains |
+| `API_TOKEN` | if set, access requires the token |
+| `GRPC_SERVER`, `GRPC_TIMEOUT` | TextProcessor address and timeout, s |
+| `MAX_UPLOAD_SIZE_MB` | file size limit |
+| `VOSK_MODEL_PATH`, `FFMPEG_BINARY` | paths to the model and ffmpeg |
+
+## How it works
+
+```mermaid
+flowchart LR
+ C[Client] -->|multipart| V[DRF APIView]
+ V -->|audio| F[FFmpeg: 16 kHz mono, denoise] --> K[Vosk]
+ V -->|PDF / DOCX| S[pdfplumber / python-docx]
+ K --> T[text]
+ S --> T
+ T -->|gRPC ProcessText| G[TextProcessor]
+ T --> R[JSON response]
+```
+
+The Vosk model is loaded once per process. Each request gets its own temporary file. Uploaded files and
+results are stored in SQLite and `media/`.
+
+## Development
+
+```bash
+pip install -r requirements-dev.txt
+ruff check --select E9,F,B --exclude proto . && pytest
+```
+
+What the tests cover:
+- recognition of the whole sample, with every phrase;
+- OGG conversion;
+- PDF and DOCX extraction with escaping;
+- limits and the token;
+- a real gRPC server and an unavailable one.
+
+The Docker image is built on `v*` tags and published to GitHub Packages and `dcr.deev.su`.
+
+## License
+
+MIT — see [LICENSE](LICENSE). The Vosk model is licensed under Apache 2.0.
+
+## Author
+
+**Egor Deev** — [GitHub](https://github.com/EDeev) · [Telegram](https://t.me/DeevEgor) · [egor@deev.space](mailto:egor@deev.space)
+
+---
+
+
+ ⭐ If you find this project useful, give it a star on GitHub!
+ Made with ❤️ — deev.space
+
diff --git a/README.md b/README.md
index 1a5cc17..f4a7bc8 100644
--- a/README.md
+++ b/README.md
@@ -1,171 +1,142 @@
# API Processor
-**Django REST API для обработки аудио и документов с интеграцией gRPC сервисов**
+**Русский** · [English](README.en.md)
-API Processor — это мощное решение для автоматической обработки мультимедийного контента. Система выполняет транскрибацию аудиофайлов в текст и извлечение данных из PDF/DOCX документов с последующей отправкой результатов на внешний gRPC сервер для дополнительной обработки.
+[](https://github.com/EDeev/api_processor/actions/workflows/ci.yml)
+[](https://github.com/EDeev/api_processor/actions/workflows/docker.yml)
+[](LICENSE)
-## 🚀 Возможности
+REST API, которое превращает аудио и документы в текст:
+- речь распознаётся офлайн моделью Vosk;
+- из PDF и DOCX извлекаются текст, таблицы и изображения.
-- **Аудио транскрипция**: Преобразование аудиофайлов в текст с использованием модели Vosk
-- **Обработка документов**: Извлечение текста, таблиц и изображений из PDF и DOCX файлов
-- **gRPC интеграция**: Автоматическая отправка обработанного текста на внешний сервер
-- **RESTful API**: Простой и понятный интерфейс для взаимодействия
-- **Поддержка форматов**: Audio (WAV, OGG и др.), PDF, DOCX
+Результат можно отправить дальше по gRPC — на сервис обработки текста.
-## 🛠 Технологический стек
+**Статус:** личный проект, завершён
-### Backend
-- **Django 4.2.6** - веб-фреймворк
-- **Django REST Framework 3.14.0** - API framework
-- **Python** - основной язык разработки
+**Стек:** Python 3.12 · Django 5.2 · Django REST Framework · Vosk · FFmpeg · pdfplumber · python-docx · gRPC · Docker
-### Обработка контента
-- **Vosk 0.3.45** - распознавание речи
-- **FFmpeg Python 0.2.0** - конвертация аудио
-- **pdfplumber 0.10.2** - извлечение данных из PDF
-- **python-docx 0.8.11** - работа с DOCX файлами
-- **Pillow 10.0.1** - обработка изображений
+## Возможности
-### Коммуникация
-- **gRPC 1.58.0** - межсервисное взаимодействие
-- **Protocol Buffers** - сериализация данных
+- **Аудио → текст.** Любой формат, который читает FFmpeg (OGG, MP3, M4A, WAV…). Перед распознаванием
+ звук приводится к 16 кГц моно с шумоподавлением и нормализацией громкости. Распознавание офлайн, без
+ внешних сервисов.
+- **Документ → HTML.** PDF и DOCX: абзацы — в ``, таблицы — CSV в `
`, изображения — в base64.
+ Текст документа экранируется, поэтому результат безопасно показывать в браузере.
+- **gRPC.** Распознанный текст уходит на сервис `TextProcessor` (`proto/text_service.proto`), его ответ
+ возвращается в поле `grpc_response`. Если адрес сервиса не задан, шаг пропускается.
+- Необязательный токен доступа и ограничение размера файла (по умолчанию 50 МБ).
-### База данных
-- **SQLite** - локальное хранение метаданных файлов
+## API
-## 📋 Требования
+```bash
+curl -F audio=@examples/sample.ogg http://localhost:8000/api/audio-to-text/
+curl -F document=@report.pdf http://localhost:8000/api/document-to-text/
+```
-- Python 3.8+
-- FFmpeg (для конвертации аудио)
-- Модель Vosk для русского языка
+```json
+{
+ "text": "раз два три проверка перевода голоса текст насколько качественно она работает один два три четыре пять шесть семь восемь девять десять",
+ "grpc_response": null
+}
+```
-## ⚡ Быстрый старт
+Если задан `API_TOKEN`, добавьте заголовок `Authorization: Bearer <токен>`. Ошибки приходят как
+`{"error": "..."}`:
-### Установка зависимостей
+| Код | Когда |
+|---|---|
+| 400 | нет файла или неподдерживаемый формат документа |
+| 403 | неверный токен |
+| 413 | файл больше лимита |
+| 422 | аудио не удалось прочитать |
+
+## Запуск
+
+```bash
+git clone https://github.com/EDeev/api_processor.git && cd api_processor
+cp .env.example .env # задайте DJANGO_SECRET_KEY
+docker compose up -d # API на http://localhost:8000
+```
+
+Модель Vosk и FFmpeg уже внутри образа. Готовый образ: `docker pull ghcr.io/edeev/api_processor` или
+`docker pull dcr.deev.su/edeev/api_processor`.
+
+Без Docker нужны:
+- FFmpeg в `PATH`;
+- модель [vosk-model-small-ru-0.22](https://alphacephei.com/vosk/models), распакованная в `models/`.
+
+Затем:
```bash
pip install -r requirements.txt
+DJANGO_DEBUG=True python manage.py migrate
+DJANGO_DEBUG=True python manage.py runserver
```
-### Настройка модели Vosk
+| Переменная | Назначение |
+|---|---|
+| `DJANGO_SECRET_KEY` | секретный ключ, обязателен без `DJANGO_DEBUG=True` |
+| `DJANGO_ALLOWED_HOSTS` | домены через запятую |
+| `API_TOKEN` | если задан — доступ только с токеном |
+| `GRPC_SERVER`, `GRPC_TIMEOUT` | адрес сервиса TextProcessor и таймаут, с |
+| `MAX_UPLOAD_SIZE_MB` | лимит размера файла |
+| `VOSK_MODEL_PATH`, `FFMPEG_BINARY` | путь к модели и к ffmpeg |
-1. Скачайте модель `vosk-model-small-ru-0.22`
-2. Разместите в папке `models/vosk-model-small-ru-0.22`
+## Как устроено
-### Настройка FFmpeg
+```mermaid
+flowchart LR
+ C[Клиент] -->|multipart| V[DRF APIView]
+ V -->|аудио| F[FFmpeg: 16 кГц моно, шумоподавление] --> K[Vosk]
+ V -->|PDF / DOCX| S[pdfplumber / python-docx]
+ K --> T[текст]
+ S --> T
+ T -->|gRPC ProcessText| G[TextProcessor]
+ T --> R[JSON-ответ]
+```
-1. Скачайте FFmpeg
-2. Разместите в `models/ffmpeg/bin/ffmpeg.exe`
+```
+api_app/views.py эндпоинты и проверки
+api_app/services/vosk_recognizer.py конвертация FFmpeg и распознавание Vosk
+api_app/services/scan.py извлечение из PDF и DOCX
+api_app/grpc_client/client.py клиент TextProcessor
+proto/ описание gRPC-сервиса и сгенерированный код
+```
-### Запуск сервера
+Модель Vosk загружается один раз на процесс. У каждого запроса свой временный файл. Загруженные файлы и
+результат сохраняются в SQLite и `media/`.
+
+## Разработка
```bash
-python manage.py migrate
-python manage.py runserver
+pip install -r requirements-dev.txt
+ruff check --select E9,F,B --exclude proto . && pytest
```
-## 📖 API Endpoints
+Что проверяют тесты:
+- распознавание образца целиком, со всеми фразами;
+- конвертацию OGG;
+- извлечение из PDF и DOCX с экранированием;
+- лимиты и токен;
+- работу с настоящим gRPC-сервером и его недоступность.
-### Транскрипция аудио
-```http
-POST /api/audio-to-text/
-Content-Type: multipart/form-data
+Docker-образ собирается по тегу `v*` и публикуется в GitHub Packages и `dcr.deev.su`.
-audio:
-```
+Код gRPC пересобирается так:
+`python -m grpc_tools.protoc -Iproto --python_out=proto --grpc_python_out=proto proto/text_service.proto`.
-**Ответ:**
-```json
-{
- "text": "Распознанный текст из аудио",
- "grpc_response": {
- "processed_text": "Обработанный текст",
- "success": true,
- "error": null
- }
-}
-```
+## Лицензия
-### Обработка документов
-```http
-POST /api/document-to-text/
-Content-Type: multipart/form-data
+MIT — см. [LICENSE](LICENSE). Модель Vosk распространяется под Apache 2.0.
-document:
-```
+## Автор
-**Ответ:**
-```json
-{
- "text": "Извлеченный текст
таблица,данные
",
- "grpc_response": {
- "processed_text": "Обработанный текст",
- "success": true,
- "error": null
- }
-}
-```
-
-## 🧪 Примеры использования
-
-### cURL команды
-
-**Транскрипция аудио:**
-```bash
-curl -X POST -F "audio=@audio.ogg" http://localhost:8000/api/audio-to-text/
-```
-
-**Обработка документа:**
-```bash
-curl -X POST -F "document=@document.pdf" http://localhost:8000/api/document-to-text/
-```
-
-## ⚙️ Конфигурация
-
-### gRPC настройки
-По умолчанию система подключается к gRPC серверу на `localhost:50051`. Для изменения адреса отредактируйте `api_app/grpc_client/client.py`.
-
-### Модели и пути
-Пути к моделям и исполняемым файлам настраиваются в `api_app/services/vosk_recognizer.py`:
-- `MODEL_PATH` - путь к модели Vosk
-- `FFMPEG_PATH` - путь к исполняемому файлу FFmpeg
-
-## 📁 Структура проекта
-
-```
-api_processor/
-├── api_app/ # Основное приложение
-│ ├── grpc_client/ # gRPC клиент
-│ ├── services/ # Сервисы обработки
-│ ├── migrations/ # Миграции БД
-│ ├── models.py # Модели данных
-│ └── views.py # API эндпоинты
-├── api_project/ # Настройки Django
-├── proto/ # Protocol Buffers схемы
-├── models/ # Модели и исполняемые файлы
-└── requirements.txt # Зависимости
-```
-
-## 🔧 Особенности реализации
-
-- **Автоматическая конвертация**: Аудиофайлы автоматически конвертируются в формат WAV 16kHz
-- **Извлечение изображений**: Из документов извлекаются изображения в формате base64
-- **Обработка таблиц**: Таблицы сохраняются в CSV формате
-- **Fallback система**: При недоступности gRPC сервера используются заглушки
-
-## 📄 Лицензия
-
-Этот проект разработан для образовательных и исследовательских целей.
-
-## 🤝 Разработчик
-
-**Деев Егор Викторович** - Backend Developer
-- GitHub: [@EDeev](https://github.com/EDeev)
-- Email: egor@deev.space
-- Telegram: [@Egor_Deev](https://t.me/Egor_Deev)
+**Деев Егор Викторович** — [GitHub](https://github.com/EDeev) · [Telegram](https://t.me/DeevEgor) · [egor@deev.space](mailto:egor@deev.space)
---
- Создано с ❤️ от вашего дорогого - deev.space ©
+ ⭐ Если проект оказался полезным, поставьте звёздочку на GitHub!
+ Сделано с ❤️ — deev.space
diff --git a/compose.yaml b/compose.yaml
new file mode 100644
index 0000000..b45b89c
--- /dev/null
+++ b/compose.yaml
@@ -0,0 +1,13 @@
+services:
+ api:
+ build: .
+ image: ghcr.io/edeev/api_processor:latest
+ env_file: .env
+ ports:
+ - "8000:8000"
+ volumes:
+ - data:/data
+ restart: unless-stopped
+
+volumes:
+ data:
diff --git a/pytest.ini b/pytest.ini
new file mode 100644
index 0000000..ebdd362
--- /dev/null
+++ b/pytest.ini
@@ -0,0 +1,6 @@
+[pytest]
+DJANGO_SETTINGS_MODULE = api_project.settings
+testpaths = tests
+env =
+ D:DJANGO_DEBUG=True
+ D:GRPC_SERVER=
diff --git a/requirements-dev.txt b/requirements-dev.txt
new file mode 100644
index 0000000..3e37687
--- /dev/null
+++ b/requirements-dev.txt
@@ -0,0 +1,7 @@
+-r requirements.txt
+grpcio-tools==1.84.0 # только для пересборки proto/*_pb2*.py
+pytest==8.4.2
+pytest-django==4.11.1
+ruff==0.14.0
+fpdf2==2.8.4
+pytest-env==1.1.5
diff --git a/requirements.txt b/requirements.txt
index 32690df..67317ac 100644
--- a/requirements.txt
+++ b/requirements.txt
@@ -1,9 +1,9 @@
-django==4.2.6
-djangorestframework==3.14.0
-grpcio==1.58.0
-grpcio-tools==1.58.0
-pdfplumber==0.10.2
-python-docx==0.8.11
-Pillow==10.0.1
+Django==5.2.17
+djangorestframework==3.18.1
+grpcio==1.84.0
+protobuf==7.36.2
+pdfplumber==0.11.10
+python-docx==1.2.0
vosk==0.3.45
ffmpeg-python==0.2.0
+gunicorn==26.2.0
diff --git a/tests/test_api.py b/tests/test_api.py
new file mode 100644
index 0000000..31fd157
--- /dev/null
+++ b/tests/test_api.py
@@ -0,0 +1,139 @@
+import io
+import os
+import shutil
+import subprocess
+from concurrent import futures
+
+import docx
+import grpc
+import pytest
+from django.core.files.uploadedfile import SimpleUploadedFile
+from django.test import override_settings
+from fpdf import FPDF
+from rest_framework.test import APIClient
+
+from api_app.grpc_client import client as grpc_client
+from api_app.services.scan import extract_text_tables
+
+import text_service_pb2
+import text_service_pb2_grpc
+
+pytestmark = pytest.mark.django_db
+
+
+@pytest.fixture(autouse=True)
+def media(tmp_path, settings):
+ settings.MEDIA_ROOT = str(tmp_path)
+
+
+def make_docx():
+ d = docx.Document()
+ d.add_paragraph("Привет & мир")
+ t = d.add_table(rows=2, cols=2)
+ t.cell(0, 0).text, t.cell(0, 1).text = "a", "b"
+ t.cell(1, 0).text, t.cell(1, 1).text = "1", "<2>"
+ buf = io.BytesIO()
+ d.save(buf)
+ return buf.getvalue()
+
+
+def make_pdf():
+ pdf = FPDF()
+ pdf.add_page()
+ pdf.set_font("Helvetica", size=12)
+ pdf.cell(text="Hello world & co")
+ return bytes(pdf.output())
+
+
+def test_docx_text_is_escaped(tmp_path):
+ path = tmp_path / "a.DOCX"
+ path.write_bytes(make_docx())
+ html = extract_text_tables(str(path))
+ assert "Привет <script>alert(1)</script> & мир
" in html
+ assert "a,b\r\n1,<2>\r\n
" in html
+
+
+def test_pdf_upper_extension_and_escaping(tmp_path):
+ path = tmp_path / "a.PDF"
+ path.write_bytes(make_pdf())
+ assert "Hello <b>world</b> & co
" in extract_text_tables(str(path))
+
+
+def test_document_endpoint():
+ client = APIClient()
+ resp = client.post("/api/document-to-text/",
+ {"document": SimpleUploadedFile("r.docx", make_docx())}, format="multipart")
+ assert resp.status_code == 200
+ assert "<script>" in resp.json()["text"] and resp.json()["grpc_response"] is None
+
+
+def test_validation_and_size_limit(settings):
+ client = APIClient()
+ assert client.post("/api/document-to-text/", {}, format="multipart").status_code == 400
+ bad = client.post("/api/document-to-text/", {"document": SimpleUploadedFile("a.txt", b"x")}, format="multipart")
+ assert bad.status_code == 400
+ settings.MAX_UPLOAD_SIZE = 10
+ big = client.post("/api/document-to-text/", {"document": SimpleUploadedFile("a.pdf", b"x" * 100)},
+ format="multipart")
+ assert big.status_code == 413
+
+
+@override_settings(API_TOKEN="secret")
+def test_api_token():
+ client = APIClient()
+ upload = {"document": SimpleUploadedFile("r.docx", make_docx())}
+ assert client.post("/api/document-to-text/", upload, format="multipart").status_code == 403
+ client.credentials(HTTP_AUTHORIZATION="Bearer secret")
+ upload = {"document": SimpleUploadedFile("r.docx", make_docx())}
+ assert client.post("/api/document-to-text/", upload, format="multipart").status_code == 200
+
+
+class Upper(text_service_pb2_grpc.TextProcessorServicer):
+ def ProcessText(self, request, context):
+ return text_service_pb2.TextResponse(processed_text=request.text.upper(), success=True)
+
+
+def test_grpc_real_server(settings):
+ server = grpc.server(futures.ThreadPoolExecutor(max_workers=1))
+ text_service_pb2_grpc.add_TextProcessorServicer_to_server(Upper(), server)
+ port = server.add_insecure_port("127.0.0.1:0")
+ server.start()
+ try:
+ settings.GRPC_SERVER = f"127.0.0.1:{port}"
+ assert grpc_client.send_to_grpc_server("привет") == {
+ "processed_text": "ПРИВЕТ", "success": True, "error": None}
+ finally:
+ server.stop(None)
+
+
+def test_grpc_unavailable(settings):
+ settings.GRPC_SERVER = "127.0.0.1:1"
+ settings.GRPC_TIMEOUT = 2
+ result = grpc_client.send_to_grpc_server("x")
+ assert result["success"] is False and result["error"].startswith("gRPC")
+
+
+@pytest.mark.skipif(not shutil.which("ffmpeg"), reason="нужен ffmpeg")
+def test_audio_endpoint_converts_ogg(tmp_path, settings):
+ if not os.path.isdir(settings.VOSK_MODEL_PATH):
+ pytest.skip("нет модели Vosk")
+ ogg = tmp_path / "tone.ogg"
+ subprocess.run(["ffmpeg", "-loglevel", "error", "-f", "lavfi", "-i", "sine=frequency=440:duration=2",
+ "-c:a", "libvorbis", str(ogg)], check=True)
+ resp = APIClient().post("/api/audio-to-text/",
+ {"audio": SimpleUploadedFile("tone.ogg", ogg.read_bytes())}, format="multipart")
+ assert resp.status_code == 200 and isinstance(resp.json()["text"], str)
+
+
+SAMPLE = os.path.join(os.path.dirname(os.path.dirname(__file__)), "examples", "sample.ogg")
+
+
+@pytest.mark.skipif(not shutil.which("ffmpeg"), reason="нужен ffmpeg")
+def test_sample_recognition_keeps_all_phrases(settings):
+ if not os.path.isdir(settings.VOSK_MODEL_PATH):
+ pytest.skip("нет модели Vosk")
+ from api_app.services.vosk_recognizer import recognize_speech
+
+ text = recognize_speech(SAMPLE)
+ # в записи несколько фраз: раньше оставалась только последняя
+ assert "проверка" in text and "десять" in text