1
0
Fork 0
mirror of https://github.com/EDeev/api_processor.git synced 2026-10-08 04:59:33 +03:00

Compare commits

..

No commits in common. "479e0c96ad635817dde329db294c6fa494ea6915" and "9dafb27c3d0d64977b321362d705bda709f690b8" have entirely different histories.

27 changed files with 439 additions and 889 deletions

View file

@ -1,7 +0,0 @@
.git
.env
media
models
db.sqlite3
__pycache__
tests

View file

@ -1,9 +0,0 @@
# Обязательно в продакшене
DJANGO_SECRET_KEY=change-me-to-a-long-random-string
DJANGO_ALLOWED_HOSTS=localhost,127.0.0.1
# DJANGO_DEBUG=True # для разработки
# Необязательно
API_TOKEN= # если задан — запросы с заголовком Authorization: Bearer <токен>
GRPC_SERVER= # адрес сервиса TextProcessor, например textproc:50051; пусто — не отправлять
MAX_UPLOAD_SIZE_MB=50

2
.gitattributes vendored
View file

@ -1,2 +0,0 @@
* text=auto eol=lf
*.ogg binary

View file

@ -1,35 +0,0 @@
name: CI
on:
push:
branches: [main]
pull_request:
jobs:
test:
runs-on: ubuntu-latest
env:
VOSK_MODEL_PATH: ${{ github.workspace }}/models/vosk-model-small-ru-0.22
steps:
- uses: actions/checkout@v4
- uses: actions/setup-python@v5
with:
python-version: "3.12"
- run: sudo apt-get update -qq && sudo apt-get install -y -qq ffmpeg
- name: Модель Vosk (кэш)
id: model
uses: actions/cache@v4
with:
path: models
key: vosk-model-small-ru-0.22
- if: steps.model.outputs.cache-hit != 'true'
run: |
mkdir -p models
curl -fsSL -o /tmp/model.zip https://alphacephei.com/vosk/models/vosk-model-small-ru-0.22.zip
unzip -q /tmp/model.zip -d models
- run: pip install -r requirements-dev.txt
- run: ruff check --select E9,F,B --exclude proto .
- run: python manage.py check && python manage.py makemigrations --check --dry-run
env:
DJANGO_DEBUG: "True"
- run: pytest -q

View file

@ -1,42 +0,0 @@
name: Docker
on:
push:
tags: ["v*"]
workflow_dispatch:
jobs:
image:
runs-on: ubuntu-latest
permissions:
contents: read
packages: write
steps:
- uses: actions/checkout@v4
- uses: docker/setup-buildx-action@v3
- uses: docker/login-action@v3
with:
registry: ghcr.io
username: ${{ github.actor }}
password: ${{ secrets.GITHUB_TOKEN }}
- uses: docker/login-action@v3
with:
registry: dcr.deev.su
username: ${{ secrets.ZOT_USERNAME }}
password: ${{ secrets.ZOT_PASSWORD }}
- id: meta
uses: docker/metadata-action@v5
with:
images: |
ghcr.io/edeev/api_processor
dcr.deev.su/edeev/api_processor
tags: |
type=semver,pattern={{version}}
type=semver,pattern={{major}}.{{minor}}
type=raw,value=latest
- uses: docker/build-push-action@v6
with:
context: .
push: true
tags: ${{ steps.meta.outputs.tags }}
labels: ${{ steps.meta.outputs.labels }}

8
.gitignore vendored
View file

@ -1,8 +0,0 @@
.env
__pycache__/
*.pyc
.pytest_cache/
db.sqlite3
media/
staticfiles/
models/

View file

@ -1,33 +0,0 @@
FROM python:3.12-slim
ENV PYTHONDONTWRITEBYTECODE=1 \
PYTHONUNBUFFERED=1 \
DJANGO_DB_PATH=/data/db.sqlite3 \
DJANGO_MEDIA_ROOT=/data/media \
VOSK_MODEL_PATH=/app/models/vosk-model-small-ru-0.22
RUN apt-get update \
&& apt-get install -y --no-install-recommends ffmpeg curl unzip \
&& rm -rf /var/lib/apt/lists/*
WORKDIR /app
# офлайн-модель распознавания русской речи (~45 МБ)
RUN mkdir -p models \
&& curl -fsSL -o /tmp/model.zip https://alphacephei.com/vosk/models/vosk-model-small-ru-0.22.zip \
&& unzip -q /tmp/model.zip -d models \
&& rm /tmp/model.zip
COPY requirements.txt .
RUN pip install --no-cache-dir -r requirements.txt
COPY manage.py ./
COPY api_project/ api_project/
COPY api_app/ api_app/
COPY proto/ proto/
RUN useradd --create-home --uid 1000 app && mkdir -p /data && chown -R app:app /app /data
USER app
VOLUME ["/data"]
EXPOSE 8000
CMD ["sh", "-c", "python manage.py migrate --noinput && gunicorn api_project.wsgi:application --bind 0.0.0.0:8000 --workers 2 --timeout 300"]

21
LICENSE
View file

@ -1,21 +0,0 @@
MIT License
Copyright (c) 2025 Egor Deev
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
SOFTWARE.

View file

@ -1,129 +0,0 @@
# API Processor
[Русский](README.md) · **English**
[![CI](https://github.com/EDeev/api_processor/actions/workflows/ci.yml/badge.svg)](https://github.com/EDeev/api_processor/actions/workflows/ci.yml)
[![Docker](https://github.com/EDeev/api_processor/actions/workflows/docker.yml/badge.svg)](https://github.com/EDeev/api_processor/actions/workflows/docker.yml)
[![License](https://img.shields.io/github/license/EDeev/api_processor)](LICENSE)
A REST API that turns audio and documents into text:
- speech is recognized offline with a Vosk model;
- text, tables and images are extracted from PDF and DOCX.
The result can be forwarded over gRPC to a text processing service. Speech recognition targets Russian.
**Status:** personal project, completed
**Stack:** Python 3.12 · Django 5.2 · Django REST Framework · Vosk · FFmpeg · pdfplumber · python-docx · gRPC · Docker
## Features
- **Audio → text.** Any format FFmpeg reads (OGG, MP3, M4A, WAV…). Audio is converted to 16 kHz mono with
noise reduction and loudness normalization first. Recognition is offline, no external services.
- **Document → HTML.** PDF and DOCX: paragraphs in `<p>`, tables as CSV in `<pre>`, images as base64. The
document text is escaped, so the result is safe to render in a browser.
- **gRPC.** The text is sent to the `TextProcessor` service (`proto/text_service.proto`), and its reply is
returned in `grpc_response`. If no service address is set, the step is skipped.
- Optional access token and a file size limit (50 MB by default).
## API
```bash
curl -F audio=@examples/sample.ogg http://localhost:8000/api/audio-to-text/
curl -F document=@report.pdf http://localhost:8000/api/document-to-text/
```
```json
{
"text": "раз два три проверка перевода голоса текст насколько качественно она работает один два три четыре пять шесть семь восемь девять десять",
"grpc_response": null
}
```
With `API_TOKEN` set, add `Authorization: Bearer <token>`. Errors come as `{"error": "..."}`:
| Code | When |
|---|---|
| 400 | no file or unsupported document format |
| 403 | wrong token |
| 413 | file over the limit |
| 422 | audio could not be read |
## Running
```bash
git clone https://github.com/EDeev/api_processor.git && cd api_processor
cp .env.example .env # set DJANGO_SECRET_KEY
docker compose up -d # API at http://localhost:8000
```
The Vosk model and FFmpeg are inside the image. Prebuilt image: `docker pull ghcr.io/edeev/api_processor` or
`docker pull dcr.deev.su/edeev/api_processor`.
Without Docker you need:
- FFmpeg in `PATH`;
- the [vosk-model-small-ru-0.22](https://alphacephei.com/vosk/models) model unpacked into `models/`.
Then:
```bash
pip install -r requirements.txt
DJANGO_DEBUG=True python manage.py migrate
DJANGO_DEBUG=True python manage.py runserver
```
| Variable | Purpose |
|---|---|
| `DJANGO_SECRET_KEY` | secret key, required unless `DJANGO_DEBUG=True` |
| `DJANGO_ALLOWED_HOSTS` | comma-separated domains |
| `API_TOKEN` | if set, access requires the token |
| `GRPC_SERVER`, `GRPC_TIMEOUT` | TextProcessor address and timeout, s |
| `MAX_UPLOAD_SIZE_MB` | file size limit |
| `VOSK_MODEL_PATH`, `FFMPEG_BINARY` | paths to the model and ffmpeg |
## How it works
```mermaid
flowchart LR
C[Client] -->|multipart| V[DRF APIView]
V -->|audio| F[FFmpeg: 16 kHz mono, denoise] --> K[Vosk]
V -->|PDF / DOCX| S[pdfplumber / python-docx]
K --> T[text]
S --> T
T -->|gRPC ProcessText| G[TextProcessor]
T --> R[JSON response]
```
The Vosk model is loaded once per process. Each request gets its own temporary file. Uploaded files and
results are stored in SQLite and `media/`.
## Development
```bash
pip install -r requirements-dev.txt
ruff check --select E9,F,B --exclude proto . && pytest
```
What the tests cover:
- recognition of the whole sample, with every phrase;
- OGG conversion;
- PDF and DOCX extraction with escaping;
- limits and the token;
- a real gRPC server and an unavailable one.
The Docker image is built on `v*` tags and published to GitHub Packages and `dcr.deev.su`.
## License
MIT — see [LICENSE](LICENSE). The Vosk model is licensed under Apache 2.0.
## Author
**Egor Deev** — [GitHub](https://github.com/EDeev) · [Telegram](https://t.me/DeevEgor) · [egor@deev.space](mailto:egor@deev.space)
---
<div align="center">
<sub>⭐ If you find this project useful, give it a star on GitHub!</sub>
<p><sub>Made with ❤️ — <a href="https://deev.space">deev.space</a></sub></p>
</div>

237
README.md
View file

@ -1,142 +1,171 @@
# API Processor # API Processor
**Русский** · [English](README.en.md) **Django REST API для обработки аудио и документов с интеграцией gRPC сервисов**
[![CI](https://github.com/EDeev/api_processor/actions/workflows/ci.yml/badge.svg)](https://github.com/EDeev/api_processor/actions/workflows/ci.yml) API Processor — это мощное решение для автоматической обработки мультимедийного контента. Система выполняет транскрибацию аудиофайлов в текст и извлечение данных из PDF/DOCX документов с последующей отправкой результатов на внешний gRPC сервер для дополнительной обработки.
[![Docker](https://github.com/EDeev/api_processor/actions/workflows/docker.yml/badge.svg)](https://github.com/EDeev/api_processor/actions/workflows/docker.yml)
[![License](https://img.shields.io/github/license/EDeev/api_processor)](LICENSE)
REST API, которое превращает аудио и документы в текст: ## 🚀 Возможности
- речь распознаётся офлайн моделью Vosk;
- из PDF и DOCX извлекаются текст, таблицы и изображения.
Результат можно отправить дальше по gRPC — на сервис обработки текста. - **Аудио транскрипция**: Преобразование аудиофайлов в текст с использованием модели Vosk
- **Обработка документов**: Извлечение текста, таблиц и изображений из PDF и DOCX файлов
- **gRPC интеграция**: Автоматическая отправка обработанного текста на внешний сервер
- **RESTful API**: Простой и понятный интерфейс для взаимодействия
- **Поддержка форматов**: Audio (WAV, OGG и др.), PDF, DOCX
**Статус:** личный проект, завершён ## 🛠 Технологический стек
**Стек:** Python 3.12 · Django 5.2 · Django REST Framework · Vosk · FFmpeg · pdfplumber · python-docx · gRPC · Docker ### Backend
- **Django 4.2.6** - веб-фреймворк
- **Django REST Framework 3.14.0** - API framework
- **Python** - основной язык разработки
## Возможности ### Обработка контента
- **Vosk 0.3.45** - распознавание речи
- **FFmpeg Python 0.2.0** - конвертация аудио
- **pdfplumber 0.10.2** - извлечение данных из PDF
- **python-docx 0.8.11** - работа с DOCX файлами
- **Pillow 10.0.1** - обработка изображений
- **Аудио → текст.** Любой формат, который читает FFmpeg (OGG, MP3, M4A, WAV…). Перед распознаванием ### Коммуникация
звук приводится к 16 кГц моно с шумоподавлением и нормализацией громкости. Распознавание офлайн, без - **gRPC 1.58.0** - межсервисное взаимодействие
внешних сервисов. - **Protocol Buffers** - сериализация данных
- **Документ → HTML.** PDF и DOCX: абзацы — в `<p>`, таблицы — CSV в `<pre>`, изображения — в base64.
Текст документа экранируется, поэтому результат безопасно показывать в браузере.
- **gRPC.** Распознанный текст уходит на сервис `TextProcessor` (`proto/text_service.proto`), его ответ
возвращается в поле `grpc_response`. Если адрес сервиса не задан, шаг пропускается.
- Необязательный токен доступа и ограничение размера файла (по умолчанию 50 МБ).
## API ### База данных
- **SQLite** - локальное хранение метаданных файлов
```bash ## 📋 Требования
curl -F audio=@examples/sample.ogg http://localhost:8000/api/audio-to-text/
curl -F document=@report.pdf http://localhost:8000/api/document-to-text/
```
```json - Python 3.8+
{ - FFmpeg (для конвертации аудио)
"text": "раз два три проверка перевода голоса текст насколько качественно она работает один два три четыре пять шесть семь восемь девять десять", - Модель Vosk для русского языка
"grpc_response": null
}
```
Если задан `API_TOKEN`, добавьте заголовок `Authorization: Bearer <токен>`. Ошибки приходят как ## ⚡ Быстрый старт
`{"error": "..."}`:
| Код | Когда | ### Установка зависимостей
|---|---|
| 400 | нет файла или неподдерживаемый формат документа |
| 403 | неверный токен |
| 413 | файл больше лимита |
| 422 | аудио не удалось прочитать |
## Запуск
```bash
git clone https://github.com/EDeev/api_processor.git && cd api_processor
cp .env.example .env # задайте DJANGO_SECRET_KEY
docker compose up -d # API на http://localhost:8000
```
Модель Vosk и FFmpeg уже внутри образа. Готовый образ: `docker pull ghcr.io/edeev/api_processor` или
`docker pull dcr.deev.su/edeev/api_processor`.
Без Docker нужны:
- FFmpeg в `PATH`;
- модель [vosk-model-small-ru-0.22](https://alphacephei.com/vosk/models), распакованная в `models/`.
Затем:
```bash ```bash
pip install -r requirements.txt pip install -r requirements.txt
DJANGO_DEBUG=True python manage.py migrate
DJANGO_DEBUG=True python manage.py runserver
``` ```
| Переменная | Назначение | ### Настройка модели Vosk
|---|---|
| `DJANGO_SECRET_KEY` | секретный ключ, обязателен без `DJANGO_DEBUG=True` |
| `DJANGO_ALLOWED_HOSTS` | домены через запятую |
| `API_TOKEN` | если задан — доступ только с токеном |
| `GRPC_SERVER`, `GRPC_TIMEOUT` | адрес сервиса TextProcessor и таймаут, с |
| `MAX_UPLOAD_SIZE_MB` | лимит размера файла |
| `VOSK_MODEL_PATH`, `FFMPEG_BINARY` | путь к модели и к ffmpeg |
## Как устроено 1. Скачайте модель `vosk-model-small-ru-0.22`
2. Разместите в папке `models/vosk-model-small-ru-0.22`
```mermaid ### Настройка FFmpeg
flowchart LR
C[Клиент] -->|multipart| V[DRF APIView]
V -->|аудио| F[FFmpeg: 16 кГц моно, шумоподавление] --> K[Vosk]
V -->|PDF / DOCX| S[pdfplumber / python-docx]
K --> T[текст]
S --> T
T -->|gRPC ProcessText| G[TextProcessor]
T --> R[JSON-ответ]
```
``` 1. Скачайте FFmpeg
api_app/views.py эндпоинты и проверки 2. Разместите в `models/ffmpeg/bin/ffmpeg.exe`
api_app/services/vosk_recognizer.py конвертация FFmpeg и распознавание Vosk
api_app/services/scan.py извлечение из PDF и DOCX
api_app/grpc_client/client.py клиент TextProcessor
proto/ описание gRPC-сервиса и сгенерированный код
```
Модель Vosk загружается один раз на процесс. У каждого запроса свой временный файл. Загруженные файлы и ### Запуск сервера
результат сохраняются в SQLite и `media/`.
## Разработка
```bash ```bash
pip install -r requirements-dev.txt python manage.py migrate
ruff check --select E9,F,B --exclude proto . && pytest python manage.py runserver
``` ```
Что проверяют тесты: ## 📖 API Endpoints
- распознавание образца целиком, со всеми фразами;
- конвертацию OGG;
- извлечение из PDF и DOCX с экранированием;
- лимиты и токен;
- работу с настоящим gRPC-сервером и его недоступность.
Docker-образ собирается по тегу `v*` и публикуется в GitHub Packages и `dcr.deev.su`. ### Транскрипция аудио
```http
POST /api/audio-to-text/
Content-Type: multipart/form-data
Код gRPC пересобирается так: audio: <audio_file>
`python -m grpc_tools.protoc -Iproto --python_out=proto --grpc_python_out=proto proto/text_service.proto`. ```
## Лицензия **Ответ:**
```json
{
"text": "Распознанный текст из аудио",
"grpc_response": {
"processed_text": "Обработанный текст",
"success": true,
"error": null
}
}
```
MIT — см. [LICENSE](LICENSE). Модель Vosk распространяется под Apache 2.0. ### Обработка документов
```http
POST /api/document-to-text/
Content-Type: multipart/form-data
## Автор document: <pdf_or_docx_file>
```
**Деев Егор Викторович** — [GitHub](https://github.com/EDeev) · [Telegram](https://t.me/DeevEgor) · [egor@deev.space](mailto:egor@deev.space) **Ответ:**
```json
{
"text": "<p>Извлеченный текст</p><pre>таблица,данные</pre>",
"grpc_response": {
"processed_text": "Обработанный текст",
"success": true,
"error": null
}
}
```
## 🧪 Примеры использования
### cURL команды
**Транскрипция аудио:**
```bash
curl -X POST -F "audio=@audio.ogg" http://localhost:8000/api/audio-to-text/
```
**Обработка документа:**
```bash
curl -X POST -F "document=@document.pdf" http://localhost:8000/api/document-to-text/
```
## ⚙️ Конфигурация
### gRPC настройки
По умолчанию система подключается к gRPC серверу на `localhost:50051`. Для изменения адреса отредактируйте `api_app/grpc_client/client.py`.
### Модели и пути
Пути к моделям и исполняемым файлам настраиваются в `api_app/services/vosk_recognizer.py`:
- `MODEL_PATH` - путь к модели Vosk
- `FFMPEG_PATH` - путь к исполняемому файлу FFmpeg
## 📁 Структура проекта
```
api_processor/
├── api_app/ # Основное приложение
│ ├── grpc_client/ # gRPC клиент
│ ├── services/ # Сервисы обработки
│ ├── migrations/ # Миграции БД
│ ├── models.py # Модели данных
│ └── views.py # API эндпоинты
├── api_project/ # Настройки Django
├── proto/ # Protocol Buffers схемы
├── models/ # Модели и исполняемые файлы
└── requirements.txt # Зависимости
```
## 🔧 Особенности реализации
- **Автоматическая конвертация**: Аудиофайлы автоматически конвертируются в формат WAV 16kHz
- **Извлечение изображений**: Из документов извлекаются изображения в формате base64
- **Обработка таблиц**: Таблицы сохраняются в CSV формате
- **Fallback система**: При недоступности gRPC сервера используются заглушки
## 📄 Лицензия
Этот проект разработан для образовательных и исследовательских целей.
## 🤝 Разработчик
**Деев Егор Викторович** - Backend Developer
- GitHub: [@EDeev](https://github.com/EDeev)
- Email: egor@deev.space
- Telegram: [@Egor_Deev](https://t.me/Egor_Deev)
--- ---
<div align="center"> <div align="center">
<sub>⭐ Если проект оказался полезным, поставьте звёздочку на GitHub!</sub> <p><sub>Создано с ❤️ от вашего дорогого - deev.space ©</sub></p>
<p><sub>Сделано с ❤️ — <a href="https://deev.space">deev.space</a></sub></p>
</div> </div>

View file

@ -1,38 +1,90 @@
import logging # api_project/api_app/grpc_client/client.py
import grpc
import os import os
import sys import sys
import grpc # Добавляем путь для импорта сгенерированных протофайлов
from django.conf import settings current_dir = os.path.dirname(os.path.abspath(__file__))
proto_dir = os.path.join(os.path.dirname(os.path.dirname(os.path.dirname(current_dir))), 'proto')
sys.path.append(proto_dir)
# сгенерированные модули лежат в proto/ в корне репозитория (раньше путь считался на уровень # Пробуем импортировать сгенерированные протофайлы
# выше репозитория — импорт всегда падал, и вместо настоящего gRPC работала заглушка # Если не получится, используем заглушки
# с поддельным success: true) try:
PROTO_DIR = os.path.join(os.path.dirname(os.path.dirname(os.path.dirname(os.path.abspath(__file__)))), "proto") import text_service_pb2
if PROTO_DIR not in sys.path: import text_service_pb2_grpc
sys.path.append(PROTO_DIR) print("Успешно импортированы сгенерированные proto файлы")
except ImportError:
import text_service_pb2 # noqa: E402 print("Не удалось импортировать сгенерированные proto файлы, используем заглушки")
import text_service_pb2_grpc # noqa: E402
# Создаем заглушки
logger = logging.getLogger(__name__) class TextRequest:
def __init__(self, text):
self.text = text
def send_to_grpc_server(text: str):
"""Отправляет текст на gRPC-сервер TextProcessor. Если адрес сервера не задан class TextResponse:
(GRPC_SERVER пустой), шаг пропускается и возвращается None""" def __init__(self, processed_text, success, error):
if not settings.GRPC_SERVER: self.processed_text = processed_text
return None self.success = success
self.error = error
class TextProcessorStub:
def __init__(self, channel):
self.channel = channel
def ProcessText(self, request):
# Эмулируем ответ от gRPC сервера
return TextResponse(
processed_text=f"ЗАГЛУШКА ОБРАБОТКИ ТЕКСТА: {request.text}",
success=True,
error=""
)
# Создаем модуль заглушки
class text_service_pb2:
TextRequest = TextRequest
TextResponse = TextResponse
class text_service_pb2_grpc:
TextProcessorStub = TextProcessorStub
def send_to_grpc_server(text: str) -> dict:
"""
Отправляет текст на gRPC сервер для обработки
Args:
text: Текст для обработки
Returns:
dict: Результат обработки
"""
try: try:
with grpc.insecure_channel(settings.GRPC_SERVER) as channel: # Создаем соединение с сервером
stub = text_service_pb2_grpc.TextProcessorStub(channel) # Для заглушки это не обязательно, но оставим для совместимости
response = stub.ProcessText(text_service_pb2.TextRequest(text=text), timeout=settings.GRPC_TIMEOUT) try:
channel = grpc.insecure_channel('localhost:50051')
except NameError:
# Если grpc не импортирован, используем заглушку
channel = "dummy_channel"
# Создаем клиент
stub = text_service_pb2_grpc.TextProcessorStub(channel)
# Создаем запрос
request = text_service_pb2.TextRequest(text=text)
# Отправляем запрос
response = stub.ProcessText(request)
# Возвращаем результат
return { return {
'processed_text': response.processed_text, 'processed_text': response.processed_text,
'success': response.success, 'success': response.success,
'error': response.error or None, 'error': response.error if hasattr(response, 'error') and response.error else None
}
except Exception as e:
return {
'processed_text': None,
'success': False,
'error': str(e)
} }
except grpc.RpcError as e:
logger.warning("gRPC-сервер %s недоступен: %s", settings.GRPC_SERVER, e.code())
return {'processed_text': None, 'success': False, 'error': f"gRPC: {e.code().name}"}

View file

@ -1,33 +1,33 @@
# Generated by Django 5.1.3 on 2025-05-06 12:35 # Generated by Django 5.1.3 on 2025-05-06 12:35
import api_app.models import api_app.models
from django.db import migrations, models from django.db import migrations, models
class Migration(migrations.Migration): class Migration(migrations.Migration):
initial = True initial = True
dependencies = [ dependencies = [
] ]
operations = [ operations = [
migrations.CreateModel( migrations.CreateModel(
name='AudioFile', name='AudioFile',
fields=[ fields=[
('id', models.AutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')), ('id', models.AutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')),
('file', models.FileField(upload_to=api_app.models.get_file_path)), ('file', models.FileField(upload_to=api_app.models.get_file_path)),
('uploaded_at', models.DateTimeField(auto_now_add=True)), ('uploaded_at', models.DateTimeField(auto_now_add=True)),
('processed_text', models.TextField(blank=True, null=True)), ('processed_text', models.TextField(blank=True, null=True)),
], ],
), ),
migrations.CreateModel( migrations.CreateModel(
name='DocumentFile', name='DocumentFile',
fields=[ fields=[
('id', models.AutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')), ('id', models.AutoField(auto_created=True, primary_key=True, serialize=False, verbose_name='ID')),
('file', models.FileField(upload_to=api_app.models.get_file_path)), ('file', models.FileField(upload_to=api_app.models.get_file_path)),
('uploaded_at', models.DateTimeField(auto_now_add=True)), ('uploaded_at', models.DateTimeField(auto_now_add=True)),
('processed_text', models.TextField(blank=True, null=True)), ('processed_text', models.TextField(blank=True, null=True)),
], ],
), ),
] ]

View file

@ -4,7 +4,7 @@ import uuid
import os import os
def get_file_path(instance, filename): def get_file_path(instance, filename):
ext = filename.split('.')[-1].lower() ext = filename.split('.')[-1]
filename = f"{uuid.uuid4()}.{ext}" filename = f"{uuid.uuid4()}.{ext}"
return os.path.join('uploads', filename) return os.path.join('uploads', filename)

View file

@ -1,16 +0,0 @@
import hmac
from django.conf import settings
from rest_framework.permissions import BasePermission
class ApiTokenPermission(BasePermission):
"""Если задан API_TOKEN, запросы должны нести заголовок Authorization: Bearer <токен>"""
message = "Нужен заголовок Authorization: Bearer <API_TOKEN>"
def has_permission(self, request, view):
if not settings.API_TOKEN:
return True
header = request.headers.get("Authorization", "")
return hmac.compare_digest(header, f"Bearer {settings.API_TOKEN}")

View file

@ -1,70 +1,66 @@
import base64 import pdfplumber
import csv import docx
from html import escape import csv
from io import StringIO import os
from io import StringIO, BytesIO
import docx from PIL import Image
import pdfplumber import base64
# сигнатуры форматов: картинка из PDF — это сырой поток, не обязательно PNG
IMAGE_SIGNATURES = { def extract_text_tables(file_path: str) -> str:
b"\x89PNG\r\n\x1a\n": "image/png", result = ""
b"\xff\xd8\xff": "image/jpeg", if file_path.endswith(".pdf"):
b"GIF8": "image/gif", with pdfplumber.open(file_path) as pdf:
} for page in pdf.pages:
text = page.extract_text()
if text:
def image_mime(data: bytes): result += "<p>" + text.replace("\n", "</p><p>") + "</p>"
for signature, mime in IMAGE_SIGNATURES.items():
if data.startswith(signature): tables = page.extract_tables()
return mime if tables:
return None for table in tables:
csv_output = StringIO()
csv_writer = csv.writer(csv_output)
def img_tag(data: bytes, mime: str) -> str: csv_writer.writerows(table)
return f'<img src="data:{mime};base64,{base64.b64encode(data).decode("utf-8")}"/>' result += f"<pre>{csv_output.getvalue()}</pre>"
# Извлечение изображений
def table_to_pre(rows) -> str: if page.images:
csv_output = StringIO() for img in page.images:
csv.writer(csv_output).writerows([[cell if cell is not None else "" for cell in row] for row in rows]) img_data = img["stream"].get_data()
return f"<pre>{escape(csv_output.getvalue())}</pre>" encoded_img = base64.b64encode(img_data).decode("utf-8")
result += f'<img src="data:image/png;base64,{encoded_img}"/>'
def extract_text_tables(file_path: str) -> str: elif file_path.endswith(".docx"):
"""Текст, таблицы (CSV) и изображения документа одной HTML-строкой. doc = docx.Document(file_path)
Текст экранируется: иначе содержимое документа становится разметкой (XSS у потребителя)"""
result = "" text_data = []
path = file_path.lower() # раньше файл «.PDF» молча давал пустой результат table_data = []
image_data = []
if path.endswith(".pdf"):
with pdfplumber.open(file_path) as pdf: for para in doc.paragraphs:
for page in pdf.pages: if para.text.strip():
text = page.extract_text() text_data.append(f"<p>{para.text}</p>")
if text:
result += "".join(f"<p>{escape(line)}</p>" for line in text.split("\n")) for table in doc.tables:
csv_output = StringIO()
for table in page.extract_tables() or []: csv_writer = csv.writer(csv_output)
result += table_to_pre(table) for row in table.rows:
csv_writer.writerow([cell.text.strip() for cell in row.cells])
# Извлечение изображений — только тех, что лежат в PDF готовым файлом (PNG/JPEG) table_data.append(f"<pre>{csv_output.getvalue()}</pre>")
for img in page.images:
data = img["stream"].get_data() # Извлечение изображений
mime = image_mime(data) for rel in doc.part.rels:
if mime: if "image" in doc.part.rels[rel].target_ref:
result += img_tag(data, mime) image_data_blob = doc.part.rels[rel].target_part.blob
encoded_img = base64.b64encode(image_data_blob).decode("utf-8")
elif path.endswith(".docx"): image_data.append(f'<img src="data:image/png;base64,{encoded_img}"/>')
doc = docx.Document(file_path)
if text_data:
result += "".join(f"<p>{escape(para.text)}</p>" for para in doc.paragraphs if para.text.strip()) result += "".join(text_data)
result += "".join(table_to_pre([[cell.text.strip() for cell in row.cells] for row in table.rows]) if table_data:
for table in doc.tables) result += "".join(table_data)
if image_data:
# Извлечение изображений с их настоящим типом result += "".join(image_data)
for rel in doc.part.rels.values():
if "image" in rel.reltype and not rel.is_external: return result
part = rel.target_part
result += img_tag(part.blob, part.content_type)
return result

View file

@ -1,81 +1,51 @@
import json import os, wave, vosk, ffmpeg
import logging
import os
import tempfile
import threading
import wave
import ffmpeg MODEL_PATH = r"models/vosk-model-small-ru-0.22"
import vosk FFMPEG_PATH = r"models/ffmpeg/bin/ffmpeg.exe"
from django.conf import settings
logger = logging.getLogger(__name__) def convert_audio_to_wav(input_file, output_file, FFMPEG_PATH):
vosk.SetLogLevel(-1)
_model = None
_model_lock = threading.Lock()
class RecognitionError(Exception):
pass
def get_model():
"""Модель Vosk загружается один раз на процесс (раньше — на каждый запрос, это секунды и сотни МБ)"""
global _model
with _model_lock:
if _model is None:
if not os.path.isdir(settings.VOSK_MODEL_PATH):
raise RecognitionError(f"Модель Vosk не найдена: {settings.VOSK_MODEL_PATH}")
_model = vosk.Model(str(settings.VOSK_MODEL_PATH))
return _model
def convert_audio_to_wav(input_file, output_file):
"""WAV 16 кГц моно для Vosk, с шумоподавлением и нормализацией громкости"""
try: try:
( (
ffmpeg ffmpeg
.input(input_file) .input(input_file)
.output(output_file, format='wav', acodec='pcm_s16le', ar='16000', ac=1, .output(output_file, format='wav', acodec='pcm_s16le', ar='16000', ac=1,
af='acompressor,afftdn,dynaudnorm,aresample=16000') # 16kHz для Vosk af='acompressor,afftdn,dynaudnorm,aresample=16000') # 16kHz для Vosk
.global_args('-loglevel', 'error') .global_args('-loglevel', 'quiet')
.run(cmd=settings.FFMPEG_BINARY, overwrite_output=True, capture_stdout=True, capture_stderr=True) .run(cmd=FFMPEG_PATH, overwrite_output=True)
) )
print(f"Конвертация завершена: {output_file}")
except ffmpeg.Error as e: except ffmpeg.Error as e:
raise RecognitionError("Не удалось прочитать аудиофайл") from e print("Ошибка при конвертации:", e.stderr.decode())
def is_vosk_ready_wav(path) -> bool: vosk.SetLogLevel(-1)
try:
with wave.open(path, "rb") as wf:
return wf.getnchannels() == 1 and wf.getsampwidth() == 2 and wf.getframerate() == 16000
except (wave.Error, EOFError):
return False
def recognize_speech(audio_path) -> str: def recognize_speech(audio_path) -> str:
model = get_model() if not os.path.exists(MODEL_PATH):
print("Ошибка: Модель не найдена!")
return ""
# временный файл — свой на каждый запрос (раньше общий audio.wav в текущей папке: model = vosk.Model(MODEL_PATH)
# одновременные запросы перезаписывали друг другу аудио)
with tempfile.TemporaryDirectory() as tmp:
if not (audio_path.lower().endswith(".wav") and is_vosk_ready_wav(audio_path)):
wav_path = os.path.join(tmp, "audio.wav")
convert_audio_to_wav(audio_path, wav_path)
audio_path = wav_path
# Vosk отдаёт текст по фразам: раньше бралось только FinalResult(), и в длинной записи if audio_path.split('.')[-1] != "wav":
# оставалась лишь последняя фраза convert.convert_audio_to_wav(audio_path, "audio.wav", FFMPEG_PATH)
parts = [] audio_path = "audio.wav"
else:
with wave.open(audio_path, "rb") as wf: with wave.open(audio_path, "rb") as wf:
recognizer = vosk.KaldiRecognizer(model, wf.getframerate()) if wf.getnchannels() != 1 or wf.getsampwidth() != 2 or wf.getframerate() != 16000:
while True: convert.convert_audio_to_wav(audio_path, "audio.wav", FFMPEG_PATH)
data = wf.readframes(4000) audio_path = "audio.wav"
if not data:
break
if recognizer.AcceptWaveform(data):
parts.append(json.loads(recognizer.Result()).get("text", ""))
parts.append(json.loads(recognizer.FinalResult()).get("text", ""))
return " ".join(p for p in parts if p)
with wave.open(audio_path, "rb") as wf: # использование vosk
recognizer = vosk.KaldiRecognizer(model, wf.getframerate())
while True:
data = wf.readframes(3200)
if not data:
break
recognizer.AcceptWaveform(data)
if audio_path == "audio.wav":
os.remove(audio_path)
return recognizer.FinalResult().split(": \"")[-1][:-3]

View file

@ -1,90 +1,77 @@
import logging # api_project/api_app/views.py
import os import os
from django.conf import settings
from rest_framework import status from rest_framework import status
from rest_framework.parsers import FormParser, MultiPartParser
from rest_framework.response import Response
from rest_framework.views import APIView from rest_framework.views import APIView
from rest_framework.response import Response
from rest_framework.parsers import MultiPartParser, FormParser
from django.conf import settings
from .grpc_client.client import send_to_grpc_server
from .models import AudioFile, DocumentFile from .models import AudioFile, DocumentFile
from .services.vosk_recognizer import recognize_speech
from .services.scan import extract_text_tables from .services.scan import extract_text_tables
from .services.vosk_recognizer import RecognitionError, recognize_speech from .grpc_client.client import send_to_grpc_server
logger = logging.getLogger(__name__) class AudioToTextView(APIView):
def too_large(uploaded):
return uploaded.size > settings.MAX_UPLOAD_SIZE
def size_error():
return Response({'error': f'Файл больше {settings.MAX_UPLOAD_SIZE // 1024 // 1024} МБ'},
status=status.HTTP_413_REQUEST_ENTITY_TOO_LARGE)
class ProcessView(APIView):
parser_classes = (MultiPartParser, FormParser) parser_classes = (MultiPartParser, FormParser)
field = ""
model = None
def process(self, path):
raise NotImplementedError
def handle(self, uploaded):
record = self.model(file=uploaded)
record.save()
try:
text = self.process(os.path.join(settings.MEDIA_ROOT, record.file.name))
except RecognitionError as e:
return Response({'error': str(e)}, status=status.HTTP_422_UNPROCESSABLE_ENTITY)
except Exception:
# подробности — в лог, а не в ответ клиенту
logger.exception("Ошибка обработки %s", record.file.name)
return Response({'error': 'Не удалось обработать файл'}, status=status.HTTP_500_INTERNAL_SERVER_ERROR)
record.processed_text = text
record.save()
return Response({'text': text, 'grpc_response': send_to_grpc_server(text)}, status=status.HTTP_200_OK)
class AudioToTextView(ProcessView):
model = AudioFile
def process(self, path):
return recognize_speech(path)
def post(self, request, *args, **kwargs): def post(self, request, *args, **kwargs):
audio_file = request.FILES.get('audio') audio_file = request.FILES.get('audio')
if not audio_file: if not audio_file:
return Response({'error': 'Нет аудио файла'}, status=status.HTTP_400_BAD_REQUEST) return Response({'error': 'Нет аудио файла'}, status=status.HTTP_400_BAD_REQUEST)
if too_large(audio_file):
return size_error() audio_model = AudioFile(file=audio_file)
audio_model.save()
return self.handle(audio_file)
try:
file_path = os.path.join(settings.MEDIA_ROOT, audio_model.file.name)
class DocumentToTextView(ProcessView):
model = DocumentFile text = recognize_speech(file_path)
def process(self, path): audio_model.processed_text = text
return extract_text_tables(path) audio_model.save()
grpc_response = send_to_grpc_server(text)
return Response({
'text': text,
'grpc_response': grpc_response
}, status=status.HTTP_200_OK)
except Exception as e:
return Response({'error': str(e)}, status=status.HTTP_500_INTERNAL_SERVER_ERROR)
class DocumentToTextView(APIView):
parser_classes = (MultiPartParser, FormParser)
def post(self, request, *args, **kwargs): def post(self, request, *args, **kwargs):
document_file = request.FILES.get('document') document_file = request.FILES.get('document')
if not document_file: if not document_file:
return Response({'error': 'Нет документа'}, status=status.HTTP_400_BAD_REQUEST) return Response({'error': 'Нет документа'}, status=status.HTTP_400_BAD_REQUEST)
file_ext = os.path.splitext(document_file.name)[1].lower() file_ext = os.path.splitext(document_file.name)[1].lower()
if file_ext not in ['.pdf', '.docx']: if file_ext not in ['.pdf', '.docx']:
return Response({'error': 'Поддерживаются только PDF и DOCX файлы'}, return Response({'error': 'Поддерживаются только PDF и DOCX файлы'},
status=status.HTTP_400_BAD_REQUEST) status=status.HTTP_400_BAD_REQUEST)
if too_large(document_file):
return size_error() doc_model = DocumentFile(file=document_file)
doc_model.save()
return self.handle(document_file)
try:
file_path = os.path.join(settings.MEDIA_ROOT, doc_model.file.name)
text = extract_text_tables(file_path)
doc_model.processed_text = text
doc_model.save()
grpc_response = send_to_grpc_server(text)
return Response({
'text': text,
'grpc_response': grpc_response
}, status=status.HTTP_200_OK)
except Exception as e:
return Response({'error': str(e)}, status=status.HTTP_500_INTERNAL_SERVER_ERROR)

View file

@ -3,20 +3,11 @@ from pathlib import Path
BASE_DIR = Path(__file__).resolve().parent.parent BASE_DIR = Path(__file__).resolve().parent.parent
SECRET_KEY = 'django-insecure-)+yykzv8cr7dbc38g2#x(8*ifs@+-f_fyan9!c%mmxg1$ekztq'
def env_bool(name, default=False): DEBUG = True
return os.environ.get(name, str(default)).lower() in ("1", "true", "yes")
ALLOWED_HOSTS = []
DEBUG = env_bool("DJANGO_DEBUG")
# Ключ по умолчанию — только для разработки (DJANGO_DEBUG=True)
DEV_SECRET_KEY = "django-insecure-dev-only-change-me"
SECRET_KEY = os.environ.get("DJANGO_SECRET_KEY", DEV_SECRET_KEY if DEBUG else "")
if not SECRET_KEY:
raise RuntimeError("Задайте DJANGO_SECRET_KEY (или DJANGO_DEBUG=True для разработки)")
ALLOWED_HOSTS = [h for h in os.environ.get("DJANGO_ALLOWED_HOSTS", "localhost,127.0.0.1").split(",") if h]
INSTALLED_APPS = [ INSTALLED_APPS = [
'django.contrib.admin', 'django.contrib.admin',
@ -24,7 +15,7 @@ INSTALLED_APPS = [
'django.contrib.contenttypes', 'django.contrib.contenttypes',
'django.contrib.sessions', 'django.contrib.sessions',
'django.contrib.messages', 'django.contrib.messages',
'django.contrib.staticfiles', # 'django.contrib.staticfiles',
'rest_framework', # Добавляем DRF 'rest_framework', # Добавляем DRF
'api_app', # Наше API приложение 'api_app', # Наше API приложение
] ]
@ -62,38 +53,17 @@ WSGI_APPLICATION = 'api_project.wsgi.application'
DATABASES = { DATABASES = {
'default': { 'default': {
'ENGINE': 'django.db.backends.sqlite3', 'ENGINE': 'django.db.backends.sqlite3',
'NAME': os.environ.get("DJANGO_DB_PATH", BASE_DIR / 'db.sqlite3'), 'NAME': BASE_DIR / 'db.sqlite3',
} }
} }
# Путь для загрузки файлов # Путь для загрузки файлов
MEDIA_URL = '/media/' MEDIA_URL = '/media/'
MEDIA_ROOT = os.environ.get("DJANGO_MEDIA_ROOT", os.path.join(BASE_DIR, 'media')) MEDIA_ROOT = os.path.join(BASE_DIR, 'media')
STATIC_URL = '/static/'
STATIC_ROOT = BASE_DIR / 'staticfiles'
# Распознавание и обработка
VOSK_MODEL_PATH = os.environ.get("VOSK_MODEL_PATH", BASE_DIR / "models" / "vosk-model-small-ru-0.22")
FFMPEG_BINARY = os.environ.get("FFMPEG_BINARY", "ffmpeg") # из PATH; раньше — Windows-путь к ffmpeg.exe
GRPC_SERVER = os.environ.get("GRPC_SERVER", "localhost:50051") # пусто — не отправлять на gRPC
GRPC_TIMEOUT = float(os.environ.get("GRPC_TIMEOUT", "10"))
API_TOKEN = os.environ.get("API_TOKEN", "") # если задан — нужен заголовок Authorization: Bearer
MAX_UPLOAD_SIZE = int(os.environ.get("MAX_UPLOAD_SIZE_MB", "50")) * 1024 * 1024
FILE_UPLOAD_MAX_MEMORY_SIZE = 5 * 1024 * 1024 # крупные файлы — во временный файл, не в память
# Настройки для REST Framework # Настройки для REST Framework
REST_FRAMEWORK = { REST_FRAMEWORK = {
'DEFAULT_PERMISSION_CLASSES': [ 'DEFAULT_PERMISSION_CLASSES': [
'api_app.permissions.ApiTokenPermission', 'rest_framework.permissions.AllowAny', # Для тестирования, в продакшне лучше ограничить
] ]
} }
LOGGING = {
"version": 1,
"disable_existing_loggers": False,
"handlers": {"console": {"class": "logging.StreamHandler"}},
"root": {"handlers": ["console"], "level": "INFO"},
}
# как в существующей миграции — без новой миграции
DEFAULT_AUTO_FIELD = 'django.db.models.AutoField'

View file

@ -1,13 +0,0 @@
services:
api:
build: .
image: ghcr.io/edeev/api_processor:latest
env_file: .env
ports:
- "8000:8000"
volumes:
- data:/data
restart: unless-stopped
volumes:
data:

View file

@ -2,7 +2,7 @@
# Generated by the protocol buffer compiler. DO NOT EDIT! # Generated by the protocol buffer compiler. DO NOT EDIT!
# NO CHECKED-IN PROTOBUF GENCODE # NO CHECKED-IN PROTOBUF GENCODE
# source: text_service.proto # source: text_service.proto
# Protobuf Python Version: 7.35.1 # Protobuf Python Version: 5.29.0
"""Generated protocol buffer code.""" """Generated protocol buffer code."""
from google.protobuf import descriptor as _descriptor from google.protobuf import descriptor as _descriptor
from google.protobuf import descriptor_pool as _descriptor_pool from google.protobuf import descriptor_pool as _descriptor_pool
@ -11,9 +11,9 @@ from google.protobuf import symbol_database as _symbol_database
from google.protobuf.internal import builder as _builder from google.protobuf.internal import builder as _builder
_runtime_version.ValidateProtobufRuntimeVersion( _runtime_version.ValidateProtobufRuntimeVersion(
_runtime_version.Domain.PUBLIC, _runtime_version.Domain.PUBLIC,
7, 5,
35, 29,
1, 0,
'', '',
'text_service.proto' 'text_service.proto'
) )

View file

@ -5,7 +5,7 @@ import warnings
import text_service_pb2 as text__service__pb2 import text_service_pb2 as text__service__pb2
GRPC_GENERATED_VERSION = '1.84.0' GRPC_GENERATED_VERSION = '1.71.0'
GRPC_VERSION = grpc.__version__ GRPC_VERSION = grpc.__version__
_version_not_supported = False _version_not_supported = False
@ -18,14 +18,14 @@ except ImportError:
if _version_not_supported: if _version_not_supported:
raise RuntimeError( raise RuntimeError(
f'The grpc package installed is at version {GRPC_VERSION},' f'The grpc package installed is at version {GRPC_VERSION},'
+ ' but the generated code in text_service_pb2_grpc.py depends on' + f' but the generated code in text_service_pb2_grpc.py depends on'
+ f' grpcio>={GRPC_GENERATED_VERSION}.' + f' grpcio>={GRPC_GENERATED_VERSION}.'
+ f' Please upgrade your grpc module to grpcio>={GRPC_GENERATED_VERSION}' + f' Please upgrade your grpc module to grpcio>={GRPC_GENERATED_VERSION}'
+ f' or downgrade your generated code using grpcio-tools<={GRPC_VERSION}.' + f' or downgrade your generated code using grpcio-tools<={GRPC_VERSION}.'
) )
class TextProcessorStub: class TextProcessorStub(object):
"""Missing associated documentation comment in .proto file.""" """Missing associated documentation comment in .proto file."""
def __init__(self, channel): def __init__(self, channel):
@ -41,7 +41,7 @@ class TextProcessorStub:
_registered_method=True) _registered_method=True)
class TextProcessorServicer: class TextProcessorServicer(object):
"""Missing associated documentation comment in .proto file.""" """Missing associated documentation comment in .proto file."""
def ProcessText(self, request, context): def ProcessText(self, request, context):
@ -66,7 +66,7 @@ def add_TextProcessorServicer_to_server(servicer, server):
# This class is part of an EXPERIMENTAL API. # This class is part of an EXPERIMENTAL API.
class TextProcessor: class TextProcessor(object):
"""Missing associated documentation comment in .proto file.""" """Missing associated documentation comment in .proto file."""
@staticmethod @staticmethod

View file

@ -1,6 +0,0 @@
[pytest]
DJANGO_SETTINGS_MODULE = api_project.settings
testpaths = tests
env =
D:DJANGO_DEBUG=True
D:GRPC_SERVER=

View file

@ -1,7 +0,0 @@
-r requirements.txt
grpcio-tools==1.84.0 # только для пересборки proto/*_pb2*.py
pytest==8.4.2
pytest-django==4.11.1
ruff==0.14.0
fpdf2==2.8.4
pytest-env==1.1.5

View file

@ -1,9 +1,9 @@
Django==5.2.17 django==4.2.6
djangorestframework==3.18.1 djangorestframework==3.14.0
grpcio==1.84.0 grpcio==1.58.0
protobuf==7.36.2 grpcio-tools==1.58.0
pdfplumber==0.11.10 pdfplumber==0.10.2
python-docx==1.2.0 python-docx==0.8.11
Pillow==10.0.1
vosk==0.3.45 vosk==0.3.45
ffmpeg-python==0.2.0 ffmpeg-python==0.2.0
gunicorn==26.2.0

View file

@ -1,139 +0,0 @@
import io
import os
import shutil
import subprocess
from concurrent import futures
import docx
import grpc
import pytest
from django.core.files.uploadedfile import SimpleUploadedFile
from django.test import override_settings
from fpdf import FPDF
from rest_framework.test import APIClient
from api_app.grpc_client import client as grpc_client
from api_app.services.scan import extract_text_tables
import text_service_pb2
import text_service_pb2_grpc
pytestmark = pytest.mark.django_db
@pytest.fixture(autouse=True)
def media(tmp_path, settings):
settings.MEDIA_ROOT = str(tmp_path)
def make_docx():
d = docx.Document()
d.add_paragraph("Привет <script>alert(1)</script> & мир")
t = d.add_table(rows=2, cols=2)
t.cell(0, 0).text, t.cell(0, 1).text = "a", "b"
t.cell(1, 0).text, t.cell(1, 1).text = "1", "<2>"
buf = io.BytesIO()
d.save(buf)
return buf.getvalue()
def make_pdf():
pdf = FPDF()
pdf.add_page()
pdf.set_font("Helvetica", size=12)
pdf.cell(text="Hello <b>world</b> & co")
return bytes(pdf.output())
def test_docx_text_is_escaped(tmp_path):
path = tmp_path / "a.DOCX"
path.write_bytes(make_docx())
html = extract_text_tables(str(path))
assert "<p>Привет &lt;script&gt;alert(1)&lt;/script&gt; &amp; мир</p>" in html
assert "<pre>a,b\r\n1,&lt;2&gt;\r\n</pre>" in html
def test_pdf_upper_extension_and_escaping(tmp_path):
path = tmp_path / "a.PDF"
path.write_bytes(make_pdf())
assert "<p>Hello &lt;b&gt;world&lt;/b&gt; &amp; co</p>" in extract_text_tables(str(path))
def test_document_endpoint():
client = APIClient()
resp = client.post("/api/document-to-text/",
{"document": SimpleUploadedFile("r.docx", make_docx())}, format="multipart")
assert resp.status_code == 200
assert "&lt;script&gt;" in resp.json()["text"] and resp.json()["grpc_response"] is None
def test_validation_and_size_limit(settings):
client = APIClient()
assert client.post("/api/document-to-text/", {}, format="multipart").status_code == 400
bad = client.post("/api/document-to-text/", {"document": SimpleUploadedFile("a.txt", b"x")}, format="multipart")
assert bad.status_code == 400
settings.MAX_UPLOAD_SIZE = 10
big = client.post("/api/document-to-text/", {"document": SimpleUploadedFile("a.pdf", b"x" * 100)},
format="multipart")
assert big.status_code == 413
@override_settings(API_TOKEN="secret")
def test_api_token():
client = APIClient()
upload = {"document": SimpleUploadedFile("r.docx", make_docx())}
assert client.post("/api/document-to-text/", upload, format="multipart").status_code == 403
client.credentials(HTTP_AUTHORIZATION="Bearer secret")
upload = {"document": SimpleUploadedFile("r.docx", make_docx())}
assert client.post("/api/document-to-text/", upload, format="multipart").status_code == 200
class Upper(text_service_pb2_grpc.TextProcessorServicer):
def ProcessText(self, request, context):
return text_service_pb2.TextResponse(processed_text=request.text.upper(), success=True)
def test_grpc_real_server(settings):
server = grpc.server(futures.ThreadPoolExecutor(max_workers=1))
text_service_pb2_grpc.add_TextProcessorServicer_to_server(Upper(), server)
port = server.add_insecure_port("127.0.0.1:0")
server.start()
try:
settings.GRPC_SERVER = f"127.0.0.1:{port}"
assert grpc_client.send_to_grpc_server("привет") == {
"processed_text": "ПРИВЕТ", "success": True, "error": None}
finally:
server.stop(None)
def test_grpc_unavailable(settings):
settings.GRPC_SERVER = "127.0.0.1:1"
settings.GRPC_TIMEOUT = 2
result = grpc_client.send_to_grpc_server("x")
assert result["success"] is False and result["error"].startswith("gRPC")
@pytest.mark.skipif(not shutil.which("ffmpeg"), reason="нужен ffmpeg")
def test_audio_endpoint_converts_ogg(tmp_path, settings):
if not os.path.isdir(settings.VOSK_MODEL_PATH):
pytest.skip("нет модели Vosk")
ogg = tmp_path / "tone.ogg"
subprocess.run(["ffmpeg", "-loglevel", "error", "-f", "lavfi", "-i", "sine=frequency=440:duration=2",
"-c:a", "libvorbis", str(ogg)], check=True)
resp = APIClient().post("/api/audio-to-text/",
{"audio": SimpleUploadedFile("tone.ogg", ogg.read_bytes())}, format="multipart")
assert resp.status_code == 200 and isinstance(resp.json()["text"], str)
SAMPLE = os.path.join(os.path.dirname(os.path.dirname(__file__)), "examples", "sample.ogg")
@pytest.mark.skipif(not shutil.which("ffmpeg"), reason="нужен ffmpeg")
def test_sample_recognition_keeps_all_phrases(settings):
if not os.path.isdir(settings.VOSK_MODEL_PATH):
pytest.skip("нет модели Vosk")
from api_app.services.vosk_recognizer import recognize_speech
text = recognize_speech(SAMPLE)
# в записи несколько фраз: раньше оставалась только последняя
assert "проверка" in text and "десять" in text

13
тесты.txt Normal file
View file

@ -0,0 +1,13 @@
python manage.py runserver
------------------------------------------
Тест расширения для сканирования:
curl -X POST -F "document=@C:\Users\egord\Desktop\API EasyAccess\api_project\media\uploads\Деев Е.В. Резюме.pdf" http://localhost:8000/api/document-to-text/
------------------------------------------
Тест для аудио транскрипции:
curl -X POST -F audio=@"C:\Users\egord\Desktop\API EasyAccess\api_project\media\uploads\audio.ogg" http://localhost:8000/api/audio-to-text/