mirror of
https://github.com/EDeev/api_processor.git
synced 2026-10-07 20:49:34 +03:00
Тесты, CI, Docker с моделью Vosk и FFmpeg, лицензия MIT, README на русском и английском
Пример записи перенесён из media/ в examples/sample.ogg — на нём проверяется распознавание. db.sqlite3, media/ и models/ — в .gitignore.
This commit is contained in:
parent
b211b10c39
commit
479e0c96ad
14 changed files with 560 additions and 140 deletions
7
.dockerignore
Normal file
7
.dockerignore
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
.git
|
||||
.env
|
||||
media
|
||||
models
|
||||
db.sqlite3
|
||||
__pycache__
|
||||
tests
|
||||
9
.env.example
Normal file
9
.env.example
Normal file
|
|
@ -0,0 +1,9 @@
|
|||
# Обязательно в продакшене
|
||||
DJANGO_SECRET_KEY=change-me-to-a-long-random-string
|
||||
DJANGO_ALLOWED_HOSTS=localhost,127.0.0.1
|
||||
# DJANGO_DEBUG=True # для разработки
|
||||
|
||||
# Необязательно
|
||||
API_TOKEN= # если задан — запросы с заголовком Authorization: Bearer <токен>
|
||||
GRPC_SERVER= # адрес сервиса TextProcessor, например textproc:50051; пусто — не отправлять
|
||||
MAX_UPLOAD_SIZE_MB=50
|
||||
35
.github/workflows/ci.yml
vendored
Normal file
35
.github/workflows/ci.yml
vendored
Normal file
|
|
@ -0,0 +1,35 @@
|
|||
name: CI
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
pull_request:
|
||||
|
||||
jobs:
|
||||
test:
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
VOSK_MODEL_PATH: ${{ github.workspace }}/models/vosk-model-small-ru-0.22
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.12"
|
||||
- run: sudo apt-get update -qq && sudo apt-get install -y -qq ffmpeg
|
||||
- name: Модель Vosk (кэш)
|
||||
id: model
|
||||
uses: actions/cache@v4
|
||||
with:
|
||||
path: models
|
||||
key: vosk-model-small-ru-0.22
|
||||
- if: steps.model.outputs.cache-hit != 'true'
|
||||
run: |
|
||||
mkdir -p models
|
||||
curl -fsSL -o /tmp/model.zip https://alphacephei.com/vosk/models/vosk-model-small-ru-0.22.zip
|
||||
unzip -q /tmp/model.zip -d models
|
||||
- run: pip install -r requirements-dev.txt
|
||||
- run: ruff check --select E9,F,B --exclude proto .
|
||||
- run: python manage.py check && python manage.py makemigrations --check --dry-run
|
||||
env:
|
||||
DJANGO_DEBUG: "True"
|
||||
- run: pytest -q
|
||||
42
.github/workflows/docker.yml
vendored
Normal file
42
.github/workflows/docker.yml
vendored
Normal file
|
|
@ -0,0 +1,42 @@
|
|||
name: Docker
|
||||
|
||||
on:
|
||||
push:
|
||||
tags: ["v*"]
|
||||
workflow_dispatch:
|
||||
|
||||
jobs:
|
||||
image:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: docker/setup-buildx-action@v3
|
||||
- uses: docker/login-action@v3
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ github.actor }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
- uses: docker/login-action@v3
|
||||
with:
|
||||
registry: dcr.deev.su
|
||||
username: ${{ secrets.ZOT_USERNAME }}
|
||||
password: ${{ secrets.ZOT_PASSWORD }}
|
||||
- id: meta
|
||||
uses: docker/metadata-action@v5
|
||||
with:
|
||||
images: |
|
||||
ghcr.io/edeev/api_processor
|
||||
dcr.deev.su/edeev/api_processor
|
||||
tags: |
|
||||
type=semver,pattern={{version}}
|
||||
type=semver,pattern={{major}}.{{minor}}
|
||||
type=raw,value=latest
|
||||
- uses: docker/build-push-action@v6
|
||||
with:
|
||||
context: .
|
||||
push: true
|
||||
tags: ${{ steps.meta.outputs.tags }}
|
||||
labels: ${{ steps.meta.outputs.labels }}
|
||||
8
.gitignore
vendored
Normal file
8
.gitignore
vendored
Normal file
|
|
@ -0,0 +1,8 @@
|
|||
.env
|
||||
__pycache__/
|
||||
*.pyc
|
||||
.pytest_cache/
|
||||
db.sqlite3
|
||||
media/
|
||||
staticfiles/
|
||||
models/
|
||||
33
Dockerfile
Normal file
33
Dockerfile
Normal file
|
|
@ -0,0 +1,33 @@
|
|||
FROM python:3.12-slim
|
||||
|
||||
ENV PYTHONDONTWRITEBYTECODE=1 \
|
||||
PYTHONUNBUFFERED=1 \
|
||||
DJANGO_DB_PATH=/data/db.sqlite3 \
|
||||
DJANGO_MEDIA_ROOT=/data/media \
|
||||
VOSK_MODEL_PATH=/app/models/vosk-model-small-ru-0.22
|
||||
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends ffmpeg curl unzip \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
WORKDIR /app
|
||||
# офлайн-модель распознавания русской речи (~45 МБ)
|
||||
RUN mkdir -p models \
|
||||
&& curl -fsSL -o /tmp/model.zip https://alphacephei.com/vosk/models/vosk-model-small-ru-0.22.zip \
|
||||
&& unzip -q /tmp/model.zip -d models \
|
||||
&& rm /tmp/model.zip
|
||||
|
||||
COPY requirements.txt .
|
||||
RUN pip install --no-cache-dir -r requirements.txt
|
||||
|
||||
COPY manage.py ./
|
||||
COPY api_project/ api_project/
|
||||
COPY api_app/ api_app/
|
||||
COPY proto/ proto/
|
||||
|
||||
RUN useradd --create-home --uid 1000 app && mkdir -p /data && chown -R app:app /app /data
|
||||
USER app
|
||||
VOLUME ["/data"]
|
||||
EXPOSE 8000
|
||||
|
||||
CMD ["sh", "-c", "python manage.py migrate --noinput && gunicorn api_project.wsgi:application --bind 0.0.0.0:8000 --workers 2 --timeout 300"]
|
||||
21
LICENSE
Normal file
21
LICENSE
Normal file
|
|
@ -0,0 +1,21 @@
|
|||
MIT License
|
||||
|
||||
Copyright (c) 2025 Egor Deev
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
129
README.en.md
Normal file
129
README.en.md
Normal file
|
|
@ -0,0 +1,129 @@
|
|||
# API Processor
|
||||
|
||||
[Русский](README.md) · **English**
|
||||
|
||||
[](https://github.com/EDeev/api_processor/actions/workflows/ci.yml)
|
||||
[](https://github.com/EDeev/api_processor/actions/workflows/docker.yml)
|
||||
[](LICENSE)
|
||||
|
||||
A REST API that turns audio and documents into text:
|
||||
- speech is recognized offline with a Vosk model;
|
||||
- text, tables and images are extracted from PDF and DOCX.
|
||||
|
||||
The result can be forwarded over gRPC to a text processing service. Speech recognition targets Russian.
|
||||
|
||||
**Status:** personal project, completed
|
||||
|
||||
**Stack:** Python 3.12 · Django 5.2 · Django REST Framework · Vosk · FFmpeg · pdfplumber · python-docx · gRPC · Docker
|
||||
|
||||
## Features
|
||||
|
||||
- **Audio → text.** Any format FFmpeg reads (OGG, MP3, M4A, WAV…). Audio is converted to 16 kHz mono with
|
||||
noise reduction and loudness normalization first. Recognition is offline, no external services.
|
||||
- **Document → HTML.** PDF and DOCX: paragraphs in `<p>`, tables as CSV in `<pre>`, images as base64. The
|
||||
document text is escaped, so the result is safe to render in a browser.
|
||||
- **gRPC.** The text is sent to the `TextProcessor` service (`proto/text_service.proto`), and its reply is
|
||||
returned in `grpc_response`. If no service address is set, the step is skipped.
|
||||
- Optional access token and a file size limit (50 MB by default).
|
||||
|
||||
## API
|
||||
|
||||
```bash
|
||||
curl -F audio=@examples/sample.ogg http://localhost:8000/api/audio-to-text/
|
||||
curl -F document=@report.pdf http://localhost:8000/api/document-to-text/
|
||||
```
|
||||
|
||||
```json
|
||||
{
|
||||
"text": "раз два три проверка перевода голоса текст насколько качественно она работает один два три четыре пять шесть семь восемь девять десять",
|
||||
"grpc_response": null
|
||||
}
|
||||
```
|
||||
|
||||
With `API_TOKEN` set, add `Authorization: Bearer <token>`. Errors come as `{"error": "..."}`:
|
||||
|
||||
| Code | When |
|
||||
|---|---|
|
||||
| 400 | no file or unsupported document format |
|
||||
| 403 | wrong token |
|
||||
| 413 | file over the limit |
|
||||
| 422 | audio could not be read |
|
||||
|
||||
## Running
|
||||
|
||||
```bash
|
||||
git clone https://github.com/EDeev/api_processor.git && cd api_processor
|
||||
cp .env.example .env # set DJANGO_SECRET_KEY
|
||||
docker compose up -d # API at http://localhost:8000
|
||||
```
|
||||
|
||||
The Vosk model and FFmpeg are inside the image. Prebuilt image: `docker pull ghcr.io/edeev/api_processor` or
|
||||
`docker pull dcr.deev.su/edeev/api_processor`.
|
||||
|
||||
Without Docker you need:
|
||||
- FFmpeg in `PATH`;
|
||||
- the [vosk-model-small-ru-0.22](https://alphacephei.com/vosk/models) model unpacked into `models/`.
|
||||
|
||||
Then:
|
||||
|
||||
```bash
|
||||
pip install -r requirements.txt
|
||||
DJANGO_DEBUG=True python manage.py migrate
|
||||
DJANGO_DEBUG=True python manage.py runserver
|
||||
```
|
||||
|
||||
| Variable | Purpose |
|
||||
|---|---|
|
||||
| `DJANGO_SECRET_KEY` | secret key, required unless `DJANGO_DEBUG=True` |
|
||||
| `DJANGO_ALLOWED_HOSTS` | comma-separated domains |
|
||||
| `API_TOKEN` | if set, access requires the token |
|
||||
| `GRPC_SERVER`, `GRPC_TIMEOUT` | TextProcessor address and timeout, s |
|
||||
| `MAX_UPLOAD_SIZE_MB` | file size limit |
|
||||
| `VOSK_MODEL_PATH`, `FFMPEG_BINARY` | paths to the model and ffmpeg |
|
||||
|
||||
## How it works
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
C[Client] -->|multipart| V[DRF APIView]
|
||||
V -->|audio| F[FFmpeg: 16 kHz mono, denoise] --> K[Vosk]
|
||||
V -->|PDF / DOCX| S[pdfplumber / python-docx]
|
||||
K --> T[text]
|
||||
S --> T
|
||||
T -->|gRPC ProcessText| G[TextProcessor]
|
||||
T --> R[JSON response]
|
||||
```
|
||||
|
||||
The Vosk model is loaded once per process. Each request gets its own temporary file. Uploaded files and
|
||||
results are stored in SQLite and `media/`.
|
||||
|
||||
## Development
|
||||
|
||||
```bash
|
||||
pip install -r requirements-dev.txt
|
||||
ruff check --select E9,F,B --exclude proto . && pytest
|
||||
```
|
||||
|
||||
What the tests cover:
|
||||
- recognition of the whole sample, with every phrase;
|
||||
- OGG conversion;
|
||||
- PDF and DOCX extraction with escaping;
|
||||
- limits and the token;
|
||||
- a real gRPC server and an unavailable one.
|
||||
|
||||
The Docker image is built on `v*` tags and published to GitHub Packages and `dcr.deev.su`.
|
||||
|
||||
## License
|
||||
|
||||
MIT — see [LICENSE](LICENSE). The Vosk model is licensed under Apache 2.0.
|
||||
|
||||
## Author
|
||||
|
||||
**Egor Deev** — [GitHub](https://github.com/EDeev) · [Telegram](https://t.me/DeevEgor) · [egor@deev.space](mailto:egor@deev.space)
|
||||
|
||||
---
|
||||
|
||||
<div align="center">
|
||||
<sub>⭐ If you find this project useful, give it a star on GitHub!</sub>
|
||||
<p><sub>Made with ❤️ — <a href="https://deev.space">deev.space</a></sub></p>
|
||||
</div>
|
||||
237
README.md
237
README.md
|
|
@ -1,171 +1,142 @@
|
|||
# API Processor
|
||||
|
||||
**Django REST API для обработки аудио и документов с интеграцией gRPC сервисов**
|
||||
**Русский** · [English](README.en.md)
|
||||
|
||||
API Processor — это мощное решение для автоматической обработки мультимедийного контента. Система выполняет транскрибацию аудиофайлов в текст и извлечение данных из PDF/DOCX документов с последующей отправкой результатов на внешний gRPC сервер для дополнительной обработки.
|
||||
[](https://github.com/EDeev/api_processor/actions/workflows/ci.yml)
|
||||
[](https://github.com/EDeev/api_processor/actions/workflows/docker.yml)
|
||||
[](LICENSE)
|
||||
|
||||
## 🚀 Возможности
|
||||
REST API, которое превращает аудио и документы в текст:
|
||||
- речь распознаётся офлайн моделью Vosk;
|
||||
- из PDF и DOCX извлекаются текст, таблицы и изображения.
|
||||
|
||||
- **Аудио транскрипция**: Преобразование аудиофайлов в текст с использованием модели Vosk
|
||||
- **Обработка документов**: Извлечение текста, таблиц и изображений из PDF и DOCX файлов
|
||||
- **gRPC интеграция**: Автоматическая отправка обработанного текста на внешний сервер
|
||||
- **RESTful API**: Простой и понятный интерфейс для взаимодействия
|
||||
- **Поддержка форматов**: Audio (WAV, OGG и др.), PDF, DOCX
|
||||
Результат можно отправить дальше по gRPC — на сервис обработки текста.
|
||||
|
||||
## 🛠 Технологический стек
|
||||
**Статус:** личный проект, завершён
|
||||
|
||||
### Backend
|
||||
- **Django 4.2.6** - веб-фреймворк
|
||||
- **Django REST Framework 3.14.0** - API framework
|
||||
- **Python** - основной язык разработки
|
||||
**Стек:** Python 3.12 · Django 5.2 · Django REST Framework · Vosk · FFmpeg · pdfplumber · python-docx · gRPC · Docker
|
||||
|
||||
### Обработка контента
|
||||
- **Vosk 0.3.45** - распознавание речи
|
||||
- **FFmpeg Python 0.2.0** - конвертация аудио
|
||||
- **pdfplumber 0.10.2** - извлечение данных из PDF
|
||||
- **python-docx 0.8.11** - работа с DOCX файлами
|
||||
- **Pillow 10.0.1** - обработка изображений
|
||||
## Возможности
|
||||
|
||||
### Коммуникация
|
||||
- **gRPC 1.58.0** - межсервисное взаимодействие
|
||||
- **Protocol Buffers** - сериализация данных
|
||||
- **Аудио → текст.** Любой формат, который читает FFmpeg (OGG, MP3, M4A, WAV…). Перед распознаванием
|
||||
звук приводится к 16 кГц моно с шумоподавлением и нормализацией громкости. Распознавание офлайн, без
|
||||
внешних сервисов.
|
||||
- **Документ → HTML.** PDF и DOCX: абзацы — в `<p>`, таблицы — CSV в `<pre>`, изображения — в base64.
|
||||
Текст документа экранируется, поэтому результат безопасно показывать в браузере.
|
||||
- **gRPC.** Распознанный текст уходит на сервис `TextProcessor` (`proto/text_service.proto`), его ответ
|
||||
возвращается в поле `grpc_response`. Если адрес сервиса не задан, шаг пропускается.
|
||||
- Необязательный токен доступа и ограничение размера файла (по умолчанию 50 МБ).
|
||||
|
||||
### База данных
|
||||
- **SQLite** - локальное хранение метаданных файлов
|
||||
## API
|
||||
|
||||
## 📋 Требования
|
||||
```bash
|
||||
curl -F audio=@examples/sample.ogg http://localhost:8000/api/audio-to-text/
|
||||
curl -F document=@report.pdf http://localhost:8000/api/document-to-text/
|
||||
```
|
||||
|
||||
- Python 3.8+
|
||||
- FFmpeg (для конвертации аудио)
|
||||
- Модель Vosk для русского языка
|
||||
```json
|
||||
{
|
||||
"text": "раз два три проверка перевода голоса текст насколько качественно она работает один два три четыре пять шесть семь восемь девять десять",
|
||||
"grpc_response": null
|
||||
}
|
||||
```
|
||||
|
||||
## ⚡ Быстрый старт
|
||||
Если задан `API_TOKEN`, добавьте заголовок `Authorization: Bearer <токен>`. Ошибки приходят как
|
||||
`{"error": "..."}`:
|
||||
|
||||
### Установка зависимостей
|
||||
| Код | Когда |
|
||||
|---|---|
|
||||
| 400 | нет файла или неподдерживаемый формат документа |
|
||||
| 403 | неверный токен |
|
||||
| 413 | файл больше лимита |
|
||||
| 422 | аудио не удалось прочитать |
|
||||
|
||||
## Запуск
|
||||
|
||||
```bash
|
||||
git clone https://github.com/EDeev/api_processor.git && cd api_processor
|
||||
cp .env.example .env # задайте DJANGO_SECRET_KEY
|
||||
docker compose up -d # API на http://localhost:8000
|
||||
```
|
||||
|
||||
Модель Vosk и FFmpeg уже внутри образа. Готовый образ: `docker pull ghcr.io/edeev/api_processor` или
|
||||
`docker pull dcr.deev.su/edeev/api_processor`.
|
||||
|
||||
Без Docker нужны:
|
||||
- FFmpeg в `PATH`;
|
||||
- модель [vosk-model-small-ru-0.22](https://alphacephei.com/vosk/models), распакованная в `models/`.
|
||||
|
||||
Затем:
|
||||
|
||||
```bash
|
||||
pip install -r requirements.txt
|
||||
DJANGO_DEBUG=True python manage.py migrate
|
||||
DJANGO_DEBUG=True python manage.py runserver
|
||||
```
|
||||
|
||||
### Настройка модели Vosk
|
||||
| Переменная | Назначение |
|
||||
|---|---|
|
||||
| `DJANGO_SECRET_KEY` | секретный ключ, обязателен без `DJANGO_DEBUG=True` |
|
||||
| `DJANGO_ALLOWED_HOSTS` | домены через запятую |
|
||||
| `API_TOKEN` | если задан — доступ только с токеном |
|
||||
| `GRPC_SERVER`, `GRPC_TIMEOUT` | адрес сервиса TextProcessor и таймаут, с |
|
||||
| `MAX_UPLOAD_SIZE_MB` | лимит размера файла |
|
||||
| `VOSK_MODEL_PATH`, `FFMPEG_BINARY` | путь к модели и к ffmpeg |
|
||||
|
||||
1. Скачайте модель `vosk-model-small-ru-0.22`
|
||||
2. Разместите в папке `models/vosk-model-small-ru-0.22`
|
||||
## Как устроено
|
||||
|
||||
### Настройка FFmpeg
|
||||
```mermaid
|
||||
flowchart LR
|
||||
C[Клиент] -->|multipart| V[DRF APIView]
|
||||
V -->|аудио| F[FFmpeg: 16 кГц моно, шумоподавление] --> K[Vosk]
|
||||
V -->|PDF / DOCX| S[pdfplumber / python-docx]
|
||||
K --> T[текст]
|
||||
S --> T
|
||||
T -->|gRPC ProcessText| G[TextProcessor]
|
||||
T --> R[JSON-ответ]
|
||||
```
|
||||
|
||||
1. Скачайте FFmpeg
|
||||
2. Разместите в `models/ffmpeg/bin/ffmpeg.exe`
|
||||
```
|
||||
api_app/views.py эндпоинты и проверки
|
||||
api_app/services/vosk_recognizer.py конвертация FFmpeg и распознавание Vosk
|
||||
api_app/services/scan.py извлечение из PDF и DOCX
|
||||
api_app/grpc_client/client.py клиент TextProcessor
|
||||
proto/ описание gRPC-сервиса и сгенерированный код
|
||||
```
|
||||
|
||||
### Запуск сервера
|
||||
Модель Vosk загружается один раз на процесс. У каждого запроса свой временный файл. Загруженные файлы и
|
||||
результат сохраняются в SQLite и `media/`.
|
||||
|
||||
## Разработка
|
||||
|
||||
```bash
|
||||
python manage.py migrate
|
||||
python manage.py runserver
|
||||
pip install -r requirements-dev.txt
|
||||
ruff check --select E9,F,B --exclude proto . && pytest
|
||||
```
|
||||
|
||||
## 📖 API Endpoints
|
||||
Что проверяют тесты:
|
||||
- распознавание образца целиком, со всеми фразами;
|
||||
- конвертацию OGG;
|
||||
- извлечение из PDF и DOCX с экранированием;
|
||||
- лимиты и токен;
|
||||
- работу с настоящим gRPC-сервером и его недоступность.
|
||||
|
||||
### Транскрипция аудио
|
||||
```http
|
||||
POST /api/audio-to-text/
|
||||
Content-Type: multipart/form-data
|
||||
Docker-образ собирается по тегу `v*` и публикуется в GitHub Packages и `dcr.deev.su`.
|
||||
|
||||
audio: <audio_file>
|
||||
```
|
||||
Код gRPC пересобирается так:
|
||||
`python -m grpc_tools.protoc -Iproto --python_out=proto --grpc_python_out=proto proto/text_service.proto`.
|
||||
|
||||
**Ответ:**
|
||||
```json
|
||||
{
|
||||
"text": "Распознанный текст из аудио",
|
||||
"grpc_response": {
|
||||
"processed_text": "Обработанный текст",
|
||||
"success": true,
|
||||
"error": null
|
||||
}
|
||||
}
|
||||
```
|
||||
## Лицензия
|
||||
|
||||
### Обработка документов
|
||||
```http
|
||||
POST /api/document-to-text/
|
||||
Content-Type: multipart/form-data
|
||||
MIT — см. [LICENSE](LICENSE). Модель Vosk распространяется под Apache 2.0.
|
||||
|
||||
document: <pdf_or_docx_file>
|
||||
```
|
||||
## Автор
|
||||
|
||||
**Ответ:**
|
||||
```json
|
||||
{
|
||||
"text": "<p>Извлеченный текст</p><pre>таблица,данные</pre>",
|
||||
"grpc_response": {
|
||||
"processed_text": "Обработанный текст",
|
||||
"success": true,
|
||||
"error": null
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
## 🧪 Примеры использования
|
||||
|
||||
### cURL команды
|
||||
|
||||
**Транскрипция аудио:**
|
||||
```bash
|
||||
curl -X POST -F "audio=@audio.ogg" http://localhost:8000/api/audio-to-text/
|
||||
```
|
||||
|
||||
**Обработка документа:**
|
||||
```bash
|
||||
curl -X POST -F "document=@document.pdf" http://localhost:8000/api/document-to-text/
|
||||
```
|
||||
|
||||
## ⚙️ Конфигурация
|
||||
|
||||
### gRPC настройки
|
||||
По умолчанию система подключается к gRPC серверу на `localhost:50051`. Для изменения адреса отредактируйте `api_app/grpc_client/client.py`.
|
||||
|
||||
### Модели и пути
|
||||
Пути к моделям и исполняемым файлам настраиваются в `api_app/services/vosk_recognizer.py`:
|
||||
- `MODEL_PATH` - путь к модели Vosk
|
||||
- `FFMPEG_PATH` - путь к исполняемому файлу FFmpeg
|
||||
|
||||
## 📁 Структура проекта
|
||||
|
||||
```
|
||||
api_processor/
|
||||
├── api_app/ # Основное приложение
|
||||
│ ├── grpc_client/ # gRPC клиент
|
||||
│ ├── services/ # Сервисы обработки
|
||||
│ ├── migrations/ # Миграции БД
|
||||
│ ├── models.py # Модели данных
|
||||
│ └── views.py # API эндпоинты
|
||||
├── api_project/ # Настройки Django
|
||||
├── proto/ # Protocol Buffers схемы
|
||||
├── models/ # Модели и исполняемые файлы
|
||||
└── requirements.txt # Зависимости
|
||||
```
|
||||
|
||||
## 🔧 Особенности реализации
|
||||
|
||||
- **Автоматическая конвертация**: Аудиофайлы автоматически конвертируются в формат WAV 16kHz
|
||||
- **Извлечение изображений**: Из документов извлекаются изображения в формате base64
|
||||
- **Обработка таблиц**: Таблицы сохраняются в CSV формате
|
||||
- **Fallback система**: При недоступности gRPC сервера используются заглушки
|
||||
|
||||
## 📄 Лицензия
|
||||
|
||||
Этот проект разработан для образовательных и исследовательских целей.
|
||||
|
||||
## 🤝 Разработчик
|
||||
|
||||
**Деев Егор Викторович** - Backend Developer
|
||||
- GitHub: [@EDeev](https://github.com/EDeev)
|
||||
- Email: egor@deev.space
|
||||
- Telegram: [@Egor_Deev](https://t.me/Egor_Deev)
|
||||
**Деев Егор Викторович** — [GitHub](https://github.com/EDeev) · [Telegram](https://t.me/DeevEgor) · [egor@deev.space](mailto:egor@deev.space)
|
||||
|
||||
---
|
||||
|
||||
<div align="center">
|
||||
<p><sub>Создано с ❤️ от вашего дорогого - deev.space ©</sub></p>
|
||||
<sub>⭐ Если проект оказался полезным, поставьте звёздочку на GitHub!</sub>
|
||||
<p><sub>Сделано с ❤️ — <a href="https://deev.space">deev.space</a></sub></p>
|
||||
</div>
|
||||
|
|
|
|||
13
compose.yaml
Normal file
13
compose.yaml
Normal file
|
|
@ -0,0 +1,13 @@
|
|||
services:
|
||||
api:
|
||||
build: .
|
||||
image: ghcr.io/edeev/api_processor:latest
|
||||
env_file: .env
|
||||
ports:
|
||||
- "8000:8000"
|
||||
volumes:
|
||||
- data:/data
|
||||
restart: unless-stopped
|
||||
|
||||
volumes:
|
||||
data:
|
||||
6
pytest.ini
Normal file
6
pytest.ini
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
[pytest]
|
||||
DJANGO_SETTINGS_MODULE = api_project.settings
|
||||
testpaths = tests
|
||||
env =
|
||||
D:DJANGO_DEBUG=True
|
||||
D:GRPC_SERVER=
|
||||
7
requirements-dev.txt
Normal file
7
requirements-dev.txt
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
-r requirements.txt
|
||||
grpcio-tools==1.84.0 # только для пересборки proto/*_pb2*.py
|
||||
pytest==8.4.2
|
||||
pytest-django==4.11.1
|
||||
ruff==0.14.0
|
||||
fpdf2==2.8.4
|
||||
pytest-env==1.1.5
|
||||
|
|
@ -1,9 +1,9 @@
|
|||
django==4.2.6
|
||||
djangorestframework==3.14.0
|
||||
grpcio==1.58.0
|
||||
grpcio-tools==1.58.0
|
||||
pdfplumber==0.10.2
|
||||
python-docx==0.8.11
|
||||
Pillow==10.0.1
|
||||
Django==5.2.17
|
||||
djangorestframework==3.18.1
|
||||
grpcio==1.84.0
|
||||
protobuf==7.36.2
|
||||
pdfplumber==0.11.10
|
||||
python-docx==1.2.0
|
||||
vosk==0.3.45
|
||||
ffmpeg-python==0.2.0
|
||||
gunicorn==26.2.0
|
||||
|
|
|
|||
139
tests/test_api.py
Normal file
139
tests/test_api.py
Normal file
|
|
@ -0,0 +1,139 @@
|
|||
import io
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
from concurrent import futures
|
||||
|
||||
import docx
|
||||
import grpc
|
||||
import pytest
|
||||
from django.core.files.uploadedfile import SimpleUploadedFile
|
||||
from django.test import override_settings
|
||||
from fpdf import FPDF
|
||||
from rest_framework.test import APIClient
|
||||
|
||||
from api_app.grpc_client import client as grpc_client
|
||||
from api_app.services.scan import extract_text_tables
|
||||
|
||||
import text_service_pb2
|
||||
import text_service_pb2_grpc
|
||||
|
||||
pytestmark = pytest.mark.django_db
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def media(tmp_path, settings):
|
||||
settings.MEDIA_ROOT = str(tmp_path)
|
||||
|
||||
|
||||
def make_docx():
|
||||
d = docx.Document()
|
||||
d.add_paragraph("Привет <script>alert(1)</script> & мир")
|
||||
t = d.add_table(rows=2, cols=2)
|
||||
t.cell(0, 0).text, t.cell(0, 1).text = "a", "b"
|
||||
t.cell(1, 0).text, t.cell(1, 1).text = "1", "<2>"
|
||||
buf = io.BytesIO()
|
||||
d.save(buf)
|
||||
return buf.getvalue()
|
||||
|
||||
|
||||
def make_pdf():
|
||||
pdf = FPDF()
|
||||
pdf.add_page()
|
||||
pdf.set_font("Helvetica", size=12)
|
||||
pdf.cell(text="Hello <b>world</b> & co")
|
||||
return bytes(pdf.output())
|
||||
|
||||
|
||||
def test_docx_text_is_escaped(tmp_path):
|
||||
path = tmp_path / "a.DOCX"
|
||||
path.write_bytes(make_docx())
|
||||
html = extract_text_tables(str(path))
|
||||
assert "<p>Привет <script>alert(1)</script> & мир</p>" in html
|
||||
assert "<pre>a,b\r\n1,<2>\r\n</pre>" in html
|
||||
|
||||
|
||||
def test_pdf_upper_extension_and_escaping(tmp_path):
|
||||
path = tmp_path / "a.PDF"
|
||||
path.write_bytes(make_pdf())
|
||||
assert "<p>Hello <b>world</b> & co</p>" in extract_text_tables(str(path))
|
||||
|
||||
|
||||
def test_document_endpoint():
|
||||
client = APIClient()
|
||||
resp = client.post("/api/document-to-text/",
|
||||
{"document": SimpleUploadedFile("r.docx", make_docx())}, format="multipart")
|
||||
assert resp.status_code == 200
|
||||
assert "<script>" in resp.json()["text"] and resp.json()["grpc_response"] is None
|
||||
|
||||
|
||||
def test_validation_and_size_limit(settings):
|
||||
client = APIClient()
|
||||
assert client.post("/api/document-to-text/", {}, format="multipart").status_code == 400
|
||||
bad = client.post("/api/document-to-text/", {"document": SimpleUploadedFile("a.txt", b"x")}, format="multipart")
|
||||
assert bad.status_code == 400
|
||||
settings.MAX_UPLOAD_SIZE = 10
|
||||
big = client.post("/api/document-to-text/", {"document": SimpleUploadedFile("a.pdf", b"x" * 100)},
|
||||
format="multipart")
|
||||
assert big.status_code == 413
|
||||
|
||||
|
||||
@override_settings(API_TOKEN="secret")
|
||||
def test_api_token():
|
||||
client = APIClient()
|
||||
upload = {"document": SimpleUploadedFile("r.docx", make_docx())}
|
||||
assert client.post("/api/document-to-text/", upload, format="multipart").status_code == 403
|
||||
client.credentials(HTTP_AUTHORIZATION="Bearer secret")
|
||||
upload = {"document": SimpleUploadedFile("r.docx", make_docx())}
|
||||
assert client.post("/api/document-to-text/", upload, format="multipart").status_code == 200
|
||||
|
||||
|
||||
class Upper(text_service_pb2_grpc.TextProcessorServicer):
|
||||
def ProcessText(self, request, context):
|
||||
return text_service_pb2.TextResponse(processed_text=request.text.upper(), success=True)
|
||||
|
||||
|
||||
def test_grpc_real_server(settings):
|
||||
server = grpc.server(futures.ThreadPoolExecutor(max_workers=1))
|
||||
text_service_pb2_grpc.add_TextProcessorServicer_to_server(Upper(), server)
|
||||
port = server.add_insecure_port("127.0.0.1:0")
|
||||
server.start()
|
||||
try:
|
||||
settings.GRPC_SERVER = f"127.0.0.1:{port}"
|
||||
assert grpc_client.send_to_grpc_server("привет") == {
|
||||
"processed_text": "ПРИВЕТ", "success": True, "error": None}
|
||||
finally:
|
||||
server.stop(None)
|
||||
|
||||
|
||||
def test_grpc_unavailable(settings):
|
||||
settings.GRPC_SERVER = "127.0.0.1:1"
|
||||
settings.GRPC_TIMEOUT = 2
|
||||
result = grpc_client.send_to_grpc_server("x")
|
||||
assert result["success"] is False and result["error"].startswith("gRPC")
|
||||
|
||||
|
||||
@pytest.mark.skipif(not shutil.which("ffmpeg"), reason="нужен ffmpeg")
|
||||
def test_audio_endpoint_converts_ogg(tmp_path, settings):
|
||||
if not os.path.isdir(settings.VOSK_MODEL_PATH):
|
||||
pytest.skip("нет модели Vosk")
|
||||
ogg = tmp_path / "tone.ogg"
|
||||
subprocess.run(["ffmpeg", "-loglevel", "error", "-f", "lavfi", "-i", "sine=frequency=440:duration=2",
|
||||
"-c:a", "libvorbis", str(ogg)], check=True)
|
||||
resp = APIClient().post("/api/audio-to-text/",
|
||||
{"audio": SimpleUploadedFile("tone.ogg", ogg.read_bytes())}, format="multipart")
|
||||
assert resp.status_code == 200 and isinstance(resp.json()["text"], str)
|
||||
|
||||
|
||||
SAMPLE = os.path.join(os.path.dirname(os.path.dirname(__file__)), "examples", "sample.ogg")
|
||||
|
||||
|
||||
@pytest.mark.skipif(not shutil.which("ffmpeg"), reason="нужен ffmpeg")
|
||||
def test_sample_recognition_keeps_all_phrases(settings):
|
||||
if not os.path.isdir(settings.VOSK_MODEL_PATH):
|
||||
pytest.skip("нет модели Vosk")
|
||||
from api_app.services.vosk_recognizer import recognize_speech
|
||||
|
||||
text = recognize_speech(SAMPLE)
|
||||
# в записи несколько фраз: раньше оставалась только последняя
|
||||
assert "проверка" in text and "десять" in text
|
||||
Loading…
Add table
Reference in a new issue