mirror of
https://github.com/EDeev/converterbot.git
synced 2026-10-07 20:49:33 +03:00
Тесты, CI, публикация на PyPI и Docker по тегу, лицензия MIT, README на русском и английском
Тесты проверяют номер страницы, шрифт заголовков, списки, структурные заголовки, таблицы, CLI, защиту от zip-бомбы и разбор архива. По тегу v* пакет md2gost уходит на PyPI, образ бота — в GHCR и dcr.deev.su. В requirements.txt убраны неиспользуемые aiohttp, aiofiles, certifi.
This commit is contained in:
parent
911b6ce233
commit
5f9ef626f1
15 changed files with 467 additions and 148 deletions
2
.env.example
Normal file
2
.env.example
Normal file
|
|
@ -0,0 +1,2 @@
|
|||
# Токен бота от @BotFather
|
||||
BOT_TOKEN=123456:your-token
|
||||
27
.github/workflows/ci.yml
vendored
Normal file
27
.github/workflows/ci.yml
vendored
Normal file
|
|
@ -0,0 +1,27 @@
|
|||
name: CI
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
pull_request:
|
||||
|
||||
jobs:
|
||||
test:
|
||||
runs-on: ubuntu-latest
|
||||
strategy:
|
||||
matrix:
|
||||
python-version: ["3.9", "3.12", "3.13"]
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: ${{ matrix.python-version }}
|
||||
- run: pip install pytest==8.4.2 ruff==0.14.0 build python-docx
|
||||
- run: ruff check --select E9,F,B .
|
||||
- name: Тесты пакета md2gost
|
||||
run: pytest -q tests/test_md2gost.py
|
||||
- name: Тесты бота
|
||||
if: matrix.python-version != '3.9'
|
||||
run: pip install -r requirements.txt && pytest -q tests/test_bot_helpers.py
|
||||
- name: Сборка пакета
|
||||
run: python -m build && pip install dist/*.whl && md2gost --version
|
||||
61
.github/workflows/release.yml
vendored
Normal file
61
.github/workflows/release.yml
vendored
Normal file
|
|
@ -0,0 +1,61 @@
|
|||
name: Release
|
||||
|
||||
on:
|
||||
push:
|
||||
tags: ["v*"]
|
||||
|
||||
jobs:
|
||||
pypi:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: write
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: "3.12"
|
||||
- run: pip install build
|
||||
- run: python -m build
|
||||
- name: Публикация md2gost на PyPI
|
||||
uses: pypa/gh-action-pypi-publish@release/v1
|
||||
with:
|
||||
password: ${{ secrets.PYPI_API_TOKEN }}
|
||||
- name: Пакет в релиз GitHub
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
run: gh release upload "$GITHUB_REF_NAME" dist/* --clobber || true
|
||||
|
||||
image:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: docker/setup-buildx-action@v3
|
||||
- uses: docker/login-action@v3
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ github.actor }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
- uses: docker/login-action@v3
|
||||
with:
|
||||
registry: dcr.deev.su
|
||||
username: ${{ secrets.ZOT_USERNAME }}
|
||||
password: ${{ secrets.ZOT_PASSWORD }}
|
||||
- id: meta
|
||||
uses: docker/metadata-action@v5
|
||||
with:
|
||||
images: |
|
||||
ghcr.io/edeev/my_converterbot
|
||||
dcr.deev.su/edeev/my_converterbot
|
||||
tags: |
|
||||
type=semver,pattern={{version}}
|
||||
type=semver,pattern={{major}}.{{minor}}
|
||||
type=raw,value=latest
|
||||
- uses: docker/build-push-action@v6
|
||||
with:
|
||||
context: .
|
||||
push: true
|
||||
tags: ${{ steps.meta.outputs.tags }}
|
||||
labels: ${{ steps.meta.outputs.labels }}
|
||||
7
.gitignore
vendored
Normal file
7
.gitignore
vendored
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
.env
|
||||
__pycache__/
|
||||
*.pyc
|
||||
.pytest_cache/
|
||||
dist/
|
||||
build/
|
||||
*.egg-info/
|
||||
15
Dockerfile
Normal file
15
Dockerfile
Normal file
|
|
@ -0,0 +1,15 @@
|
|||
FROM python:3.12-slim
|
||||
|
||||
ENV PYTHONDONTWRITEBYTECODE=1 \
|
||||
PYTHONUNBUFFERED=1
|
||||
|
||||
WORKDIR /app
|
||||
COPY requirements.txt .
|
||||
RUN pip install --no-cache-dir -r requirements.txt
|
||||
|
||||
COPY bot.py rep_to_txt.py ./
|
||||
COPY md2gost/ md2gost/
|
||||
RUN useradd --create-home --uid 1000 app && chown -R app:app /app
|
||||
USER app
|
||||
|
||||
CMD ["python", "bot.py"]
|
||||
21
LICENSE
Normal file
21
LICENSE
Normal file
|
|
@ -0,0 +1,21 @@
|
|||
MIT License
|
||||
|
||||
Copyright (c) 2025 Egor Deev
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
92
README.en.md
Normal file
92
README.en.md
Normal file
|
|
@ -0,0 +1,92 @@
|
|||
# My Converter Bot · md2gost
|
||||
|
||||
[Русский](https://github.com/EDeev/my_converterbot/blob/main/README.md) · **English**
|
||||
|
||||
[](https://github.com/EDeev/my_converterbot/actions/workflows/ci.yml)
|
||||
[](https://pypi.org/project/md2gost/)
|
||||
[](https://pypi.org/project/md2gost/)
|
||||
[](https://github.com/EDeev/my_converterbot/blob/main/LICENSE)
|
||||
|
||||
A Markdown to DOCX converter that follows GOST 7.32-2017 (the Russian standard for research and student
|
||||
reports), plus a Telegram bot for study routine: send a `.md` and get a report ready to submit, send a
|
||||
project `.zip` and get a `.txt` with the folder tree and file contents. The converter is installable on its
|
||||
own as the `md2gost` package on PyPI. The bot speaks Russian.
|
||||
|
||||
**Status:** personal project, maintained · bot [@my_convbot](https://t.me/my_convbot) ·
|
||||
package [md2gost](https://pypi.org/project/md2gost/)
|
||||
|
||||

|
||||
|
||||
**Stack:** Python 3.9+ · python-docx · aiogram 3 · Docker
|
||||
|
||||
## md2gost — Markdown → DOCX per GOST
|
||||
|
||||
```bash
|
||||
pip install md2gost
|
||||
md2gost report.md # writes report.docx next to it
|
||||
md2gost report.md -o out.docx --no-heading-numbers
|
||||
```
|
||||
|
||||
What it does to the document:
|
||||
|
||||
- margins: left 30 mm, right 15, top and bottom 20; Times New Roman 14 pt, 1.5 line spacing, 1.25 cm first-line indent, justified text
|
||||
- page numbers at the bottom center, none on the title page
|
||||
- section numbering `1.`, `1.1.`, `1.1.1.`; structural elements ("Введение", "Заключение", "Список
|
||||
литературы" and others) unnumbered, uppercase, centered
|
||||
- a page break before every second-level section
|
||||
- bulleted lists with dashes and nesting, numbered lists as `1)` with numbering restarted per list
|
||||
- tables with a "Таблица N" caption on the top left, code blocks and inline code in a monospace font,
|
||||
quotes, footnotes `[^1]`, bold and italic
|
||||
|
||||
Command-line options: `--font`, `--size`, `--spacing`, `--no-heading-numbers`, `--no-page-numbers`,
|
||||
`--number-title-page`. From Python:
|
||||
|
||||
```python
|
||||
from md2gost import DocumentSettings, MarkdownToDocxConverter
|
||||
|
||||
settings = DocumentSettings()
|
||||
settings.auto_numbering_headings = True
|
||||
MarkdownToDocxConverter(settings).convert("report.md", "report.docx")
|
||||
```
|
||||
|
||||
The Markdown parser is custom and line-based: nested tables and lists inside tables are not supported.
|
||||
|
||||
## The bot
|
||||
|
||||
| Send | Get |
|
||||
|---|---|
|
||||
| `.md` | a GOST-formatted `.docx` (the same md2gost, with heading numbers) |
|
||||
| project `.zip` | a `.txt`: folder tree and contents of text files with line numbers, handy for an LLM or a report appendix |
|
||||
|
||||
Files up to 20 MB. Archives are checked before extraction: at most 5000 files and 200 MB unpacked. Service
|
||||
folders (`.git`, `node_modules`, `__pycache__`, `build`…) and binary files are skipped. Conversion runs in
|
||||
a separate thread, so the bot never freezes on big files.
|
||||
|
||||
```bash
|
||||
git clone https://github.com/EDeev/my_converterbot.git && cd my_converterbot
|
||||
cp .env.example .env # BOT_TOKEN from @BotFather
|
||||
docker compose up -d
|
||||
```
|
||||
|
||||
Prebuilt image: `docker pull ghcr.io/edeev/my_converterbot` or `docker pull dcr.deev.su/edeev/my_converterbot`.
|
||||
Without Docker: `pip install -r requirements.txt`, then `BOT_TOKEN=… python bot.py`.
|
||||
|
||||
`rep_to_txt.py` also works on its own: `python rep_to_txt.py path/to/project`.
|
||||
|
||||
## Development
|
||||
|
||||
```bash
|
||||
pip install -r requirements-dev.txt
|
||||
ruff check --select E9,F,B . && pytest
|
||||
```
|
||||
|
||||
CI tests the package on Python 3.9, 3.12 and 3.13 and builds it. On `v*` tags the package is published to
|
||||
PyPI and the bot's Docker image to GitHub Packages and `dcr.deev.su`.
|
||||
|
||||
## License
|
||||
|
||||
MIT — see [LICENSE](https://github.com/EDeev/my_converterbot/blob/main/LICENSE).
|
||||
|
||||
## Author
|
||||
|
||||
**Egor Deev** — [GitHub](https://github.com/EDeev) · [Telegram](https://t.me/DeevEgor) · [egor@deev.space](mailto:egor@deev.space)
|
||||
223
README.md
223
README.md
|
|
@ -1,171 +1,108 @@
|
|||
# 📄 My Converter Bot
|
||||
# My Converter Bot · md2gost
|
||||
|
||||
[](https://www.python.org/)
|
||||
[](https://docs.aiogram.dev/)
|
||||
[](LICENSE)
|
||||
**Русский** · [English](README.en.md)
|
||||
|
||||
Телеграм-бот для автоматизированной конвертации документов с поддержкой форматирования по ГОСТ 7.32-2017 и анализа структуры проектов.
|
||||
[](https://github.com/EDeev/my_converterbot/actions/workflows/ci.yml)
|
||||
[](https://pypi.org/project/md2gost/)
|
||||
[](https://pypi.org/project/md2gost/)
|
||||
[](LICENSE)
|
||||
|
||||
## 🎯 Функциональные возможности
|
||||
Конвертер Markdown в DOCX по ГОСТ 7.32-2017 и Telegram-бот для учебной рутины: присылаешь `.md` —
|
||||
получаешь отчёт, готовый к сдаче, присылаешь `.zip` с проектом — получаешь `.txt` с деревом папок и
|
||||
содержимым файлов. Конвертер ставится отдельно, пакетом `md2gost` с PyPI.
|
||||
|
||||
### Конвертация Markdown → DOCX
|
||||
- **Полная поддержка ГОСТ 7.32-2017**: автоматическое форматирование научно-технической документации
|
||||
- **Интеллектуальная обработка синтаксиса**: заголовки, списки, таблицы, блоки кода
|
||||
- **Автоматическая нумерация**: иерархическая нумерация разделов (1.1.1, 1.1.2)
|
||||
- **Управление сносками**: интеграция footnotes с автоматическим форматированием
|
||||
- **Настраиваемая типографика**: Times New Roman 14pt, межстрочный интервал 1.5
|
||||
**Статус:** личный проект, поддерживается · бот [@my_convbot](https://t.me/my_convbot) ·
|
||||
пакет [md2gost](https://pypi.org/project/md2gost/)
|
||||
|
||||
### Анализ архивов → TXT
|
||||
- **Древовидная визуализация**: полная структура проекта с UTF-8 оформлением
|
||||
- **Извлечение содержимого**: автоматический экспорт кода из всех текстовых файлов
|
||||
- **Интеллектуальная фильтрация**: игнорирование служебных директорий (node_modules, __pycache__)
|
||||
- **Обработка бинарных файлов**: детектирование и генерация placeholder для медиа
|
||||

|
||||
|
||||
## 🔧 Технологический стек
|
||||
**Стек:** Python 3.9+ · python-docx · aiogram 3 · Docker
|
||||
|
||||
| Компонент | Технология | Назначение |
|
||||
|-----------|------------|------------|
|
||||
| **Bot Framework** | aiogram 3.x | Асинхронная обработка Telegram API |
|
||||
| **Document Processing** | python-docx | Генерация DOCX с программным управлением стилями |
|
||||
| **Parsing Engine** | re (regex) | Парсинг Markdown синтаксиса |
|
||||
| **Archive Handling** | zipfile | Распаковка и анализ архивов |
|
||||
| **Async Runtime** | asyncio | Конкурентная обработка запросов |
|
||||
|
||||
## 📦 Установка и развертывание
|
||||
|
||||
### Системные требования
|
||||
- Python 3.10 или выше
|
||||
- pip package manager
|
||||
- Telegram Bot Token (получить у [@BotFather](https://t.me/botfather))
|
||||
|
||||
### Процедура установки
|
||||
## md2gost — Markdown → DOCX по ГОСТ
|
||||
|
||||
```bash
|
||||
# Клонирование репозитория
|
||||
git clone https://github.com/EDeev/my_converterbot.git
|
||||
cd my_converterbot
|
||||
|
||||
# Установка зависимостей
|
||||
pip install -r requirements.txt
|
||||
|
||||
# Конфигурация токена
|
||||
# Отредактируйте bot.py, установите ваш BOT_TOKEN
|
||||
# BOT_TOKEN = "your_telegram_bot_token_here"
|
||||
|
||||
# Запуск бота
|
||||
python bot.py
|
||||
pip install md2gost
|
||||
md2gost report.md # рядом появится report.docx
|
||||
md2gost report.md -o out.docx --no-heading-numbers
|
||||
```
|
||||
|
||||
## 🚀 Использование
|
||||
Что делает с документом:
|
||||
|
||||
### Базовые команды
|
||||
- `/start` — инициализация и приветственное сообщение
|
||||
- `/help` — детальная документация по функциям
|
||||
- поля: левое 30 мм, правое 15, верхнее и нижнее 20; Times New Roman 14 пт, интервал 1,5, абзацный отступ 1,25 см, выравнивание по ширине
|
||||
- номера страниц внизу по центру, без номера на титульном листе
|
||||
- нумерация разделов `1.`, `1.1.`, `1.1.1.`; «Введение», «Заключение», «Список литературы» и другие
|
||||
структурные элементы — без номера, прописными, по центру
|
||||
- разрыв страницы перед каждым разделом второго уровня
|
||||
- маркированные списки с тире и вложенностью, нумерованные — `1)`, своя нумерация у каждого списка
|
||||
- таблицы с подписью «Таблица N» слева сверху, блоки и вставки кода моноширинным шрифтом, цитаты,
|
||||
сноски `[^1]`, жирный и курсив
|
||||
|
||||
### Рабочий процесс
|
||||
|
||||
#### Markdown → DOCX конвертация
|
||||
1. Отправьте `.md` файл боту
|
||||
2. Система автоматически применит ГОСТ форматирование
|
||||
3. Получите готовый `.docx` документ
|
||||
|
||||
**Пример входного Markdown:**
|
||||
```markdown
|
||||
# Введение
|
||||
|
||||
Основной текст с **жирным** и *курсивным* форматированием[^1].
|
||||
|
||||
## 1. Методология
|
||||
|
||||
- Пункт списка 1
|
||||
- Пункт списка 2
|
||||
|
||||
[^1]: Текст сноски
|
||||
```
|
||||
|
||||
#### ZIP → TXT анализ
|
||||
1. Отправьте `.zip` архив с проектом
|
||||
2. Бот извлечет и проанализирует структуру
|
||||
3. Получите `project_structure.txt` с полным содержимым
|
||||
|
||||
## ⚙️ Архитектурные особенности
|
||||
|
||||
### Модульная структура
|
||||
|
||||
```
|
||||
my_converterbot/
|
||||
├── bot.py # Основной модуль Telegram бота
|
||||
├── md_to_docx.py # Конвертер Markdown с ГОСТ движком
|
||||
├── rep_to_txt.py # Анализатор проектных структур
|
||||
├── requirements.txt # Спецификация зависимостей
|
||||
└── README.md # Текущая документация
|
||||
```
|
||||
|
||||
### DocumentSettings: Параметрическая конфигурация
|
||||
|
||||
Класс `DocumentSettings` обеспечивает гранулярное управление:
|
||||
- Размеры шрифтов (14pt основной текст, 16pt заголовки первого уровня)
|
||||
- Отступы документа (левый: 3.0 см для переплета)
|
||||
- Режимы нумерации (decimal: 1.1.1 или simple: 1)
|
||||
- Позиционирование номеров страниц
|
||||
|
||||
### Интеллектуальная обработка
|
||||
|
||||
**Алгоритм обработки списков:**
|
||||
- Распознавание вложенности через отступы
|
||||
- Автоматическая замена bullet points на длинное тире (ГОСТ)
|
||||
- Сохранение иерархической структуры
|
||||
|
||||
**Система обработки сносок:**
|
||||
- Inline маркеры `[^1]` → верхний индекс в тексте
|
||||
- Автоматическая агрегация определений
|
||||
- Размещение в конце документа с разделителем
|
||||
|
||||
## 🔒 Ограничения и constraints
|
||||
|
||||
- **Максимальный размер файла**: 20 МБ (Telegram API limitation)
|
||||
- **Поддерживаемые форматы входных данных**: `.md`, `.zip`
|
||||
- **Кодировки**: UTF-8, UTF-8-sig, CP1251, Latin1 (fallback цепочка)
|
||||
|
||||
## 📊 Производительность
|
||||
|
||||
- **Обработка Markdown**: ~0.5-2 сек для документов до 50 страниц
|
||||
- **Анализ ZIP архивов**: ~1-5 сек для проектов до 1000 файлов
|
||||
- **Конкурентная обработка**: до 10 одновременных запросов
|
||||
|
||||
## 🛠️ Расширение функциональности
|
||||
|
||||
### Кастомизация ГОСТ параметров
|
||||
Параметры командной строки: `--font`, `--size`, `--spacing`, `--no-heading-numbers`, `--no-page-numbers`,
|
||||
`--number-title-page`. Из Python:
|
||||
|
||||
```python
|
||||
from md_to_docx import MarkdownToDocxConverter, DocumentSettings
|
||||
from md2gost import DocumentSettings, MarkdownToDocxConverter
|
||||
|
||||
settings = DocumentSettings()
|
||||
settings.font_name = "Times New Roman"
|
||||
settings.font_size = 14
|
||||
settings.line_spacing = 1.5
|
||||
settings.margin_left = 3.0
|
||||
settings.auto_numbering_headings = True
|
||||
settings.numbering_format = "decimal"
|
||||
|
||||
converter = MarkdownToDocxConverter(settings)
|
||||
converter.convert("input.md", "output.docx")
|
||||
MarkdownToDocxConverter(settings).convert("report.md", "report.docx")
|
||||
```
|
||||
|
||||
## 📄 Лицензия
|
||||
Разбор Markdown свой и построчный: вложенные таблицы и списки внутри таблиц не поддерживаются.
|
||||
|
||||
Этот проект является некоммерческим и распространяется под лицензией MIT.
|
||||
## Бот
|
||||
|
||||
## 👨💻 Автор
|
||||
| Прислать | Получить |
|
||||
|---|---|
|
||||
| `.md` | `.docx` по ГОСТ (тот же md2gost с нумерацией заголовков) |
|
||||
| `.zip` с проектом | `.txt`: дерево папок и содержимое текстовых файлов с номерами строк — удобно отдать в LLM или приложить к отчёту |
|
||||
|
||||
**Деев Егор Викторович** - Backend Developer
|
||||
- GitHub: [@EDeev](https://github.com/EDeev)
|
||||
- Email: egor@deev.space
|
||||
- Telegram: [@Egor_Deev](https://t.me/Egor_Deev)
|
||||
Файлы — до 20 МБ. Архив проверяется до распаковки: не больше 5000 файлов и 200 МБ в распакованном виде.
|
||||
Служебные папки (`.git`, `node_modules`, `__pycache__`, `build`…) и бинарные файлы пропускаются.
|
||||
Конвертация идёт в отдельном потоке, поэтому бот не замирает на больших файлах.
|
||||
|
||||
```bash
|
||||
git clone https://github.com/EDeev/my_converterbot.git && cd my_converterbot
|
||||
cp .env.example .env # BOT_TOKEN от @BotFather
|
||||
docker compose up -d
|
||||
```
|
||||
|
||||
Готовый образ: `docker pull ghcr.io/edeev/my_converterbot` или `docker pull dcr.deev.su/edeev/my_converterbot`.
|
||||
Без Docker: `pip install -r requirements.txt`, затем `BOT_TOKEN=… python bot.py`.
|
||||
|
||||
`rep_to_txt.py` работает и сам по себе: `python rep_to_txt.py путь/к/проекту`.
|
||||
|
||||
## Структура
|
||||
|
||||
```
|
||||
md2gost/converter.py конвертер: настройки DocumentSettings и MarkdownToDocxConverter
|
||||
md2gost/cli.py командная строка md2gost
|
||||
bot.py Telegram-бот
|
||||
rep_to_txt.py дерево проекта и содержимое файлов в один .txt
|
||||
tests/ тесты конвертера и бота
|
||||
```
|
||||
|
||||
## Разработка
|
||||
|
||||
```bash
|
||||
pip install -r requirements-dev.txt
|
||||
ruff check --select E9,F,B . && pytest
|
||||
```
|
||||
|
||||
CI проверяет пакет на Python 3.9, 3.12 и 3.13 и собирает его. По тегу `v*` пакет публикуется на PyPI, а
|
||||
Docker-образ бота — в GitHub Packages и `dcr.deev.su`.
|
||||
|
||||
## Лицензия
|
||||
|
||||
MIT — см. [LICENSE](LICENSE).
|
||||
|
||||
## Автор
|
||||
|
||||
**Деев Егор Викторович** — [GitHub](https://github.com/EDeev) · [Telegram](https://t.me/DeevEgor) · [egor@deev.space](mailto:egor@deev.space)
|
||||
|
||||
---
|
||||
|
||||
<div align="center">
|
||||
<sub>⭐ Если проект оказался полезным, поставьте звездочку на GitHub!</sub>
|
||||
<p><sub>Создано с ❤️ от вашего дорогого - deev.space ©</sub></p>
|
||||
<sub>⭐ Если проект оказался полезным, поставьте звёздочку на GitHub!</sub>
|
||||
<p><sub>Сделано с ❤️ — <a href="https://deev.space">deev.space</a></sub></p>
|
||||
</div>
|
||||
|
|
|
|||
6
compose.yaml
Normal file
6
compose.yaml
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
services:
|
||||
bot:
|
||||
build: .
|
||||
image: ghcr.io/edeev/my_converterbot:latest
|
||||
env_file: .env
|
||||
restart: unless-stopped
|
||||
BIN
docs/demo.png
Normal file
BIN
docs/demo.png
Normal file
Binary file not shown.
|
After Width: | Height: | Size: 98 KiB |
4
requirements-dev.txt
Normal file
4
requirements-dev.txt
Normal file
|
|
@ -0,0 +1,4 @@
|
|||
-r requirements.txt
|
||||
pytest==8.4.2
|
||||
ruff==0.14.0
|
||||
build==1.3.0
|
||||
|
|
@ -1,5 +1,2 @@
|
|||
aiogram==3.13.1
|
||||
python-docx==1.1.2
|
||||
aiohttp>=3.9.0
|
||||
aiofiles>=23.0.0
|
||||
certifi>=2023.0.0
|
||||
aiogram==3.23.0
|
||||
python-docx==1.2.0
|
||||
|
|
|
|||
4
tests/conftest.py
Normal file
4
tests/conftest.py
Normal file
|
|
@ -0,0 +1,4 @@
|
|||
import os
|
||||
import sys
|
||||
|
||||
sys.path.insert(0, os.path.join(os.path.dirname(__file__), ".."))
|
||||
41
tests/test_bot_helpers.py
Normal file
41
tests/test_bot_helpers.py
Normal file
|
|
@ -0,0 +1,41 @@
|
|||
import io
|
||||
import os
|
||||
import zipfile
|
||||
|
||||
import pytest
|
||||
|
||||
os.environ.setdefault("BOT_TOKEN", "123456:TEST")
|
||||
|
||||
import bot # noqa: E402
|
||||
from rep_to_txt import generate_complete_project_structure # noqa: E402
|
||||
|
||||
|
||||
def test_zip_bomb_is_rejected(tmp_path, monkeypatch):
|
||||
monkeypatch.setattr(bot, "MAX_UNPACKED_SIZE", 1000)
|
||||
archive = tmp_path / "a.zip"
|
||||
with zipfile.ZipFile(archive, "w", zipfile.ZIP_DEFLATED) as z:
|
||||
z.writestr("big.txt", "0" * 10_000)
|
||||
with pytest.raises(bot.ArchiveTooLarge):
|
||||
bot.analyze_archive(str(archive), str(tmp_path))
|
||||
|
||||
|
||||
def test_archive_structure(tmp_path):
|
||||
archive = tmp_path / "p.zip"
|
||||
with zipfile.ZipFile(archive, "w") as z:
|
||||
z.writestr("proj/main.py", "print(1)\n")
|
||||
z.writestr("proj/img.png", b"\x89PNG\x00")
|
||||
z.writestr("proj/node_modules/x.js", "ignored")
|
||||
text = open(bot.analyze_archive(str(archive), str(tmp_path)), encoding="utf-8").read()
|
||||
assert "main.py" in text and " 1 | print(1)" in text
|
||||
assert "[Image - content not displayed]" in text and "node_modules" not in text
|
||||
|
||||
|
||||
def test_structure_of_missing_path():
|
||||
assert generate_complete_project_structure("/no/such/dir").startswith("Error")
|
||||
|
||||
|
||||
def test_docx_conversion_in_bot(tmp_path):
|
||||
md = tmp_path / "a.md"
|
||||
md.write_text("# Тест\n", encoding="utf-8")
|
||||
out = bot.convert_md_to_docx(str(md), str(tmp_path))
|
||||
assert zipfile.is_zipfile(io.BytesIO(open(out, "rb").read()))
|
||||
105
tests/test_md2gost.py
Normal file
105
tests/test_md2gost.py
Normal file
|
|
@ -0,0 +1,105 @@
|
|||
import zipfile
|
||||
|
||||
from docx import Document
|
||||
|
||||
from md2gost import DocumentSettings, MarkdownToDocxConverter
|
||||
from md2gost.cli import main as cli_main
|
||||
|
||||
SAMPLE = """# Отчёт о практике
|
||||
|
||||
## Введение
|
||||
|
||||
Абзац с **жирным**, *курсивом*, `кодом` и сноской[^1].
|
||||
|
||||
### Цели
|
||||
|
||||
- Первый пункт
|
||||
- Второй пункт
|
||||
- Вложенный пункт
|
||||
|
||||
1. Раз
|
||||
2. Два
|
||||
|
||||
Второй список:
|
||||
|
||||
1. Снова один
|
||||
2. Снова два
|
||||
|
||||
| Параметр | Значение |
|
||||
|---|---|
|
||||
| A | 1 |
|
||||
|
||||
Строка с | вертикальной чертой, но не таблица.
|
||||
|
||||
## Список литературы
|
||||
|
||||
1. Иванов И. И. Книга. — М., 2020.
|
||||
2. Петров П. П. Статья. — СПб., 2021.
|
||||
|
||||
[^1]: Текст сноски.
|
||||
"""
|
||||
|
||||
|
||||
def convert(tmp_path, text=SAMPLE, **options):
|
||||
src = tmp_path / "in.md"
|
||||
src.write_text(text, encoding="utf-8")
|
||||
settings = DocumentSettings()
|
||||
settings.auto_numbering_headings = True
|
||||
for key, value in options.items():
|
||||
setattr(settings, key, value)
|
||||
out = tmp_path / "out.docx"
|
||||
MarkdownToDocxConverter(settings).convert(str(src), str(out))
|
||||
return out
|
||||
|
||||
|
||||
def texts(path):
|
||||
return [p.text for p in Document(str(path)).paragraphs if p.text.strip()]
|
||||
|
||||
|
||||
def test_page_number_field_and_title_page(tmp_path):
|
||||
out = convert(tmp_path)
|
||||
with zipfile.ZipFile(out) as z:
|
||||
footers = [z.read(n).decode() for n in z.namelist() if n.startswith("word/footer")]
|
||||
document = z.read("word/document.xml").decode()
|
||||
assert any("PAGE" in f for f in footers)
|
||||
assert "<w:titlePg/>" in document
|
||||
|
||||
|
||||
def test_heading_font_is_not_theme_font(tmp_path):
|
||||
out = convert(tmp_path)
|
||||
with zipfile.ZipFile(out) as z:
|
||||
styles = z.read("word/styles.xml").decode()
|
||||
heading = styles[styles.index('w:styleId="Heading1"'):]
|
||||
heading = heading[:heading.index("</w:style>")]
|
||||
assert "asciiTheme" not in heading and 'w:ascii="Times New Roman"' in heading
|
||||
|
||||
|
||||
def test_lists_have_single_dash_nesting_and_restart(tmp_path):
|
||||
lines = texts(convert(tmp_path))
|
||||
assert "– Первый пункт" in lines
|
||||
nested = next(p for p in Document(str(convert(tmp_path))).paragraphs if p.text == "– Вложенный пункт")
|
||||
assert nested.paragraph_format.left_indent.cm > 0
|
||||
assert lines.count("1) Раз") == 1 and "1) Снова один" in lines
|
||||
|
||||
|
||||
def test_structural_headings_are_not_numbered(tmp_path):
|
||||
lines = texts(convert(tmp_path))
|
||||
assert "ВВЕДЕНИЕ" in lines and "СПИСОК ЛИТЕРАТУРЫ" in lines
|
||||
assert "1.1. Цели" in lines # нумерация разделов идёт мимо «Введения»
|
||||
assert "1. Иванов И. И. Книга. — М., 2020." in lines
|
||||
|
||||
|
||||
def test_table_and_pipe_paragraph(tmp_path):
|
||||
out = convert(tmp_path)
|
||||
doc = Document(str(out))
|
||||
assert len(doc.tables) == 1 and doc.tables[0].cell(1, 1).text == "1"
|
||||
assert "Строка с | вертикальной чертой, но не таблица." in texts(out)
|
||||
|
||||
|
||||
def test_cli(tmp_path, capsys):
|
||||
src = tmp_path / "doc.md"
|
||||
src.write_text("# Заголовок\n\nТекст\n", encoding="utf-8")
|
||||
assert cli_main([str(src), "--no-heading-numbers"]) == 0
|
||||
out = tmp_path / "doc.docx"
|
||||
assert out.exists() and "Заголовок" in texts(out)
|
||||
assert cli_main([str(tmp_path / "nope.md")]) == 1
|
||||
Loading…
Add table
Reference in a new issue