mirror of
https://github.com/EDeev/converterbot.git
synced 2026-10-07 20:49:33 +03:00
Compare commits
No commits in common. "main" and "v1.0.1" have entirely different histories.
4 changed files with 16 additions and 10 deletions
6
.github/workflows/release.yml
vendored
6
.github/workflows/release.yml
vendored
|
|
@ -38,11 +38,17 @@ jobs:
|
||||||
registry: ghcr.io
|
registry: ghcr.io
|
||||||
username: ${{ github.actor }}
|
username: ${{ github.actor }}
|
||||||
password: ${{ secrets.GITHUB_TOKEN }}
|
password: ${{ secrets.GITHUB_TOKEN }}
|
||||||
|
- uses: docker/login-action@v3
|
||||||
|
with:
|
||||||
|
registry: dcr.deev.su
|
||||||
|
username: ${{ secrets.ZOT_USERNAME }}
|
||||||
|
password: ${{ secrets.ZOT_PASSWORD }}
|
||||||
- id: meta
|
- id: meta
|
||||||
uses: docker/metadata-action@v5
|
uses: docker/metadata-action@v5
|
||||||
with:
|
with:
|
||||||
images: |
|
images: |
|
||||||
ghcr.io/edeev/converterbot
|
ghcr.io/edeev/converterbot
|
||||||
|
dcr.deev.su/edeev/converterbot
|
||||||
tags: |
|
tags: |
|
||||||
type=semver,pattern={{version}}
|
type=semver,pattern={{version}}
|
||||||
type=semver,pattern={{major}}.{{minor}}
|
type=semver,pattern={{major}}.{{minor}}
|
||||||
|
|
|
||||||
|
|
@ -58,7 +58,7 @@ The Markdown parser is custom and line-based: nested tables and lists inside tab
|
||||||
| `.md` | a GOST-formatted `.docx` (the same md2gost, with heading numbers) |
|
| `.md` | a GOST-formatted `.docx` (the same md2gost, with heading numbers) |
|
||||||
| project `.zip` | a `.txt`: folder tree and contents of text files with line numbers, handy for an LLM or a report appendix |
|
| project `.zip` | a `.txt`: folder tree and contents of text files with line numbers, handy for an LLM or a report appendix |
|
||||||
|
|
||||||
Files up to 20 MB. Archives are checked before extraction: at most 10,000,000 files and 15 GB unpacked. Service
|
Files up to 20 MB. Archives are checked before extraction: at most 5000 files and 200 MB unpacked. Service
|
||||||
folders (`.git`, `node_modules`, `__pycache__`, `build`…) and binary files are skipped. Conversion runs in
|
folders (`.git`, `node_modules`, `__pycache__`, `build`…) and binary files are skipped. Conversion runs in
|
||||||
a separate thread, so the bot never freezes on big files.
|
a separate thread, so the bot never freezes on big files.
|
||||||
|
|
||||||
|
|
@ -68,7 +68,7 @@ cp .env.example .env # BOT_TOKEN from @BotFather
|
||||||
docker compose up -d
|
docker compose up -d
|
||||||
```
|
```
|
||||||
|
|
||||||
Prebuilt image: `docker pull ghcr.io/edeev/converterbot` or `docker pull git.deev.su/edeev/converterbot`.
|
Prebuilt image: `docker pull ghcr.io/edeev/converterbot` or `docker pull dcr.deev.su/edeev/converterbot`.
|
||||||
Without Docker: `pip install -r requirements.txt`, then `BOT_TOKEN=… python bot.py`.
|
Without Docker: `pip install -r requirements.txt`, then `BOT_TOKEN=… python bot.py`.
|
||||||
|
|
||||||
`rep_to_txt.py` also works on its own: `python rep_to_txt.py path/to/project`.
|
`rep_to_txt.py` also works on its own: `python rep_to_txt.py path/to/project`.
|
||||||
|
|
@ -81,7 +81,7 @@ ruff check --select E9,F,B . && pytest
|
||||||
```
|
```
|
||||||
|
|
||||||
CI tests the package on Python 3.9, 3.12 and 3.13 and builds it. On `v*` tags the package is published to
|
CI tests the package on Python 3.9, 3.12 and 3.13 and builds it. On `v*` tags the package is published to
|
||||||
PyPI and the bot's Docker image to GitHub Packages and `git.deev.su`.
|
PyPI and the bot's Docker image to GitHub Packages and `dcr.deev.su`.
|
||||||
|
|
||||||
## License
|
## License
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -57,7 +57,7 @@ MarkdownToDocxConverter(settings).convert("report.md", "report.docx")
|
||||||
| `.md` | `.docx` по ГОСТ (тот же md2gost с нумерацией заголовков) |
|
| `.md` | `.docx` по ГОСТ (тот же md2gost с нумерацией заголовков) |
|
||||||
| `.zip` с проектом | `.txt`: дерево папок и содержимое текстовых файлов с номерами строк — удобно отдать в LLM или приложить к отчёту |
|
| `.zip` с проектом | `.txt`: дерево папок и содержимое текстовых файлов с номерами строк — удобно отдать в LLM или приложить к отчёту |
|
||||||
|
|
||||||
Файлы — до 20 МБ. Архив проверяется до распаковки: не больше 10 000 000 файлов и 15 ГБ в распакованном виде.
|
Файлы — до 20 МБ. Архив проверяется до распаковки: не больше 5000 файлов и 200 МБ в распакованном виде.
|
||||||
Служебные папки (`.git`, `node_modules`, `__pycache__`, `build`…) и бинарные файлы пропускаются.
|
Служебные папки (`.git`, `node_modules`, `__pycache__`, `build`…) и бинарные файлы пропускаются.
|
||||||
Конвертация идёт в отдельном потоке, поэтому бот не замирает на больших файлах.
|
Конвертация идёт в отдельном потоке, поэтому бот не замирает на больших файлах.
|
||||||
|
|
||||||
|
|
@ -67,7 +67,7 @@ cp .env.example .env # BOT_TOKEN от @BotFather
|
||||||
docker compose up -d
|
docker compose up -d
|
||||||
```
|
```
|
||||||
|
|
||||||
Готовый образ: `docker pull ghcr.io/edeev/converterbot` или `docker pull git.deev.su/edeev/converterbot`.
|
Готовый образ: `docker pull ghcr.io/edeev/converterbot` или `docker pull dcr.deev.su/edeev/converterbot`.
|
||||||
Без Docker: `pip install -r requirements.txt`, затем `BOT_TOKEN=… python bot.py`.
|
Без Docker: `pip install -r requirements.txt`, затем `BOT_TOKEN=… python bot.py`.
|
||||||
|
|
||||||
`rep_to_txt.py` работает и сам по себе: `python rep_to_txt.py путь/к/проекту`.
|
`rep_to_txt.py` работает и сам по себе: `python rep_to_txt.py путь/к/проекту`.
|
||||||
|
|
@ -90,7 +90,7 @@ ruff check --select E9,F,B . && pytest
|
||||||
```
|
```
|
||||||
|
|
||||||
CI проверяет пакет на Python 3.9, 3.12 и 3.13 и собирает его. По тегу `v*` пакет публикуется на PyPI, а
|
CI проверяет пакет на Python 3.9, 3.12 и 3.13 и собирает его. По тегу `v*` пакет публикуется на PyPI, а
|
||||||
Docker-образ бота — в GitHub Packages и `git.deev.su`.
|
Docker-образ бота — в GitHub Packages и `dcr.deev.su`.
|
||||||
|
|
||||||
## Лицензия
|
## Лицензия
|
||||||
|
|
||||||
|
|
|
||||||
8
bot.py
8
bot.py
|
|
@ -20,8 +20,8 @@ from rep_to_txt import generate_complete_project_structure
|
||||||
BOT_TOKEN = os.getenv("BOT_TOKEN", "XXXXXXXXXXXXXXXXXXXXXXXX") # @my_convbot
|
BOT_TOKEN = os.getenv("BOT_TOKEN", "XXXXXXXXXXXXXXXXXXXXXXXX") # @my_convbot
|
||||||
|
|
||||||
MAX_FILE_SIZE = 20 * 1024 * 1024 # больше Telegram-боту не скачать
|
MAX_FILE_SIZE = 20 * 1024 * 1024 # больше Telegram-боту не скачать
|
||||||
MAX_UNPACKED_SIZE = 15 * 1024 ** 3 # защита от zip-бомбы: 15 ГБ в распакованном виде
|
MAX_UNPACKED_SIZE = 200 * 1024 * 1024 # защита от zip-бомбы
|
||||||
MAX_FILES_IN_ARCHIVE = 10_000_000
|
MAX_FILES_IN_ARCHIVE = 5000
|
||||||
|
|
||||||
# Инициализация бота
|
# Инициализация бота
|
||||||
bot = Bot(token=BOT_TOKEN, default=DefaultBotProperties(parse_mode=ParseMode.HTML))
|
bot = Bot(token=BOT_TOKEN, default=DefaultBotProperties(parse_mode=ParseMode.HTML))
|
||||||
|
|
@ -139,9 +139,9 @@ def check_archive(zip_ref: zipfile.ZipFile) -> None:
|
||||||
"""Архив на 20 МБ может распаковаться в гигабайты и забить диск — проверяем до распаковки"""
|
"""Архив на 20 МБ может распаковаться в гигабайты и забить диск — проверяем до распаковки"""
|
||||||
infos = zip_ref.infolist()
|
infos = zip_ref.infolist()
|
||||||
if len(infos) > MAX_FILES_IN_ARCHIVE:
|
if len(infos) > MAX_FILES_IN_ARCHIVE:
|
||||||
raise ArchiveTooLarge(f"В архиве больше {MAX_FILES_IN_ARCHIVE:,} файлов".replace(",", " "))
|
raise ArchiveTooLarge(f"В архиве больше {MAX_FILES_IN_ARCHIVE} файлов")
|
||||||
if sum(info.file_size for info in infos) > MAX_UNPACKED_SIZE:
|
if sum(info.file_size for info in infos) > MAX_UNPACKED_SIZE:
|
||||||
raise ArchiveTooLarge(f"Распакованный архив больше {MAX_UNPACKED_SIZE / 1024 ** 3:g} ГБ")
|
raise ArchiveTooLarge(f"Распакованный архив больше {MAX_UNPACKED_SIZE // 1024 // 1024} МБ")
|
||||||
|
|
||||||
|
|
||||||
def analyze_archive(archive_path: str, temp_dir: str) -> str:
|
def analyze_archive(archive_path: str, temp_dir: str) -> str:
|
||||||
|
|
|
||||||
Loading…
Add table
Reference in a new issue