mirror of
https://github.com/EDeev/converterbot.git
synced 2026-10-07 20:49:33 +03:00
Конвертер перенесён в пакет md2gost, переводы строк — LF
Отдельный пакет — чтобы конвертер можно было ставить из PyPI и запускать из командной строки без бота.
This commit is contained in:
parent
af19aa999f
commit
bc351e995a
3 changed files with 166 additions and 164 deletions
2
.gitattributes
vendored
Normal file
2
.gitattributes
vendored
Normal file
|
|
@ -0,0 +1,2 @@
|
||||||
|
* text=auto eol=lf
|
||||||
|
*.docx binary
|
||||||
328
rep_to_txt.py
328
rep_to_txt.py
|
|
@ -1,165 +1,165 @@
|
||||||
import os
|
import os
|
||||||
|
|
||||||
IGNORE_PATTERNS = {
|
IGNORE_PATTERNS = {
|
||||||
'.git', '.svn', '.hg', # Version control systems
|
'.git', '.svn', '.hg', # Version control systems
|
||||||
'__pycache__', '.pytest_cache', # Python artifacts
|
'__pycache__', '.pytest_cache', # Python artifacts
|
||||||
'node_modules', '.npm', # Node.js dependencies
|
'node_modules', '.npm', # Node.js dependencies
|
||||||
'target', 'build', 'dist', # Build outputs
|
'target', 'build', 'dist', # Build outputs
|
||||||
'.idea', '.vscode', # IDE metadata
|
'.idea', '.vscode', # IDE metadata
|
||||||
'.DS_Store', 'Thumbs.db', # OS metadata
|
'.DS_Store', 'Thumbs.db', # OS metadata
|
||||||
'.pro.user' # QT user config
|
'.pro.user' # QT user config
|
||||||
}
|
}
|
||||||
|
|
||||||
BINARY_EXTENSIONS = {
|
BINARY_EXTENSIONS = {
|
||||||
'.exe', '.dll', '.so', '.dylib', '.zip', '.tar', '.gz', '.rar', '.7z',
|
'.exe', '.dll', '.so', '.dylib', '.zip', '.tar', '.gz', '.rar', '.7z',
|
||||||
'.jpg', '.jpeg', '.png', '.gif', '.bmp', '.ico', '.svg', '.webp',
|
'.jpg', '.jpeg', '.png', '.gif', '.bmp', '.ico', '.svg', '.webp',
|
||||||
'.mp3', '.mp4', '.avi', '.mov', '.mkv', '.wmv', '.flv',
|
'.mp3', '.mp4', '.avi', '.mov', '.mkv', '.wmv', '.flv',
|
||||||
'.pdf', '.doc', '.docx', '.xls', '.xlsx', '.ppt', '.pptx',
|
'.pdf', '.doc', '.docx', '.xls', '.xlsx', '.ppt', '.pptx',
|
||||||
'.bin', '.dat', '.db', '.sqlite', '.mdb'
|
'.bin', '.dat', '.db', '.sqlite', '.mdb'
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
def scan_directory(path, prefix=""):
|
def scan_directory(path, prefix=""):
|
||||||
"""Рекурсивное сканирование с форматированием дерева"""
|
"""Рекурсивное сканирование с форматированием дерева"""
|
||||||
items = []
|
items = []
|
||||||
try:
|
try:
|
||||||
entries = sorted(os.listdir(path))
|
entries = sorted(os.listdir(path))
|
||||||
# Critical filtering layer для performance optimization
|
# Critical filtering layer для performance optimization
|
||||||
dirs = [e for e in entries if os.path.isdir(os.path.join(path, e)) and e not in IGNORE_PATTERNS]
|
dirs = [e for e in entries if os.path.isdir(os.path.join(path, e)) and e not in IGNORE_PATTERNS]
|
||||||
files = [e for e in entries if os.path.isfile(os.path.join(path, e)) and e not in IGNORE_PATTERNS]
|
files = [e for e in entries if os.path.isfile(os.path.join(path, e)) and e not in IGNORE_PATTERNS]
|
||||||
all_items = dirs + files
|
all_items = dirs + files
|
||||||
|
|
||||||
for i, item in enumerate(all_items):
|
for i, item in enumerate(all_items):
|
||||||
item_path = os.path.join(path, item)
|
item_path = os.path.join(path, item)
|
||||||
is_last_item = (i == len(all_items) - 1)
|
is_last_item = (i == len(all_items) - 1)
|
||||||
|
|
||||||
if is_last_item:
|
if is_last_item:
|
||||||
current_prefix = prefix + "└── "
|
current_prefix = prefix + "└── "
|
||||||
next_prefix = prefix + " "
|
next_prefix = prefix + " "
|
||||||
else:
|
else:
|
||||||
current_prefix = prefix + "├── "
|
current_prefix = prefix + "├── "
|
||||||
next_prefix = prefix + "│ "
|
next_prefix = prefix + "│ "
|
||||||
|
|
||||||
items.append(current_prefix + item)
|
items.append(current_prefix + item)
|
||||||
|
|
||||||
if os.path.isdir(item_path):
|
if os.path.isdir(item_path):
|
||||||
items.extend(scan_directory(item_path, next_prefix))
|
items.extend(scan_directory(item_path, next_prefix))
|
||||||
|
|
||||||
except PermissionError:
|
except PermissionError:
|
||||||
items.append(prefix + "└── [Access Denied]")
|
items.append(prefix + "└── [Access Denied]")
|
||||||
|
|
||||||
return items
|
return items
|
||||||
|
|
||||||
|
|
||||||
def generate_complete_project_structure(root_path):
|
def generate_complete_project_structure(root_path):
|
||||||
"""Генератор проектной документации корпоративного уровня"""
|
"""Генератор проектной документации корпоративного уровня"""
|
||||||
if not os.path.exists(root_path):
|
if not os.path.exists(root_path):
|
||||||
return f"Error: Path {root_path} does not exist"
|
return f"Error: Path {root_path} does not exist"
|
||||||
|
|
||||||
result = []
|
result = []
|
||||||
|
|
||||||
# Этап 1: Создание древовидной структуры
|
# Этап 1: Создание древовидной структуры
|
||||||
root_name = os.path.basename(root_path) or root_path
|
root_name = os.path.basename(root_path) or root_path
|
||||||
result.append(root_name)
|
result.append(root_name)
|
||||||
result.extend(scan_directory(root_path))
|
result.extend(scan_directory(root_path))
|
||||||
|
|
||||||
# Этап 2: Полное извлечение содержимого файла
|
# Этап 2: Полное извлечение содержимого файла
|
||||||
result.append("\n") # Separator между разделами дерева и содержимым
|
result.append("\n") # Separator между разделами дерева и содержимым
|
||||||
result.extend(extract_all_file_contents(root_path))
|
result.extend(extract_all_file_contents(root_path))
|
||||||
|
|
||||||
return "\n".join(result)
|
return "\n".join(result)
|
||||||
|
|
||||||
|
|
||||||
def extract_all_file_contents(root_path):
|
def extract_all_file_contents(root_path):
|
||||||
"""Механизм извлечения контента с обработкой файлов"""
|
"""Механизм извлечения контента с обработкой файлов"""
|
||||||
content_lines = []
|
content_lines = []
|
||||||
|
|
||||||
for root, dirs, files in os.walk(root_path):
|
for root, dirs, files in os.walk(root_path):
|
||||||
dirs[:] = [d for d in dirs if d not in IGNORE_PATTERNS]
|
dirs[:] = [d for d in dirs if d not in IGNORE_PATTERNS]
|
||||||
|
|
||||||
for file in sorted(files):
|
for file in sorted(files):
|
||||||
if file in IGNORE_PATTERNS:
|
if file in IGNORE_PATTERNS:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
file_path = os.path.join(root, file)
|
file_path = os.path.join(root, file)
|
||||||
relative_path = os.path.relpath(file_path, root_path)
|
relative_path = os.path.relpath(file_path, root_path)
|
||||||
|
|
||||||
content_lines.extend(process_single_file(relative_path, file_path))
|
content_lines.extend(process_single_file(relative_path, file_path))
|
||||||
|
|
||||||
return content_lines
|
return content_lines
|
||||||
|
|
||||||
|
|
||||||
def process_single_file(relative_path, file_path):
|
def process_single_file(relative_path, file_path):
|
||||||
"""Обработка файлов"""
|
"""Обработка файлов"""
|
||||||
content_lines = []
|
content_lines = []
|
||||||
|
|
||||||
# Раздел заголовка
|
# Раздел заголовка
|
||||||
content_lines.append("\n" + "-" * 80)
|
content_lines.append("\n" + "-" * 80)
|
||||||
content_lines.append(f"{relative_path}:")
|
content_lines.append(f"{relative_path}:")
|
||||||
content_lines.append("-" * 80)
|
content_lines.append("-" * 80)
|
||||||
|
|
||||||
file_ext = os.path.splitext(relative_path)[1].lower()
|
file_ext = os.path.splitext(relative_path)[1].lower()
|
||||||
|
|
||||||
# Обнаружение двоичных файлов и генерация URL-адресов
|
# Обнаружение двоичных файлов и генерация URL-адресов
|
||||||
if file_ext in BINARY_EXTENSIONS or is_likely_binary(file_path):
|
if file_ext in BINARY_EXTENSIONS or is_likely_binary(file_path):
|
||||||
# GitHub raw URL
|
# GitHub raw URL
|
||||||
if file_ext in {'.png', '.jpg', '.jpeg', '.gif', '.svg', '.ico'}:
|
if file_ext in {'.png', '.jpg', '.jpeg', '.gif', '.svg', '.ico'}:
|
||||||
# Структура URL - настраивается на основе фактического хранилища
|
# Структура URL - настраивается на основе фактического хранилища
|
||||||
github_url = f"https://raw.githubusercontent.com/.../{relative_path.replace(os.sep, '/')}"
|
github_url = f"https://raw.githubusercontent.com/.../{relative_path.replace(os.sep, '/')}"
|
||||||
content_lines.append(github_url)
|
content_lines.append(github_url)
|
||||||
else:
|
else:
|
||||||
content_lines.append("[Binary file - content not displayed]")
|
content_lines.append("[Binary file - content not displayed]")
|
||||||
else:
|
else:
|
||||||
# Извлечение содержимого текстового файла
|
# Извлечение содержимого текстового файла
|
||||||
content_lines.extend(extract_text_content(file_path))
|
content_lines.extend(extract_text_content(file_path))
|
||||||
|
|
||||||
content_lines.append("")
|
content_lines.append("")
|
||||||
return content_lines
|
return content_lines
|
||||||
|
|
||||||
|
|
||||||
def extract_text_content(file_path):
|
def extract_text_content(file_path):
|
||||||
"""Резервное извлечение с несколькими кодировками"""
|
"""Резервное извлечение с несколькими кодировками"""
|
||||||
encodings_priority = ['utf-8', 'utf-8-sig', 'cp1251', 'latin1', 'cp1252']
|
encodings_priority = ['utf-8', 'utf-8-sig', 'cp1251', 'latin1', 'cp1252']
|
||||||
|
|
||||||
for encoding in encodings_priority:
|
for encoding in encodings_priority:
|
||||||
try:
|
try:
|
||||||
with open(file_path, 'r', encoding=encoding) as f:
|
with open(file_path, 'r', encoding=encoding) as f:
|
||||||
lines = f.readlines()
|
lines = f.readlines()
|
||||||
return [f"{i:4} | {line.rstrip()}" for i, line in enumerate(lines, 1)]
|
return [f"{i:4} | {line.rstrip()}" for i, line in enumerate(lines, 1)]
|
||||||
except (UnicodeDecodeError, UnicodeError):
|
except (UnicodeDecodeError, UnicodeError):
|
||||||
continue
|
continue
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
return [f"ERROR: Не удается прочитать файл - {e}"]
|
return [f"ERROR: Не удается прочитать файл - {e}"]
|
||||||
|
|
||||||
return ["WARNING: Кодировка файла, не поддерживаемая для извлечения текста"]
|
return ["WARNING: Кодировка файла, не поддерживаемая для извлечения текста"]
|
||||||
|
|
||||||
|
|
||||||
def is_likely_binary(file_path):
|
def is_likely_binary(file_path):
|
||||||
"""Эвристическое обнаружение двоичных файлов для крайних случаев"""
|
"""Эвристическое обнаружение двоичных файлов для крайних случаев"""
|
||||||
try:
|
try:
|
||||||
with open(file_path, 'rb') as f:
|
with open(file_path, 'rb') as f:
|
||||||
chunk = f.read(8192)
|
chunk = f.read(8192)
|
||||||
# Обнаружение нулевого байта - надежный бинарный индикатор
|
# Обнаружение нулевого байта - надежный бинарный индикатор
|
||||||
return b'\x00' in chunk
|
return b'\x00' in chunk
|
||||||
except:
|
except:
|
||||||
return True
|
return True
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
# Конфигурация: измените путь к целевому каталогу проекта
|
# Конфигурация: измените путь к целевому каталогу проекта
|
||||||
project_path = r"D:\Programs\GitHub\deev.space\static"
|
project_path = r"D:\Programs\GitHub\deev.space\static"
|
||||||
# project_path = r"D:/Programs/GitHub/openoffice"
|
# project_path = r"D:/Programs/GitHub/openoffice"
|
||||||
# project_path = "."
|
# project_path = "."
|
||||||
|
|
||||||
print("Приступаем к формированию комплексной структуры проекта...")
|
print("Приступаем к формированию комплексной структуры проекта...")
|
||||||
tree_output = generate_complete_project_structure(project_path)
|
tree_output = generate_complete_project_structure(project_path)
|
||||||
|
|
||||||
output_filename = project_path.split('\\')[-1] + "_rep.txt"
|
output_filename = project_path.split('\\')[-1] + "_rep.txt"
|
||||||
try:
|
try:
|
||||||
with open(output_filename, "w", encoding="utf-8") as f:
|
with open(output_filename, "w", encoding="utf-8") as f:
|
||||||
f.write(tree_output)
|
f.write(tree_output)
|
||||||
print(f"\nПолная проектная документация, сохраненная в: {output_filename}")
|
print(f"\nПолная проектная документация, сохраненная в: {output_filename}")
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
print(f"Предупреждение: Не удалось сохранить файл - {e}")
|
print(f"Предупреждение: Не удалось сохранить файл - {e}")
|
||||||
|
|
||||||
print("Формирование структуры проекта успешно завершено!")
|
print("Формирование структуры проекта успешно завершено!")
|
||||||
Loading…
Add table
Reference in a new issue