From bc351e995a4022cea058f67b4ccd71ca4cdb6d56 Mon Sep 17 00:00:00 2001 From: Egor Deev Date: Mon, 5 Oct 2026 22:54:01 +0000 Subject: [PATCH] =?UTF-8?q?=D0=9A=D0=BE=D0=BD=D0=B2=D0=B5=D1=80=D1=82?= =?UTF-8?q?=D0=B5=D1=80=20=D0=BF=D0=B5=D1=80=D0=B5=D0=BD=D0=B5=D1=81=D1=91?= =?UTF-8?q?=D0=BD=20=D0=B2=20=D0=BF=D0=B0=D0=BA=D0=B5=D1=82=20md2gost,=20?= =?UTF-8?q?=D0=BF=D0=B5=D1=80=D0=B5=D0=B2=D0=BE=D0=B4=D1=8B=20=D1=81=D1=82?= =?UTF-8?q?=D1=80=D0=BE=D0=BA=20=E2=80=94=20LF?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Отдельный пакет — чтобы конвертер можно было ставить из PyPI и запускать из командной строки без бота. --- .gitattributes | 2 + md_to_docx.py => md2gost/converter.py | 0 rep_to_txt.py | 328 +++++++++++++------------- 3 files changed, 166 insertions(+), 164 deletions(-) create mode 100644 .gitattributes rename md_to_docx.py => md2gost/converter.py (100%) diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..7fc65c6 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,2 @@ +* text=auto eol=lf +*.docx binary diff --git a/md_to_docx.py b/md2gost/converter.py similarity index 100% rename from md_to_docx.py rename to md2gost/converter.py diff --git a/rep_to_txt.py b/rep_to_txt.py index e320764..e1f90d9 100644 --- a/rep_to_txt.py +++ b/rep_to_txt.py @@ -1,165 +1,165 @@ -import os - -IGNORE_PATTERNS = { - '.git', '.svn', '.hg', # Version control systems - '__pycache__', '.pytest_cache', # Python artifacts - 'node_modules', '.npm', # Node.js dependencies - 'target', 'build', 'dist', # Build outputs - '.idea', '.vscode', # IDE metadata - '.DS_Store', 'Thumbs.db', # OS metadata - '.pro.user' # QT user config -} - -BINARY_EXTENSIONS = { - '.exe', '.dll', '.so', '.dylib', '.zip', '.tar', '.gz', '.rar', '.7z', - '.jpg', '.jpeg', '.png', '.gif', '.bmp', '.ico', '.svg', '.webp', - '.mp3', '.mp4', '.avi', '.mov', '.mkv', '.wmv', '.flv', - '.pdf', '.doc', '.docx', '.xls', '.xlsx', '.ppt', '.pptx', - '.bin', '.dat', '.db', '.sqlite', '.mdb' -} - - -def scan_directory(path, prefix=""): - """Рекурсивное сканирование с форматированием дерева""" - items = [] - try: - entries = sorted(os.listdir(path)) - # Critical filtering layer для performance optimization - dirs = [e for e in entries if os.path.isdir(os.path.join(path, e)) and e not in IGNORE_PATTERNS] - files = [e for e in entries if os.path.isfile(os.path.join(path, e)) and e not in IGNORE_PATTERNS] - all_items = dirs + files - - for i, item in enumerate(all_items): - item_path = os.path.join(path, item) - is_last_item = (i == len(all_items) - 1) - - if is_last_item: - current_prefix = prefix + "└── " - next_prefix = prefix + " " - else: - current_prefix = prefix + "├── " - next_prefix = prefix + "│ " - - items.append(current_prefix + item) - - if os.path.isdir(item_path): - items.extend(scan_directory(item_path, next_prefix)) - - except PermissionError: - items.append(prefix + "└── [Access Denied]") - - return items - - -def generate_complete_project_structure(root_path): - """Генератор проектной документации корпоративного уровня""" - if not os.path.exists(root_path): - return f"Error: Path {root_path} does not exist" - - result = [] - - # Этап 1: Создание древовидной структуры - root_name = os.path.basename(root_path) or root_path - result.append(root_name) - result.extend(scan_directory(root_path)) - - # Этап 2: Полное извлечение содержимого файла - result.append("\n") # Separator между разделами дерева и содержимым - result.extend(extract_all_file_contents(root_path)) - - return "\n".join(result) - - -def extract_all_file_contents(root_path): - """Механизм извлечения контента с обработкой файлов""" - content_lines = [] - - for root, dirs, files in os.walk(root_path): - dirs[:] = [d for d in dirs if d not in IGNORE_PATTERNS] - - for file in sorted(files): - if file in IGNORE_PATTERNS: - continue - - file_path = os.path.join(root, file) - relative_path = os.path.relpath(file_path, root_path) - - content_lines.extend(process_single_file(relative_path, file_path)) - - return content_lines - - -def process_single_file(relative_path, file_path): - """Обработка файлов""" - content_lines = [] - - # Раздел заголовка - content_lines.append("\n" + "-" * 80) - content_lines.append(f"{relative_path}:") - content_lines.append("-" * 80) - - file_ext = os.path.splitext(relative_path)[1].lower() - - # Обнаружение двоичных файлов и генерация URL-адресов - if file_ext in BINARY_EXTENSIONS or is_likely_binary(file_path): - # GitHub raw URL - if file_ext in {'.png', '.jpg', '.jpeg', '.gif', '.svg', '.ico'}: - # Структура URL - настраивается на основе фактического хранилища - github_url = f"https://raw.githubusercontent.com/.../{relative_path.replace(os.sep, '/')}" - content_lines.append(github_url) - else: - content_lines.append("[Binary file - content not displayed]") - else: - # Извлечение содержимого текстового файла - content_lines.extend(extract_text_content(file_path)) - - content_lines.append("") - return content_lines - - -def extract_text_content(file_path): - """Резервное извлечение с несколькими кодировками""" - encodings_priority = ['utf-8', 'utf-8-sig', 'cp1251', 'latin1', 'cp1252'] - - for encoding in encodings_priority: - try: - with open(file_path, 'r', encoding=encoding) as f: - lines = f.readlines() - return [f"{i:4} | {line.rstrip()}" for i, line in enumerate(lines, 1)] - except (UnicodeDecodeError, UnicodeError): - continue - except Exception as e: - return [f"ERROR: Не удается прочитать файл - {e}"] - - return ["WARNING: Кодировка файла, не поддерживаемая для извлечения текста"] - - -def is_likely_binary(file_path): - """Эвристическое обнаружение двоичных файлов для крайних случаев""" - try: - with open(file_path, 'rb') as f: - chunk = f.read(8192) - # Обнаружение нулевого байта - надежный бинарный индикатор - return b'\x00' in chunk - except: - return True - - -if __name__ == "__main__": - # Конфигурация: измените путь к целевому каталогу проекта - project_path = r"D:\Programs\GitHub\deev.space\static" - # project_path = r"D:/Programs/GitHub/openoffice" - # project_path = "." - - print("Приступаем к формированию комплексной структуры проекта...") - tree_output = generate_complete_project_structure(project_path) - - output_filename = project_path.split('\\')[-1] + "_rep.txt" - try: - with open(output_filename, "w", encoding="utf-8") as f: - f.write(tree_output) - print(f"\nПолная проектная документация, сохраненная в: {output_filename}") - except Exception as e: - print(f"Предупреждение: Не удалось сохранить файл - {e}") - +import os + +IGNORE_PATTERNS = { + '.git', '.svn', '.hg', # Version control systems + '__pycache__', '.pytest_cache', # Python artifacts + 'node_modules', '.npm', # Node.js dependencies + 'target', 'build', 'dist', # Build outputs + '.idea', '.vscode', # IDE metadata + '.DS_Store', 'Thumbs.db', # OS metadata + '.pro.user' # QT user config +} + +BINARY_EXTENSIONS = { + '.exe', '.dll', '.so', '.dylib', '.zip', '.tar', '.gz', '.rar', '.7z', + '.jpg', '.jpeg', '.png', '.gif', '.bmp', '.ico', '.svg', '.webp', + '.mp3', '.mp4', '.avi', '.mov', '.mkv', '.wmv', '.flv', + '.pdf', '.doc', '.docx', '.xls', '.xlsx', '.ppt', '.pptx', + '.bin', '.dat', '.db', '.sqlite', '.mdb' +} + + +def scan_directory(path, prefix=""): + """Рекурсивное сканирование с форматированием дерева""" + items = [] + try: + entries = sorted(os.listdir(path)) + # Critical filtering layer для performance optimization + dirs = [e for e in entries if os.path.isdir(os.path.join(path, e)) and e not in IGNORE_PATTERNS] + files = [e for e in entries if os.path.isfile(os.path.join(path, e)) and e not in IGNORE_PATTERNS] + all_items = dirs + files + + for i, item in enumerate(all_items): + item_path = os.path.join(path, item) + is_last_item = (i == len(all_items) - 1) + + if is_last_item: + current_prefix = prefix + "└── " + next_prefix = prefix + " " + else: + current_prefix = prefix + "├── " + next_prefix = prefix + "│ " + + items.append(current_prefix + item) + + if os.path.isdir(item_path): + items.extend(scan_directory(item_path, next_prefix)) + + except PermissionError: + items.append(prefix + "└── [Access Denied]") + + return items + + +def generate_complete_project_structure(root_path): + """Генератор проектной документации корпоративного уровня""" + if not os.path.exists(root_path): + return f"Error: Path {root_path} does not exist" + + result = [] + + # Этап 1: Создание древовидной структуры + root_name = os.path.basename(root_path) or root_path + result.append(root_name) + result.extend(scan_directory(root_path)) + + # Этап 2: Полное извлечение содержимого файла + result.append("\n") # Separator между разделами дерева и содержимым + result.extend(extract_all_file_contents(root_path)) + + return "\n".join(result) + + +def extract_all_file_contents(root_path): + """Механизм извлечения контента с обработкой файлов""" + content_lines = [] + + for root, dirs, files in os.walk(root_path): + dirs[:] = [d for d in dirs if d not in IGNORE_PATTERNS] + + for file in sorted(files): + if file in IGNORE_PATTERNS: + continue + + file_path = os.path.join(root, file) + relative_path = os.path.relpath(file_path, root_path) + + content_lines.extend(process_single_file(relative_path, file_path)) + + return content_lines + + +def process_single_file(relative_path, file_path): + """Обработка файлов""" + content_lines = [] + + # Раздел заголовка + content_lines.append("\n" + "-" * 80) + content_lines.append(f"{relative_path}:") + content_lines.append("-" * 80) + + file_ext = os.path.splitext(relative_path)[1].lower() + + # Обнаружение двоичных файлов и генерация URL-адресов + if file_ext in BINARY_EXTENSIONS or is_likely_binary(file_path): + # GitHub raw URL + if file_ext in {'.png', '.jpg', '.jpeg', '.gif', '.svg', '.ico'}: + # Структура URL - настраивается на основе фактического хранилища + github_url = f"https://raw.githubusercontent.com/.../{relative_path.replace(os.sep, '/')}" + content_lines.append(github_url) + else: + content_lines.append("[Binary file - content not displayed]") + else: + # Извлечение содержимого текстового файла + content_lines.extend(extract_text_content(file_path)) + + content_lines.append("") + return content_lines + + +def extract_text_content(file_path): + """Резервное извлечение с несколькими кодировками""" + encodings_priority = ['utf-8', 'utf-8-sig', 'cp1251', 'latin1', 'cp1252'] + + for encoding in encodings_priority: + try: + with open(file_path, 'r', encoding=encoding) as f: + lines = f.readlines() + return [f"{i:4} | {line.rstrip()}" for i, line in enumerate(lines, 1)] + except (UnicodeDecodeError, UnicodeError): + continue + except Exception as e: + return [f"ERROR: Не удается прочитать файл - {e}"] + + return ["WARNING: Кодировка файла, не поддерживаемая для извлечения текста"] + + +def is_likely_binary(file_path): + """Эвристическое обнаружение двоичных файлов для крайних случаев""" + try: + with open(file_path, 'rb') as f: + chunk = f.read(8192) + # Обнаружение нулевого байта - надежный бинарный индикатор + return b'\x00' in chunk + except: + return True + + +if __name__ == "__main__": + # Конфигурация: измените путь к целевому каталогу проекта + project_path = r"D:\Programs\GitHub\deev.space\static" + # project_path = r"D:/Programs/GitHub/openoffice" + # project_path = "." + + print("Приступаем к формированию комплексной структуры проекта...") + tree_output = generate_complete_project_structure(project_path) + + output_filename = project_path.split('\\')[-1] + "_rep.txt" + try: + with open(output_filename, "w", encoding="utf-8") as f: + f.write(tree_output) + print(f"\nПолная проектная документация, сохраненная в: {output_filename}") + except Exception as e: + print(f"Предупреждение: Не удалось сохранить файл - {e}") + print("Формирование структуры проекта успешно завершено!") \ No newline at end of file