#!/usr/bin/env python3 """ =============================================================================== FILE: scripts/diagnostics/make_etl_snapshot.py ROLE: Компактная динамическая генерация слепка ETL-конвейера, сервисов и БД. Исключает исторические манифесты docs, диагностический шум и пустые файлы. =============================================================================== """ import os ROOT_DIR = os.path.abspath(os.path.join(os.path.dirname(__file__), "..", "..")) OUTPUT_FILE = os.path.join(ROOT_DIR, "etl_code_snapshot.md") # 1. Отдельные ключевые файлы в корне проекта ROOT_EXPLICIT_FILES = [ "config.py", "exceptions.json", "main_etl.py" ] # 2. Директории для автоматического сканирования INCLUDED_DIRS = [ "core", "services", "scripts", "docs" ] # 3. Разрешенные расширения файлов ALLOWED_EXTENSIONS = { ".py": "py", ".sh": "bash", ".json": "json", ".md": "markdown" } # 4. Папки, которые категорически игнорируются IGNORE_DIRS = { "venv", ".venv", ".git", "__pycache__", "data", "output", "logs", "node_modules", "static", ".idea", ".vscode", "diagnostics" } # 5. Файлы, исключаемые для экономии контекста (тяжелые исторические манифесты и временные дампы) IGNORE_FILES = { "etl_code_snapshot.md", "web_api_code_snapshot.md", "db_dump_full.xlsx", # Исключаем исторические манифесты из docs/ (~60 КБ дублирующего текста) "PROJECT BRAIN_ SCUD Orion AI & Context API (Master Manifesto v5.0).md", "SCUD Orion AI — Полная энциклопедическая хроника, архитектурный паспорт и технический контекст (v4.0).md" } def collect_target_files(): """Автоматически собирает список файлов проекта без шума и устаревших манифестов.""" target_files = [] # Добавляем ключевые файлы из корня for fname in ROOT_EXPLICIT_FILES: fpath = os.path.join(ROOT_DIR, fname) if os.path.isfile(fpath): target_files.append(fname) # Рекурсивный обход разрешенных каталогов for d_name in INCLUDED_DIRS: base_dir = os.path.join(ROOT_DIR, d_name) if not os.path.exists(base_dir): continue for root, dirs, files in os.walk(base_dir): dirs[:] = [d for d in dirs if d not in IGNORE_DIRS and not d.startswith(".")] for file in sorted(files): if file in IGNORE_FILES or file.startswith("."): continue _, ext = os.path.splitext(file) if ext.lower() in ALLOWED_EXTENSIONS: full_path = os.path.join(root, file) # Пропускаем пустые __init__.py (0 байт) if file == "__init__.py" and os.path.getsize(full_path) == 0: continue rel_path = os.path.relpath(full_path, ROOT_DIR) target_files.append(rel_path) return sorted(target_files) def create_etl_snapshot(): files_to_pack = collect_target_files() content = ["# 📦 КОМПАКТНЫЙ ETL-СЛЕПОК ИСХОДНОГО КОДА (СКУД ⟷ 1С & DB CORE)\n"] included_count = 0 for rel_path in files_to_pack: full_path = os.path.join(ROOT_DIR, rel_path) ext = os.path.splitext(rel_path)[1].lower() lang = ALLOWED_EXTENSIONS.get(ext, "text") try: with open(full_path, "r", encoding="utf-8") as f: file_text = f.read() content.append(f"## File: `./{rel_path}`\n```{lang}\n{file_text}\n```\n") included_count += 1 except Exception as e: print(f"[⚠️] Ошибка чтения {rel_path}: {e}") with open(OUTPUT_FILE, "w", encoding="utf-8") as f: f.write("\n".join(content)) size_kb = os.path.getsize(OUTPUT_FILE) / 1024 print(f"\n[✓] Компактный ETL-слепок успешно создан: {OUTPUT_FILE}") print(f" Включено файлов: {included_count} | Размер: {size_kb:.1f} KB\n") if __name__ == "__main__": create_etl_snapshot()