119 lines
4.6 KiB
Python
119 lines
4.6 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
===============================================================================
|
|
FILE: scripts/diagnostics/make_etl_snapshot.py
|
|
ROLE: Компактная динамическая генерация слепка ETL-конвейера, сервисов и БД.
|
|
Исключает исторические манифесты docs, диагностический шум и пустые файлы.
|
|
===============================================================================
|
|
"""
|
|
|
|
import os
|
|
|
|
ROOT_DIR = os.path.abspath(os.path.join(os.path.dirname(__file__), "..", ".."))
|
|
OUTPUT_FILE = os.path.join(ROOT_DIR, "etl_code_snapshot.md")
|
|
|
|
# 1. Отдельные ключевые файлы в корне проекта
|
|
ROOT_EXPLICIT_FILES = [
|
|
"config.py",
|
|
"exceptions.json",
|
|
"main_etl.py"
|
|
]
|
|
|
|
# 2. Директории для автоматического сканирования
|
|
INCLUDED_DIRS = [
|
|
"core",
|
|
"services",
|
|
"scripts",
|
|
"docs"
|
|
]
|
|
|
|
# 3. Разрешенные расширения файлов
|
|
ALLOWED_EXTENSIONS = {
|
|
".py": "py",
|
|
".sh": "bash",
|
|
".json": "json",
|
|
".md": "markdown"
|
|
}
|
|
|
|
# 4. Папки, которые категорически игнорируются
|
|
IGNORE_DIRS = {
|
|
"venv", ".venv", ".git", "__pycache__", "data", "output", "logs",
|
|
"node_modules", "static", ".idea", ".vscode", "diagnostics"
|
|
}
|
|
|
|
# 5. Файлы, исключаемые для экономии контекста (тяжелые исторические манифесты и временные дампы)
|
|
IGNORE_FILES = {
|
|
"etl_code_snapshot.md",
|
|
"web_api_code_snapshot.md",
|
|
"db_dump_full.xlsx",
|
|
# Исключаем исторические манифесты из docs/ (~60 КБ дублирующего текста)
|
|
"PROJECT BRAIN_ SCUD Orion AI & Context API (Master Manifesto v5.0).md",
|
|
"SCUD Orion AI — Полная энциклопедическая хроника, архитектурный паспорт и технический контекст (v4.0).md"
|
|
}
|
|
|
|
|
|
def collect_target_files():
|
|
"""Автоматически собирает список файлов проекта без шума и устаревших манифестов."""
|
|
target_files = []
|
|
|
|
# Добавляем ключевые файлы из корня
|
|
for fname in ROOT_EXPLICIT_FILES:
|
|
fpath = os.path.join(ROOT_DIR, fname)
|
|
if os.path.isfile(fpath):
|
|
target_files.append(fname)
|
|
|
|
# Рекурсивный обход разрешенных каталогов
|
|
for d_name in INCLUDED_DIRS:
|
|
base_dir = os.path.join(ROOT_DIR, d_name)
|
|
if not os.path.exists(base_dir):
|
|
continue
|
|
|
|
for root, dirs, files in os.walk(base_dir):
|
|
dirs[:] = [d for d in dirs if d not in IGNORE_DIRS and not d.startswith(".")]
|
|
|
|
for file in sorted(files):
|
|
if file in IGNORE_FILES or file.startswith("."):
|
|
continue
|
|
|
|
_, ext = os.path.splitext(file)
|
|
if ext.lower() in ALLOWED_EXTENSIONS:
|
|
full_path = os.path.join(root, file)
|
|
|
|
# Пропускаем пустые __init__.py (0 байт)
|
|
if file == "__init__.py" and os.path.getsize(full_path) == 0:
|
|
continue
|
|
|
|
rel_path = os.path.relpath(full_path, ROOT_DIR)
|
|
target_files.append(rel_path)
|
|
|
|
return sorted(target_files)
|
|
|
|
|
|
def create_etl_snapshot():
|
|
files_to_pack = collect_target_files()
|
|
content = ["# 📦 КОМПАКТНЫЙ ETL-СЛЕПОК ИСХОДНОГО КОДА (СКУД ⟷ 1С & DB CORE)\n"]
|
|
included_count = 0
|
|
|
|
for rel_path in files_to_pack:
|
|
full_path = os.path.join(ROOT_DIR, rel_path)
|
|
ext = os.path.splitext(rel_path)[1].lower()
|
|
lang = ALLOWED_EXTENSIONS.get(ext, "text")
|
|
|
|
try:
|
|
with open(full_path, "r", encoding="utf-8") as f:
|
|
file_text = f.read()
|
|
content.append(f"## File: `./{rel_path}`\n```{lang}\n{file_text}\n```\n")
|
|
included_count += 1
|
|
except Exception as e:
|
|
print(f"[⚠️] Ошибка чтения {rel_path}: {e}")
|
|
|
|
with open(OUTPUT_FILE, "w", encoding="utf-8") as f:
|
|
f.write("\n".join(content))
|
|
|
|
size_kb = os.path.getsize(OUTPUT_FILE) / 1024
|
|
print(f"\n[✓] Компактный ETL-слепок успешно создан: {OUTPUT_FILE}")
|
|
print(f" Включено файлов: {included_count} | Размер: {size_kb:.1f} KB\n")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
create_etl_snapshot() |