#!/usr/bin/env python3 """Atualiza incrementalmente o índice RAG do G-ART. As credenciais são fornecidas pelo ambiente; este arquivo nunca deve conter senha. O indexador usa o banco ``rag_gart`` e o schema ``gart`` por padrão. """ from __future__ import annotations import hashlib import os import re import sys from pathlib import Path import psycopg2 import requests ROOT = Path(__file__).resolve().parents[1] DB_NAME = os.environ.get("RAG_DB_NAME", "rag_gart") DB_SCHEMA = os.environ.get("RAG_DB_SCHEMA", "gart") OLLAMA_URL = os.environ.get("OLLAMA_URL", "http://127.0.0.1:11434") EMBED_MODEL = os.environ.get("RAG_EMBED_MODEL", "nomic-embed-text") EMBED_DIM = int(os.environ.get("RAG_EMBED_DIM", "768")) INCLUDE_EXTENSIONS = { ".command", ".md", ".py", ".sh", ".sql", ".swift", ".txt", ".yml", ".yaml", } EXCLUDE_DIRS = { ".git", ".venv", ".pytest_cache", ".ruff_cache", "__pycache__", "build", "dist", "node_modules", "graphify-out", "bm", "models", "whisper", } EXCLUDE_FILES = {".env", "admin/genial-crm.env", "admin/genial-crm.local.env"} CHUNK_LINES = 60 CHUNK_OVERLAP = 10 CHUNK_MAX_CHARS = 5000 def _sql_id(value: str) -> str: return '"' + value.replace('"', '""') + '"' def _connect(): password = os.environ.get("RAG_DB_PASSWORD") if not password: raise RuntimeError("RAG_DB_PASSWORD não foi definida") return psycopg2.connect( host=os.environ.get("RAG_DB_HOST", "127.0.0.1"), port=os.environ.get("RAG_DB_PORT", "55435"), dbname=DB_NAME, user=os.environ.get("RAG_DB_USER", "gart_rag_indexer"), password=password, connect_timeout=5, ) def _iter_files(): for path in ROOT.rglob("*"): if not path.is_file() or path.suffix.lower() not in INCLUDE_EXTENSIONS: continue rel = path.relative_to(ROOT).as_posix() parts = set(path.relative_to(ROOT).parts) if parts & EXCLUDE_DIRS or rel in EXCLUDE_FILES or path.name in EXCLUDE_FILES: continue if any(part.startswith(".") for part in path.relative_to(ROOT).parts[:-1]): continue yield path, rel def _chunks(text: str): lines = text.splitlines() if not lines: return [] step = max(1, CHUNK_LINES - CHUNK_OVERLAP) result = [] for start in range(0, len(lines), step): window_start = start buffer = [] size = 0 for offset, line in enumerate(lines[start:start + CHUNK_LINES]): if buffer and size + len(line) + 1 > CHUNK_MAX_CHARS: result.append((window_start + 1, window_start + len(buffer), "\n".join(buffer).strip())) buffer = [] window_start = start + offset size = 0 buffer.append(line) size += len(line) + 1 if buffer: result.append((window_start + 1, window_start + len(buffer), "\n".join(buffer).strip())) if start + CHUNK_LINES >= len(lines): break return [(start, end, content) for start, end, content in result if content] def _facts(text: str, rel_path: str): lines = text.splitlines() summary = next( (line.strip().lstrip("#! ").strip() for line in lines[:30] if line.strip()), None, ) symbols = re.findall( r"^\s*(?:class|def|async\s+def|func|struct|enum|protocol|actor|interface)\s+([A-Za-z_]\w*)", text, re.MULTILINE, ) parts = Path(rel_path).parts module = parts[0] if len(parts) > 1 else None return module, Path(rel_path).stem, summary, sorted(set(symbols)), len(lines) class ChunkTooLargeError(Exception): """Chunk excede o contexto do modelo de embedding (ver EXCLUDE_FILES/CHUNK_MAX_CHARS).""" def _embed(text: str): response = requests.post( f"{OLLAMA_URL.rstrip('/')}/api/embeddings", json={"model": EMBED_MODEL, "prompt": f"search_document: {text}"}, timeout=60, ) if response.status_code == 500 and "context length" in response.text.lower(): raise ChunkTooLargeError(response.text) response.raise_for_status() vector = response.json()["embedding"] if len(vector) != EMBED_DIM: raise ValueError(f"embedding com {len(vector)} dimensões; esperado {EMBED_DIM}") return vector def _hash(text: str) -> str: return hashlib.md5(text.encode("utf-8")).hexdigest() def index(): schema = _sql_id(DB_SCHEMA) conn = _connect() conn.autocommit = False indexed = skipped = deleted = chunks_written = 0 seen = set() try: with conn.cursor() as cur: for path, rel_path in sorted(_iter_files(), key=lambda item: item[1]): try: text = path.read_text(encoding="utf-8", errors="ignore") except OSError as exc: print(f"[RAG] ignorado {rel_path}: {exc}", file=sys.stderr) continue seen.add(rel_path) digest = _hash(text) cur.execute(f"SELECT content_hash FROM {schema}.indexed_files WHERE file_path = %s", (rel_path,)) row = cur.fetchone() if row and row[0] == digest: skipped += 1 continue module, main_type, summary, symbols, n_lines = _facts(text, rel_path) cur.execute(f"DELETE FROM {schema}.code_chunks WHERE file_path = %s", (rel_path,)) for index_number, (start, end, content) in enumerate(_chunks(text)): try: vector = _embed(content) except ChunkTooLargeError: # Chunks densos em tokens (ex: tabelas de dados numéricas # como font_metrics.py) podem passar de CHUNK_MAX_CHARS em # caracteres mas estourar o contexto do modelo em tokens. # Pular o chunk em vez de abortar a transação inteira. print(f"[RAG] chunk grande demais, pulado: {rel_path}:{start}-{end}", file=sys.stderr) continue cur.execute( f"""INSERT INTO {schema}.code_chunks (file_path, content, chunk_index, embedding, content_hash, file_mtime, start_line, end_line, symbols, module, kind) VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)""", (rel_path, content, index_number, vector, digest, path.stat().st_mtime, start, end, ", ".join(symbols), module, "window"), ) chunks_written += 1 cur.execute( f"""INSERT INTO {schema}.file_index (file_path, module, main_type, public_symbols, summary, n_lines, content_hash) VALUES (%s, %s, %s, %s, %s, %s, %s) ON CONFLICT (file_path) DO UPDATE SET module = EXCLUDED.module, main_type = EXCLUDED.main_type, public_symbols = EXCLUDED.public_symbols, summary = EXCLUDED.summary, n_lines = EXCLUDED.n_lines, content_hash = EXCLUDED.content_hash, updated_at = CURRENT_TIMESTAMP""", (rel_path, module, main_type, symbols, summary, n_lines, digest), ) cur.execute( f"""INSERT INTO {schema}.indexed_files (file_path, content_hash) VALUES (%s, %s) ON CONFLICT (file_path) DO UPDATE SET content_hash = EXCLUDED.content_hash, updated_at = CURRENT_TIMESTAMP""", (rel_path, digest), ) indexed += 1 print(f"[RAG] {rel_path}") cur.execute(f"SELECT file_path FROM {schema}.indexed_files") for (rel_path,) in cur.fetchall(): if rel_path not in seen: cur.execute(f"DELETE FROM {schema}.code_chunks WHERE file_path = %s", (rel_path,)) cur.execute(f"DELETE FROM {schema}.file_index WHERE file_path = %s", (rel_path,)) cur.execute(f"DELETE FROM {schema}.indexed_files WHERE file_path = %s", (rel_path,)) deleted += 1 print(f"[RAG] removido {rel_path}") conn.commit() except Exception: conn.rollback() raise finally: conn.close() print(f"[RAG] concluído: {indexed} atualizado(s), {skipped} sem mudança, {deleted} removido(s), {chunks_written} chunk(s).") if __name__ == "__main__": index()