Files
gart/admin/update_rag.py
João HenriqueandClaude Sonnet 5 e9a17c1b62 feat(rag): provisiona busca RAG do G-ART e corrige indexação que abortava em chunk grande
Cria rag/ (schema, busca híbrida densa+lexical com RRF em search.py/
search_gart.sh, SETUP.md) — o projeto já tinha admin/update_rag.py para
indexar, mas nenhuma forma de consultar o índice. Corrige admin/update_rag.py:
um chunk denso em tokens (code/fcpxml/font_metrics.py) estourava o contexto
do modelo de embedding e derrubava a transação inteira; agora só aquele
chunk é pulado. Banco rag_gart provisionado no rag-hub-db compartilhado e
primeira indexação completa rodada (304 arquivos, 1702 chunks).

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-09-23 09:38:46 -04:00

217 lines
8.5 KiB
Python
Executable File

#!/usr/bin/env python3
"""Atualiza incrementalmente o índice RAG do G-ART.
As credenciais são fornecidas pelo ambiente; este arquivo nunca deve conter
senha. O indexador usa o banco ``rag_gart`` e o schema ``gart`` por padrão.
"""
from __future__ import annotations
import hashlib
import os
import re
import sys
from pathlib import Path
import psycopg2
import requests
ROOT = Path(__file__).resolve().parents[1]
DB_NAME = os.environ.get("RAG_DB_NAME", "rag_gart")
DB_SCHEMA = os.environ.get("RAG_DB_SCHEMA", "gart")
OLLAMA_URL = os.environ.get("OLLAMA_URL", "http://127.0.0.1:11434")
EMBED_MODEL = os.environ.get("RAG_EMBED_MODEL", "nomic-embed-text")
EMBED_DIM = int(os.environ.get("RAG_EMBED_DIM", "768"))
INCLUDE_EXTENSIONS = {
".command", ".md", ".py", ".sh", ".sql", ".swift", ".txt", ".yml", ".yaml",
}
EXCLUDE_DIRS = {
".git", ".venv", ".pytest_cache", ".ruff_cache", "__pycache__", "build",
"dist", "node_modules", "graphify-out", "bm", "models", "whisper",
}
EXCLUDE_FILES = {".env", "admin/genial-crm.env", "admin/genial-crm.local.env"}
CHUNK_LINES = 60
CHUNK_OVERLAP = 10
CHUNK_MAX_CHARS = 5000
def _sql_id(value: str) -> str:
return '"' + value.replace('"', '""') + '"'
def _connect():
password = os.environ.get("RAG_DB_PASSWORD")
if not password:
raise RuntimeError("RAG_DB_PASSWORD não foi definida")
return psycopg2.connect(
host=os.environ.get("RAG_DB_HOST", "127.0.0.1"),
port=os.environ.get("RAG_DB_PORT", "55435"),
dbname=DB_NAME,
user=os.environ.get("RAG_DB_USER", "gart_rag_indexer"),
password=password,
connect_timeout=5,
)
def _iter_files():
for path in ROOT.rglob("*"):
if not path.is_file() or path.suffix.lower() not in INCLUDE_EXTENSIONS:
continue
rel = path.relative_to(ROOT).as_posix()
parts = set(path.relative_to(ROOT).parts)
if parts & EXCLUDE_DIRS or rel in EXCLUDE_FILES or path.name in EXCLUDE_FILES:
continue
if any(part.startswith(".") for part in path.relative_to(ROOT).parts[:-1]):
continue
yield path, rel
def _chunks(text: str):
lines = text.splitlines()
if not lines:
return []
step = max(1, CHUNK_LINES - CHUNK_OVERLAP)
result = []
for start in range(0, len(lines), step):
window_start = start
buffer = []
size = 0
for offset, line in enumerate(lines[start:start + CHUNK_LINES]):
if buffer and size + len(line) + 1 > CHUNK_MAX_CHARS:
result.append((window_start + 1, window_start + len(buffer), "\n".join(buffer).strip()))
buffer = []
window_start = start + offset
size = 0
buffer.append(line)
size += len(line) + 1
if buffer:
result.append((window_start + 1, window_start + len(buffer), "\n".join(buffer).strip()))
if start + CHUNK_LINES >= len(lines):
break
return [(start, end, content) for start, end, content in result if content]
def _facts(text: str, rel_path: str):
lines = text.splitlines()
summary = next(
(line.strip().lstrip("#! ").strip() for line in lines[:30] if line.strip()),
None,
)
symbols = re.findall(
r"^\s*(?:class|def|async\s+def|func|struct|enum|protocol|actor|interface)\s+([A-Za-z_]\w*)",
text,
re.MULTILINE,
)
parts = Path(rel_path).parts
module = parts[0] if len(parts) > 1 else None
return module, Path(rel_path).stem, summary, sorted(set(symbols)), len(lines)
class ChunkTooLargeError(Exception):
"""Chunk excede o contexto do modelo de embedding (ver EXCLUDE_FILES/CHUNK_MAX_CHARS)."""
def _embed(text: str):
response = requests.post(
f"{OLLAMA_URL.rstrip('/')}/api/embeddings",
json={"model": EMBED_MODEL, "prompt": f"search_document: {text}"},
timeout=60,
)
if response.status_code == 500 and "context length" in response.text.lower():
raise ChunkTooLargeError(response.text)
response.raise_for_status()
vector = response.json()["embedding"]
if len(vector) != EMBED_DIM:
raise ValueError(f"embedding com {len(vector)} dimensões; esperado {EMBED_DIM}")
return vector
def _hash(text: str) -> str:
return hashlib.md5(text.encode("utf-8")).hexdigest()
def index():
schema = _sql_id(DB_SCHEMA)
conn = _connect()
conn.autocommit = False
indexed = skipped = deleted = chunks_written = 0
seen = set()
try:
with conn.cursor() as cur:
for path, rel_path in sorted(_iter_files(), key=lambda item: item[1]):
try:
text = path.read_text(encoding="utf-8", errors="ignore")
except OSError as exc:
print(f"[RAG] ignorado {rel_path}: {exc}", file=sys.stderr)
continue
seen.add(rel_path)
digest = _hash(text)
cur.execute(f"SELECT content_hash FROM {schema}.indexed_files WHERE file_path = %s", (rel_path,))
row = cur.fetchone()
if row and row[0] == digest:
skipped += 1
continue
module, main_type, summary, symbols, n_lines = _facts(text, rel_path)
cur.execute(f"DELETE FROM {schema}.code_chunks WHERE file_path = %s", (rel_path,))
for index_number, (start, end, content) in enumerate(_chunks(text)):
try:
vector = _embed(content)
except ChunkTooLargeError:
# Chunks densos em tokens (ex: tabelas de dados numéricas
# como font_metrics.py) podem passar de CHUNK_MAX_CHARS em
# caracteres mas estourar o contexto do modelo em tokens.
# Pular o chunk em vez de abortar a transação inteira.
print(f"[RAG] chunk grande demais, pulado: {rel_path}:{start}-{end}", file=sys.stderr)
continue
cur.execute(
f"""INSERT INTO {schema}.code_chunks
(file_path, content, chunk_index, embedding, content_hash,
file_mtime, start_line, end_line, symbols, module, kind)
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)""",
(rel_path, content, index_number, vector, digest,
path.stat().st_mtime, start, end, ", ".join(symbols), module, "window"),
)
chunks_written += 1
cur.execute(
f"""INSERT INTO {schema}.file_index
(file_path, module, main_type, public_symbols, summary, n_lines, content_hash)
VALUES (%s, %s, %s, %s, %s, %s, %s)
ON CONFLICT (file_path) DO UPDATE SET
module = EXCLUDED.module, main_type = EXCLUDED.main_type,
public_symbols = EXCLUDED.public_symbols, summary = EXCLUDED.summary,
n_lines = EXCLUDED.n_lines, content_hash = EXCLUDED.content_hash,
updated_at = CURRENT_TIMESTAMP""",
(rel_path, module, main_type, symbols, summary, n_lines, digest),
)
cur.execute(
f"""INSERT INTO {schema}.indexed_files (file_path, content_hash)
VALUES (%s, %s)
ON CONFLICT (file_path) DO UPDATE SET
content_hash = EXCLUDED.content_hash, updated_at = CURRENT_TIMESTAMP""",
(rel_path, digest),
)
indexed += 1
print(f"[RAG] {rel_path}")
cur.execute(f"SELECT file_path FROM {schema}.indexed_files")
for (rel_path,) in cur.fetchall():
if rel_path not in seen:
cur.execute(f"DELETE FROM {schema}.code_chunks WHERE file_path = %s", (rel_path,))
cur.execute(f"DELETE FROM {schema}.file_index WHERE file_path = %s", (rel_path,))
cur.execute(f"DELETE FROM {schema}.indexed_files WHERE file_path = %s", (rel_path,))
deleted += 1
print(f"[RAG] removido {rel_path}")
conn.commit()
except Exception:
conn.rollback()
raise
finally:
conn.close()
print(f"[RAG] concluído: {indexed} atualizado(s), {skipped} sem mudança, {deleted} removido(s), {chunks_written} chunk(s).")
if __name__ == "__main__":
index()