feat(rag): provisiona busca RAG do G-ART e corrige indexação que abortava em chunk grande

Cria rag/ (schema, busca híbrida densa+lexical com RRF em search.py/
search_gart.sh, SETUP.md) — o projeto já tinha admin/update_rag.py para
indexar, mas nenhuma forma de consultar o índice. Corrige admin/update_rag.py:
um chunk denso em tokens (code/fcpxml/font_metrics.py) estourava o contexto
do modelo de embedding e derrubava a transação inteira; agora só aquele
chunk é pulado. Banco rag_gart provisionado no rag-hub-db compartilhado e
primeira indexação completa rodada (304 arquivos, 1702 chunks).

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
João Henrique
2026-09-23 09:38:46 -04:00
co-authored by Claude Sonnet 5
parent d13f643ebc
commit e9a17c1b62
11 changed files with 1066 additions and 0 deletions
+12
View File
@@ -0,0 +1,12 @@
# Copie para admin/gart-rag.env e preencha a senha. Este arquivo é apenas um
# modelo; admin/gart-rag.env é ignorado pelo git.
RAG_DB_HOST=127.0.0.1
RAG_DB_PORT=55435
RAG_DB_NAME=rag_gart
RAG_DB_SCHEMA=gart
RAG_DB_USER=gart_rag_indexer
RAG_DB_PASSWORD=
# Ollama que fornece nomic-embed-text.
OLLAMA_URL=http://127.0.0.1:11434
RAG_EMBED_MODEL=nomic-embed-text
+68
View File
@@ -0,0 +1,68 @@
#!/bin/zsh
# Atualiza incrementalmente a RAG do G-ART usando o banco compartilhado.
#
# Credenciais: defina RAG_DB_PASSWORD no ambiente ou crie
# admin/gart-rag.env (ignorado pelo git). O arquivo pode conter também
# RAG_DB_USER, RAG_DB_PORT, OLLAMA_URL e RAG_EMBED_MODEL.
set -euo pipefail
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
ENV_FILE="$ROOT/admin/gart-rag.env"
if [[ -f "$ENV_FILE" ]]; then
set -a
source "$ENV_FILE"
set +a
fi
PYTHON="${RAG_PYTHON:-}"
if [[ -z "$PYTHON" ]]; then
for candidate in "$ROOT/admin/.venv/bin/python3" "$ROOT/rag/.venv/bin/python3"; do
if [[ -x "$candidate" ]]; then PYTHON="$candidate"; break; fi
done
fi
PYTHON="${PYTHON:-$(command -v python3)}"
if ! "$PYTHON" -c 'import psycopg2, requests' >/dev/null 2>&1; then
echo "ERRO: o Python da RAG precisa dos pacotes psycopg2 e requests." >&2
echo "Instale-os no ambiente indicado por RAG_PYTHON e tente novamente." >&2
exit 1
fi
if [[ -z "${RAG_DB_PASSWORD:-}" ]]; then
echo "ERRO: defina RAG_DB_PASSWORD ou configure $ENV_FILE" >&2
exit 1
fi
HOST="${RAG_VPS_HOST:-179.197.228.240}"
LOCAL_PORT="${RAG_DB_PORT:-55435}"
REMOTE_PORT="${RAG_REMOTE_PORT:-55435}"
TUNNEL_PID=""
cleanup() {
if [[ -n "$TUNNEL_PID" ]] && kill -0 "$TUNNEL_PID" 2>/dev/null; then
kill "$TUNNEL_PID" 2>/dev/null || true
wait "$TUNNEL_PID" 2>/dev/null || true
fi
}
trap cleanup EXIT
if ! nc -z 127.0.0.1 "$LOCAL_PORT" 2>/dev/null; then
echo "==> Abrindo túnel RAG (127.0.0.1:$LOCAL_PORT)..."
ssh -N -o ExitOnForwardFailure=yes -o ServerAliveInterval=30 \
-o ServerAliveCountMax=3 -L "127.0.0.1:$LOCAL_PORT:127.0.0.1:$REMOTE_PORT" \
"${RAG_VPS_USER:-root}@$HOST" &
TUNNEL_PID=$!
for _ in {1..20}; do
nc -z 127.0.0.1 "$LOCAL_PORT" 2>/dev/null && break
kill -0 "$TUNNEL_PID" 2>/dev/null || break
sleep 0.25
done
fi
if ! nc -z 127.0.0.1 "$LOCAL_PORT" 2>/dev/null; then
echo "ERRO: não foi possível abrir o túnel RAG." >&2
exit 1
fi
echo "==> Atualizando RAG do G-ART (incremental)..."
cd "$ROOT"
exec "$PYTHON" "$ROOT/admin/update_rag.py"
+216
View File
@@ -0,0 +1,216 @@
#!/usr/bin/env python3
"""Atualiza incrementalmente o índice RAG do G-ART.
As credenciais são fornecidas pelo ambiente; este arquivo nunca deve conter
senha. O indexador usa o banco ``rag_gart`` e o schema ``gart`` por padrão.
"""
from __future__ import annotations
import hashlib
import os
import re
import sys
from pathlib import Path
import psycopg2
import requests
ROOT = Path(__file__).resolve().parents[1]
DB_NAME = os.environ.get("RAG_DB_NAME", "rag_gart")
DB_SCHEMA = os.environ.get("RAG_DB_SCHEMA", "gart")
OLLAMA_URL = os.environ.get("OLLAMA_URL", "http://127.0.0.1:11434")
EMBED_MODEL = os.environ.get("RAG_EMBED_MODEL", "nomic-embed-text")
EMBED_DIM = int(os.environ.get("RAG_EMBED_DIM", "768"))
INCLUDE_EXTENSIONS = {
".command", ".md", ".py", ".sh", ".sql", ".swift", ".txt", ".yml", ".yaml",
}
EXCLUDE_DIRS = {
".git", ".venv", ".pytest_cache", ".ruff_cache", "__pycache__", "build",
"dist", "node_modules", "graphify-out", "bm", "models", "whisper",
}
EXCLUDE_FILES = {".env", "admin/genial-crm.env", "admin/genial-crm.local.env"}
CHUNK_LINES = 60
CHUNK_OVERLAP = 10
CHUNK_MAX_CHARS = 5000
def _sql_id(value: str) -> str:
return '"' + value.replace('"', '""') + '"'
def _connect():
password = os.environ.get("RAG_DB_PASSWORD")
if not password:
raise RuntimeError("RAG_DB_PASSWORD não foi definida")
return psycopg2.connect(
host=os.environ.get("RAG_DB_HOST", "127.0.0.1"),
port=os.environ.get("RAG_DB_PORT", "55435"),
dbname=DB_NAME,
user=os.environ.get("RAG_DB_USER", "gart_rag_indexer"),
password=password,
connect_timeout=5,
)
def _iter_files():
for path in ROOT.rglob("*"):
if not path.is_file() or path.suffix.lower() not in INCLUDE_EXTENSIONS:
continue
rel = path.relative_to(ROOT).as_posix()
parts = set(path.relative_to(ROOT).parts)
if parts & EXCLUDE_DIRS or rel in EXCLUDE_FILES or path.name in EXCLUDE_FILES:
continue
if any(part.startswith(".") for part in path.relative_to(ROOT).parts[:-1]):
continue
yield path, rel
def _chunks(text: str):
lines = text.splitlines()
if not lines:
return []
step = max(1, CHUNK_LINES - CHUNK_OVERLAP)
result = []
for start in range(0, len(lines), step):
window_start = start
buffer = []
size = 0
for offset, line in enumerate(lines[start:start + CHUNK_LINES]):
if buffer and size + len(line) + 1 > CHUNK_MAX_CHARS:
result.append((window_start + 1, window_start + len(buffer), "\n".join(buffer).strip()))
buffer = []
window_start = start + offset
size = 0
buffer.append(line)
size += len(line) + 1
if buffer:
result.append((window_start + 1, window_start + len(buffer), "\n".join(buffer).strip()))
if start + CHUNK_LINES >= len(lines):
break
return [(start, end, content) for start, end, content in result if content]
def _facts(text: str, rel_path: str):
lines = text.splitlines()
summary = next(
(line.strip().lstrip("#! ").strip() for line in lines[:30] if line.strip()),
None,
)
symbols = re.findall(
r"^\s*(?:class|def|async\s+def|func|struct|enum|protocol|actor|interface)\s+([A-Za-z_]\w*)",
text,
re.MULTILINE,
)
parts = Path(rel_path).parts
module = parts[0] if len(parts) > 1 else None
return module, Path(rel_path).stem, summary, sorted(set(symbols)), len(lines)
class ChunkTooLargeError(Exception):
"""Chunk excede o contexto do modelo de embedding (ver EXCLUDE_FILES/CHUNK_MAX_CHARS)."""
def _embed(text: str):
response = requests.post(
f"{OLLAMA_URL.rstrip('/')}/api/embeddings",
json={"model": EMBED_MODEL, "prompt": f"search_document: {text}"},
timeout=60,
)
if response.status_code == 500 and "context length" in response.text.lower():
raise ChunkTooLargeError(response.text)
response.raise_for_status()
vector = response.json()["embedding"]
if len(vector) != EMBED_DIM:
raise ValueError(f"embedding com {len(vector)} dimensões; esperado {EMBED_DIM}")
return vector
def _hash(text: str) -> str:
return hashlib.md5(text.encode("utf-8")).hexdigest()
def index():
schema = _sql_id(DB_SCHEMA)
conn = _connect()
conn.autocommit = False
indexed = skipped = deleted = chunks_written = 0
seen = set()
try:
with conn.cursor() as cur:
for path, rel_path in sorted(_iter_files(), key=lambda item: item[1]):
try:
text = path.read_text(encoding="utf-8", errors="ignore")
except OSError as exc:
print(f"[RAG] ignorado {rel_path}: {exc}", file=sys.stderr)
continue
seen.add(rel_path)
digest = _hash(text)
cur.execute(f"SELECT content_hash FROM {schema}.indexed_files WHERE file_path = %s", (rel_path,))
row = cur.fetchone()
if row and row[0] == digest:
skipped += 1
continue
module, main_type, summary, symbols, n_lines = _facts(text, rel_path)
cur.execute(f"DELETE FROM {schema}.code_chunks WHERE file_path = %s", (rel_path,))
for index_number, (start, end, content) in enumerate(_chunks(text)):
try:
vector = _embed(content)
except ChunkTooLargeError:
# Chunks densos em tokens (ex: tabelas de dados numéricas
# como font_metrics.py) podem passar de CHUNK_MAX_CHARS em
# caracteres mas estourar o contexto do modelo em tokens.
# Pular o chunk em vez de abortar a transação inteira.
print(f"[RAG] chunk grande demais, pulado: {rel_path}:{start}-{end}", file=sys.stderr)
continue
cur.execute(
f"""INSERT INTO {schema}.code_chunks
(file_path, content, chunk_index, embedding, content_hash,
file_mtime, start_line, end_line, symbols, module, kind)
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)""",
(rel_path, content, index_number, vector, digest,
path.stat().st_mtime, start, end, ", ".join(symbols), module, "window"),
)
chunks_written += 1
cur.execute(
f"""INSERT INTO {schema}.file_index
(file_path, module, main_type, public_symbols, summary, n_lines, content_hash)
VALUES (%s, %s, %s, %s, %s, %s, %s)
ON CONFLICT (file_path) DO UPDATE SET
module = EXCLUDED.module, main_type = EXCLUDED.main_type,
public_symbols = EXCLUDED.public_symbols, summary = EXCLUDED.summary,
n_lines = EXCLUDED.n_lines, content_hash = EXCLUDED.content_hash,
updated_at = CURRENT_TIMESTAMP""",
(rel_path, module, main_type, symbols, summary, n_lines, digest),
)
cur.execute(
f"""INSERT INTO {schema}.indexed_files (file_path, content_hash)
VALUES (%s, %s)
ON CONFLICT (file_path) DO UPDATE SET
content_hash = EXCLUDED.content_hash, updated_at = CURRENT_TIMESTAMP""",
(rel_path, digest),
)
indexed += 1
print(f"[RAG] {rel_path}")
cur.execute(f"SELECT file_path FROM {schema}.indexed_files")
for (rel_path,) in cur.fetchall():
if rel_path not in seen:
cur.execute(f"DELETE FROM {schema}.code_chunks WHERE file_path = %s", (rel_path,))
cur.execute(f"DELETE FROM {schema}.file_index WHERE file_path = %s", (rel_path,))
cur.execute(f"DELETE FROM {schema}.indexed_files WHERE file_path = %s", (rel_path,))
deleted += 1
print(f"[RAG] removido {rel_path}")
conn.commit()
except Exception:
conn.rollback()
raise
finally:
conn.close()
print(f"[RAG] concluído: {indexed} atualizado(s), {skipped} sem mudança, {deleted} removido(s), {chunks_written} chunk(s).")
if __name__ == "__main__":
index()