Cria rag/ (schema, busca híbrida densa+lexical com RRF em search.py/ search_gart.sh, SETUP.md) — o projeto já tinha admin/update_rag.py para indexar, mas nenhuma forma de consultar o índice. Corrige admin/update_rag.py: um chunk denso em tokens (code/fcpxml/font_metrics.py) estourava o contexto do modelo de embedding e derrubava a transação inteira; agora só aquele chunk é pulado. Banco rag_gart provisionado no rag-hub-db compartilhado e primeira indexação completa rodada (304 arquivos, 1702 chunks). Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
217 lines
8.5 KiB
Python
Executable File
217 lines
8.5 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""Atualiza incrementalmente o índice RAG do G-ART.
|
|
|
|
As credenciais são fornecidas pelo ambiente; este arquivo nunca deve conter
|
|
senha. O indexador usa o banco ``rag_gart`` e o schema ``gart`` por padrão.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import hashlib
|
|
import os
|
|
import re
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
import psycopg2
|
|
import requests
|
|
|
|
ROOT = Path(__file__).resolve().parents[1]
|
|
DB_NAME = os.environ.get("RAG_DB_NAME", "rag_gart")
|
|
DB_SCHEMA = os.environ.get("RAG_DB_SCHEMA", "gart")
|
|
OLLAMA_URL = os.environ.get("OLLAMA_URL", "http://127.0.0.1:11434")
|
|
EMBED_MODEL = os.environ.get("RAG_EMBED_MODEL", "nomic-embed-text")
|
|
EMBED_DIM = int(os.environ.get("RAG_EMBED_DIM", "768"))
|
|
|
|
INCLUDE_EXTENSIONS = {
|
|
".command", ".md", ".py", ".sh", ".sql", ".swift", ".txt", ".yml", ".yaml",
|
|
}
|
|
EXCLUDE_DIRS = {
|
|
".git", ".venv", ".pytest_cache", ".ruff_cache", "__pycache__", "build",
|
|
"dist", "node_modules", "graphify-out", "bm", "models", "whisper",
|
|
}
|
|
EXCLUDE_FILES = {".env", "admin/genial-crm.env", "admin/genial-crm.local.env"}
|
|
CHUNK_LINES = 60
|
|
CHUNK_OVERLAP = 10
|
|
CHUNK_MAX_CHARS = 5000
|
|
|
|
|
|
def _sql_id(value: str) -> str:
|
|
return '"' + value.replace('"', '""') + '"'
|
|
|
|
|
|
def _connect():
|
|
password = os.environ.get("RAG_DB_PASSWORD")
|
|
if not password:
|
|
raise RuntimeError("RAG_DB_PASSWORD não foi definida")
|
|
return psycopg2.connect(
|
|
host=os.environ.get("RAG_DB_HOST", "127.0.0.1"),
|
|
port=os.environ.get("RAG_DB_PORT", "55435"),
|
|
dbname=DB_NAME,
|
|
user=os.environ.get("RAG_DB_USER", "gart_rag_indexer"),
|
|
password=password,
|
|
connect_timeout=5,
|
|
)
|
|
|
|
|
|
def _iter_files():
|
|
for path in ROOT.rglob("*"):
|
|
if not path.is_file() or path.suffix.lower() not in INCLUDE_EXTENSIONS:
|
|
continue
|
|
rel = path.relative_to(ROOT).as_posix()
|
|
parts = set(path.relative_to(ROOT).parts)
|
|
if parts & EXCLUDE_DIRS or rel in EXCLUDE_FILES or path.name in EXCLUDE_FILES:
|
|
continue
|
|
if any(part.startswith(".") for part in path.relative_to(ROOT).parts[:-1]):
|
|
continue
|
|
yield path, rel
|
|
|
|
|
|
def _chunks(text: str):
|
|
lines = text.splitlines()
|
|
if not lines:
|
|
return []
|
|
step = max(1, CHUNK_LINES - CHUNK_OVERLAP)
|
|
result = []
|
|
for start in range(0, len(lines), step):
|
|
window_start = start
|
|
buffer = []
|
|
size = 0
|
|
for offset, line in enumerate(lines[start:start + CHUNK_LINES]):
|
|
if buffer and size + len(line) + 1 > CHUNK_MAX_CHARS:
|
|
result.append((window_start + 1, window_start + len(buffer), "\n".join(buffer).strip()))
|
|
buffer = []
|
|
window_start = start + offset
|
|
size = 0
|
|
buffer.append(line)
|
|
size += len(line) + 1
|
|
if buffer:
|
|
result.append((window_start + 1, window_start + len(buffer), "\n".join(buffer).strip()))
|
|
if start + CHUNK_LINES >= len(lines):
|
|
break
|
|
return [(start, end, content) for start, end, content in result if content]
|
|
|
|
|
|
def _facts(text: str, rel_path: str):
|
|
lines = text.splitlines()
|
|
summary = next(
|
|
(line.strip().lstrip("#! ").strip() for line in lines[:30] if line.strip()),
|
|
None,
|
|
)
|
|
symbols = re.findall(
|
|
r"^\s*(?:class|def|async\s+def|func|struct|enum|protocol|actor|interface)\s+([A-Za-z_]\w*)",
|
|
text,
|
|
re.MULTILINE,
|
|
)
|
|
parts = Path(rel_path).parts
|
|
module = parts[0] if len(parts) > 1 else None
|
|
return module, Path(rel_path).stem, summary, sorted(set(symbols)), len(lines)
|
|
|
|
|
|
class ChunkTooLargeError(Exception):
|
|
"""Chunk excede o contexto do modelo de embedding (ver EXCLUDE_FILES/CHUNK_MAX_CHARS)."""
|
|
|
|
|
|
def _embed(text: str):
|
|
response = requests.post(
|
|
f"{OLLAMA_URL.rstrip('/')}/api/embeddings",
|
|
json={"model": EMBED_MODEL, "prompt": f"search_document: {text}"},
|
|
timeout=60,
|
|
)
|
|
if response.status_code == 500 and "context length" in response.text.lower():
|
|
raise ChunkTooLargeError(response.text)
|
|
response.raise_for_status()
|
|
vector = response.json()["embedding"]
|
|
if len(vector) != EMBED_DIM:
|
|
raise ValueError(f"embedding com {len(vector)} dimensões; esperado {EMBED_DIM}")
|
|
return vector
|
|
|
|
|
|
def _hash(text: str) -> str:
|
|
return hashlib.md5(text.encode("utf-8")).hexdigest()
|
|
|
|
|
|
def index():
|
|
schema = _sql_id(DB_SCHEMA)
|
|
conn = _connect()
|
|
conn.autocommit = False
|
|
indexed = skipped = deleted = chunks_written = 0
|
|
seen = set()
|
|
try:
|
|
with conn.cursor() as cur:
|
|
for path, rel_path in sorted(_iter_files(), key=lambda item: item[1]):
|
|
try:
|
|
text = path.read_text(encoding="utf-8", errors="ignore")
|
|
except OSError as exc:
|
|
print(f"[RAG] ignorado {rel_path}: {exc}", file=sys.stderr)
|
|
continue
|
|
seen.add(rel_path)
|
|
digest = _hash(text)
|
|
cur.execute(f"SELECT content_hash FROM {schema}.indexed_files WHERE file_path = %s", (rel_path,))
|
|
row = cur.fetchone()
|
|
if row and row[0] == digest:
|
|
skipped += 1
|
|
continue
|
|
|
|
module, main_type, summary, symbols, n_lines = _facts(text, rel_path)
|
|
cur.execute(f"DELETE FROM {schema}.code_chunks WHERE file_path = %s", (rel_path,))
|
|
for index_number, (start, end, content) in enumerate(_chunks(text)):
|
|
try:
|
|
vector = _embed(content)
|
|
except ChunkTooLargeError:
|
|
# Chunks densos em tokens (ex: tabelas de dados numéricas
|
|
# como font_metrics.py) podem passar de CHUNK_MAX_CHARS em
|
|
# caracteres mas estourar o contexto do modelo em tokens.
|
|
# Pular o chunk em vez de abortar a transação inteira.
|
|
print(f"[RAG] chunk grande demais, pulado: {rel_path}:{start}-{end}", file=sys.stderr)
|
|
continue
|
|
cur.execute(
|
|
f"""INSERT INTO {schema}.code_chunks
|
|
(file_path, content, chunk_index, embedding, content_hash,
|
|
file_mtime, start_line, end_line, symbols, module, kind)
|
|
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)""",
|
|
(rel_path, content, index_number, vector, digest,
|
|
path.stat().st_mtime, start, end, ", ".join(symbols), module, "window"),
|
|
)
|
|
chunks_written += 1
|
|
cur.execute(
|
|
f"""INSERT INTO {schema}.file_index
|
|
(file_path, module, main_type, public_symbols, summary, n_lines, content_hash)
|
|
VALUES (%s, %s, %s, %s, %s, %s, %s)
|
|
ON CONFLICT (file_path) DO UPDATE SET
|
|
module = EXCLUDED.module, main_type = EXCLUDED.main_type,
|
|
public_symbols = EXCLUDED.public_symbols, summary = EXCLUDED.summary,
|
|
n_lines = EXCLUDED.n_lines, content_hash = EXCLUDED.content_hash,
|
|
updated_at = CURRENT_TIMESTAMP""",
|
|
(rel_path, module, main_type, symbols, summary, n_lines, digest),
|
|
)
|
|
cur.execute(
|
|
f"""INSERT INTO {schema}.indexed_files (file_path, content_hash)
|
|
VALUES (%s, %s)
|
|
ON CONFLICT (file_path) DO UPDATE SET
|
|
content_hash = EXCLUDED.content_hash, updated_at = CURRENT_TIMESTAMP""",
|
|
(rel_path, digest),
|
|
)
|
|
indexed += 1
|
|
print(f"[RAG] {rel_path}")
|
|
|
|
cur.execute(f"SELECT file_path FROM {schema}.indexed_files")
|
|
for (rel_path,) in cur.fetchall():
|
|
if rel_path not in seen:
|
|
cur.execute(f"DELETE FROM {schema}.code_chunks WHERE file_path = %s", (rel_path,))
|
|
cur.execute(f"DELETE FROM {schema}.file_index WHERE file_path = %s", (rel_path,))
|
|
cur.execute(f"DELETE FROM {schema}.indexed_files WHERE file_path = %s", (rel_path,))
|
|
deleted += 1
|
|
print(f"[RAG] removido {rel_path}")
|
|
conn.commit()
|
|
except Exception:
|
|
conn.rollback()
|
|
raise
|
|
finally:
|
|
conn.close()
|
|
print(f"[RAG] concluído: {indexed} atualizado(s), {skipped} sem mudança, {deleted} removido(s), {chunks_written} chunk(s).")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
index()
|