feat(rag): provisiona busca RAG do G-ART e corrige indexação que abortava em chunk grande
Cria rag/ (schema, busca híbrida densa+lexical com RRF em search.py/ search_gart.sh, SETUP.md) — o projeto já tinha admin/update_rag.py para indexar, mas nenhuma forma de consultar o índice. Corrige admin/update_rag.py: um chunk denso em tokens (code/fcpxml/font_metrics.py) estourava o contexto do modelo de embedding e derrubava a transação inteira; agora só aquele chunk é pulado. Banco rag_gart provisionado no rag-hub-db compartilhado e primeira indexação completa rodada (304 arquivos, 1702 chunks). Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
d13f643ebc
commit
e9a17c1b62
Executable
+216
@@ -0,0 +1,216 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Atualiza incrementalmente o índice RAG do G-ART.
|
||||
|
||||
As credenciais são fornecidas pelo ambiente; este arquivo nunca deve conter
|
||||
senha. O indexador usa o banco ``rag_gart`` e o schema ``gart`` por padrão.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import psycopg2
|
||||
import requests
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
DB_NAME = os.environ.get("RAG_DB_NAME", "rag_gart")
|
||||
DB_SCHEMA = os.environ.get("RAG_DB_SCHEMA", "gart")
|
||||
OLLAMA_URL = os.environ.get("OLLAMA_URL", "http://127.0.0.1:11434")
|
||||
EMBED_MODEL = os.environ.get("RAG_EMBED_MODEL", "nomic-embed-text")
|
||||
EMBED_DIM = int(os.environ.get("RAG_EMBED_DIM", "768"))
|
||||
|
||||
INCLUDE_EXTENSIONS = {
|
||||
".command", ".md", ".py", ".sh", ".sql", ".swift", ".txt", ".yml", ".yaml",
|
||||
}
|
||||
EXCLUDE_DIRS = {
|
||||
".git", ".venv", ".pytest_cache", ".ruff_cache", "__pycache__", "build",
|
||||
"dist", "node_modules", "graphify-out", "bm", "models", "whisper",
|
||||
}
|
||||
EXCLUDE_FILES = {".env", "admin/genial-crm.env", "admin/genial-crm.local.env"}
|
||||
CHUNK_LINES = 60
|
||||
CHUNK_OVERLAP = 10
|
||||
CHUNK_MAX_CHARS = 5000
|
||||
|
||||
|
||||
def _sql_id(value: str) -> str:
|
||||
return '"' + value.replace('"', '""') + '"'
|
||||
|
||||
|
||||
def _connect():
|
||||
password = os.environ.get("RAG_DB_PASSWORD")
|
||||
if not password:
|
||||
raise RuntimeError("RAG_DB_PASSWORD não foi definida")
|
||||
return psycopg2.connect(
|
||||
host=os.environ.get("RAG_DB_HOST", "127.0.0.1"),
|
||||
port=os.environ.get("RAG_DB_PORT", "55435"),
|
||||
dbname=DB_NAME,
|
||||
user=os.environ.get("RAG_DB_USER", "gart_rag_indexer"),
|
||||
password=password,
|
||||
connect_timeout=5,
|
||||
)
|
||||
|
||||
|
||||
def _iter_files():
|
||||
for path in ROOT.rglob("*"):
|
||||
if not path.is_file() or path.suffix.lower() not in INCLUDE_EXTENSIONS:
|
||||
continue
|
||||
rel = path.relative_to(ROOT).as_posix()
|
||||
parts = set(path.relative_to(ROOT).parts)
|
||||
if parts & EXCLUDE_DIRS or rel in EXCLUDE_FILES or path.name in EXCLUDE_FILES:
|
||||
continue
|
||||
if any(part.startswith(".") for part in path.relative_to(ROOT).parts[:-1]):
|
||||
continue
|
||||
yield path, rel
|
||||
|
||||
|
||||
def _chunks(text: str):
|
||||
lines = text.splitlines()
|
||||
if not lines:
|
||||
return []
|
||||
step = max(1, CHUNK_LINES - CHUNK_OVERLAP)
|
||||
result = []
|
||||
for start in range(0, len(lines), step):
|
||||
window_start = start
|
||||
buffer = []
|
||||
size = 0
|
||||
for offset, line in enumerate(lines[start:start + CHUNK_LINES]):
|
||||
if buffer and size + len(line) + 1 > CHUNK_MAX_CHARS:
|
||||
result.append((window_start + 1, window_start + len(buffer), "\n".join(buffer).strip()))
|
||||
buffer = []
|
||||
window_start = start + offset
|
||||
size = 0
|
||||
buffer.append(line)
|
||||
size += len(line) + 1
|
||||
if buffer:
|
||||
result.append((window_start + 1, window_start + len(buffer), "\n".join(buffer).strip()))
|
||||
if start + CHUNK_LINES >= len(lines):
|
||||
break
|
||||
return [(start, end, content) for start, end, content in result if content]
|
||||
|
||||
|
||||
def _facts(text: str, rel_path: str):
|
||||
lines = text.splitlines()
|
||||
summary = next(
|
||||
(line.strip().lstrip("#! ").strip() for line in lines[:30] if line.strip()),
|
||||
None,
|
||||
)
|
||||
symbols = re.findall(
|
||||
r"^\s*(?:class|def|async\s+def|func|struct|enum|protocol|actor|interface)\s+([A-Za-z_]\w*)",
|
||||
text,
|
||||
re.MULTILINE,
|
||||
)
|
||||
parts = Path(rel_path).parts
|
||||
module = parts[0] if len(parts) > 1 else None
|
||||
return module, Path(rel_path).stem, summary, sorted(set(symbols)), len(lines)
|
||||
|
||||
|
||||
class ChunkTooLargeError(Exception):
|
||||
"""Chunk excede o contexto do modelo de embedding (ver EXCLUDE_FILES/CHUNK_MAX_CHARS)."""
|
||||
|
||||
|
||||
def _embed(text: str):
|
||||
response = requests.post(
|
||||
f"{OLLAMA_URL.rstrip('/')}/api/embeddings",
|
||||
json={"model": EMBED_MODEL, "prompt": f"search_document: {text}"},
|
||||
timeout=60,
|
||||
)
|
||||
if response.status_code == 500 and "context length" in response.text.lower():
|
||||
raise ChunkTooLargeError(response.text)
|
||||
response.raise_for_status()
|
||||
vector = response.json()["embedding"]
|
||||
if len(vector) != EMBED_DIM:
|
||||
raise ValueError(f"embedding com {len(vector)} dimensões; esperado {EMBED_DIM}")
|
||||
return vector
|
||||
|
||||
|
||||
def _hash(text: str) -> str:
|
||||
return hashlib.md5(text.encode("utf-8")).hexdigest()
|
||||
|
||||
|
||||
def index():
|
||||
schema = _sql_id(DB_SCHEMA)
|
||||
conn = _connect()
|
||||
conn.autocommit = False
|
||||
indexed = skipped = deleted = chunks_written = 0
|
||||
seen = set()
|
||||
try:
|
||||
with conn.cursor() as cur:
|
||||
for path, rel_path in sorted(_iter_files(), key=lambda item: item[1]):
|
||||
try:
|
||||
text = path.read_text(encoding="utf-8", errors="ignore")
|
||||
except OSError as exc:
|
||||
print(f"[RAG] ignorado {rel_path}: {exc}", file=sys.stderr)
|
||||
continue
|
||||
seen.add(rel_path)
|
||||
digest = _hash(text)
|
||||
cur.execute(f"SELECT content_hash FROM {schema}.indexed_files WHERE file_path = %s", (rel_path,))
|
||||
row = cur.fetchone()
|
||||
if row and row[0] == digest:
|
||||
skipped += 1
|
||||
continue
|
||||
|
||||
module, main_type, summary, symbols, n_lines = _facts(text, rel_path)
|
||||
cur.execute(f"DELETE FROM {schema}.code_chunks WHERE file_path = %s", (rel_path,))
|
||||
for index_number, (start, end, content) in enumerate(_chunks(text)):
|
||||
try:
|
||||
vector = _embed(content)
|
||||
except ChunkTooLargeError:
|
||||
# Chunks densos em tokens (ex: tabelas de dados numéricas
|
||||
# como font_metrics.py) podem passar de CHUNK_MAX_CHARS em
|
||||
# caracteres mas estourar o contexto do modelo em tokens.
|
||||
# Pular o chunk em vez de abortar a transação inteira.
|
||||
print(f"[RAG] chunk grande demais, pulado: {rel_path}:{start}-{end}", file=sys.stderr)
|
||||
continue
|
||||
cur.execute(
|
||||
f"""INSERT INTO {schema}.code_chunks
|
||||
(file_path, content, chunk_index, embedding, content_hash,
|
||||
file_mtime, start_line, end_line, symbols, module, kind)
|
||||
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)""",
|
||||
(rel_path, content, index_number, vector, digest,
|
||||
path.stat().st_mtime, start, end, ", ".join(symbols), module, "window"),
|
||||
)
|
||||
chunks_written += 1
|
||||
cur.execute(
|
||||
f"""INSERT INTO {schema}.file_index
|
||||
(file_path, module, main_type, public_symbols, summary, n_lines, content_hash)
|
||||
VALUES (%s, %s, %s, %s, %s, %s, %s)
|
||||
ON CONFLICT (file_path) DO UPDATE SET
|
||||
module = EXCLUDED.module, main_type = EXCLUDED.main_type,
|
||||
public_symbols = EXCLUDED.public_symbols, summary = EXCLUDED.summary,
|
||||
n_lines = EXCLUDED.n_lines, content_hash = EXCLUDED.content_hash,
|
||||
updated_at = CURRENT_TIMESTAMP""",
|
||||
(rel_path, module, main_type, symbols, summary, n_lines, digest),
|
||||
)
|
||||
cur.execute(
|
||||
f"""INSERT INTO {schema}.indexed_files (file_path, content_hash)
|
||||
VALUES (%s, %s)
|
||||
ON CONFLICT (file_path) DO UPDATE SET
|
||||
content_hash = EXCLUDED.content_hash, updated_at = CURRENT_TIMESTAMP""",
|
||||
(rel_path, digest),
|
||||
)
|
||||
indexed += 1
|
||||
print(f"[RAG] {rel_path}")
|
||||
|
||||
cur.execute(f"SELECT file_path FROM {schema}.indexed_files")
|
||||
for (rel_path,) in cur.fetchall():
|
||||
if rel_path not in seen:
|
||||
cur.execute(f"DELETE FROM {schema}.code_chunks WHERE file_path = %s", (rel_path,))
|
||||
cur.execute(f"DELETE FROM {schema}.file_index WHERE file_path = %s", (rel_path,))
|
||||
cur.execute(f"DELETE FROM {schema}.indexed_files WHERE file_path = %s", (rel_path,))
|
||||
deleted += 1
|
||||
print(f"[RAG] removido {rel_path}")
|
||||
conn.commit()
|
||||
except Exception:
|
||||
conn.rollback()
|
||||
raise
|
||||
finally:
|
||||
conn.close()
|
||||
print(f"[RAG] concluído: {indexed} atualizado(s), {skipped} sem mudança, {deleted} removido(s), {chunks_written} chunk(s).")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
index()
|
||||
Reference in New Issue
Block a user