feat(rag): provisiona busca RAG do G-ART e corrige indexação que abortava em chunk grande
Cria rag/ (schema, busca híbrida densa+lexical com RRF em search.py/ search_gart.sh, SETUP.md) — o projeto já tinha admin/update_rag.py para indexar, mas nenhuma forma de consultar o índice. Corrige admin/update_rag.py: um chunk denso em tokens (code/fcpxml/font_metrics.py) estourava o contexto do modelo de embedding e derrubava a transação inteira; agora só aquele chunk é pulado. Banco rag_gart provisionado no rag-hub-db compartilhado e primeira indexação completa rodada (304 arquivos, 1702 chunks). Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
d13f643ebc
commit
e9a17c1b62
@@ -0,0 +1,12 @@
|
||||
# Copie para admin/gart-rag.env e preencha a senha. Este arquivo é apenas um
|
||||
# modelo; admin/gart-rag.env é ignorado pelo git.
|
||||
RAG_DB_HOST=127.0.0.1
|
||||
RAG_DB_PORT=55435
|
||||
RAG_DB_NAME=rag_gart
|
||||
RAG_DB_SCHEMA=gart
|
||||
RAG_DB_USER=gart_rag_indexer
|
||||
RAG_DB_PASSWORD=
|
||||
|
||||
# Ollama que fornece nomic-embed-text.
|
||||
OLLAMA_URL=http://127.0.0.1:11434
|
||||
RAG_EMBED_MODEL=nomic-embed-text
|
||||
Executable
+68
@@ -0,0 +1,68 @@
|
||||
#!/bin/zsh
|
||||
# Atualiza incrementalmente a RAG do G-ART usando o banco compartilhado.
|
||||
#
|
||||
# Credenciais: defina RAG_DB_PASSWORD no ambiente ou crie
|
||||
# admin/gart-rag.env (ignorado pelo git). O arquivo pode conter também
|
||||
# RAG_DB_USER, RAG_DB_PORT, OLLAMA_URL e RAG_EMBED_MODEL.
|
||||
set -euo pipefail
|
||||
|
||||
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||
ENV_FILE="$ROOT/admin/gart-rag.env"
|
||||
if [[ -f "$ENV_FILE" ]]; then
|
||||
set -a
|
||||
source "$ENV_FILE"
|
||||
set +a
|
||||
fi
|
||||
|
||||
PYTHON="${RAG_PYTHON:-}"
|
||||
if [[ -z "$PYTHON" ]]; then
|
||||
for candidate in "$ROOT/admin/.venv/bin/python3" "$ROOT/rag/.venv/bin/python3"; do
|
||||
if [[ -x "$candidate" ]]; then PYTHON="$candidate"; break; fi
|
||||
done
|
||||
fi
|
||||
PYTHON="${PYTHON:-$(command -v python3)}"
|
||||
|
||||
if ! "$PYTHON" -c 'import psycopg2, requests' >/dev/null 2>&1; then
|
||||
echo "ERRO: o Python da RAG precisa dos pacotes psycopg2 e requests." >&2
|
||||
echo "Instale-os no ambiente indicado por RAG_PYTHON e tente novamente." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if [[ -z "${RAG_DB_PASSWORD:-}" ]]; then
|
||||
echo "ERRO: defina RAG_DB_PASSWORD ou configure $ENV_FILE" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
HOST="${RAG_VPS_HOST:-179.197.228.240}"
|
||||
LOCAL_PORT="${RAG_DB_PORT:-55435}"
|
||||
REMOTE_PORT="${RAG_REMOTE_PORT:-55435}"
|
||||
TUNNEL_PID=""
|
||||
cleanup() {
|
||||
if [[ -n "$TUNNEL_PID" ]] && kill -0 "$TUNNEL_PID" 2>/dev/null; then
|
||||
kill "$TUNNEL_PID" 2>/dev/null || true
|
||||
wait "$TUNNEL_PID" 2>/dev/null || true
|
||||
fi
|
||||
}
|
||||
trap cleanup EXIT
|
||||
|
||||
if ! nc -z 127.0.0.1 "$LOCAL_PORT" 2>/dev/null; then
|
||||
echo "==> Abrindo túnel RAG (127.0.0.1:$LOCAL_PORT)..."
|
||||
ssh -N -o ExitOnForwardFailure=yes -o ServerAliveInterval=30 \
|
||||
-o ServerAliveCountMax=3 -L "127.0.0.1:$LOCAL_PORT:127.0.0.1:$REMOTE_PORT" \
|
||||
"${RAG_VPS_USER:-root}@$HOST" &
|
||||
TUNNEL_PID=$!
|
||||
for _ in {1..20}; do
|
||||
nc -z 127.0.0.1 "$LOCAL_PORT" 2>/dev/null && break
|
||||
kill -0 "$TUNNEL_PID" 2>/dev/null || break
|
||||
sleep 0.25
|
||||
done
|
||||
fi
|
||||
|
||||
if ! nc -z 127.0.0.1 "$LOCAL_PORT" 2>/dev/null; then
|
||||
echo "ERRO: não foi possível abrir o túnel RAG." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "==> Atualizando RAG do G-ART (incremental)..."
|
||||
cd "$ROOT"
|
||||
exec "$PYTHON" "$ROOT/admin/update_rag.py"
|
||||
Executable
+216
@@ -0,0 +1,216 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Atualiza incrementalmente o índice RAG do G-ART.
|
||||
|
||||
As credenciais são fornecidas pelo ambiente; este arquivo nunca deve conter
|
||||
senha. O indexador usa o banco ``rag_gart`` e o schema ``gart`` por padrão.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import psycopg2
|
||||
import requests
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[1]
|
||||
DB_NAME = os.environ.get("RAG_DB_NAME", "rag_gart")
|
||||
DB_SCHEMA = os.environ.get("RAG_DB_SCHEMA", "gart")
|
||||
OLLAMA_URL = os.environ.get("OLLAMA_URL", "http://127.0.0.1:11434")
|
||||
EMBED_MODEL = os.environ.get("RAG_EMBED_MODEL", "nomic-embed-text")
|
||||
EMBED_DIM = int(os.environ.get("RAG_EMBED_DIM", "768"))
|
||||
|
||||
INCLUDE_EXTENSIONS = {
|
||||
".command", ".md", ".py", ".sh", ".sql", ".swift", ".txt", ".yml", ".yaml",
|
||||
}
|
||||
EXCLUDE_DIRS = {
|
||||
".git", ".venv", ".pytest_cache", ".ruff_cache", "__pycache__", "build",
|
||||
"dist", "node_modules", "graphify-out", "bm", "models", "whisper",
|
||||
}
|
||||
EXCLUDE_FILES = {".env", "admin/genial-crm.env", "admin/genial-crm.local.env"}
|
||||
CHUNK_LINES = 60
|
||||
CHUNK_OVERLAP = 10
|
||||
CHUNK_MAX_CHARS = 5000
|
||||
|
||||
|
||||
def _sql_id(value: str) -> str:
|
||||
return '"' + value.replace('"', '""') + '"'
|
||||
|
||||
|
||||
def _connect():
|
||||
password = os.environ.get("RAG_DB_PASSWORD")
|
||||
if not password:
|
||||
raise RuntimeError("RAG_DB_PASSWORD não foi definida")
|
||||
return psycopg2.connect(
|
||||
host=os.environ.get("RAG_DB_HOST", "127.0.0.1"),
|
||||
port=os.environ.get("RAG_DB_PORT", "55435"),
|
||||
dbname=DB_NAME,
|
||||
user=os.environ.get("RAG_DB_USER", "gart_rag_indexer"),
|
||||
password=password,
|
||||
connect_timeout=5,
|
||||
)
|
||||
|
||||
|
||||
def _iter_files():
|
||||
for path in ROOT.rglob("*"):
|
||||
if not path.is_file() or path.suffix.lower() not in INCLUDE_EXTENSIONS:
|
||||
continue
|
||||
rel = path.relative_to(ROOT).as_posix()
|
||||
parts = set(path.relative_to(ROOT).parts)
|
||||
if parts & EXCLUDE_DIRS or rel in EXCLUDE_FILES or path.name in EXCLUDE_FILES:
|
||||
continue
|
||||
if any(part.startswith(".") for part in path.relative_to(ROOT).parts[:-1]):
|
||||
continue
|
||||
yield path, rel
|
||||
|
||||
|
||||
def _chunks(text: str):
|
||||
lines = text.splitlines()
|
||||
if not lines:
|
||||
return []
|
||||
step = max(1, CHUNK_LINES - CHUNK_OVERLAP)
|
||||
result = []
|
||||
for start in range(0, len(lines), step):
|
||||
window_start = start
|
||||
buffer = []
|
||||
size = 0
|
||||
for offset, line in enumerate(lines[start:start + CHUNK_LINES]):
|
||||
if buffer and size + len(line) + 1 > CHUNK_MAX_CHARS:
|
||||
result.append((window_start + 1, window_start + len(buffer), "\n".join(buffer).strip()))
|
||||
buffer = []
|
||||
window_start = start + offset
|
||||
size = 0
|
||||
buffer.append(line)
|
||||
size += len(line) + 1
|
||||
if buffer:
|
||||
result.append((window_start + 1, window_start + len(buffer), "\n".join(buffer).strip()))
|
||||
if start + CHUNK_LINES >= len(lines):
|
||||
break
|
||||
return [(start, end, content) for start, end, content in result if content]
|
||||
|
||||
|
||||
def _facts(text: str, rel_path: str):
|
||||
lines = text.splitlines()
|
||||
summary = next(
|
||||
(line.strip().lstrip("#! ").strip() for line in lines[:30] if line.strip()),
|
||||
None,
|
||||
)
|
||||
symbols = re.findall(
|
||||
r"^\s*(?:class|def|async\s+def|func|struct|enum|protocol|actor|interface)\s+([A-Za-z_]\w*)",
|
||||
text,
|
||||
re.MULTILINE,
|
||||
)
|
||||
parts = Path(rel_path).parts
|
||||
module = parts[0] if len(parts) > 1 else None
|
||||
return module, Path(rel_path).stem, summary, sorted(set(symbols)), len(lines)
|
||||
|
||||
|
||||
class ChunkTooLargeError(Exception):
|
||||
"""Chunk excede o contexto do modelo de embedding (ver EXCLUDE_FILES/CHUNK_MAX_CHARS)."""
|
||||
|
||||
|
||||
def _embed(text: str):
|
||||
response = requests.post(
|
||||
f"{OLLAMA_URL.rstrip('/')}/api/embeddings",
|
||||
json={"model": EMBED_MODEL, "prompt": f"search_document: {text}"},
|
||||
timeout=60,
|
||||
)
|
||||
if response.status_code == 500 and "context length" in response.text.lower():
|
||||
raise ChunkTooLargeError(response.text)
|
||||
response.raise_for_status()
|
||||
vector = response.json()["embedding"]
|
||||
if len(vector) != EMBED_DIM:
|
||||
raise ValueError(f"embedding com {len(vector)} dimensões; esperado {EMBED_DIM}")
|
||||
return vector
|
||||
|
||||
|
||||
def _hash(text: str) -> str:
|
||||
return hashlib.md5(text.encode("utf-8")).hexdigest()
|
||||
|
||||
|
||||
def index():
|
||||
schema = _sql_id(DB_SCHEMA)
|
||||
conn = _connect()
|
||||
conn.autocommit = False
|
||||
indexed = skipped = deleted = chunks_written = 0
|
||||
seen = set()
|
||||
try:
|
||||
with conn.cursor() as cur:
|
||||
for path, rel_path in sorted(_iter_files(), key=lambda item: item[1]):
|
||||
try:
|
||||
text = path.read_text(encoding="utf-8", errors="ignore")
|
||||
except OSError as exc:
|
||||
print(f"[RAG] ignorado {rel_path}: {exc}", file=sys.stderr)
|
||||
continue
|
||||
seen.add(rel_path)
|
||||
digest = _hash(text)
|
||||
cur.execute(f"SELECT content_hash FROM {schema}.indexed_files WHERE file_path = %s", (rel_path,))
|
||||
row = cur.fetchone()
|
||||
if row and row[0] == digest:
|
||||
skipped += 1
|
||||
continue
|
||||
|
||||
module, main_type, summary, symbols, n_lines = _facts(text, rel_path)
|
||||
cur.execute(f"DELETE FROM {schema}.code_chunks WHERE file_path = %s", (rel_path,))
|
||||
for index_number, (start, end, content) in enumerate(_chunks(text)):
|
||||
try:
|
||||
vector = _embed(content)
|
||||
except ChunkTooLargeError:
|
||||
# Chunks densos em tokens (ex: tabelas de dados numéricas
|
||||
# como font_metrics.py) podem passar de CHUNK_MAX_CHARS em
|
||||
# caracteres mas estourar o contexto do modelo em tokens.
|
||||
# Pular o chunk em vez de abortar a transação inteira.
|
||||
print(f"[RAG] chunk grande demais, pulado: {rel_path}:{start}-{end}", file=sys.stderr)
|
||||
continue
|
||||
cur.execute(
|
||||
f"""INSERT INTO {schema}.code_chunks
|
||||
(file_path, content, chunk_index, embedding, content_hash,
|
||||
file_mtime, start_line, end_line, symbols, module, kind)
|
||||
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)""",
|
||||
(rel_path, content, index_number, vector, digest,
|
||||
path.stat().st_mtime, start, end, ", ".join(symbols), module, "window"),
|
||||
)
|
||||
chunks_written += 1
|
||||
cur.execute(
|
||||
f"""INSERT INTO {schema}.file_index
|
||||
(file_path, module, main_type, public_symbols, summary, n_lines, content_hash)
|
||||
VALUES (%s, %s, %s, %s, %s, %s, %s)
|
||||
ON CONFLICT (file_path) DO UPDATE SET
|
||||
module = EXCLUDED.module, main_type = EXCLUDED.main_type,
|
||||
public_symbols = EXCLUDED.public_symbols, summary = EXCLUDED.summary,
|
||||
n_lines = EXCLUDED.n_lines, content_hash = EXCLUDED.content_hash,
|
||||
updated_at = CURRENT_TIMESTAMP""",
|
||||
(rel_path, module, main_type, symbols, summary, n_lines, digest),
|
||||
)
|
||||
cur.execute(
|
||||
f"""INSERT INTO {schema}.indexed_files (file_path, content_hash)
|
||||
VALUES (%s, %s)
|
||||
ON CONFLICT (file_path) DO UPDATE SET
|
||||
content_hash = EXCLUDED.content_hash, updated_at = CURRENT_TIMESTAMP""",
|
||||
(rel_path, digest),
|
||||
)
|
||||
indexed += 1
|
||||
print(f"[RAG] {rel_path}")
|
||||
|
||||
cur.execute(f"SELECT file_path FROM {schema}.indexed_files")
|
||||
for (rel_path,) in cur.fetchall():
|
||||
if rel_path not in seen:
|
||||
cur.execute(f"DELETE FROM {schema}.code_chunks WHERE file_path = %s", (rel_path,))
|
||||
cur.execute(f"DELETE FROM {schema}.file_index WHERE file_path = %s", (rel_path,))
|
||||
cur.execute(f"DELETE FROM {schema}.indexed_files WHERE file_path = %s", (rel_path,))
|
||||
deleted += 1
|
||||
print(f"[RAG] removido {rel_path}")
|
||||
conn.commit()
|
||||
except Exception:
|
||||
conn.rollback()
|
||||
raise
|
||||
finally:
|
||||
conn.close()
|
||||
print(f"[RAG] concluído: {indexed} atualizado(s), {skipped} sem mudança, {deleted} removido(s), {chunks_written} chunk(s).")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
index()
|
||||
@@ -0,0 +1,89 @@
|
||||
# RAG deste projeto (G-ART)
|
||||
|
||||
Banco de RAG próprio do G-ART — usado só para a IA indexar código/documentação
|
||||
e responder consultas gastando menos tokens, sem precisar reler o repositório
|
||||
inteiro a cada tarefa. **Não é o banco de dados do sistema**: é infraestrutura
|
||||
de apoio ao desenvolvimento, mantida à parte da aplicação.
|
||||
|
||||
Segue o mesmo padrão dos projetos irmãos (Doza, Tigre, Jhonny): **um único
|
||||
container Postgres + pgvector compartilhado** (`rag-hub-db`) na VPS da
|
||||
equipe, e **um banco por sistema** dentro dele.
|
||||
|
||||
```
|
||||
rag-hub-db (container único na VPS)
|
||||
├── rag_doza ← banco do Doza
|
||||
├── rag_tigre ← banco do Tigre
|
||||
├── jhonny-rag ← banco do Jhonny
|
||||
└── rag_gart ← banco deste projeto
|
||||
```
|
||||
|
||||
## Divisão de responsabilidades neste projeto
|
||||
|
||||
Diferente dos projetos irmãos, aqui a indexação **já existia antes desta
|
||||
pasta** e mora em `admin/`, não em `rag/`:
|
||||
|
||||
- **Indexação** — [`admin/update_rag.py`](../admin/update_rag.py), chamado
|
||||
por `admin/update_rag.command` (túnel SSH + execução) e por
|
||||
`admin/run.command` (roda junto com o app). Varre `INCLUDE_EXTENSIONS`
|
||||
(`.command .md .py .sh .sql .swift .txt .yml .yaml`) a partir da raiz do
|
||||
projeto, corta por janela de linhas (`CHUNK_LINES`), grava embeddings via
|
||||
Ollama e é incremental (hash por arquivo em `gart.indexed_files`).
|
||||
- **Schema** — [`schema.sql`](schema.sql) nesta pasta: é o que
|
||||
`admin/update_rag.py` espera encontrar (`gart.code_chunks`,
|
||||
`gart.file_index`, `gart.indexed_files`). Rodar uma vez para provisionar
|
||||
um banco novo.
|
||||
- **Busca** — [`search.py`](search.py) e o wrapper
|
||||
[`search_gart.sh`](search_gart.sh) nesta pasta: é o que os projetos irmãos
|
||||
chamam de `search_<projeto>.sh`. Não existia ainda para o G-ART.
|
||||
- **Credenciais** — reaproveitadas de `admin/gart-rag.env` (mesmo arquivo que
|
||||
`admin/update_rag.command` já usa), para não duplicar a senha em dois
|
||||
lugares. Ver `admin/gart-rag.env.example` para o formato.
|
||||
|
||||
## Como a busca funciona
|
||||
|
||||
Duas listas em paralelo, fundidas com RRF ponderado (parâmetros herdados dos
|
||||
projetos irmãos, calibrados lá via `rag/bench.py` sobre consultas douradas):
|
||||
|
||||
1. **densa** — embedding do trecho de código/texto;
|
||||
2. **lexical** — `pg_trgm` sobre os símbolos declarados (nomes de
|
||||
classe/função extraídos por regex em `admin/update_rag.py`), para
|
||||
consultas que citam o nome exato de algo;
|
||||
3. **resumo** (`file_index.summary_embedding`) — hoje não é preenchido por
|
||||
`admin/update_rag.py` (só grava `summary` em texto, sem embedding), então
|
||||
essa lista fica vazia até alguém adicionar isso ao indexador. A busca
|
||||
funciona normalmente sem ela.
|
||||
|
||||
Cada trecho guarda `start_line`/`end_line`, então o resultado aponta a janela
|
||||
exata (`code/fcpxml/writer/modifier.py:120-180`) em vez de mandar ler o
|
||||
arquivo inteiro.
|
||||
|
||||
### Modos de saída
|
||||
|
||||
| Comando | O que traz |
|
||||
|---|---|
|
||||
| `rag/search_gart.sh "consulta"` | caminho, faixa de linhas e uma linha de descrição (padrão) |
|
||||
| `… --snippet` | + 300 chars do trecho |
|
||||
| `… --full` | + o trecho inteiro |
|
||||
| `… --json` | saída estruturada |
|
||||
| `… --module X` / `--path Y` / `--ext .py` | restringe o escopo |
|
||||
| `… --map [termo]` | inventário de arquivos, sem nenhum código |
|
||||
|
||||
## Arquivos desta pasta
|
||||
|
||||
- `README.md` — este arquivo.
|
||||
- `SETUP.md` — passo a passo para provisionar o banco `rag_gart` na primeira
|
||||
vez.
|
||||
- `schema.sql` — schema do G-ART (extensões, tabelas, índices). Estado FINAL
|
||||
desejado: num banco novo basta rodá-lo.
|
||||
- `embed.py` — chamada ao Ollama compartilhada entre indexador e busca (só
|
||||
os prefixos `search_document:`/`search_query:` do nomic-embed-text).
|
||||
- `search.py` / `search_gart.sh` — busca híbrida e seu wrapper.
|
||||
- `ensure_tunnel.sh` — abre o túnel SSH até `rag-hub-db` se ainda não estiver
|
||||
aberto (idempotente).
|
||||
|
||||
## Onde ficam as credenciais reais
|
||||
|
||||
Nunca nesta pasta. Credenciais de indexação (usuário/senha do Postgres) ficam
|
||||
em `admin/gart-rag.env` (fora do git). Acesso SSH à VPS e senha do usuário
|
||||
admin do `rag-hub-db` ficam documentados no `VPS-ACCESS.md` de outro projeto
|
||||
da equipe que já usa a mesma VPS — peça a quem provisionou o banco.
|
||||
@@ -0,0 +1,71 @@
|
||||
# Provisionar o banco `rag_gart`
|
||||
|
||||
Passo a passo para criar o banco deste projeto no container compartilhado
|
||||
`rag-hub-db` (mesma VPS usada por Doza/Tigre/Jhonny). Só precisa ser feito
|
||||
uma vez (ou para reprovisionar do zero).
|
||||
|
||||
Precisa de acesso SSH à VPS (`root@179.197.228.240`, chave já autorizada) e
|
||||
das credenciais do usuário admin do `rag-hub-db` (`rag_admin` — senha em
|
||||
`/docker/rag-hub/.env` na própria VPS; não é duplicada em nenhum projeto).
|
||||
|
||||
## 1. Criar o banco e a role de indexação
|
||||
|
||||
Via túnel SSH ou `docker exec` na VPS, como `rag_admin`:
|
||||
|
||||
```sql
|
||||
CREATE DATABASE rag_gart OWNER rag_admin;
|
||||
\c rag_gart
|
||||
CREATE EXTENSION IF NOT EXISTS vector;
|
||||
CREATE EXTENSION IF NOT EXISTS pg_trgm;
|
||||
|
||||
-- Role de indexação, sem DDL (só o que admin/update_rag.py e rag/search.py
|
||||
-- precisam: SELECT/INSERT/UPDATE/DELETE no schema gart).
|
||||
CREATE ROLE gart_rag_indexer LOGIN PASSWORD '<senha forte gerada aqui>';
|
||||
```
|
||||
|
||||
## 2. Criar o schema
|
||||
|
||||
Rode [`schema.sql`](schema.sql) neste banco (idempotente, `CREATE ... IF NOT
|
||||
EXISTS` em tudo):
|
||||
|
||||
```bash
|
||||
psql "postgresql://rag_admin@127.0.0.1:55435/rag_gart" -f rag/schema.sql
|
||||
```
|
||||
|
||||
Depois conceda os privilégios ao usuário de indexação:
|
||||
|
||||
```sql
|
||||
GRANT USAGE ON SCHEMA gart TO gart_rag_indexer;
|
||||
GRANT SELECT, INSERT, UPDATE, DELETE ON ALL TABLES IN SCHEMA gart TO gart_rag_indexer;
|
||||
GRANT USAGE, SELECT ON ALL SEQUENCES IN SCHEMA gart TO gart_rag_indexer;
|
||||
ALTER DEFAULT PRIVILEGES IN SCHEMA gart
|
||||
GRANT SELECT, INSERT, UPDATE, DELETE ON TABLES TO gart_rag_indexer;
|
||||
```
|
||||
|
||||
## 3. Preencher as credenciais locais
|
||||
|
||||
```bash
|
||||
cp admin/gart-rag.env.example admin/gart-rag.env
|
||||
```
|
||||
|
||||
Editar `admin/gart-rag.env` e colocar a senha gerada no passo 1 em
|
||||
`RAG_DB_PASSWORD`. Esse arquivo é ignorado pelo git — nunca commitar.
|
||||
|
||||
## 4. Indexar pela primeira vez
|
||||
|
||||
```bash
|
||||
admin/update_rag.command
|
||||
```
|
||||
|
||||
Abre o túnel SSH (se não estiver aberto), roda `admin/update_rag.py` e grava
|
||||
os chunks + o mapa de arquivos em `rag_gart`.
|
||||
|
||||
## 5. Testar a busca
|
||||
|
||||
```bash
|
||||
rag/search_gart.sh "como funciona o export para DaVinci"
|
||||
rag/search_gart.sh --map fcpxml
|
||||
```
|
||||
|
||||
Se vier `[RAG vazio, buscando local]`, confira se o passo 4 rodou sem erro e
|
||||
se `RAG_DB_PASSWORD` está correta.
|
||||
@@ -0,0 +1,37 @@
|
||||
"""Embeddings compartilhados entre indexador e busca do RAG do G-ART.
|
||||
|
||||
admin/update_rag.py já indexa (chunking por janela de linhas, incremental,
|
||||
rodado por admin/run.command). Este módulo só isola a chamada ao Ollama para
|
||||
que rag/search.py use exatamente o mesmo modelo/prefixo na consulta.
|
||||
|
||||
Prefixos do embedding
|
||||
----------------------
|
||||
O nomic-embed-text espera `search_document: ` no que é indexado e
|
||||
`search_query: ` no que é consultado — sem isso a qualidade da busca cai.
|
||||
admin/update_rag.py já indexa com `search_document: ` (ver `_embed` lá).
|
||||
Trocar esse regime invalida os vetores antigos e exige reindexar tudo.
|
||||
"""
|
||||
|
||||
import os
|
||||
|
||||
import requests
|
||||
|
||||
OLLAMA_URL = os.environ.get("OLLAMA_URL", "http://127.0.0.1:11434")
|
||||
EMBED_MODEL = os.environ.get("RAG_EMBED_MODEL", "nomic-embed-text")
|
||||
EMBED_DIM = int(os.environ.get("RAG_EMBED_DIM", "768"))
|
||||
|
||||
DOC_PREFIX = "search_document: "
|
||||
QUERY_PREFIX = "search_query: "
|
||||
|
||||
|
||||
def _embed(text: str, prefix: str = DOC_PREFIX):
|
||||
response = requests.post(
|
||||
f"{OLLAMA_URL.rstrip('/')}/api/embeddings",
|
||||
json={"model": EMBED_MODEL, "prompt": f"{prefix}{text}"},
|
||||
timeout=60,
|
||||
)
|
||||
response.raise_for_status()
|
||||
vector = response.json()["embedding"]
|
||||
if len(vector) != EMBED_DIM:
|
||||
raise ValueError(f"embedding com {len(vector)} dimensões; esperado {EMBED_DIM}")
|
||||
return vector
|
||||
Executable
+28
@@ -0,0 +1,28 @@
|
||||
#!/bin/bash
|
||||
# Garante que o túnel SSH até o Postgres da VPS (rag-hub-db) está aberto em
|
||||
# 127.0.0.1:55435. Idempotente: se já estiver escutando, não faz nada.
|
||||
# Mesmo container compartilhado usado por rag_doza/rag_tigre/jhonny-rag.
|
||||
|
||||
HOST="root@179.197.228.240"
|
||||
LOCAL_PORT=55435
|
||||
REMOTE_PORT=55435
|
||||
|
||||
if nc -z 127.0.0.1 "$LOCAL_PORT" 2>/dev/null; then
|
||||
echo "[rag] tunel ja aberto em 127.0.0.1:$LOCAL_PORT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
echo "[rag] abrindo tunel SSH ate $HOST ($LOCAL_PORT -> $REMOTE_PORT)..."
|
||||
ssh -f -N -o ExitOnForwardFailure=yes -o ServerAliveInterval=30 -o ServerAliveCountMax=3 \
|
||||
-L "127.0.0.1:${LOCAL_PORT}:127.0.0.1:${REMOTE_PORT}" "$HOST"
|
||||
|
||||
for _ in $(seq 1 10); do
|
||||
if nc -z 127.0.0.1 "$LOCAL_PORT" 2>/dev/null; then
|
||||
echo "[rag] tunel ativo."
|
||||
exit 0
|
||||
fi
|
||||
sleep 0.5
|
||||
done
|
||||
|
||||
echo "[rag] AVISO: nao foi possivel confirmar o tunel." >&2
|
||||
exit 1
|
||||
@@ -0,0 +1,6 @@
|
||||
# Dependências só das ferramentas de RAG (dev, não da aplicação).
|
||||
# admin/update_rag.py usa psycopg2 + requests diretamente (sem dotenv, pois
|
||||
# admin/update_rag.command já faz `source` do .env antes de chamá-lo).
|
||||
psycopg2-binary>=2.9
|
||||
python-dotenv>=1.0
|
||||
requests
|
||||
@@ -0,0 +1,86 @@
|
||||
-- Schema RAG do projeto G-ART (fcp-mcp-server).
|
||||
-- Idempotente: seguro rodar múltiplas vezes (CREATE ... IF NOT EXISTS).
|
||||
-- Segue o padrão dos bancos irmãos (rag_doza, rag_tigre, jhonny-rag): um
|
||||
-- banco por sistema dentro do container compartilhado rag-hub-db, schema
|
||||
-- próprio. Banco: rag_gart. Schema: gart.
|
||||
--
|
||||
-- Colunas e tabelas espelham exatamente o que admin/update_rag.py grava
|
||||
-- (code_chunks, file_index, indexed_files) e o que rag/search.py lê.
|
||||
|
||||
CREATE EXTENSION IF NOT EXISTS vector;
|
||||
CREATE EXTENSION IF NOT EXISTS pg_trgm;
|
||||
|
||||
CREATE SCHEMA IF NOT EXISTS gart;
|
||||
|
||||
CREATE TABLE IF NOT EXISTS gart.code_chunks (
|
||||
id bigserial PRIMARY KEY,
|
||||
file_path text NOT NULL,
|
||||
content text NOT NULL,
|
||||
chunk_index int,
|
||||
embedding vector(768),
|
||||
content_hash text,
|
||||
file_mtime double precision,
|
||||
-- Faixa de linhas do trecho no arquivo original. É o que permite ao
|
||||
-- agente ler só a janela relevante em vez do arquivo inteiro.
|
||||
start_line int,
|
||||
end_line int,
|
||||
-- Nomes declarados no trecho (class/def/func...), separados por vírgula —
|
||||
-- o lado lexical (pg_trgm) da busca híbrida casa contra isto.
|
||||
symbols text,
|
||||
-- Módulo derivado do caminho relativo (primeiro segmento, ex: code/fcpxml -> code).
|
||||
module text,
|
||||
-- 'window': admin/update_rag.py corta por janela de linhas, não por
|
||||
-- declaração (o projeto é majoritariamente Python/Swift/Markdown/shell).
|
||||
kind text,
|
||||
updated_at timestamp DEFAULT now()
|
||||
);
|
||||
|
||||
-- HNSW, não ivfflat: com poucas centenas/milhares de chunks o ivfflat
|
||||
-- particiona o espaço em listas quase vazias e a busca com probes baixo
|
||||
-- varre quase nada (ver rag/README.md para os números de referência).
|
||||
CREATE INDEX IF NOT EXISTS code_chunks_embedding_hnsw_idx
|
||||
ON gart.code_chunks USING hnsw (embedding vector_cosine_ops)
|
||||
WITH (m = 16, ef_construction = 64);
|
||||
|
||||
-- Acelera a reindexação incremental (busca por file_path) e a limpeza de
|
||||
-- chunks de um arquivo antes de reinserir.
|
||||
CREATE INDEX IF NOT EXISTS idx_code_chunks_file_path
|
||||
ON gart.code_chunks (file_path);
|
||||
|
||||
-- Lado lexical da busca híbrida.
|
||||
CREATE INDEX IF NOT EXISTS code_chunks_file_path_trgm_idx
|
||||
ON gart.code_chunks USING gin (file_path gin_trgm_ops);
|
||||
CREATE INDEX IF NOT EXISTS code_chunks_symbols_trgm_idx
|
||||
ON gart.code_chunks USING gin (symbols gin_trgm_ops);
|
||||
CREATE INDEX IF NOT EXISTS code_chunks_module_idx
|
||||
ON gart.code_chunks (module);
|
||||
|
||||
-- Hash de conteúdo por arquivo, usado pelo indexador para pular arquivos
|
||||
-- que não mudaram desde a última rodada (reindexação incremental).
|
||||
CREATE TABLE IF NOT EXISTS gart.indexed_files (
|
||||
file_path text PRIMARY KEY,
|
||||
content_hash text NOT NULL,
|
||||
updated_at timestamp DEFAULT now()
|
||||
);
|
||||
|
||||
-- Mapa de arquivos: 1 linha por arquivo. Responde "onde fica X" e "o que
|
||||
-- tem no módulo Y" sem trazer nenhum corpo de código.
|
||||
CREATE TABLE IF NOT EXISTS gart.file_index (
|
||||
file_path text PRIMARY KEY,
|
||||
module text,
|
||||
main_type text,
|
||||
public_symbols text[],
|
||||
summary text,
|
||||
n_lines int,
|
||||
content_hash text,
|
||||
summary_embedding vector(768),
|
||||
updated_at timestamp DEFAULT now()
|
||||
);
|
||||
|
||||
CREATE INDEX IF NOT EXISTS file_index_module_idx
|
||||
ON gart.file_index (module);
|
||||
CREATE INDEX IF NOT EXISTS file_index_path_trgm_idx
|
||||
ON gart.file_index USING gin (file_path gin_trgm_ops);
|
||||
CREATE INDEX IF NOT EXISTS file_index_summary_hnsw_idx
|
||||
ON gart.file_index USING hnsw (summary_embedding vector_cosine_ops)
|
||||
WITH (m = 16, ef_construction = 64);
|
||||
+434
@@ -0,0 +1,434 @@
|
||||
"""Busca semântica RAG do projeto G-ART.
|
||||
|
||||
Consulta `gart.code_chunks` no Postgres (via túnel SSH) combinando dois
|
||||
sinais e devolvendo faixas de linha, para o agente ler só o trecho relevante
|
||||
em vez do arquivo inteiro.
|
||||
|
||||
Uso:
|
||||
rag/search_gart.sh "como funciona o export FCPXML"
|
||||
rag/search_gart.sh "FCPXMLModifier" --snippet
|
||||
rag/search_gart.sh "voice_actions" --module fcpxml --json
|
||||
rag/search_gart.sh --map fcpxml # inventário do módulo
|
||||
|
||||
Como módulo:
|
||||
from search import rag_search
|
||||
rag_search("consulta", top_k=5)
|
||||
|
||||
Por que busca híbrida
|
||||
---------------------
|
||||
A busca puramente densa erra nomes exatos: procurar `FCPXMLModifier` pode não
|
||||
trazer `modifier.py` no topo. Por isso rodamos duas listas em paralelo —
|
||||
densa (embedding) e lexical (pg_trgm sobre os símbolos declarados) — e
|
||||
fundimos com RRF, que soma 1/(k+posição) de cada lista e portanto não exige
|
||||
normalizar escalas diferentes. Adaptado do rag/search.py dos projetos irmãos
|
||||
(Doza, Tigre, Jhonny) — mesmos parâmetros, calibrados lá via rag/bench.py.
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
|
||||
import psycopg2
|
||||
from dotenv import load_dotenv
|
||||
|
||||
RAG_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
ROOT = os.path.dirname(RAG_DIR)
|
||||
# Por padrão reaproveita as mesmas credenciais do indexador
|
||||
# (admin/gart-rag.env), para não duplicar a senha em dois arquivos.
|
||||
_load_path = os.environ.get("RAG_ENV_FILE", os.path.join(ROOT, "admin", "gart-rag.env"))
|
||||
load_dotenv(_load_path)
|
||||
|
||||
if RAG_DIR not in sys.path:
|
||||
sys.path.insert(0, RAG_DIR)
|
||||
|
||||
from embed import QUERY_PREFIX, _embed # noqa: E402
|
||||
|
||||
# Constante do Reciprocal Rank Fusion e pesos por lista — valores herdados
|
||||
# dos projetos irmãos (calibrados lá via varredura sobre consultas douradas,
|
||||
# ver rag/bench.py no projeto Doza/Tigre). Ao mexer nesses números, monte um
|
||||
# conjunto de consultas de referência para o G-ART antes.
|
||||
RRF_K = 8
|
||||
W_DENSE = 1.0
|
||||
W_LEX = 1.0 # multiplicado pela similaridade bruta do casamento
|
||||
W_SUMMARY = 1.5 # o resumo é curto e preciso: um acerto ali vale mais
|
||||
|
||||
# Piso do casamento lexical: abaixo disso o lado lexical se cala em vez de
|
||||
# afogar a lista densa com ruído.
|
||||
LEX_MIN = 0.5
|
||||
|
||||
# Quantos chunks do mesmo arquivo podem ocupar o top-k, para um arquivo
|
||||
# grande não tomar todos os lugares.
|
||||
MAX_PER_FILE = 2
|
||||
|
||||
# Quanto código o modo --snippet mostra por resultado.
|
||||
SNIPPET_CHARS = 300
|
||||
|
||||
_EXTRA_COLS = ("start_line", "end_line", "symbols", "module", "kind")
|
||||
_cols_cache = {}
|
||||
|
||||
|
||||
def _available_cols(cur, schema):
|
||||
if schema not in _cols_cache:
|
||||
cur.execute(
|
||||
"""SELECT column_name FROM information_schema.columns
|
||||
WHERE table_schema = %s AND table_name = 'code_chunks'""",
|
||||
(schema,),
|
||||
)
|
||||
_cols_cache[schema] = {r[0] for r in cur.fetchall()}
|
||||
return _cols_cache[schema]
|
||||
|
||||
|
||||
def _select_cols(cur, schema):
|
||||
have = _available_cols(cur, schema)
|
||||
extra = ", ".join(f"c.{c}" if c in have else f"NULL AS {c}"
|
||||
for c in _EXTRA_COLS)
|
||||
return f"c.id, c.file_path, c.content, {extra}"
|
||||
|
||||
|
||||
def _db_connect():
|
||||
return psycopg2.connect(
|
||||
host=os.environ.get("RAG_DB_HOST", "127.0.0.1"),
|
||||
port=os.environ.get("RAG_DB_PORT", "55435"),
|
||||
dbname=os.environ.get("RAG_DB_NAME", "rag_gart"),
|
||||
user=os.environ.get("RAG_DB_USER", "gart_rag_indexer"),
|
||||
password=os.environ["RAG_DB_PASSWORD"],
|
||||
connect_timeout=5,
|
||||
)
|
||||
|
||||
|
||||
def _schema():
|
||||
return os.environ.get("RAG_DB_SCHEMA", "gart")
|
||||
|
||||
|
||||
def _qid(schema):
|
||||
"""Schema como identificador SQL seguro (aspas duplas)."""
|
||||
return '"' + schema.replace('"', '""') + '"'
|
||||
|
||||
|
||||
def _filters(module, path, ext, have=()):
|
||||
"""Cláusulas de escopo aplicadas antes do ranqueamento."""
|
||||
clauses, params = [], []
|
||||
if module and "module" in have:
|
||||
clauses.append("c.module = %s")
|
||||
params.append(module)
|
||||
if path:
|
||||
clauses.append("c.file_path ILIKE %s")
|
||||
params.append(f"%{path}%")
|
||||
if ext:
|
||||
clauses.append("c.file_path LIKE %s")
|
||||
params.append(f"%{ext}")
|
||||
return (" AND " + " AND ".join(clauses) if clauses else ""), params
|
||||
|
||||
|
||||
def _probe_terms(query):
|
||||
"""Termos que o lado lexical tenta casar contra os símbolos: a consulta
|
||||
inteira e a versão sem espaços (faz "voice timeline" casar com
|
||||
`VoiceTimeline`)."""
|
||||
terms = [query, query.replace(" ", "")]
|
||||
return list(dict.fromkeys(t for t in terms if t))
|
||||
|
||||
|
||||
def _row_to_dict(row):
|
||||
return {
|
||||
"id": row[0], "file_path": row[1], "content": row[2],
|
||||
"start_line": row[3], "end_line": row[4],
|
||||
"symbols": row[5], "module": row[6], "kind": row[7],
|
||||
}
|
||||
|
||||
|
||||
def _dense(cur, schema, q_emb, limit, where, params):
|
||||
cur.execute("SET LOCAL hnsw.ef_search = 64")
|
||||
cur.execute(
|
||||
f"""
|
||||
SELECT {_select_cols(cur, schema)}, 1 - (c.embedding <=> %s::vector) AS s
|
||||
FROM {_qid(schema)}.code_chunks c
|
||||
WHERE c.embedding IS NOT NULL {where}
|
||||
ORDER BY c.embedding <=> %s::vector
|
||||
LIMIT %s
|
||||
""",
|
||||
[q_emb] + params + [q_emb, limit],
|
||||
)
|
||||
return [(_row_to_dict(r), float(r[8])) for r in cur.fetchall()]
|
||||
|
||||
|
||||
def _lexical(cur, schema, terms, limit, where, params):
|
||||
have = _available_cols(cur, schema)
|
||||
stem = "regexp_replace(c.file_path, '^.*/|\\.[^.]*$', '', 'g')"
|
||||
target = f"coalesce(c.symbols, {stem})" if "symbols" in have else stem
|
||||
tiebreak = "c.start_line" if "start_line" in have else "c.id"
|
||||
score = f"(SELECT max(word_similarity(t, {target})) FROM unnest(%s::text[]) t)"
|
||||
cur.execute(
|
||||
f"""
|
||||
SELECT {_select_cols(cur, schema)}, {score} AS s
|
||||
FROM {_qid(schema)}.code_chunks c
|
||||
WHERE {score} >= %s {where}
|
||||
ORDER BY s DESC, {tiebreak} ASC
|
||||
LIMIT %s
|
||||
""",
|
||||
[terms] + [terms, LEX_MIN] + params + [limit],
|
||||
)
|
||||
return [(_row_to_dict(r), float(r[8])) for r in cur.fetchall()]
|
||||
|
||||
|
||||
def _summary_dense(cur, schema, q_emb, limit, where, params):
|
||||
"""Terceira lista: busca sobre o resumo do arquivo (file_index), não
|
||||
sobre o código — resgata arquivos pequenos e precisos que a lista por
|
||||
trecho, dominada por arquivos grandes, deixa passar."""
|
||||
cur.execute("SELECT to_regclass(%s)", (f"{_qid(schema)}.file_index",))
|
||||
if cur.fetchone()[0] is None:
|
||||
return []
|
||||
cur.execute("SET LOCAL hnsw.ef_search = 64")
|
||||
cur.execute(
|
||||
f"""
|
||||
SELECT file_path, 1 - (summary_embedding <=> %s::vector) AS s
|
||||
FROM {_qid(schema)}.file_index
|
||||
WHERE summary_embedding IS NOT NULL
|
||||
ORDER BY summary_embedding <=> %s::vector
|
||||
LIMIT %s
|
||||
""",
|
||||
(q_emb, q_emb, limit),
|
||||
)
|
||||
hits = cur.fetchall()
|
||||
if not hits:
|
||||
return []
|
||||
order = {fp: i for i, (fp, _) in enumerate(hits)}
|
||||
raw = {fp: s for fp, s in hits}
|
||||
|
||||
cur.execute(
|
||||
f"""
|
||||
SELECT DISTINCT ON (c.file_path) {_select_cols(cur, schema)}
|
||||
FROM {_qid(schema)}.code_chunks c
|
||||
WHERE c.file_path = ANY(%s) AND c.embedding IS NOT NULL {where}
|
||||
ORDER BY c.file_path, c.embedding <=> %s::vector
|
||||
""",
|
||||
[list(order)] + params + [q_emb],
|
||||
)
|
||||
rows = [_row_to_dict(r) for r in cur.fetchall()]
|
||||
rows.sort(key=lambda r: order[r["file_path"]])
|
||||
return [(r, raw[r["file_path"]]) for r in rows]
|
||||
|
||||
|
||||
def _fuse(ranked_lists):
|
||||
"""Reciprocal Rank Fusion ponderada sobre listas já ordenadas."""
|
||||
scores, best = {}, {}
|
||||
for results, weight_of in ranked_lists:
|
||||
for rank, (row, raw) in enumerate(results, 1):
|
||||
contrib = weight_of(raw) / (RRF_K + rank)
|
||||
scores[row["id"]] = scores.get(row["id"], 0.0) + contrib
|
||||
best.setdefault(row["id"], row)
|
||||
ordered = sorted(scores.items(), key=lambda kv: -kv[1])
|
||||
return [dict(best[i], score=s) for i, s in ordered]
|
||||
|
||||
|
||||
def _dedupe(rows, top_k, max_per_file=MAX_PER_FILE):
|
||||
"""Limita chunks por arquivo; o excedente vira uma nota de localização."""
|
||||
kept, counts, extras = [], {}, {}
|
||||
for row in rows:
|
||||
fp = row["file_path"]
|
||||
if counts.get(fp, 0) < max_per_file:
|
||||
counts[fp] = counts.get(fp, 0) + 1
|
||||
row["also_at"] = []
|
||||
kept.append(row)
|
||||
else:
|
||||
extras.setdefault(fp, []).append((row["start_line"], row["end_line"]))
|
||||
for row in kept:
|
||||
row["also_at"] = extras.get(row["file_path"], [])[:3]
|
||||
return kept[:top_k]
|
||||
|
||||
|
||||
def _attach_map(cur, schema, rows):
|
||||
"""Anexa tipo principal e resumo de `file_index`, quando existir."""
|
||||
cur.execute("SELECT to_regclass(%s)", (f"{_qid(schema)}.file_index",))
|
||||
if cur.fetchone()[0] is None or not rows:
|
||||
return rows
|
||||
paths = list({r["file_path"] for r in rows})
|
||||
cur.execute(
|
||||
f"SELECT file_path, main_type, summary FROM {_qid(schema)}.file_index "
|
||||
f"WHERE file_path = ANY(%s)",
|
||||
(paths,),
|
||||
)
|
||||
info = {p: (t, s) for p, t, s in cur.fetchall()}
|
||||
for row in rows:
|
||||
main_type, summary = info.get(row["file_path"], (None, None))
|
||||
row["main_type"] = main_type
|
||||
row["summary"] = summary
|
||||
return rows
|
||||
|
||||
|
||||
def rag_search(query, top_k=5, module=None, path=None, ext=None, dense_only=False):
|
||||
"""Busca híbrida. Devolve dicts com file_path, faixa de linhas e score."""
|
||||
schema = _schema()
|
||||
pool = max(top_k * 4, 20)
|
||||
|
||||
conn = _db_connect()
|
||||
try:
|
||||
with conn.cursor() as cur:
|
||||
where, params = _filters(module, path, ext,
|
||||
_available_cols(cur, schema))
|
||||
q_emb = _embed(query, prefix=QUERY_PREFIX)
|
||||
lists = [(_dense(cur, schema, q_emb, pool, where, params),
|
||||
lambda raw: W_DENSE)]
|
||||
if not dense_only:
|
||||
lists.append((_lexical(cur, schema, _probe_terms(query),
|
||||
pool, where, params),
|
||||
lambda raw: W_LEX * (0.5 + raw)))
|
||||
lists.append((_summary_dense(cur, schema, q_emb, pool,
|
||||
where, params),
|
||||
lambda raw: W_SUMMARY))
|
||||
rows = _dedupe(_fuse(lists), top_k)
|
||||
rows = _attach_map(cur, schema, rows)
|
||||
conn.commit()
|
||||
finally:
|
||||
conn.close()
|
||||
return rows
|
||||
|
||||
|
||||
def map_files(term=None, module=None, limit=60):
|
||||
"""Inventário de arquivos — responde "onde fica X" sem corpo de código."""
|
||||
schema = _schema()
|
||||
conn = _db_connect()
|
||||
try:
|
||||
with conn.cursor() as cur:
|
||||
cur.execute("SELECT to_regclass(%s)", (f"{_qid(schema)}.file_index",))
|
||||
if cur.fetchone()[0] is None:
|
||||
return []
|
||||
clauses, params = [], []
|
||||
if module:
|
||||
clauses.append("module = %s")
|
||||
params.append(module)
|
||||
if term:
|
||||
clauses.append("(file_path ILIKE %s OR main_type ILIKE %s "
|
||||
"OR summary ILIKE %s)")
|
||||
params += [f"%{term}%"] * 3
|
||||
where = "WHERE " + " AND ".join(clauses) if clauses else ""
|
||||
cur.execute(
|
||||
f"""SELECT file_path, module, main_type, summary, n_lines
|
||||
FROM {_qid(schema)}.file_index {where}
|
||||
ORDER BY module, file_path LIMIT %s""",
|
||||
params + [limit],
|
||||
)
|
||||
return [
|
||||
{"file_path": r[0], "module": r[1], "main_type": r[2],
|
||||
"summary": r[3], "n_lines": r[4]}
|
||||
for r in cur.fetchall()
|
||||
]
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Formatação
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
_NOISE = {'"""', "'''", "# ---", "---", "/*", "*/", "{", "}", "*"}
|
||||
|
||||
|
||||
def _first_doc_line(content):
|
||||
"""Primeira linha que serve de descrição do trecho, usada quando o
|
||||
arquivo não tem resumo em `file_index`."""
|
||||
for line in content.splitlines():
|
||||
s = line.strip()
|
||||
if s.startswith(("///", "//", "#", "*")):
|
||||
s = s.lstrip("/#* ").strip()
|
||||
if s and s not in _NOISE:
|
||||
return s
|
||||
for line in content.splitlines():
|
||||
s = line.strip().strip("\"'").strip()
|
||||
if s and s not in _NOISE and not s.startswith("import"):
|
||||
return s
|
||||
return ""
|
||||
|
||||
|
||||
def format_results(results, mode="map"):
|
||||
"""Renderiza o resultado no modo pedido. O padrão é `map`: caminho +
|
||||
faixa de linhas + uma linha de descrição, sem corpo de código."""
|
||||
if not results:
|
||||
return "[RAG vazio, buscando local]\n"
|
||||
|
||||
if mode == "json":
|
||||
return json.dumps(results, ensure_ascii=False, indent=2) + "\n"
|
||||
|
||||
out = []
|
||||
for i, r in enumerate(results, 1):
|
||||
loc = r["file_path"]
|
||||
if r.get("start_line") and r.get("end_line"):
|
||||
loc += f":{r['start_line']}-{r['end_line']}"
|
||||
out.append(f"{i} {r['score']:.2f} {loc}")
|
||||
|
||||
desc = r.get("summary") or _first_doc_line(r["content"])
|
||||
stem = os.path.splitext(os.path.basename(r["file_path"]))[0]
|
||||
label = r.get("main_type") or ""
|
||||
if label == stem:
|
||||
label = ""
|
||||
if label and desc:
|
||||
out.append(f" {label} · {desc[:78]}")
|
||||
elif desc:
|
||||
out.append(f" {desc[:88]}")
|
||||
|
||||
spans = [f"{a}-{b}" for a, b in r.get("also_at", []) if a and b]
|
||||
if spans:
|
||||
out.append(f" (+ tambem em {', '.join(spans)})")
|
||||
|
||||
if mode == "snippet":
|
||||
body, size = [], 0
|
||||
for ln in r["content"].splitlines():
|
||||
if body and size + len(ln) > SNIPPET_CHARS:
|
||||
body.append("…")
|
||||
break
|
||||
body.append(ln)
|
||||
size += len(ln) + 1
|
||||
out.append("".join(f" | {ln}\n" for ln in body))
|
||||
elif mode == "full":
|
||||
out.append("".join(f" | {ln}\n" for ln in r["content"].splitlines()))
|
||||
return "\n".join(out) + "\n"
|
||||
|
||||
|
||||
def format_map(rows):
|
||||
if not rows:
|
||||
return "[RAG vazio, buscando local]\n"
|
||||
out = []
|
||||
current = None
|
||||
for r in rows:
|
||||
if r["module"] != current:
|
||||
current = r["module"]
|
||||
out.append(f"\n{current or '(sem modulo)'}")
|
||||
name = os.path.basename(r["file_path"])
|
||||
desc = (r["summary"] or "")[:78]
|
||||
out.append(f" {name:<38} {r['n_lines']:>5}L {desc}")
|
||||
return "\n".join(out) + "\n"
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(
|
||||
description="Busca RAG hibrida (densa + lexical, fundidas com RRF) do G-ART")
|
||||
ap.add_argument("query", nargs="?", help="consulta")
|
||||
ap.add_argument("top_k", nargs="?", type=int, default=5)
|
||||
ap.add_argument("--snippet", action="store_true", help="mostra 300 chars do trecho")
|
||||
ap.add_argument("--full", action="store_true", help="mostra o trecho inteiro")
|
||||
ap.add_argument("--json", action="store_true", help="saida estruturada")
|
||||
ap.add_argument("--module", help="restringe a um modulo (ex: fcpxml)")
|
||||
ap.add_argument("--path", help="restringe a caminhos contendo este texto")
|
||||
ap.add_argument("--ext", help="restringe a uma extensao (ex: .py)")
|
||||
ap.add_argument("--map", dest="map_term", nargs="?", const="",
|
||||
help="inventario de arquivos em vez de busca por trecho")
|
||||
ap.add_argument("--dense-only", action="store_true",
|
||||
help="desliga o lado lexical (para comparacao)")
|
||||
args = ap.parse_args()
|
||||
|
||||
if args.map_term is not None:
|
||||
print(format_map(map_files(term=args.map_term or None, module=args.module)))
|
||||
return
|
||||
|
||||
if not args.query:
|
||||
ap.error("informe a consulta, ou use --map")
|
||||
|
||||
mode = ("json" if args.json else "full" if args.full
|
||||
else "snippet" if args.snippet else "map")
|
||||
results = rag_search(args.query, args.top_k, module=args.module,
|
||||
path=args.path, ext=args.ext, dense_only=args.dense_only)
|
||||
print(format_results(results, mode=mode))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Executable
+19
@@ -0,0 +1,19 @@
|
||||
#!/bin/zsh
|
||||
# Busca RAG do G-ART. Abre o túnel SSH se preciso e chama rag/search.py com
|
||||
# as credenciais de admin/gart-rag.env (mesmas do indexador incremental).
|
||||
set -euo pipefail
|
||||
|
||||
ROOT="$(cd "$(dirname "$0")/.." && pwd -P)"
|
||||
RAG_DIR="$ROOT/rag"
|
||||
|
||||
"$RAG_DIR/ensure_tunnel.sh"
|
||||
|
||||
PYTHON="${RAG_PYTHON:-}"
|
||||
if [[ -z "$PYTHON" ]]; then
|
||||
for candidate in "$ROOT/admin/.venv/bin/python3" "$ROOT/rag/.venv/bin/python3"; do
|
||||
if [[ -x "$candidate" ]]; then PYTHON="$candidate"; break; fi
|
||||
done
|
||||
fi
|
||||
PYTHON="${PYTHON:-$(command -v python3)}"
|
||||
|
||||
exec "$PYTHON" "$RAG_DIR/search.py" "$@"
|
||||
Reference in New Issue
Block a user