chore: remove lixo do repositório (bm/ duplicado, rag/tigre-*, .prev do painel, pastas vazias)
Remove a cópia zipada/extraída redundante do premiere-pro-mcp em bm/, os scripts e schema do RAG de outro sistema (Tigre) que foram parar aqui por engano, os backups manuais .prev do cep-plugin já superados pelo git, um arquivo solto ":memory:.ses" e pastas vazias sem uso em code/engine (domain, skills, dominio/objetos_de_valor, scanner/modelos, scanner/contratos, integracoes/premiere/contratos). Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 5
parent
5f3c7f6a24
commit
29c85be5fe
@@ -1,20 +0,0 @@
|
||||
#!/bin/bash
|
||||
# Indexa o código do novo sistema Tigre (codeclass/Sources) no banco RAG
|
||||
# dedicado (rag_tigre, schema tigre). Usa o mesmo container compartilhado
|
||||
# rag-hub-db — um banco por sistema.
|
||||
set -e
|
||||
|
||||
RAG_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
PROJECT_ROOT="$(dirname "$RAG_DIR")"
|
||||
|
||||
# Garante o túnel SSH até o Postgres da VPS (idempotente).
|
||||
"$RAG_DIR/ensure_tunnel.sh" >&2
|
||||
|
||||
# usa o python do ag .venv se existir, senão o python3 global
|
||||
PY="${RAG_DIR}/.venv/bin/python3"
|
||||
if [ ! -x "$PY" ]; then
|
||||
PY="$(command -v python3)"
|
||||
fi
|
||||
|
||||
export RAG_ENV_FILE="$RAG_DIR/tigre.env"
|
||||
exec "$PY" "$RAG_DIR/index_code.py" "$@"
|
||||
@@ -1,10 +0,0 @@
|
||||
#!/bin/bash
|
||||
# Aplica uma migration no banco do sistema novo (rag_tigre, schema tigre).
|
||||
# Uso: rag/migrate_tigre.sh [arquivo.sql] (padrao: a 002)
|
||||
set -euo pipefail
|
||||
RAG_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
"$RAG_DIR/migrate.sh" rag_tigre tigre \
|
||||
"${1:-$RAG_DIR/migrations/002-tigre-search.sql}"
|
||||
echo
|
||||
echo "[rag] PROXIMO PASSO OBRIGATORIO: reindexar do zero."
|
||||
echo " rag/index_tigre.sh --full"
|
||||
@@ -1,96 +0,0 @@
|
||||
-- Migration 002 — busca hibrida e leitura cirurgica no schema `tigre`.
|
||||
--
|
||||
-- Motivo (medido em 02/09/2026, 346 chunks / 190 arquivos):
|
||||
-- 1. O indice ivfflat com lists=100 sobre 346 linhas da ~3,5 linhas por
|
||||
-- lista. Com o padrao ivfflat.probes=1 a busca varre ~3 de 346 linhas,
|
||||
-- e varias consultas voltavam VAZIAS. recall@5 do conjunto dourado
|
||||
-- (rag/bench.py): 33%. Trocamos por HNSW, que nao particiona.
|
||||
-- 2. A busca densa pura erra nomes exatos de tipo (`FCPXMLValidator` nao
|
||||
-- aparecia no top-5 do proprio arquivo). Adicionamos `symbols` +
|
||||
-- indices GIN trgm para o lado lexico da fusao.
|
||||
-- 3. Os chunks nao guardavam faixa de linhas, entao o agente lia o
|
||||
-- arquivo inteiro depois da busca. `start_line`/`end_line` permitem
|
||||
-- ler so a janela relevante.
|
||||
--
|
||||
-- Idempotente: seguro rodar mais de uma vez.
|
||||
--
|
||||
-- COMO RODAR (precisa ser rag_admin — o rag_tigre_indexer nao tem DDL):
|
||||
--
|
||||
-- rag/ensure_tunnel.sh
|
||||
-- psql -h 127.0.0.1 -p 55435 -U rag_admin -d rag_tigre \
|
||||
-- -f rag/migrations/002-tigre-search.sql
|
||||
--
|
||||
-- A senha do rag_admin esta em admin/VPS-ACCESS.md.
|
||||
--
|
||||
-- DEPOIS DE RODAR: reindexar do zero e obrigatorio.
|
||||
-- rag/reindex_tigre.sh --full
|
||||
-- Os embeddings passam a usar o prefixo `search_document: ` exigido pelo
|
||||
-- nomic-embed-text; vetores antigos e novos nao sao comparaveis entre si.
|
||||
|
||||
BEGIN;
|
||||
|
||||
-- ---------------------------------------------------------------------
|
||||
-- 1. Indice vetorial: ivfflat -> HNSW
|
||||
-- ---------------------------------------------------------------------
|
||||
DROP INDEX IF EXISTS tigre.code_chunks_embedding_idx;
|
||||
|
||||
CREATE INDEX IF NOT EXISTS code_chunks_embedding_hnsw_idx
|
||||
ON tigre.code_chunks USING hnsw (embedding vector_cosine_ops)
|
||||
WITH (m = 16, ef_construction = 64);
|
||||
|
||||
-- ---------------------------------------------------------------------
|
||||
-- 2. Metadados de chunk: faixa de linhas, simbolos, modulo, tipo
|
||||
-- ---------------------------------------------------------------------
|
||||
ALTER TABLE tigre.code_chunks ADD COLUMN IF NOT EXISTS start_line int;
|
||||
ALTER TABLE tigre.code_chunks ADD COLUMN IF NOT EXISTS end_line int;
|
||||
ALTER TABLE tigre.code_chunks ADD COLUMN IF NOT EXISTS symbols text;
|
||||
ALTER TABLE tigre.code_chunks ADD COLUMN IF NOT EXISTS module text;
|
||||
-- kind: 'decl' (declaracao Swift), 'header' (imports + doc do tipo),
|
||||
-- 'window' (fallback por janela de linhas, usado fora do Swift)
|
||||
ALTER TABLE tigre.code_chunks ADD COLUMN IF NOT EXISTS kind text;
|
||||
|
||||
-- ---------------------------------------------------------------------
|
||||
-- 3. Indices lexicos (pg_trgm) — o lado keyword da busca hibrida
|
||||
-- ---------------------------------------------------------------------
|
||||
CREATE INDEX IF NOT EXISTS code_chunks_file_path_trgm_idx
|
||||
ON tigre.code_chunks USING gin (file_path gin_trgm_ops);
|
||||
|
||||
CREATE INDEX IF NOT EXISTS code_chunks_symbols_trgm_idx
|
||||
ON tigre.code_chunks USING gin (symbols gin_trgm_ops);
|
||||
|
||||
-- Filtro de escopo por modulo (--module TigreAI).
|
||||
CREATE INDEX IF NOT EXISTS code_chunks_module_idx
|
||||
ON tigre.code_chunks (module);
|
||||
|
||||
-- ---------------------------------------------------------------------
|
||||
-- 4. Mapa de arquivos — 1 linha por arquivo, para responder "onde fica X"
|
||||
-- sem trazer nenhum corpo de codigo.
|
||||
-- ---------------------------------------------------------------------
|
||||
CREATE TABLE IF NOT EXISTS tigre.file_index (
|
||||
file_path text PRIMARY KEY,
|
||||
module text,
|
||||
main_type text,
|
||||
public_symbols text[],
|
||||
summary text,
|
||||
n_lines int,
|
||||
content_hash text,
|
||||
updated_at timestamp DEFAULT now()
|
||||
);
|
||||
|
||||
CREATE INDEX IF NOT EXISTS file_index_module_idx
|
||||
ON tigre.file_index (module);
|
||||
|
||||
CREATE INDEX IF NOT EXISTS file_index_path_trgm_idx
|
||||
ON tigre.file_index USING gin (file_path gin_trgm_ops);
|
||||
|
||||
-- ---------------------------------------------------------------------
|
||||
-- 5. Permissoes do usuario de indexacao/busca
|
||||
-- ---------------------------------------------------------------------
|
||||
GRANT SELECT, INSERT, UPDATE, DELETE ON tigre.file_index TO rag_tigre_indexer;
|
||||
GRANT USAGE, SELECT ON ALL SEQUENCES IN SCHEMA tigre TO rag_tigre_indexer;
|
||||
|
||||
COMMIT;
|
||||
|
||||
-- Conferencia rapida (rode a mao depois):
|
||||
-- \d tigre.code_chunks
|
||||
-- SELECT indexname FROM pg_indexes WHERE schemaname = 'tigre';
|
||||
@@ -1,81 +0,0 @@
|
||||
-- Schema RAG do projeto Tigre (novo sistema em classes, Swift em codeclass/).
|
||||
-- Idempotente: seguro rodar múltiplas vezes (CREATE ... IF NOT EXISTS).
|
||||
-- Segue o padrão do banco irmão rag_doza (schema doza): um banco por sistema
|
||||
-- dentro do container compartilhado rag-hub-db, schema próprio.
|
||||
--
|
||||
-- Este arquivo é o estado FINAL desejado do schema — inclui o que a
|
||||
-- migration `migrations/002-tigre-search.sql` aplicou num banco já
|
||||
-- existente. Num banco novo, rodar só este arquivo basta.
|
||||
|
||||
CREATE EXTENSION IF NOT EXISTS vector;
|
||||
CREATE EXTENSION IF NOT EXISTS pg_trgm;
|
||||
|
||||
CREATE SCHEMA IF NOT EXISTS tigre;
|
||||
|
||||
CREATE TABLE IF NOT EXISTS tigre.code_chunks (
|
||||
id bigserial PRIMARY KEY,
|
||||
file_path text NOT NULL,
|
||||
content text NOT NULL,
|
||||
chunk_index int,
|
||||
embedding vector(768),
|
||||
content_hash text,
|
||||
file_mtime double precision,
|
||||
-- Faixa de linhas do trecho no arquivo original. É o que permite ao
|
||||
-- agente ler só a janela relevante em vez do arquivo inteiro.
|
||||
start_line int,
|
||||
end_line int,
|
||||
-- Nomes declarados no trecho, separados por espaço — o lado lexical
|
||||
-- (pg_trgm) da busca híbrida casa contra isto.
|
||||
symbols text,
|
||||
-- Módulo SwiftPM derivado de codeclass/Sources/<Módulo>/…
|
||||
module text,
|
||||
-- 'decl' (declaração Swift) | 'header' (imports + doc do tipo)
|
||||
-- | 'window' (fallback por janela de linhas, usado fora do Swift)
|
||||
kind text,
|
||||
updated_at timestamp DEFAULT now()
|
||||
);
|
||||
|
||||
-- HNSW, não ivfflat: com poucas centenas de chunks o ivfflat particiona o
|
||||
-- espaço em listas quase vazias e a busca com probes=1 varre quase nada
|
||||
-- (media 002 registra o recall@5 de 33% que isso causava).
|
||||
CREATE INDEX IF NOT EXISTS code_chunks_embedding_hnsw_idx
|
||||
ON tigre.code_chunks USING hnsw (embedding vector_cosine_ops)
|
||||
WITH (m = 16, ef_construction = 64);
|
||||
|
||||
-- Acelera a reindexação incremental (busca por file_path).
|
||||
CREATE INDEX IF NOT EXISTS idx_code_chunks_file_path
|
||||
ON tigre.code_chunks (file_path);
|
||||
|
||||
-- Lado lexical da busca híbrida.
|
||||
CREATE INDEX IF NOT EXISTS code_chunks_file_path_trgm_idx
|
||||
ON tigre.code_chunks USING gin (file_path gin_trgm_ops);
|
||||
CREATE INDEX IF NOT EXISTS code_chunks_symbols_trgm_idx
|
||||
ON tigre.code_chunks USING gin (symbols gin_trgm_ops);
|
||||
CREATE INDEX IF NOT EXISTS code_chunks_module_idx
|
||||
ON tigre.code_chunks (module);
|
||||
|
||||
-- Hash de conteúdo por arquivo, usado pelo indexador para pular arquivos
|
||||
-- que não mudaram desde a última rodada (reindexação incremental).
|
||||
CREATE TABLE IF NOT EXISTS tigre.indexed_files (
|
||||
file_path text PRIMARY KEY,
|
||||
content_hash text NOT NULL,
|
||||
updated_at timestamp DEFAULT now()
|
||||
);
|
||||
|
||||
-- Mapa de arquivos: 1 linha por arquivo. Responde "onde fica X" e "o que
|
||||
-- tem no módulo Y" sem trazer nenhum corpo de código.
|
||||
CREATE TABLE IF NOT EXISTS tigre.file_index (
|
||||
file_path text PRIMARY KEY,
|
||||
module text,
|
||||
main_type text,
|
||||
public_symbols text[],
|
||||
summary text,
|
||||
n_lines int,
|
||||
content_hash text,
|
||||
updated_at timestamp DEFAULT now()
|
||||
);
|
||||
|
||||
CREATE INDEX IF NOT EXISTS file_index_module_idx
|
||||
ON tigre.file_index (module);
|
||||
CREATE INDEX IF NOT EXISTS file_index_path_trgm_idx
|
||||
ON tigre.file_index USING gin (file_path gin_trgm_ops);
|
||||
@@ -1,16 +0,0 @@
|
||||
#!/bin/bash
|
||||
# Busca semântica no banco RAG do sistema Tigre (codeclass/Sources).
|
||||
# Uso: rag/search_tigre.sh "consulta" [top_k]
|
||||
set -e
|
||||
|
||||
RAG_DIR="$(cd "$(dirname "$0")" && pwd)"
|
||||
|
||||
"$RAG_DIR/ensure_tunnel.sh" >&2
|
||||
|
||||
PY="${RAG_DIR}/.venv/bin/python3"
|
||||
if [ ! -x "$PY" ]; then
|
||||
PY="$(command -v python3)"
|
||||
fi
|
||||
|
||||
export RAG_ENV_FILE="$RAG_DIR/tigre.env"
|
||||
exec "$PY" "$RAG_DIR/search.py" "$@"
|
||||
@@ -1,10 +0,0 @@
|
||||
RAG_DB_HOST=127.0.0.1
|
||||
RAG_DB_PORT=55435
|
||||
RAG_DB_NAME=rag_tigre
|
||||
RAG_DB_USER=rag_tigre_indexer
|
||||
RAG_DB_PASSWORD=TghUpFOJGazA9EJyCHbuNEfs_OC5hH4d
|
||||
RAG_DB_SCHEMA=tigre
|
||||
RAG_TARGET_DIR=codeclass/Sources
|
||||
|
||||
RAG_ENV_FILE=tigre.env
|
||||
RAG_EMBED_PREFIX=1
|
||||
Reference in New Issue
Block a user