feat: initial commit - Jhonny Editor

- Adicionado estrutura completa do projeto
- Configurado MCP server para Premiere Pro
- Adicionado documentação e skills
- Configurado Gitignore para o projeto
This commit is contained in:
João Henrique
2026-09-08 09:59:31 -04:00
commit b541f502ba
1507 changed files with 387650 additions and 0 deletions
+75
View File
@@ -0,0 +1,75 @@
-- Migration 002 (doza) — busca híbrida e leitura cirúrgica no schema `doza`.
--
-- É a mesma mudança já aplicada ao schema `tigre` (migrations 002 e 003),
-- reunida num arquivo só porque aqui as duas vão juntas desde o início.
--
-- Motivo, medido no índice legado (694 trechos / 93 arquivos):
-- O índice ivfflat com lists=100 sobre 694 linhas dá ~7 linhas por lista.
-- Com o padrão ivfflat.probes=1 a busca varre quase nada, e o resultado é
-- ruído: procurar `SpineSegment` devolvia requirements.txt e CLA.md,
-- enquanto a busca exata põe doza_assist/fcpxml/parser.py em 1º.
-- recall@5 do conjunto dourado (rag/bench_doza.sh): 0%. Zero de 18.
--
-- Idempotente: seguro rodar mais de uma vez.
--
-- COMO RODAR (precisa ser rag_admin — o rag_doza_indexer não tem DDL):
-- rag/migrate_doza.sh
--
-- DEPOIS DE RODAR: reindexar do zero é obrigatório.
-- rag/index_doza.sh --full
-- Os embeddings passam a usar os prefixos do nomic-embed-text; vetores
-- antigos e novos não são comparáveis entre si.
BEGIN;
-- 1. Índice vetorial: ivfflat -> HNSW ---------------------------------
DROP INDEX IF EXISTS doza.code_chunks_embedding_idx;
CREATE INDEX IF NOT EXISTS code_chunks_embedding_hnsw_idx
ON doza.code_chunks USING hnsw (embedding vector_cosine_ops)
WITH (m = 16, ef_construction = 64);
-- 2. Metadados de trecho ----------------------------------------------
ALTER TABLE doza.code_chunks ADD COLUMN IF NOT EXISTS start_line int;
ALTER TABLE doza.code_chunks ADD COLUMN IF NOT EXISTS end_line int;
ALTER TABLE doza.code_chunks ADD COLUMN IF NOT EXISTS symbols text;
ALTER TABLE doza.code_chunks ADD COLUMN IF NOT EXISTS module text;
ALTER TABLE doza.code_chunks ADD COLUMN IF NOT EXISTS kind text;
ALTER TABLE doza.code_chunks ADD COLUMN IF NOT EXISTS file_mtime double precision;
-- 3. Índices lexicos (pg_trgm) — o lado keyword da busca híbrida -------
CREATE INDEX IF NOT EXISTS code_chunks_file_path_trgm_idx
ON doza.code_chunks USING gin (file_path gin_trgm_ops);
CREATE INDEX IF NOT EXISTS code_chunks_symbols_trgm_idx
ON doza.code_chunks USING gin (symbols gin_trgm_ops);
CREATE INDEX IF NOT EXISTS code_chunks_module_idx
ON doza.code_chunks (module);
CREATE INDEX IF NOT EXISTS idx_code_chunks_file_path
ON doza.code_chunks (file_path);
-- 4. Mapa de arquivos, com embedding do resumo ------------------------
CREATE TABLE IF NOT EXISTS doza.file_index (
file_path text PRIMARY KEY,
module text,
main_type text,
public_symbols text[],
summary text,
n_lines int,
content_hash text,
summary_embedding vector(768),
updated_at timestamp DEFAULT now()
);
CREATE INDEX IF NOT EXISTS file_index_module_idx
ON doza.file_index (module);
CREATE INDEX IF NOT EXISTS file_index_path_trgm_idx
ON doza.file_index USING gin (file_path gin_trgm_ops);
CREATE INDEX IF NOT EXISTS file_index_summary_hnsw_idx
ON doza.file_index USING hnsw (summary_embedding vector_cosine_ops)
WITH (m = 16, ef_construction = 64);
-- 5. Permissões do usuário de indexação/busca -------------------------
GRANT SELECT, INSERT, UPDATE, DELETE ON doza.file_index TO rag_doza_indexer;
GRANT USAGE, SELECT ON ALL SEQUENCES IN SCHEMA doza TO rag_doza_indexer;
COMMIT;
+96
View File
@@ -0,0 +1,96 @@
-- Migration 002 — busca hibrida e leitura cirurgica no schema `tigre`.
--
-- Motivo (medido em 02/09/2026, 346 chunks / 190 arquivos):
-- 1. O indice ivfflat com lists=100 sobre 346 linhas da ~3,5 linhas por
-- lista. Com o padrao ivfflat.probes=1 a busca varre ~3 de 346 linhas,
-- e varias consultas voltavam VAZIAS. recall@5 do conjunto dourado
-- (rag/bench.py): 33%. Trocamos por HNSW, que nao particiona.
-- 2. A busca densa pura erra nomes exatos de tipo (`FCPXMLValidator` nao
-- aparecia no top-5 do proprio arquivo). Adicionamos `symbols` +
-- indices GIN trgm para o lado lexico da fusao.
-- 3. Os chunks nao guardavam faixa de linhas, entao o agente lia o
-- arquivo inteiro depois da busca. `start_line`/`end_line` permitem
-- ler so a janela relevante.
--
-- Idempotente: seguro rodar mais de uma vez.
--
-- COMO RODAR (precisa ser rag_admin — o rag_tigre_indexer nao tem DDL):
--
-- rag/ensure_tunnel.sh
-- psql -h 127.0.0.1 -p 55435 -U rag_admin -d rag_tigre \
-- -f rag/migrations/002-tigre-search.sql
--
-- A senha do rag_admin esta em admin/VPS-ACCESS.md.
--
-- DEPOIS DE RODAR: reindexar do zero e obrigatorio.
-- rag/reindex_tigre.sh --full
-- Os embeddings passam a usar o prefixo `search_document: ` exigido pelo
-- nomic-embed-text; vetores antigos e novos nao sao comparaveis entre si.
BEGIN;
-- ---------------------------------------------------------------------
-- 1. Indice vetorial: ivfflat -> HNSW
-- ---------------------------------------------------------------------
DROP INDEX IF EXISTS tigre.code_chunks_embedding_idx;
CREATE INDEX IF NOT EXISTS code_chunks_embedding_hnsw_idx
ON tigre.code_chunks USING hnsw (embedding vector_cosine_ops)
WITH (m = 16, ef_construction = 64);
-- ---------------------------------------------------------------------
-- 2. Metadados de chunk: faixa de linhas, simbolos, modulo, tipo
-- ---------------------------------------------------------------------
ALTER TABLE tigre.code_chunks ADD COLUMN IF NOT EXISTS start_line int;
ALTER TABLE tigre.code_chunks ADD COLUMN IF NOT EXISTS end_line int;
ALTER TABLE tigre.code_chunks ADD COLUMN IF NOT EXISTS symbols text;
ALTER TABLE tigre.code_chunks ADD COLUMN IF NOT EXISTS module text;
-- kind: 'decl' (declaracao Swift), 'header' (imports + doc do tipo),
-- 'window' (fallback por janela de linhas, usado fora do Swift)
ALTER TABLE tigre.code_chunks ADD COLUMN IF NOT EXISTS kind text;
-- ---------------------------------------------------------------------
-- 3. Indices lexicos (pg_trgm) — o lado keyword da busca hibrida
-- ---------------------------------------------------------------------
CREATE INDEX IF NOT EXISTS code_chunks_file_path_trgm_idx
ON tigre.code_chunks USING gin (file_path gin_trgm_ops);
CREATE INDEX IF NOT EXISTS code_chunks_symbols_trgm_idx
ON tigre.code_chunks USING gin (symbols gin_trgm_ops);
-- Filtro de escopo por modulo (--module TigreAI).
CREATE INDEX IF NOT EXISTS code_chunks_module_idx
ON tigre.code_chunks (module);
-- ---------------------------------------------------------------------
-- 4. Mapa de arquivos — 1 linha por arquivo, para responder "onde fica X"
-- sem trazer nenhum corpo de codigo.
-- ---------------------------------------------------------------------
CREATE TABLE IF NOT EXISTS tigre.file_index (
file_path text PRIMARY KEY,
module text,
main_type text,
public_symbols text[],
summary text,
n_lines int,
content_hash text,
updated_at timestamp DEFAULT now()
);
CREATE INDEX IF NOT EXISTS file_index_module_idx
ON tigre.file_index (module);
CREATE INDEX IF NOT EXISTS file_index_path_trgm_idx
ON tigre.file_index USING gin (file_path gin_trgm_ops);
-- ---------------------------------------------------------------------
-- 5. Permissoes do usuario de indexacao/busca
-- ---------------------------------------------------------------------
GRANT SELECT, INSERT, UPDATE, DELETE ON tigre.file_index TO rag_tigre_indexer;
GRANT USAGE, SELECT ON ALL SEQUENCES IN SCHEMA tigre TO rag_tigre_indexer;
COMMIT;
-- Conferencia rapida (rode a mao depois):
-- \d tigre.code_chunks
-- SELECT indexname FROM pg_indexes WHERE schemaname = 'tigre';
@@ -0,0 +1,28 @@
-- Migration 003 — embedding do resumo de cada arquivo.
--
-- Motivo (medido depois da 002): os scores do nomic-embed-text ficam
-- comprimidos numa faixa estreita (0,50–0,65 neste corpus), e arquivos
-- grandes viram "hubs" — casam morno com qualquer consulta e ocupam o topo.
-- Dois alvos legítimos caíam para as posições 33 e 63 da lista densa com
-- score praticamente colado no do 1º colocado:
--
-- "gravar arquivo em disco de forma atomica" -> AtomicFileIO (33º)
-- "impedir caminho de projeto fora da pasta ..." -> ProjectPathSafety (63º)
--
-- Um embedding só do resumo (caminho + tipo + doc-comment, sem corpo de
-- código para diluir) não sofre disso, e entra como terceira lista da fusão.
--
-- COMO RODAR:
-- rag/migrate_tigre.sh rag/migrations/003-file-summary-embedding.sql
-- DEPOIS: rag/index_tigre.sh --full
BEGIN;
ALTER TABLE tigre.file_index
ADD COLUMN IF NOT EXISTS summary_embedding vector(768);
CREATE INDEX IF NOT EXISTS file_index_summary_hnsw_idx
ON tigre.file_index USING hnsw (summary_embedding vector_cosine_ops)
WITH (m = 16, ef_construction = 64);
COMMIT;