feat: initial commit - Jhonny Editor
- Adicionado estrutura completa do projeto - Configurado MCP server para Premiere Pro - Adicionado documentação e skills - Configurado Gitignore para o projeto
This commit is contained in:
@@ -0,0 +1,444 @@
|
||||
"""Indexador RAG do projeto.
|
||||
|
||||
Varre um diretório de código, quebra cada arquivo em trechos, gera embeddings
|
||||
via Ollama (`nomic-embed-text`, 768 dims — bate com `code_chunks.embedding
|
||||
vector(768)`) e grava tudo no Postgres da VPS (via túnel SSH).
|
||||
|
||||
Uso:
|
||||
rag/index_tigre.sh # incremental (só o que mudou)
|
||||
rag/index_tigre.sh --full # reindexa tudo, ignorando o hash
|
||||
rag/index_tigre.sh --file codeclass/Sources/TigreAI/OllamaClient.swift
|
||||
|
||||
Reindexação é incremental: um hash (md5) do conteúdo de cada arquivo fica
|
||||
salvo em `code_chunks.content_hash`. Se o hash não mudou, o arquivo é pulado.
|
||||
Arquivos apagados do repo também têm seus chunks removidos.
|
||||
|
||||
Como o corte é feito
|
||||
--------------------
|
||||
Arquivos `.swift` são cortados por declaração (`swift_chunker`), não por
|
||||
janela cega de linhas: cada trecho começa e termina em fronteira de código e
|
||||
guarda `start_line`/`end_line`. É isso que permite ao agente ler só a janela
|
||||
relevante depois da busca, em vez do arquivo inteiro. Os demais formatos
|
||||
usam janela de linhas, também com faixa registrada.
|
||||
|
||||
Prefixos do embedding
|
||||
---------------------
|
||||
O nomic-embed-text espera `search_document: ` no que é indexado e
|
||||
`search_query: ` no que é consultado. Sem isso a qualidade cai. Trocar de
|
||||
regime invalida os vetores antigos, então quem muda isso precisa reindexar
|
||||
com `--full`.
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import os
|
||||
import sys
|
||||
|
||||
import psycopg2
|
||||
import requests
|
||||
from dotenv import load_dotenv
|
||||
|
||||
RAG_DIR = os.path.dirname(os.path.abspath(__file__))
|
||||
PROJECT_ROOT = os.path.dirname(RAG_DIR)
|
||||
DEFAULT_TARGET = os.path.join(PROJECT_ROOT, "code")
|
||||
|
||||
if RAG_DIR not in sys.path:
|
||||
sys.path.insert(0, RAG_DIR)
|
||||
|
||||
# Arquivo de env a carregar. Por padrão usa o .env do rag (sistema "doza",
|
||||
# alvo code/). Um wrapper por sistema pode apontar para outro arquivo env
|
||||
# (ex: RAG_ENV_FILE=tigre.env) e setar RAG_TARGET_DIR para o seu diretório.
|
||||
_env_file = os.environ.get("RAG_ENV_FILE", os.path.join(RAG_DIR, ".env"))
|
||||
load_dotenv(_env_file)
|
||||
|
||||
from swift_chunker import chunk_swift # noqa: E402
|
||||
from swift_chunker import file_facts as swift_facts # noqa: E402
|
||||
from python_chunker import chunk_python # noqa: E402
|
||||
from python_chunker import file_facts as python_facts # noqa: E402
|
||||
from ts_chunker import chunk_ts # noqa: E402
|
||||
from ts_chunker import file_facts as ts_facts # noqa: E402
|
||||
|
||||
OLLAMA_URL = os.environ.get("OLLAMA_URL", "http://localhost:11434")
|
||||
EMBED_MODEL = os.environ.get("RAG_EMBED_MODEL", "nomic-embed-text")
|
||||
EMBED_DIM = 768
|
||||
|
||||
# Prefixos exigidos pelo nomic-embed-text. Exportados para o search.py usar
|
||||
# o par correspondente na consulta.
|
||||
DOC_PREFIX = "search_document: "
|
||||
QUERY_PREFIX = "search_query: "
|
||||
|
||||
# Os prefixos só podem ser usados se o índice TAMBÉM foi construído com eles
|
||||
# — misturar os dois regimes deixa consulta e documento em espaços
|
||||
# diferentes. Ligado por padrão (é o comportamento correto para um índice
|
||||
# novo); o `.env` do banco legado `doza`, indexado antes disso, desliga com
|
||||
# RAG_EMBED_PREFIX=0.
|
||||
USE_PREFIX = os.environ.get("RAG_EMBED_PREFIX", "1").lower() not in ("0", "false", "no")
|
||||
|
||||
# Extensões consideradas "código/documentação" para fins de RAG. Permite
|
||||
# sobrescrever por env (ex: RAG_INCLUDE_EXT=".swift,.py,.md").
|
||||
_include_ext_csv = os.environ.get("RAG_INCLUDE_EXT", ".py,.md,.sql,.sh,.command,.txt,.swift")
|
||||
_INCLUDE_EXT = {e.strip() for e in _include_ext_csv.split(",") if e.strip()}
|
||||
|
||||
# Diretórios que nunca devem ser indexados (ambientes virtuais, caches,
|
||||
# artefatos de build, dados de usuário).
|
||||
_exclude_dir_csv = os.environ.get(
|
||||
"RAG_EXCLUDE_DIRS",
|
||||
".venv,.venv.bak,__pycache__,.pytest_cache,.git,node_modules,projects,"
|
||||
"exports,dist,build,icon_build,.github,.build,DerivedData,.swiftpm"
|
||||
)
|
||||
_EXCLUDE_DIRS = {e.strip() for e in _exclude_dir_csv.split(",") if e.strip()}
|
||||
|
||||
# Lockfiles y artefactos de dependencias por nombre: ruido puro para RAG
|
||||
# (JSON de resolución de dependencias, no el código que importa).
|
||||
_EXCLUDE_FILES = {
|
||||
"package-lock.json", "pnpm-lock.yaml", "yarn.lock", "npm-shrinkwrap.json",
|
||||
"poetry.lock", "Pipfile.lock", "composer.lock", "composer.lock.json",
|
||||
"Cargo.lock", "Gemfile.lock", "go.sum", "uv.lock",
|
||||
}
|
||||
|
||||
_CHUNK_LINES = 60
|
||||
_CHUNK_OVERLAP = 10
|
||||
# nomic-embed-text's context window is 2048 tokens (~4 chars/token). Keep a
|
||||
# safety margin below that so long lines (minified JS, long docstrings)
|
||||
# don't blow the limit even when the line count is small.
|
||||
_CHUNK_MAX_CHARS = 5000
|
||||
|
||||
|
||||
def _qid(schema):
|
||||
"""Schema como identificador SQL seguro (aspas duplas) — aceita guífen
|
||||
(ex: `jhonny-rag`) sem quebrar a sintaxe. Retroativo: produtos sem guífen
|
||||
voltam idênticos (`doza` -> `"doza"`)."""
|
||||
return '"' + schema.replace('"', '""') + '"'
|
||||
|
||||
|
||||
def _db_connect():
|
||||
return psycopg2.connect(
|
||||
host=os.environ["RAG_DB_HOST"],
|
||||
port=os.environ["RAG_DB_PORT"],
|
||||
dbname=os.environ["RAG_DB_NAME"],
|
||||
user=os.environ["RAG_DB_USER"],
|
||||
password=os.environ["RAG_DB_PASSWORD"],
|
||||
connect_timeout=5,
|
||||
)
|
||||
|
||||
|
||||
def _iter_files(target_dir):
|
||||
for root, dirs, files in os.walk(target_dir):
|
||||
dirs[:] = [d for d in dirs if d not in _EXCLUDE_DIRS and not d.startswith(".")]
|
||||
for name in files:
|
||||
if name in _EXCLUDE_FILES:
|
||||
continue
|
||||
if os.path.splitext(name)[1] in _INCLUDE_EXT:
|
||||
yield os.path.join(root, name)
|
||||
|
||||
|
||||
def _module_of(rel_path):
|
||||
"""Módulo SwiftPM do arquivo: o diretório logo abaixo de `Sources/`."""
|
||||
parts = rel_path.split(os.sep)
|
||||
if "Sources" in parts:
|
||||
i = parts.index("Sources")
|
||||
if i + 1 < len(parts) - 1:
|
||||
return parts[i + 1]
|
||||
return parts[1] if len(parts) > 2 else None
|
||||
|
||||
|
||||
def _chunk_windows(text, chunk_lines=_CHUNK_LINES, overlap=_CHUNK_OVERLAP,
|
||||
max_chars=_CHUNK_MAX_CHARS):
|
||||
"""Corte por janela de linhas — usado fora do Swift.
|
||||
|
||||
Devolve dicts no mesmo formato do `chunk_swift`, com a faixa de linhas
|
||||
preenchida, para todo o resto do pipeline não precisar saber qual dos
|
||||
dois cortadores produziu o trecho.
|
||||
"""
|
||||
lines = text.splitlines()
|
||||
if not lines:
|
||||
return []
|
||||
chunks = []
|
||||
step = max(1, chunk_lines - overlap)
|
||||
for start in range(0, len(lines), step):
|
||||
window = lines[start:start + chunk_lines]
|
||||
# Teto de caracteres aplicado POR LINHA, nunca no meio de uma linha,
|
||||
# para start_line/end_line continuarem verdadeiros.
|
||||
buf, buf_start, size = [], start, 0
|
||||
for offset, line in enumerate(window):
|
||||
if buf and size + len(line) + 1 > max_chars:
|
||||
body = "\n".join(buf).strip()
|
||||
if body:
|
||||
chunks.append({
|
||||
"content": body,
|
||||
"start_line": buf_start + 1,
|
||||
"end_line": buf_start + len(buf),
|
||||
"symbols": None,
|
||||
"kind": "window",
|
||||
})
|
||||
buf, buf_start, size = [], start + offset, 0
|
||||
buf.append(line)
|
||||
size += len(line) + 1
|
||||
body = "\n".join(buf).strip()
|
||||
if body:
|
||||
chunks.append({
|
||||
"content": body,
|
||||
"start_line": buf_start + 1,
|
||||
"end_line": buf_start + len(buf),
|
||||
"symbols": None,
|
||||
"kind": "window",
|
||||
})
|
||||
if start + chunk_lines >= len(lines):
|
||||
break
|
||||
return chunks
|
||||
|
||||
|
||||
def chunk_file(text, rel_path):
|
||||
"""Escolhe o cortador certo para o arquivo."""
|
||||
ext = os.path.splitext(rel_path)[1]
|
||||
cutters = {
|
||||
".swift": chunk_swift, ".py": chunk_python,
|
||||
".ts": chunk_ts, ".tsx": chunk_ts, ".js": chunk_ts, ".jsx": chunk_ts,
|
||||
}
|
||||
cutter = cutters.get(ext)
|
||||
if cutter:
|
||||
chunks = cutter(text, rel_path)
|
||||
if chunks:
|
||||
return chunks
|
||||
return _chunk_windows(text)
|
||||
|
||||
|
||||
def _generic_facts(text, rel_path):
|
||||
"""Fatos de arquivos sem cortador próprio (.md, .sh, .sql, .txt).
|
||||
|
||||
O resumo é a primeira linha com conteúdo — num Markdown isso é o título,
|
||||
num shell script o comentário de cabeçalho.
|
||||
"""
|
||||
lines = text.splitlines()
|
||||
summary = None
|
||||
for line in lines[:30]:
|
||||
stripped = line.strip().lstrip("#!").lstrip("# ").strip()
|
||||
if stripped and not stripped.startswith(("/bin/", "/usr/")):
|
||||
summary = stripped
|
||||
break
|
||||
return {
|
||||
"main_type": os.path.splitext(os.path.basename(rel_path))[0],
|
||||
"summary": summary,
|
||||
"public_symbols": [],
|
||||
"n_lines": len(lines),
|
||||
}
|
||||
|
||||
|
||||
def file_facts(text, rel_path):
|
||||
"""Fatos do arquivo para o mapa (`file_index`), por linguagem."""
|
||||
ext = os.path.splitext(rel_path)[1]
|
||||
if ext == ".swift":
|
||||
return swift_facts(text, rel_path)
|
||||
if ext == ".py":
|
||||
return python_facts(text, rel_path)
|
||||
if ext in (".ts", ".tsx", ".js", ".jsx"):
|
||||
return ts_facts(text, rel_path)
|
||||
return _generic_facts(text, rel_path)
|
||||
|
||||
|
||||
def _embed(text, prefix=QUERY_PREFIX):
|
||||
"""Embedding via Ollama.
|
||||
|
||||
O prefixo padrão é o de CONSULTA porque quem importa esta função de fora
|
||||
(search.py) está sempre consultando; a indexação passa `DOC_PREFIX`
|
||||
explicitamente.
|
||||
"""
|
||||
prompt = f"{prefix}{text}" if USE_PREFIX else text
|
||||
resp = requests.post(
|
||||
f"{OLLAMA_URL}/api/embeddings",
|
||||
json={"model": EMBED_MODEL, "prompt": prompt},
|
||||
timeout=60,
|
||||
)
|
||||
resp.raise_for_status()
|
||||
vec = resp.json()["embedding"]
|
||||
if len(vec) != EMBED_DIM:
|
||||
raise ValueError(f"embedding com {len(vec)} dims, esperado {EMBED_DIM}")
|
||||
return vec
|
||||
|
||||
|
||||
def _file_hash(text):
|
||||
return hashlib.md5(text.encode("utf-8")).hexdigest()
|
||||
|
||||
|
||||
def _has_file_index(cur, schema):
|
||||
"""O mapa de arquivos só existe onde a migration 002 rodou (schema tigre).
|
||||
|
||||
O banco legado (doza) segue sem ele; o indexador é o mesmo para os dois.
|
||||
"""
|
||||
cur.execute("SELECT to_regclass(%s)", (f"{_qid(schema)}.file_index",))
|
||||
return cur.fetchone()[0] is not None
|
||||
|
||||
|
||||
def index_project(target_dir=DEFAULT_TARGET, full=False, only_file=None):
|
||||
conn = _db_connect()
|
||||
conn.autocommit = False
|
||||
schema = os.environ.get("RAG_DB_SCHEMA", "doza")
|
||||
files_done = files_skipped = chunks_done = 0
|
||||
stale_paths = set()
|
||||
try:
|
||||
with conn.cursor() as cur:
|
||||
has_map = _has_file_index(cur, schema)
|
||||
if not has_map:
|
||||
print(" [aviso] sem tabela file_index neste schema — modo mapa "
|
||||
"indisponivel (rode rag/migrate_tigre.sh)")
|
||||
|
||||
seen_paths = set()
|
||||
paths = ([os.path.join(PROJECT_ROOT, only_file)] if only_file
|
||||
else sorted(_iter_files(target_dir)))
|
||||
|
||||
for path in paths:
|
||||
rel_path = os.path.relpath(path, PROJECT_ROOT)
|
||||
try:
|
||||
with open(path, encoding="utf-8", errors="ignore") as f:
|
||||
text = f.read()
|
||||
except OSError as exc:
|
||||
print(f" skip {rel_path}: {exc}", file=sys.stderr)
|
||||
continue
|
||||
|
||||
seen_paths.add(rel_path)
|
||||
content_hash = _file_hash(text)
|
||||
mtime = os.path.getmtime(path)
|
||||
|
||||
if not full:
|
||||
cur.execute(
|
||||
f"SELECT DISTINCT content_hash FROM {_qid(schema)}.code_chunks "
|
||||
f"WHERE file_path = %s LIMIT 1",
|
||||
(rel_path,),
|
||||
)
|
||||
row = cur.fetchone()
|
||||
if row and row[0] == content_hash:
|
||||
files_skipped += 1
|
||||
print(f" [SKIP] {rel_path} sem alteracao")
|
||||
continue
|
||||
|
||||
module = _module_of(rel_path)
|
||||
chunks = chunk_file(text, rel_path)
|
||||
facts = file_facts(text, rel_path)
|
||||
# Prefixo de contexto no texto QUE VAI PARA O EMBEDDING (o
|
||||
# `content` gravado continua sendo o código puro). Sem isso o
|
||||
# doc-comment do tipo fica diluído no meio do código e
|
||||
# consultas conceituais erram: "detectar rostos" não achava o
|
||||
# FacePerceiver, cujo doc diz exatamente "Percepção de rostos".
|
||||
context = " · ".join(
|
||||
p for p in (rel_path, facts["main_type"], facts["summary"]) if p
|
||||
)
|
||||
|
||||
cur.execute(
|
||||
f"DELETE FROM {_qid(schema)}.code_chunks WHERE file_path = %s",
|
||||
(rel_path,),
|
||||
)
|
||||
inserted = 0
|
||||
for i, chunk in enumerate(chunks):
|
||||
try:
|
||||
embedding = _embed(f"{context}\n{chunk['content']}",
|
||||
prefix=DOC_PREFIX)
|
||||
except (requests.RequestException, ValueError) as exc:
|
||||
print(f" skip chunk {rel_path}#{i}: {exc}", file=sys.stderr)
|
||||
continue
|
||||
cur.execute(
|
||||
f"""
|
||||
INSERT INTO {_qid(schema)}.code_chunks
|
||||
(file_path, content, chunk_index, embedding,
|
||||
content_hash, file_mtime, start_line, end_line,
|
||||
symbols, module, kind)
|
||||
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s)
|
||||
""",
|
||||
(rel_path, chunk["content"], i, embedding, content_hash,
|
||||
mtime, chunk["start_line"], chunk["end_line"],
|
||||
chunk["symbols"], module, chunk["kind"]),
|
||||
)
|
||||
inserted += 1
|
||||
chunks_done += 1
|
||||
|
||||
if has_map:
|
||||
# Embedding só do resumo — sem corpo de código para
|
||||
# diluir. É a terceira lista da busca, a que resgata
|
||||
# arquivos pequenos e precisos que os arquivos grandes
|
||||
# abafam na lista densa por chunk.
|
||||
try:
|
||||
summary_emb = _embed(context, prefix=DOC_PREFIX)
|
||||
except (requests.RequestException, ValueError) as exc:
|
||||
print(f" skip resumo {rel_path}: {exc}", file=sys.stderr)
|
||||
summary_emb = None
|
||||
cur.execute(
|
||||
f"""
|
||||
INSERT INTO {_qid(schema)}.file_index
|
||||
(file_path, module, main_type, public_symbols,
|
||||
summary, n_lines, content_hash,
|
||||
summary_embedding, updated_at)
|
||||
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, now())
|
||||
ON CONFLICT (file_path) DO UPDATE SET
|
||||
module = EXCLUDED.module,
|
||||
main_type = EXCLUDED.main_type,
|
||||
public_symbols = EXCLUDED.public_symbols,
|
||||
summary = EXCLUDED.summary,
|
||||
n_lines = EXCLUDED.n_lines,
|
||||
content_hash = EXCLUDED.content_hash,
|
||||
summary_embedding = EXCLUDED.summary_embedding,
|
||||
updated_at = now()
|
||||
""",
|
||||
(rel_path, module, facts["main_type"],
|
||||
facts["public_symbols"], facts["summary"],
|
||||
facts["n_lines"], content_hash, summary_emb),
|
||||
)
|
||||
|
||||
files_done += 1
|
||||
print(f" {rel_path}: {inserted} trecho(s)")
|
||||
|
||||
# Arquivos que sumiram do repo desde a última rodada: limpa os
|
||||
# chunks, senão o índice acumula lixo morto. Não se aplica ao
|
||||
# modo --file, que só enxerga um arquivo.
|
||||
if not only_file:
|
||||
cur.execute(f"SELECT DISTINCT file_path FROM {_qid(schema)}.code_chunks")
|
||||
stale_paths = {r[0] for r in cur.fetchall()} - seen_paths
|
||||
for rel_path in stale_paths:
|
||||
cur.execute(
|
||||
f"DELETE FROM {_qid(schema)}.code_chunks WHERE file_path = %s",
|
||||
(rel_path,),
|
||||
)
|
||||
if has_map:
|
||||
cur.execute(
|
||||
f"DELETE FROM {_qid(schema)}.file_index WHERE file_path = %s",
|
||||
(rel_path,),
|
||||
)
|
||||
print(f" [DELETE] removido (nao existe mais): {rel_path}")
|
||||
|
||||
conn.commit()
|
||||
except Exception:
|
||||
conn.rollback()
|
||||
raise
|
||||
finally:
|
||||
conn.close()
|
||||
print(
|
||||
f"\nOK: {files_done} arquivo(s) reindexado(s), {files_skipped} sem mudanca "
|
||||
f"(pulado(s)), {len(stale_paths)} removido(s), {chunks_done} trecho(s) gravado(s)."
|
||||
)
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description="Indexador RAG")
|
||||
ap.add_argument("target", nargs="?", default=None,
|
||||
help="diretorio a indexar (padrao: RAG_TARGET_DIR ou code/)")
|
||||
ap.add_argument("--full", action="store_true",
|
||||
help="reindexa tudo, ignorando o hash de conteudo")
|
||||
ap.add_argument("--file", default=None,
|
||||
help="reindexa um unico arquivo (caminho relativo a raiz)")
|
||||
args = ap.parse_args()
|
||||
|
||||
# Alvo prioritário: RAG_TARGET_DIR da env (ex: codeclass/Sources). Pode
|
||||
# ser relativo ao project root; senão, argumento CLI; por último code/.
|
||||
env_target = os.environ.get("RAG_TARGET_DIR")
|
||||
default_target = (os.path.join(PROJECT_ROOT, env_target) if env_target
|
||||
else DEFAULT_TARGET)
|
||||
target = args.target or default_target
|
||||
|
||||
db = os.environ.get("RAG_DB_NAME", "?")
|
||||
schema = os.environ.get("RAG_DB_SCHEMA", "?")
|
||||
what = args.file or target
|
||||
mode = " (--full)" if args.full else ""
|
||||
print(f"Indexando {what} -> {db}.{schema}{mode}...")
|
||||
index_project(target, full=args.full, only_file=args.file)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user