"""Indexador RAG do projeto. Varre um diretório de código, quebra cada arquivo em trechos, gera embeddings via Ollama (`nomic-embed-text`, 768 dims — bate com `code_chunks.embedding vector(768)`) e grava tudo no Postgres da VPS (via túnel SSH). Uso: rag/index_tigre.sh # incremental (só o que mudou) rag/index_tigre.sh --full # reindexa tudo, ignorando o hash rag/index_tigre.sh --file codeclass/Sources/TigreAI/OllamaClient.swift Reindexação é incremental: um hash (md5) do conteúdo de cada arquivo fica salvo em `code_chunks.content_hash`. Se o hash não mudou, o arquivo é pulado. Arquivos apagados do repo também têm seus chunks removidos. Como o corte é feito -------------------- Arquivos `.swift` são cortados por declaração (`swift_chunker`), não por janela cega de linhas: cada trecho começa e termina em fronteira de código e guarda `start_line`/`end_line`. É isso que permite ao agente ler só a janela relevante depois da busca, em vez do arquivo inteiro. Os demais formatos usam janela de linhas, também com faixa registrada. Prefixos do embedding --------------------- O nomic-embed-text espera `search_document: ` no que é indexado e `search_query: ` no que é consultado. Sem isso a qualidade cai. Trocar de regime invalida os vetores antigos, então quem muda isso precisa reindexar com `--full`. """ import argparse import hashlib import os import sys import psycopg2 import requests from dotenv import load_dotenv RAG_DIR = os.path.dirname(os.path.abspath(__file__)) PROJECT_ROOT = os.path.dirname(RAG_DIR) DEFAULT_TARGET = os.path.join(PROJECT_ROOT, "code") if RAG_DIR not in sys.path: sys.path.insert(0, RAG_DIR) # Arquivo de env a carregar. Por padrão usa o .env do rag (sistema "doza", # alvo code/). Um wrapper por sistema pode apontar para outro arquivo env # (ex: RAG_ENV_FILE=tigre.env) e setar RAG_TARGET_DIR para o seu diretório. _env_file = os.environ.get("RAG_ENV_FILE", os.path.join(RAG_DIR, ".env")) load_dotenv(_env_file) from swift_chunker import chunk_swift # noqa: E402 from swift_chunker import file_facts as swift_facts # noqa: E402 from python_chunker import chunk_python # noqa: E402 from python_chunker import file_facts as python_facts # noqa: E402 from ts_chunker import chunk_ts # noqa: E402 from ts_chunker import file_facts as ts_facts # noqa: E402 OLLAMA_URL = os.environ.get("OLLAMA_URL", "http://localhost:11434") EMBED_MODEL = os.environ.get("RAG_EMBED_MODEL", "nomic-embed-text") EMBED_DIM = 768 # Prefixos exigidos pelo nomic-embed-text. Exportados para o search.py usar # o par correspondente na consulta. DOC_PREFIX = "search_document: " QUERY_PREFIX = "search_query: " # Os prefixos só podem ser usados se o índice TAMBÉM foi construído com eles # — misturar os dois regimes deixa consulta e documento em espaços # diferentes. Ligado por padrão (é o comportamento correto para um índice # novo); o `.env` do banco legado `doza`, indexado antes disso, desliga com # RAG_EMBED_PREFIX=0. USE_PREFIX = os.environ.get("RAG_EMBED_PREFIX", "1").lower() not in ("0", "false", "no") # Extensões consideradas "código/documentação" para fins de RAG. Permite # sobrescrever por env (ex: RAG_INCLUDE_EXT=".swift,.py,.md"). _include_ext_csv = os.environ.get("RAG_INCLUDE_EXT", ".py,.md,.sql,.sh,.command,.txt,.swift") _INCLUDE_EXT = {e.strip() for e in _include_ext_csv.split(",") if e.strip()} # Diretórios que nunca devem ser indexados (ambientes virtuais, caches, # artefatos de build, dados de usuário). _exclude_dir_csv = os.environ.get( "RAG_EXCLUDE_DIRS", ".venv,.venv.bak,__pycache__,.pytest_cache,.git,node_modules,projects," "exports,dist,build,icon_build,.github,.build,DerivedData,.swiftpm" ) _EXCLUDE_DIRS = {e.strip() for e in _exclude_dir_csv.split(",") if e.strip()} # Lockfiles y artefactos de dependencias por nombre: ruido puro para RAG # (JSON de resolución de dependencias, no el código que importa). _EXCLUDE_FILES = { "package-lock.json", "pnpm-lock.yaml", "yarn.lock", "npm-shrinkwrap.json", "poetry.lock", "Pipfile.lock", "composer.lock", "composer.lock.json", "Cargo.lock", "Gemfile.lock", "go.sum", "uv.lock", } _CHUNK_LINES = 60 _CHUNK_OVERLAP = 10 # nomic-embed-text's context window is 2048 tokens (~4 chars/token). Keep a # safety margin below that so long lines (minified JS, long docstrings) # don't blow the limit even when the line count is small. _CHUNK_MAX_CHARS = 5000 def _qid(schema): """Schema como identificador SQL seguro (aspas duplas) — aceita guífen (ex: `jhonny-rag`) sem quebrar a sintaxe. Retroativo: produtos sem guífen voltam idênticos (`doza` -> `"doza"`).""" return '"' + schema.replace('"', '""') + '"' def _db_connect(): return psycopg2.connect( host=os.environ["RAG_DB_HOST"], port=os.environ["RAG_DB_PORT"], dbname=os.environ["RAG_DB_NAME"], user=os.environ["RAG_DB_USER"], password=os.environ["RAG_DB_PASSWORD"], connect_timeout=5, ) def _iter_files(target_dir): for root, dirs, files in os.walk(target_dir): dirs[:] = [d for d in dirs if d not in _EXCLUDE_DIRS and not d.startswith(".")] for name in files: if name in _EXCLUDE_FILES: continue if os.path.splitext(name)[1] in _INCLUDE_EXT: yield os.path.join(root, name) def _module_of(rel_path): """Módulo SwiftPM do arquivo: o diretório logo abaixo de `Sources/`.""" parts = rel_path.split(os.sep) if "Sources" in parts: i = parts.index("Sources") if i + 1 < len(parts) - 1: return parts[i + 1] return parts[1] if len(parts) > 2 else None def _chunk_windows(text, chunk_lines=_CHUNK_LINES, overlap=_CHUNK_OVERLAP, max_chars=_CHUNK_MAX_CHARS): """Corte por janela de linhas — usado fora do Swift. Devolve dicts no mesmo formato do `chunk_swift`, com a faixa de linhas preenchida, para todo o resto do pipeline não precisar saber qual dos dois cortadores produziu o trecho. """ lines = text.splitlines() if not lines: return [] chunks = [] step = max(1, chunk_lines - overlap) for start in range(0, len(lines), step): window = lines[start:start + chunk_lines] # Teto de caracteres aplicado POR LINHA, nunca no meio de uma linha, # para start_line/end_line continuarem verdadeiros. buf, buf_start, size = [], start, 0 for offset, line in enumerate(window): if buf and size + len(line) + 1 > max_chars: body = "\n".join(buf).strip() if body: chunks.append({ "content": body, "start_line": buf_start + 1, "end_line": buf_start + len(buf), "symbols": None, "kind": "window", }) buf, buf_start, size = [], start + offset, 0 buf.append(line) size += len(line) + 1 body = "\n".join(buf).strip() if body: chunks.append({ "content": body, "start_line": buf_start + 1, "end_line": buf_start + len(buf), "symbols": None, "kind": "window", }) if start + chunk_lines >= len(lines): break return chunks def chunk_file(text, rel_path): """Escolhe o cortador certo para o arquivo.""" ext = os.path.splitext(rel_path)[1] cutters = { ".swift": chunk_swift, ".py": chunk_python, ".ts": chunk_ts, ".tsx": chunk_ts, ".js": chunk_ts, ".jsx": chunk_ts, } cutter = cutters.get(ext) if cutter: chunks = cutter(text, rel_path) if chunks: return chunks return _chunk_windows(text) def _generic_facts(text, rel_path): """Fatos de arquivos sem cortador próprio (.md, .sh, .sql, .txt). O resumo é a primeira linha com conteúdo — num Markdown isso é o título, num shell script o comentário de cabeçalho. """ lines = text.splitlines() summary = None for line in lines[:30]: stripped = line.strip().lstrip("#!").lstrip("# ").strip() if stripped and not stripped.startswith(("/bin/", "/usr/")): summary = stripped break return { "main_type": os.path.splitext(os.path.basename(rel_path))[0], "summary": summary, "public_symbols": [], "n_lines": len(lines), } def file_facts(text, rel_path): """Fatos do arquivo para o mapa (`file_index`), por linguagem.""" ext = os.path.splitext(rel_path)[1] if ext == ".swift": return swift_facts(text, rel_path) if ext == ".py": return python_facts(text, rel_path) if ext in (".ts", ".tsx", ".js", ".jsx"): return ts_facts(text, rel_path) return _generic_facts(text, rel_path) def _embed(text, prefix=QUERY_PREFIX): """Embedding via Ollama. O prefixo padrão é o de CONSULTA porque quem importa esta função de fora (search.py) está sempre consultando; a indexação passa `DOC_PREFIX` explicitamente. """ prompt = f"{prefix}{text}" if USE_PREFIX else text resp = requests.post( f"{OLLAMA_URL}/api/embeddings", json={"model": EMBED_MODEL, "prompt": prompt}, timeout=60, ) resp.raise_for_status() vec = resp.json()["embedding"] if len(vec) != EMBED_DIM: raise ValueError(f"embedding com {len(vec)} dims, esperado {EMBED_DIM}") return vec def _file_hash(text): return hashlib.md5(text.encode("utf-8")).hexdigest() def _has_file_index(cur, schema): """O mapa de arquivos só existe onde a migration 002 rodou (schema tigre). O banco legado (doza) segue sem ele; o indexador é o mesmo para os dois. """ cur.execute("SELECT to_regclass(%s)", (f"{_qid(schema)}.file_index",)) return cur.fetchone()[0] is not None def index_project(target_dir=DEFAULT_TARGET, full=False, only_file=None): conn = _db_connect() conn.autocommit = False schema = os.environ.get("RAG_DB_SCHEMA", "doza") files_done = files_skipped = chunks_done = 0 stale_paths = set() try: with conn.cursor() as cur: has_map = _has_file_index(cur, schema) if not has_map: print(" [aviso] sem tabela file_index neste schema — modo mapa " "indisponivel (rode rag/migrate_tigre.sh)") seen_paths = set() paths = ([os.path.join(PROJECT_ROOT, only_file)] if only_file else sorted(_iter_files(target_dir))) for path in paths: rel_path = os.path.relpath(path, PROJECT_ROOT) try: with open(path, encoding="utf-8", errors="ignore") as f: text = f.read() except OSError as exc: print(f" skip {rel_path}: {exc}", file=sys.stderr) continue seen_paths.add(rel_path) content_hash = _file_hash(text) mtime = os.path.getmtime(path) if not full: cur.execute( f"SELECT DISTINCT content_hash FROM {_qid(schema)}.code_chunks " f"WHERE file_path = %s LIMIT 1", (rel_path,), ) row = cur.fetchone() if row and row[0] == content_hash: files_skipped += 1 print(f" [SKIP] {rel_path} sem alteracao") continue module = _module_of(rel_path) chunks = chunk_file(text, rel_path) facts = file_facts(text, rel_path) # Prefixo de contexto no texto QUE VAI PARA O EMBEDDING (o # `content` gravado continua sendo o código puro). Sem isso o # doc-comment do tipo fica diluído no meio do código e # consultas conceituais erram: "detectar rostos" não achava o # FacePerceiver, cujo doc diz exatamente "Percepção de rostos". context = " · ".join( p for p in (rel_path, facts["main_type"], facts["summary"]) if p ) cur.execute( f"DELETE FROM {_qid(schema)}.code_chunks WHERE file_path = %s", (rel_path,), ) inserted = 0 for i, chunk in enumerate(chunks): try: embedding = _embed(f"{context}\n{chunk['content']}", prefix=DOC_PREFIX) except (requests.RequestException, ValueError) as exc: print(f" skip chunk {rel_path}#{i}: {exc}", file=sys.stderr) continue cur.execute( f""" INSERT INTO {_qid(schema)}.code_chunks (file_path, content, chunk_index, embedding, content_hash, file_mtime, start_line, end_line, symbols, module, kind) VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s, %s) """, (rel_path, chunk["content"], i, embedding, content_hash, mtime, chunk["start_line"], chunk["end_line"], chunk["symbols"], module, chunk["kind"]), ) inserted += 1 chunks_done += 1 if has_map: # Embedding só do resumo — sem corpo de código para # diluir. É a terceira lista da busca, a que resgata # arquivos pequenos e precisos que os arquivos grandes # abafam na lista densa por chunk. try: summary_emb = _embed(context, prefix=DOC_PREFIX) except (requests.RequestException, ValueError) as exc: print(f" skip resumo {rel_path}: {exc}", file=sys.stderr) summary_emb = None cur.execute( f""" INSERT INTO {_qid(schema)}.file_index (file_path, module, main_type, public_symbols, summary, n_lines, content_hash, summary_embedding, updated_at) VALUES (%s, %s, %s, %s, %s, %s, %s, %s, now()) ON CONFLICT (file_path) DO UPDATE SET module = EXCLUDED.module, main_type = EXCLUDED.main_type, public_symbols = EXCLUDED.public_symbols, summary = EXCLUDED.summary, n_lines = EXCLUDED.n_lines, content_hash = EXCLUDED.content_hash, summary_embedding = EXCLUDED.summary_embedding, updated_at = now() """, (rel_path, module, facts["main_type"], facts["public_symbols"], facts["summary"], facts["n_lines"], content_hash, summary_emb), ) files_done += 1 print(f" {rel_path}: {inserted} trecho(s)") # Arquivos que sumiram do repo desde a última rodada: limpa os # chunks, senão o índice acumula lixo morto. Não se aplica ao # modo --file, que só enxerga um arquivo. if not only_file: cur.execute(f"SELECT DISTINCT file_path FROM {_qid(schema)}.code_chunks") stale_paths = {r[0] for r in cur.fetchall()} - seen_paths for rel_path in stale_paths: cur.execute( f"DELETE FROM {_qid(schema)}.code_chunks WHERE file_path = %s", (rel_path,), ) if has_map: cur.execute( f"DELETE FROM {_qid(schema)}.file_index WHERE file_path = %s", (rel_path,), ) print(f" [DELETE] removido (nao existe mais): {rel_path}") conn.commit() except Exception: conn.rollback() raise finally: conn.close() print( f"\nOK: {files_done} arquivo(s) reindexado(s), {files_skipped} sem mudanca " f"(pulado(s)), {len(stale_paths)} removido(s), {chunks_done} trecho(s) gravado(s)." ) def main(): ap = argparse.ArgumentParser(description="Indexador RAG") ap.add_argument("target", nargs="?", default=None, help="diretorio a indexar (padrao: RAG_TARGET_DIR ou code/)") ap.add_argument("--full", action="store_true", help="reindexa tudo, ignorando o hash de conteudo") ap.add_argument("--file", default=None, help="reindexa um unico arquivo (caminho relativo a raiz)") args = ap.parse_args() # Alvo prioritário: RAG_TARGET_DIR da env (ex: codeclass/Sources). Pode # ser relativo ao project root; senão, argumento CLI; por último code/. env_target = os.environ.get("RAG_TARGET_DIR") default_target = (os.path.join(PROJECT_ROOT, env_target) if env_target else DEFAULT_TARGET) target = args.target or default_target db = os.environ.get("RAG_DB_NAME", "?") schema = os.environ.get("RAG_DB_SCHEMA", "?") what = args.file or target mode = " (--full)" if args.full else "" print(f"Indexando {what} -> {db}.{schema}{mode}...") index_project(target, full=args.full, only_file=args.file) if __name__ == "__main__": main()