feat: initial commit - Jhonny Editor
- Adicionado estrutura completa do projeto - Configurado MCP server para Premiere Pro - Adicionado documentação e skills - Configurado Gitignore para o projeto
This commit is contained in:
@@ -0,0 +1,262 @@
|
||||
"""Corte de arquivos Swift em trechos por declaração, com faixa de linhas.
|
||||
|
||||
O indexador antigo cortava em janelas cegas de 60 linhas, o que partia
|
||||
funções ao meio e não dizia em que linha cada trecho começava. Aqui o corte
|
||||
segue a estrutura do código: o cabeçalho do arquivo (imports + doc do tipo)
|
||||
vira um trecho, e cada membro de topo do tipo vira outro, já com a faixa de
|
||||
linhas que o agente usa para ler só aquela janela.
|
||||
|
||||
Combina com a regra do projeto de UMA CLASSE POR ARQUIVO (ver
|
||||
codeclass/README.md): o "tipo principal" de um arquivo é o primeiro tipo
|
||||
público declarado nele.
|
||||
"""
|
||||
|
||||
import os
|
||||
import re
|
||||
|
||||
# Palavras que abrem uma declaração. `case` entra por causa de enums, que
|
||||
# neste projeto carregam doc-comment por caso.
|
||||
_DECL_KW = (
|
||||
"func|init|deinit|subscript|var|let|case|enum|struct|class|protocol"
|
||||
"|actor|typealias|extension"
|
||||
)
|
||||
|
||||
# Modificadores que podem preceder a palavra-chave da declaração.
|
||||
_MODIFIERS = (
|
||||
"public|internal|private|fileprivate|open|final|static|class|override"
|
||||
"|mutating|nonmutating|required|convenience|lazy|weak|unowned|indirect"
|
||||
"|nonisolated|isolated|dynamic|optional|async|throws"
|
||||
)
|
||||
|
||||
_DECL_RE = re.compile(
|
||||
rf"^(?P<indent>[ \t]*)(?:(?:{_MODIFIERS})\s+)*(?P<kw>{_DECL_KW})\b"
|
||||
rf"(?:\s+(?P<name>[A-Za-z_][A-Za-z0-9_]*))?"
|
||||
)
|
||||
|
||||
# Tipos declarados no nível 0 do arquivo.
|
||||
_TYPE_RE = re.compile(
|
||||
rf"^(?:(?:{_MODIFIERS})\s+)*(?P<kw>enum|struct|class|protocol|actor|extension)"
|
||||
rf"\s+(?P<name>[A-Za-z_][A-Za-z0-9_]*)"
|
||||
)
|
||||
|
||||
_PUBLIC_RE = re.compile(r"^\s*(?:public|open)\b")
|
||||
|
||||
# Nomes declarados em um trecho — alimenta a coluna `symbols`, que é o lado
|
||||
# lexical (pg_trgm) da busca híbrida.
|
||||
_SYMBOL_RE = re.compile(
|
||||
rf"\b(?:{_DECL_KW})\s+(?P<name>[A-Za-z_][A-Za-z0-9_]*)"
|
||||
)
|
||||
|
||||
# Linhas que não abrem nem fecham escopo de verdade, para a contagem de
|
||||
# chaves não ser enganada por comentário e string literal.
|
||||
_LINE_COMMENT_RE = re.compile(r"//.*$")
|
||||
_STRING_RE = re.compile(r'"(?:[^"\\]|\\.)*"')
|
||||
|
||||
# Arquivos até este tamanho viram um único trecho: o arquivo inteiro. A
|
||||
# maioria dos arquivos do projeto cai aqui (média ~70 linhas), e um trecho
|
||||
# igual ao arquivo é o melhor resultado possível de busca.
|
||||
WHOLE_FILE_MAX_LINES = 120
|
||||
|
||||
# Teto rígido de caracteres por trecho. O nomic-embed-text tem janela de
|
||||
# 2048 tokens (~4 chars/token); ficamos com folga. Quando um trecho passa
|
||||
# disso ele é dividido POR LINHA (nunca no meio de uma linha), para
|
||||
# start_line/end_line continuarem verdadeiros.
|
||||
MAX_CHARS = 5000
|
||||
|
||||
# Tamanho que um trecho tenta alcançar juntando declarações vizinhas. Sem
|
||||
# isso cada `var` de uma linha viraria um trecho próprio: mais embeddings
|
||||
# para gerar, e cada resultado de busca carregando pouco contexto útil.
|
||||
TARGET_CHARS = 2000
|
||||
|
||||
|
||||
def _strip_noise(line):
|
||||
return _LINE_COMMENT_RE.sub("", _STRING_RE.sub('""', line))
|
||||
|
||||
|
||||
def _preamble_start(lines, i):
|
||||
"""Recua de `i` para incluir doc-comment e atributos da declaração."""
|
||||
j = i
|
||||
while j > 0:
|
||||
prev = lines[j - 1].strip()
|
||||
if prev.startswith("///") or prev.startswith("//") or prev.startswith("@") \
|
||||
or prev.startswith("*") or prev.startswith("/*"):
|
||||
j -= 1
|
||||
else:
|
||||
break
|
||||
return j
|
||||
|
||||
|
||||
def _split_long(chunk_lines, start_line, max_chars=MAX_CHARS):
|
||||
"""Divide um trecho grande em pedaços, sempre em fronteira de linha."""
|
||||
out = []
|
||||
buf, buf_start = [], start_line
|
||||
size = 0
|
||||
for offset, line in enumerate(chunk_lines):
|
||||
if buf and size + len(line) + 1 > max_chars:
|
||||
out.append((buf, buf_start))
|
||||
buf, buf_start, size = [], start_line + offset, 0
|
||||
buf.append(line)
|
||||
size += len(line) + 1
|
||||
if buf:
|
||||
out.append((buf, buf_start))
|
||||
return out
|
||||
|
||||
|
||||
def _merge_small(chunks):
|
||||
"""Junta declarações vizinhas curtas até `TARGET_CHARS`.
|
||||
|
||||
Só funde trechos contíguos (o fim de um encosta no começo do outro), para
|
||||
a faixa de linhas do trecho fundido continuar sendo uma janela única e
|
||||
legível com `Read(offset=…, limit=…)`.
|
||||
"""
|
||||
merged = []
|
||||
for c in chunks:
|
||||
prev = merged[-1] if merged else None
|
||||
contiguous = prev is not None and prev["end_line"] + 1 == c["start_line"]
|
||||
same_kind = prev is not None and prev["kind"] == c["kind"] == "decl"
|
||||
fits = prev is not None and \
|
||||
len(prev["content"]) + len(c["content"]) + 1 <= TARGET_CHARS
|
||||
if contiguous and same_kind and fits:
|
||||
prev["content"] += "\n" + c["content"]
|
||||
prev["end_line"] = c["end_line"]
|
||||
prev["symbols"] = " ".join(
|
||||
sorted(set(prev["symbols"].split()) | set(c["symbols"].split()))
|
||||
)
|
||||
else:
|
||||
merged.append(dict(c))
|
||||
return merged
|
||||
|
||||
|
||||
def _symbols_of(text, file_stem, main_type):
|
||||
"""Nomes buscáveis do trecho.
|
||||
|
||||
O nome do arquivo entra sempre: é o que faz uma consulta pelo nome do
|
||||
tipo (`FCPXMLValidator`) casar com o arquivo certo — exatamente o caso
|
||||
em que a busca puramente densa errava.
|
||||
"""
|
||||
names = {file_stem}
|
||||
if main_type:
|
||||
names.add(main_type)
|
||||
names.update(m.group("name") for m in _SYMBOL_RE.finditer(text))
|
||||
return " ".join(sorted(n for n in names if n))
|
||||
|
||||
|
||||
def file_facts(text, rel_path):
|
||||
"""Tipo principal, símbolos públicos e resumo — alimenta `file_index`."""
|
||||
lines = text.splitlines()
|
||||
main_type = None
|
||||
summary = None
|
||||
public_symbols = []
|
||||
|
||||
for i, line in enumerate(lines):
|
||||
m = _TYPE_RE.match(line)
|
||||
if m and main_type is None and m.group("kw") != "extension":
|
||||
main_type = m.group("name")
|
||||
# Resumo: primeira linha do doc-comment logo acima do tipo.
|
||||
for k in range(_preamble_start(lines, i), i):
|
||||
doc = lines[k].strip()
|
||||
if doc.startswith("///"):
|
||||
summary = doc.lstrip("/").strip()
|
||||
break
|
||||
if _PUBLIC_RE.match(line):
|
||||
d = _DECL_RE.match(line)
|
||||
if d and d.group("name"):
|
||||
public_symbols.append(d.group("name"))
|
||||
|
||||
if main_type is None:
|
||||
main_type = os.path.splitext(os.path.basename(rel_path))[0]
|
||||
if summary is None:
|
||||
for line in lines[:40]:
|
||||
s = line.strip()
|
||||
if s.startswith("///"):
|
||||
summary = s.lstrip("/").strip()
|
||||
break
|
||||
|
||||
return {
|
||||
"main_type": main_type,
|
||||
"summary": summary,
|
||||
"public_symbols": sorted(set(public_symbols)),
|
||||
"n_lines": len(lines),
|
||||
}
|
||||
|
||||
|
||||
def chunk_swift(text, rel_path):
|
||||
"""Divide um arquivo Swift em trechos com faixa de linhas e símbolos.
|
||||
|
||||
Devolve lista de dicts: content, start_line, end_line, symbols, kind.
|
||||
Linhas são 1-indexadas e inclusivas nas duas pontas, iguais ao que o
|
||||
`Read(offset=…, limit=…)` do agente espera.
|
||||
"""
|
||||
lines = text.splitlines()
|
||||
if not lines:
|
||||
return []
|
||||
|
||||
stem = os.path.splitext(os.path.basename(rel_path))[0]
|
||||
facts = file_facts(text, rel_path)
|
||||
main_type = facts["main_type"]
|
||||
|
||||
def build(chunk_lines, start_line, kind):
|
||||
out = []
|
||||
for piece, piece_start in _split_long(chunk_lines, start_line):
|
||||
body = "\n".join(piece).strip()
|
||||
if not body:
|
||||
continue
|
||||
out.append({
|
||||
"content": body,
|
||||
"start_line": piece_start,
|
||||
"end_line": piece_start + len(piece) - 1,
|
||||
"symbols": _symbols_of(body, stem, main_type),
|
||||
"kind": kind,
|
||||
})
|
||||
return out
|
||||
|
||||
# Arquivo curto: um trecho só, o arquivo inteiro. É o resultado de busca
|
||||
# ideal — nada de emendar pedaços depois.
|
||||
if len(lines) <= WHOLE_FILE_MAX_LINES:
|
||||
return build(lines, 1, "file")
|
||||
|
||||
# Acha onde o corpo do tipo principal abre (primeira ida de 0 para 1 na
|
||||
# contagem de chaves) para separar o cabeçalho do arquivo.
|
||||
depth = 0
|
||||
body_start = None
|
||||
depths = []
|
||||
for i, line in enumerate(lines):
|
||||
clean = _strip_noise(line)
|
||||
opened = clean.count("{")
|
||||
closed = clean.count("}")
|
||||
depths.append(depth) # profundidade NO INÍCIO da linha
|
||||
if body_start is None and depth == 0 and opened > closed:
|
||||
body_start = i + 1
|
||||
depth += opened - closed
|
||||
|
||||
if body_start is None:
|
||||
# Sem corpo de tipo reconhecível (arquivo só de typealias, por ex).
|
||||
return build(lines, 1, "file")
|
||||
|
||||
chunks = build(lines[:body_start], 1, "header")
|
||||
|
||||
# Membros: declarações que começam com o corpo do tipo aberto (nível 1).
|
||||
starts = []
|
||||
for i in range(body_start, len(lines)):
|
||||
if depths[i] != 1:
|
||||
continue
|
||||
if not _DECL_RE.match(lines[i]):
|
||||
continue
|
||||
s = _preamble_start(lines, i)
|
||||
if s >= body_start and (not starts or s > starts[-1]):
|
||||
starts.append(s)
|
||||
|
||||
if not starts:
|
||||
chunks.extend(build(lines[body_start:], body_start + 1, "window"))
|
||||
return chunks
|
||||
|
||||
# O que sobra entre o cabeçalho e o primeiro membro (propriedades soltas,
|
||||
# por exemplo) é anexado ao cabeçalho como trecho próprio.
|
||||
if starts[0] > body_start:
|
||||
chunks.extend(build(lines[body_start:starts[0]], body_start + 1, "decl"))
|
||||
|
||||
for idx, s in enumerate(starts):
|
||||
e = starts[idx + 1] if idx + 1 < len(starts) else len(lines)
|
||||
chunks.extend(build(lines[s:e], s + 1, "decl"))
|
||||
|
||||
return _merge_small(chunks)
|
||||
Reference in New Issue
Block a user