- Adicionado estrutura completa do projeto - Configurado MCP server para Premiere Pro - Adicionado documentação e skills - Configurado Gitignore para o projeto
235 lines
8.1 KiB
Python
235 lines
8.1 KiB
Python
"""Corte de arquivos Python em trechos por declaração, com faixa de linhas.
|
|
|
|
Irmão do `swift_chunker`, para o índice do sistema legado (`code/`). O
|
|
problema aqui é diferente: o legado NÃO segue "uma classe por arquivo" —
|
|
`ai_analysis.py` e `app.py` têm mais de 5000 linhas cada, com dezenas de
|
|
funções soltas. Cortar isso em janelas cegas de 60 linhas produz trechos que
|
|
começam no meio de uma função e não dizem a que função pertencem.
|
|
|
|
Python facilita: o nível de indentação já marca a estrutura, então um bloco
|
|
`def`/`class` vai da sua linha de cabeçalho até a próxima linha não vazia com
|
|
indentação menor ou igual.
|
|
"""
|
|
|
|
import os
|
|
import re
|
|
|
|
# Início de um bloco nomeado, em qualquer nível de indentação.
|
|
_BLOCK_RE = re.compile(
|
|
r"^(?P<indent>[ \t]*)(?:async\s+)?(?P<kw>def|class)\s+(?P<name>[A-Za-z_]\w*)"
|
|
)
|
|
# Decoradores e comentários que pertencem ao bloco logo abaixo.
|
|
_PREAMBLE_RE = re.compile(r"^[ \t]*(?:@|#)")
|
|
|
|
# Abre uma docstring: prefixo opcional de string (r/b/u/f) seguido das
|
|
# aspas. O prefixo é capturado como grupo para ser removido com precisão —
|
|
# um `strip()` de conjunto de caracteres comeria o "F" de "FCPXML parser".
|
|
_DOCSTRING_RE = re.compile(r'^(?:[rRbBuUfF]{0,2})("""|\'\'\'|"|\')(?P<rest>.*)$')
|
|
|
|
WHOLE_FILE_MAX_LINES = 120
|
|
TARGET_CHARS = 2000
|
|
MAX_CHARS = 5000
|
|
|
|
|
|
def _blocks_at(lines, start, end, indent):
|
|
"""Fronteiras dos blocos `def`/`class` no nível de indentação dado.
|
|
|
|
Devolve lista de (início, fim) meio-aberta, já incluindo decoradores e
|
|
comentários que precedem cada bloco.
|
|
"""
|
|
starts = []
|
|
for i in range(start, end):
|
|
m = _BLOCK_RE.match(lines[i])
|
|
if not m or len(m.group("indent").expandtabs(4)) != indent:
|
|
continue
|
|
j = i
|
|
while j > start and _PREAMBLE_RE.match(lines[j - 1]):
|
|
j -= 1
|
|
if not starts or j > starts[-1]:
|
|
starts.append(j)
|
|
out = []
|
|
for idx, s in enumerate(starts):
|
|
e = starts[idx + 1] if idx + 1 < len(starts) else end
|
|
out.append((s, e))
|
|
return out
|
|
|
|
|
|
def _body_indent(lines, start, end):
|
|
"""Indentação do corpo do bloco que começa em `start`."""
|
|
for i in range(start, end):
|
|
stripped = lines[i].strip()
|
|
if not stripped or _PREAMBLE_RE.match(lines[i]):
|
|
continue
|
|
if _BLOCK_RE.match(lines[i]):
|
|
continue
|
|
return len(lines[i]) - len(lines[i].lstrip())
|
|
return 4
|
|
|
|
|
|
def _split_long(chunk_lines, start_line, max_chars=MAX_CHARS):
|
|
"""Divide um trecho grande em pedaços, sempre em fronteira de linha."""
|
|
out, buf, buf_start, size = [], [], start_line, 0
|
|
for offset, line in enumerate(chunk_lines):
|
|
if buf and size + len(line) + 1 > max_chars:
|
|
out.append((buf, buf_start))
|
|
buf, buf_start, size = [], start_line + offset, 0
|
|
buf.append(line)
|
|
size += len(line) + 1
|
|
if buf:
|
|
out.append((buf, buf_start))
|
|
return out
|
|
|
|
|
|
def _symbols_of(text, file_stem, module_parts):
|
|
"""Nomes buscáveis do trecho.
|
|
|
|
O nome do arquivo e as pastas do módulo entram sempre: no legado é comum
|
|
procurar por `exporters/edl` ou `media_probe`, que são caminho, não
|
|
identificador declarado.
|
|
"""
|
|
names = {file_stem} | set(module_parts)
|
|
names.update(m.group("name") for m in _BLOCK_RE.finditer(text))
|
|
return " ".join(sorted(n for n in names if n))
|
|
|
|
|
|
def _docstring_first_line(lines, start, end):
|
|
for i in range(start, min(end, start + 40)):
|
|
s = lines[i].strip()
|
|
if not s or s.startswith(("#", "@")):
|
|
continue
|
|
m = _DOCSTRING_RE.match(s)
|
|
if m:
|
|
quote = m.group(1)
|
|
text = m.group("rest")
|
|
if text.endswith(quote):
|
|
text = text[:-len(quote)]
|
|
text = text.strip()
|
|
if text:
|
|
return text
|
|
# Docstring cuja primeira linha é só as aspas: pega a seguinte.
|
|
if i + 1 < end:
|
|
nxt = lines[i + 1].strip()
|
|
if nxt.endswith(quote):
|
|
nxt = nxt[:-len(quote)]
|
|
return nxt.strip() or None
|
|
return None
|
|
return None
|
|
return None
|
|
|
|
|
|
def file_facts(text, rel_path):
|
|
"""Tipo principal, símbolos públicos e resumo — alimenta `file_index`."""
|
|
lines = text.splitlines()
|
|
stem = os.path.splitext(os.path.basename(rel_path))[0]
|
|
|
|
main_type = None
|
|
public_symbols = []
|
|
for s, e in _blocks_at(lines, 0, len(lines), 0):
|
|
m = _BLOCK_RE.match(lines[s]) or next(
|
|
(_BLOCK_RE.match(lines[k]) for k in range(s, e)
|
|
if _BLOCK_RE.match(lines[k])), None)
|
|
if not m:
|
|
continue
|
|
name = m.group("name")
|
|
if not name.startswith("_"):
|
|
public_symbols.append(name)
|
|
if main_type is None and m.group("kw") == "class":
|
|
main_type = name
|
|
|
|
# Resumo: docstring do módulo; se não houver, a do primeiro bloco.
|
|
summary = _docstring_first_line(lines, 0, len(lines))
|
|
if not summary:
|
|
blocks = _blocks_at(lines, 0, len(lines), 0)
|
|
if blocks:
|
|
s, e = blocks[0]
|
|
body = _body_indent(lines, s, e)
|
|
summary = _docstring_first_line(lines, s + 1, e)
|
|
del body
|
|
|
|
return {
|
|
"main_type": main_type or stem,
|
|
"summary": summary,
|
|
"public_symbols": sorted(set(public_symbols)),
|
|
"n_lines": len(lines),
|
|
}
|
|
|
|
|
|
def _merge_small(chunks):
|
|
"""Junta blocos vizinhos curtos até `TARGET_CHARS`.
|
|
|
|
Sem isso cada função de três linhas (`_timebase`, `_actual_fps`…) vira um
|
|
trecho próprio: muitos embeddings e pouco contexto em cada resultado.
|
|
"""
|
|
merged = []
|
|
for c in chunks:
|
|
prev = merged[-1] if merged else None
|
|
if (prev is not None
|
|
and prev["end_line"] + 1 == c["start_line"]
|
|
and prev["kind"] == c["kind"] == "decl"
|
|
and len(prev["content"]) + len(c["content"]) + 1 <= TARGET_CHARS):
|
|
prev["content"] += "\n" + c["content"]
|
|
prev["end_line"] = c["end_line"]
|
|
prev["symbols"] = " ".join(
|
|
sorted(set(prev["symbols"].split()) | set(c["symbols"].split()))
|
|
)
|
|
else:
|
|
merged.append(dict(c))
|
|
return merged
|
|
|
|
|
|
def chunk_python(text, rel_path):
|
|
"""Divide um arquivo Python em trechos com faixa de linhas e símbolos.
|
|
|
|
Linhas são 1-indexadas e inclusivas nas duas pontas, iguais ao que o
|
|
`Read(offset=…, limit=…)` do agente espera.
|
|
"""
|
|
lines = text.splitlines()
|
|
if not lines:
|
|
return []
|
|
|
|
stem = os.path.splitext(os.path.basename(rel_path))[0]
|
|
module_parts = [p for p in os.path.dirname(rel_path).split(os.sep)
|
|
if p not in ("", "code")]
|
|
|
|
def build(start, end, kind):
|
|
out = []
|
|
for piece, piece_start in _split_long(lines[start:end], start + 1):
|
|
body = "\n".join(piece).strip()
|
|
if not body:
|
|
continue
|
|
out.append({
|
|
"content": body,
|
|
"start_line": piece_start,
|
|
"end_line": piece_start + len(piece) - 1,
|
|
"symbols": _symbols_of(body, stem, module_parts),
|
|
"kind": kind,
|
|
})
|
|
return out
|
|
|
|
if len(lines) <= WHOLE_FILE_MAX_LINES:
|
|
return build(0, len(lines), "file")
|
|
|
|
top = _blocks_at(lines, 0, len(lines), 0)
|
|
if not top:
|
|
return build(0, len(lines), "window")
|
|
|
|
# Cabeçalho: docstring do módulo + imports, antes do primeiro bloco.
|
|
chunks = build(0, top[0][0], "header") if top[0][0] > 0 else []
|
|
|
|
for s, e in top:
|
|
size = sum(len(lines[i]) + 1 for i in range(s, e))
|
|
if size <= TARGET_CHARS:
|
|
chunks.extend(build(s, e, "decl"))
|
|
continue
|
|
# Bloco grande (classe com muitos métodos): desce um nível e corta
|
|
# por método, mantendo a assinatura da classe no primeiro pedaço.
|
|
inner = _blocks_at(lines, s + 1, e, _body_indent(lines, s, e))
|
|
if not inner:
|
|
chunks.extend(build(s, e, "decl"))
|
|
continue
|
|
chunks.extend(build(s, inner[0][0], "decl"))
|
|
for si, ei in inner:
|
|
chunks.extend(build(si, ei, "decl"))
|
|
|
|
return _merge_small(chunks)
|