commit 330060ca89cd74562a11e2ced012ae53766ebe25
parent 6d66ec1ecb1d8f9c081a11eaf1e6f7cf034045e6
Author: Sivaldo <gxixtx@xsxixvxaxlxdxoxdxaxvxix.xcxoxm>
Date: Sun, 13 Sep 2026 11:21:58 -0300
feat(scripts): adiciona script yt-sub para download/gerenciamento de legendas do YouTube
Diffstat:
| A | .local/bin/yt-sub | | | 148 | +++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++ |
1 file changed, 148 insertions(+), 0 deletions(-)
diff --git a/.local/bin/yt-sub b/.local/bin/yt-sub
@@ -0,0 +1,148 @@
+import sys
+import subprocess
+import glob
+import re
+import argparse
+
+def clean_vtt(filepath, remove_time=True, words_per_paragraph=150):
+ """
+ Limpa o arquivo VTT, remove duplicatas e junta o texto em blocos baseados
+ em uma quantidade limite de palavras por parágrafo.
+ """
+ with open(filepath, "r", encoding="utf-8") as f:
+ lines = f.readlines()
+
+ cleaned_content = []
+ last_text_line = ""
+
+ # Padrões Regex para identificar lixo e tempos
+ time_pattern = re.compile(r'-->')
+ header_pattern = re.compile(r'^WEBVTT|^Kind:|^Language:|^Align:|Style:')
+ number_pattern = re.compile(r'^\d+$')
+ tags_pattern = re.compile(r'<[^>]+>') # Remove tags como <00:00:01.000> ou <c>
+
+ for line in lines:
+ original_line = line.strip()
+
+ # Pula linhas vazias
+ if not original_line:
+ if not remove_time:
+ cleaned_content.append("\n")
+ continue
+
+ # Ignora cabeçalhos e números de indexação
+ if header_pattern.search(original_line) or number_pattern.match(original_line):
+ if not remove_time:
+ cleaned_content.append(original_line + "\n")
+ continue
+
+ # Lida com o tempo
+ if time_pattern.search(original_line):
+ if not remove_time:
+ cleaned_content.append(original_line + "\n")
+ continue
+
+ # Extrai apenas o texto limpo
+ text = tags_pattern.sub('', original_line).strip()
+
+ if not text:
+ continue
+
+ # Lógica para evitar as duplicações do YouTube (legendas rolantes)
+ if text != last_text_line:
+ if remove_time:
+ # Se for sem tempo, junta tudo na mesma linha separando por espaço
+ cleaned_content.append(text + " ")
+ else:
+ # Se for com tempo, mantém a quebra de linha do VTT
+ cleaned_content.append(text + "\n")
+
+ last_text_line = text
+
+ final_text = "".join(cleaned_content)
+
+ # Se estiver no modo texto (sem tempo), organiza em parágrafos por contagem de palavras
+ if remove_time:
+ # Remove espaços duplos que possam ter sobrado
+ final_text = re.sub(r'\s+', ' ', final_text).strip()
+
+ # Divide o texto todo em uma lista de palavras
+ palavras = final_text.split()
+
+ # Agrupa as palavras em blocos (parágrafos) do tamanho definido
+ paragrafos = []
+ for i in range(0, len(palavras), words_per_paragraph):
+ bloco_palavras = palavras[i : i + words_per_paragraph]
+ paragrafo = " ".join(bloco_palavras)
+ paragrafos.append(paragrafo)
+
+ # Junta os parágrafos com duas quebras de linha entre eles
+ final_text = "\n\n".join(paragrafos)
+
+ # Sobrescreve o arquivo com a versão final
+ with open(filepath, "w", encoding="utf-8") as f:
+ f.write(final_text)
+
+def process_url(url, keep_time, words_per_paragraph):
+ print(f"\n--- Processando: {url} ---")
+
+ # Lista arquivos antes para saber qual foi baixado agora
+ vtts_antes = set(glob.glob("*.vtt"))
+
+ # Comando base (silencioso)
+ comando_base = [
+ "yt-dlp",
+ "--quiet",
+ "--no-warnings",
+ "--write-subs",
+ "--write-auto-subs",
+ "--sub-format", "vtt",
+ "--skip-download",
+ url
+ ]
+
+ # 1. TENTA BAIXAR EM PORTUGUÊS
+ print("Buscando legenda em PT...")
+ comando_pt = comando_base[:5] + ["--sub-langs", "pt"] + comando_base[5:]
+ subprocess.run(comando_pt)
+
+ vtts_depois = set(glob.glob("*.vtt"))
+ novos_vtts = vtts_depois - vtts_antes
+
+ # 2. SE NÃO ACHOU, TENTA EM INGLÊS
+ if not novos_vtts:
+ print("Legenda em PT não encontrada. Tentando EN...")
+ comando_en = comando_base[:5] + ["--sub-langs", "en"] + comando_base[5:]
+ subprocess.run(comando_en)
+
+ vtts_depois = set(glob.glob("*.vtt"))
+ novos_vtts = vtts_depois - vtts_antes
+
+ # SE CONTINUAR SEM ACHAR, ABORTA
+ if not novos_vtts:
+ print("Nenhuma legenda encontrada (nem PT, nem EN).")
+ return
+
+ # Limpa os arquivos recém-baixados
+ for arquivo in novos_vtts:
+ print(f"Limpando e formatando arquivo em blocos de {words_per_paragraph} palavras: {arquivo}")
+ clean_vtt(arquivo, remove_time=not keep_time, words_per_paragraph=words_per_paragraph)
+
+ print("Processo concluído com sucesso!")
+
+def main():
+ parser = argparse.ArgumentParser(description="Baixa legendas (PT ou EN) e formata em grandes blocos de texto.")
+
+ parser.add_argument("urls", nargs="+", help="Uma ou mais URLs do YouTube.")
+ parser.add_argument("-t", "--keep-time", action="store_true",
+ help="Mantém as tags de tempo (formato VTT original, sem juntar os blocos).")
+ parser.add_argument("-w", "--words", type=int, default=150,
+ help="Quantidade de palavras por parágrafo (Padrão: 150).")
+
+ args = parser.parse_args()
+
+ for url in args.urls:
+ process_url(url, keep_time=args.keep_time, words_per_paragraph=args.words)
+
+if __name__ == "__main__":
+ main()