dotfiles

Repositório dotfiles
git clone git://git.sivaldodavi.com/dotfiles.git
Log | Files | Refs | README

commit 330060ca89cd74562a11e2ced012ae53766ebe25
parent 6d66ec1ecb1d8f9c081a11eaf1e6f7cf034045e6
Author: Sivaldo <gxixtx@xsxixvxaxlxdxoxdxaxvxix.xcxoxm>
Date:   Sun, 13 Sep 2026 11:21:58 -0300

feat(scripts): adiciona script yt-sub para download/gerenciamento de legendas do YouTube

Diffstat:
A.local/bin/yt-sub | 148+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
1 file changed, 148 insertions(+), 0 deletions(-)

diff --git a/.local/bin/yt-sub b/.local/bin/yt-sub @@ -0,0 +1,148 @@ +import sys +import subprocess +import glob +import re +import argparse + +def clean_vtt(filepath, remove_time=True, words_per_paragraph=150): + """ + Limpa o arquivo VTT, remove duplicatas e junta o texto em blocos baseados + em uma quantidade limite de palavras por parágrafo. + """ + with open(filepath, "r", encoding="utf-8") as f: + lines = f.readlines() + + cleaned_content = [] + last_text_line = "" + + # Padrões Regex para identificar lixo e tempos + time_pattern = re.compile(r'-->') + header_pattern = re.compile(r'^WEBVTT|^Kind:|^Language:|^Align:|Style:') + number_pattern = re.compile(r'^\d+$') + tags_pattern = re.compile(r'<[^>]+>') # Remove tags como <00:00:01.000> ou <c> + + for line in lines: + original_line = line.strip() + + # Pula linhas vazias + if not original_line: + if not remove_time: + cleaned_content.append("\n") + continue + + # Ignora cabeçalhos e números de indexação + if header_pattern.search(original_line) or number_pattern.match(original_line): + if not remove_time: + cleaned_content.append(original_line + "\n") + continue + + # Lida com o tempo + if time_pattern.search(original_line): + if not remove_time: + cleaned_content.append(original_line + "\n") + continue + + # Extrai apenas o texto limpo + text = tags_pattern.sub('', original_line).strip() + + if not text: + continue + + # Lógica para evitar as duplicações do YouTube (legendas rolantes) + if text != last_text_line: + if remove_time: + # Se for sem tempo, junta tudo na mesma linha separando por espaço + cleaned_content.append(text + " ") + else: + # Se for com tempo, mantém a quebra de linha do VTT + cleaned_content.append(text + "\n") + + last_text_line = text + + final_text = "".join(cleaned_content) + + # Se estiver no modo texto (sem tempo), organiza em parágrafos por contagem de palavras + if remove_time: + # Remove espaços duplos que possam ter sobrado + final_text = re.sub(r'\s+', ' ', final_text).strip() + + # Divide o texto todo em uma lista de palavras + palavras = final_text.split() + + # Agrupa as palavras em blocos (parágrafos) do tamanho definido + paragrafos = [] + for i in range(0, len(palavras), words_per_paragraph): + bloco_palavras = palavras[i : i + words_per_paragraph] + paragrafo = " ".join(bloco_palavras) + paragrafos.append(paragrafo) + + # Junta os parágrafos com duas quebras de linha entre eles + final_text = "\n\n".join(paragrafos) + + # Sobrescreve o arquivo com a versão final + with open(filepath, "w", encoding="utf-8") as f: + f.write(final_text) + +def process_url(url, keep_time, words_per_paragraph): + print(f"\n--- Processando: {url} ---") + + # Lista arquivos antes para saber qual foi baixado agora + vtts_antes = set(glob.glob("*.vtt")) + + # Comando base (silencioso) + comando_base = [ + "yt-dlp", + "--quiet", + "--no-warnings", + "--write-subs", + "--write-auto-subs", + "--sub-format", "vtt", + "--skip-download", + url + ] + + # 1. TENTA BAIXAR EM PORTUGUÊS + print("Buscando legenda em PT...") + comando_pt = comando_base[:5] + ["--sub-langs", "pt"] + comando_base[5:] + subprocess.run(comando_pt) + + vtts_depois = set(glob.glob("*.vtt")) + novos_vtts = vtts_depois - vtts_antes + + # 2. SE NÃO ACHOU, TENTA EM INGLÊS + if not novos_vtts: + print("Legenda em PT não encontrada. Tentando EN...") + comando_en = comando_base[:5] + ["--sub-langs", "en"] + comando_base[5:] + subprocess.run(comando_en) + + vtts_depois = set(glob.glob("*.vtt")) + novos_vtts = vtts_depois - vtts_antes + + # SE CONTINUAR SEM ACHAR, ABORTA + if not novos_vtts: + print("Nenhuma legenda encontrada (nem PT, nem EN).") + return + + # Limpa os arquivos recém-baixados + for arquivo in novos_vtts: + print(f"Limpando e formatando arquivo em blocos de {words_per_paragraph} palavras: {arquivo}") + clean_vtt(arquivo, remove_time=not keep_time, words_per_paragraph=words_per_paragraph) + + print("Processo concluído com sucesso!") + +def main(): + parser = argparse.ArgumentParser(description="Baixa legendas (PT ou EN) e formata em grandes blocos de texto.") + + parser.add_argument("urls", nargs="+", help="Uma ou mais URLs do YouTube.") + parser.add_argument("-t", "--keep-time", action="store_true", + help="Mantém as tags de tempo (formato VTT original, sem juntar os blocos).") + parser.add_argument("-w", "--words", type=int, default=150, + help="Quantidade de palavras por parágrafo (Padrão: 150).") + + args = parser.parse_args() + + for url in args.urls: + process_url(url, keep_time=args.keep_time, words_per_paragraph=args.words) + +if __name__ == "__main__": + main()