#!/usr/bin/env python3 """course-toolkit 辅助工具。 子命令: manifest 将 index|id|title 转为 NN|id|slug|title clean transcripts/*.txt -> clean/*.txt(去时间轴、并段) transcribe 本地 ASR 转写(env: MODEL, LANG, THREADS) build 组装 md + README(guides/.txt 可选作中文导读) """ import os, re, sys def slugify(s): s = s.lower() s = re.sub(r'[^a-z0-9]+', '-', s) return re.sub(r'-+', '-', s).strip('-')[:60] or 'video' def cmd_manifest(rawfile): out = [] for line in open(rawfile, encoding='utf-8'): line = line.strip() if not line: continue parts = line.split('|') if len(parts) < 3: continue idx, vid, title = parts[0], parts[1], '|'.join(parts[2:]) try: nn = f"{int(idx):02d}" except Exception: nn = slugify(idx)[:2] out.append(f"{nn}|{vid}|{slugify(title)}|{title}") sys.stdout.write("\n".join(out) + "\n") def cmd_clean(outdir): tdir = os.path.join(outdir, 'transcripts'); cdir = os.path.join(outdir, 'clean') os.makedirs(cdir, exist_ok=True) for f in sorted(os.listdir(tdir)): if not f.endswith('.txt'): continue nn = f.split('-')[0] lines = [l for l in open(os.path.join(tdir, f), encoding='utf-8').read().splitlines() if l.strip()] txt = " ".join(re.sub(r'^\[\s*[\d.]+\s*-\s*[\d.]+\s*\]\s*', '', l) for l in lines) open(os.path.join(cdir, f"{nn}.txt"), 'w', encoding='utf-8').write(txt) print(f"[clean] {len(os.listdir(cdir))} files -> {cdir}") def cmd_transcribe(wav, out): os.environ.setdefault("HF_ENDPOINT", "https://hf-mirror.com") os.environ.setdefault("HF_HUB_DISABLE_XET", "1") from faster_whisper import WhisperModel model = os.environ.get("MODEL", "base.en") lang = os.environ.get("LANG", "en") threads = int(os.environ.get("THREADS", "1")) m = WhisperModel(model, device="cpu", compute_type="int8", cpu_threads=threads) segs, info = m.transcribe(wav, language=lang, beam_size=1, vad_filter=True) lines = [f"[{s.start:7.1f}-{s.end:7.1f}] {s.text.strip()}" for s in segs] open(out, 'w', encoding='utf-8').write("\n".join(lines)) print(f"[transcribe] {out} segs={len(lines)} dur={info.duration:.0f}") def cmd_build(outdir): import subprocess, json man = [l.split('|') for l in open(os.path.join(outdir, 'manifest.txt'), encoding='utf-8').read().splitlines() if l.strip()] gdir = os.path.join(outdir, 'guides') rows = [] for p in man: NN, ID, SLUG = p[0], p[1], p[2] title = p[3] if len(p) > 3 else SLUG vid = f"videos/{NN}-{SLUG}.mp4" txt = open(os.path.join(outdir, 'clean', f'{NN}.txt'), encoding='utf-8').read().strip() gpath = os.path.join(gdir, f'{NN}.txt') guide = open(gpath, encoding='utf-8').read().strip() if os.path.exists(gpath) else "> TODO:待补充中文导读。" try: dur = float(subprocess.check_output(["ffprobe", "-v", "error", "-show_entries", "format=duration", "-of", "csv=p=0", os.path.join(outdir, vid)]).decode()) mm, ss = int(dur // 60), int(dur % 60) except Exception: mm = ss = 0 md = f"""# {NN} · {title} - **原始视频**:https://youtu.be/{ID} - **时长**:{mm} 分 {ss} 秒 - **本地视频**:[{vid}]({vid}) ## 🎯 本集要点(中文导读) {guide} ## 📝 完整文稿(自动转写) > 由本地 ASR 转写,未人工校对,供检索/精读使用。 {txt} """ open(os.path.join(outdir, f'{NN}-{SLUG}.md'), 'w', encoding='utf-8').write(md) rows.append((NN, title, mm, ss, SLUG)) idx = "\n".join(f"| {n} | {t} | {m}:{s:02d} | [📄 笔记]({n}-{sl}.md) · [🎬 视频](videos/{n}-{sl}.mp4) |" for n, t, m, s, sl in rows) open(os.path.join(outdir, 'README.md'), 'w', encoding='utf-8').write( "# 课程教程\n\n> 由 course-toolkit 生成。\n\n| 集 | 标题 | 时长 | 链接 |\n|---|---|---|---|\n" + idx + "\n") print(f"[build] {len(rows)} 篇 + README -> {outdir}") if __name__ == '__main__': if len(sys.argv) < 2: print(__doc__); sys.exit(1) cmd = sys.argv[1] if cmd == 'manifest': cmd_manifest(sys.argv[2]) elif cmd == 'clean': cmd_clean(sys.argv[2]) elif cmd == 'transcribe': cmd_transcribe(sys.argv[2], sys.argv[3]) elif cmd == 'build': cmd_build(sys.argv[2]) else: print(__doc__); sys.exit(1)