ops: 新增 course-toolkit(YouTube 课程下载+转写+成册工具,含踩坑记录)

This commit is contained in:
xiaowu
2026-09-12 19:03:17 +08:00
parent 9a43ecfedf
commit 5629d57ea7
4 changed files with 214 additions and 0 deletions
+102
View File
@@ -0,0 +1,102 @@
#!/usr/bin/env python3
"""course-toolkit 辅助工具。
子命令:
manifest <rawfile> 将 index|id|title 转为 NN|id|slug|title
clean <outdir> transcripts/*.txt -> clean/*.txt(去时间轴、并段)
transcribe <wav> <out> 本地 ASR 转写(env: MODEL, LANG, THREADS)
build <outdir> 组装 md + README(guides/<NN>.txt 可选作中文导读)
"""
import os, re, sys
def slugify(s):
s = s.lower()
s = re.sub(r'[^a-z0-9]+', '-', s)
return re.sub(r'-+', '-', s).strip('-')[:60] or 'video'
def cmd_manifest(rawfile):
out = []
for line in open(rawfile, encoding='utf-8'):
line = line.strip()
if not line: continue
parts = line.split('|')
if len(parts) < 3: continue
idx, vid, title = parts[0], parts[1], '|'.join(parts[2:])
try: nn = f"{int(idx):02d}"
except Exception: nn = slugify(idx)[:2]
out.append(f"{nn}|{vid}|{slugify(title)}|{title}")
sys.stdout.write("\n".join(out) + "\n")
def cmd_clean(outdir):
tdir = os.path.join(outdir, 'transcripts'); cdir = os.path.join(outdir, 'clean')
os.makedirs(cdir, exist_ok=True)
for f in sorted(os.listdir(tdir)):
if not f.endswith('.txt'): continue
nn = f.split('-')[0]
lines = [l for l in open(os.path.join(tdir, f), encoding='utf-8').read().splitlines() if l.strip()]
txt = " ".join(re.sub(r'^\[\s*[\d.]+\s*-\s*[\d.]+\s*\]\s*', '', l) for l in lines)
open(os.path.join(cdir, f"{nn}.txt"), 'w', encoding='utf-8').write(txt)
print(f"[clean] {len(os.listdir(cdir))} files -> {cdir}")
def cmd_transcribe(wav, out):
os.environ.setdefault("HF_ENDPOINT", "https://hf-mirror.com")
os.environ.setdefault("HF_HUB_DISABLE_XET", "1")
from faster_whisper import WhisperModel
model = os.environ.get("MODEL", "base.en")
lang = os.environ.get("LANG", "en")
threads = int(os.environ.get("THREADS", "1"))
m = WhisperModel(model, device="cpu", compute_type="int8", cpu_threads=threads)
segs, info = m.transcribe(wav, language=lang, beam_size=1, vad_filter=True)
lines = [f"[{s.start:7.1f}-{s.end:7.1f}] {s.text.strip()}" for s in segs]
open(out, 'w', encoding='utf-8').write("\n".join(lines))
print(f"[transcribe] {out} segs={len(lines)} dur={info.duration:.0f}")
def cmd_build(outdir):
import subprocess, json
man = [l.split('|') for l in open(os.path.join(outdir, 'manifest.txt'), encoding='utf-8').read().splitlines() if l.strip()]
gdir = os.path.join(outdir, 'guides')
rows = []
for p in man:
NN, ID, SLUG = p[0], p[1], p[2]
title = p[3] if len(p) > 3 else SLUG
vid = f"videos/{NN}-{SLUG}.mp4"
txt = open(os.path.join(outdir, 'clean', f'{NN}.txt'), encoding='utf-8').read().strip()
gpath = os.path.join(gdir, f'{NN}.txt')
guide = open(gpath, encoding='utf-8').read().strip() if os.path.exists(gpath) else "> TODO:待补充中文导读。"
try:
dur = float(subprocess.check_output(["ffprobe", "-v", "error", "-show_entries", "format=duration", "-of", "csv=p=0", os.path.join(outdir, vid)]).decode())
mm, ss = int(dur // 60), int(dur % 60)
except Exception:
mm = ss = 0
md = f"""# {NN} · {title}
- **原始视频**:https://youtu.be/{ID}
- **时长**:{mm} 分 {ss} 秒
- **本地视频**:[{vid}]({vid})
## 🎯 本集要点(中文导读)
{guide}
## 📝 完整文稿(自动转写)
> 由本地 ASR 转写,未人工校对,供检索/精读使用。
{txt}
"""
open(os.path.join(outdir, f'{NN}-{SLUG}.md'), 'w', encoding='utf-8').write(md)
rows.append((NN, title, mm, ss, SLUG))
idx = "\n".join(f"| {n} | {t} | {m}:{s:02d} | [📄 笔记]({n}-{sl}.md) · [🎬 视频](videos/{n}-{sl}.mp4) |" for n, t, m, s, sl in rows)
open(os.path.join(outdir, 'README.md'), 'w', encoding='utf-8').write(
"# 课程教程\n\n> 由 course-toolkit 生成。\n\n| 集 | 标题 | 时长 | 链接 |\n|---|---|---|---|\n" + idx + "\n")
print(f"[build] {len(rows)} 篇 + README -> {outdir}")
if __name__ == '__main__':
if len(sys.argv) < 2:
print(__doc__); sys.exit(1)
cmd = sys.argv[1]
if cmd == 'manifest': cmd_manifest(sys.argv[2])
elif cmd == 'clean': cmd_clean(sys.argv[2])
elif cmd == 'transcribe': cmd_transcribe(sys.argv[2], sys.argv[3])
elif cmd == 'build': cmd_build(sys.argv[2])
else: print(__doc__); sys.exit(1)