完成任务16:编写 04_长期发展/环境保护.md
This commit is contained in:
175
skills/video-use/helpers/transcribe.py
Normal file
175
skills/video-use/helpers/transcribe.py
Normal file
@@ -0,0 +1,175 @@
|
||||
"""Transcribe a video with ElevenLabs Scribe.
|
||||
|
||||
Extracts mono 16kHz audio via ffmpeg, uploads to Scribe with verbatim +
|
||||
diarize + audio events + word-level timestamps, writes the full response
|
||||
to <edit_dir>/transcripts/<video_stem>.json.
|
||||
|
||||
Cached: if the output file already exists, the upload is skipped.
|
||||
|
||||
Usage:
|
||||
python helpers/transcribe.py <video_path>
|
||||
python helpers/transcribe.py <video_path> --edit-dir /custom/edit
|
||||
python helpers/transcribe.py <video_path> --language en
|
||||
python helpers/transcribe.py <video_path> --num-speakers 2
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
import requests
|
||||
|
||||
|
||||
SCRIBE_URL = "https://api.elevenlabs.io/v1/speech-to-text"
|
||||
|
||||
|
||||
def load_api_key() -> str:
|
||||
for candidate in [Path(__file__).resolve().parent.parent / ".env", Path(".env")]:
|
||||
if candidate.exists():
|
||||
for line in candidate.read_text().splitlines():
|
||||
line = line.strip()
|
||||
if not line or line.startswith("#") or "=" not in line:
|
||||
continue
|
||||
k, v = line.split("=", 1)
|
||||
if k.strip() == "ELEVENLABS_API_KEY":
|
||||
return v.strip().strip('"').strip("'")
|
||||
v = os.environ.get("ELEVENLABS_API_KEY", "")
|
||||
if not v:
|
||||
sys.exit("ELEVENLABS_API_KEY not found in .env or environment")
|
||||
return v
|
||||
|
||||
|
||||
def extract_audio(video_path: Path, dest: Path) -> None:
|
||||
cmd = [
|
||||
"ffmpeg", "-y", "-i", str(video_path),
|
||||
"-vn", "-ac", "1", "-ar", "16000", "-c:a", "pcm_s16le",
|
||||
str(dest),
|
||||
]
|
||||
subprocess.run(cmd, check=True, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
|
||||
|
||||
|
||||
def call_scribe(
|
||||
audio_path: Path,
|
||||
api_key: str,
|
||||
language: str | None = None,
|
||||
num_speakers: int | None = None,
|
||||
) -> dict:
|
||||
data: dict[str, str] = {
|
||||
"model_id": "scribe_v1",
|
||||
"diarize": "true",
|
||||
"tag_audio_events": "true",
|
||||
"timestamps_granularity": "word",
|
||||
}
|
||||
if language:
|
||||
data["language_code"] = language
|
||||
if num_speakers:
|
||||
data["num_speakers"] = str(num_speakers)
|
||||
|
||||
with open(audio_path, "rb") as f:
|
||||
resp = requests.post(
|
||||
SCRIBE_URL,
|
||||
headers={"xi-api-key": api_key},
|
||||
files={"file": (audio_path.name, f, "audio/wav")},
|
||||
data=data,
|
||||
timeout=1800,
|
||||
)
|
||||
|
||||
if resp.status_code != 200:
|
||||
raise RuntimeError(f"Scribe returned {resp.status_code}: {resp.text[:500]}")
|
||||
|
||||
return resp.json()
|
||||
|
||||
|
||||
def transcribe_one(
|
||||
video: Path,
|
||||
edit_dir: Path,
|
||||
api_key: str,
|
||||
language: str | None = None,
|
||||
num_speakers: int | None = None,
|
||||
verbose: bool = True,
|
||||
) -> Path:
|
||||
"""Transcribe a single video. Returns path to transcript JSON.
|
||||
|
||||
Cached: returns existing path immediately if the transcript already exists.
|
||||
"""
|
||||
transcripts_dir = edit_dir / "transcripts"
|
||||
transcripts_dir.mkdir(parents=True, exist_ok=True)
|
||||
out_path = transcripts_dir / f"{video.stem}.json"
|
||||
|
||||
if out_path.exists():
|
||||
if verbose:
|
||||
print(f"cached: {out_path.name}")
|
||||
return out_path
|
||||
|
||||
if verbose:
|
||||
print(f" extracting audio from {video.name}", flush=True)
|
||||
|
||||
t0 = time.time()
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
audio = Path(tmp) / f"{video.stem}.wav"
|
||||
extract_audio(video, audio)
|
||||
size_mb = audio.stat().st_size / (1024 * 1024)
|
||||
if verbose:
|
||||
print(f" uploading {video.stem}.wav ({size_mb:.1f} MB)", flush=True)
|
||||
payload = call_scribe(audio, api_key, language, num_speakers)
|
||||
|
||||
out_path.write_text(json.dumps(payload, indent=2))
|
||||
dt = time.time() - t0
|
||||
|
||||
if verbose:
|
||||
kb = out_path.stat().st_size / 1024
|
||||
print(f" saved: {out_path.name} ({kb:.1f} KB) in {dt:.1f}s")
|
||||
if isinstance(payload, dict) and "words" in payload:
|
||||
print(f" words: {len(payload['words'])}")
|
||||
|
||||
return out_path
|
||||
|
||||
|
||||
def main() -> None:
|
||||
ap = argparse.ArgumentParser(description="Transcribe a video with ElevenLabs Scribe")
|
||||
ap.add_argument("video", type=Path, help="Path to video file")
|
||||
ap.add_argument(
|
||||
"--edit-dir",
|
||||
type=Path,
|
||||
default=None,
|
||||
help="Edit output directory (default: <video_parent>/edit)",
|
||||
)
|
||||
ap.add_argument(
|
||||
"--language",
|
||||
type=str,
|
||||
default=None,
|
||||
help="Optional ISO language code (e.g., 'en'). Omit to auto-detect.",
|
||||
)
|
||||
ap.add_argument(
|
||||
"--num-speakers",
|
||||
type=int,
|
||||
default=None,
|
||||
help="Optional number of speakers when known. Improves diarization accuracy.",
|
||||
)
|
||||
args = ap.parse_args()
|
||||
|
||||
video = args.video.resolve()
|
||||
if not video.exists():
|
||||
sys.exit(f"video not found: {video}")
|
||||
|
||||
edit_dir = (args.edit_dir or (video.parent / "edit")).resolve()
|
||||
api_key = load_api_key()
|
||||
|
||||
transcribe_one(
|
||||
video=video,
|
||||
edit_dir=edit_dir,
|
||||
api_key=api_key,
|
||||
language=args.language,
|
||||
num_speakers=args.num_speakers,
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user