-
Notifications
You must be signed in to change notification settings - Fork 3.3k
Add Grok STT as the default transcription provider #130
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
base: main
Are you sure you want to change the base?
Changes from all commits
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -1 +1,4 @@ | ||
| # Either key is enough. If both are set, Grok STT is used unless you pass | ||
| # --provider elevenlabs. | ||
| XAI_API_KEY= | ||
| ELEVENLABS_API_KEY= |
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -1,13 +1,20 @@ | ||
| """Transcribe a video with ElevenLabs Scribe. | ||
| """Transcribe a video with Grok STT or ElevenLabs Scribe. | ||
|
|
||
| Extracts mono 16kHz audio via ffmpeg, uploads to Scribe with verbatim + | ||
| diarize + audio events + word-level timestamps, writes the full response | ||
| to <edit_dir>/transcripts/<video_stem>.json. | ||
| Accepts either XAI_API_KEY or ELEVENLABS_API_KEY. An existing ElevenLabs-only | ||
| .env keeps working with no flags. If both keys are set, Grok STT is used | ||
| unless you pass --provider elevenlabs. | ||
|
|
||
| Extracts mono 16kHz audio via ffmpeg, uploads with diarization + word-level | ||
| timestamps + filler-word retention, writes a Scribe-shaped transcript to | ||
| <edit_dir>/transcripts/<video_stem>.json so pack/render/timeline_view stay | ||
| provider-agnostic. | ||
|
|
||
| Cached: if the output file already exists, the upload is skipped. | ||
|
|
||
| Usage: | ||
| python helpers/transcribe.py <video_path> | ||
| python helpers/transcribe.py <video_path> --provider grok | ||
| python helpers/transcribe.py <video_path> --provider elevenlabs | ||
| python helpers/transcribe.py <video_path> --edit-dir /custom/edit | ||
| python helpers/transcribe.py <video_path> --language en | ||
| python helpers/transcribe.py <video_path> --num-speakers 2 | ||
|
|
@@ -27,22 +34,53 @@ | |
| import requests | ||
|
|
||
|
|
||
| GROK_STT_URL = "https://api.x.ai/v1/stt" | ||
| SCRIBE_URL = "https://api.elevenlabs.io/v1/speech-to-text" | ||
| PROVIDERS = ("grok", "elevenlabs") | ||
| KEY_FOR_PROVIDER = { | ||
| "grok": "XAI_API_KEY", | ||
| "elevenlabs": "ELEVENLABS_API_KEY", | ||
| } | ||
|
|
||
|
|
||
| def load_api_key() -> str: | ||
| def _read_key(name: str) -> str: | ||
| for candidate in [Path(__file__).resolve().parent.parent / ".env", Path(".env")]: | ||
| if candidate.exists(): | ||
| for line in candidate.read_text().splitlines(): | ||
| line = line.strip() | ||
| if not line or line.startswith("#") or "=" not in line: | ||
| continue | ||
| k, v = line.split("=", 1) | ||
| if k.strip() == "ELEVENLABS_API_KEY": | ||
| return v.strip().strip('"').strip("'") | ||
| v = os.environ.get("ELEVENLABS_API_KEY", "") | ||
| if not candidate.exists(): | ||
| continue | ||
| for line in candidate.read_text().splitlines(): | ||
| line = line.strip() | ||
| if not line or line.startswith("#") or "=" not in line: | ||
| continue | ||
| k, v = line.split("=", 1) | ||
| if k.strip() != name: | ||
| continue | ||
| val = v.strip().strip('"').strip("'") | ||
| if val: | ||
| return val | ||
| return os.environ.get(name, "").strip() | ||
|
|
||
|
|
||
| def resolve_provider(explicit: str | None) -> str: | ||
| if explicit: | ||
| if explicit not in PROVIDERS: | ||
| sys.exit(f"unknown --provider {explicit!r} (want {'|'.join(PROVIDERS)})") | ||
| return explicit | ||
| if _read_key("XAI_API_KEY"): | ||
| return "grok" | ||
| if _read_key("ELEVENLABS_API_KEY"): | ||
| return "elevenlabs" | ||
| sys.exit( | ||
| "need XAI_API_KEY or ELEVENLABS_API_KEY in .env or the environment " | ||
| "(existing ElevenLabs-only setups keep working; pass --provider to force one)" | ||
| ) | ||
|
|
||
|
|
||
| def load_api_key(provider: str | None = None) -> str: | ||
| provider = resolve_provider(provider) | ||
|
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. P2: Adding a second provider makes the existing filename-only transcript cache return stale cross-provider data. A user who previously transcribed with ElevenLabs and now runs the new grok default will get their old Scribe transcripts back (and vice versa) because the cache checks only that Prompt for AI agents |
||
| name = KEY_FOR_PROVIDER[provider] | ||
| v = _read_key(name) | ||
| if not v: | ||
| sys.exit("ELEVENLABS_API_KEY not found in .env or environment") | ||
| sys.exit(f"{name} not found in .env or environment") | ||
| return v | ||
|
|
||
|
|
||
|
|
@@ -55,6 +93,75 @@ def extract_audio(video_path: Path, dest: Path) -> None: | |
| subprocess.run(cmd, check=True, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) | ||
|
|
||
|
|
||
| def grok_to_scribe(payload: dict) -> dict: | ||
| """Normalize Grok STT words into the Scribe-shaped schema pack/render expect. | ||
|
|
||
| Grok returns {text, start, end, speaker?} with no type / spacing / audio_event | ||
| entries. We add type=word, map speaker→speaker_id, and synthesize spacing | ||
| tokens from inter-word gaps so silence-aware packing still works. | ||
| """ | ||
| words_out: list[dict] = [] | ||
| prev_end: float | None = None | ||
| for w in payload.get("words") or []: | ||
| start = w.get("start") | ||
| if start is None: | ||
| continue | ||
| end = w.get("end", start) | ||
| if prev_end is not None and start > prev_end: | ||
| words_out.append({ | ||
| "type": "spacing", | ||
| "text": " ", | ||
| "start": prev_end, | ||
| "end": start, | ||
| }) | ||
| entry: dict = { | ||
| "type": "word", | ||
| "text": w.get("text") or "", | ||
| "start": start, | ||
| "end": end, | ||
| } | ||
| speaker = w.get("speaker") | ||
| if speaker is not None: | ||
| entry["speaker_id"] = f"speaker_{speaker}" | ||
| words_out.append(entry) | ||
| prev_end = end | ||
| return { | ||
| "text": payload.get("text", ""), | ||
| "language": payload.get("language"), | ||
| "duration": payload.get("duration"), | ||
| "provider": "grok", | ||
| "words": words_out, | ||
| } | ||
|
|
||
|
|
||
| def call_grok_stt( | ||
| audio_path: Path, | ||
| api_key: str, | ||
| language: str | None = None, | ||
| ) -> dict: | ||
| # Verbatim + fillers: do not set format=true (that runs ITN). | ||
| form: list[tuple[str, str]] = [ | ||
| ("diarize", "true"), | ||
| ("filler_words", "true"), | ||
| ] | ||
| if language: | ||
| form.append(("language", language)) | ||
|
|
||
| with open(audio_path, "rb") as f: | ||
| resp = requests.post( | ||
| GROK_STT_URL, | ||
| headers={"Authorization": f"Bearer {api_key}"}, | ||
| data=form, | ||
| files={"file": (audio_path.name, f, "audio/wav")}, | ||
| timeout=1800, | ||
| ) | ||
|
|
||
| if resp.status_code != 200: | ||
| raise RuntimeError(f"Grok STT returned {resp.status_code}: {resp.text[:500]}") | ||
|
|
||
| return grok_to_scribe(resp.json()) | ||
|
|
||
|
|
||
| def call_scribe( | ||
| audio_path: Path, | ||
| api_key: str, | ||
|
|
@@ -84,7 +191,10 @@ def call_scribe( | |
| if resp.status_code != 200: | ||
| raise RuntimeError(f"Scribe returned {resp.status_code}: {resp.text[:500]}") | ||
|
|
||
| return resp.json() | ||
| payload = resp.json() | ||
| if isinstance(payload, dict): | ||
| payload.setdefault("provider", "elevenlabs") | ||
| return payload | ||
|
|
||
|
|
||
| def transcribe_one( | ||
|
|
@@ -94,11 +204,13 @@ def transcribe_one( | |
| language: str | None = None, | ||
| num_speakers: int | None = None, | ||
| verbose: bool = True, | ||
| provider: str | None = None, | ||
| ) -> Path: | ||
| """Transcribe a single video. Returns path to transcript JSON. | ||
|
|
||
| Cached: returns existing path immediately if the transcript already exists. | ||
| """ | ||
| provider = resolve_provider(provider) | ||
| transcripts_dir = edit_dir / "transcripts" | ||
| transcripts_dir.mkdir(parents=True, exist_ok=True) | ||
| out_path = transcripts_dir / f"{video.stem}.json" | ||
|
|
@@ -109,7 +221,7 @@ def transcribe_one( | |
| return out_path | ||
|
|
||
| if verbose: | ||
| print(f" extracting audio from {video.name}", flush=True) | ||
| print(f" extracting audio from {video.name} [{provider}]", flush=True) | ||
|
|
||
| t0 = time.time() | ||
| with tempfile.TemporaryDirectory() as tmp: | ||
|
|
@@ -118,7 +230,10 @@ def transcribe_one( | |
| size_mb = audio.stat().st_size / (1024 * 1024) | ||
| if verbose: | ||
| print(f" uploading {video.stem}.wav ({size_mb:.1f} MB)", flush=True) | ||
| payload = call_scribe(audio, api_key, language, num_speakers) | ||
| if provider == "grok": | ||
| payload = call_grok_stt(audio, api_key, language) | ||
| else: | ||
| payload = call_scribe(audio, api_key, language, num_speakers) | ||
|
|
||
| out_path.write_text(json.dumps(payload, indent=2)) | ||
| dt = time.time() - t0 | ||
|
|
@@ -133,14 +248,20 @@ def transcribe_one( | |
|
|
||
|
|
||
| def main() -> None: | ||
| ap = argparse.ArgumentParser(description="Transcribe a video with ElevenLabs Scribe") | ||
| ap = argparse.ArgumentParser(description="Transcribe a video with Grok STT or ElevenLabs Scribe") | ||
| ap.add_argument("video", type=Path, help="Path to video file") | ||
| ap.add_argument( | ||
| "--edit-dir", | ||
| type=Path, | ||
| default=None, | ||
| help="Edit output directory (default: <video_parent>/edit)", | ||
| ) | ||
| ap.add_argument( | ||
| "--provider", | ||
| choices=PROVIDERS, | ||
| default=None, | ||
| help="STT backend. Default: grok if XAI_API_KEY is set, else elevenlabs.", | ||
| ) | ||
| ap.add_argument( | ||
| "--language", | ||
| type=str, | ||
|
|
@@ -151,7 +272,7 @@ def main() -> None: | |
| "--num-speakers", | ||
| type=int, | ||
| default=None, | ||
| help="Optional number of speakers when known. Improves diarization accuracy.", | ||
| help="Optional speaker count (ElevenLabs only). Improves diarization accuracy.", | ||
| ) | ||
| args = ap.parse_args() | ||
|
|
||
|
|
@@ -160,14 +281,16 @@ def main() -> None: | |
| sys.exit(f"video not found: {video}") | ||
|
|
||
| edit_dir = (args.edit_dir or (video.parent / "edit")).resolve() | ||
| api_key = load_api_key() | ||
| provider = resolve_provider(args.provider) | ||
| api_key = load_api_key(provider) | ||
|
|
||
| transcribe_one( | ||
| video=video, | ||
| edit_dir=edit_dir, | ||
| api_key=api_key, | ||
| language=args.language, | ||
| num_speakers=args.num_speakers, | ||
| provider=provider, | ||
| ) | ||
|
|
||
|
|
||
|
|
||
Uh oh!
There was an error while loading. Please reload this page.
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
P2: When a caller supplies an ElevenLabs
api_keybut omitsprovider, this ambient-key lookup can select Grok and send the wrong credential, causing authentication failures. Preserve the existing direct-call behavior or require and validate the provider alongside the key.Prompt for AI agents