Remove Anthropic API call from transcribe.py - agent generates Whisper prompt itself
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Sonnet 4.6
parent
2c21bc0488
commit
699e9960ce
+15
-15
@@ -115,30 +115,30 @@ Skip this step entirely if `detect` returned zero `video` files.
|
|||||||
|
|
||||||
Video and audio files cannot be read directly. Transcribe them to text first, then treat the transcripts as doc files in Step 3.
|
Video and audio files cannot be read directly. Transcribe them to text first, then treat the transcripts as doc files in Step 3.
|
||||||
|
|
||||||
**Strategy:** Run non-video semantic extraction first (Step 3B) to get god nodes, use those to build a domain hint for Whisper, then transcribe. This keeps the prompt relevant without guessing the corpus topic from filenames.
|
**Strategy:** Read the god nodes from the detect output or analysis file. You are already a language model - write a one-sentence domain hint yourself from those labels. Then pass it to Whisper as the initial prompt. No separate API call needed.
|
||||||
|
|
||||||
**However**, if the corpus has *only* video files and no other docs/code, skip the god-node step and transcribe with the generic fallback prompt immediately.
|
**However**, if the corpus has *only* video files and no other docs/code, use the generic fallback prompt: `"Use proper punctuation and paragraph breaks."`
|
||||||
|
|
||||||
**Transcription command:**
|
**Step 1 - Write the Whisper prompt yourself.**
|
||||||
|
|
||||||
|
Read the top god node labels from detect output or analysis, then compose a short domain hint sentence, for example:
|
||||||
|
|
||||||
|
- Labels: `transformer, attention, encoder, decoder` -> `"Machine learning research on transformer architectures and attention mechanisms. Use proper punctuation and paragraph breaks."`
|
||||||
|
- Labels: `kubernetes, deployment, pod, helm` -> `"DevOps discussion about Kubernetes deployments and Helm charts. Use proper punctuation and paragraph breaks."`
|
||||||
|
|
||||||
|
Set it as `GRAPHIFY_WHISPER_PROMPT` in the environment before running the transcription command.
|
||||||
|
|
||||||
|
**Step 2 - Transcribe:**
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
$(cat graphify-out/.graphify_python) -c "
|
$(cat graphify-out/.graphify_python) -c "
|
||||||
import json
|
import json, os
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from graphify.transcribe import build_whisper_prompt, transcribe_all
|
from graphify.transcribe import transcribe_all
|
||||||
|
|
||||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text())
|
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text())
|
||||||
video_files = detect.get('files', {}).get('video', [])
|
video_files = detect.get('files', {}).get('video', [])
|
||||||
|
prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and paragraph breaks.')
|
||||||
# Try to load god nodes from a previous partial run or pass [] if not yet available
|
|
||||||
try:
|
|
||||||
analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text())
|
|
||||||
god_nodes = analysis.get('god_nodes', [])
|
|
||||||
except Exception:
|
|
||||||
god_nodes = []
|
|
||||||
|
|
||||||
prompt = build_whisper_prompt(god_nodes)
|
|
||||||
print(f'Whisper prompt: {prompt}')
|
|
||||||
|
|
||||||
transcript_paths = transcribe_all(video_files, initial_prompt=prompt)
|
transcript_paths = transcribe_all(video_files, initial_prompt=prompt)
|
||||||
print(json.dumps(transcript_paths))
|
print(json.dumps(transcript_paths))
|
||||||
|
|||||||
+15
-15
@@ -115,30 +115,30 @@ Skip this step entirely if `detect` returned zero `video` files.
|
|||||||
|
|
||||||
Video and audio files cannot be read directly. Transcribe them to text first, then treat the transcripts as doc files in Step 3.
|
Video and audio files cannot be read directly. Transcribe them to text first, then treat the transcripts as doc files in Step 3.
|
||||||
|
|
||||||
**Strategy:** Run non-video semantic extraction first (Step 3B) to get god nodes, use those to build a domain hint for Whisper, then transcribe. This keeps the prompt relevant without guessing the corpus topic from filenames.
|
**Strategy:** Read the god nodes from the detect output or analysis file. You are already a language model - write a one-sentence domain hint yourself from those labels. Then pass it to Whisper as the initial prompt. No separate API call needed.
|
||||||
|
|
||||||
**However**, if the corpus has *only* video files and no other docs/code, skip the god-node step and transcribe with the generic fallback prompt immediately.
|
**However**, if the corpus has *only* video files and no other docs/code, use the generic fallback prompt: `"Use proper punctuation and paragraph breaks."`
|
||||||
|
|
||||||
**Transcription command:**
|
**Step 1 - Write the Whisper prompt yourself.**
|
||||||
|
|
||||||
|
Read the top god node labels from detect output or analysis, then compose a short domain hint sentence, for example:
|
||||||
|
|
||||||
|
- Labels: `transformer, attention, encoder, decoder` -> `"Machine learning research on transformer architectures and attention mechanisms. Use proper punctuation and paragraph breaks."`
|
||||||
|
- Labels: `kubernetes, deployment, pod, helm` -> `"DevOps discussion about Kubernetes deployments and Helm charts. Use proper punctuation and paragraph breaks."`
|
||||||
|
|
||||||
|
Set it as `GRAPHIFY_WHISPER_PROMPT` in the environment before running the transcription command.
|
||||||
|
|
||||||
|
**Step 2 - Transcribe:**
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
$(cat graphify-out/.graphify_python) -c "
|
$(cat graphify-out/.graphify_python) -c "
|
||||||
import json
|
import json, os
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from graphify.transcribe import build_whisper_prompt, transcribe_all
|
from graphify.transcribe import transcribe_all
|
||||||
|
|
||||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text())
|
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text())
|
||||||
video_files = detect.get('files', {}).get('video', [])
|
video_files = detect.get('files', {}).get('video', [])
|
||||||
|
prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and paragraph breaks.')
|
||||||
# Try to load god nodes from a previous partial run or pass [] if not yet available
|
|
||||||
try:
|
|
||||||
analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text())
|
|
||||||
god_nodes = analysis.get('god_nodes', [])
|
|
||||||
except Exception:
|
|
||||||
god_nodes = []
|
|
||||||
|
|
||||||
prompt = build_whisper_prompt(god_nodes)
|
|
||||||
print(f'Whisper prompt: {prompt}')
|
|
||||||
|
|
||||||
transcript_paths = transcribe_all(video_files, initial_prompt=prompt)
|
transcript_paths = transcribe_all(video_files, initial_prompt=prompt)
|
||||||
print(json.dumps(transcript_paths))
|
print(json.dumps(transcript_paths))
|
||||||
|
|||||||
+15
-14
@@ -114,29 +114,30 @@ Skip this step entirely if `detect` returned zero `video` files.
|
|||||||
|
|
||||||
Video and audio files cannot be read directly. Transcribe them to text first, then treat the transcripts as doc files in Step 3.
|
Video and audio files cannot be read directly. Transcribe them to text first, then treat the transcripts as doc files in Step 3.
|
||||||
|
|
||||||
**Strategy:** Run non-video semantic extraction first (Step 3B) to get god nodes, use those to build a domain hint for Whisper, then transcribe. This keeps the prompt relevant without guessing the corpus topic from filenames.
|
**Strategy:** Read the god nodes from the detect output or analysis file. You are already a language model — write a one-sentence domain hint yourself from those labels. Then pass it to Whisper as the initial prompt. No separate API call needed.
|
||||||
|
|
||||||
**However**, if the corpus has *only* video files and no other docs/code, skip the god-node step and transcribe with the generic fallback prompt immediately.
|
**However**, if the corpus has *only* video files and no other docs/code, use the generic fallback prompt: `"Use proper punctuation and paragraph breaks."`
|
||||||
|
|
||||||
**Transcription command:**
|
**Step 1 - Write the Whisper prompt yourself.**
|
||||||
|
|
||||||
|
Read the top god node labels from detect output or analysis, then compose a short domain hint sentence, for example:
|
||||||
|
|
||||||
|
- Labels: `transformer, attention, encoder, decoder` → `"Machine learning research on transformer architectures and attention mechanisms. Use proper punctuation and paragraph breaks."`
|
||||||
|
- Labels: `kubernetes, deployment, pod, helm` → `"DevOps discussion about Kubernetes deployments and Helm charts. Use proper punctuation and paragraph breaks."`
|
||||||
|
|
||||||
|
Set it as `GRAPHIFY_WHISPER_PROMPT` in the environment before running the transcription command.
|
||||||
|
|
||||||
|
**Step 2 - Transcribe:**
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
$(cat graphify-out/.graphify_python) -c "
|
$(cat graphify-out/.graphify_python) -c "
|
||||||
import json
|
import json, os
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from graphify.transcribe import build_whisper_prompt, transcribe_all
|
from graphify.transcribe import transcribe_all
|
||||||
|
|
||||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text())
|
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text())
|
||||||
video_files = detect.get('files', {}).get('video', [])
|
video_files = detect.get('files', {}).get('video', [])
|
||||||
|
prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and paragraph breaks.')
|
||||||
try:
|
|
||||||
analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text())
|
|
||||||
god_nodes = analysis.get('god_nodes', [])
|
|
||||||
except Exception:
|
|
||||||
god_nodes = []
|
|
||||||
|
|
||||||
prompt = build_whisper_prompt(god_nodes)
|
|
||||||
print(f'Whisper prompt: {prompt}')
|
|
||||||
|
|
||||||
transcript_paths = transcribe_all(video_files, initial_prompt=prompt)
|
transcript_paths = transcribe_all(video_files, initial_prompt=prompt)
|
||||||
print(json.dumps(transcript_paths))
|
print(json.dumps(transcript_paths))
|
||||||
|
|||||||
+15
-15
@@ -117,30 +117,30 @@ Skip this step entirely if `detect` returned zero `video` files.
|
|||||||
|
|
||||||
Video and audio files cannot be read directly. Transcribe them to text first, then treat the transcripts as doc files in Step 3.
|
Video and audio files cannot be read directly. Transcribe them to text first, then treat the transcripts as doc files in Step 3.
|
||||||
|
|
||||||
**Strategy:** Run non-video semantic extraction first (Step 3B) to get god nodes, use those to build a domain hint for Whisper, then transcribe. This keeps the prompt relevant without guessing the corpus topic from filenames.
|
**Strategy:** Read the god nodes from the detect output or analysis file. You are already a language model - write a one-sentence domain hint yourself from those labels. Then pass it to Whisper as the initial prompt. No separate API call needed.
|
||||||
|
|
||||||
**However**, if the corpus has *only* video files and no other docs/code, skip the god-node step and transcribe with the generic fallback prompt immediately.
|
**However**, if the corpus has *only* video files and no other docs/code, use the generic fallback prompt: `"Use proper punctuation and paragraph breaks."`
|
||||||
|
|
||||||
**Transcription command:**
|
**Step 1 - Write the Whisper prompt yourself.**
|
||||||
|
|
||||||
|
Read the top god node labels from detect output or analysis, then compose a short domain hint sentence, for example:
|
||||||
|
|
||||||
|
- Labels: `transformer, attention, encoder, decoder` -> `"Machine learning research on transformer architectures and attention mechanisms. Use proper punctuation and paragraph breaks."`
|
||||||
|
- Labels: `kubernetes, deployment, pod, helm` -> `"DevOps discussion about Kubernetes deployments and Helm charts. Use proper punctuation and paragraph breaks."`
|
||||||
|
|
||||||
|
Set it as `GRAPHIFY_WHISPER_PROMPT` in the environment before running the transcription command.
|
||||||
|
|
||||||
|
**Step 2 - Transcribe:**
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
$(cat graphify-out/.graphify_python) -c "
|
$(cat graphify-out/.graphify_python) -c "
|
||||||
import json
|
import json, os
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from graphify.transcribe import build_whisper_prompt, transcribe_all
|
from graphify.transcribe import transcribe_all
|
||||||
|
|
||||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text())
|
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text())
|
||||||
video_files = detect.get('files', {}).get('video', [])
|
video_files = detect.get('files', {}).get('video', [])
|
||||||
|
prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and paragraph breaks.')
|
||||||
# Try to load god nodes from a previous partial run or pass [] if not yet available
|
|
||||||
try:
|
|
||||||
analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text())
|
|
||||||
god_nodes = analysis.get('god_nodes', [])
|
|
||||||
except Exception:
|
|
||||||
god_nodes = []
|
|
||||||
|
|
||||||
prompt = build_whisper_prompt(god_nodes)
|
|
||||||
print(f'Whisper prompt: {prompt}')
|
|
||||||
|
|
||||||
transcript_paths = transcribe_all(video_files, initial_prompt=prompt)
|
transcript_paths = transcribe_all(video_files, initial_prompt=prompt)
|
||||||
print(json.dumps(transcript_paths))
|
print(json.dumps(transcript_paths))
|
||||||
|
|||||||
+15
-15
@@ -115,30 +115,30 @@ Skip this step entirely if `detect` returned zero `video` files.
|
|||||||
|
|
||||||
Video and audio files cannot be read directly. Transcribe them to text first, then treat the transcripts as doc files in Step 3.
|
Video and audio files cannot be read directly. Transcribe them to text first, then treat the transcripts as doc files in Step 3.
|
||||||
|
|
||||||
**Strategy:** Run non-video semantic extraction first (Step 3B) to get god nodes, use those to build a domain hint for Whisper, then transcribe. This keeps the prompt relevant without guessing the corpus topic from filenames.
|
**Strategy:** Read the god nodes from the detect output or analysis file. You are already a language model - write a one-sentence domain hint yourself from those labels. Then pass it to Whisper as the initial prompt. No separate API call needed.
|
||||||
|
|
||||||
**However**, if the corpus has *only* video files and no other docs/code, skip the god-node step and transcribe with the generic fallback prompt immediately.
|
**However**, if the corpus has *only* video files and no other docs/code, use the generic fallback prompt: `"Use proper punctuation and paragraph breaks."`
|
||||||
|
|
||||||
**Transcription command:**
|
**Step 1 - Write the Whisper prompt yourself.**
|
||||||
|
|
||||||
|
Read the top god node labels from detect output or analysis, then compose a short domain hint sentence, for example:
|
||||||
|
|
||||||
|
- Labels: `transformer, attention, encoder, decoder` -> `"Machine learning research on transformer architectures and attention mechanisms. Use proper punctuation and paragraph breaks."`
|
||||||
|
- Labels: `kubernetes, deployment, pod, helm` -> `"DevOps discussion about Kubernetes deployments and Helm charts. Use proper punctuation and paragraph breaks."`
|
||||||
|
|
||||||
|
Set it as `GRAPHIFY_WHISPER_PROMPT` in the environment before running the transcription command.
|
||||||
|
|
||||||
|
**Step 2 - Transcribe:**
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
$(cat graphify-out/.graphify_python) -c "
|
$(cat graphify-out/.graphify_python) -c "
|
||||||
import json
|
import json, os
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from graphify.transcribe import build_whisper_prompt, transcribe_all
|
from graphify.transcribe import transcribe_all
|
||||||
|
|
||||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text())
|
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text())
|
||||||
video_files = detect.get('files', {}).get('video', [])
|
video_files = detect.get('files', {}).get('video', [])
|
||||||
|
prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and paragraph breaks.')
|
||||||
# Try to load god nodes from a previous partial run or pass [] if not yet available
|
|
||||||
try:
|
|
||||||
analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text())
|
|
||||||
god_nodes = analysis.get('god_nodes', [])
|
|
||||||
except Exception:
|
|
||||||
god_nodes = []
|
|
||||||
|
|
||||||
prompt = build_whisper_prompt(god_nodes)
|
|
||||||
print(f'Whisper prompt: {prompt}')
|
|
||||||
|
|
||||||
transcript_paths = transcribe_all(video_files, initial_prompt=prompt)
|
transcript_paths = transcribe_all(video_files, initial_prompt=prompt)
|
||||||
print(json.dumps(transcript_paths))
|
print(json.dumps(transcript_paths))
|
||||||
|
|||||||
+15
-15
@@ -115,30 +115,30 @@ Skip this step entirely if `detect` returned zero `video` files.
|
|||||||
|
|
||||||
Video and audio files cannot be read directly. Transcribe them to text first, then treat the transcripts as doc files in Step 3.
|
Video and audio files cannot be read directly. Transcribe them to text first, then treat the transcripts as doc files in Step 3.
|
||||||
|
|
||||||
**Strategy:** Run non-video semantic extraction first (Step 3B) to get god nodes, use those to build a domain hint for Whisper, then transcribe. This keeps the prompt relevant without guessing the corpus topic from filenames.
|
**Strategy:** Read the god nodes from the detect output or analysis file. You are already a language model - write a one-sentence domain hint yourself from those labels. Then pass it to Whisper as the initial prompt. No separate API call needed.
|
||||||
|
|
||||||
**However**, if the corpus has *only* video files and no other docs/code, skip the god-node step and transcribe with the generic fallback prompt immediately.
|
**However**, if the corpus has *only* video files and no other docs/code, use the generic fallback prompt: `"Use proper punctuation and paragraph breaks."`
|
||||||
|
|
||||||
**Transcription command:**
|
**Step 1 - Write the Whisper prompt yourself.**
|
||||||
|
|
||||||
|
Read the top god node labels from detect output or analysis, then compose a short domain hint sentence, for example:
|
||||||
|
|
||||||
|
- Labels: `transformer, attention, encoder, decoder` -> `"Machine learning research on transformer architectures and attention mechanisms. Use proper punctuation and paragraph breaks."`
|
||||||
|
- Labels: `kubernetes, deployment, pod, helm` -> `"DevOps discussion about Kubernetes deployments and Helm charts. Use proper punctuation and paragraph breaks."`
|
||||||
|
|
||||||
|
Set it as `GRAPHIFY_WHISPER_PROMPT` in the environment before running the transcription command.
|
||||||
|
|
||||||
|
**Step 2 - Transcribe:**
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
$(cat graphify-out/.graphify_python) -c "
|
$(cat graphify-out/.graphify_python) -c "
|
||||||
import json
|
import json, os
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from graphify.transcribe import build_whisper_prompt, transcribe_all
|
from graphify.transcribe import transcribe_all
|
||||||
|
|
||||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text())
|
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text())
|
||||||
video_files = detect.get('files', {}).get('video', [])
|
video_files = detect.get('files', {}).get('video', [])
|
||||||
|
prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and paragraph breaks.')
|
||||||
# Try to load god nodes from a previous partial run or pass [] if not yet available
|
|
||||||
try:
|
|
||||||
analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text())
|
|
||||||
god_nodes = analysis.get('god_nodes', [])
|
|
||||||
except Exception:
|
|
||||||
god_nodes = []
|
|
||||||
|
|
||||||
prompt = build_whisper_prompt(god_nodes)
|
|
||||||
print(f'Whisper prompt: {prompt}')
|
|
||||||
|
|
||||||
transcript_paths = transcribe_all(video_files, initial_prompt=prompt)
|
transcript_paths = transcribe_all(video_files, initial_prompt=prompt)
|
||||||
print(json.dumps(transcript_paths))
|
print(json.dumps(transcript_paths))
|
||||||
|
|||||||
+15
-15
@@ -114,30 +114,30 @@ Skip this step entirely if `detect` returned zero `video` files.
|
|||||||
|
|
||||||
Video and audio files cannot be read directly. Transcribe them to text first, then treat the transcripts as doc files in Step 3.
|
Video and audio files cannot be read directly. Transcribe them to text first, then treat the transcripts as doc files in Step 3.
|
||||||
|
|
||||||
**Strategy:** Run non-video semantic extraction first (Step 3B) to get god nodes, use those to build a domain hint for Whisper, then transcribe. This keeps the prompt relevant without guessing the corpus topic from filenames.
|
**Strategy:** Read the god nodes from the detect output or analysis file. You are already a language model - write a one-sentence domain hint yourself from those labels. Then pass it to Whisper as the initial prompt. No separate API call needed.
|
||||||
|
|
||||||
**However**, if the corpus has *only* video files and no other docs/code, skip the god-node step and transcribe with the generic fallback prompt immediately.
|
**However**, if the corpus has *only* video files and no other docs/code, use the generic fallback prompt: `"Use proper punctuation and paragraph breaks."`
|
||||||
|
|
||||||
**Transcription command:**
|
**Step 1 - Write the Whisper prompt yourself.**
|
||||||
|
|
||||||
|
Read the top god node labels from detect output or analysis, then compose a short domain hint sentence, for example:
|
||||||
|
|
||||||
|
- Labels: `transformer, attention, encoder, decoder` -> `"Machine learning research on transformer architectures and attention mechanisms. Use proper punctuation and paragraph breaks."`
|
||||||
|
- Labels: `kubernetes, deployment, pod, helm` -> `"DevOps discussion about Kubernetes deployments and Helm charts. Use proper punctuation and paragraph breaks."`
|
||||||
|
|
||||||
|
Set it as `GRAPHIFY_WHISPER_PROMPT` in the environment before running the transcription command.
|
||||||
|
|
||||||
|
**Step 2 - Transcribe:**
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
$(cat graphify-out/.graphify_python) -c "
|
$(cat graphify-out/.graphify_python) -c "
|
||||||
import json
|
import json, os
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from graphify.transcribe import build_whisper_prompt, transcribe_all
|
from graphify.transcribe import transcribe_all
|
||||||
|
|
||||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text())
|
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text())
|
||||||
video_files = detect.get('files', {}).get('video', [])
|
video_files = detect.get('files', {}).get('video', [])
|
||||||
|
prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and paragraph breaks.')
|
||||||
# Try to load god nodes from a previous partial run or pass [] if not yet available
|
|
||||||
try:
|
|
||||||
analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text())
|
|
||||||
god_nodes = analysis.get('god_nodes', [])
|
|
||||||
except Exception:
|
|
||||||
god_nodes = []
|
|
||||||
|
|
||||||
prompt = build_whisper_prompt(god_nodes)
|
|
||||||
print(f'Whisper prompt: {prompt}')
|
|
||||||
|
|
||||||
transcript_paths = transcribe_all(video_files, initial_prompt=prompt)
|
transcript_paths = transcribe_all(video_files, initial_prompt=prompt)
|
||||||
print(json.dumps(transcript_paths))
|
print(json.dumps(transcript_paths))
|
||||||
|
|||||||
+15
-14
@@ -107,29 +107,30 @@ Skip this step entirely if `detect` returned zero `video` files.
|
|||||||
|
|
||||||
Video and audio files cannot be read directly. Transcribe them to text first, then treat the transcripts as doc files in Step 3.
|
Video and audio files cannot be read directly. Transcribe them to text first, then treat the transcripts as doc files in Step 3.
|
||||||
|
|
||||||
**Strategy:** Run non-video semantic extraction first (Step 3B) to get god nodes, use those to build a domain hint for Whisper, then transcribe. This keeps the prompt relevant without guessing the corpus topic from filenames.
|
**Strategy:** Read the god nodes from the detect output or analysis file. You are already a language model - write a one-sentence domain hint yourself from those labels. Then pass it to Whisper as the initial prompt. No separate API call needed.
|
||||||
|
|
||||||
**However**, if the corpus has *only* video files and no other docs/code, skip the god-node step and transcribe with the generic fallback prompt immediately.
|
**However**, if the corpus has *only* video files and no other docs/code, use the generic fallback prompt: `"Use proper punctuation and paragraph breaks."`
|
||||||
|
|
||||||
**Transcription command (PowerShell):**
|
**Step 1 - Write the Whisper prompt yourself.**
|
||||||
|
|
||||||
|
Read the top god node labels from detect output or analysis, then compose a short domain hint sentence, for example:
|
||||||
|
|
||||||
|
- Labels: `transformer, attention, encoder, decoder` -> `"Machine learning research on transformer architectures and attention mechanisms. Use proper punctuation and paragraph breaks."`
|
||||||
|
- Labels: `kubernetes, deployment, pod, helm` -> `"DevOps discussion about Kubernetes deployments and Helm charts. Use proper punctuation and paragraph breaks."`
|
||||||
|
|
||||||
|
Set it as `$env:GRAPHIFY_WHISPER_PROMPT` before running the transcription command.
|
||||||
|
|
||||||
|
**Step 2 - Transcribe (PowerShell):**
|
||||||
|
|
||||||
```powershell
|
```powershell
|
||||||
& (Get-Content graphify-out\.graphify_python) -c "
|
& (Get-Content graphify-out\.graphify_python) -c "
|
||||||
import json
|
import json, os
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from graphify.transcribe import build_whisper_prompt, transcribe_all
|
from graphify.transcribe import transcribe_all
|
||||||
|
|
||||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text())
|
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text())
|
||||||
video_files = detect.get('files', {}).get('video', [])
|
video_files = detect.get('files', {}).get('video', [])
|
||||||
|
prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and paragraph breaks.')
|
||||||
try:
|
|
||||||
analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text())
|
|
||||||
god_nodes = analysis.get('god_nodes', [])
|
|
||||||
except Exception:
|
|
||||||
god_nodes = []
|
|
||||||
|
|
||||||
prompt = build_whisper_prompt(god_nodes)
|
|
||||||
print(f'Whisper prompt: {prompt}')
|
|
||||||
|
|
||||||
transcript_paths = transcribe_all(video_files, initial_prompt=prompt)
|
transcript_paths = transcribe_all(video_files, initial_prompt=prompt)
|
||||||
print(json.dumps(transcript_paths))
|
print(json.dumps(transcript_paths))
|
||||||
|
|||||||
+16
-15
@@ -119,30 +119,31 @@ Skip this step entirely if `detect` returned zero `video` files.
|
|||||||
|
|
||||||
Video and audio files cannot be read directly. Transcribe them to text first, then treat the transcripts as doc files in Step 3.
|
Video and audio files cannot be read directly. Transcribe them to text first, then treat the transcripts as doc files in Step 3.
|
||||||
|
|
||||||
**Strategy:** Run non-video semantic extraction first (Step 3B) to get god nodes, use those to build a domain hint for Whisper, then transcribe. This keeps the prompt relevant without guessing the corpus topic from filenames.
|
**Strategy:** Read the god nodes from `graphify-out/.graphify_detect.json` (or the analysis file if it exists from a previous run). You are already a language model — write a one-sentence domain hint yourself from those labels. Then pass it to Whisper as the initial prompt. No separate API call needed.
|
||||||
|
|
||||||
**However**, if the corpus has *only* video files and no other docs/code, skip the god-node step and transcribe with the generic fallback prompt immediately.
|
**However**, if the corpus has *only* video files and no other docs/code, use the generic fallback prompt: `"Use proper punctuation and paragraph breaks."`
|
||||||
|
|
||||||
**Transcription command:**
|
**Step 1 - Write the Whisper prompt yourself.**
|
||||||
|
|
||||||
|
Read the top god node labels from detect output or analysis, then compose a short domain hint sentence, for example:
|
||||||
|
|
||||||
|
- Labels: `transformer, attention, encoder, decoder` → `"Machine learning research on transformer architectures and attention mechanisms. Use proper punctuation and paragraph breaks."`
|
||||||
|
- Labels: `kubernetes, deployment, pod, helm` → `"DevOps discussion about Kubernetes deployments and Helm charts. Use proper punctuation and paragraph breaks."`
|
||||||
|
|
||||||
|
Set it as `WHISPER_PROMPT` to use in the next command.
|
||||||
|
|
||||||
|
**Step 2 - Transcribe:**
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
|
GRAPHIFY_WHISPER_MODEL=base # or whatever --whisper-model the user passed
|
||||||
$(cat graphify-out/.graphify_python) -c "
|
$(cat graphify-out/.graphify_python) -c "
|
||||||
import json
|
import json, os
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from graphify.transcribe import build_whisper_prompt, transcribe_all
|
from graphify.transcribe import transcribe_all
|
||||||
|
|
||||||
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text())
|
detect = json.loads(Path('graphify-out/.graphify_detect.json').read_text())
|
||||||
video_files = detect.get('files', {}).get('video', [])
|
video_files = detect.get('files', {}).get('video', [])
|
||||||
|
prompt = os.environ.get('GRAPHIFY_WHISPER_PROMPT', 'Use proper punctuation and paragraph breaks.')
|
||||||
# Try to load god nodes from a previous partial run or pass [] if not yet available
|
|
||||||
try:
|
|
||||||
analysis = json.loads(Path('graphify-out/.graphify_analysis.json').read_text())
|
|
||||||
god_nodes = analysis.get('god_nodes', [])
|
|
||||||
except Exception:
|
|
||||||
god_nodes = []
|
|
||||||
|
|
||||||
prompt = build_whisper_prompt(god_nodes)
|
|
||||||
print(f'Whisper prompt: {prompt}')
|
|
||||||
|
|
||||||
transcript_paths = transcribe_all(video_files, initial_prompt=prompt)
|
transcript_paths = transcribe_all(video_files, initial_prompt=prompt)
|
||||||
print(json.dumps(transcript_paths))
|
print(json.dumps(transcript_paths))
|
||||||
|
|||||||
+4
-24
@@ -91,14 +91,14 @@ def download_audio(url: str, output_dir: Path) -> Path:
|
|||||||
def build_whisper_prompt(god_nodes: list[dict]) -> str:
|
def build_whisper_prompt(god_nodes: list[dict]) -> str:
|
||||||
"""Build a domain hint for Whisper from god nodes extracted from the corpus.
|
"""Build a domain hint for Whisper from god nodes extracted from the corpus.
|
||||||
|
|
||||||
Takes the top god nodes (most connected concepts) already extracted from
|
Formats the top god node labels into a topic string for Whisper.
|
||||||
non-video files and asks the LLM to summarise them into a one-sentence
|
The coding agent (Claude Code, Codex, etc.) generates the actual one-sentence
|
||||||
speech-to-text hint. Falls back to a generic prompt if no nodes available.
|
domain hint from these labels and passes it via GRAPHIFY_WHISPER_PROMPT or
|
||||||
|
as initial_prompt — no separate API call needed here.
|
||||||
"""
|
"""
|
||||||
if not god_nodes:
|
if not god_nodes:
|
||||||
return _FALLBACK_PROMPT
|
return _FALLBACK_PROMPT
|
||||||
|
|
||||||
# Use env override if set
|
|
||||||
override = os.environ.get("GRAPHIFY_WHISPER_PROMPT")
|
override = os.environ.get("GRAPHIFY_WHISPER_PROMPT")
|
||||||
if override:
|
if override:
|
||||||
return override
|
return override
|
||||||
@@ -107,26 +107,6 @@ def build_whisper_prompt(god_nodes: list[dict]) -> str:
|
|||||||
if not labels:
|
if not labels:
|
||||||
return _FALLBACK_PROMPT
|
return _FALLBACK_PROMPT
|
||||||
|
|
||||||
try:
|
|
||||||
import anthropic
|
|
||||||
client = anthropic.Anthropic()
|
|
||||||
msg = client.messages.create(
|
|
||||||
model="claude-haiku-4-5-20251001",
|
|
||||||
max_tokens=60,
|
|
||||||
messages=[{
|
|
||||||
"role": "user",
|
|
||||||
"content": (
|
|
||||||
f"These are the key concepts from a document corpus: {', '.join(labels)}. "
|
|
||||||
"Write a single short sentence (under 20 words) that describes the domain "
|
|
||||||
"for a speech-to-text model. Start with 'Technical' or the domain name. "
|
|
||||||
"No explanation, just the sentence."
|
|
||||||
),
|
|
||||||
}],
|
|
||||||
)
|
|
||||||
prompt = msg.content[0].text.strip().strip('"')
|
|
||||||
return prompt + " Use proper punctuation and paragraph breaks."
|
|
||||||
except Exception:
|
|
||||||
# If LLM call fails for any reason, fall back gracefully
|
|
||||||
topics = ", ".join(labels[:5])
|
topics = ", ".join(labels[:5])
|
||||||
return f"Technical discussion about {topics}. Use proper punctuation and paragraph breaks."
|
return f"Technical discussion about {topics}. Use proper punctuation and paragraph breaks."
|
||||||
|
|
||||||
|
|||||||
@@ -1,13 +1,10 @@
|
|||||||
"""Tests for graphify.transcribe — video/audio transcription support."""
|
"""Tests for graphify.transcribe — video/audio transcription support."""
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import json
|
|
||||||
import os
|
import os
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from unittest.mock import MagicMock, patch
|
from unittest.mock import MagicMock, patch
|
||||||
|
|
||||||
import sys
|
|
||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
|
|
||||||
from graphify.transcribe import (
|
from graphify.transcribe import (
|
||||||
@@ -47,38 +44,13 @@ def test_build_whisper_prompt_env_override(monkeypatch):
|
|||||||
assert prompt == "Custom domain hint."
|
assert prompt == "Custom domain hint."
|
||||||
|
|
||||||
|
|
||||||
def test_build_whisper_prompt_llm_success():
|
def test_build_whisper_prompt_returns_topic_string():
|
||||||
"""Successful LLM call returns generated prompt with punctuation suffix."""
|
"""Returns a topic-based prompt from god node labels — no LLM call."""
|
||||||
god_nodes = [{"label": "neural networks"}, {"label": "transformers"}, {"label": "attention"}]
|
god_nodes = [{"label": "neural networks"}, {"label": "transformers"}, {"label": "attention"}]
|
||||||
|
|
||||||
fake_response = MagicMock()
|
|
||||||
fake_response.content = [MagicMock(text="Machine learning and deep learning research")]
|
|
||||||
|
|
||||||
mock_anthropic = MagicMock()
|
|
||||||
mock_anthropic.Anthropic.return_value.messages.create.return_value = fake_response
|
|
||||||
|
|
||||||
with patch.dict(os.environ, {}, clear=False):
|
with patch.dict(os.environ, {}, clear=False):
|
||||||
os.environ.pop("GRAPHIFY_WHISPER_PROMPT", None)
|
os.environ.pop("GRAPHIFY_WHISPER_PROMPT", None)
|
||||||
with patch.dict(sys.modules, {"anthropic": mock_anthropic}):
|
|
||||||
prompt = build_whisper_prompt(god_nodes)
|
prompt = build_whisper_prompt(god_nodes)
|
||||||
|
assert "neural networks" in prompt.lower() or "transformers" in prompt.lower()
|
||||||
assert "Machine learning" in prompt
|
|
||||||
assert "punctuation" in prompt.lower()
|
|
||||||
|
|
||||||
|
|
||||||
def test_build_whisper_prompt_llm_failure_fallback():
|
|
||||||
"""If LLM call raises, falls back to topic-based prompt."""
|
|
||||||
god_nodes = [{"label": "kubernetes"}, {"label": "docker"}, {"label": "helm"}]
|
|
||||||
|
|
||||||
mock_anthropic = MagicMock()
|
|
||||||
mock_anthropic.Anthropic.return_value.messages.create.side_effect = Exception("API error")
|
|
||||||
|
|
||||||
with patch.dict(os.environ, {}, clear=False):
|
|
||||||
os.environ.pop("GRAPHIFY_WHISPER_PROMPT", None)
|
|
||||||
with patch.dict(sys.modules, {"anthropic": mock_anthropic}):
|
|
||||||
prompt = build_whisper_prompt(god_nodes)
|
|
||||||
|
|
||||||
assert "kubernetes" in prompt.lower() or "docker" in prompt.lower()
|
|
||||||
assert "punctuation" in prompt.lower()
|
assert "punctuation" in prompt.lower()
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user