admin can split a sentence into 2 sentences

This commit is contained in:
2026-08-30 09:25:34 +08:00
parent abc052e42d
commit a99c0cfd17
8 changed files with 577 additions and 13 deletions

View File

@@ -1,6 +1,11 @@
import numpy as np
from sentence_api.processing import MOSS_ALGORITHM_VERSION, document_from_transcript
from sentence_api.processing import (
MOSS_ALGORITHM_VERSION,
document_from_transcript,
split_sentence_text,
)
from sentence_api.models import SentenceBoundary
from sentence_api.transcription import (
Transcript,
TranscriptionSegment,
@@ -13,6 +18,18 @@ SAMPLE_RATE = 16000
VIDEO_HASH = "a" * 64
class WordTimestampTranscriber:
def __init__(self, transcript):
self.transcript = transcript
@property
def available(self):
return True
def transcribe(self, audio_path, language=None):
return self.transcript
def _speech(seconds: float) -> np.ndarray:
t = np.arange(int(seconds * SAMPLE_RATE)) / SAMPLE_RATE
return 0.25 * np.sin(2 * np.pi * 220 * t)
@@ -99,3 +116,45 @@ def test_document_from_transcript_splits_sentences_at_punctuation(tmp_path):
assert document.sentences[1].start_ms == 9500
assert document.sentences[1].end_ms == 24000
assert all(sentence.reference_speech_duration_ms > 0 for sentence in document.sentences)
def test_split_sentence_text_uses_whisper_word_timestamps(tmp_path):
transcript = Transcript(
text="Alpha beta. Gamma delta.",
segments=[
TranscriptionSegment(
start_seconds=0.0,
end_seconds=1.0,
text="Alpha beta. Gamma delta.",
words=[
WordTimestamp(0.0, 0.25, "Alpha"),
WordTimestamp(0.25, 0.5, "beta."),
WordTimestamp(0.5, 0.75, "Gamma"),
WordTimestamp(0.75, 1.0, "delta."),
],
)
],
)
audio_path = tmp_path / "sentence.wav"
audio_path.write_bytes(_wav_bytes(_speech(1.0), SAMPLE_RATE))
source = SentenceBoundary(
index=2,
start_ms=1000,
end_ms=5000,
text="Alpha beta Gamma delta",
language="en",
reference_speech_duration_ms=3800,
)
replacements = split_sentence_text(
transcriber=WordTimestampTranscriber(transcript),
media_path=audio_path,
sentence=source,
text="Alpha beta\nGamma delta",
language="en",
)
assert [(item.text, item.start_ms, item.end_ms) for item in replacements] == [
("Alpha beta", 1000, 3000),
("Gamma delta", 3000, 5000),
]