allow ajust sentences

This commit is contained in:
2026-08-30 11:07:50 +08:00
parent 508a26ba02
commit 0d0b82e095
9 changed files with 270 additions and 3 deletions

View File

@@ -157,7 +157,8 @@ curl -X POST http://127.0.0.1:8001/v1/audio/transcriptions \
逗号前内容不足 50 个单词时不在此处断句会继续与后续内容合并。Whisper 在没有 逗号前内容不足 50 个单词时不在此处断句会继续与后续内容合并。Whisper 在没有
标点处断开的 segment 会自动合并成同一句。管理后台编辑句子文本时,换行并保存会把 标点处断开的 segment 会自动合并成同一句。管理后台编辑句子文本时,换行并保存会把
该句拆成多句;拆分点优先使用对该句音频再次转写得到的词级时间戳,转写不可用时按 该句拆成多句;拆分点优先使用对该句音频再次转写得到的词级时间戳,转写不可用时按
文本长度估算。若转写服务支持词级时间戳 文本长度估算。管理后台的结束时间旁支持微调按钮;调整后下一句的起始时间联动更新,
并保证相邻句子不重叠。若转写服务支持词级时间戳
`timestamp_granularities[]=word`Whisper/Speaches 支持),句子的结束时间用 `timestamp_granularities[]=word`Whisper/Speaches 支持),句子的结束时间用
标点所在单词的时间戳;不支持时自动按文字长度在 segment 内估算,请求失败会回退为 标点所在单词的时间戳;不支持时自动按文字长度在 segment 内估算,请求失败会回退为
不带词级时间戳的普通请求。MOSS 不支持该参数时同样自动回退,不影响转写。 不带词级时间戳的普通请求。MOSS 不支持该参数时同样自动回退,不影响转写。

View File

@@ -57,6 +57,7 @@ PUT /api/v1/admin/videos/raw
POST /api/v1/admin/videos/{sha256}/process POST /api/v1/admin/videos/{sha256}/process
PUT /api/v1/admin/videos/{sha256}/sentences/{index} PUT /api/v1/admin/videos/{sha256}/sentences/{index}
PUT /api/v1/admin/videos/{sha256}/sentences/{index}/split PUT /api/v1/admin/videos/{sha256}/sentences/{index}/split
PUT /api/v1/admin/videos/{sha256}/sentences/{index}/boundary
DELETE /api/v1/admin/videos/{sha256} DELETE /api/v1/admin/videos/{sha256}
``` ```

View File

@@ -32,6 +32,7 @@ from .audio_metrics import AudioAnalysisError
from .config import Settings from .config import Settings
from .models import ( from .models import (
AssessmentResult, AssessmentResult,
SentenceBoundaryAdjust,
SentenceBoundary, SentenceBoundary,
SentenceBoundaryDocument, SentenceBoundaryDocument,
SentenceTextUpdate, SentenceTextUpdate,
@@ -363,6 +364,28 @@ def create_app(
raise HTTPException(status_code=404, detail="Sentence was not found.") raise HTTPException(status_code=404, detail="Sentence was not found.")
return result return result
@application.put(
"/api/v1/admin/videos/{video_hash}/sentences/{sentence_index}/boundary",
response_model=List[SentenceBoundary],
dependencies=[Depends(require_admin)],
)
def adjust_sentence_boundary(
payload: SentenceBoundaryAdjust,
video_hash: str = SHA256_PATH,
sentence_index: int = ApiPath(ge=0),
) -> List[SentenceBoundary]:
try:
result = video_repository.adjust_sentence_boundary(
video_hash,
sentence_index,
payload.delta_ms,
)
except ValueError as exc:
raise HTTPException(status_code=400, detail=str(exc)) from exc
if result is None:
raise HTTPException(status_code=404, detail="Sentence or video was not found.")
return result
@application.delete( @application.delete(
"/api/v1/admin/videos/{video_hash}", "/api/v1/admin/videos/{video_hash}",
status_code=204, status_code=204,

View File

@@ -92,6 +92,12 @@ class SentenceTextUpdate(BaseModel):
language: Optional[str] = Field(default=None, max_length=32) language: Optional[str] = Field(default=None, max_length=32)
class SentenceBoundaryAdjust(BaseModel):
model_config = ConfigDict(extra="forbid")
delta_ms: int = Field(ge=-10_000, le=10_000)
class TextSubstitution(BaseModel): class TextSubstitution(BaseModel):
expected: str expected: str
actual: str actual: str

View File

@@ -491,6 +491,121 @@ class VideoRepository:
for row in result_rows for row in result_rows
] ]
def adjust_sentence_boundary(
self,
video_hash: str,
sentence_index: int,
delta_ms: int,
) -> Optional[List[SentenceBoundary]]:
"""Move one sentence end and link the next sentence's start to it."""
normalized_hash = normalize_video_hash(video_hash)
with self._connect() as connection:
video = connection.execute(
"SELECT duration_ms FROM videos WHERE video_hash = ?",
(normalized_hash,),
).fetchone()
if video is None or video["duration_ms"] is None:
return None
current = connection.execute(
"""
SELECT * FROM sentences
WHERE video_hash = ? AND sentence_index = ?
""",
(normalized_hash, sentence_index),
).fetchone()
if current is None:
return None
following = connection.execute(
"""
SELECT * FROM sentences
WHERE video_hash = ? AND sentence_index = ?
""",
(normalized_hash, sentence_index + 1),
).fetchone()
duration_ms = int(video["duration_ms"])
minimum_end = current["start_ms"] + 1
maximum_end = duration_ms
if following is not None:
maximum_end = min(maximum_end, following["end_ms"] - 1)
new_end_ms = current["end_ms"] + delta_ms
if new_end_ms < minimum_end or new_end_ms > maximum_end:
raise ValueError(
"The adjusted boundary must remain inside both sentences."
)
current_duration_ms = new_end_ms - current["start_ms"]
current_reference_ms = current["reference_speech_duration_ms"] or current_duration_ms
current_reference_ms = min(
current_duration_ms,
max(1, current_reference_ms + delta_ms),
)
connection.execute(
"""
UPDATE sentences
SET end_ms = ?, reference_speech_duration_ms = ?
WHERE video_hash = ? AND sentence_index = ?
""",
(
new_end_ms,
current_reference_ms,
normalized_hash,
sentence_index,
),
)
updated_indexes = [sentence_index]
if following is not None:
old_start_ms = following["start_ms"]
next_duration_ms = following["end_ms"] - new_end_ms
next_reference_ms = following["reference_speech_duration_ms"] or next_duration_ms
next_reference_ms = min(
next_duration_ms,
max(1, next_reference_ms - (new_end_ms - old_start_ms)),
)
connection.execute(
"""
UPDATE sentences
SET start_ms = ?, reference_speech_duration_ms = ?
WHERE video_hash = ? AND sentence_index = ?
""",
(
new_end_ms,
next_reference_ms,
normalized_hash,
sentence_index + 1,
),
)
updated_indexes.append(sentence_index + 1)
connection.execute(
"UPDATE videos SET updated_at = ? WHERE video_hash = ?",
(utc_now(), normalized_hash),
)
placeholders = ", ".join("?" for _ in updated_indexes)
result_rows = connection.execute(
f"""
SELECT * FROM sentences
WHERE video_hash = ? AND sentence_index IN ({placeholders})
ORDER BY sentence_index
""",
(normalized_hash, *updated_indexes),
).fetchall()
return [
SentenceBoundary(
index=row["sentence_index"],
start_ms=row["start_ms"],
end_ms=row["end_ms"],
text=row["text"],
language=row["language"],
reference_speech_duration_ms=row["reference_speech_duration_ms"],
)
for row in result_rows
]
def create_dub_share( def create_dub_share(
self, self,
*, *,

View File

@@ -106,6 +106,10 @@ dialog::backdrop { background: rgb(20 25 28 / 55%); }
.sentence-scroll { max-height: calc(100vh - 120px); overflow: auto; } .sentence-scroll { max-height: calc(100vh - 120px); overflow: auto; }
.sentence-scroll textarea { width: min(600px, 45vw); min-width: 260px; min-height: 58px; resize: vertical; padding: 7px; border: 1px solid #b8c0c7; border-radius: 3px; } .sentence-scroll textarea { width: min(600px, 45vw); min-width: 260px; min-height: 58px; resize: vertical; padding: 7px; border: 1px solid #b8c0c7; border-radius: 3px; }
.sentence-scroll select { min-width: 90px; } .sentence-scroll select { min-width: 90px; }
.boundary-control { display: inline-flex; align-items: center; gap: 4px; margin-left: 6px; white-space: nowrap; }
.boundary-control button { min-height: 24px; padding: 2px 6px; font-size: 11px; line-height: 1; }
.boundary-time { display: inline-block; min-width: 56px; text-align: center; font-variant-numeric: tabular-nums; }
.boundary-status { min-width: 26px; color: #a12622; font-size: 11px; }
@media (max-width: 820px) { @media (max-width: 820px) {
.topbar { align-items: stretch; flex-direction: column; padding: 14px 16px; } .topbar { align-items: stretch; flex-direction: column; padding: 14px 16px; }

View File

@@ -91,6 +91,6 @@
</div> </div>
</dialog> </dialog>
<script src="/static/admin.js?v=manual-split-v1" defer></script> <script src="/static/admin.js?v=boundary-adjust-v1" defer></script>
</body> </body>
</html> </html>

View File

@@ -239,9 +239,13 @@ function sentenceRow(video, sentence) {
} }
}); });
saveContainer.append(save); saveContainer.append(save);
const timeContainer = document.createElement("td");
const startLabel = document.createElement("span");
startLabel.textContent = formatDuration(sentence.start_ms);
timeContainer.append(startLabel, " — ", boundaryControl(video, sentence));
row.append( row.append(
textCell(String(sentence.index + 1)), textCell(String(sentence.index + 1)),
textCell(`${formatDuration(sentence.start_ms)} - ${formatDuration(sentence.end_ms)}`), timeContainer,
textContainer, textContainer,
languageContainer, languageContainer,
saveContainer, saveContainer,
@@ -249,6 +253,50 @@ function sentenceRow(video, sentence) {
return row; return row;
} }
function boundaryControl(video, sentence) {
const container = document.createElement("div");
container.className = "boundary-control";
const label = document.createElement("span");
label.className = "boundary-time";
label.textContent = formatDuration(sentence.end_ms);
const status = document.createElement("span");
status.className = "boundary-status";
async function adjust(deltaMs, changedButton) {
const buttons = [minus, plus];
buttons.forEach((item) => { item.disabled = true; });
status.textContent = "";
try {
await api(
`/api/v1/admin/videos/${video.video_hash}/sentences/${sentence.index}/boundary`,
{
method: "PUT",
headers: { "Content-Type": "application/json" },
body: JSON.stringify({ delta_ms: deltaMs }),
},
);
await openSentences(video);
} catch (error) {
elements.serviceState.textContent = `分割点调整失败: ${error.message}`;
status.textContent = "失败";
buttons.forEach((item) => { item.disabled = false; });
window.setTimeout(() => { status.textContent = ""; }, 1800);
}
if (changedButton) {
window.setTimeout(() => {
changedButton.textContent = deltaMs < 0 ? "0.1" : "+0.1";
}, 900);
}
}
const minus = button("0.1", (event) => adjust(event.shiftKey ? -10 : -100, minus));
const plus = button("+0.1", (event) => adjust(event.shiftKey ? 10 : 100, plus));
minus.title = "结束点提前 0.1 秒;按住 Shift 微调 0.01 秒";
plus.title = "结束点延后 0.1 秒;按住 Shift 微调 0.01 秒";
container.append(minus, label, plus, status);
return container;
}
async function reprocess(videoHash) { async function reprocess(videoHash) {
try { try {
await api(`/api/v1/admin/videos/${videoHash}/process`, { method: "POST" }); await api(`/api/v1/admin/videos/${videoHash}/process`, { method: "POST" });

View File

@@ -259,3 +259,72 @@ def test_admin_can_split_sentence_and_reindexes_following_sentences(tmp_path):
"Next sentence.", "Next sentence.",
] ]
assert [item.index for item in document.sentences] == [0, 1, 2] assert [item.index for item in document.sentences] == [0, 1, 2]
def test_admin_boundary_adjustment_updates_linked_next_start(tmp_path):
settings = replace(Settings.from_env(), data_dir=tmp_path / "data")
settings.ensure_directories()
repository = VideoRepository(settings.database_path)
repository.upsert_upload(
video_hash=VIDEO_HASH,
title="Boundary lesson",
filename="lesson.mp4",
stored_filename=f"{VIDEO_HASH}.mp4",
content_type="video/mp4",
size_bytes=5,
language="en",
)
repository.save_processing_result(
SentenceBoundaryDocument(
video_hash=VIDEO_HASH,
duration_ms=9000,
algorithm_version="test",
sentences=[
SentenceBoundary(
index=0,
start_ms=0,
end_ms=4000,
text="First.",
language="en",
reference_speech_duration_ms=3600,
),
SentenceBoundary(
index=1,
start_ms=4000,
end_ms=9000,
text="Second.",
language="en",
reference_speech_duration_ms=4600,
),
],
),
None,
)
class UnavailableTranscriber:
available = False
client = TestClient(
create_app(
BoundaryStore(tmp_path / "boundaries.json"),
settings=settings,
repository=repository,
transcriber=UnavailableTranscriber(),
)
)
response = client.put(
f"/api/v1/admin/videos/{VIDEO_HASH}/sentences/0/boundary",
json={"delta_ms": 250},
)
assert response.status_code == 200
updated = response.json()
assert [item["index"] for item in updated] == [0, 1]
assert updated[0]["end_ms"] == 4250
assert updated[1]["start_ms"] == 4250
assert updated[0]["reference_speech_duration_ms"] == 3850
assert updated[1]["reference_speech_duration_ms"] == 4350
document = repository.get_document(VIDEO_HASH)
assert [item.end_ms for item in document.sentences] == [4250, 9000]
assert [item.start_ms for item in document.sentences] == [0, 4250]