add sentence service

This commit is contained in:
2026-08-14 19:06:04 +08:00
parent f8a6bf24e4
commit 69d2ba986f
19 changed files with 2379 additions and 109 deletions

View File

@@ -121,3 +121,11 @@ SDK 基于 AndroidX Media3 / ExoPlayer支持在线视频流、本地缓存、
暂停/继续、左划/右划按句或按步长跳转,并预留了学生模仿质量评分算法接口。 暂停/继续、左划/右划按句或按步长跳转,并预留了学生模仿质量评分算法接口。
详见 [android/README.md](android/README.md)。 详见 [android/README.md](android/README.md)。
## 句子边界 API
仓库新增了 `sentence_api/` 服务。它复用桌面播放器的静音检测算法,提前为视频
生成句子边界,并根据视频 SHA-256 哈希向 Android 返回毫秒级的
`start_ms` / `end_ms` 列表。服务不会在移动端请求时临时分析视频。
详见 [sentence_api/README.md](sentence_api/README.md)。

View File

@@ -17,6 +17,10 @@ and a placeholder interface for future imitation-quality scoring.
- Swipe right: next sentence; if there is no sentence data, fast-forward by the - Swipe right: next sentence; if there is no sentence data, fast-forward by the
configured seek step. configured seek step.
The sample app's previous/next sentence buttons use the same fallback behavior.
Actual sentence navigation requires `SentenceBoundary` timestamps supplied by
the course service or extracted from subtitle/speech analysis data.
These defaults match the desktop player's arrow-key workflow while fitting a These defaults match the desktop player's arrow-key workflow while fitting a
tablet touch screen. tablet touch screen.
@@ -62,6 +66,50 @@ OralTrainerSdk.init(
) )
``` ```
## Sentence Boundary API
The SDK defaults to `https://video_service.d1kt.cn` and requests:
```text
GET /api/v1/videos/{sha256}/sentence-boundaries
```
For a known hash:
```kotlin
sdk.sentenceBoundaryApi.fetch(videoHash, callback)
```
For a local `content://` video, the SDK can hash the file in streaming chunks
before querying the API:
```kotlin
sdk.sentenceBoundaryApi.fetchForUri(videoUri, contentResolver, callback)
```
Override the service only when a staging or private deployment is required:
```kotlin
OralTrainerSdk.init(
context,
OralTrainerSdkConfig(
sentenceBoundaryApiBaseUrl = "https://video_service.d1kt.cn"
)
)
```
## Local Video Testing
The computer path `/Users/...` is not visible to an Android device. For a
quick test, copy the video to the device or use the `选择本地视频` button in the
sample app. The Android document picker returns a `content://` URI, which the
SDK supports without requesting broad storage permissions.
For production courses, keep the video on an HTTPS CDN or object-storage
service and pass its URL as `TrainingMediaItem.uri`. The SDK streams it and
caches downloaded ranges locally. Copying large 4K files to each device is
better reserved for explicitly offline courses.
## Future Imitation Scoring ## Future Imitation Scoring
Provide an implementation of `ImitationQualityAssessor` when the speech Provide an implementation of `ImitationQualityAssessor` when the speech

View File

@@ -242,6 +242,18 @@ class OralTrainerController internal constructor(
return true return true
} }
fun previousSentenceOrRewind() {
if (!config.sentenceMode || !seekToPreviousSentence()) {
rewind()
}
}
fun nextSentenceOrForward() {
if (!config.sentenceMode || !seekToNextSentence()) {
fastForward()
}
}
fun performSwipeAction(action: SwipeAction) { fun performSwipeAction(action: SwipeAction) {
when (action) { when (action) {
SwipeAction.NONE -> Unit SwipeAction.NONE -> Unit
@@ -249,16 +261,8 @@ class OralTrainerController internal constructor(
SwipeAction.FORWARD -> fastForward() SwipeAction.FORWARD -> fastForward()
SwipeAction.PREVIOUS_SENTENCE -> seekToPreviousSentence() SwipeAction.PREVIOUS_SENTENCE -> seekToPreviousSentence()
SwipeAction.NEXT_SENTENCE -> seekToNextSentence() SwipeAction.NEXT_SENTENCE -> seekToNextSentence()
SwipeAction.PREVIOUS_SENTENCE_OR_REWIND -> { SwipeAction.PREVIOUS_SENTENCE_OR_REWIND -> previousSentenceOrRewind()
if (!config.sentenceMode || !seekToPreviousSentence()) { SwipeAction.NEXT_SENTENCE_OR_FORWARD -> nextSentenceOrForward()
rewind()
}
}
SwipeAction.NEXT_SENTENCE_OR_FORWARD -> {
if (!config.sentenceMode || !seekToNextSentence()) {
fastForward()
}
}
} }
} }

View File

@@ -9,6 +9,7 @@ class OralTrainerSdk private constructor(
private val appContext = context.applicationContext private val appContext = context.applicationContext
val cache: OralTrainerCache = OralTrainerCache(appContext, config) val cache: OralTrainerCache = OralTrainerCache(appContext, config)
val sentenceBoundaryApi: SentenceBoundaryApi = SentenceBoundaryApi(appContext, config)
@JvmOverloads @JvmOverloads
fun createController( fun createController(
@@ -24,6 +25,7 @@ class OralTrainerSdk private constructor(
} }
fun release() { fun release() {
sentenceBoundaryApi.release()
StreamingCache.release() StreamingCache.release()
} }

View File

@@ -8,4 +8,5 @@ data class OralTrainerSdkConfig @JvmOverloads constructor(
val userAgent: String = "OralTrainerSdk/0.1.0", val userAgent: String = "OralTrainerSdk/0.1.0",
val connectTimeoutMs: Int = 15_000, val connectTimeoutMs: Int = 15_000,
val readTimeoutMs: Int = 30_000, val readTimeoutMs: Int = 30_000,
val sentenceBoundaryApiBaseUrl: String = "https://video_service.d1kt.cn",
) )

View File

@@ -0,0 +1,196 @@
package cn.learningpad.oraltrainer.sdk
import android.content.ContentResolver
import android.content.Context
import android.net.Uri
import android.os.Handler
import android.os.Looper
import org.json.JSONObject
import java.io.BufferedInputStream
import java.io.IOException
import java.io.InterruptedIOException
import java.net.HttpURLConnection
import java.net.URL
import java.security.MessageDigest
import java.util.concurrent.ExecutorService
import java.util.concurrent.Executors
import java.util.concurrent.Future
data class SentenceBoundaryApiResult(
val videoHash: String,
val durationMs: Long,
val algorithmVersion: String,
val sentences: List<SentenceBoundary>,
)
interface SentenceBoundaryApiCallback {
fun onSuccess(result: SentenceBoundaryApiResult)
fun onError(error: Throwable)
}
class CancellableRequest internal constructor(
private val future: Future<*>,
) {
fun cancel() {
future.cancel(true)
}
}
class SentenceBoundaryApi internal constructor(
context: Context,
private val config: OralTrainerSdkConfig,
) {
private val appContext = context.applicationContext
private val mainHandler = Handler(Looper.getMainLooper())
private val executor: ExecutorService = Executors.newCachedThreadPool()
private val baseUri = Uri.parse(config.sentenceBoundaryApiBaseUrl.trimEnd('/'))
init {
require(baseUri.scheme == "https" || baseUri.scheme == "http") {
"sentenceBoundaryApiBaseUrl must use http or https."
}
require(!baseUri.host.isNullOrBlank()) {
"sentenceBoundaryApiBaseUrl must include a host."
}
}
fun fetch(
videoHash: String,
callback: SentenceBoundaryApiCallback,
): CancellableRequest {
validateHash(videoHash)
return submit(callback) { fetchBlocking(videoHash.lowercase()) }
}
fun fetchForUri(
uri: Uri,
contentResolver: ContentResolver = appContext.contentResolver,
callback: SentenceBoundaryApiCallback,
): CancellableRequest {
return submit(callback) {
val videoHash = sha256(uri, contentResolver)
fetchBlocking(videoHash)
}
}
fun release() {
executor.shutdownNow()
}
private fun submit(
callback: SentenceBoundaryApiCallback,
operation: () -> SentenceBoundaryApiResult,
): CancellableRequest {
val future = executor.submit {
try {
val result = operation()
if (!Thread.currentThread().isInterrupted) {
mainHandler.post { callback.onSuccess(result) }
}
} catch (error: Throwable) {
if (!Thread.currentThread().isInterrupted) {
mainHandler.post { callback.onError(error) }
}
}
}
return CancellableRequest(future)
}
private fun fetchBlocking(videoHash: String): SentenceBoundaryApiResult {
val endpoint = baseUri.buildUpon()
.appendPath("api")
.appendPath("v1")
.appendPath("videos")
.appendPath(videoHash)
.appendPath("sentence-boundaries")
.build()
val connection = URL(endpoint.toString()).openConnection() as HttpURLConnection
return try {
connection.requestMethod = "GET"
connection.connectTimeout = config.connectTimeoutMs
connection.readTimeout = config.readTimeoutMs
connection.setRequestProperty("Accept", "application/json")
connection.setRequestProperty("User-Agent", config.userAgent)
val statusCode = connection.responseCode
val stream = if (statusCode in 200..299) {
connection.inputStream
} else {
connection.errorStream
}
val body = stream?.bufferedReader(Charsets.UTF_8)?.use { it.readText() }.orEmpty()
if (statusCode !in 200..299) {
throw SentenceBoundaryApiException(statusCode, body)
}
parseResponse(body, videoHash)
} finally {
connection.disconnect()
}
}
private fun parseResponse(
body: String,
requestedHash: String,
): SentenceBoundaryApiResult {
val root = JSONObject(body)
val responseHash = root.getString("video_hash").lowercase()
if (responseHash != requestedHash.lowercase()) {
throw IOException("Sentence boundary response hash does not match the request.")
}
val jsonSentences = root.getJSONArray("sentences")
val sentences = buildList(jsonSentences.length()) {
for (index in 0 until jsonSentences.length()) {
val sentence = jsonSentences.getJSONObject(index)
add(
SentenceBoundary(
index = sentence.getInt("index"),
startMs = sentence.getLong("start_ms"),
endMs = sentence.getLong("end_ms"),
text = if (sentence.isNull("text")) null else sentence.getString("text"),
)
)
}
}
return SentenceBoundaryApiResult(
videoHash = responseHash,
durationMs = root.getLong("duration_ms"),
algorithmVersion = root.getString("algorithm_version"),
sentences = sentences,
)
}
private fun sha256(uri: Uri, contentResolver: ContentResolver): String {
val digest = MessageDigest.getInstance("SHA-256")
contentResolver.openInputStream(uri)?.use { input ->
BufferedInputStream(input).use { buffered ->
val buffer = ByteArray(1024 * 1024)
while (true) {
if (Thread.currentThread().isInterrupted) {
throw InterruptedIOException()
}
val count = buffered.read(buffer)
if (count < 0) {
break
}
digest.update(buffer, 0, count)
}
}
} ?: throw IOException("Cannot open video URI: $uri")
return digest.digest().joinToString("") { byte -> "%02x".format(byte.toInt() and 0xff) }
}
private fun validateHash(videoHash: String) {
require(SHA256_PATTERN.matches(videoHash)) {
"videoHash must be a 64-character SHA-256 hex digest."
}
}
companion object {
private val SHA256_PATTERN = Regex("^[0-9a-fA-F]{64}$")
}
}
class SentenceBoundaryApiException(
val statusCode: Int,
responseBody: String,
) : IOException("Sentence boundary API returned HTTP $statusCode: $responseBody")

View File

@@ -5,6 +5,7 @@ package cn.learningpad.oraltrainer.sdk
import android.content.Context import android.content.Context
import androidx.media3.database.StandaloneDatabaseProvider import androidx.media3.database.StandaloneDatabaseProvider
import androidx.media3.datasource.DataSource import androidx.media3.datasource.DataSource
import androidx.media3.datasource.DefaultDataSource
import androidx.media3.datasource.cache.CacheDataSource import androidx.media3.datasource.cache.CacheDataSource
import androidx.media3.datasource.cache.LeastRecentlyUsedCacheEvictor import androidx.media3.datasource.cache.LeastRecentlyUsedCacheEvictor
import androidx.media3.datasource.cache.SimpleCache import androidx.media3.datasource.cache.SimpleCache
@@ -32,11 +33,12 @@ internal object StreamingCache {
context: Context, context: Context,
config: OralTrainerSdkConfig, config: OralTrainerSdkConfig,
): DataSource.Factory { ): DataSource.Factory {
val upstreamFactory = DefaultHttpDataSource.Factory() val httpDataSourceFactory = DefaultHttpDataSource.Factory()
.setUserAgent(config.userAgent) .setUserAgent(config.userAgent)
.setConnectTimeoutMs(config.connectTimeoutMs) .setConnectTimeoutMs(config.connectTimeoutMs)
.setReadTimeoutMs(config.readTimeoutMs) .setReadTimeoutMs(config.readTimeoutMs)
.setAllowCrossProtocolRedirects(true) .setAllowCrossProtocolRedirects(true)
val upstreamFactory = DefaultDataSource.Factory(context.applicationContext, httpDataSourceFactory)
return CacheDataSource.Factory() return CacheDataSource.Factory()
.setCache(get(context, config)) .setCache(get(context, config))

View File

@@ -1,6 +1,7 @@
package cn.learningpad.oraltrainer.sample package cn.learningpad.oraltrainer.sample
import android.app.Activity import android.app.Activity
import android.content.Intent
import android.graphics.Color import android.graphics.Color
import android.net.Uri import android.net.Uri
import android.os.Bundle import android.os.Bundle
@@ -19,9 +20,16 @@ import cn.learningpad.oraltrainer.sdk.OralTrainerSdk
import cn.learningpad.oraltrainer.sdk.PlaybackSnapshot import cn.learningpad.oraltrainer.sdk.PlaybackSnapshot
import cn.learningpad.oraltrainer.sdk.PlayerConfig import cn.learningpad.oraltrainer.sdk.PlayerConfig
import cn.learningpad.oraltrainer.sdk.SentenceBoundary import cn.learningpad.oraltrainer.sdk.SentenceBoundary
import cn.learningpad.oraltrainer.sdk.SentenceBoundaryApiCallback
import cn.learningpad.oraltrainer.sdk.TrainingMediaItem import cn.learningpad.oraltrainer.sdk.TrainingMediaItem
import cn.learningpad.oraltrainer.sdk.SentenceBoundaryApiResult
class MainActivity : Activity() { class MainActivity : Activity() {
private companion object {
const val PICK_VIDEO_REQUEST = 1001
}
private lateinit var sdk: OralTrainerSdk
private lateinit var controller: OralTrainerController private lateinit var controller: OralTrainerController
private lateinit var statusText: TextView private lateinit var statusText: TextView
private lateinit var sentenceText: TextView private lateinit var sentenceText: TextView
@@ -29,7 +37,7 @@ class MainActivity : Activity() {
override fun onCreate(savedInstanceState: Bundle?) { override fun onCreate(savedInstanceState: Bundle?) {
super.onCreate(savedInstanceState) super.onCreate(savedInstanceState)
val sdk = OralTrainerSdk.init(this) sdk = OralTrainerSdk.init(this)
controller = sdk.createController( controller = sdk.createController(
playerConfig = PlayerConfig( playerConfig = PlayerConfig(
sentenceMode = true, sentenceMode = true,
@@ -66,8 +74,8 @@ class MainActivity : Activity() {
addView(commandButton("快退") { controller.rewind() }) addView(commandButton("快退") { controller.rewind() })
addView(commandButton("播放/暂停") { controller.togglePlayPause() }) addView(commandButton("播放/暂停") { controller.togglePlayPause() })
addView(commandButton("快进") { controller.fastForward() }) addView(commandButton("快进") { controller.fastForward() })
addView(commandButton("上一句") { controller.seekToPreviousSentence() }) addView(commandButton("上一句") { controller.previousSentenceOrRewind() })
addView(commandButton("下一句") { controller.seekToNextSentence() }) addView(commandButton("下一句") { controller.nextSentenceOrForward() })
} }
val root = LinearLayout(this).apply { val root = LinearLayout(this).apply {
@@ -77,6 +85,10 @@ class MainActivity : Activity() {
addView(statusText) addView(statusText)
addView(sentenceText) addView(sentenceText)
addView(controls) addView(controls)
addView(Button(this@MainActivity).apply {
text = "选择本地视频"
setOnClickListener { openLocalVideoPicker() }
})
} }
setContentView(root) setContentView(root)
@@ -108,6 +120,37 @@ class MainActivity : Activity() {
controller.loadItem(sampleOnlineLesson()) controller.loadItem(sampleOnlineLesson())
} }
override fun onActivityResult(requestCode: Int, resultCode: Int, data: Intent?) {
super.onActivityResult(requestCode, resultCode, data)
if (requestCode != PICK_VIDEO_REQUEST || resultCode != RESULT_OK) {
return
}
val uri = data?.data ?: return
val persistableFlags = data.flags and Intent.FLAG_GRANT_READ_URI_PERMISSION
if (persistableFlags != 0) {
runCatching {
contentResolver.takePersistableUriPermission(uri, persistableFlags)
}
}
val item = TrainingMediaItem(
id = uri.toString(),
title = uri.lastPathSegment ?: "本地视频",
uri = uri,
)
controller.loadItem(item)
statusText.text = "正在获取句子边界..."
sdk.sentenceBoundaryApi.fetchForUri(uri, contentResolver, object : SentenceBoundaryApiCallback {
override fun onSuccess(result: SentenceBoundaryApiResult) {
controller.loadItem(item.copy(sentences = result.sentences))
statusText.text = "句子边界已加载:${result.sentences.size}"
}
override fun onError(error: Throwable) {
statusText.text = "句子边界获取失败,使用 10 秒跳转:${error.message.orEmpty()}"
}
})
}
override fun onDestroy() { override fun onDestroy() {
controller.release() controller.release()
super.onDestroy() super.onDestroy()
@@ -128,6 +171,18 @@ class MainActivity : Activity() {
} }
} }
private fun openLocalVideoPicker() {
startActivityForResult(
Intent(Intent.ACTION_OPEN_DOCUMENT).apply {
addCategory(Intent.CATEGORY_OPENABLE)
type = "video/*"
addFlags(Intent.FLAG_GRANT_READ_URI_PERMISSION)
addFlags(Intent.FLAG_GRANT_PERSISTABLE_URI_PERMISSION)
},
PICK_VIDEO_REQUEST,
)
}
private fun sampleOnlineLesson(): TrainingMediaItem { private fun sampleOnlineLesson(): TrainingMediaItem {
return TrainingMediaItem( return TrainingMediaItem(
id = "online_sample_01", id = "online_sample_01",

View File

@@ -38,6 +38,8 @@ from pathlib import Path
import tkinter as tk import tkinter as tk
from tkinter import filedialog, messagebox, ttk from tkinter import filedialog, messagebox, ttk
from sentence_analysis import detect_sentence_boundaries
try: try:
from PIL import Image as PILImage, ImageTk from PIL import Image as PILImage, ImageTk
except ImportError: except ImportError:
@@ -71,11 +73,6 @@ LOOP_LABELS = {"off": "循环: 关", "one": "循环: 单曲", "list": "循环:
SPEED_PRESETS = [0.5, 0.75, 1.0, 1.25, 1.5, 2.0] SPEED_PRESETS = [0.5, 0.75, 1.0, 1.25, 1.5, 2.0]
SEEK_STEP = 10 SEEK_STEP = 10
SEEK_STEP_BIG = 60 SEEK_STEP_BIG = 60
SILENCE_FLOOR_PERCENTILE = 10 # 用能量低分位估计背景噪声底
SILENCE_THRESHOLD_FACTOR = 1.5 # 静音阈值 = 噪声底 × 系数
SILENCE_BRIDGE_GAP = 0.06 # 停顿中 60ms 内的短促残响/噪声也按静音处理
def fmt_time(ms): def fmt_time(ms):
"""把毫秒格式化为 mm:ss 或 h:mm:ss。""" """把毫秒格式化为 mm:ss 或 h:mm:ss。"""
if ms is None or ms < 0: if ms is None or ms < 0:
@@ -88,96 +85,6 @@ def fmt_time(ms):
return f"{minutes:02d}:{seconds:02d}" return f"{minutes:02d}:{seconds:02d}"
def detect_sentence_boundaries(path, min_silence=0.30, min_sentence=0.35):
"""基于音频静音自动切分句子,返回每句话开始的时间(秒)列表。
语音之间通常有短暂停顿,把停顿前后的连续声音视为一句话。
没有音频轨、无法解码或缺少依赖时返回 None。
"""
try:
import av
import numpy as np
except ImportError:
return None
try:
container = av.open(str(path))
audio = next((s for s in container.streams if s.type == "audio"), None)
if audio is None:
container.close()
return None
sample_rate = 16000
resampler = av.AudioResampler(format="fltp", layout="mono", rate=sample_rate)
window = 480 # 30ms @ 16kHz
frame_seconds = window / sample_rate
energies = []
buf = []
def push(data):
buf.append(data)
total = sum(len(x) for x in buf)
if total < window:
return
arr = np.concatenate(buf)
buf.clear()
n = len(arr) // window * window
win = arr[:n].reshape(-1, window)
energies.extend((np.mean(win * win, axis=1) ** 0.5).tolist())
if len(arr) > n:
buf.append(arr[n:])
for packet in container.demux(audio):
for frame in packet.decode():
for out in resampler.resample(frame):
push(out.to_ndarray()[0])
for out in resampler.resample(None):
push(out.to_ndarray()[0])
if buf:
push(np.zeros(window, dtype=np.float32))
duration = container.duration
container.close()
if len(energies) < 3:
return None
energies = np.asarray(energies, dtype=np.float64)
floor = float(np.percentile(energies, 95))
if floor <= 0.0:
return None
noise_floor = float(np.percentile(energies, SILENCE_FLOOR_PERCENTILE))
threshold = max(0.008, SILENCE_THRESHOLD_FACTOR * noise_floor)
silence = energies < threshold
bridge_frames = int(round(SILENCE_BRIDGE_GAP / frame_seconds))
if bridge_frames > 0:
bridged = silence.copy()
run_start = None
for idx, is_silent in enumerate(silence):
if not is_silent and run_start is None:
run_start = idx
elif is_silent and run_start is not None:
if run_start > 0 and idx - run_start <= bridge_frames:
bridged[run_start:idx] = True
run_start = None
silence = bridged
boundaries = [0.0]
run_start = None
for idx, is_silent in enumerate(silence):
if is_silent and run_start is None:
run_start = idx
elif not is_silent and run_start is not None:
if (idx - run_start) * frame_seconds >= min_silence:
end_sec = idx * frame_seconds
if end_sec - boundaries[-1] >= min_sentence:
boundaries.append(end_sec)
run_start = None
if run_start is not None and (len(silence) - run_start) * frame_seconds >= min_silence:
end_sec = len(silence) * frame_seconds
if end_sec - boundaries[-1] >= min_sentence:
boundaries.append(end_sec)
if duration and duration > 0:
boundaries = [b for b in boundaries if b < duration / 1e6 - 0.1]
return boundaries
except Exception:
return None
class _BaseEngine: class _BaseEngine:
"""播放内核统一接口。""" """播放内核统一接口。"""

113
sentence_analysis.py Normal file
View File

@@ -0,0 +1,113 @@
"""Shared audio-silence sentence boundary detection.
This module intentionally has no GUI or web-service dependencies so the
desktop player, the pre-generation CLI, and the API worker use the same
algorithm.
"""
SILENCE_FLOOR_PERCENTILE = 10
SILENCE_THRESHOLD_FACTOR = 1.5
SILENCE_BRIDGE_GAP = 0.06
def detect_sentence_boundaries(path, min_silence=0.30, min_sentence=0.35):
"""Return sentence start times in seconds, or None when unavailable.
Audio is resampled to mono 16 kHz, divided into 30 ms windows, and
silence runs are converted into boundaries. This matches the original
desktop player's behavior.
"""
try:
import av
import numpy as np
except ImportError:
return None
container = None
try:
container = av.open(str(path))
audio = next((stream for stream in container.streams if stream.type == "audio"), None)
if audio is None:
return None
sample_rate = 16_000
resampler = av.AudioResampler(format="fltp", layout="mono", rate=sample_rate)
window = 480
frame_seconds = window / sample_rate
energies = []
buffer = []
def push(data):
buffer.append(data)
total = sum(len(chunk) for chunk in buffer)
if total < window:
return
array = np.concatenate(buffer)
buffer.clear()
complete = len(array) // window * window
windows = array[:complete].reshape(-1, window)
energies.extend((np.mean(windows * windows, axis=1) ** 0.5).tolist())
if len(array) > complete:
buffer.append(array[complete:])
for packet in container.demux(audio):
for frame in packet.decode():
for output in resampler.resample(frame):
push(output.to_ndarray()[0])
for output in resampler.resample(None):
push(output.to_ndarray()[0])
if buffer:
push(np.zeros(window, dtype=np.float32))
duration = container.duration
if len(energies) < 3:
return None
energies = np.asarray(energies, dtype=np.float64)
signal_floor = float(np.percentile(energies, 95))
if signal_floor <= 0.0:
return None
noise_floor = float(np.percentile(energies, SILENCE_FLOOR_PERCENTILE))
threshold = max(0.008, SILENCE_THRESHOLD_FACTOR * noise_floor)
silence = energies < threshold
bridge_frames = int(round(SILENCE_BRIDGE_GAP / frame_seconds))
if bridge_frames > 0:
bridged = silence.copy()
run_start = None
for index, is_silent in enumerate(silence):
if not is_silent and run_start is None:
run_start = index
elif is_silent and run_start is not None:
if run_start > 0 and index - run_start <= bridge_frames:
bridged[run_start:index] = True
run_start = None
silence = bridged
boundaries = [0.0]
run_start = None
for index, is_silent in enumerate(silence):
if is_silent and run_start is None:
run_start = index
elif not is_silent and run_start is not None:
if (index - run_start) * frame_seconds >= min_silence:
end_seconds = index * frame_seconds
if end_seconds - boundaries[-1] >= min_sentence:
boundaries.append(end_seconds)
run_start = None
if run_start is not None and (len(silence) - run_start) * frame_seconds >= min_silence:
end_seconds = len(silence) * frame_seconds
if end_seconds - boundaries[-1] >= min_sentence:
boundaries.append(end_seconds)
if duration and duration > 0:
boundaries = [boundary for boundary in boundaries if boundary < duration / 1e6 - 0.1]
return boundaries
except Exception:
return None
finally:
if container is not None:
container.close()
__all__ = ["detect_sentence_boundaries"]

92
sentence_api/README.md Normal file
View File

@@ -0,0 +1,92 @@
# Sentence Boundary API
This service looks up pre-generated sentence boundaries by the SHA-256 hash of
the exact video bytes. It does not analyze media during an API request.
## Install
From the repository root:
```bash
python3 -m venv .venv-sentence-api
. .venv-sentence-api/bin/activate
python -m pip install -r sentence_api/requirements.txt
```
## Generate The Index
Generate boundaries with the same silence detector used by the desktop player:
```bash
python -m sentence_api.generate_boundaries \
"/path/to/lesson.mp4" \
--index sentence_api/data/sentence_boundaries.json
```
The command calculates the SHA-256 hash, detects boundaries, converts seconds
to milliseconds, infers each `end_ms` from the next sentence start, and writes
the result atomically into the JSON index. The last sentence ends at the media
duration.
For a large course library, run this command in an ingestion worker and store
the same document in a database or object storage instead of committing the
JSON file to the application image.
## Run
```bash
SENTENCE_BOUNDARIES_FILE=sentence_api/data/sentence_boundaries.json \
python -m uvicorn sentence_api.main:app --host 0.0.0.0 --port 8000
```
The interactive API documentation is available at `/docs`.
## Request
```http
GET /api/v1/videos/{sha256}/sentence-boundaries
```
Example using the demo record in the checked-in index:
```bash
curl http://127.0.0.1:8000/api/v1/videos/468a4d064f6ec49942b45e25ab93c500d31870f978c4f28ff8b3b408852326e0/sentence-boundaries
```
The MP4 used during development is also indexed. Its hash is
`b6631d5cf48f37fed0ecc623563dd48b7ed660689b4d25d2ebac7fecab807ddc`, and its
generated index contains 244 boundaries.
The response is:
```json
{
"video_hash": "...",
"duration_ms": 16000,
"algorithm_version": "silence-rms-v1",
"sentences": [
{
"index": 0,
"start_ms": 0,
"end_ms": 4230,
"text": null
}
]
}
```
Unknown hashes return `404`. A hash must be a 64-character hexadecimal
SHA-256 digest; malformed values return `422`.
When testing from a physical Android phone, replace `127.0.0.1` with the
computer's LAN IP address. `127.0.0.1` on the phone refers to the phone itself.
For production, expose the API over HTTPS.
## Android Request Flow
The mobile app should calculate the hash from the selected `content://` URI in
streaming chunks, request the endpoint, map the returned `sentences` to
`SentenceBoundary`, and then call `controller.loadItem`. The hash must be
calculated from the exact bytes of the same video served to the player. For
HTTPS course videos, the course manifest can carry the hash and avoid hashing
the entire remote file on every device.

2
sentence_api/__init__.py Normal file
View File

@@ -0,0 +1,2 @@
"""Sentence boundary lookup service."""

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,116 @@
import argparse
import hashlib
import json
from pathlib import Path
from typing import Any, Dict, List, Optional, Tuple
from sentence_analysis import detect_sentence_boundaries
from sentence_api.store import normalize_video_hash
ALGORITHM_VERSION = "silence-rms-v1"
DEFAULT_INDEX_PATH = Path(__file__).resolve().parent / "data" / "sentence_boundaries.json"
def sha256_file(path: Path) -> str:
digest = hashlib.sha256()
with path.open("rb") as media_file:
for chunk in iter(lambda: media_file.read(1024 * 1024), b""):
digest.update(chunk)
return digest.hexdigest()
def media_duration_ms(path: Path) -> int:
try:
import av
except ImportError as exc:
raise RuntimeError("PyAV is required to generate sentence boundaries.") from exc
container = av.open(str(path))
try:
if not container.duration or container.duration <= 0:
raise RuntimeError("Could not determine media duration.")
return int(round(container.duration / 1000))
finally:
container.close()
def make_entry(path: Path, video_hash: Optional[str] = None, min_silence: float = 0.30,
min_sentence: float = 0.35) -> Tuple[Dict[str, Any], str]:
duration_ms = media_duration_ms(path)
starts = detect_sentence_boundaries(
path,
min_silence=min_silence,
min_sentence=min_sentence,
)
if not starts:
raise RuntimeError("No sentence boundaries could be detected.")
starts_ms = sorted({max(0, int(round(start * 1000))) for start in starts})
starts_ms = [start for start in starts_ms if start < duration_ms]
sentences: List[Dict[str, Any]] = []
for index, start_ms in enumerate(starts_ms):
end_ms = starts_ms[index + 1] if index + 1 < len(starts_ms) else duration_ms
if end_ms <= start_ms:
continue
sentences.append({
"index": len(sentences),
"start_ms": start_ms,
"end_ms": end_ms,
"text": None,
})
if not sentences:
raise RuntimeError("Detected boundaries do not form valid sentence ranges.")
entry = {
"duration_ms": duration_ms,
"algorithm_version": ALGORITHM_VERSION,
"sentences": sentences,
}
resolved_hash = normalize_video_hash(video_hash) if video_hash else sha256_file(path)
return entry, resolved_hash
def update_index(index_path: Path, video_hash: str, entry: Dict[str, Any]) -> None:
if index_path.exists():
raw = json.loads(index_path.read_text(encoding="utf-8"))
else:
raw = {"videos": {}}
videos = raw.setdefault("videos", {})
if not isinstance(videos, dict):
raise ValueError("index JSON must contain a 'videos' object")
videos[normalize_video_hash(video_hash)] = entry
index_path.parent.mkdir(parents=True, exist_ok=True)
temporary_path = index_path.with_suffix(index_path.suffix + ".tmp")
temporary_path.write_text(
json.dumps(raw, ensure_ascii=False, indent=2) + "\n",
encoding="utf-8",
)
temporary_path.replace(index_path)
def main() -> None:
parser = argparse.ArgumentParser(description="Generate API sentence boundaries for a video.")
parser.add_argument("video", type=Path)
parser.add_argument("--index", type=Path, default=DEFAULT_INDEX_PATH)
parser.add_argument("--video-hash", help="Override the calculated SHA-256 hash.")
parser.add_argument("--min-silence", type=float, default=0.30)
parser.add_argument("--min-sentence", type=float, default=0.35)
args = parser.parse_args()
if not args.video.is_file():
parser.error(f"Video file does not exist: {args.video}")
entry, video_hash = make_entry(
args.video,
video_hash=args.video_hash,
min_silence=args.min_silence,
min_sentence=args.min_sentence,
)
update_index(args.index, video_hash, entry)
print(f"video_hash={video_hash}")
print(f"sentences={len(entry['sentences'])}")
print(f"index={args.index}")
if __name__ == "__main__":
main()

62
sentence_api/main.py Normal file
View File

@@ -0,0 +1,62 @@
import os
from pathlib import Path
from typing import Any, Dict
from fastapi import Depends, FastAPI, HTTPException, Path as ApiPath
from .models import SentenceBoundaryDocument
from .store import BoundaryStore
DEFAULT_INDEX_PATH = Path(__file__).resolve().parent / "data" / "sentence_boundaries.json"
def create_app(store: BoundaryStore = None) -> FastAPI:
application = FastAPI(
title="Oral Trainer Sentence Boundary API",
version="1.0.0",
description="Looks up pre-generated sentence boundaries by video SHA-256.",
)
index_path = Path(os.getenv("SENTENCE_BOUNDARIES_FILE", str(DEFAULT_INDEX_PATH)))
application.state.boundary_store = store or BoundaryStore(index_path)
def get_store() -> BoundaryStore:
return application.state.boundary_store
@application.get("/healthz")
def healthz(boundary_store: BoundaryStore = Depends(get_store)) -> Dict[str, Any]:
return {"status": "ok", "video_count": boundary_store.count()}
@application.get(
"/api/v1/videos/{video_hash}/sentence-boundaries",
response_model=SentenceBoundaryDocument,
)
@application.get(
"/api/v1/sentence-boundaries/{video_hash}",
response_model=SentenceBoundaryDocument,
include_in_schema=False,
)
def get_sentence_boundaries(
video_hash: str = ApiPath(
min_length=64,
max_length=64,
pattern=r"^[A-Fa-f0-9]{64}$",
description="SHA-256 hex digest of the exact video bytes",
),
boundary_store: BoundaryStore = Depends(get_store),
) -> SentenceBoundaryDocument:
document = boundary_store.get(video_hash)
if document is None:
raise HTTPException(
status_code=404,
detail={
"code": "SENTENCE_BOUNDARIES_NOT_FOUND",
"message": "No sentence boundaries are registered for this video hash.",
},
)
return document
return application
app = create_app()

48
sentence_api/models.py Normal file
View File

@@ -0,0 +1,48 @@
from typing import List, Optional
from pydantic import BaseModel, ConfigDict, Field, field_validator, model_validator
class SentenceBoundary(BaseModel):
model_config = ConfigDict(extra="forbid")
index: int = Field(ge=0)
start_ms: int = Field(ge=0)
end_ms: int = Field(gt=0)
text: Optional[str] = None
@model_validator(mode="after")
def validate_range(self):
if self.end_ms <= self.start_ms:
raise ValueError("end_ms must be greater than start_ms")
return self
class SentenceBoundaryDocument(BaseModel):
model_config = ConfigDict(extra="forbid")
video_hash: str
duration_ms: int = Field(gt=0)
algorithm_version: str = Field(min_length=1)
sentences: List[SentenceBoundary]
@field_validator("video_hash")
@classmethod
def validate_video_hash(cls, value: str) -> str:
normalized = value.strip().lower()
if len(normalized) != 64 or any(char not in "0123456789abcdef" for char in normalized):
raise ValueError("video_hash must be a SHA-256 hex digest")
return normalized
@model_validator(mode="after")
def validate_sentences(self):
previous_end = 0
for expected_index, sentence in enumerate(self.sentences):
if sentence.index != expected_index:
raise ValueError("sentence indexes must be contiguous and zero-based")
if sentence.start_ms < previous_end:
raise ValueError("sentences must be sorted and non-overlapping")
if sentence.end_ms > self.duration_ms:
raise ValueError("sentence end_ms cannot exceed duration_ms")
previous_end = sentence.end_ms
return self

View File

@@ -0,0 +1,7 @@
fastapi>=0.115,<1
uvicorn[standard]>=0.30,<1
pydantic>=2.7,<3
pytest>=8,<9
httpx>=0.27,<1
av>=12.0
numpy>=1.26

54
sentence_api/store.py Normal file
View File

@@ -0,0 +1,54 @@
import json
from pathlib import Path
from threading import RLock
from typing import Dict, Optional
from .models import SentenceBoundaryDocument
def normalize_video_hash(video_hash: str) -> str:
normalized = video_hash.strip().lower()
if len(normalized) != 64 or any(char not in "0123456789abcdef" for char in normalized):
raise ValueError("video_hash must be a SHA-256 hex digest")
return normalized
class BoundaryStore:
"""Read-only in-memory index loaded from a JSON file."""
def __init__(self, index_path: Path):
self.index_path = Path(index_path)
self._lock = RLock()
self._documents: Dict[str, SentenceBoundaryDocument] = {}
self.reload()
def reload(self) -> None:
raw = json.loads(self.index_path.read_text(encoding="utf-8"))
entries = raw.get("videos", raw)
if not isinstance(entries, dict):
raise ValueError("index JSON must contain a 'videos' object")
documents: Dict[str, SentenceBoundaryDocument] = {}
for video_hash, entry in entries.items():
normalized_hash = normalize_video_hash(video_hash)
if not isinstance(entry, dict):
raise ValueError(f"entry for {normalized_hash} must be an object")
document = SentenceBoundaryDocument(
video_hash=normalized_hash,
duration_ms=entry.get("duration_ms"),
algorithm_version=entry.get("algorithm_version", "unknown"),
sentences=entry.get("sentences", []),
)
documents[normalized_hash] = document
with self._lock:
self._documents = documents
def get(self, video_hash: str) -> Optional[SentenceBoundaryDocument]:
normalized_hash = normalize_video_hash(video_hash)
with self._lock:
return self._documents.get(normalized_hash)
def count(self) -> int:
with self._lock:
return len(self._documents)

View File

@@ -0,0 +1,55 @@
import json
from fastapi.testclient import TestClient
from sentence_api.main import create_app
from sentence_api.store import BoundaryStore
VIDEO_HASH = "a" * 64
def make_client(tmp_path):
index_path = tmp_path / "boundaries.json"
index_path.write_text(
json.dumps({
"videos": {
VIDEO_HASH: {
"duration_ms": 9000,
"algorithm_version": "test-v1",
"sentences": [
{"index": 0, "start_ms": 0, "end_ms": 4000, "text": "One."},
{"index": 1, "start_ms": 4000, "end_ms": 9000, "text": "Two."},
],
}
}
}),
encoding="utf-8",
)
return TestClient(create_app(BoundaryStore(index_path)))
def test_lookup_returns_boundaries(tmp_path):
client = make_client(tmp_path)
response = client.get(f"/api/v1/videos/{VIDEO_HASH}/sentence-boundaries")
assert response.status_code == 200
assert response.json()["sentences"][1]["start_ms"] == 4000
def test_unknown_hash_returns_not_found(tmp_path):
client = make_client(tmp_path)
response = client.get(f"/api/v1/videos/{'b' * 64}/sentence-boundaries")
assert response.status_code == 404
assert response.json()["detail"]["code"] == "SENTENCE_BOUNDARIES_NOT_FOUND"
def test_invalid_hash_is_rejected(tmp_path):
client = make_client(tmp_path)
response = client.get("/api/v1/videos/not-a-sha256/sentence-boundaries")
assert response.status_code == 422