111 lines
4.1 KiB
Plaintext
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""
POST /api/calibrate
字幕校准服务 - 用LLM将WhisperX识别的歌词时间戳与原始歌词对齐
Parameters:
original_lyrics: 原始歌词(准确文字)
whisperx_json: WhisperX输出的JSON时间戳准文字不准
Returns:
JSON with calibrated subtitles (accurate text + precise timestamps)
"""
import json
import os
import sys
CALIBRATE_PROMPT = """你是一个专业的歌词字幕校准专家。
任务将WhisperX语音识别输出的时间戳与原始歌词文字进行精确对齐。
规则:
1. 保留WhisperX输出的所有时间戳start/end这些时间是准确的
2. 将WhisperX识别的文字替换为原始歌词中对应的文字
3. 按段落顺序匹配WhisperX的第N段对应原歌词的第N段
4. 处理副歌重复如果WhisperX识别出重复段落映射到同一歌词段落
5. 忽略纯音乐段落(无歌词的时间段)
输出格式严格JSON
{
"segments": [
{
"text": "校准后的歌词文字",
"start": 1.234,
"end": 4.567,
"chars": [
{"char": "爱", "start": 1.234, "end": 1.500},
{"char": "上", "start": 1.500, "end": 1.800}
]
}
]
}
原始歌词:
{original_lyrics}
WhisperX输出JSON
{whisperx_json}
请输出校准后的JSON不要markdown代码块直接输出JSON
"""
try:
original_lyrics = params_kw.get('original_lyrics', '')
whisperx_json = params_kw.get('whisperx_json', '')
if not original_lyrics or not whisperx_json:
result = json.dumps({"status": "error", "error": "missing original_lyrics or whisperx_json"}, ensure_ascii=False)
else:
# Build the prompt
prompt = CALIBRATE_PROMPT.format(
original_lyrics=original_lyrics,
whisperx_json=whisperx_json
)
# Call LLM via Sage llmage API
import aiohttp
LLM_API_BASE = os.environ.get('LLM_API_BASE', 'https://token.opencomputing.cn/llmage/v1')
LLM_API_KEY = os.environ.get('LLM_API_KEY', '')
if not LLM_API_KEY:
# Try to get from config
from ahserver.serverenv import ServerEnv
env = ServerEnv()
LLM_API_KEY = getattr(env, 'llm_api_key', '') or ''
async with aiohttp.ClientSession() as session:
payload = {
"model": "qwen3-235b-a22b",
"catelogid": "t2t",
"messages": [{"role": "user", "content": prompt}],
"temperature": 0.1,
"max_tokens": 4096
}
headers = {
"Authorization": f"Bearer {LLM_API_KEY}",
"Content-Type": "application/json"
}
async with session.post(f"{LLM_API_BASE}/chat/completions", json=payload, headers=headers, timeout=120) as resp:
if resp.status == 200:
data = await resp.json()
content = data.get('choices', [{}])[0].get('message', {}).get('content', '')
# Parse the LLM response
content = content.strip()
if content.startswith('```'):
content = content.split('```')[1]
if content.startswith('json'):
content = content[4:]
content = content.strip()
try:
calibrated = json.loads(content)
result = json.dumps({"status": "success", "data": calibrated}, ensure_ascii=False)
except json.JSONDecodeError:
result = json.dumps({"status": "error", "error": "LLM response not valid JSON", "raw": content[:500]}, ensure_ascii=False)
else:
text = await resp.text()
result = json.dumps({"status": "error", "error": f"LLM API returned {resp.status}", "detail": text[:300]}, ensure_ascii=False)
except Exception as e:
import traceback
result = json.dumps({"status": "error", "error": str(e), "traceback": traceback.format_exc()}, ensure_ascii=False)