Initial clean project import
This commit is contained in:
@@ -0,0 +1,468 @@
|
||||
import axios from 'axios';
|
||||
import * as fs from 'fs';
|
||||
import * as path from 'path';
|
||||
import { app } from 'electron';
|
||||
import logger from '../log/main';
|
||||
import { Apps } from '../app/index';
|
||||
import { withoutUnsupportedProxy } from '../httpClient';
|
||||
|
||||
/**
|
||||
* 在线 TTS(千问 Qwen TTS)客户端
|
||||
*/
|
||||
|
||||
export interface CosyVoiceConfig {
|
||||
apiKey: string;
|
||||
model: string;
|
||||
voice: string;
|
||||
format?: string;
|
||||
sampleRate?: number;
|
||||
rate?: number;
|
||||
volume?: number;
|
||||
pitch?: number;
|
||||
instruction?: string;
|
||||
language_hints?: string;
|
||||
language?: string;
|
||||
emotion?: string;
|
||||
outputPath?: string;
|
||||
}
|
||||
|
||||
export interface CosyVoiceResult {
|
||||
success: boolean;
|
||||
audioPath?: string;
|
||||
error?: string;
|
||||
requestId?: string;
|
||||
}
|
||||
|
||||
const QWEN_TTS_URL = 'https://dashscope.aliyuncs.com/api/v1/services/aigc/multimodal-generation/generation';
|
||||
|
||||
function getOutputFormat(outputPath: string, configFormat?: string): 'wav' | 'mp3' {
|
||||
const ext = path.extname(outputPath).toLowerCase();
|
||||
if (ext === '.wav') return 'wav';
|
||||
if (ext === '.mp3') return 'mp3';
|
||||
if ((configFormat || '').toLowerCase() === 'wav') return 'wav';
|
||||
return 'mp3';
|
||||
}
|
||||
|
||||
function detectAudioFormat(
|
||||
buffer: Buffer,
|
||||
contentType?: string,
|
||||
fallback: 'wav' | 'mp3' = 'mp3'
|
||||
): 'wav' | 'mp3' {
|
||||
const normalizedType = (contentType || '').toLowerCase();
|
||||
if (normalizedType.includes('wav') || normalizedType.includes('wave')) return 'wav';
|
||||
if (normalizedType.includes('mpeg') || normalizedType.includes('mp3')) return 'mp3';
|
||||
|
||||
if (buffer.length >= 12) {
|
||||
const riff = buffer.toString('ascii', 0, 4);
|
||||
const wave = buffer.toString('ascii', 8, 12);
|
||||
if (riff === 'RIFF' && wave === 'WAVE') return 'wav';
|
||||
}
|
||||
|
||||
if (buffer.length >= 3 && buffer.toString('ascii', 0, 3) === 'ID3') return 'mp3';
|
||||
if (buffer.length >= 2 && buffer[0] === 0xff && (buffer[1] & 0xe0) === 0xe0) return 'mp3';
|
||||
|
||||
return fallback;
|
||||
}
|
||||
|
||||
async function transcodeAudioFile(
|
||||
inputPath: string,
|
||||
outputPath: string,
|
||||
targetFormat: 'wav' | 'mp3'
|
||||
): Promise<void> {
|
||||
const args = targetFormat === 'wav'
|
||||
? ['-y', '-i', inputPath, '-c:a', 'pcm_s16le', outputPath]
|
||||
: ['-y', '-i', inputPath, '-c:a', 'libmp3lame', '-b:a', '128k', outputPath];
|
||||
|
||||
await Apps.spawnBinary('ffmpeg', args, {
|
||||
shell: false,
|
||||
});
|
||||
}
|
||||
|
||||
async function writeAudioBufferAsTargetFormat(
|
||||
audioBuffer: Buffer,
|
||||
outputPath: string,
|
||||
targetFormat: 'wav' | 'mp3',
|
||||
contentType?: string
|
||||
): Promise<void> {
|
||||
const sourceFormat = detectAudioFormat(audioBuffer, contentType, targetFormat);
|
||||
if (sourceFormat === targetFormat) {
|
||||
fs.writeFileSync(outputPath, audioBuffer);
|
||||
return;
|
||||
}
|
||||
|
||||
const parsedOutput = path.parse(outputPath);
|
||||
const tempInputPath = path.join(
|
||||
parsedOutput.dir,
|
||||
`${parsedOutput.name}_source_${Date.now()}${Math.random().toString(36).slice(2, 8)}.${sourceFormat}`
|
||||
);
|
||||
|
||||
fs.writeFileSync(tempInputPath, audioBuffer);
|
||||
try {
|
||||
await transcodeAudioFile(tempInputPath, outputPath, targetFormat);
|
||||
} finally {
|
||||
try { fs.unlinkSync(tempInputPath); } catch (_error) { /* ignore */ }
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* 将长文本按自然断句分割为不超过 maxLen 的段落
|
||||
* qwen3-tts-vc 限制每次最多600(含 instructions),保守设为300
|
||||
* 普通 qwen3-tts 模型可适当放宽
|
||||
*/
|
||||
function splitTextForTTS(text: string, maxLen: number = 300): string[] {
|
||||
if (text.length <= maxLen) return [text];
|
||||
|
||||
const segments: string[] = [];
|
||||
let remaining = text;
|
||||
|
||||
while (remaining.length > 0) {
|
||||
if (remaining.length <= maxLen) {
|
||||
segments.push(remaining);
|
||||
break;
|
||||
}
|
||||
// 在 maxLen 范围内找最后一个自然断句点
|
||||
const chunk = remaining.slice(0, maxLen);
|
||||
let splitIdx = -1;
|
||||
// 优先按句号/问号/感叹号分割
|
||||
for (const sep of ['。', '!', '?', '!', '?', ';', ';', '\n']) {
|
||||
const idx = chunk.lastIndexOf(sep);
|
||||
if (idx > maxLen * 0.3) { splitIdx = idx + 1; break; }
|
||||
}
|
||||
// 其次按逗号/顿号分割
|
||||
if (splitIdx === -1) {
|
||||
for (const sep of [',', ',', '、', ':', ':']) {
|
||||
const idx = chunk.lastIndexOf(sep);
|
||||
if (idx > maxLen * 0.3) { splitIdx = idx + 1; break; }
|
||||
}
|
||||
}
|
||||
// 实在找不到就硬切
|
||||
if (splitIdx === -1) splitIdx = maxLen;
|
||||
|
||||
segments.push(remaining.slice(0, splitIdx).trim());
|
||||
remaining = remaining.slice(splitIdx).trim();
|
||||
}
|
||||
|
||||
return segments.filter(s => s.length > 0);
|
||||
}
|
||||
|
||||
export async function synthesizeSpeech(
|
||||
text: string,
|
||||
config: CosyVoiceConfig
|
||||
): Promise<CosyVoiceResult> {
|
||||
logger.info('[QwenTTS] 开始语音合成', {
|
||||
model: config.model,
|
||||
voice: config.voice,
|
||||
instruction: config.instruction,
|
||||
language: config.language,
|
||||
language_hints: config.language_hints,
|
||||
textLength: text.length,
|
||||
});
|
||||
|
||||
try {
|
||||
// ✅ 关键修复:长文本自动分段(避免 "Range of input length should be [0, 600]" 错误)
|
||||
// VC 模型(声音克隆)的限制更严格,API 可能将 text + instructions 合并计算
|
||||
const instructionLen = config.instruction ? config.instruction.length : 0;
|
||||
const isVcModel = config.model?.includes('-vc') || config.model?.includes('_vc');
|
||||
// VC 模型更保守(200字),普通模型 300 字,减去 instruction 长度
|
||||
const baseMaxLen = isVcModel ? 200 : 300;
|
||||
const effectiveMaxLen = Math.max(100, baseMaxLen - instructionLen);
|
||||
const segments = splitTextForTTS(text, effectiveMaxLen);
|
||||
logger.info(`[QwenTTS] 文本分段: ${segments.length}段, 各段长度: [${segments.map(s => s.length).join(', ')}]`, {
|
||||
isVcModel, instructionLen, effectiveMaxLen
|
||||
});
|
||||
|
||||
// 准备输出路径
|
||||
let outputPath: string;
|
||||
if (config.outputPath) {
|
||||
outputPath = config.outputPath;
|
||||
const outputDir = path.dirname(outputPath);
|
||||
if (!fs.existsSync(outputDir)) {
|
||||
fs.mkdirSync(outputDir, { recursive: true });
|
||||
}
|
||||
} else {
|
||||
const outputDir = path.join(app.getPath('userData'), 'temp', 'tts');
|
||||
if (!fs.existsSync(outputDir)) {
|
||||
fs.mkdirSync(outputDir, { recursive: true });
|
||||
}
|
||||
const defaultExt = (config.format || '').toLowerCase() === 'wav' ? 'wav' : 'mp3';
|
||||
outputPath = path.join(outputDir, `qwen_tts_${Date.now()}.${defaultExt}`);
|
||||
}
|
||||
|
||||
const outputFormat = getOutputFormat(outputPath, config.format);
|
||||
|
||||
// 如果只有一段,直接合成
|
||||
if (segments.length === 1) {
|
||||
return await synthesizeOneSegment(segments[0], config, outputPath);
|
||||
}
|
||||
|
||||
// 多段:逐段合成,然后拼接 WAV
|
||||
const segmentPaths: string[] = [];
|
||||
const parsedOutput = path.parse(outputPath);
|
||||
for (let i = 0; i < segments.length; i++) {
|
||||
const segPath = path.join(
|
||||
parsedOutput.dir,
|
||||
`${parsedOutput.name}_seg${i}${parsedOutput.ext || (outputFormat === 'wav' ? '.wav' : '.mp3')}`
|
||||
);
|
||||
logger.info(`[QwenTTS] 合成第 ${i + 1}/${segments.length} 段 (${segments[i].length}字)...`);
|
||||
const segResult = await synthesizeOneSegment(segments[i], config, segPath);
|
||||
if (!segResult.success) {
|
||||
return segResult; // 任一段失败则返回错误
|
||||
}
|
||||
segmentPaths.push(segPath);
|
||||
}
|
||||
|
||||
// 拼接所有段的 WAV 文件(简单拼接 PCM 数据)
|
||||
try {
|
||||
await concatAudioFiles(segmentPaths, outputPath, outputFormat);
|
||||
// 清理临时分段文件
|
||||
for (const p of segmentPaths) {
|
||||
try { fs.unlinkSync(p); } catch (e) { /* ignore */ }
|
||||
}
|
||||
logger.info('[QwenTTS] 多段音频拼接完成', { outputPath, segments: segments.length });
|
||||
} catch (concatError) {
|
||||
// 拼接失败则使用第一段
|
||||
logger.error('[QwenTTS] 音频拼接失败,使用第一段', concatError);
|
||||
if (segmentPaths.length > 0 && fs.existsSync(segmentPaths[0])) {
|
||||
fs.copyFileSync(segmentPaths[0], outputPath);
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
success: true,
|
||||
audioPath: outputPath,
|
||||
};
|
||||
|
||||
} catch (error: any) {
|
||||
logger.error('[QwenTTS] 语音合成失败', error.response?.data || error.message || error);
|
||||
return {
|
||||
success: false,
|
||||
error: error.response?.data?.message || error.message || '语音合成失败',
|
||||
};
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* 合成单段文本
|
||||
*/
|
||||
async function synthesizeOneSegment(
|
||||
text: string,
|
||||
config: CosyVoiceConfig,
|
||||
outputPath: string
|
||||
): Promise<CosyVoiceResult> {
|
||||
const instrLen = config.instruction ? config.instruction.length : 0;
|
||||
const outputFormat = getOutputFormat(outputPath, config.format);
|
||||
logger.info(`[QwenTTS] synthesizeOneSegment: textLen=${text.length}, instructionLen=${instrLen}, totalLen=${text.length + instrLen}, model=${config.model}`);
|
||||
|
||||
const response = await axios.post(
|
||||
QWEN_TTS_URL,
|
||||
{
|
||||
model: config.model,
|
||||
input: {
|
||||
text,
|
||||
voice: config.voice,
|
||||
...(mapLanguageType(config.language_hints || config.language)
|
||||
? { language_type: mapLanguageType(config.language_hints || config.language) }
|
||||
: {}),
|
||||
...(config.instruction ? { instructions: config.instruction, optimize_instructions: true } : {}),
|
||||
},
|
||||
},
|
||||
withoutUnsupportedProxy({
|
||||
headers: {
|
||||
Authorization: `Bearer ${config.apiKey}`,
|
||||
'Content-Type': 'application/json',
|
||||
},
|
||||
timeout: 600000,
|
||||
})
|
||||
);
|
||||
|
||||
const audioUrl = response.data?.output?.audio?.url;
|
||||
const requestId = response.data?.request_id;
|
||||
if (!audioUrl) {
|
||||
return {
|
||||
success: false,
|
||||
error: response.data?.message || '未获取到音频 URL',
|
||||
requestId,
|
||||
};
|
||||
}
|
||||
|
||||
const audioResponse = await axios.get(audioUrl, withoutUnsupportedProxy({
|
||||
responseType: 'arraybuffer',
|
||||
timeout: 600000,
|
||||
}));
|
||||
await writeAudioBufferAsTargetFormat(
|
||||
Buffer.from(audioResponse.data),
|
||||
outputPath,
|
||||
outputFormat,
|
||||
audioResponse.headers?.['content-type']
|
||||
);
|
||||
|
||||
logger.info('[QwenTTS] 单段音频已保存', { outputPath, requestId, textLen: text.length });
|
||||
return {
|
||||
success: true,
|
||||
audioPath: outputPath,
|
||||
requestId,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* 拼接多个 WAV 文件(假设格式相同:都是 PCM WAV)
|
||||
*/
|
||||
async function concatWavFiles(inputPaths: string[], outputPath: string): Promise<void> {
|
||||
if (inputPaths.length === 0) throw new Error('无输入文件');
|
||||
if (inputPaths.length === 1) {
|
||||
fs.copyFileSync(inputPaths[0], outputPath);
|
||||
return;
|
||||
}
|
||||
|
||||
// 读取第一个文件的 WAV 头(44字节)
|
||||
const firstFile = fs.readFileSync(inputPaths[0]);
|
||||
const headerSize = 44;
|
||||
const header = Buffer.from(firstFile.buffer, 0, headerSize);
|
||||
|
||||
// 收集所有 PCM 数据(跳过 WAV 头)
|
||||
const pcmBuffers: Buffer[] = [];
|
||||
let totalPcmSize = 0;
|
||||
for (const p of inputPaths) {
|
||||
const buf = fs.readFileSync(p);
|
||||
const pcm = buf.slice(headerSize);
|
||||
pcmBuffers.push(pcm);
|
||||
totalPcmSize += pcm.length;
|
||||
}
|
||||
|
||||
// 更新 WAV 头中的文件大小字段
|
||||
const newHeader = Buffer.from(header);
|
||||
newHeader.writeUInt32LE(totalPcmSize + headerSize - 8, 4); // ChunkSize
|
||||
newHeader.writeUInt32LE(totalPcmSize, 40); // Subchunk2Size
|
||||
|
||||
// 写入文件
|
||||
const outputBuf = Buffer.concat([newHeader, ...pcmBuffers]);
|
||||
fs.writeFileSync(outputPath, outputBuf);
|
||||
}
|
||||
|
||||
async function concatMp3Files(inputPaths: string[], outputPath: string): Promise<void> {
|
||||
const concatListPath = path.join(
|
||||
path.dirname(outputPath),
|
||||
`concat_${Date.now()}_${Math.random().toString(36).slice(2)}.txt`
|
||||
);
|
||||
const concatList = inputPaths
|
||||
.map(filePath => {
|
||||
const normalized = filePath.replace(/\\/g, '/').replace(/'/g, "'\\''");
|
||||
return `file '${normalized}'`;
|
||||
})
|
||||
.join('\n');
|
||||
|
||||
fs.writeFileSync(concatListPath, concatList, 'utf-8');
|
||||
try {
|
||||
await Apps.spawnBinary('ffmpeg', [
|
||||
'-y',
|
||||
'-f', 'concat',
|
||||
'-safe', '0',
|
||||
'-i', concatListPath,
|
||||
'-c:a', 'libmp3lame',
|
||||
'-b:a', '128k',
|
||||
outputPath,
|
||||
], {
|
||||
shell: false,
|
||||
});
|
||||
} finally {
|
||||
try { fs.unlinkSync(concatListPath); } catch (_error) { /* ignore */ }
|
||||
}
|
||||
}
|
||||
|
||||
async function concatAudioFiles(
|
||||
inputPaths: string[],
|
||||
outputPath: string,
|
||||
format: 'wav' | 'mp3'
|
||||
): Promise<void> {
|
||||
if (format === 'wav') {
|
||||
await concatWavFiles(inputPaths, outputPath);
|
||||
return;
|
||||
}
|
||||
|
||||
await concatMp3Files(inputPaths, outputPath);
|
||||
}
|
||||
|
||||
export function getAvailableVoices(model: string): Array<{ id: string; name: string; desc: string }> {
|
||||
if (model.startsWith('qwen3-tts') || model.startsWith('qwen-tts')) {
|
||||
return [
|
||||
{ id: 'Cherry', name: '芊悦', desc: '普通话,女声,通用播报' },
|
||||
{ id: 'Serena', name: '苏瑶', desc: '普通话,女声,温柔自然' },
|
||||
{ id: 'Ethan', name: '晨煦', desc: '普通话,男声,稳重清晰' },
|
||||
{ id: 'Jada', name: '上海-阿珍', desc: '上海话,女声,方言播报' },
|
||||
{ id: 'Dylan', name: '北京-晓东', desc: '北京话,男声,方言播报' },
|
||||
{ id: 'Li', name: '南京-老李', desc: '南京话,男声,方言播报' },
|
||||
{ id: 'Marcus', name: '陕西-秦川', desc: '陕西话,男声,方言播报' },
|
||||
{ id: 'Roy', name: '闽南-阿杰', desc: '闽南语,男声,方言播报' },
|
||||
{ id: 'Peter', name: '天津-李彼得', desc: '天津话,男声,方言播报' },
|
||||
{ id: 'Sunny', name: '四川-晴儿', desc: '四川话,女声,方言播报' },
|
||||
{ id: 'Eric', name: '四川-程川', desc: '四川话,男声,方言播报' },
|
||||
{ id: 'Rocky', name: '粤语-阿强', desc: '粤语,男声,方言播报' },
|
||||
{ id: 'Kiki', name: '粤语-阿清', desc: '粤语,女声,方言播报' },
|
||||
];
|
||||
}
|
||||
|
||||
return [];
|
||||
}
|
||||
|
||||
export function getInstructionForLanguage(language: string): string | undefined {
|
||||
if (!language || language === '中文' || language === '中文(普通话)') {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
const dialectMap: Record<string, string> = {
|
||||
'上海话': '请用上海话表达。',
|
||||
'北京话': '请用北京话表达。',
|
||||
'四川话': '请用四川话表达。',
|
||||
'南京话': '请用南京话表达。',
|
||||
'陕西话': '请用陕西话表达。',
|
||||
'闽南话': '请用闽南话表达。',
|
||||
'天津话': '请用天津话表达。',
|
||||
'粤语': '请用粤语表达。',
|
||||
'广东话': '请用粤语表达。',
|
||||
};
|
||||
if (dialectMap[language]) {
|
||||
return dialectMap[language];
|
||||
}
|
||||
|
||||
const languageMap: Record<string, string> = {
|
||||
'英语': 'Please speak in English.',
|
||||
'德语': 'Please speak in German.',
|
||||
'意大利语': 'Please speak in Italian.',
|
||||
'葡萄牙语': 'Please speak in Portuguese.',
|
||||
'西班牙语': 'Please speak in Spanish.',
|
||||
'日语': 'Please speak in Japanese.',
|
||||
'韩语': 'Please speak in Korean.',
|
||||
'法语': 'Please speak in French.',
|
||||
'俄语': 'Please speak in Russian.',
|
||||
};
|
||||
|
||||
return languageMap[language];
|
||||
}
|
||||
|
||||
function mapLanguageType(language?: string): string | undefined {
|
||||
const languageMap: Record<string, string> = {
|
||||
'中文': 'Chinese',
|
||||
'中文(普通话)': 'Chinese',
|
||||
'Chinese': 'Chinese',
|
||||
'英语': 'English',
|
||||
'English': 'English',
|
||||
'德语': 'German',
|
||||
'German': 'German',
|
||||
'意大利语': 'Italian',
|
||||
'Italian': 'Italian',
|
||||
'葡萄牙语': 'Portuguese',
|
||||
'Portuguese': 'Portuguese',
|
||||
'西班牙语': 'Spanish',
|
||||
'Spanish': 'Spanish',
|
||||
'日语': 'Japanese',
|
||||
'Japanese': 'Japanese',
|
||||
'韩语': 'Korean',
|
||||
'Korean': 'Korean',
|
||||
'法语': 'French',
|
||||
'French': 'French',
|
||||
'俄语': 'Russian',
|
||||
'Russian': 'Russian',
|
||||
};
|
||||
|
||||
return language ? languageMap[language] : undefined;
|
||||
}
|
||||
Reference in New Issue
Block a user