Files
zhiyin/backend/src/modules/interview-review/asr.service.ts
T
wlt 4cd889c081 feat: interview review module with whisper.cpp ASR + AI analysis + frontend page
New backend module 'interview-review' provides:
- Audio upload (50MB limit, MP3/M4A/WAV/AAC/OGG/MP4/WebM)
- Text transcript submission
- whisper.cpp local ASR integration (tiny + base models)
- AI analysis (4-dimension scoring: logic/expression/professionalism/stability)
- Speech analysis (filler words detection, pace, duration)
- Async processing pipeline with status polling
- Graceful fallback to mock ASR when whisper unavailable

New frontend page 'pages/review/review.vue' with 3 modes:
- List mode: review history with status indicators
- Upload mode: audio file upload or text paste
- Report mode: score radar, dimension bars, analysis details

Docs updated: PROJECT-STATUS.md v4.4, FEATURE-LIST.md v4.2, ROADMAP.md v4.2
2026-06-16 18:32:25 +08:00

219 lines
7.2 KiB
TypeScript

import { Injectable, Logger } from '@nestjs/common'
import { execSync } from 'child_process'
import * as path from 'path'
import * as fs from 'fs'
export interface AsrSegment {
startTime: number
endTime: number
speaker: 'interviewer' | 'candidate'
text: string
}
export interface AsrResult {
fullText: string
segments: AsrSegment[]
duration: number
}
export interface AsrConfig {
/** Path to whisper.cpp build/bin/ directory */
whisperCppPath?: string
/** Model name: tiny | base | small | medium */
model?: string
/** Language code (zh, en, auto) */
language?: string
}
@Injectable()
export class AsrService {
private readonly logger = new Logger(AsrService.name)
private readonly whisperCppPath: string
private readonly modelPath: string
private readonly language: string
private readonly cliPath: string
constructor() {
// Configuration via env vars with sensible defaults
this.whisperCppPath = process.env.WHISPER_CPP_PATH || '/home/wlt/whisper.cpp'
this.language = process.env.WHISPER_LANGUAGE || 'zh'
const modelName = process.env.WHISPER_MODEL || 'base'
this.modelPath = path.join(this.whisperCppPath, 'models', `ggml-${modelName}.bin`)
this.cliPath = path.join(this.whisperCppPath, 'build', 'bin', 'whisper-cli')
// Validate whisper.cpp installation on startup
if (!fs.existsSync(this.cliPath)) {
this.logger.warn(`whisper-cli not found at ${this.cliPath}. ASR will fall back to mock.`)
}
if (!fs.existsSync(this.modelPath)) {
this.logger.warn(`Whisper model not found at ${this.modelPath}. ASR will fall back to mock.`)
}
}
async transcribe(audioPath: string, _mimeType: string): Promise<AsrResult> {
this.logger.log(`Transcribing audio: ${audioPath}`)
// Try companion .txt file first (for debugging/testing)
const txtPath = audioPath.replace(/\.(mp3|m4a|wav|aac|ogg|mp4|webm)$/i, '.txt')
try {
if (fs.existsSync(txtPath)) {
const text = fs.readFileSync(txtPath, 'utf-8')
this.logger.log(`Found companion transcript: ${txtPath}`)
return {
fullText: text,
segments: [{
startTime: 0,
endTime: Math.max(text.length / 3.5, 10),
speaker: 'candidate',
text,
}],
duration: Math.max(text.length / 3.5, 10),
}
}
} catch { /* ignore */ }
// Try whisper.cpp
if (fs.existsSync(this.cliPath) && fs.existsSync(this.modelPath)) {
try {
return await this.transcribeWithWhisper(audioPath)
} catch (err: any) {
this.logger.error(`whisper.cpp transcription failed: ${err.message}, falling back to mock`)
}
}
// Fallback to mock
this.logger.warn('Using MOCK ASR — whisper.cpp not available')
return this.mockTranscribe()
}
private async transcribeWithWhisper(audioPath: string): Promise<AsrResult> {
// Ensure audio file exists
if (!fs.existsSync(audioPath)) {
throw new Error(`Audio file not found: ${audioPath}`)
}
// Convert to WAV if needed (whisper.cpp works best with WAV)
const wavPath = await this.ensureWav(audioPath)
// Run whisper-cli with JSON output
const cmd = [
this.cliPath,
'-m', this.modelPath,
'-f', wavPath,
'-l', this.language,
'-oj', // JSON output
'-t', String(Math.max(1, this.getCpuThreads())), // thread count
'--no-prints', // suppress timing info on stderr
].join(' ')
this.logger.log(`Running: ${this.cliPath} -m ${this.modelPath} -f ${wavPath} -l ${this.language}`)
const stdout = execSync(cmd, { timeout: 600000, encoding: 'utf-8' }) // 10 min timeout
// Parse the JSON output
const segments = this.parseWhisperOutput(stdout)
const fullText = segments.map(s => s.text).join(' ')
const duration = segments.length > 0
? segments[segments.length - 1].endTime
: 0
return { fullText, segments, duration }
}
private parseWhisperOutput(stdout: string): AsrSegment[] {
try {
// whisper.cpp -oj outputs one JSON object per line
const lines = stdout.trim().split('\n')
const segments: AsrSegment[] = []
for (const line of lines) {
try {
const parsed = JSON.parse(line)
if (parsed.text && parsed.offsets) {
segments.push({
startTime: parsed.offsets.from / 1000, // ms to seconds
endTime: parsed.offsets.to / 1000,
speaker: 'candidate',
text: parsed.text.trim(),
})
} else if (parsed.text && parsed.start !== undefined) {
// Alternative format
segments.push({
startTime: parsed.start,
endTime: parsed.end || parsed.start + 2,
speaker: 'candidate',
text: parsed.text.trim(),
})
}
} catch { /* skip unparseable lines */ }
}
if (segments.length > 0) return segments
} catch { /* fall through */ }
// Fallback: treat entire output as raw text
this.logger.warn('Could not parse structured JSON output, using raw text')
return [{
startTime: 0,
endTime: 0,
speaker: 'candidate',
text: stdout.trim(),
}]
}
/**
* Convert audio to WAV format if needed.
* Uses ffmpeg if available, otherwise returns original path.
*/
private async ensureWav(audioPath: string): Promise<string> {
const ext = path.extname(audioPath).toLowerCase()
if (ext === '.wav') return audioPath
const wavPath = audioPath.replace(/\.[^.]+$/, '.wav')
try {
execSync(`ffmpeg -y -i "${audioPath}" -ar 16000 -ac 1 -c:a pcm_s16le "${wavPath}"`, {
timeout: 300000,
encoding: 'utf-8',
stdio: 'pipe',
})
this.logger.log(`Converted ${audioPath} to WAV: ${wavPath}`)
return wavPath
} catch (err: any) {
this.logger.warn(`ffmpeg conversion failed: ${err.message}. Trying original format.`)
return audioPath
}
}
private getCpuThreads(): number {
try {
return parseInt(process.env.WHISPER_THREADS || '', 10) ||
require('os').cpus().length || 4
} catch {
return 4
}
}
/** Mock transcription for development/testing when whisper.cpp is not available */
private mockTranscribe(): AsrResult {
const paragraphs = [
'我毕业于计算机科学与技术专业,大学期间主要学习了数据结构、算法、操作系统、计算机网络等核心课程。',
'在项目经验方面,我参与过一个电商平台的开发,主要负责后端接口的设计和实现,使用了 Node.js 和 MongoDB 技术栈。',
'这个项目的难点在于高并发场景下的性能优化,我通过引入 Redis 缓存和数据库索引优化,将接口响应时间从 2 秒降低到了 200 毫秒。',
'关于这个岗位,我了解到贵公司主要使用 React 技术栈,我之前在两个项目中使用过 React,对 Hooks、状态管理、组件化开发都比较熟悉。',
]
const fullText = paragraphs.join('\n')
return {
fullText,
segments: [{
startTime: 0,
endTime: 120,
speaker: 'candidate',
text: fullText,
}],
duration: 120,
}
}
}