#!/usr/bin/env python3 """Pronunciation assessment engine — поверх Whisper API.""" from phonemizer import phonemize import json, re, math # ── Phoneme conversion ────────────────────────────────────────────── def text_to_phonemes(text, lang="it"): """text → list of phoneme strings.""" raw = phonemize(text, language=lang, backend="espeak", strip=True, preserve_punctuation=False, with_stress=False) # espeak returns space-separated phonemes, but some are multi-char # Parse carefully: split on spaces, merge tied phonemes tokens = raw.split() return tokens def phonemes_to_syllables(phonemes): """Group phonemes into rough syllables (vowel = nucleus).""" vowels = set("aeɛiouɔɑɒʌəɨʉɯʊɤeøɘɵɐœɶʏɪɞ") syllables = [] cur = [] for p in phonemes: cur.append(p) # Any vowel-ish character means syllable nucleus if any(c in vowels for c in p): pass # keep collecting consonants after vowel in same syllable # Simple heuristic: two consonants in a row → split before second # Simpler approach: just split on vowel positions result = [] buf = [] for p in phonemes: buf.append(p) has_vowel = any(c in vowels for c in p) if has_vowel: result.append("-".join(buf)) buf = [] if buf: if result: result[-1] = result[-1] + "-" + "-".join(buf) else: result.append("-".join(buf)) return result # ── Expected pronunciation model ──────────────────────────────────── def build_expected(text, lang="it"): """Build expected pronunciation model for given text.""" phonemes = text_to_phonemes(text, lang) syllables = phonemes_to_syllables(phonemes) return { "text": text, "language": lang, "phonemes": phonemes, "syllables": syllables, "phoneme_count": len(phonemes), "syllable_count": len(syllables), } # ── Timing analysis ───────────────────────────────────────────────── def analyze_timing(words, expected_duration=None): """Analyze word timing from Whisper output.""" if not words: return {"error": "no words"} times = [] for w in words: dur = w.get("end", 0) - w.get("start", 0) times.append({ "word": w.get("word", ""), "start": w.get("start", 0), "end": w.get("end", 0), "duration": dur, }) total = times[-1]["end"] - times[0]["start"] if times else 0 avg_speed = sum(t["duration"] for t in times) / len(times) if times else 0 # Rhythm score: lower std deviation = more natural rhythm durs = [t["duration"] for t in times] mean_dur = sum(durs) / len(durs) if durs else 1 variance = sum((d - mean_dur)**2 for d in durs) / len(durs) if durs else 0 rhythm_score = max(0, 100 - math.sqrt(variance) * 50) return { "words": times, "total_duration": total, "avg_word_duration": avg_speed, "rhythm_score": round(rhythm_score, 1), "timing_quality": "good" if rhythm_score > 70 else "ok" if rhythm_score > 40 else "poor", } # ── Phoneme comparison ────────────────────────────────────────────── def levenshtein_ops(a, b): """Levenshtein with backtrace — returns list of (op, char_a, char_b).""" m, n = len(a), len(b) dp = [[0]*(n+1) for _ in range(m+1)] for i in range(m+1): dp[i][0] = i for j in range(n+1): dp[0][j] = j for i in range(1, m+1): for j in range(1, n+1): cost = 0 if a[i-1] == b[j-1] else 1 dp[i][j] = min(dp[i-1][j]+1, dp[i][j-1]+1, dp[i-1][j-1]+cost) # Backtrace ops = [] i, j = m, n while i > 0 or j > 0: if i > 0 and j > 0 and a[i-1] == b[j-1]: ops.append(("match", a[i-1], b[j-1])) i -= 1; j -= 1 elif i > 0 and j > 0 and dp[i][j] == dp[i-1][j-1] + 1: ops.append(("sub", a[i-1], b[j-1])) i -= 1; j -= 1 elif i > 0 and dp[i][j] == dp[i-1][j] + 1: ops.append(("del", a[i-1], "")) i -= 1 else: ops.append(("ins", "", b[j-1])) j -= 1 ops.reverse() return ops def compare_phonemes(expected_phonemes, actual_phonemes): """Compare expected vs actual phonemes.""" ops = levenshtein_ops(expected_phonemes, actual_phonemes) matches = sum(1 for o in ops if o[0] == "match") substitutions = sum(1 for o in ops if o[0] == "sub") deletions = sum(1 for o in ops if o[0] == "del") insertions = sum(1 for o in ops if o[0] == "ins") total = len(ops) accuracy = round(matches / max(total, 1) * 100, 1) errors = [] for op, expected, actual in ops: if op != "match": errors.append({"type": op, "expected": expected, "actual": actual}) return { "accuracy": accuracy, "total_phonemes": total, "matches": matches, "substitutions": substitutions, "deletions": deletions, "insertions": insertions, "errors": errors, } # ── Italian-specific confusion penalties ──────────────────────────── ITALIAN_CONFUSION = { # Russian → Italian common errors ("r", "ɾ"): 0.5, # rolled R ("l", "ʎ"): 0.7, # gli sound ("n", "ɲ"): 0.7, # gn sound ("e", "ɛ"): 0.3, # open e ("e", "e"): 0.0, # same ("o", "ɔ"): 0.3, # open o # Double consonants (gemination) ("t", "tt"): 0.6, ("l", "ll"): 0.6, ("n", "nn"): 0.6, } def apply_confusion_penalty(errors): """Apply language-specific penalties to phoneme errors.""" weighted = 0 total = len(errors) if errors else 1 for err in errors: key = (err.get("expected", ""), err.get("actual", "")) penalty = ITALIAN_CONFUSION.get(key, 1.0) weighted += penalty return round(weighted / total, 2) # ── Scoring ───────────────────────────────────────────────────────── def compute_score(text_comparison, phoneme_comparison, timing_analysis, confusion_penalty): """Compute overall pronunciation score 0-100.""" # Text accuracy weight: 30% text_score = text_comparison.get("score", 0) # Phoneme accuracy weight: 35% phoneme_score = phoneme_comparison.get("accuracy", 0) # Timing/rhythm weight: 20% timing_score = timing_analysis.get("rhythm_score", 0) # Confusion penalty weight: 15% (inverted — lower penalty = higher score) penalty_score = max(0, 100 - confusion_penalty * 100) overall = text_score * 0.30 + phoneme_score * 0.35 + timing_score * 0.20 + penalty_score * 0.15 return round(overall, 1) def quality_label(score): if score >= 90: return "Отлично! 🇮🇹" if score >= 75: return "Хорошо 👍" if score >= 60: return "Неплохо 🙂" if score >= 40: return "Нужна практика 📚" return "Попробуй ещё раз 💪" # ── AI Feedback ───────────────────────────────────────────────────── def generate_feedback(assessment): """Generate human-readable feedback from assessment data.""" parts = [] score = assessment.get("overall_score", 0) if score >= 90: parts.append("🎉 Отличное произношение!") elif score >= 75: parts.append("👍 Хорошее произношение, есть небольшие недочёты.") elif score >= 60: parts.append("📚 Неплохо, но нужно поработать над звуками.") else: parts.append("💪 Требуется практика произношения.") # Phoneme errors errors = assessment.get("phoneme_comparison", {}).get("errors", []) sub_errors = [e for e in errors if e["type"] == "sub"] if sub_errors: sample = sub_errors[:3] parts.append("Звуки для отработки: " + ", ".join( f"{e['expected']}→{e['actual']}" for e in sample )) del_errors = [e for e in errors if e["type"] == "del"] if del_errors: parts.append(f"Пропущено звуков: {len(del_errors)}.") ins_errors = [e for e in errors if e["type"] == "ins"] if ins_errors: parts.append(f"Лишних звуков: {len(ins_errors)}.") # Timing timing = assessment.get("timing", {}) if timing.get("timing_quality") == "poor": parts.append("⏱ Ритм неравномерный — попробуй говорить плавнее.") return " ".join(parts) # ── Full assessment ───────────────────────────────────────────────── def assess(expected_text, whisper_result, lang="it"): """ Full pronunciation assessment. Args: expected_text: "buona sera" whisper_result: {"text": "...", "words": [{word, start, end}, ...]} Returns: Full assessment dict with scores, phonemes, errors, feedback. """ # 1. Build expected model expected = build_expected(expected_text, lang) # 2. Build actual model from Whisper output actual_text = whisper_result.get("text", "").strip() actual_words = whisper_result.get("words", []) actual_phonemes = text_to_phonemes(actual_text, lang) if actual_text else [] # 3. Text-level comparison (Levenshtein from frontend, or compute here) # Use simple word accuracy expected_words_norm = re.sub(r"[^\w\s]", "", expected_text.lower()).split() actual_words_norm = re.sub(r"[^\w\s]", "", actual_text.lower()).split() if actual_text else [] word_matches = sum(1 for e, a in zip(expected_words_norm, actual_words_norm) if e == a) text_score = round(word_matches / max(len(expected_words_norm), 1) * 100, 1) text_comparison = { "expected": expected_text, "actual": actual_text, "expected_words": expected_words_norm, "actual_words": actual_words_norm, "word_matches": word_matches, "total_words": len(expected_words_norm), "score": text_score, } # 4. Phoneme comparison phoneme_comparison = compare_phonemes(expected["phonemes"], actual_phonemes) # 5. Timing analysis timing = analyze_timing(actual_words) # 6. Confusion penalty confusion_penalty = apply_confusion_penalty(phoneme_comparison.get("errors", [])) # 7. Overall score overall = compute_score(text_comparison, phoneme_comparison, timing, confusion_penalty) assessment = { "overall_score": overall, "quality": quality_label(overall), "expected": expected, "actual": { "text": actual_text, "phonemes": actual_phonemes, "words": actual_words, }, "text_comparison": text_comparison, "phoneme_comparison": phoneme_comparison, "timing": timing, "confusion_penalty": confusion_penalty, "language": lang, } # 8. Feedback assessment["feedback"] = generate_feedback(assessment) return assessment # ── Flask endpoint ────────────────────────────────────────────────── def register_routes(app): from flask import request @app.route("/pronounce/assess", methods=["POST"]) def pronounce_assess(): try: data = request.get_json(force=True) expected = data.get("expected", "").strip() whisper_result = data.get("whisper", {}) lang = data.get("lang", "it") if not expected: return {"error": "no expected text"}, 400 if not whisper_result.get("text"): return {"error": "no whisper result"}, 400 result = assess(expected, whisper_result, lang) result["_version"] = "1.0.0" print(f"[PRONOUNCE] \"{expected}\" → score={result['overall_score']} " f"phonemes={result['phoneme_comparison']['accuracy']}% " f"rhythm={result['timing']['rhythm_score']}", flush=True) resp = app.make_response((json.dumps(result, ensure_ascii=False), 200)) resp.headers.update({ "Access-Control-Allow-Origin": "*", "Content-Type": "application/json; charset=utf-8", }) return resp except Exception as e: print(f"[PRONOUNCE] ERROR: {e}", flush=True) return {"error": str(e)}, 500 @app.route("/pronounce/health", methods=["GET"]) def pronounce_health(): return { "status": "ok", "phonemizer": "espeak", "languages": ["it", "en", "fr", "de", "es", "ru"], "version": "1.0.0", }