From 415b27edc26a9916e328b413bb263aad231c85fa Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E2=80=9CNaeel=E2=80=9D?= Date: Sat, 23 May 2026 17:34:45 +0400 Subject: [PATCH] MVP pronunciation assessment: phonemizer + alignment + scoring + AI feedback. Endpoint: POST /pronounce/assess --- deploy_lang.sh | 4 + server/pronounce.py | 371 ++++++++++++++++++++++++++++++++++++++++++++ server/proxy.py | 7 + 3 files changed, 382 insertions(+) create mode 100644 server/pronounce.py diff --git a/deploy_lang.sh b/deploy_lang.sh index 3e25717..1521a5d 100755 --- a/deploy_lang.sh +++ b/deploy_lang.sh @@ -18,4 +18,8 @@ echo " done" echo "=== rsync -> VM ===" rsync -az -e "$SSH" /home/naeel/lang/dist/ ${VM}:/var/www/lyngvo/ +echo "=== update server ===" +rsync -az -e "$SSH" /home/naeel/lang/server/ ${VM}:/opt/groq-proxy/ +$SSH $VM 'systemctl restart groq-proxy && echo OK' + echo "=== https://lang.kube5s.ru ===" diff --git a/server/pronounce.py b/server/pronounce.py new file mode 100644 index 0000000..9518854 --- /dev/null +++ b/server/pronounce.py @@ -0,0 +1,371 @@ +#!/usr/bin/env python3 +"""Pronunciation assessment engine — поверх Whisper API.""" + +from phonemizer import phonemize +import json, re, math + + +# ── Phoneme conversion ────────────────────────────────────────────── + +def text_to_phonemes(text, lang="it"): + """text → list of phoneme strings.""" + raw = phonemize(text, language=lang, backend="espeak", strip=True, + preserve_punctuation=False, with_stress=False) + # espeak returns space-separated phonemes, but some are multi-char + # Parse carefully: split on spaces, merge tied phonemes + tokens = raw.split() + return tokens + + +def phonemes_to_syllables(phonemes): + """Group phonemes into rough syllables (vowel = nucleus).""" + vowels = set("aeɛiouɔɑɒʌəɨʉɯʊɤeøɘɵɐœɶʏɪɞ") + syllables = [] + cur = [] + for p in phonemes: + cur.append(p) + # Any vowel-ish character means syllable nucleus + if any(c in vowels for c in p): + pass # keep collecting consonants after vowel in same syllable + # Simple heuristic: two consonants in a row → split before second + # Simpler approach: just split on vowel positions + result = [] + buf = [] + for p in phonemes: + buf.append(p) + has_vowel = any(c in vowels for c in p) + if has_vowel: + result.append("-".join(buf)) + buf = [] + if buf: + if result: + result[-1] = result[-1] + "-" + "-".join(buf) + else: + result.append("-".join(buf)) + return result + + +# ── Expected pronunciation model ──────────────────────────────────── + +def build_expected(text, lang="it"): + """Build expected pronunciation model for given text.""" + phonemes = text_to_phonemes(text, lang) + syllables = phonemes_to_syllables(phonemes) + return { + "text": text, + "language": lang, + "phonemes": phonemes, + "syllables": syllables, + "phoneme_count": len(phonemes), + "syllable_count": len(syllables), + } + + +# ── Timing analysis ───────────────────────────────────────────────── + +def analyze_timing(words, expected_duration=None): + """Analyze word timing from Whisper output.""" + if not words: + return {"error": "no words"} + + times = [] + for w in words: + dur = w.get("end", 0) - w.get("start", 0) + times.append({ + "word": w.get("word", ""), + "start": w.get("start", 0), + "end": w.get("end", 0), + "duration": dur, + }) + + total = times[-1]["end"] - times[0]["start"] if times else 0 + avg_speed = sum(t["duration"] for t in times) / len(times) if times else 0 + + # Rhythm score: lower std deviation = more natural rhythm + durs = [t["duration"] for t in times] + mean_dur = sum(durs) / len(durs) if durs else 1 + variance = sum((d - mean_dur)**2 for d in durs) / len(durs) if durs else 0 + rhythm_score = max(0, 100 - math.sqrt(variance) * 50) + + return { + "words": times, + "total_duration": total, + "avg_word_duration": avg_speed, + "rhythm_score": round(rhythm_score, 1), + "timing_quality": "good" if rhythm_score > 70 else "ok" if rhythm_score > 40 else "poor", + } + + +# ── Phoneme comparison ────────────────────────────────────────────── + +def levenshtein_ops(a, b): + """Levenshtein with backtrace — returns list of (op, char_a, char_b).""" + m, n = len(a), len(b) + dp = [[0]*(n+1) for _ in range(m+1)] + for i in range(m+1): dp[i][0] = i + for j in range(n+1): dp[0][j] = j + for i in range(1, m+1): + for j in range(1, n+1): + cost = 0 if a[i-1] == b[j-1] else 1 + dp[i][j] = min(dp[i-1][j]+1, dp[i][j-1]+1, dp[i-1][j-1]+cost) + + # Backtrace + ops = [] + i, j = m, n + while i > 0 or j > 0: + if i > 0 and j > 0 and a[i-1] == b[j-1]: + ops.append(("match", a[i-1], b[j-1])) + i -= 1; j -= 1 + elif i > 0 and j > 0 and dp[i][j] == dp[i-1][j-1] + 1: + ops.append(("sub", a[i-1], b[j-1])) + i -= 1; j -= 1 + elif i > 0 and dp[i][j] == dp[i-1][j] + 1: + ops.append(("del", a[i-1], "")) + i -= 1 + else: + ops.append(("ins", "", b[j-1])) + j -= 1 + ops.reverse() + return ops + + +def compare_phonemes(expected_phonemes, actual_phonemes): + """Compare expected vs actual phonemes.""" + ops = levenshtein_ops(expected_phonemes, actual_phonemes) + + matches = sum(1 for o in ops if o[0] == "match") + substitutions = sum(1 for o in ops if o[0] == "sub") + deletions = sum(1 for o in ops if o[0] == "del") + insertions = sum(1 for o in ops if o[0] == "ins") + total = len(ops) + + accuracy = round(matches / max(total, 1) * 100, 1) + + errors = [] + for op, expected, actual in ops: + if op != "match": + errors.append({"type": op, "expected": expected, "actual": actual}) + + return { + "accuracy": accuracy, + "total_phonemes": total, + "matches": matches, + "substitutions": substitutions, + "deletions": deletions, + "insertions": insertions, + "errors": errors, + } + + +# ── Italian-specific confusion penalties ──────────────────────────── + +ITALIAN_CONFUSION = { + # Russian → Italian common errors + ("r", "ɾ"): 0.5, # rolled R + ("l", "ʎ"): 0.7, # gli sound + ("n", "ɲ"): 0.7, # gn sound + ("e", "ɛ"): 0.3, # open e + ("e", "e"): 0.0, # same + ("o", "ɔ"): 0.3, # open o + # Double consonants (gemination) + ("t", "tt"): 0.6, + ("l", "ll"): 0.6, + ("n", "nn"): 0.6, +} + + +def apply_confusion_penalty(errors): + """Apply language-specific penalties to phoneme errors.""" + weighted = 0 + total = len(errors) if errors else 1 + for err in errors: + key = (err.get("expected", ""), err.get("actual", "")) + penalty = ITALIAN_CONFUSION.get(key, 1.0) + weighted += penalty + return round(weighted / total, 2) + + +# ── Scoring ───────────────────────────────────────────────────────── + +def compute_score(text_comparison, phoneme_comparison, timing_analysis, confusion_penalty): + """Compute overall pronunciation score 0-100.""" + # Text accuracy weight: 30% + text_score = text_comparison.get("score", 0) + + # Phoneme accuracy weight: 35% + phoneme_score = phoneme_comparison.get("accuracy", 0) + + # Timing/rhythm weight: 20% + timing_score = timing_analysis.get("rhythm_score", 0) + + # Confusion penalty weight: 15% (inverted — lower penalty = higher score) + penalty_score = max(0, 100 - confusion_penalty * 100) + + overall = text_score * 0.30 + phoneme_score * 0.35 + timing_score * 0.20 + penalty_score * 0.15 + return round(overall, 1) + + +def quality_label(score): + if score >= 90: return "Отлично! 🇮🇹" + if score >= 75: return "Хорошо 👍" + if score >= 60: return "Неплохо 🙂" + if score >= 40: return "Нужна практика 📚" + return "Попробуй ещё раз 💪" + + +# ── AI Feedback ───────────────────────────────────────────────────── + +def generate_feedback(assessment): + """Generate human-readable feedback from assessment data.""" + parts = [] + score = assessment.get("overall_score", 0) + + if score >= 90: + parts.append("🎉 Отличное произношение!") + elif score >= 75: + parts.append("👍 Хорошее произношение, есть небольшие недочёты.") + elif score >= 60: + parts.append("📚 Неплохо, но нужно поработать над звуками.") + else: + parts.append("💪 Требуется практика произношения.") + + # Phoneme errors + errors = assessment.get("phoneme_comparison", {}).get("errors", []) + sub_errors = [e for e in errors if e["type"] == "sub"] + if sub_errors: + sample = sub_errors[:3] + parts.append("Звуки для отработки: " + ", ".join( + f"{e['expected']}→{e['actual']}" for e in sample + )) + + del_errors = [e for e in errors if e["type"] == "del"] + if del_errors: + parts.append(f"Пропущено звуков: {len(del_errors)}.") + + ins_errors = [e for e in errors if e["type"] == "ins"] + if ins_errors: + parts.append(f"Лишних звуков: {len(ins_errors)}.") + + # Timing + timing = assessment.get("timing", {}) + if timing.get("timing_quality") == "poor": + parts.append("⏱ Ритм неравномерный — попробуй говорить плавнее.") + + return " ".join(parts) + + +# ── Full assessment ───────────────────────────────────────────────── + +def assess(expected_text, whisper_result, lang="it"): + """ + Full pronunciation assessment. + + Args: + expected_text: "buona sera" + whisper_result: {"text": "...", "words": [{word, start, end}, ...]} + + Returns: + Full assessment dict with scores, phonemes, errors, feedback. + """ + # 1. Build expected model + expected = build_expected(expected_text, lang) + + # 2. Build actual model from Whisper output + actual_text = whisper_result.get("text", "").strip() + actual_words = whisper_result.get("words", []) + actual_phonemes = text_to_phonemes(actual_text, lang) if actual_text else [] + + # 3. Text-level comparison (Levenshtein from frontend, or compute here) + # Use simple word accuracy + expected_words_norm = re.sub(r"[^\w\s]", "", expected_text.lower()).split() + actual_words_norm = re.sub(r"[^\w\s]", "", actual_text.lower()).split() if actual_text else [] + word_matches = sum(1 for e, a in zip(expected_words_norm, actual_words_norm) if e == a) + text_score = round(word_matches / max(len(expected_words_norm), 1) * 100, 1) + + text_comparison = { + "expected": expected_text, + "actual": actual_text, + "expected_words": expected_words_norm, + "actual_words": actual_words_norm, + "word_matches": word_matches, + "total_words": len(expected_words_norm), + "score": text_score, + } + + # 4. Phoneme comparison + phoneme_comparison = compare_phonemes(expected["phonemes"], actual_phonemes) + + # 5. Timing analysis + timing = analyze_timing(actual_words) + + # 6. Confusion penalty + confusion_penalty = apply_confusion_penalty(phoneme_comparison.get("errors", [])) + + # 7. Overall score + overall = compute_score(text_comparison, phoneme_comparison, timing, confusion_penalty) + + assessment = { + "overall_score": overall, + "quality": quality_label(overall), + "expected": expected, + "actual": { + "text": actual_text, + "phonemes": actual_phonemes, + "words": actual_words, + }, + "text_comparison": text_comparison, + "phoneme_comparison": phoneme_comparison, + "timing": timing, + "confusion_penalty": confusion_penalty, + "language": lang, + } + + # 8. Feedback + assessment["feedback"] = generate_feedback(assessment) + + return assessment + + +# ── Flask endpoint ────────────────────────────────────────────────── + +def register_routes(app): + from flask import request + + @app.route("/pronounce/assess", methods=["POST"]) + def pronounce_assess(): + try: + data = request.get_json(force=True) + expected = data.get("expected", "").strip() + whisper_result = data.get("whisper", {}) + lang = data.get("lang", "it") + + if not expected: + return {"error": "no expected text"}, 400 + if not whisper_result.get("text"): + return {"error": "no whisper result"}, 400 + + result = assess(expected, whisper_result, lang) + result["_version"] = "1.0.0" + + print(f"[PRONOUNCE] \"{expected}\" → score={result['overall_score']} " + f"phonemes={result['phoneme_comparison']['accuracy']}% " + f"rhythm={result['timing']['rhythm_score']}", flush=True) + + resp = app.make_response((json.dumps(result, ensure_ascii=False), 200)) + resp.headers.update({ + "Access-Control-Allow-Origin": "*", + "Content-Type": "application/json; charset=utf-8", + }) + return resp + except Exception as e: + print(f"[PRONOUNCE] ERROR: {e}", flush=True) + return {"error": str(e)}, 500 + + @app.route("/pronounce/health", methods=["GET"]) + def pronounce_health(): + return { + "status": "ok", + "phonemizer": "espeak", + "languages": ["it", "en", "fr", "de", "es", "ru"], + "version": "1.0.0", + } diff --git a/server/proxy.py b/server/proxy.py index c1a1dcd..5964423 100644 --- a/server/proxy.py +++ b/server/proxy.py @@ -1,6 +1,7 @@ #!/usr/bin/env python3 from flask import Flask, request, Response import requests +import json app = Flask(__name__) GROQ_BASE = "https://api.proxyapi.ru/openai" @@ -11,6 +12,12 @@ CORS = { "Access-Control-Allow-Headers": "Authorization, Content-Type", } +# Импорт pronounce после создания app +import sys, os +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +from pronounce import register_routes +register_routes(app) + @app.route("/", methods=["OPTIONS"]) def options(path): return Response(status=204, headers=CORS)