diff --git a/anki_exporter.py b/anki_exporter.py
index 57a5df7..9e301d2 100644
--- a/anki_exporter.py
+++ b/anki_exporter.py
@@ -6,6 +6,7 @@ import genanki
import edge_tts
import shutil
import database
+from tts_utils import parse_text_for_edgetts, get_configured_tts_rate
async def generate_edge_audio(text, voice, output_path, rate_modifier="+0%"):
"""Asynchronously streams data packages via the Microsoft Edge API pipeline."""
@@ -29,14 +30,7 @@ def compile_anki_package(records, output_path, deck_name):
# Global Configuration Pace Resolver Mapping
settings = database.load_all_settings() or {}
- config_speed = settings.get("tts_playback_speed", "1.0")
-
- # Transform numeric string floats (e.g., 1.2) into Edge-TTS percentage strings (e.g., +20%)
- try:
- pct = int((float(config_speed) - 1.0) * 100)
- rate_string = f"{'+' if pct >= 0 else ''}{pct}%"
- except Exception:
- rate_string = "+0%"
+ rate_string = get_configured_tts_rate(settings)
anki_model = genanki.Model(
model_id,
@@ -83,10 +77,14 @@ def compile_anki_package(records, output_path, deck_name):
with tempfile.TemporaryDirectory() as tmpdir:
for idx, record in enumerate(records):
- en_text = record["en_text"]
- es_text = record["es_text"]
+ en_raw = record["en_text"]
+ es_raw = record["es_text"]
gender_flag = record.get("gender", "Female")
+ # Sanitize text payloads for TTS engine
+ en_tts_text = parse_text_for_edgetts(en_raw)
+ es_tts_text = parse_text_for_edgetts(es_raw)
+
# Map native neural voice files matching gender settings
spanish_voice = "es-ES-AlvaroNeural" if gender_flag == "Male" else "es-ES-ElviraNeural"
english_voice = "en-US-EmmaNeural"
@@ -105,14 +103,14 @@ def compile_anki_package(records, output_path, deck_name):
es_audio_path = os.path.join(tmpdir, es_audio_filename)
# English Audio Synthesis
- if loop.run_until_complete(generate_edge_audio(en_text, english_voice, en_audio_path, rate_string)):
+ if en_tts_text.strip() and loop.run_until_complete(generate_edge_audio(en_tts_text, english_voice, en_audio_path, rate_string)):
media_files_to_pack.append(en_audio_path)
en_audio_field = f"[sound:{en_audio_filename}]"
else:
en_audio_field = ""
# Spanish Audio Synthesis
- if loop.run_until_complete(generate_edge_audio(es_text, spanish_voice, es_audio_path, rate_string)):
+ if es_tts_text.strip() and loop.run_until_complete(generate_edge_audio(es_tts_text, spanish_voice, es_audio_path, rate_string)):
media_files_to_pack.append(es_audio_path)
es_audio_field = f"[sound:{es_audio_filename}]"
else:
@@ -121,7 +119,7 @@ def compile_anki_package(records, output_path, deck_name):
# Clean sequential matching fields array matching schema mapping above
note = genanki.Note(
model=anki_model,
- fields=[en_text, es_text, anki_notes_html, en_audio_field, es_audio_field],
+ fields=[en_raw, es_raw, anki_notes_html, en_audio_field, es_audio_field],
tags=note_tags
)
deck.add_note(note)
diff --git a/spanish_trainer.db b/spanish_trainer.db
index aecf0c3..1fc4531 100644
Binary files a/spanish_trainer.db and b/spanish_trainer.db differ
diff --git a/tabs/review_tab.py b/tabs/review_tab.py
index 61ecdaf..75fc449 100644
--- a/tabs/review_tab.py
+++ b/tabs/review_tab.py
@@ -2,14 +2,19 @@
import random
import os
import subprocess
+import tempfile
+import threading
+import asyncio
from PyQt6.QtWidgets import (
QWidget, QVBoxLayout, QHBoxLayout, QLabel,
QLineEdit, QPushButton, QTableWidget, QTableWidgetItem,
QHeaderView, QFormLayout, QMessageBox, QFrame, QStackedWidget
)
from PyQt6.QtCore import Qt, pyqtSlot
+import edge_tts
import database
import anki_exporter
+from tts_utils import parse_text_for_edgetts, get_configured_tts_rate
class ReviewTab(QWidget):
def __init__(self, parent=None):
@@ -262,21 +267,74 @@ class ReviewTab(QWidget):
self.is_flipped = False
self.display_current_card()
+ def _async_edge_speech_worker(self, text, voice, rate_modifier):
+ """Background thread worker to render neural speech with terminal debug logging."""
+ async def stream_audio():
+ temp_file = os.path.join(tempfile.gettempdir(), "review_card_audio.mp3")
+
+ # Clean up old file if present
+ if os.path.exists(temp_file):
+ try:
+ os.remove(temp_file)
+ except Exception:
+ pass
+
+ try:
+ print(f"[TTS Debug] Generating TTS -> Voice: {voice} | Rate: {rate_modifier} | Text: '{text}'")
+ communicate = edge_tts.Communicate(text=text, voice=voice, rate=rate_modifier)
+ await communicate.save(temp_file)
+
+ if os.path.exists(temp_file) and os.path.getsize(temp_file) > 0:
+ print(f"[TTS Debug] Audio ready ({os.path.getsize(temp_file)} bytes). Playing via afplay...")
+ result = subprocess.run(["afplay", temp_file], capture_output=True, text=True)
+ if result.returncode != 0:
+ print(f"[TTS Debug] afplay failed: {result.stderr}")
+ else:
+ print("[TTS Debug] Playback finished successfully.")
+ else:
+ print("[TTS Debug] Error: Audio file was not created or is 0 bytes.")
+ except Exception as e:
+ print(f"[TTS Debug] Exception during speech synthesis: {e}")
+
+ # Explicitly set up and run a clean event loop for this thread
+ try:
+ loop = asyncio.new_event_loop()
+ asyncio.set_event_loop(loop)
+ loop.run_until_complete(stream_audio())
+ loop.close()
+ except Exception as e:
+ print(f"[TTS Debug] Event loop error: {e}")
+
@pyqtSlot()
def play_card_audio(self):
- """Auditions native voice files based on active visual canvas sides."""
+ """Auditions neural edge-tts voice based on active flashcard side."""
if not (0 <= self.current_index < len(self.filtered_review_pool)):
return
record = self.filtered_review_pool[self.current_index]
+ settings = database.load_all_settings() or {}
+ rate_string = get_configured_tts_rate(settings)
+
if not self.is_flipped:
- # Play English text utilizing standard native subsystem default output
- if record["en_text"]:
- subprocess.Popen(["say", record["en_text"]])
+ raw_text = record.get("en_text", "")
+ spoken_text = parse_text_for_edgetts(raw_text)
+ voice = "en-US-EmmaNeural"
else:
- # Play target translation explicitly targeting the Castilian voice profile Mónica
- if record["es_text"]:
- subprocess.Popen(["say", "-v", "Monica", record["es_text"]])
+ raw_text = record.get("es_text", "")
+ spoken_text = parse_text_for_edgetts(raw_text)
+ is_male = (record.get("gender") == "Male")
+ voice = "es-ES-AlvaroNeural" if is_male else "es-ES-ElviraNeural"
+
+ # Respect flags or empty entries
+ if not spoken_text.strip():
+ print("[TTS Debug] Skipped: Parsed text is empty or muted via sound-off tag.")
+ return
+
+ threading.Thread(
+ target=self._async_edge_speech_worker,
+ args=(spoken_text, voice, rate_string),
+ daemon=True
+ ).start()
@pyqtSlot()
def generate_deck_action(self):
@@ -288,14 +346,10 @@ class ReviewTab(QWidget):
try:
# Load active settings dictionary directly from your database configurations
- settings = database.load_all_settings()
- # print("--- CURRENT DATABASE SETTINGS ---")
- # for key, value in settings.items():
- # print(f"{key}: {value}")
- # print("---------------------------------")
+ settings = database.load_all_settings() or {}
+
# Extract configurations targeting exact database schema names found in settings
target_dir = settings.get("anki_export_directory")
- #root_deck_name = settings.get("default_deck_name")
root_deck_name = settings.get("anki_root_deck_name")
sub_deck_hierarchy = settings.get("anki_sub_deck_name")
@@ -316,8 +370,6 @@ class ReviewTab(QWidget):
if sub_deck_hierarchy and str(sub_deck_hierarchy).strip():
deck_tree_parts.append(str(sub_deck_hierarchy).strip())
- ## deck_tree_parts.append("Filtered Review Session")
-
# Join parts using Anki double-colon syntax (::)
full_deck_namespace = "::".join(deck_tree_parts)
diff --git a/tabs/sandbox_tab.py b/tabs/sandbox_tab.py
index 2a3fa92..55cb254 100644
--- a/tabs/sandbox_tab.py
+++ b/tabs/sandbox_tab.py
@@ -12,6 +12,8 @@ from PyQt6.QtWidgets import (
from PyQt6.QtCore import pyqtSignal, pyqtSlot, Qt
import edge_tts
import database
+from tts_utils import parse_text_for_edgetts, get_configured_tts_rate
+
class TextEditorDialog(QDialog):
"""A pop-up modal containing a large text field workspace for copy-pasting extra text blocks or drafting HTML content."""
@@ -360,42 +362,35 @@ class SandboxTab(QWidget):
@pyqtSlot()
def audition_english(self):
- txt = self.txt_english.text().strip()
- if not txt:
+ raw_txt = self.txt_english.text().strip()
+ spoken_text = parse_text_for_edgetts(raw_txt)
+
+ if not spoken_text.strip():
return
settings = database.load_all_settings() or {}
- config_speed = settings.get("tts_playback_speed", "1.0")
- try:
- pct = int((float(config_speed) - 1.0) * 100)
- rate_string = f"{'+' if pct >= 0 else ''}{pct}%"
- except Exception:
- rate_string = "+0%"
+ rate_string = get_configured_tts_rate(settings)
threading.Thread(
target=self._async_edge_speech_worker,
- args=(txt, "en-US-EmmaNeural", rate_string),
+ args=(spoken_text, "en-US-EmmaNeural", rate_string),
daemon=True
).start()
@pyqtSlot()
def audition_spanish(self):
- txt = self.txt_spanish.text().strip()
- if not txt:
+ raw_txt = self.txt_spanish.text().strip()
+ spoken_text = parse_text_for_edgetts(raw_txt)
+
+ if not spoken_text.strip():
return
voice = "es-ES-AlvaroNeural" if self.rb_male.isChecked() else "es-ES-ElviraNeural"
-
settings = database.load_all_settings() or {}
- config_speed = settings.get("tts_playback_speed", "1.0")
- try:
- pct = int((float(config_speed) - 1.0) * 100)
- rate_string = f"{'+' if pct >= 0 else ''}{pct}%"
- except Exception:
- rate_string = "+0%"
+ rate_string = get_configured_tts_rate(settings)
threading.Thread(
target=self._async_edge_speech_worker,
- args=(txt, voice, rate_string),
+ args=(spoken_text, voice, rate_string),
daemon=True
).start()
\ No newline at end of file
diff --git a/test_tts.py b/test_tts.py
index 2f47f65..6543cc0 100644
--- a/test_tts.py
+++ b/test_tts.py
@@ -1,28 +1,55 @@
-import asyncio
-import os
-import edge_tts
+# tts_utils.py
+import re
+from bs4 import BeautifulSoup
-# 1. Define the phrase, output path, and target Castellano voice
-SPANISH_TEXT = "¡Buenos días! ¿Cómo estás? Bienvenido a tu curso de español."
-OUTPUT_FILE = "media/buenos_dias_castellano-Female.mp3"
-#VOICE = "es-ES-AlvaroNeural" # Swap to "es-ES-ElviraNeural" if you prefer a female tone
-VOICE = "es-ES-ElviraNeural" # Swap to "es-ES-ElviraNeural" if you prefer a female tone
-async def generate_castilian_audio():
- # Ensure our target media folder exists locally
- os.makedirs(os.path.dirname(OUTPUT_FILE), exist_ok=True)
-
- print(f"🔄 Synthesizing text using Castilian voice: {VOICE}...")
-
- # 2. Configure the Communicate engine
- communicate = edge_tts.Communicate(SPANISH_TEXT, VOICE)
-
- # 3. Stream and write the data packets to disk
- await communicate.save(OUTPUT_FILE)
-
- print(f"✨ Success! MP3 file exported safely to: {OUTPUT_FILE}")
- print(f"📂 File size: {os.path.getsize(OUTPUT_FILE)} bytes")
+def get_configured_tts_rate(settings: dict) -> str:
+ """Converts a numerical speed multiplier (e.g., 0.75, 1.0, 1.25) from app settings
+ into Edge TTS percentage format string (e.g., '-25%', '+0%', '+25%').
+ """
+ # Check the actual database key 'tts_playback_speed' with fallbacks
+ raw_val = (
+ settings.get("tts_playback_speed")
+ or settings.get("tts_speed_multiplier")
+ or 1.0
+ )
-if __name__ == "__main__":
- # Run the async loop loop natively
- asyncio.run(generate_castilian_audio())
\ No newline at end of file
+ try:
+ raw_rate = float(raw_val)
+ except (TypeError, ValueError):
+ raw_rate = 1.0
+
+ # Calculate percentage shift relative to baseline 1.0
+ pct = int(round((raw_rate - 1.0) * 100))
+ if pct >= 0:
+ return f"+{pct}%"
+ return f"{pct}%"
+
+
+def parse_text_for_edgetts(html_content: str) -> str:
+ """Strips content between and tags,
+ handling optional whitespace inside the tag brackets (e.g., ).
+ Converts remaining HTML tags to clean spoken text.
+ """
+ if not html_content:
+ return ""
+
+ # Flexible regex to slice out everything from through
+ pattern = re.compile(
+ r".*?",
+ re.DOTALL | re.IGNORECASE,
+ )
+ cleaned_html = re.sub(pattern, "", html_content)
+
+ # Handle unclosed (mute rest of string from that point)
+ if re.search(r"", cleaned_html, re.IGNORECASE):
+ cleaned_html = re.split(
+ r"", cleaned_html, flags=re.IGNORECASE
+ )[0]
+
+ # Convert remaining HTML into plain text for speech
+ soup = BeautifulSoup(cleaned_html, "html.parser")
+ text = soup.get_text(separator=" ")
+
+ # Normalize extra whitespace
+ return re.sub(r"\s+", " ", text).strip()
\ No newline at end of file
diff --git a/tts_utils.py b/tts_utils.py
new file mode 100644
index 0000000..71a34d3
--- /dev/null
+++ b/tts_utils.py
@@ -0,0 +1,52 @@
+# tts_utils.py
+import re
+from bs4 import BeautifulSoup
+
+
+def get_configured_tts_rate(settings: dict) -> str:
+ """Extracts speed multiplier from settings dict and converts to Edge TTS percentage string (e.g., '-25%')."""
+ # Query the exact key name shown in SQLite database: 'tts_playback_speed'
+ raw_val = settings.get("tts_playback_speed", 1.0) if settings else 1.0
+
+ try:
+ raw_rate = float(raw_val)
+ except (TypeError, ValueError):
+ raw_rate = 1.0
+
+ # Calculate percentage shift relative to baseline 1.0 (e.g., 0.75 -> -25%)
+ pct = int(round((raw_rate - 1.0) * 100))
+ rate_str = f"+{pct}%" if pct >= 0 else f"{pct}%"
+
+ print(
+ f"[TTS Debug] DB Key 'tts_playback_speed': {raw_val} -> Formatted Rate: {rate_str}"
+ )
+ return rate_str
+
+
+def parse_text_for_edgetts(html_content: str) -> str:
+ """Strips content between and tags,
+ handling optional whitespace inside the tag brackets (e.g., ).
+ Converts remaining HTML tags to clean spoken text.
+ """
+ if not html_content:
+ return ""
+
+ # Flexible regex to slice out everything from through
+ pattern = re.compile(
+ r".*?",
+ re.DOTALL | re.IGNORECASE,
+ )
+ cleaned_html = re.sub(pattern, "", html_content)
+
+ # Handle unclosed (mute rest of string from that point)
+ if re.search(r"", cleaned_html, re.IGNORECASE):
+ cleaned_html = re.split(
+ r"", cleaned_html, flags=re.IGNORECASE
+ )[0]
+
+ # Convert remaining HTML into plain text for speech
+ soup = BeautifulSoup(cleaned_html, "html.parser")
+ text = soup.get_text(separator=" ")
+
+ # Normalize extra whitespace
+ return re.sub(r"\s+", " ", text).strip()
\ No newline at end of file