"""Local-only PDF narration with Chatterbox. No voice or document uploads to a cloud service.""" import os from pathlib import Path BASE = Path(__file__).resolve().parent os.environ.setdefault('HF_HUB_DISABLE_XET', '1') os.environ.setdefault('HF_HOME', str(BASE / 'models')) os.environ.setdefault('GRADIO_ANALYTICS_ENABLED', 'False') os.environ.setdefault('PYTORCH_ENABLE_MPS_FALLBACK', '1') import re import uuid import threading import gradio as gr from pypdf import PdfReader MAX_CHARS = 25000 MODEL = None DEFAULT_CONDS = None LOCK = threading.Lock() def read_document(path): if not path: return '' file = Path(path) if file.stat().st_size > 20 * 1024 * 1024: raise gr.Error('Please use a document under 20 MB.') try: if file.suffix.lower() == '.txt': text = file.read_text(encoding='utf-8') else: pdf = PdfReader(file) if pdf.is_encrypted: raise ValueError('Please upload an unencrypted PDF.') if len(pdf.pages) > 60: raise ValueError('Please use a PDF of 60 pages or fewer.') attached = pdf.attachments.get('narration.txt') text = attached[0].decode('utf-8') if attached else '\n\n'.join(p.extract_text() or '' for p in pdf.pages) text = text.strip() if not text: raise ValueError('No readable text found. Scanned PDFs need OCR first.') if len(text) > MAX_CHARS: raise ValueError('This document is too long. Split it into sections under 25,000 characters.') return text except gr.Error: raise except Exception as exc: raise gr.Error(f'Could not read the document: {exc}') from exc def chunks(text, limit=320): pieces = re.split(r'(?<=[.!?])\s+|\n+', text.strip()) current = '' for sentence in pieces: for word in sentence.split(): if len(current) + len(word) + 1 > limit and current: yield current current = '' current = (current + ' ' + word).strip() if len(current) > limit // 2: yield current current = '' if current: yield current def get_model(): global MODEL, DEFAULT_CONDS if MODEL is None: import torch from chatterbox.tts import ChatterboxTTS device = 'mps' if torch.backends.mps.is_available() else ('cuda' if torch.cuda.is_available() else 'cpu') MODEL = ChatterboxTTS.from_pretrained(device=device) DEFAULT_CONDS = MODEL.conds return MODEL def generate(text, voice, consent, progress=gr.Progress()): if not text or not text.strip(): raise gr.Error('Upload a PDF or enter your narration first.') if len(text) > MAX_CHARS: raise gr.Error('Please keep each narration under 25,000 characters.') if voice and not consent: raise gr.Error('Confirm that this is your voice or you have permission to use it.') import numpy as np import soundfile as sf if voice: info = sf.info(voice) if not 5 <= info.duration <= 60: raise gr.Error('Use a clean voice sample between 5 and 60 seconds; 10–20 seconds is a good start.') with LOCK: progress(0, desc='Loading Chatterbox (first run downloads model files)…') try: model = get_model() if not voice: model.conds = DEFAULT_CONDS parts = list(chunks(text)) audio = [] for i, part in enumerate(parts): progress(i / len(parts), desc=f'Narrating section {i+1} of {len(parts)}') wav = model.generate(part, audio_prompt_path=voice or None, exaggeration=0.45, cfg_weight=0.5) samples = wav.detach().cpu().numpy().reshape(-1) audio.extend([samples, np.zeros(int(model.sr * 0.35), dtype=np.float32)]) out = BASE / 'output' out.mkdir(exist_ok=True) filename = out / f'f5cyber-briefing-{uuid.uuid4().hex[:10]}.wav' sf.write(filename, np.concatenate(audio), model.sr) progress(1, desc='Audio ready') return str(filename), str(filename) except Exception as exc: raise gr.Error(f'Audio generation failed: {exc}. Your script is still available. Check your connection on the first run and try a shorter passage.') from exc def make_app(): with gr.Blocks(title='F5 Cyber Audio', analytics_enabled=False) as app: gr.Markdown('# F5 Cyber Audio\nUpload your daily briefing, review the script, and generate a local audio edition. **Your PDF and voice stay on this computer.** Model files download on first use; generation then runs locally.') doc = gr.File(label='1. Upload briefing PDF or text', file_types=['.pdf', '.txt'], type='filepath') script = gr.Textbox(label='2. Review or edit the narration', lines=12, max_lines=25) doc.change(read_document, doc, script) voice = gr.Audio(label='3. Your voice (optional, 10–20 seconds recommended)', sources=['upload'], type='filepath') consent = gr.Checkbox(label='This is my voice, or I have permission to reproduce it.') gr.Markdown('Leave the voice sample empty to use the default voice. Length follows your script; generation may take several minutes. Review facts and pronunciation before publishing. Chatterbox watermarks generated audio.') button = gr.Button('Generate audio', variant='primary') player = gr.Audio(label='Listen to your briefing', type='filepath') download = gr.File(label='Download WAV') button.click(generate, [script, voice, consent], [player, download], concurrency_limit=1) gr.Markdown('Files are saved in this app’s output folder. Stop the app by closing its Terminal window. Local uploads are temporary; delete output files when no longer needed.') return app if __name__ == '__main__': make_app().queue().launch(server_name='127.0.0.1', server_port=7860, share=False, inbrowser=True, max_file_size='20mb')