import gradio as gr import re import numpy as np import os import tempfile import soundfile as sf import zipfile import gc from num2words import num2words from kokoro_vietnamese import KokoroVietnamese print("Đang khởi động nền tảng Web UI trên CPU (Tối ưu hóa bộ nhớ cho HF Spaces)...") MAX_CHARS = 10 def xoa_ky_tu_dac_biet(text): return re.sub(r'[\\/*?:"<>|]', "", text) # 1. HÀM XỬ LÝ VĂN BẢN def chuan_hoa_van_ban(van_ban): van_ban = re.sub(r'[;:]', ',', van_ban) van_ban = re.sub(r'["\'“”‘’]', '', van_ban) cac_dong = van_ban.split('\n') cac_dong_moi = [] for dong in cac_dong: dong_strip = dong.strip() if dong_strip and dong_strip[-1] not in ['.', '!', '?']: dong = dong.rstrip() + '.' cac_dong_moi.append(dong) van_ban = '\n'.join(cac_dong_moi) van_ban = re.sub(r'\b(\d+)/(\d+)/QH(\d+)\b', r'số \1 năm \2 Quốc hội khóa \3', van_ban) van_ban = re.sub(r'\b(\d+)/(\d+)/([A-Za-zĐđ]+)-CP\b', r'số \1 năm \2 \3 chính phủ', van_ban) van_ban = re.sub(r'\b(\d+)/(\d+)/([A-Za-zĐđ-]+)\b', lambda m: f"số {m.group(1)} năm {m.group(2)} {m.group(3).replace('-', ' ')}", van_ban) # --- TỪ ĐIỂN VIẾT TẮT --- tu_viet_tat = { # === NHÓM SỐ LA MÃ === "I": "một", "II": "hai", "III": "ba", "IV": "bốn", "V": "năm", "VI": "sáu", "VII": "bảy", "VIII": "tám", "IX": "chín", "X": "mười", "XI": "mười một", "XII": "mười hai", "XIII": "mười ba", "XIV": "mười bốn", "XV": "mười lăm", "XVI": "mười sáu", "XVII": "mười bảy", "XVIII": "mười tám", "XIX": "mười chín", "XX": "hai mươi", "XXI": "hai mươi mốt", "XXII": "hai mươi hai", "XXIII": "hai mươi ba", "XXIV": "hai mươi bốn", # === NHÓM CHÍNH TRỊ - TỔ CHỨC - ĐOÀN THỂ === "ĐCS": "đảng cộng sản", "T.Ư": "trung ương", "UBND": "ủy ban nhân dân", "HĐND": "hội đồng nhân dân", "MTTQ": "mặt trận tổ quốc", "TNCS": "thanh niên cộng sản", "ĐTN": "đoàn thanh niên", "LĐLĐ": "liên đoàn lao động", "HPN": "hội phụ nữ", "BCH": "ban chấp hành", # Đã xóa "CP": "chính phủ" ở đây để không bị đụng hàng # === NHÓM PHÁP LUẬT - TÒA ÁN - CÔNG AN === "VKSND": "viện kiểm sát nhân dân", "TAND": "tòa án nhân dân", "CQĐT": "cơ quan điều tra", "CSĐT": "cảnh sát điều tra", "CSGT": "cảnh sát giao thông", "CSCĐ": "cảnh sát cơ động", # === NHÓM BỘ - CƠ QUAN - BẢO HIỂM === "GD&ĐT": "giáo dục và đào tạo", "LĐ-TB&XH": "lao động thương binh và xã hội", "UBATGTQG": "ủy ban an toàn giao thông quốc gia", "BHXH": "bảo hiểm xã hội", "BHYT": "bảo hiểm y tế", # === NHÓM VĂN BẢN HÀNH CHÍNH === "NQ": "nghị quyết", "QĐ": "quyết định", "CT": "chỉ thị", "TB": "thông báo", "BC": "báo cáo", "CV": "công văn", "TT": "thông tư", "NĐ": "nghị định", "PL": "pháp luật", "LT": "luật", "ĐA": "đề án", "KH": "kế hoạch", # === NHÓM GIÁO DỤC & ĐÀO TẠO === "ĐH": "đại học", "CĐ": "cao đẳng", "THPT": "trung học phổ thông", "THCS": "trung học cơ sở", "TH": "tiểu học", "MN": "mầm non", "GDPT": "giáo dục phổ thông", "GDTX": "giáo dục thường xuyên", # === NHÓM ĐỊA DANH - HÀNH CHÍNH === "VN": "Việt Nam", "HN": "Hà Nội", "TP.HCM": "thành phố Hồ Chí Minh", "TP": "thành phố", "P.": "phòng", "Q.": "quận", # === NHÓM CHỨC DANH - HỌC VỊ === "PGS.TS": "phó giáo sư tiến sĩ", "TS": "tiến sĩ", "ThS": "thạc sĩ", "CN": "cử nhân", "BS": "bác sĩ", "GV": "giáo viên", "GVCN": "giáo viên chủ nhiệm", # === NHÓM ĐƠN VỊ ĐO LƯỜNG & TIỀN TỆ === "km/h": "ki lô mét trên giờ", "km": "ki lô mét", "m": "mét", "cm": "xăng ti mét", "mm": "mi li mét", "kg": "ký", "ml": "mi li lít", "l": "lít", "VNĐ": "Việt Nam đồng", "USD": "đô la Mỹ", # === NHÓM ĐỜI SỐNG - GIAO TIẾP THÔNG DỤNG === "SN": "sinh năm", "BKS": "biển kiểm soát", "STT": "số thứ tự", "Đ/c": "đồng chí", "đ/c": "đồng chí", "K/g": "kính gửi", "ĐT": "điện thoại", # === NHÓM DOANH NGHIỆP - TÀI CHÍNH - KINH TẾ === "Cty": "công ty", "TNHH": "trách nhiệm hữu hạn", "CP": "cổ phần", # Nhờ Regex bên trên bảo vệ, chữ CP ở đây chỉ áp dụng cho doanh nghiệp "TMCP": "thương mại cổ phần", "DN": "doanh nghiệp", "DNNN": "doanh nghiệp nhà nước", "NHNN": "ngân hàng nhà nước", "NHTM": "ngân hàng thương mại", "BĐS": "bất động sản", "HĐQT": "hội đồng quản trị", "GĐ": "giám đốc", "PGĐ": "phó giám đốc", "TGĐ": "tổng giám đốc", "KTT": "kế toán trưởng", "GDP": "gi-đi-pi", "FDI": "ép-đê-i", # === NHÓM ĐỜI SỐNG - XÃ HỘI - HÀNH CHÍNH CÔNG === "CCCD": "căn cước công dân", "CMND": "chứng minh nhân dân", "HKTT": "hộ khẩu thường trú", "PCCC": "phòng cháy chữa cháy", "VSATTP": "vệ sinh an toàn thực phẩm", "TTTM": "trung tâm thương mại", "KĐT": "khu đô thị", "KCN": "khu công nghiệp", "KCX": "khu chế xuất", "BQL": "ban quản lý", "NƠXH": "nhà ở xã hội", # === NHÓM Y TẾ - SỨC KHỎE === "BV": "bệnh viện", "PK": "phòng khám", "SYT": "sở y tế", "BYT": "bộ y tế", "BN": "bệnh nhân", "NVYT": "nhân viên y tế", "TBYT": "thiết bị y tế", "WHO": "tổ chức y tế thế giới", # === NHÓM TRUYỀN THÔNG - BÁO CHÍ - THỂ THAO === "BTV": "biên tập viên", "PV": "phóng viên", "MC": "em xi", "NXB": "nhà xuất bản", "ĐTH": "đài truyền hình", "ĐPTTH": "đài phát thanh truyền hình", "MXH": "mạng xã hội", "HLV": "huấn luyện viên", "VĐV": "vận động viên", "CLB": "câu lạc bộ", "HCV": "huy chương vàng", "HCB": "huy chương bạc", "HCĐ": "huy chương đồng", # === NHÓM GIAO THÔNG - XÂY DỰNG === "QL": "quốc lộ", "TL": "tỉnh lộ", "HL": "hương lộ", "BOT": "bê ô tê", "ETC": "i ti xi", "GPMB": "giải phóng mặt bằng", "OĐ": "ô đê", # === NHÓM BỔ SUNG VỀ HÀNH CHÍNH - ĐỊA LÝ === "T.": "tỉnh", "H.": "huyện", "X.": "xã", "TX": "thị xã", "TT": "thị trấn", # === NHÓM BỔ SUNG VỀ GIÁO DỤC === "HS": "học sinh", "SV": "sinh viên", "HSSV": "học sinh sinh viên", "NCS": "nghiên cứu sinh", "SGK": "sách giáo khoa", "HĐND": "hội đồng nhà trường", "ĐHQG": "đại học quốc gia", # === NHÓM CÔNG NGHỆ THÔNG TIN === "CNTT": "công nghệ thông tin", "VT": "viễn thông", "CĐS": "chuyển đổi số", "ATTT": "an toàn thông tin", "CSDL": "cơ sở dữ liệu", "AI": "trí tuệ nhân tạo", "IT": "ai ti" } for tat, day_du in tu_viet_tat.items(): tat_escaped = re.escape(tat) pattern = r'(? 0: danh_sach_audio.append(silence) if i < len(doan_vans) - 1 and thoi_gian_nghi > 0: danh_sach_audio.append(silence) if danh_sach_audio: audio_final = np.concatenate(danh_sach_audio) else: audio_final = np.zeros(0, dtype=np.float32) temp_dir = tempfile.mkdtemp() output_path = os.path.join(temp_dir, "Audio_DocDon.wav") sf.write(output_path, audio_final, sample_rate) del tts gc.collect() return output_path, gr.update(visible=False), gr.update(interactive=True, value=output_path) # 4. HÀM TẠO AUDIO HỘI THOẠI NHIỀU GIỌNG def text_to_speech_hoi_thoai(text, g_voice, g_rate, g_nghi, *args): if not text.strip(): return None, gr.update(visible=True, value="Vui lòng nhập văn bản."), None char_names = args[0:MAX_CHARS] char_voices = args[MAX_CHARS:2*MAX_CHARS] char_rates = args[2*MAX_CHARS:3*MAX_CHARS] char_map = {} for name, voice, rate in zip(char_names, char_voices, char_rates): if name and name.strip(): char_map[name.strip()] = {"voice": voice, "rate": rate} blocks = [] current_char = None current_text = [] for line in text.split('\n'): line = line.strip() if not line: continue match = re.match(r'^([A-ZÀ-Ỹa-zà-ỹ0-9\s_]{1,40}):(.*)', line) if match: if current_text: blocks.append({"char": current_char, "text": " ".join(current_text)}) current_char = match.group(1).strip() dialogue = match.group(2).strip() current_text = [dialogue] if dialogue else [] else: current_text.append(line) if current_text: blocks.append({"char": current_char, "text": " ".join(current_text)}) if not blocks: return None, gr.update(visible=True, value="Không tìm thấy cấu trúc hội thoại."), None temp_dir = tempfile.mkdtemp() zip_path = os.path.join(temp_dir, "Audio_HoiThoai_Full.zip") merged_path = os.path.join(temp_dir, "00_ToanBo_CauChuyen_Gop.wav") wav_files = [] danh_sach_audio_gop = [] sample_rate = 24000 silence = np.zeros(int(sample_rate * g_nghi), dtype=np.float32) tts_cache = {} for idx, block in enumerate(blocks): char = block["char"] text_chunk = block["text"] if not text_chunk: continue voice = char_map.get(char, {}).get("voice", g_voice) rate = char_map.get(char, {}).get("rate", g_rate) if voice not in tts_cache: tts_cache[voice] = KokoroVietnamese(device="cpu", voice=voice) tts = tts_cache[voice] van_ban_da_xu_ly = chuan_hoa_van_ban(text_chunk) chunk_audio_list = [] doan_vans = [p.strip() for p in van_ban_da_xu_ly.split('\n') if p.strip()] for i, doan in enumerate(doan_vans): caus = cat_cau_thong_minh(doan) for j, cau in enumerate(caus): audio, _ = tts.synthesize(cau, speed=rate) chunk_audio_list.append(audio) if j < len(caus) - 1 and cau[-1] in '.!?' and g_nghi > 0: chunk_audio_list.append(silence) if i < len(doan_vans) - 1 and g_nghi > 0: chunk_audio_list.append(silence) if chunk_audio_list: final_chunk_audio = np.concatenate(chunk_audio_list) else: final_chunk_audio = np.zeros(0, dtype=np.float32) short_text = xoa_ky_tu_dac_biet(text_chunk[:25]) char_clean = xoa_ky_tu_dac_biet(char) if char else "NguoiKe" file_name = f"{idx+1:02d}_{char_clean}_{short_text}.wav" file_path = os.path.join(temp_dir, file_name) sf.write(file_path, final_chunk_audio, sample_rate) wav_files.append(file_path) danh_sach_audio_gop.append(final_chunk_audio) if idx < len(blocks) - 1 and g_nghi > 0: danh_sach_audio_gop.append(silence) if danh_sach_audio_gop: audio_final_gop = np.concatenate(danh_sach_audio_gop) else: audio_final_gop = np.zeros(0, dtype=np.float32) sf.write(merged_path, audio_final_gop, sample_rate) with zipfile.ZipFile(zip_path, 'w') as zipf: zipf.write(merged_path, "00_ToanBo_CauChuyen_Gop.wav") for file in wav_files: zipf.write(file, f"Tach_Rieng/{os.path.basename(file)}") for v_name in list(tts_cache.keys()): del tts_cache[v_name] del tts_cache gc.collect() return merged_path, gr.update(visible=False), gr.update(interactive=True, value=zip_path) # HÀM QUÉT NHÂN VẬT def quet_nhan_vat(text, default_voice): if not text: return [gr.update(visible=False)] * MAX_CHARS + [gr.update()] * MAX_CHARS + [gr.update()] * MAX_CHARS + [gr.update()] * MAX_CHARS + [gr.update(visible=False)] pattern = r'^([A-ZÀ-Ỹa-zà-ỹ0-9\s_]{1,40}):' matches = re.findall(pattern, text, re.MULTILINE) unique_chars = [] for m in matches: name = m.strip() if name and name not in unique_chars: unique_chars.append(name) has_chars = len(unique_chars) > 0 row_updates = [] name_updates = [] voice_updates = [] rate_updates = [] for i in range(MAX_CHARS): if i < len(unique_chars): row_updates.append(gr.update(visible=True)) name_updates.append(gr.update(value=unique_chars[i])) voice_updates.append(gr.update(value=default_voice)) rate_updates.append(gr.update(value=1.0)) else: row_updates.append(gr.update(visible=False)) name_updates.append(gr.update()) voice_updates.append(gr.update()) rate_updates.append(gr.update()) # Trả về thêm 1 biến cập nhật hiển thị thanh trượt thời gian nghỉ ở cuối return tuple(row_updates + name_updates + voice_updates + rate_updates + [gr.update(visible=has_chars)]) def doc_noi_dung_file(file_path): if not file_path: return "" try: ext = os.path.splitext(file_path)[1].lower() noidung = "" if ext in ['.txt', '.srt']: with open(file_path, "r", encoding="utf-8") as f: noidung = f.read() elif ext == '.docx': try: import docx doc = docx.Document(file_path) noidung = "\n".join([para.text for para in doc.paragraphs]) except ImportError: return "Lỗi: Hệ thống chưa cài đặt 'python-docx'." else: return f"Định dạng {ext} chưa được hỗ trợ." if len(noidung) > 1500: noidung = noidung[:1500] return noidung except Exception as e: return f"Lỗi khi đọc file: {e}" # 5. DANH SÁCH GIỌNG ĐỌC danh_sach_giong = [ ("Đức Anh (0,9)", "hung_thinh"), ("Tuấn Kiệt (1)", "manh_dung"), ("Thế Vinh (1)", "phat_tai"), ("Hồng Sơn (1,1)", "thanh_dat"), ("Duy Mạnh (0,9)", "duc_duy"), ("────────────────", "duong_phan_cach"), ("Khánh Vy (0,9)", "diem_trinh"), ("Tuyết Nhung (0,9)", "mai_linh"), ("Như Huỳnh (1,2)", "mai_loan"), ("Minh Thư (1,1)", "my_yen"), ("Thanh Thúy (1)", "ngoc_huyen"), ("Thúy Diễm (0,9)", "thuc_trinh"), ("Tú Uyên (1,3)", "storyvert"), ("Hà Giang (0,9)", "duc_an"), ("Ngọc Mai (0,9)", "tuan_ngoc") ] css_tuy_chinh = """ #cot_phai * { font-weight: normal !important; } #cot_phai label span, #cot_phai .block-title, #cot_phai .wrap-inner { font-size: 14px !important; color: #111 !important; } #nhom_nut_chuc_nang { display: flex; flex-direction: column; gap: 16px; margin-top: 5px; margin-bottom: 0px; } #nut_tao_giong { color: #ffffff !important; font-size: 18px !important; font-weight: bold !important; margin: 0 !important; } #nut_tai_xuong { font-size: 18px !important; font-weight: bold !important; margin: 0 !important; transition: all 0.3s ease; } #nut_tai_xuong:not([disabled]) { background: #2563eb !important; color: white !important; border: none !important; box-shadow: 0 4px 6px rgba(37, 99, 235, 0.2) !important; } #khu_vuc_quet { background: #f8fafc; padding: 12px; border-radius: 8px; border: 1px solid #e2e8f0; } #nut_quet { font-weight: bold !important; background: #10b981 !important; color: white !important; margin-bottom: 15px !important; border: none !important;} #nut_tao_hoi_thoai { background: #8b5cf6 !important; color: white !important; font-size: 16px !important; font-weight: bold !important; margin-top: 15px !important; border: none !important;} /* Ép nhỏ khung upload file */ #tai_file_box > div, #tai_file_box .wrap { min-height: 100px !important; height: 100px !important; padding: 5px !important; } #tai_file_box .wrap svg, #tai_file_box .wrap span { display: none !important; } /* Làm đậm tiêu đề nhân vật */ #khu_vuc_quet label span { font-weight: bold !important; color: #111 !important; } /* Thêm đoạn này để ép ô Audio luôn hiển thị với chiều cao 120px và có khung viền */ #audio_ketqua, #audio_ketqua *, #audio_ketqua > div { box-shadow: none !important; -webkit-box-shadow: none !important; } #audio_ketqua { min-height: 120px !important; background-color: #ffffff !important; border: 2px solid rgba(128, 128, 128, 0.4) !important; border-radius: 8px !important; margin-bottom: 12px !important; } #audio_ketqua > div { border: none !important; } #noidung_text textarea { font-size: 20px; line-height: 1.6; height: 65vh !important; border: 2px solid rgba(128, 128, 128, 0.4) !important; border-radius: 8px !important; padding: 16px; resize: none; } /* Ẩn nút share trên thanh kết quả audio */ #audio_ketqua button[aria-label="Share"], #audio_ketqua .share-button, #audio_ketqua a[title="Share"] { display: none !important; } /* Ẩn các dòng chữ Use via API, Built with Gradio... ở cuối trang */ footer { display: none !important; } """ with gr.Blocks(title="AI Voice Studio") as giao_dien: default_voice = "hung_thinh" hidden_default_voice = gr.Textbox(value=default_voice, visible=False) with gr.Row(): with gr.Column(scale=4): ket_qua_audio = gr.Audio(show_label=False, elem_id="audio_ketqua") nhap_van_ban = gr.Textbox(show_label=False, container=False, lines=25, placeholder="Nhập văn bản:\nCông chúa: Xin chào hoàng tử.\nHoàng tử: Xin chào công chúa xinh đẹp.", elem_id="noidung_text", max_length=1500) # Thêm link web, thanh chỉnh cỡ chữ và bộ đếm ngang hàng bằng Flexbox gr.HTML("""
phankimtu.com
0 / 1500 ký tự
""") with gr.Column(scale=1, elem_id="cot_phai"): tai_file = gr.File(label="📂 Chèn file (.txt, .srt, .docx)", file_types=[".txt", ".srt", ".docx"], type="filepath", elem_id="tai_file_box") chon_giong = gr.Dropdown(choices=danh_sach_giong, value=default_voice, label="Chọn giọng đọc") chinh_toc_do = gr.Slider(minimum=0.5, maximum=2.0, value=1.0, step=0.05, label="Tốc độ đọc chung") thoi_gian_nghi = gr.Slider(minimum=0.0, maximum=3.0, value=0.5, step=0.1, label="Nghỉ sau câu/dòng (giây)") with gr.Column(elem_id="nhom_nut_chuc_nang"): nut_bam = gr.Button("🔊 Tạo Giọng Đọc Đơn", variant="primary", elem_id="nut_tao_giong") nut_tai_xuong = gr.DownloadButton("⬇️ Tải xuống âm thanh", interactive=False, elem_id="nut_tai_xuong") canh_bao_md = gr.Markdown(visible=False) with gr.Column(elem_id="khu_vuc_quet"): nut_quet = gr.Button("🔍 Quét Nhân Vật", elem_id="nut_quet") # Thanh trượt nghỉ cho chế độ hội thoại (mặc định ẩn) thoi_gian_nghi_hoi_thoai = gr.Slider(minimum=0.0, maximum=3.0, value=0.5, step=0.1, label="Nghỉ sau câu/dòng (giây)", visible=False) char_rows, char_names, char_voices, char_rates = [], [], [], [] for i in range(MAX_CHARS): with gr.Row(visible=False, elem_classes="dong_nhan_vat") as row: name = gr.Textbox(label=f"Nhân vật {i+1}", interactive=False, scale=1) voice = gr.Dropdown(choices=danh_sach_giong, value=default_voice, show_label=False, scale=2) rate = gr.Number(value=1.0, step=0.05, show_label=False, scale=1) char_rows.append(row) char_names.append(name) char_voices.append(voice) char_rates.append(rate) nut_tao_hoi_thoai = gr.Button("⚡ Tạo Audio Hội Thoại (ZIP)", elem_id="nut_tao_hoi_thoai") js_dem_ky_tu = "(text) => { let count = text ? text.length : 0; let el = document.getElementById('hien_thi_ky_tu'); if (el) { el.innerText = count + ' / 1500 ký tự'; el.style.color = count >= 1500 ? '#dc2626' : '#555'; } }" tai_file.upload(fn=doc_noi_dung_file, inputs=tai_file, outputs=nhap_van_ban) nhap_van_ban.change(fn=None, inputs=nhap_van_ban, js=js_dem_ky_tu) nut_bam.click( fn=text_to_speech_don, inputs=[nhap_van_ban, chon_giong, chinh_toc_do, thoi_gian_nghi], outputs=[ket_qua_audio, canh_bao_md, nut_tai_xuong] ) # Cập nhật danh sách outputs để nhận trạng thái ẩn/hiện của thanh trượt hội thoại all_scan_outputs = char_rows + char_names + char_voices + char_rates + [thoi_gian_nghi_hoi_thoai] nut_quet.click( fn=quet_nhan_vat, inputs=[nhap_van_ban, hidden_default_voice], outputs=all_scan_outputs ) # Nối thanh trượt hội thoại vào inputs thay vì thanh trượt chung hoithoai_inputs = [nhap_van_ban, chon_giong, chinh_toc_do, thoi_gian_nghi_hoi_thoai] + char_names + char_voices + char_rates nut_tao_hoi_thoai.click( fn=text_to_speech_hoi_thoai, inputs=hoithoai_inputs, outputs=[ket_qua_audio, canh_bao_md, nut_tai_xuong] ) if __name__ == "__main__": giao_dien.launch(css=css_tuy_chinh)