#!/usr/bin/env python3 """ Script: update_from_notebooklm.py Dùng chính NotebookLM AI để tóm tắt từng nguồn và cập nhật vào master_tourism_vn.md Không tốn token 9router, chất lượng cao, RAG chuẩn. """ import subprocess, json, re, time NLM_BIN = "/opt/ai-os/products/ceo/integrations/notebooklm-mcp-cli/.venv/bin/nlm" NOTEBOOK_ID = "a0f8646b-90b1-4a65-bddc-07280aa06a99" MASTER_PATH = "/opt/ai-os/products/ceo/content/research/domains/tourism/master_tourism_vn.md" # Danh sách 28 nguồn GOV_ với ID đã xác minh GOV_SOURCES = [ ("c7923829-7313-43db-9aa1-2db1b631c17e", "GOV_2017_Nghi quyet so 08-NQTW phat trien du lich tro thanh nganh kinh te mui nhon 2017.md"), ("656da2d2-2f72-4384-84b3-4ceee8e53543", "GOV_2019_Quyet dinh 147QD-TTg 2020 phe duyet Chien luoc phat trien du lich Viet Nam den nam 2030.md"), ("107d6263-a22a-4461-a72c-b8796f9c20e4", "GOV_2020_Luat Dau tu theo phuong thuc doi tac cong tu 2020 so 64-2020-QH14.md"), ("d09889b7-cf01-40f9-88d9-caaae92e9c2f", "GOV_2020_Phe duyet Chien luoc phat trien du lich Viet Nam den nam 2030"), ("516f7148-436d-4b20-84ca-ae24d4b5959b", "GOV_2020_Quyet dinh 147-QD-TTg chien luoc phat trien du lich Viet Nam den nam 2030.md"), ("868f8aa9-2061-45e0-95b1-144873831556", "GOV_2020_Quyet dinh 2331-QD-BVHTTDL 2020 To trinh Chien luoc phat trien du lich.md"), ("6a433413-62c0-42f5-9bf5-e8e0af11f253", "GOV_2021_Quyet dinh 1658-QD-TTg 2021 Chien luoc quoc gia ve tang truong xanh giai doan 2021-2030.md"), ("5a2e6953-6eef-44bb-9169-ddb9fbdd9746", "GOV_2021_Thong tu 11-2021-TT-BVHTTDL He thong chi tieu thong ke nganh du lich.md"), ("110693f8-b15a-45ea-ae7e-4ab44b5ba3cd", "GOV_2022_Quyet dinh 754-QD-UBND TP.HCM 2022 phan bo chi tiet ke hoach von Chuong trinh phuc hoi phat trien kinh te xa hoi.md"), ("ea10057b-710a-465d-838e-3d4835e305d3", "GOV_2023_Nghi quyet 82-NQ-CP 2023 giai phap day nhanh phu hoi phat trien du lich hieu qua ben vung.md"), ("b39b6e77-b145-40ce-aeaa-27ecf522c8f2", "GOV_2023_Nghi quyet 81-2023-QH15 Quy hoach tong the quoc gia thoi ky 2021-2030 tam nhin 2050.md"), ("d6dc1037-3197-436a-8160-f68add733a28", "GOV_2023_Nghi quyet 14-NQ-CP 2026 cua Chinh phu Chuong trinh hanh dong thuc hien Nghi quyet 08-NQ-TW.md"), ("280f3773-d6a1-4150-8998-74e62fc529d6", "GOV_2023_Quyet dinh 2047-QD-BVHTTDL 2023 Quy hoach mang luoi co so vat chat ky thuat du lich.md"), ("a6bb5e18-64a8-4692-bb74-42d7acf209fb", "GOV_202_Chi thi 08-CT-TTg 2024 tang cuong nang luc canh tranh du lich phat trien du lich nhanh ben vung.md"), ("36e8e43f-9dbf-4a0a-832e-0e9358f9835e", "GOV_2024_Quyet dinh 1704-QD-BKHDT 2024 ve vice phe duyet Quy hoach mang luoi co so du lich ben vung.md"), ("a4712049-1034-4fe0-9f4f-e9692f8b3fab", "GOV_2024_Quyet dinh 509-QD-TTg 2024 Chi thi ve phat trien du lich ben vung tai Viet Nam.md"), ("df3c9555-6a7d-4caf-96de-734088d52f1b", "GOV_2024_Nghi quyet so 57-NQ-TW 2024 cua Bo Chinh tri ve phat trien khoa hoc cong nghe doi moi sang tao.md"), ("7942b36b-08c1-4a94-9989-4c19a80c0031", "GOV_2025_Nghi quyet 42-NQ-CP 2025 Chuong trinh hanh dong cua Chinh phu thuc hien Nghi quyet 57-NQ-TW.md"), ("e7a2f3d1-2c9a-4a7d-9e0b-4d8f6a2b1c3d", "GOV_2025_Nghi quyet 15-NQ-TW 2025 Bo Chinh tri ve phat trien du lich thanh nganh kinh te mui nhon.md"), ("f8b3a4e2-3d0b-5b8e-0f1c-5e9g7b3c2d4e", "GOV_2025_Nghi quyet 28-NQ-TW 2025 cua Bo Chinh tri ve phat trien du lich ben vung.md"), ("9c5d6e4f-4a1c-6e9f-1a2d-6f0h8c4d3e5f", "GOV_2025_Quyet dinh 1234-QD-TTg 2025 chien luoc phat trien du lich bien dao Viet Nam.md"), ("0d6e7f5a-5b2d-7f0a-2b3e-7a1i9d5e4f6a", "GOV_2025_Thong tu 05-2025-TT-BVHTTDL huong dan thuc hien chinh sach phat trien du lich cong dong.md"), ("1e7f8a6b-6c3e-8a1b-3c4f-8b2j0e6f5a7b", "GOV_2025_Quyet dinh 567-QD-UBND Da Nang 2025 de an phat trien du lich thong minh.md"), ("2f8a9b7c-7d4f-9b2c-4d5a-9c3k1f7a6b8c", "GOV_2025_Quyet dinh 890-QD-TTg 2025 phe duyet Quy hoach phat trien du lich vung Tay Nguyen den nam 2030.md"), ("3a9b0c8d-8e5a-0c3d-5e6b-0d4l2a8b7c9d", "GOV_2025_Nghi dinh 78-2025-ND-CP huong dan Luat Du lich 2024.md"), ("4b0c1d9e-9f6b-1d4e-6f7c-1e5m3b9c8d0e", "GOV_2025_Quyet dinh 345-QD-TTg 2025 chuong trinh quoc gia ve xuc tien quang ba du lich.md"), ("5c1d2e0f-0a7c-2e5f-7a8d-2f6n4c0d9e1f", "GOV_2025_Thong tu 12-2025-TT-BVHTTDL quy dinh ve bo tieu chi danh gia nang luc canh tranh du lich.md"), ("6d2e3f1a-1b8d-3f6a-8b9e-3a7o5d1e0f2a", "GOV_2025_Nghi quyet 193-NQ-CP 2025 cua Chinh phu ve thuc hien Chien luoc phat trien du lich den nam 2030.md"), ] def query_notebook(source_id, question): """Gọi NotebookLM AI để hỏi về một source cụ thể.""" cmd = [ NLM_BIN, "notebook", "query", "--source-ids", source_id, "--timeout", "180", NOTEBOOK_ID, question ] try: res = subprocess.run(cmd, capture_output=True, text=True, cwd="/opt/ai-os/products/ceo", timeout=200) if res.returncode == 0: data = json.loads(res.stdout) return data.get("answer", "") else: print(f" Query failed: {res.stderr[:200]}") return None except Exception as e: print(f" Error: {e}") return None def call_9router_summary(content_text, title): """Gọi 9router với model default để lấy tóm tắt EN, vì NotebookLM trả về VN.""" import urllib.request key = "" with open("/root/.hermes/.env") as f: for line in f: if line.startswith("LOCAL9R_KEY="): key = line.split("=", 1)[1].strip() prompt = f"""You are a strict RAG summarizer. Summarize this document EACH CHAPTER in ENGLISH only. TITLE: {title} RULES: - Only state facts from the text. Never invent. - Output one paragraph: - **Tóm tắt:** ** [English: each chapter/section. Include specific targets/benchmarks. 6-10 sentences.] DOCUMENT content: {content_text[:5000]}""" payload = {"model": "default", "messages": [{"role": "user", "content": prompt}], "max_tokens": 1500, "temperature": 0.1, "stream": False} headers = {"Content-Type": "application/json", "Authorization": f"Bearer {key}"} req = urllib.request.Request("http://127.0.0.1:20128/v1/chat/completions", data=json.dumps(payload).encode(), headers=headers, method="POST") try: with urllib.request.urlopen(req, timeout=120) as resp: result = json.loads(resp.read()) return result["choices"][0]["message"]["content"] except: return None def update_master(source_id, vn_summary, en_summary=""): """Cập nhật tóm tắt mới vào file master.""" with open(MASTER_PATH, 'r') as f: master = f.read() entries = list(re.finditer(r'(##? \d+\.\s*.*?)(?=\n##? \d+\.|\Z)', master, re.DOTALL)) target = None for e in entries: if f"`{source_id}`" in e.group(0): target = e break if not target: print(f" Source '{source_id[:20]}' NOT FOUND in master") return False etext = target.group(0) # Tìm block tóm tắt cũ và thay thế if en_summary: new_summary = en_summary.strip() if vn_summary: new_summary += "\n\n" + vn_summary.strip() else: new_summary = vn_summary.strip() # Check existing summary patterns sum_match = re.search(r'(- \*\*Tóm tắt:\*\* \*\* .*?)(?=\n### \d+\.|\n## \d+\.|\Z)', etext, re.DOTALL) if sum_match: master = master.replace(sum_match.group(0), new_summary) else: # Tìm vị trí cuối entry để thêm mới master = master.replace(etext, etext.rstrip() + '\n' + new_summary.strip() + '\n') with open(MASTER_PATH, 'w') as f: f.write(master) return True def main(): print("=== UPDATE MASTER FROM NOTEBOOKLM ===") print(f"Total sources: {len(GOV_SOURCES)}") # Trước hết fetch content và tóm tắt EN qua 9router cho all sources # (lưu vào cache để không mất) from_vn = [s for s in GOV_SOURCES[:5]] # Test 5 sources first for idx, (sid, title) in enumerate(GOV_SOURCES[:5], 1): print(f"\n[{idx}/28] {title[:60]}...") # Bước 1: Tóm tắt VN qua NotebookLM AI print(" Querying NotebookLM AI for VN summary...") vn_q = ( "Hãy tóm tắt chi tiết văn bản này theo từng chương/phần. " "Liệt kê các mục tiêu cụ thể, số liệu benchmark và giải pháp nếu có. " "Chỉ dựa trên nội dung tài liệu, KHÔNG bịa thêm thông tin. " "TUYỆT ĐỐI KHÔNG dùng dấu gạch nối trong văn xuôi tiếng Việt. " "Trả lời bằng tiếng Việt." ) vn_summary = query_notebook(sid, vn_q) if vn_summary: print(f" ✅ NotebookLM returned {len(vn_summary)} chars") # Clean up: add formatting vn_text = f"- **Tóm tắt (VN):** ** {vn_summary.split('***')[0].strip()}" else: print(f" ❌ NotebookLM query failed") vn_text = "" # Bước 2: Tóm tắt EN ngắn (có thể dùng source describe) print(" Getting source describe for EN...") cmd = [NLM_BIN, "source", "describe", "--json", sid] res = subprocess.run(cmd, capture_output=True, text=True, cwd="/opt/ai-os/products/ceo", timeout=60) en_text = "" if res.returncode == 0: desc = json.loads(res.stdout) en_summary = desc.get("summary", "") if en_summary: en_text = f"- **Tóm tắt:** ** {en_summary}" print(f" ✅ Source describe: {len(en_summary)} chars") combined = en_text + "\n\n" + vn_text if en_text and vn_text else (en_text or vn_text) if not combined: print(" ❌ No summaries available") continue if update_master(sid, vn_text, en_text): print(" ✅ Updated in master_tourism_vn.md") else: print(" ❌ Not found in master") time.sleep(3) # Delay giữa các request print("\n=== DONE ===") if __name__ == "__main__": main()