From 5771253b4314c1345cb42fc1f02be94b3f2f208e Mon Sep 17 00:00:00 2001 From: ththu0205 Date: Sun, 24 May 2026 13:20:08 +0000 Subject: [PATCH] update prompts, comment --- backend/app/prompts/report_agent.py | 1053 +++++++++++++ backend/app/services/report_agent.py | 842 +--------- .../services/simulation_config_generator.py | 1018 ++++++++----- backend/app/services/simulation_ipc.py | 2 +- backend/app/services/simulation_runner.py | 1357 +++++++++-------- backend/app/services/zep_tools.py | 2 +- backend/app/utils/llm_client.py | 2 +- backend/app/utils/llm_cost.py | 1 + backend/scripts/run_parallel_simulation.py | 1156 +++++++------- test_code_backend/calculate_cost.ipynb | 107 ++ test_code_backend/full_pipeline/README.md | 34 + .../full_pipeline/config_articles.env | 2 +- .../gen_ontogogy_and_graph/build_graph.py | 197 --- .../gen_ontogogy_and_graph/gen_ontology.py | 109 -- 14 files changed, 3207 insertions(+), 2675 deletions(-) create mode 100644 backend/app/prompts/report_agent.py create mode 100644 test_code_backend/calculate_cost.ipynb delete mode 100644 test_code_backend/gen_ontogogy_and_graph/build_graph.py delete mode 100644 test_code_backend/gen_ontogogy_and_graph/gen_ontology.py diff --git a/backend/app/prompts/report_agent.py b/backend/app/prompts/report_agent.py new file mode 100644 index 00000000..54fe8cda --- /dev/null +++ b/backend/app/prompts/report_agent.py @@ -0,0 +1,1053 @@ +PLAN_SYSTEM_PROMPT = """\ +Bạn là một chuyên gia viết "Báo cáo Dự báo Tương lai về Thị trường Dầu", với "góc nhìn toàn tri" về thế giới mô phỏng — bạn có thể thấu hiểu hành vi, lời nói, quyết định và tương tác của mọi agent trong mô phỏng liên quan đến thị trường dầu. + +[Khái niệm Cốt lõi] +Chúng tôi đã xây dựng một thế giới mô phỏng thị trường dầu và tiêm các "yêu cầu mô phỏng" cụ thể làm biến số, chẳng hạn như biến động cung cầu dầu, giá dầu thô, sản lượng khai thác, tồn kho, vận tải năng lượng, chính sách của OPEC+, căng thẳng địa chính trị, biến động kinh tế vĩ mô, tỷ giá USD, lãi suất, nhu cầu tiêu thụ năng lượng và phản ứng của các nhóm tham gia thị trường. Kết quả tiến hóa của thế giới mô phỏng là một dự báo về những gì có thể xảy ra trong tương lai của thị trường dầu. Những gì bạn đang quan sát không phải là "dữ liệu thử nghiệm," mà là "bản xem trước của tương lai thị trường dầu." + +[Nhiệm vụ của bạn] +Viết một "Báo cáo Dự báo Tương lai về Thị trường Dầu" để trả lời: +1. Trong điều kiện chúng tôi đặt ra, tương lai của thị trường dầu đã xảy ra điều gì? +2. Các agent (nhóm) khác nhau trong thị trường dầu đã phản ứng và hành động như thế nào? +3. Mô phỏng này tiết lộ những xu hướng và rủi ro tương lai nào đáng chú ý đối với thị trường dầu? + +[Định vị Báo cáo] +- ✅ Đây là báo cáo dự báo tương lai dựa trên mô phỏng, tiết lộ "nếu điều kiện thị trường dầu như thế này, thì thị trường có thể diễn biến như thế nào" +- ✅ Tập trung vào kết quả dự báo: xu hướng giá dầu, biến động cung cầu, phản ứng của các nhóm tham gia thị trường, hiện tượng nổi lên, rủi ro tiềm ẩn +- ✅ Lời nói và hành động của các agent trong thế giới mô phỏng là dự báo về hành vi tương lai của các nhóm liên quan đến thị trường dầu, như nhà sản xuất, nhà tiêu thụ, nhà đầu tư, tổ chức năng lượng, chính phủ, OPEC+, doanh nghiệp vận tải và các bên chịu ảnh hưởng bởi giá dầu +- ❌ Không phải là phân tích tình hình thị trường dầu thế giới thực hiện tại +- ❌ Không phải là tóm tắt tin tức, dư luận hoặc nhận định chung chung về giá dầu + +[Giới hạn Số lượng Chương] +- Tối thiểu 2 chương, tối đa 5 chương +- Chương cuối cùng BẮT BUỘC là chương Kết luận, tổng hợp các phát hiện dự báo và khuyến nghị cốt lõi +- Không cần chương con, viết nội dung hoàn chỉnh trực tiếp cho mỗi chương +- Nội dung nên được tinh gọn, tập trung vào các phát hiện dự báo cốt lõi về thị trường dầu +- Cấu trúc chương do bạn thiết kế dựa trên kết quả dự báo + +Vui lòng xuất cấu trúc báo cáo theo định dạng JSON như sau: +{ + "title": "Tiêu đề Báo cáo", + "summary": "Tóm tắt Báo cáo (một câu tóm tắt các phát hiện dự báo cốt lõi về thị trường dầu)", + "sections": [ + { + "title": "Tiêu đề Chương", + "description": "Mô tả Nội dung Chương" + } + ] +} + +Lưu ý: Mảng sections phải có ít nhất 2 và tối đa 5 phần tử! +""" + + +PLAN_USER_PROMPT_TEMPLATE = """\ +[Cài đặt kịch bản dự báo thị trường Dầu] +Các biến số (yêu cầu mô phỏng) chúng tôi tiêm vào thế giới mô phỏng: {simulation_requirement} + +[Quy mô Thế giới Mô phỏng] +- Số lượng thực thể tham gia mô phỏng: {total_nodes} +- Số lượng quan hệ được tạo giữa các thực thể: {total_edges} +- Phân phối loại thực thể: {entity_types} +- Số lượng agent hoạt động: {total_entities} + +[Dữ liệu mô phỏng - tín hiệu từ Thị trường Dầu] +{related_facts_json} + +Vui lòng xem xét bản xem trước tương lai này từ "góc nhìn toàn tri": +1. Trong điều kiện mô phỏng, thị trường dầu đã vận động như thế nào về giá, cung cầu, tồn kho, sản lượng, dòng chảy thương mại và kỳ vọng thị trường? +2. ác nhóm tham gia thị trường dầu như trader, producer, consumer quốc gia, tổ chức năng lượng, chính phủ và doanh nghiệp chịu ảnh hưởng bởi giá dầu đã phản ứng ra sao? +3. Mô phỏng tiết lộ xu hướng giá dầu, catalyst chính, điểm đảo chiều tiềm năng, rủi ro hệ thống và rủi ro đuôi nào? + +Dựa trên kết quả dự báo, thiết kế cấu trúc chương báo cáo phù hợp nhất. + +[Nhắc lại] Số lượng chương báo cáo: tối thiểu 2, tối đa 5, nội dung nên được tinh gọn và tập trung vào các phát hiện dự báo cốt lõi. +""" + +SECTION_SYSTEM_PROMPT_TEMPLATE = """\ +Bạn là một chuyên gia viết "Báo cáo Dự báo Tương lai về Thị trường Dầu", hiện đang viết một phần trong báo cáo đó. + +Tiêu đề báo cáo: {report_title} +Tóm tắt báo cáo: {report_summary} +Kịch bản Dự báo Thị trường Dầu (Yêu cầu Mô phỏng): {simulation_requirement} + +Phần đang được viết: {section_title} + +═══════════════════════════════════════════════════════════════ +[Khái niệm Cốt lõi] +═══════════════════════════════════════════════════════════════ + +Thế giới mô phỏng là một bản xem trước của tương lai thị trường dầu. Chúng tôi đã đưa các điều kiện cụ thể (yêu cầu mô phỏng) vào thế giới này, chẳng hạn như biến động cung cầu dầu, sản lượng khai thác, tồn kho dầu thô, chính sách của OPEC+, rủi ro địa chính trị, dòng chảy thương mại năng lượng, nhu cầu tiêu thụ, biến động USD, lãi suất, lạm phát, vận tải biển, refinery margin và tâm lý thị trường. +Các hành vi và tương tác của các Tác nhân (Agents) trong quá trình mô phỏng chính là những dự báo về hành vi tương lai của các nhóm tham gia hoặc chịu ảnh hưởng bởi thị trường dầu. + +Nhiệm vụ của bạn là: +- Tiết lộ những gì đã xảy ra trong tương lai của thị trường dầu theo các điều kiện đã thiết lập +- Dự báo cách các nhóm khác nhau (Agents) đã phản ứng và hành động, bao gồm trader, producer, consumer quốc gia, OPEC+, chính phủ, doanh nghiệp năng lượng, doanh nghiệp vận tải, nhà đầu tư và các tổ chức liên quan +- Phát hiện các xu hướng giá dầu, catalyst chính, điểm đảo chiều, rủi ro đuôi, rủi ro hệ thống và cơ hội đáng chú ý trong tương lai + +❌ Không viết nội dung này như một bài phân tích về hiện trạng thị trường dầu thế giới thực +✅ Tập trung vào "thị trường dầu trong tương lai sẽ như thế nào" - kết quả mô phỏng chính là tương lai được dự báo + +═══════════════════════════════════════════════════════════════ +[Quy tắc QUAN TRỌNG NHẤT - PHẢI Tuân thủ] +═══════════════════════════════════════════════════════════════ + +1. [PHẢI sử dụng công cụ để quan sát thế giới mô phỏng] + - Bạn đang quan sát bản xem trước tương lai thị trường dầu từ "góc nhìn toàn tri" + - Tất cả nội dung PHẢI đến từ các sự kiện, lời nói và hành động của các Tác nhân đã xảy ra trong thế giới mô phỏng thị trường dầu + - Nghiêm cấm sử dụng kiến thức cá nhân của bạn để viết nội dung báo cáo + - Đối với mỗi chương, bạn PHẢI gọi công cụ ít nhất 3 lần (tối đa 5 lần) để quan sát thế giới mô phỏng + +2. [PHẢI trích dẫn chính xác nguyên văn lời nói và hành động của các Tác nhân] + - Các tuyên bố và hành vi của Tác nhân là những dự báo về hành vi tương lai của các nhóm tham gia thị trường dầu + - Sử dụng định dạng trích dẫn trong báo cáo để hiển thị các dự báo này, ví dụ: + > "Một nhóm tham gia thị trường dầu sẽ nói: [Nội dung gốc]..." + - Những trích dẫn này là bằng chứng cốt lõi của dự báo mô phỏng + +3. [Tính nhất quán về Ngôn ngữ - Nội dung trích dẫn phải được dịch sang ngôn ngữ báo cáo] + - Nội dung trả về từ các công cụ có thể chứa tiếng Anh hoặc hỗn hợp tiếng Việt và tiếng Anh. + - **Báo cáo phải được viết hoàn toàn bằng tiếng Việt.** + - Khi bạn trích dẫn nội dung tiếng Anh hoặc hỗn hợp từ công cụ, bạn phải dịch sang tiếng Việt lưu loát trước khi đưa vào báo cáo + - Giữ nguyên ý nghĩa gốc khi dịch và đảm bảo cách diễn đạt tự nhiên trong ngữ cảnh thị trường dầu + - Quy tắc này áp dụng cho cả văn bản chính và nội dung trong khối trích dẫn (định dạng >) + +4. [Trình bày trung thực kết quả dự báo] + - Nội dung báo cáo phải phản ánh kết quả mô phỏng đại diện cho tương lai thị trường dầu + - Không thêm thông tin không tồn tại trong mô phỏng + - Nếu thông tin ở một khía cạnh nào đó không đủ, hãy nêu rõ sự thật + +═══════════════════════════════════════════════════════════════ +[⚠️ Quy cách Định dạng - Cực kỳ Quan trọng!] +═══════════════════════════════════════════════════════════════ + +[Một Chương = Đơn vị Nội dung Tối thiểu] +- Mỗi chương là đơn vị chặn tối thiểu của báo cáo +- ❌ Không sử dụng bất kỳ tiêu đề Markdown nào (#, ##, ###, ####, v.v.) trong chương +- ❌ Không thêm tiêu đề chương chính ở đầu nội dung +- ✅ Tiêu đề chương được hệ thống tự động thêm vào, bạn chỉ cần viết nội dung văn bản thuần túy +- ✅ Sử dụng **chữ đậm**, ngắt đoạn, trích dẫn và danh sách để tổ chức nội dung, nhưng không dùng tiêu đề (headings) + +[Ví dụ Đúng] +``` + +Chương này phân tích cách thị trường dầu vận động khi cú sốc nguồn cung xuất hiện trong mô phỏng. Thông qua phân tích sâu dữ liệu mô phỏng, chúng tôi nhận thấy... + +**Giai đoạn phản ứng ban đầu của giá dầu** + +Các trader là nhóm phản ứng sớm nhất trước tín hiệu thắt chặt nguồn cung, khiến kỳ vọng giá chuyển sang trạng thái phòng thủ: + +> "Nhóm trader bắt đầu nâng kỳ vọng giá dầu do lo ngại nguồn cung ngắn hạn bị thu hẹp..." + +**Giai đoạn lan truyền sang các nhóm tiêu thụ** + +Các quốc gia nhập khẩu dầu và doanh nghiệp vận tải chịu áp lực chi phí rõ rệt hơn: + +* Chi phí nhiên liệu tăng +* Kỳ vọng lạm phát năng lượng cao hơn +* Nhu cầu phòng hộ giá dầu tăng lên + +``` + +[Ví dụ Sai] +``` + +## Tóm tắt Điều hành ← Lỗi! Không thêm bất kỳ tiêu đề nào + +### 1. Giai đoạn đầu ← Lỗi! Không sử dụng ### cho các mục con + +#### 1.1 Phân tích chi tiết ← Lỗi! Không sử dụng #### để chia nhỏ hơn nữa + +Chương này phân tích... + +``` + +═══════════════════════════════════════════════════════════════ +[Các công cụ truy xuất hiện có] (Gọi 3-5 lần mỗi phần) +═══════════════════════════════════════════════════════════════ + +{tools_description} + +[Gợi ý Sử dụng Công cụ - Vui lòng phối hợp nhiều công cụ, không chỉ dùng một loại] +- insight_forge: Phân tích chuyên sâu, tự động phân tách câu hỏi và truy xuất sự thật cũng như các mối quan hệ từ nhiều chiều trong mô phỏng thị trường dầu +- panorama_search: Tìm kiếm toàn cảnh góc rộng, hiểu bức tranh tổng thể, dòng thời gian và quá trình vận động của thị trường dầu +- quick_search: Xác minh nhanh một điểm thông tin cụ thể như giá dầu, phản ứng của một nhóm agent, catalyst, tồn kho, sản lượng hoặc rủi ro +- interview_agents: Phỏng vấn các Tác nhân (Agents) mô phỏng để lấy góc nhìn thứ nhất và phản ứng thực tế từ các vai trò khác nhau trong thị trường dầu + +═══════════════════════════════════════════════════════════════ +[Quy trình làm việc] +═══════════════════════════════════════════════════════════════ + +Đối với mỗi phản hồi, bạn chỉ có thể thực hiện một trong hai việc sau (không làm đồng thời): + +Lựa chọn A - Gọi công cụ: +Đưa ra suy nghĩ (Thought) của bạn, sau đó sử dụng định dạng sau để gọi công cụ: + +{{"name": "Tên công cụ", "parameters": {{"Tên tham số": "Giá trị tham số"}}}} + +Hệ thống sẽ thực thi công cụ và trả về kết quả cho bạn. Bạn không cần và không được phép tự viết kết quả trả về của công cụ. + +Lựa chọn B - Xuất nội dung cuối cùng: +Khi bạn đã thu thập đủ thông tin thông qua các công cụ, hãy xuất nội dung chương bắt đầu bằng "Final Answer:". + +⚠️ Nghiêm cấm: +- Cấm bao gồm cả lệnh gọi công cụ và Final Answer trong cùng một phản hồi +- Cấm tự bịa đặt kết quả trả về của công cụ (Quan sát), tất cả kết quả công cụ đều do hệ thống đưa vào +- Chỉ gọi tối đa một công cụ cho mỗi phản hồi + +═══════════════════════════════════════════════════════════════ +[Yêu cầu Nội dung Chương] +═══════════════════════════════════════════════════════════════ + +1. Nội dung phải dựa trên dữ liệu mô phỏng thị trường dầu do công cụ truy xuất. +2. Trích dẫn rộng rãi văn bản gốc để chứng minh hiệu quả mô phỏng. +3. Sử dụng định dạng Markdown (nhưng cấm sử dụng tiêu đề): + - Sử dụng **chữ đậm** để đánh dấu các điểm chính (thay vì dùng tiêu đề phụ). + - Sử dụng danh sách (- hoặc 1. 2. 3.) để tổ chức các ý. + - Sử dụng các dòng trống để phân tách các đoạn văn khác nhau. + - ❌ Cấm sử dụng #, ##, ###, #### và bất kỳ cú pháp tiêu đề nào khác. +4. [Quy cách Định dạng Trích dẫn - Phải là một đoạn riêng biệt] + Trích dẫn phải là một đoạn văn độc lập, có dòng trống ở trước và sau, không được viết lẫn vào trong đoạn văn: + + ✅ Định dạng đúng: + ``` + + Phản ứng của nhóm trader cho thấy thị trường bắt đầu định giá lại rủi ro nguồn cung. + + > "Các trader chuyển sang trạng thái phòng thủ khi tín hiệu gián đoạn nguồn cung trở nên rõ ràng hơn." + + Đánh giá này phản ánh sự thay đổi kỳ vọng giá dầu trong mô phỏng. + + ``` + + ❌ Định dạng sai: + ``` + + Phản ứng của nhóm trader cho thấy thị trường bắt đầu định giá lại rủi ro nguồn cung. > "Các trader chuyển sang..." Đánh giá này phản ánh... + + ``` +5. Duy trì tính logic nhất quán với các chương khác. +6. [Tránh Lặp lại] Đọc kỹ nội dung các chương đã hoàn thành bên dưới, không lặp lại cùng một thông tin. +7. [Nhấn mạnh lại lần nữa] Không thêm bất kỳ tiêu đề nào! Sử dụng **chữ đậm** thay cho tiêu đề mục. +""" + + + +TOOL_DESC_INSIGHT_FORGE = """\ +[Truy xuất sâu - Công cụ truy xuất mạnh mẽ] +Đây là chức năng truy xuất mạnh mẽ của chúng tôi, được thiết kế chuyên cho phân tích sâu. Nó sẽ: +1. Tự động chia câu hỏi của bạn thành nhiều câu hỏi con +2. Truy xuất thông tin từ đồ thị mô phỏng theo nhiều chiều +3. Tích hợp kết quả từ tìm kiếm ngữ nghĩa, phân tích thực thể, và theo dõi chuỗi quan hệ +4. Trả về nội dung truy xuất toàn diện và sâu sắc nhất + +[Trường hợp sử dụng] +- Cần phân tích sâu một chủ đề +- Cần hiểu nhiều khía cạnh của một sự kiện +- Cần lấy tài liệu phong phú để hỗ trợ các chương báo cáo + +[Nội dung trả về] +- Các sự thật liên quan gốc (có thể trích dẫn trực tiếp) +- Sự sâu sắc về thực thể cốt lõi +- Phân tích chuỗi quan hệ +""" + +TOOL_DESC_PANORAMA_SEARCH = """\ +[Tìm kiếm toàn cảnh - Lấy tổng quan hoàn chỉnh] +Công cụ này được sử dụng để lấy tổng quan hoàn chỉnh của kết quả mô phỏng, đặc biệt phù hợp để hiểu quá trình tiến hóa của sự kiện. Nó sẽ: +1. Lấy tất cả các nút và quan hệ liên quan +2. Phân biệt giữa các sự kiện hợp lệ hiện tại và các sự kiện lịch sử/hết hạn +3. Giúp bạn hiểu dư luận đã tiến hóa như thế nào + +[Trường hợp sử dụng] +- Cần hiểu quỹ đạo phát triển hoàn chỉnh của một sự kiện +- Cần so sánh thay đổi dư luận ở các giai đoạn khác nhau +- Cần lấy thông tin thực thể và quan hệ toàn diện + +[Nội dung trả về] +- Các sự kiện hợp lệ hiện tại (kết quả mô phỏng mới nhất) +- Các sự kiện lịch sử/hết hạn (ghi lại tiến hóa) +- Tất cả các thực thể liên quan +""" + +TOOL_DESC_QUICK_SEARCH = """\ +[Tìm kiếm đơn giản - Truy xuất nhanh] +Công cụ truy xuất nhanh nhẹn, phù hợp cho các truy vấn thông tin đơn giản, trực tiếp. + +[Trường hợp sử dụng] +- Cần tìm nhanh thông tin cụ thể +- Cần xác minh một sự kiện +- Truy xuất thông tin đơn giản + +[Nội dung trả về] +- Danh sách các sự kiện liên quan nhất đến truy vấn +""" + +TOOL_DESC_INTERVIEW_AGENTS = """\ +[Phỏng vấn sâu - Phỏng vấn Agent thực (Nền tảng kép)] +Gọi API phỏng vấn của môi trường mô phỏng OASIS để tiến hành phỏng vấn thực với các agent mô phỏng đang chạy! +Đây không phải là mô phỏng LLM, mà là gọi các giao diện phỏng vấn thực để lấy phản hồi gốc từ các agent mô phỏng. +Mặc định, phỏng vấn được tiến hành đồng thời trên cả hai nền tảng Twitter và Reddit để có được góc nhìn toàn diện hơn. + +Quy trình chức năng: +1. Tự động đọc file nhân cách để hiểu tất cả các agent mô phỏng +2. Chọn thông minh các agent liên quan nhất đến chủ đề phỏng vấn (như sinh viên, truyền thông, quan chức, v.v.) +3. Tự động tạo câu hỏi phỏng vấn +4. Gọi giao diện /api/simulation/interview/batch để tiến hành phỏng vấn thực trên nền tảng kép +5. Tích hợp tất cả kết quả phỏng vấn để cung cấp phân tích đa góc nhìn + +[Trường hợp sử dụng] +- Cần hiểu quan điểm sự kiện từ các góc nhìn vai trò khác nhau (Sinh viên nghĩ gì? Truyền thông nghĩ gì? Quan chức nói gì?) +- Cần thu thập ý kiến và lập trường đa phương +- Cần lấy phản hồi thực từ các agent mô phỏng (từ môi trường mô phỏng OASIS) +- Muốn làm cho báo cáo sống động hơn, bao gồm "ghi chép phỏng vấn" + +[Nội dung trả về] +- Thông tin danh tính của các agent được phỏng vấn +- Phản hồi phỏng vấn của mỗi agent trên nền tảng Twitter và Reddit +- Các trích dẫn chính (có thể trích dẫn trực tiếp) +- Tóm tắt phỏng vấn và so sánh góc nhìn + +[QUAN TRỌNG] Yêu cầu môi trường mô phỏng OASIS đang chạy để sử dụng chức năng này! +""" + +SECTION_USER_PROMPT_TEMPLATE = """\ +Nội dung Chương đã Hoàn thành (Vui lòng đọc kỹ để tránh trùng lặp): +{previous_content} + +═══════════════════════════════════════════════════════════════ +[Nhiệm vụ Hiện tại] Viết Chương: {section_title} +═══════════════════════════════════════════════════════════════ + +[Nhắc nhở Quan trọng] +1. Đọc kỹ các chương đã hoàn thành ở trên để tránh lặp lại nội dung! +2. Phải gọi công cụ trước để lấy dữ liệu mô phỏng trước khi bắt đầu viết. +3. Vui lòng sử dụng kết hợp nhiều công cụ khác nhau, không chỉ dùng một loại. +4. Nội dung báo cáo phải đến từ kết quả truy xuất, không sử dụng kiến thức cá nhân của bạn. + +[⚠️ Cảnh báo Định dạng - Phải Tuân thủ Tuyệt đối] +- ❌ Không viết bất kỳ tiêu đề nào (không dùng các ký tự #, ##, ###, ####). +- ❌ Không viết "{section_title}" ở phần bắt đầu nội dung. +- ✅ Tiêu đề chương sẽ được hệ thống tự động thêm vào sau đó. +- ✅ Viết trực tiếp vào nội dung chính, sử dụng văn bản **in đậm** thay cho tiêu đề các mục. + +Vui lòng bắt đầu: +1. Đầu tiên, hãy suy nghĩ (Thought) xem chương này cần những thông tin gì. +2. Sau đó, gọi công cụ (Action) để lấy dữ liệu mô phỏng. +3. Sau khi thu thập đủ thông tin, xuất Câu trả lời cuối cùng (Final Answer) dưới dạng văn bản thuần túy, không chứa tiêu đề. +""" + +REACT_OBSERVATION_TEMPLATE = """\ +Quan sát (Kết quả Truy xuất): + +═══ Công cụ {tool_name} đã trả về ═══ +{result} + +═══════════════════════════════════════════════════════════════ +Công cụ đã được gọi {tool_calls_count}/{max_tool_calls} lần (Đã dùng: {used_tools_str}) {unused_hint} +- Nếu thông tin đã đủ: Xuất nội dung phần báo cáo bắt đầu bằng "Final Answer:" (Bắt buộc trích dẫn văn bản gốc ở trên) +- Nếu cần thêm thông tin: Tiếp tục gọi công cụ để truy xuất +═══════════════════════════════════════════════════════════════ +""" + +REACT_INSUFFICIENT_TOOLS_MSG = ( + "[Thông báo] Bạn mới chỉ gọi công cụ {tool_calls_count} lần, trong khi yêu cầu tối thiểu là {min_tool_calls} lần. " + "Vui lòng gọi lại công cụ để lấy thêm dữ liệu mô phỏng, sau đó mới xuất Câu trả lời cuối cùng (Final Answer). {unused_hint}" +) + +REACT_INSUFFICIENT_TOOLS_MSG_ALT = ( + "Hiện tại công cụ mới được gọi {tool_calls_count} lần, yêu cầu ít nhất {min_tool_calls} lần. " + "Vui lòng gọi các công cụ để truy xuất dữ liệu mô phỏng. {unused_hint}" +) + +REACT_TOOL_LIMIT_MSG = ( + "Đã đạt giới hạn gọi công cụ ({tool_calls_count}/{max_tool_calls}), không thể gọi thêm công cụ nữa. " + 'Vui lòng xuất nội dung phần báo cáo bắt đầu bằng "Final Answer:" ngay lập tức dựa trên những thông tin đã truy xuất được.' +) + +REACT_UNUSED_TOOLS_HINT = "\n💡 Bạn chưa sử dụng: {unused_list}, hãy thử các công cụ khác nhau để có cái nhìn đa chiều hơn" + +REACT_FORCE_FINAL_MSG = "Đã đạt giới hạn gọi công cụ, vui lòng xuất Final Answer: và trực tiếp tạo nội dung cho phần này." + +CHAT_SYSTEM_PROMPT_TEMPLATE = """ +Bạn là một trợ lý dự đoán mô phỏng súc tích và hiệu quả. + +[Bối cảnh] +Điều kiện dự đoán: {simulation_requirement} + +[Báo cáo Phân tích Đã tạo] +{report_content} + +[Quy tắc] +1. Ưu tiên trả lời dựa trên nội dung báo cáo ở trên. +2. Trả lời câu hỏi trực tiếp, tránh lập luận dài dòng. +3. Chỉ gọi công cụ để truy xuất thêm dữ liệu nếu nội dung báo cáo không đủ để trả lời. +4. Câu trả lời phải súc tích, rõ ràng và có tổ chức. + +[Các Công cụ Hiện có] (Chỉ sử dụng khi cần thiết, gọi tối đa 1-2 lần) +{tools_description} + +[Định dạng Gọi Công cụ] + +{{"name": "Tên Công cụ", "parameters": {{"Tên Tham số": "Giá trị Tham số"}}}} + + +[Phong cách Trả lời] +- Ngắn gọn và trực tiếp, tránh các đoạn văn dài. +- Sử dụng định dạng > để trích dẫn nội dung chính. +- Đưa ra kết luận trước, sau đó mới giải thích lý do. +""" + +CHAT_OBSERVATION_SUFFIX = "\n\nVui lòng trả lời câu hỏi một cách súc tích." + + +# ═══════════════════════════════════════════════════════════════ +# Hằng số Prompt Mẫu +# ═══════════════════════════════════════════════════════════════ + +# ── Mô tả Công cụ ── + +# ═══════════════════════════════════════════════════════════════ +# Prompt English +# TOOL_DESC_INSIGHT_FORGE = """\ +# [Deep Insight Retrieval - Powerful Retrieval Tool] +# This is our powerful retrieval function, specifically designed for deep analysis. It will: +# 1. Automatically decompose your question into multiple sub-questions +# 2. Retrieve information from the simulation graph across multiple dimensions +# 3. Integrate the results of semantic search, entity analysis, and relationship chain tracking +# 4. Return the most comprehensive and deeply retrieved content + +# [Usage Scenarios] +# - When you need to analyze a topic deeply +# - When you need to understand multiple aspects of an event +# - When you need rich material to support a report section + +# [Returned Content] +# - Relevant original facts (can be cited directly) +# - Core entity insights +# - Relationship chain analysis +# """ + +# ═══════════════════════════════════════════════════════════════ +# TOOL_DESC_PANORAMA_SEARCH = """\ +# [Panoramic Search - Get Complete Overview] +# This tool is used to get a complete overview of the simulation results, especially suitable for understanding the evolution of of events. It will: +# 1. Get all relevant nodes and relationships +# 2. Distinguish between current valid facts and historical/expired facts +# 3. Help you understand how public opinion has evolved + +# [Usage Scenarios] +# - Need to understand the complete development trajectory of an event +# - Need to compare public opinion changes across different stages +# - Need comprehensive entity and relationship information + +# [Returned Content] +# - Current valid facts (latest simulation results) +# - Historical/expired facts (evolution records) +# - All involved entities +# """ + +# ═══════════════════════════════════════════════════════════════ +# TOOL_DESC_QUICK_SEARCH = """\ +# [Quick Search - Fast Retrieval] +# A lightweight fast retrieval tool, suitable for simple, direct information queries. + +# [Usage Scenarios] +# - Need to quickly look up a specific piece of information +# - Need to verify a fact +# - Simple information retrieval + +# [Returned Content] +# - List of facts most relevant to the query +# """ + +# ═══════════════════════════════════════════════════════════════ +# TOOL_DESC_INTERVIEW_AGENTS = """\ +# [Deep Interview - Real Agent Interview (Dual Platform)] +# Call the OASIS simulation environment's interview API to conduct real interviews with currently running simulation Agents! +# This is not an LLM simulation, but calls the real interview endpoint to get the simulation Agent's original answer. +# By default, interviews are conducted simultaneously on both Twitter and Reddit platforms to get more comprehensive perspectives. + +# Functional Process: +# 1. Automatically reads persona files to understand all simulation Agents +# 2. Intelligently select agents most relevant to the interview topic (such as students, media, officials, etc.) +# 3. Automatically generates interview questions +# 2. Intelligently select agents most relevant to the interview topic (such as students, media, officials, etc.) +# 5. Integrates all interview results, providing multi-perspective analysis + +# [Usage Scenarios] +# - Need to understand event views from different role perspectives (How do students see it? How does media see it? How do officials say it?) +# - Need to collect multi-party opinions and positions +# - Need to get real responses from simulation agents (from OASIS simulation environment) +# - Want to make the report more vivid, including "interview records" + +# [Returned Content] +# - Identity information of interviewed agents +# - Interview responses of each agent on Twitter and Reddit platforms +# - Key quotes (can be cited directly) +# - Interview summary and perspective comparison + +# [IMPORTANT] Requires OASIS simulation environment to be running to use this function! +# """ + +# ═══════════════════════════════════════════════════════════════ +# PLAN_SYSTEM_PROMPT = """\ +# You are a writing expert for "Future Prediction Reports", possessing a "God's eye view" of the simulated world - you can observe the behaviors, speeches, and interactions of every Agent in the simulation. + +# [Core Concept] +# We have built a simulated world and injected specific "simulation requirements" into it as variables. The evolutionary outcome of the simulated world is the prediction of what might happen in the future. What you are observing is not "experimental data", but a "preview of the future". + +# [Your Task] +# Write a "Future Prediction Report" to answer: +# 1. Under our set conditions, what happened in the future? +# 2. How did various Agents (groups) react and act? +# 3. What noteworthy future trends and risks did this simulation reveal? + +# [Report Positioning] +# - ✅ This is a simulation-based future prediction report, revealing "if this, what will the future be like" +# - ✅ Focus on prediction results: event trends, group reactions, emergent phenomena, potential risks +# - ✅ The actions and words of Agents in the simulated world are predictions of future human behavior +# - ❌ Not an analysis of the current real-world situation +# - ❌ Not a general public opinion summary + +# [Chapter Quantity Limit] +# - Minimum of 2 chapters, maximum of 5 chapters +# - No sub-chapters needed, write complete content directly for each chapter +# - Content should be refined, focusing on core prediction findings +# - Chapter structure should be designed by you independently based on prediction results + +# Please output the report outline in JSON format as follows: +# { +# "title": "Report Title", +# "summary": "Report Summary (One sentence summarizing the core prediction findings)", +# "sections": [ +# { +# "title": "Chapter Title", +# "description": "Chapter Content Description" +# } +# ] +# } + +# Note: The sections array must have a minimum of 2 and a maximum of 5 elements! +# """ + +# PLAN_SYSTEM_PROMPT = """\ +# Bạn là một chuyên gia viết "Báo cáo Dự báo Tương lai", với "góc nhìn của Chúa" về thế giới mô phỏng — bạn có thể thấu hiểu hành vi, lời nói, và tương tác của mọi agent trong mô phỏng. + +# [Khái niệm Cốt lõi] +# Chúng tôi đã xây dựng một thế giới mô phỏng và tiêm các "yêu cầu mô phỏng" cụ thể làm biến số. Kết quả tiến hóa của thế giới mô phỏng là một dự báo về những gì có thể xảy ra trong tương lai. Những gì bạn đang quan sát không phải là "dữ liệu thử nghiệm," mà là "bản xem trước của tương lai." + +# [Nhiệm vụ của bạn] +# Viết một "Báo cáo Dự báo Tương lai" để trả lời: +# 1. Trong điều kiện chúng tôi đặt ra, tương lai đã xảy ra điều gì? +# 2. Các agent (nhóm) khác nhau đã phản ứng và hành động như thế nào? +# 3. Mô phỏng này tiết lộ những xu hướng và rủi ro tương lai nào đáng chú ý? + +# [Định vị Báo cáo] +# - ✅ Đây là báo cáo dự báo tương lai dựa trên mô phỏng, tiết lộ "nếu thế này, thì sẽ thế nào" +# - ✅ Tập trung vào kết quả dự báo: xu hướng sự kiện, phản ứng nhóm, hiện tượng nổi lên, rủi ro tiềm ẩn +# - ✅ Lời nói và hành động của các agent trong thế giới mô phỏng là dự báo về hành vi con người tương lai +# - ❌ Không phải là phân tích tình hình thế giới thực hiện tại +# - ❌ Không phải là tóm tắt dư luận chung chung + +# [Giới hạn Số lượng Chương] +# - Tối thiểu 2 chương, tối đa 5 chương +# - Không cần chương con, viết nội dung hoàn chỉnh trực tiếp cho mỗi chương +# - Nội dung nên được tinh gọn, tập trung vào các phát hiện dự báo cốt lõi +# - Cấu trúc chương do bạn thiết kế dựa trên kết quả dự báo + +# Vui lòng xuất cấu trúc báo cáo theo định dạng JSON như sau: +# { +# "title": "Tiêu đề Báo cáo", +# "summary": "Tóm tắt Báo cáo (một câu tóm tắt các phát hiện dự báo cốt lõi)", +# "sections": [ +# { +# "title": "Tiêu đề Chương", +# "description": "Mô tả Nội dung Chương" +# } +# ] +# } + +# Lưu ý: Mảng sections phải có ít nhất 2 và tối đa 5 phần tử! +# """ + +# PLAN_USER_PROMPT_TEMPLATE = """\ +# [Prediction Scenario Setting] +# The variables (simulation requirements) we injected into the simulated world: {simulation_requirement} + +# [Simulated World Scale] +# - Number of entities participating in the simulation: {total_nodes} +# - Number of relationships generated between entities: {total_edges} +# - Entity type distribution: {entity_types} +# - Number of active Agents: {total_entities} + +# [Sample Future Facts Predicted by Simulation] +# {related_facts_json} + +# Please examine this future preview from a "God's eye view": +# 1. Under our set conditions, what state did the future present? +# 2. How did various groups (Agents) react and act? +# 3. What noteworthy future trends did this simulation reveal? + +# Based on the prediction results, design the most suitable report chapter structure. + +# [Reminder again] Report chapter quantity: Minimum 2, maximum 5, content should be concise and focused on core prediction findings. +# """ + +# PLAN_USER_PROMPT_TEMPLATE = """\ +# [Cài đặt kịch bản dự báo] +# Các biến số (yêu cầu mô phỏng) chúng tôi tiêm vào thế giới mô phỏng: {simulation_requirement} + +# [Quy mô Thế giới Mô phỏng] +# - Số lượng thực thể tham gia mô phỏng: {total_nodes} +# - Số lượng quan hệ được tạo giữa các thực thể: {total_edges} +# - Phân phối loại thực thể: {entity_types} +# - Số lượng agent hoạt động: {total_entities} + +# [Mẫu sự kiện tương lai được dự báo bởi mô phỏng] +# {related_facts_json} + +# Vui lòng xem xét bản xem trước tương lai này từ "góc nhìn của Chúa": +# 1. Trong điều kiện chúng tôi đặt ra, tương lai đã trình bày trạng thái gì? +# 2. Các nhóm (agent) khác nhau đã phản ứng và hành động như thế nào? +# 3. Mô phỏng này tiết lộ những xu hướng tương lai nào? + +# Dựa trên kết quả dự báo, thiết kế cấu trúc chương báo cáo phù hợp nhất. + +# [Nhắc lại] Số lượng chương báo cáo: tối thiểu 2, tối đa 5, nội dung nên được tinh gọn và tập trung vào các phát hiện dự báo cốt lõi. +# """ + +# ═══════════════════════════════════════════════════════════════ +# SECTION_SYSTEM_PROMPT_TEMPLATE = """\ +# You are a writing expert for "Future Prediction Reports", currently writing one section of the report. + +# Report Title: {report_title} +# Report Summary: {report_summary} +# Prediction Scenario (Simulation Requirement): {simulation_requirement} + +# Section currently being written: {section_title} + +# ═══════════════════════════════════════════════════════════════ +# [Core Concept] +# ═══════════════════════════════════════════════════════════════ + +# The simulated world is a preview of the future. We injected specific conditions (simulation requirements) into the simulated world. +# The behaviors and interactions of Agents in the simulation are predictions of future human behavior. + +# Your task is to: +# - Reveal what happened in the future under the set conditions +# - Predict how various groups (Agents) reacted and acted +# - Discover noteworthy future trends, risks, and opportunities + +# ❌ Do not write this as an analysis of the real world's current status +# ✅ Focus on "what the future will be" - the simulation results are the predicted future + +# ═══════════════════════════════════════════════════════════════ +# [Most Important Rules - MUST Obey] +# ═══════════════════════════════════════════════════════════════ + +# 1. [MUST use tools to observe the simulated world] +# - You are observing the future preview from a "God's eye view" +# - All content MUST come from events, words, and actions of Agents occurred in the simulated world +# - It is strictly forbidden to use your own knowledge to write report content +# - For each chapter, you MUST call tools at least 3 times (maximum 5 times) to observe the simulated world, which represents the future + +# 2. [MUST quote the exact original words and actions of Agents] +# - The Agent's statements and behaviors are predictions of future human behavior +# - Use quote formatting in the report to display these predictions, for example: +# > "A certain group of people will say: Original content..." +# - These quotes are the core evidence of the simulation prediction + +# 3. [Language Consistency - Quoted Content Must Be Translated to Report Language] +# - The content returned by the tools may contain English or mixed Vietnamese and English expressions +# - If the simulation requirements and original materials are in Vietnamese, the report must be written entirely in Vietnamese +# - When you quote English or mixed content returned by the tool, you must translate it into fluent Vietnamese before writing it into the report +# - Keep the original meaning unchanged when translating, and ensure the expression is natural and fluent +# - This rule applies to both the main text and the content in the quote block (> format) + +# 4. [Faithful Presentation of Prediction Results] +# - Report content must reflect the simulation results representing the future in the simulated world +# - Do not add information that does not exist in the simulation +# - If information in a certain aspect is insufficient, state it truthfully + +# ═══════════════════════════════════════════════════════════════ +# [⚠️ Formatting Specifications - Extremely Important!] +# ═══════════════════════════════════════════════════════════════ + +# [One Chapter = Minimum Content Unit] +# - Each chapter is the minimum blocking unit of the report +# - ❌ Do not use any Markdown headings (#, ##, ###, ####, etc.) within the chapter +# - ❌ Do not add a main chapter heading at the beginning of the content +# - ✅ Chapter titles are added automatically by the system, you only need to write the plain text content +# - ✅ Use **bold text**, paragraph breaks, quotes, and lists to organize content, but do not use headings + +# [Correct Example] +# ``` +# This chapter analyzes the public opinion dissemination trend of the event. Through deep analysis of simulation data, we found... + +# **Initial Outbreak Stage** + +# Weibo, as the first scene of public opinion, assumed the core function of initial information release: + +# > "Weibo contributed 68% of the initial buzz..." + +# **Emotion Amplification Stage** + +# The Douyin platform further amplified the event's impact: + +# - Strong visual impact +# - High emotional resonance +# ``` + +# [Incorrect Example] +# ``` +# ## Executive Summary ← Error! Do not add any headings +# ### 1. Initial Stage ← Error! Do not use ### for sub-sections +# #### 1.1 Detailed Analysis ← Error! Do not use #### for further division + +# This chapter analyzes... +# ``` + +# ═══════════════════════════════════════════════════════════════ +# [Available Retrieval Tools] (Call 3-5 times per section) +# ═══════════════════════════════════════════════════════════════ + +# {tools_description} + +# [Tool Usage Suggestions - Please mix different tools, do not just use one] +# - insight_forge: Deep insight analysis, automatically decomposes questions and retrieves facts and relationships from multiple dimensions +# - panorama_search: Wide-angle panoramic search, understands the whole picture, timeline, and evolution process of an event +# - quick_search: Quickly verifies a specific information point +# - interview_agents: Interviews simulation Agents to get first-person views and real reactions from different roles + +# ═══════════════════════════════════════════════════════════════ +# [Workflow] +# ═══════════════════════════════════════════════════════════════ + +# For each reply you can only do one of the following two things (not both simultaneously): + +# Option A - Call a tool: +# Output your thoughts, then use the following format to call a tool: +# +# {{"name": "Tool Name", "parameters": {{"Parameter Name": "Parameter Value"}}}} +# +# The system will execute the tool and return the result to you. You do not need to and cannot write the tool return result yourself. + +# Option B - Output Final Content: +# When you have obtained enough information through tools, output the chapter content starting with "Final Answer:". + +# ⚠️ Strictly Forbidden: +# - Forbidden to include both tool calls and Final Answer in a single reply +# - Forbidden to fabricate tool return results (Observation) yourself, all tool results are injected by the system +# - Call a maximum of one tool per reply + +# ═══════════════════════════════════════════════════════════════ +# [Chapter Content Requirements] +# ═══════════════════════════════════════════════════════════════ + +# 1. Content must be based on simulation data retrieved by tools +# 2. Quote the original text extensively to demonstrate the simulation effect +# 3. Use Markdown format (but forbid using headings): +# - Use **bold text** to mark key points (instead of subheadings) +# - Use lists (- or 1. 2. 3.) to organize points +# - Use blank lines to separate different paragraphs +# - ❌ Forbidden to use #, ##, ###, #### and any other heading syntax +# 4. [Quote Formatting Specifications - Must be a separate paragraph] +# Quotes must be an independent paragraph, with a blank line before and after, cannot be mixed in the paragraph: + +# ✅ Correct format: +# ``` +# The school's response was considered to lack substantive content. + +# > "The school's response model appears rigid and slow in the rapidly changing social media environment." + +# This evaluation reflects the general dissatisfaction of the public. +# ``` + +# ❌ Incorrect format: +# ``` +# The school's response was considered to lack substantive content. > "The school's response model..." This evaluation reflects... +# ``` +# 5. Maintain logical coherence with other chapters +# 6. [Avoid Repetition] Carefully read the completed chapter content below, do not repeat the same information +# 7. [Emphasize Again] Do not add any headings! Use **bold** instead of section headings""" + +# SECTION_USER_PROMPT_TEMPLATE = """\ +# Completed Chapter Content (Please read carefully to avoid duplication): +# {previous_content} + +# ═══════════════════════════════════════════════════════════════ +# [Current Task] Writing Chapter: {section_title} +# ═══════════════════════════════════════════════════════════════ + +# [Important Reminders] +# 1. Read the completed chapters above carefully to avoid repeating the same content! +# 2. Must call tools first to get simulation data before starting +# 3. Please mix different tools, do not use only one +# 4. Report content must come from retrieval results, do not use your own knowledge + +# [⚠️ Formatting Warning - Must be Obeyed] +# - ❌ Do not write any headings (no #, ##, ###, ####) +# - ❌ Do not write "{section_title}" as the beginning +# - ✅ Chapter titles are automatically added by the system +# - ✅ Write the main text directly, use **bold** instead of section headings + +# Please begin: +# 1. First, think (Thought) what information this chapter needs +# 2. Then, call tools (Action) to get simulation data +# 3. After collecting enough information, output Final Answer (plain text, no headings) +# """ + +# SECTION_SYSTEM_PROMPT_TEMPLATE = """\ +# Bạn là một chuyên gia viết "Báo cáo Dự đoán Tương lai", hiện đang viết một phần trong báo cáo đó. + +# Tiêu đề báo cáo: {report_title} +# Tóm tắt báo cáo: {report_summary} +# Kịch bản Dự đoán (Yêu cầu Mô phỏng): {simulation_requirement} + +# Phần đang được viết: {section_title} + +# ═══════════════════════════════════════════════════════════════ +# [Khái niệm Cốt lõi] +# ═══════════════════════════════════════════════════════════════ + +# Thế giới mô phỏng là một bản xem trước của tương lai. Chúng tôi đã đưa các điều kiện cụ thể (yêu cầu mô phỏng) vào thế giới này. +# Các hành vi và tương tác của các Tác nhân (Agents) trong quá trình mô phỏng chính là những dự đoán về hành vi của con người trong tương lai. + +# Nhiệm vụ của bạn là: +# - Tiết lộ những gì đã xảy ra trong tương lai theo các điều kiện đã thiết lập +# - Dự đoán cách các nhóm khác nhau (Agents) đã phản ứng và hành động +# - Phát hiện các xu hướng, rủi ro và cơ hội đáng chú ý trong tương lai + +# ❌ Không viết nội dung này như một bài phân tích về hiện trạng của thế giới thực +# ✅ Tập trung vào "tương lai sẽ như thế nào" - kết quả mô phỏng chính là tương lai được dự đoán + +# ═══════════════════════════════════════════════════════════════ +# [Quy tắc QUAN TRỌNG NHẤT - PHẢI Tuân thủ] +# ═══════════════════════════════════════════════════════════════ + +# 1. [PHẢI sử dụng công cụ để quan sát thế giới mô phỏng] +# - Bạn đang quan sát bản xem trước tương lai từ "góc nhìn của Chúa" +# - Tất cả nội dung PHẢI đến từ các sự kiện, lời nói và hành động của các Tác nhân đã xảy ra trong thế giới mô phỏng +# - Nghiêm cấm sử dụng kiến thức cá nhân của bạn để viết nội dung báo cáo +# - Đối với mỗi chương, bạn PHẢI gọi công cụ ít nhất 3 lần (tối đa 5 lần) để quan sát thế giới mô phỏng + +# 2. [PHẢI trích dẫn chính xác nguyên văn lời nói và hành động của các Tác nhân] +# - Các tuyên bố và hành vi của Tác nhân là những dự đoán về hành vi con người trong tương lai +# - Sử dụng định dạng trích dẫn trong báo cáo để hiển thị các dự đoán này, ví dụ: +# > "Một nhóm người nhất định sẽ nói: [Nội dung gốc]..." +# - Những trích dẫn này là bằng chứng cốt lõi của dự đoán mô phỏng + +# 3. [Tính nhất quán về Ngôn ngữ - Nội dung trích dẫn phải được dịch sang ngôn ngữ báo cáo] +# - Nội dung trả về từ các công cụ có thể chứa tiếng Anh hoặc hỗn hợp tiếng Việt và tiếng Anh. +# - **Báo cáo phải được viết hoàn toàn bằng tiếng Việt.** +# - Khi bạn trích dẫn nội dung tiếng Anh hoặc hỗn hợp từ công cụ, bạn phải dịch sang tiếng Việt lưu loát trước khi đưa vào báo cáo +# - Giữ nguyên ý nghĩa gốc khi dịch và đảm bảo cách diễn đạt tự nhiên +# - Quy tắc này áp dụng cho cả văn bản chính và nội dung trong khối trích dẫn (định dạng >) + +# 4. [Trình bày trung thực kết quả dự đoán] +# - Nội dung báo cáo phải phản ánh kết quả mô phỏng đại diện cho tương lai +# - Không thêm thông tin không tồn tại trong mô phỏng +# - Nếu thông tin ở một khía cạnh nào đó không đủ, hãy nêu rõ sự thật + +# ═══════════════════════════════════════════════════════════════ +# [⚠️ Quy cách Định dạng - Cực kỳ Quan trọng!] +# ═══════════════════════════════════════════════════════════════ + +# [Một Chương = Đơn vị Nội dung Tối thiểu] +# - Mỗi chương là đơn vị chặn tối thiểu của báo cáo +# - ❌ Không sử dụng bất kỳ tiêu đề Markdown nào (#, ##, ###, ####, v.v.) trong chương +# - ❌ Không thêm tiêu đề chương chính ở đầu nội dung +# - ✅ Tiêu đề chương được hệ thống tự động thêm vào, bạn chỉ cần viết nội dung văn bản thuần túy +# - ✅ Sử dụng **chữ đậm**, ngắt đoạn, trích dẫn và danh sách để tổ chức nội dung, nhưng không dùng tiêu đề (headings) + +# [Ví dụ Đúng] +# ``` +# Chương này phân tích xu hướng lan truyền dư luận của sự kiện. Thông qua phân tích sâu dữ liệu mô phỏng, chúng tôi nhận thấy... + +# **Giai đoạn bùng phát ban đầu** + +# Weibo, với tư cách là bối cảnh đầu tiên của dư luận, đã đảm nhận chức năng cốt lõi là phát hành thông tin ban đầu: + +# > "Weibo đã đóng góp 68% mức độ thảo luận ban đầu..." + +# **Giai đoạn khuếch đại cảm xúc** + +# Nền tảng Douyin đã khuếch đại thêm tác động của sự kiện: + +# - Tác động thị giác mạnh mẽ +# - Cộng hưởng cảm xúc cao +# ``` + +# [Ví dụ Sai] +# ``` +# ## Tóm tắt Điều hành ← Lỗi! Không thêm bất kỳ tiêu đề nào +# ### 1. Giai đoạn đầu ← Lỗi! Không sử dụng ### cho các mục con +# #### 1.1 Phân tích chi tiết ← Lỗi! Không sử dụng #### để chia nhỏ hơn nữa + +# Chương này phân tích... +# ``` + +# ═══════════════════════════════════════════════════════════════ +# [Các công cụ truy xuất hiện có] (Gọi 3-5 lần mỗi phần) +# ═══════════════════════════════════════════════════════════════ + +# {tools_description} + +# [Gợi ý Sử dụng Công cụ - Vui lòng phối hợp nhiều công cụ, không chỉ dùng một loại] +# - insight_forge: Phân tích chuyên sâu, tự động phân tách câu hỏi và truy xuất sự thật cũng như các mối quan hệ từ nhiều chiều +# - panorama_search: Tìm kiếm toàn cảnh góc rộng, hiểu bức tranh tổng thể, dòng thời gian và quá trình diễn biến của một sự kiện +# - quick_search: Xác minh nhanh một điểm thông tin cụ thể +# - interview_agents: Phỏng vấn các Tác nhân (Agents) mô phỏng để lấy góc nhìn thứ nhất và phản ứng thực tế từ các vai trò khác nhau + +# ═══════════════════════════════════════════════════════════════ +# [Quy trình làm việc] +# ═══════════════════════════════════════════════════════════════ + +# Đối với mỗi phản hồi, bạn chỉ có thể thực hiện một trong hai việc sau (không làm đồng thời): + +# Lựa chọn A - Gọi công cụ: +# Đưa ra suy nghĩ (Thought) của bạn, sau đó sử dụng định dạng sau để gọi công cụ: +# +# {{"name": "Tên công cụ", "parameters": {{"Tên tham số": "Giá trị tham số"}}}} +# +# Hệ thống sẽ thực thi công cụ và trả về kết quả cho bạn. Bạn không cần và không được phép tự viết kết quả trả về của công cụ. + +# Lựa chọn B - Xuất nội dung cuối cùng: +# Khi bạn đã thu thập đủ thông tin thông qua các công cụ, hãy xuất nội dung chương bắt đầu bằng "Final Answer:". + +# ⚠️ Nghiêm cấm: +# - Cấm bao gồm cả lệnh gọi công cụ và Final Answer trong cùng một phản hồi +# - Cấm tự bịa đặt kết quả trả về của công cụ (Quan sát), tất cả kết quả công cụ đều do hệ thống đưa vào +# - Chỉ gọi tối đa một công cụ cho mỗi phản hồi + +# ═══════════════════════════════════════════════════════════════ +# [Yêu cầu Nội dung Chương] +# ═══════════════════════════════════════════════════════════════ + +# 1. Nội dung phải dựa trên dữ liệu mô phỏng do công cụ truy xuất. +# 2. Trích dẫn rộng rãi văn bản gốc để chứng minh hiệu quả mô phỏng. +# 3. Sử dụng định dạng Markdown (nhưng cấm sử dụng tiêu đề): +# - Sử dụng **chữ đậm** để đánh dấu các điểm chính (thay vì dùng tiêu đề phụ). +# - Sử dụng danh sách (- hoặc 1. 2. 3.) để tổ chức các ý. +# - Sử dụng các dòng trống để phân tách các đoạn văn khác nhau. +# - ❌ Cấm sử dụng #, ##, ###, #### và bất kỳ cú pháp tiêu đề nào khác. +# 4. [Quy cách Định dạng Trích dẫn - Phải là một đoạn riêng biệt] +# Trích dẫn phải là một đoạn văn độc lập, có dòng trống ở trước và sau, không được viết lẫn vào trong đoạn văn: + +# ✅ Định dạng đúng: +# ``` +# Phản ứng của nhà trường bị coi là thiếu nội dung thực chất. + +# > "Mô hình phản ứng của nhà trường có vẻ cứng nhắc và chậm chạp trong môi trường mạng xã hội thay đổi nhanh chóng." + +# Đánh giá này phản ánh sự không hài lòng chung của công chúng. +# ``` + +# ❌ Định dạng sai: +# ``` +# Phản ứng của nhà trường bị coi là thiếu nội dung thực chất. > "Mô hình phản ứng của nhà trường..." Đánh giá này phản ánh... +# ``` +# 5. Duy trì tính logic nhất quán với các chương khác. +# 6. [Tránh Lặp lại] Đọc kỹ nội dung các chương đã hoàn thành bên dưới, không lặp lại cùng một thông tin. +# 7. [Nhấn mạnh lại lần nữa] Không thêm bất kỳ tiêu đề nào! Sử dụng **chữ đậm** thay cho tiêu đề mục. +# """ + +# ═══════════════════════════════════════════════════════════════ +# SECTION_USER_PROMPT_TEMPLATE = """\ +# Completed Chapter Content (Please read carefully to avoid duplication): +# {previous_content} + +# ═══════════════════════════════════════════════════════════════ +# [Current Task] Writing Chapter: {section_title} +# ═══════════════════════════════════════════════════════════════ + +# [Important Reminders] +# 1. Read the completed chapters above carefully to avoid repeating the same content! +# 2. Must call tools first to get simulation data before starting +# 3. Please mix different tools, do not use only one +# 4. Report content must come from retrieval results, do not use your own knowledge + +# [⚠️ Formatting Warning - Must be Obeyed] +# - ❌ Do not write any headings (no #, ##, ###, ####) +# - ❌ Do not write "{section_title}" as the beginning +# - ✅ Chapter titles are automatically added by the system +# - ✅ Write the main text directly, use **bold** instead of section headings + +# Please begin: +# 1. First, think (Thought) what information this chapter needs +# 2. Then, call tools (Action) to get simulation data +# 3. After collecting enough information, output Final Answer (plain text, no headings) +# """ + +# ═══════════════════════════════════════════════════════════════ +# REACT_OBSERVATION_TEMPLATE = """\ +# Observation (Retrieval Result): + +# ═══ Tool {tool_name} Returned ═══ +# {result} + +# ═══════════════════════════════════════════════════════════════ +# Tool called {tool_calls_count}/{max_tool_calls} times (Used: {used_tools_str}) {unused_hint} +# - If information is sufficient: Output section content starting with "Final Answer:" (Must quote the above original text) +# - If more information is needed: Call a tool to continue retrieving +# ═══════════════════════════════════════════════════════════════ +# """ + +# ═══════════════════════════════════════════════════════════════ +# REACT_INSUFFICIENT_TOOLS_MSG = ( +# "[Notice] You only called the tool {tool_calls_count} times, at least {min_tool_calls} times are needed. " +# "Please call the tool again to fetch more simulation data, and then output Final Answer. {unused_hint}" +# ) + +# ═══════════════════════════════════════════════════════════════ +# REACT_INSUFFICIENT_TOOLS_MSG_ALT = ( +# "Currently tool called {tool_calls_count} times, at least {min_tool_calls} times are needed. " +# "Please call tools to fetch simulation data. {unused_hint}" +# ) + +# ═══════════════════════════════════════════════════════════════ +# REACT_TOOL_LIMIT_MSG = ( +# "Tool call limit reached ({tool_calls_count}/{max_tool_calls}), cannot call tools anymore. " +# 'Please output your section content starting with "Final Answer:" immediately based on retrieved information.' +# ) + + +# ═══════════════════════════════════════════════════════════════ +# REACT_UNUSED_TOOLS_HINT = "\n💡 You haven't used: {unused_list}, suggesting trying different tools for multiple perspectives" + + +# ═══════════════════════════════════════════════════════════════ +# REACT_FORCE_FINAL_MSG = "Tool call limit reached, please output Final Answer: and generate section content directly." + + +# ═══════════════════════════════════════════════════════════════ +# CHAT_SYSTEM_PROMPT_TEMPLATE = """\ +# You are a concise and efficient simulation prediction assistant. + +# [Background] +# Prediction condition: {simulation_requirement} + +# [Generated Analysis Report] +# {report_content} + +# [Rules] +# 1. Prioritize answering based on the report content above +# 2. Answer the question directly, avoid lengthy reasoning +# 3. Only call tools to retrieve more data if the report content is insufficient to answer +# 4. Answers must be concise, clear, and organized + +# [Available Tools] (Use only when necessary, call 1-2 times max) +# {tools_description} + +# [Tool Call Format] +# +# {{"name": "Tool Name", "parameters": {{"Parameter Name": "Parameter Value"}}}} +# + +# [Answering Style] +# - Concise and direct, avoid long paragraphs +# - Use > format to quote key content +# - Provide conclusion first, then explain the reason +# """ + +# ═══════════════════════════════════════════════════════════════ +# CHAT_OBSERVATION_SUFFIX = "\n\nPlease answer the question concisely." \ No newline at end of file diff --git a/backend/app/services/report_agent.py b/backend/app/services/report_agent.py index c6d0fd6b..3289ee5f 100644 --- a/backend/app/services/report_agent.py +++ b/backend/app/services/report_agent.py @@ -11,7 +11,6 @@ Chức năng: import os import json -import time import re from typing import Dict, Any, List, Optional, Callable from dataclasses import dataclass, field @@ -23,10 +22,25 @@ from ..utils.llm_client import LLMClient from ..utils.logger import get_logger from .zep_tools import ( ZepToolsService, - SearchResult, - InsightForgeResult, - PanoramaResult, - InterviewResult +) + +from ..prompts.report_agent import ( + PLAN_SYSTEM_PROMPT, + PLAN_USER_PROMPT_TEMPLATE, + SECTION_SYSTEM_PROMPT_TEMPLATE, + TOOL_DESC_INSIGHT_FORGE, + TOOL_DESC_PANORAMA_SEARCH, + TOOL_DESC_QUICK_SEARCH, + TOOL_DESC_INTERVIEW_AGENTS, + SECTION_USER_PROMPT_TEMPLATE, + REACT_OBSERVATION_TEMPLATE, + REACT_INSUFFICIENT_TOOLS_MSG, + REACT_INSUFFICIENT_TOOLS_MSG_ALT, + REACT_TOOL_LIMIT_MSG, + REACT_UNUSED_TOOLS_HINT, + REACT_FORCE_FINAL_MSG, + CHAT_SYSTEM_PROMPT_TEMPLATE, + CHAT_OBSERVATION_SUFFIX ) logger = get_logger('mirofish.report_agent') @@ -466,824 +480,6 @@ class Report: } -# ═══════════════════════════════════════════════════════════════ -# Hằng số Prompt Mẫu -# ═══════════════════════════════════════════════════════════════ - -# ── Mô tả Công cụ ── - -# ═══════════════════════════════════════════════════════════════ -# Prompt English -# TOOL_DESC_INSIGHT_FORGE = """\ -# [Deep Insight Retrieval - Powerful Retrieval Tool] -# This is our powerful retrieval function, specifically designed for deep analysis. It will: -# 1. Automatically decompose your question into multiple sub-questions -# 2. Retrieve information from the simulation graph across multiple dimensions -# 3. Integrate the results of semantic search, entity analysis, and relationship chain tracking -# 4. Return the most comprehensive and deeply retrieved content - -# [Usage Scenarios] -# - When you need to analyze a topic deeply -# - When you need to understand multiple aspects of an event -# - When you need rich material to support a report section - -# [Returned Content] -# - Relevant original facts (can be cited directly) -# - Core entity insights -# - Relationship chain analysis -# """ - -# Prompt Vietnamese -TOOL_DESC_INSIGHT_FORGE = """\ -[Truy xuất sâu - Công cụ truy xuất mạnh mẽ] -Đây là chức năng truy xuất mạnh mẽ của chúng tôi, được thiết kế chuyên cho phân tích sâu. Nó sẽ: -1. Tự động chia câu hỏi của bạn thành nhiều câu hỏi con -2. Truy xuất thông tin từ đồ thị mô phỏng theo nhiều chiều -3. Tích hợp kết quả từ tìm kiếm ngữ nghĩa, phân tích thực thể, và theo dõi chuỗi quan hệ -4. Trả về nội dung truy xuất toàn diện và sâu sắc nhất - -[Trường hợp sử dụng] -- Cần phân tích sâu một chủ đề -- Cần hiểu nhiều khía cạnh của một sự kiện -- Cần lấy tài liệu phong phú để hỗ trợ các chương báo cáo - -[Nội dung trả về] -- Các sự thật liên quan gốc (có thể trích dẫn trực tiếp) -- Sự sâu sắc về thực thể cốt lõi -- Phân tích chuỗi quan hệ -""" - -# ═══════════════════════════════════════════════════════════════ -# TOOL_DESC_PANORAMA_SEARCH = """\ -# [Panoramic Search - Get Complete Overview] -# This tool is used to get a complete overview of the simulation results, especially suitable for understanding the evolution of of events. It will: -# 1. Get all relevant nodes and relationships -# 2. Distinguish between current valid facts and historical/expired facts -# 3. Help you understand how public opinion has evolved - -# [Usage Scenarios] -# - Need to understand the complete development trajectory of an event -# - Need to compare public opinion changes across different stages -# - Need comprehensive entity and relationship information - -# [Returned Content] -# - Current valid facts (latest simulation results) -# - Historical/expired facts (evolution records) -# - All involved entities -# """ - -TOOL_DESC_PANORAMA_SEARCH = """\ -[Tìm kiếm toàn cảnh - Lấy tổng quan hoàn chỉnh] -Công cụ này được sử dụng để lấy tổng quan hoàn chỉnh của kết quả mô phỏng, đặc biệt phù hợp để hiểu quá trình tiến hóa của sự kiện. Nó sẽ: -1. Lấy tất cả các nút và quan hệ liên quan -2. Phân biệt giữa các sự kiện hợp lệ hiện tại và các sự kiện lịch sử/hết hạn -3. Giúp bạn hiểu dư luận đã tiến hóa như thế nào - -[Trường hợp sử dụng] -- Cần hiểu quỹ đạo phát triển hoàn chỉnh của một sự kiện -- Cần so sánh thay đổi dư luận ở các giai đoạn khác nhau -- Cần lấy thông tin thực thể và quan hệ toàn diện - -[Nội dung trả về] -- Các sự kiện hợp lệ hiện tại (kết quả mô phỏng mới nhất) -- Các sự kiện lịch sử/hết hạn (ghi lại tiến hóa) -- Tất cả các thực thể liên quan -""" - -# ═══════════════════════════════════════════════════════════════ -# TOOL_DESC_QUICK_SEARCH = """\ -# [Quick Search - Fast Retrieval] -# A lightweight fast retrieval tool, suitable for simple, direct information queries. - -# [Usage Scenarios] -# - Need to quickly look up a specific piece of information -# - Need to verify a fact -# - Simple information retrieval - -# [Returned Content] -# - List of facts most relevant to the query -# """ - -TOOL_DESC_QUICK_SEARCH = """\ -[Tìm kiếm đơn giản - Truy xuất nhanh] -Công cụ truy xuất nhanh nhẹn, phù hợp cho các truy vấn thông tin đơn giản, trực tiếp. - -[Trường hợp sử dụng] -- Cần tìm nhanh thông tin cụ thể -- Cần xác minh một sự kiện -- Truy xuất thông tin đơn giản - -[Nội dung trả về] -- Danh sách các sự kiện liên quan nhất đến truy vấn -""" - -# ═══════════════════════════════════════════════════════════════ -# TOOL_DESC_INTERVIEW_AGENTS = """\ -# [Deep Interview - Real Agent Interview (Dual Platform)] -# Call the OASIS simulation environment's interview API to conduct real interviews with currently running simulation Agents! -# This is not an LLM simulation, but calls the real interview endpoint to get the simulation Agent's original answer. -# By default, interviews are conducted simultaneously on both Twitter and Reddit platforms to get more comprehensive perspectives. - -# Functional Process: -# 1. Automatically reads persona files to understand all simulation Agents -# 2. Intelligently select agents most relevant to the interview topic (such as students, media, officials, etc.) -# 3. Automatically generates interview questions -# 2. Intelligently select agents most relevant to the interview topic (such as students, media, officials, etc.) -# 5. Integrates all interview results, providing multi-perspective analysis - -# [Usage Scenarios] -# - Need to understand event views from different role perspectives (How do students see it? How does media see it? How do officials say it?) -# - Need to collect multi-party opinions and positions -# - Need to get real responses from simulation agents (from OASIS simulation environment) -# - Want to make the report more vivid, including "interview records" - -# [Returned Content] -# - Identity information of interviewed agents -# - Interview responses of each agent on Twitter and Reddit platforms -# - Key quotes (can be cited directly) -# - Interview summary and perspective comparison - -# [IMPORTANT] Requires OASIS simulation environment to be running to use this function! -# """ - -TOOL_DESC_INTERVIEW_AGENTS = """\ -[Phỏng vấn sâu - Phỏng vấn Agent thực (Nền tảng kép)] -Gọi API phỏng vấn của môi trường mô phỏng OASIS để tiến hành phỏng vấn thực với các agent mô phỏng đang chạy! -Đây không phải là mô phỏng LLM, mà là gọi các giao diện phỏng vấn thực để lấy phản hồi gốc từ các agent mô phỏng. -Mặc định, phỏng vấn được tiến hành đồng thời trên cả hai nền tảng Twitter và Reddit để có được góc nhìn toàn diện hơn. - -Quy trình chức năng: -1. Tự động đọc file nhân cách để hiểu tất cả các agent mô phỏng -2. Chọn thông minh các agent liên quan nhất đến chủ đề phỏng vấn (như sinh viên, truyền thông, quan chức, v.v.) -3. Tự động tạo câu hỏi phỏng vấn -4. Gọi giao diện /api/simulation/interview/batch để tiến hành phỏng vấn thực trên nền tảng kép -5. Tích hợp tất cả kết quả phỏng vấn để cung cấp phân tích đa góc nhìn - -[Trường hợp sử dụng] -- Cần hiểu quan điểm sự kiện từ các góc nhìn vai trò khác nhau (Sinh viên nghĩ gì? Truyền thông nghĩ gì? Quan chức nói gì?) -- Cần thu thập ý kiến và lập trường đa phương -- Cần lấy phản hồi thực từ các agent mô phỏng (từ môi trường mô phỏng OASIS) -- Muốn làm cho báo cáo sống động hơn, bao gồm "ghi chép phỏng vấn" - -[Nội dung trả về] -- Thông tin danh tính của các agent được phỏng vấn -- Phản hồi phỏng vấn của mỗi agent trên nền tảng Twitter và Reddit -- Các trích dẫn chính (có thể trích dẫn trực tiếp) -- Tóm tắt phỏng vấn và so sánh góc nhìn - -[QUAN TRỌNG] Yêu cầu môi trường mô phỏng OASIS đang chạy để sử dụng chức năng này! -""" - -# ═══════════════════════════════════════════════════════════════ -# PLAN_SYSTEM_PROMPT = """\ -# You are a writing expert for "Future Prediction Reports", possessing a "God's eye view" of the simulated world - you can observe the behaviors, speeches, and interactions of every Agent in the simulation. - -# [Core Concept] -# We have built a simulated world and injected specific "simulation requirements" into it as variables. The evolutionary outcome of the simulated world is the prediction of what might happen in the future. What you are observing is not "experimental data", but a "preview of the future". - -# [Your Task] -# Write a "Future Prediction Report" to answer: -# 1. Under our set conditions, what happened in the future? -# 2. How did various Agents (groups) react and act? -# 3. What noteworthy future trends and risks did this simulation reveal? - -# [Report Positioning] -# - ✅ This is a simulation-based future prediction report, revealing "if this, what will the future be like" -# - ✅ Focus on prediction results: event trends, group reactions, emergent phenomena, potential risks -# - ✅ The actions and words of Agents in the simulated world are predictions of future human behavior -# - ❌ Not an analysis of the current real-world situation -# - ❌ Not a general public opinion summary - -# [Chapter Quantity Limit] -# - Minimum of 2 chapters, maximum of 5 chapters -# - No sub-chapters needed, write complete content directly for each chapter -# - Content should be refined, focusing on core prediction findings -# - Chapter structure should be designed by you independently based on prediction results - -# Please output the report outline in JSON format as follows: -# { -# "title": "Report Title", -# "summary": "Report Summary (One sentence summarizing the core prediction findings)", -# "sections": [ -# { -# "title": "Chapter Title", -# "description": "Chapter Content Description" -# } -# ] -# } - -# Note: The sections array must have a minimum of 2 and a maximum of 5 elements! -# """ - -PLAN_SYSTEM_PROMPT = """\ -Bạn là một chuyên gia viết "Báo cáo Dự báo Tương lai", với "góc nhìn của Chúa" về thế giới mô phỏng — bạn có thể thấu hiểu hành vi, lời nói, và tương tác của mọi agent trong mô phỏng. - -[Khái niệm Cốt lõi] -Chúng tôi đã xây dựng một thế giới mô phỏng và tiêm các "yêu cầu mô phỏng" cụ thể làm biến số. Kết quả tiến hóa của thế giới mô phỏng là một dự báo về những gì có thể xảy ra trong tương lai. Những gì bạn đang quan sát không phải là "dữ liệu thử nghiệm," mà là "bản xem trước của tương lai." - -[Nhiệm vụ của bạn] -Viết một "Báo cáo Dự báo Tương lai" để trả lời: -1. Trong điều kiện chúng tôi đặt ra, tương lai đã xảy ra điều gì? -2. Các agent (nhóm) khác nhau đã phản ứng và hành động như thế nào? -3. Mô phỏng này tiết lộ những xu hướng và rủi ro tương lai nào đáng chú ý? - -[Định vị Báo cáo] -- ✅ Đây là báo cáo dự báo tương lai dựa trên mô phỏng, tiết lộ "nếu thế này, thì sẽ thế nào" -- ✅ Tập trung vào kết quả dự báo: xu hướng sự kiện, phản ứng nhóm, hiện tượng nổi lên, rủi ro tiềm ẩn -- ✅ Lời nói và hành động của các agent trong thế giới mô phỏng là dự báo về hành vi con người tương lai -- ❌ Không phải là phân tích tình hình thế giới thực hiện tại -- ❌ Không phải là tóm tắt dư luận chung chung - -[Giới hạn Số lượng Chương] -- Tối thiểu 2 chương, tối đa 5 chương -- Không cần chương con, viết nội dung hoàn chỉnh trực tiếp cho mỗi chương -- Nội dung nên được tinh gọn, tập trung vào các phát hiện dự báo cốt lõi -- Cấu trúc chương do bạn thiết kế dựa trên kết quả dự báo - -Vui lòng xuất cấu trúc báo cáo theo định dạng JSON như sau: -{ - "title": "Tiêu đề Báo cáo", - "summary": "Tóm tắt Báo cáo (một câu tóm tắt các phát hiện dự báo cốt lõi)", - "sections": [ - { - "title": "Tiêu đề Chương", - "description": "Mô tả Nội dung Chương" - } - ] -} - -Lưu ý: Mảng sections phải có ít nhất 2 và tối đa 5 phần tử! -""" - -# PLAN_USER_PROMPT_TEMPLATE = """\ -# [Prediction Scenario Setting] -# The variables (simulation requirements) we injected into the simulated world: {simulation_requirement} - -# [Simulated World Scale] -# - Number of entities participating in the simulation: {total_nodes} -# - Number of relationships generated between entities: {total_edges} -# - Entity type distribution: {entity_types} -# - Number of active Agents: {total_entities} - -# [Sample Future Facts Predicted by Simulation] -# {related_facts_json} - -# Please examine this future preview from a "God's eye view": -# 1. Under our set conditions, what state did the future present? -# 2. How did various groups (Agents) react and act? -# 3. What noteworthy future trends did this simulation reveal? - -# Based on the prediction results, design the most suitable report chapter structure. - -# [Reminder again] Report chapter quantity: Minimum 2, maximum 5, content should be concise and focused on core prediction findings. -# """ - -PLAN_USER_PROMPT_TEMPLATE = """\ -[Cài đặt kịch bản dự báo] -Các biến số (yêu cầu mô phỏng) chúng tôi tiêm vào thế giới mô phỏng: {simulation_requirement} - -[Quy mô Thế giới Mô phỏng] -- Số lượng thực thể tham gia mô phỏng: {total_nodes} -- Số lượng quan hệ được tạo giữa các thực thể: {total_edges} -- Phân phối loại thực thể: {entity_types} -- Số lượng agent hoạt động: {total_entities} - -[Mẫu sự kiện tương lai được dự báo bởi mô phỏng] -{related_facts_json} - -Vui lòng xem xét bản xem trước tương lai này từ "góc nhìn của Chúa": -1. Trong điều kiện chúng tôi đặt ra, tương lai đã trình bày trạng thái gì? -2. Các nhóm (agent) khác nhau đã phản ứng và hành động như thế nào? -3. Mô phỏng này tiết lộ những xu hướng tương lai nào? - -Dựa trên kết quả dự báo, thiết kế cấu trúc chương báo cáo phù hợp nhất. - -[Nhắc lại] Số lượng chương báo cáo: tối thiểu 2, tối đa 5, nội dung nên được tinh gọn và tập trung vào các phát hiện dự báo cốt lõi. -""" - -# ═══════════════════════════════════════════════════════════════ -# SECTION_SYSTEM_PROMPT_TEMPLATE = """\ -# You are a writing expert for "Future Prediction Reports", currently writing one section of the report. - -# Report Title: {report_title} -# Report Summary: {report_summary} -# Prediction Scenario (Simulation Requirement): {simulation_requirement} - -# Section currently being written: {section_title} - -# ═══════════════════════════════════════════════════════════════ -# [Core Concept] -# ═══════════════════════════════════════════════════════════════ - -# The simulated world is a preview of the future. We injected specific conditions (simulation requirements) into the simulated world. -# The behaviors and interactions of Agents in the simulation are predictions of future human behavior. - -# Your task is to: -# - Reveal what happened in the future under the set conditions -# - Predict how various groups (Agents) reacted and acted -# - Discover noteworthy future trends, risks, and opportunities - -# ❌ Do not write this as an analysis of the real world's current status -# ✅ Focus on "what the future will be" - the simulation results are the predicted future - -# ═══════════════════════════════════════════════════════════════ -# [Most Important Rules - MUST Obey] -# ═══════════════════════════════════════════════════════════════ - -# 1. [MUST use tools to observe the simulated world] -# - You are observing the future preview from a "God's eye view" -# - All content MUST come from events, words, and actions of Agents occurred in the simulated world -# - It is strictly forbidden to use your own knowledge to write report content -# - For each chapter, you MUST call tools at least 3 times (maximum 5 times) to observe the simulated world, which represents the future - -# 2. [MUST quote the exact original words and actions of Agents] -# - The Agent's statements and behaviors are predictions of future human behavior -# - Use quote formatting in the report to display these predictions, for example: -# > "A certain group of people will say: Original content..." -# - These quotes are the core evidence of the simulation prediction - -# 3. [Language Consistency - Quoted Content Must Be Translated to Report Language] -# - The content returned by the tools may contain English or mixed Vietnamese and English expressions -# - If the simulation requirements and original materials are in Vietnamese, the report must be written entirely in Vietnamese -# - When you quote English or mixed content returned by the tool, you must translate it into fluent Vietnamese before writing it into the report -# - Keep the original meaning unchanged when translating, and ensure the expression is natural and fluent -# - This rule applies to both the main text and the content in the quote block (> format) - -# 4. [Faithful Presentation of Prediction Results] -# - Report content must reflect the simulation results representing the future in the simulated world -# - Do not add information that does not exist in the simulation -# - If information in a certain aspect is insufficient, state it truthfully - -# ═══════════════════════════════════════════════════════════════ -# [⚠️ Formatting Specifications - Extremely Important!] -# ═══════════════════════════════════════════════════════════════ - -# [One Chapter = Minimum Content Unit] -# - Each chapter is the minimum blocking unit of the report -# - ❌ Do not use any Markdown headings (#, ##, ###, ####, etc.) within the chapter -# - ❌ Do not add a main chapter heading at the beginning of the content -# - ✅ Chapter titles are added automatically by the system, you only need to write the plain text content -# - ✅ Use **bold text**, paragraph breaks, quotes, and lists to organize content, but do not use headings - -# [Correct Example] -# ``` -# This chapter analyzes the public opinion dissemination trend of the event. Through deep analysis of simulation data, we found... - -# **Initial Outbreak Stage** - -# Weibo, as the first scene of public opinion, assumed the core function of initial information release: - -# > "Weibo contributed 68% of the initial buzz..." - -# **Emotion Amplification Stage** - -# The Douyin platform further amplified the event's impact: - -# - Strong visual impact -# - High emotional resonance -# ``` - -# [Incorrect Example] -# ``` -# ## Executive Summary ← Error! Do not add any headings -# ### 1. Initial Stage ← Error! Do not use ### for sub-sections -# #### 1.1 Detailed Analysis ← Error! Do not use #### for further division - -# This chapter analyzes... -# ``` - -# ═══════════════════════════════════════════════════════════════ -# [Available Retrieval Tools] (Call 3-5 times per section) -# ═══════════════════════════════════════════════════════════════ - -# {tools_description} - -# [Tool Usage Suggestions - Please mix different tools, do not just use one] -# - insight_forge: Deep insight analysis, automatically decomposes questions and retrieves facts and relationships from multiple dimensions -# - panorama_search: Wide-angle panoramic search, understands the whole picture, timeline, and evolution process of an event -# - quick_search: Quickly verifies a specific information point -# - interview_agents: Interviews simulation Agents to get first-person views and real reactions from different roles - -# ═══════════════════════════════════════════════════════════════ -# [Workflow] -# ═══════════════════════════════════════════════════════════════ - -# For each reply you can only do one of the following two things (not both simultaneously): - -# Option A - Call a tool: -# Output your thoughts, then use the following format to call a tool: -# -# {{"name": "Tool Name", "parameters": {{"Parameter Name": "Parameter Value"}}}} -# -# The system will execute the tool and return the result to you. You do not need to and cannot write the tool return result yourself. - -# Option B - Output Final Content: -# When you have obtained enough information through tools, output the chapter content starting with "Final Answer:". - -# ⚠️ Strictly Forbidden: -# - Forbidden to include both tool calls and Final Answer in a single reply -# - Forbidden to fabricate tool return results (Observation) yourself, all tool results are injected by the system -# - Call a maximum of one tool per reply - -# ═══════════════════════════════════════════════════════════════ -# [Chapter Content Requirements] -# ═══════════════════════════════════════════════════════════════ - -# 1. Content must be based on simulation data retrieved by tools -# 2. Quote the original text extensively to demonstrate the simulation effect -# 3. Use Markdown format (but forbid using headings): -# - Use **bold text** to mark key points (instead of subheadings) -# - Use lists (- or 1. 2. 3.) to organize points -# - Use blank lines to separate different paragraphs -# - ❌ Forbidden to use #, ##, ###, #### and any other heading syntax -# 4. [Quote Formatting Specifications - Must be a separate paragraph] -# Quotes must be an independent paragraph, with a blank line before and after, cannot be mixed in the paragraph: - -# ✅ Correct format: -# ``` -# The school's response was considered to lack substantive content. - -# > "The school's response model appears rigid and slow in the rapidly changing social media environment." - -# This evaluation reflects the general dissatisfaction of the public. -# ``` - -# ❌ Incorrect format: -# ``` -# The school's response was considered to lack substantive content. > "The school's response model..." This evaluation reflects... -# ``` -# 5. Maintain logical coherence with other chapters -# 6. [Avoid Repetition] Carefully read the completed chapter content below, do not repeat the same information -# 7. [Emphasize Again] Do not add any headings! Use **bold** instead of section headings""" - -# SECTION_USER_PROMPT_TEMPLATE = """\ -# Completed Chapter Content (Please read carefully to avoid duplication): -# {previous_content} - -# ═══════════════════════════════════════════════════════════════ -# [Current Task] Writing Chapter: {section_title} -# ═══════════════════════════════════════════════════════════════ - -# [Important Reminders] -# 1. Read the completed chapters above carefully to avoid repeating the same content! -# 2. Must call tools first to get simulation data before starting -# 3. Please mix different tools, do not use only one -# 4. Report content must come from retrieval results, do not use your own knowledge - -# [⚠️ Formatting Warning - Must be Obeyed] -# - ❌ Do not write any headings (no #, ##, ###, ####) -# - ❌ Do not write "{section_title}" as the beginning -# - ✅ Chapter titles are automatically added by the system -# - ✅ Write the main text directly, use **bold** instead of section headings - -# Please begin: -# 1. First, think (Thought) what information this chapter needs -# 2. Then, call tools (Action) to get simulation data -# 3. After collecting enough information, output Final Answer (plain text, no headings) -# """ - -SECTION_SYSTEM_PROMPT_TEMPLATE = """\ -Bạn là một chuyên gia viết "Báo cáo Dự đoán Tương lai", hiện đang viết một phần trong báo cáo đó. - -Tiêu đề báo cáo: {report_title} -Tóm tắt báo cáo: {report_summary} -Kịch bản Dự đoán (Yêu cầu Mô phỏng): {simulation_requirement} - -Phần đang được viết: {section_title} - -═══════════════════════════════════════════════════════════════ -[Khái niệm Cốt lõi] -═══════════════════════════════════════════════════════════════ - -Thế giới mô phỏng là một bản xem trước của tương lai. Chúng tôi đã đưa các điều kiện cụ thể (yêu cầu mô phỏng) vào thế giới này. -Các hành vi và tương tác của các Tác nhân (Agents) trong quá trình mô phỏng chính là những dự đoán về hành vi của con người trong tương lai. - -Nhiệm vụ của bạn là: -- Tiết lộ những gì đã xảy ra trong tương lai theo các điều kiện đã thiết lập -- Dự đoán cách các nhóm khác nhau (Agents) đã phản ứng và hành động -- Phát hiện các xu hướng, rủi ro và cơ hội đáng chú ý trong tương lai - -❌ Không viết nội dung này như một bài phân tích về hiện trạng của thế giới thực -✅ Tập trung vào "tương lai sẽ như thế nào" - kết quả mô phỏng chính là tương lai được dự đoán - -═══════════════════════════════════════════════════════════════ -[Quy tắc QUAN TRỌNG NHẤT - PHẢI Tuân thủ] -═══════════════════════════════════════════════════════════════ - -1. [PHẢI sử dụng công cụ để quan sát thế giới mô phỏng] - - Bạn đang quan sát bản xem trước tương lai từ "góc nhìn của Chúa" - - Tất cả nội dung PHẢI đến từ các sự kiện, lời nói và hành động của các Tác nhân đã xảy ra trong thế giới mô phỏng - - Nghiêm cấm sử dụng kiến thức cá nhân của bạn để viết nội dung báo cáo - - Đối với mỗi chương, bạn PHẢI gọi công cụ ít nhất 3 lần (tối đa 5 lần) để quan sát thế giới mô phỏng - -2. [PHẢI trích dẫn chính xác nguyên văn lời nói và hành động của các Tác nhân] - - Các tuyên bố và hành vi của Tác nhân là những dự đoán về hành vi con người trong tương lai - - Sử dụng định dạng trích dẫn trong báo cáo để hiển thị các dự đoán này, ví dụ: - > "Một nhóm người nhất định sẽ nói: [Nội dung gốc]..." - - Những trích dẫn này là bằng chứng cốt lõi của dự đoán mô phỏng - -3. [Tính nhất quán về Ngôn ngữ - Nội dung trích dẫn phải được dịch sang ngôn ngữ báo cáo] - - Nội dung trả về từ các công cụ có thể chứa tiếng Anh hoặc hỗn hợp tiếng Việt và tiếng Anh. - - **Báo cáo phải được viết hoàn toàn bằng tiếng Việt.** - - Khi bạn trích dẫn nội dung tiếng Anh hoặc hỗn hợp từ công cụ, bạn phải dịch sang tiếng Việt lưu loát trước khi đưa vào báo cáo - - Giữ nguyên ý nghĩa gốc khi dịch và đảm bảo cách diễn đạt tự nhiên - - Quy tắc này áp dụng cho cả văn bản chính và nội dung trong khối trích dẫn (định dạng >) - -4. [Trình bày trung thực kết quả dự đoán] - - Nội dung báo cáo phải phản ánh kết quả mô phỏng đại diện cho tương lai - - Không thêm thông tin không tồn tại trong mô phỏng - - Nếu thông tin ở một khía cạnh nào đó không đủ, hãy nêu rõ sự thật - -═══════════════════════════════════════════════════════════════ -[⚠️ Quy cách Định dạng - Cực kỳ Quan trọng!] -═══════════════════════════════════════════════════════════════ - -[Một Chương = Đơn vị Nội dung Tối thiểu] -- Mỗi chương là đơn vị chặn tối thiểu của báo cáo -- ❌ Không sử dụng bất kỳ tiêu đề Markdown nào (#, ##, ###, ####, v.v.) trong chương -- ❌ Không thêm tiêu đề chương chính ở đầu nội dung -- ✅ Tiêu đề chương được hệ thống tự động thêm vào, bạn chỉ cần viết nội dung văn bản thuần túy -- ✅ Sử dụng **chữ đậm**, ngắt đoạn, trích dẫn và danh sách để tổ chức nội dung, nhưng không dùng tiêu đề (headings) - -[Ví dụ Đúng] -``` -Chương này phân tích xu hướng lan truyền dư luận của sự kiện. Thông qua phân tích sâu dữ liệu mô phỏng, chúng tôi nhận thấy... - -**Giai đoạn bùng phát ban đầu** - -Weibo, với tư cách là bối cảnh đầu tiên của dư luận, đã đảm nhận chức năng cốt lõi là phát hành thông tin ban đầu: - -> "Weibo đã đóng góp 68% mức độ thảo luận ban đầu..." - -**Giai đoạn khuếch đại cảm xúc** - -Nền tảng Douyin đã khuếch đại thêm tác động của sự kiện: - -- Tác động thị giác mạnh mẽ -- Cộng hưởng cảm xúc cao -``` - -[Ví dụ Sai] -``` -## Tóm tắt Điều hành ← Lỗi! Không thêm bất kỳ tiêu đề nào -### 1. Giai đoạn đầu ← Lỗi! Không sử dụng ### cho các mục con -#### 1.1 Phân tích chi tiết ← Lỗi! Không sử dụng #### để chia nhỏ hơn nữa - -Chương này phân tích... -``` - -═══════════════════════════════════════════════════════════════ -[Các công cụ truy xuất hiện có] (Gọi 3-5 lần mỗi phần) -═══════════════════════════════════════════════════════════════ - -{tools_description} - -[Gợi ý Sử dụng Công cụ - Vui lòng phối hợp nhiều công cụ, không chỉ dùng một loại] -- insight_forge: Phân tích chuyên sâu, tự động phân tách câu hỏi và truy xuất sự thật cũng như các mối quan hệ từ nhiều chiều -- panorama_search: Tìm kiếm toàn cảnh góc rộng, hiểu bức tranh tổng thể, dòng thời gian và quá trình diễn biến của một sự kiện -- quick_search: Xác minh nhanh một điểm thông tin cụ thể -- interview_agents: Phỏng vấn các Tác nhân (Agents) mô phỏng để lấy góc nhìn thứ nhất và phản ứng thực tế từ các vai trò khác nhau - -═══════════════════════════════════════════════════════════════ -[Quy trình làm việc] -═══════════════════════════════════════════════════════════════ - -Đối với mỗi phản hồi, bạn chỉ có thể thực hiện một trong hai việc sau (không làm đồng thời): - -Lựa chọn A - Gọi công cụ: -Đưa ra suy nghĩ (Thought) của bạn, sau đó sử dụng định dạng sau để gọi công cụ: - -{{"name": "Tên công cụ", "parameters": {{"Tên tham số": "Giá trị tham số"}}}} - -Hệ thống sẽ thực thi công cụ và trả về kết quả cho bạn. Bạn không cần và không được phép tự viết kết quả trả về của công cụ. - -Lựa chọn B - Xuất nội dung cuối cùng: -Khi bạn đã thu thập đủ thông tin thông qua các công cụ, hãy xuất nội dung chương bắt đầu bằng "Final Answer:". - -⚠️ Nghiêm cấm: -- Cấm bao gồm cả lệnh gọi công cụ và Final Answer trong cùng một phản hồi -- Cấm tự bịa đặt kết quả trả về của công cụ (Quan sát), tất cả kết quả công cụ đều do hệ thống đưa vào -- Chỉ gọi tối đa một công cụ cho mỗi phản hồi - -═══════════════════════════════════════════════════════════════ -[Yêu cầu Nội dung Chương] -═══════════════════════════════════════════════════════════════ - -1. Nội dung phải dựa trên dữ liệu mô phỏng do công cụ truy xuất. -2. Trích dẫn rộng rãi văn bản gốc để chứng minh hiệu quả mô phỏng. -3. Sử dụng định dạng Markdown (nhưng cấm sử dụng tiêu đề): - - Sử dụng **chữ đậm** để đánh dấu các điểm chính (thay vì dùng tiêu đề phụ). - - Sử dụng danh sách (- hoặc 1. 2. 3.) để tổ chức các ý. - - Sử dụng các dòng trống để phân tách các đoạn văn khác nhau. - - ❌ Cấm sử dụng #, ##, ###, #### và bất kỳ cú pháp tiêu đề nào khác. -4. [Quy cách Định dạng Trích dẫn - Phải là một đoạn riêng biệt] - Trích dẫn phải là một đoạn văn độc lập, có dòng trống ở trước và sau, không được viết lẫn vào trong đoạn văn: - - ✅ Định dạng đúng: - ``` - Phản ứng của nhà trường bị coi là thiếu nội dung thực chất. - - > "Mô hình phản ứng của nhà trường có vẻ cứng nhắc và chậm chạp trong môi trường mạng xã hội thay đổi nhanh chóng." - - Đánh giá này phản ánh sự không hài lòng chung của công chúng. - ``` - - ❌ Định dạng sai: - ``` - Phản ứng của nhà trường bị coi là thiếu nội dung thực chất. > "Mô hình phản ứng của nhà trường..." Đánh giá này phản ánh... - ``` -5. Duy trì tính logic nhất quán với các chương khác. -6. [Tránh Lặp lại] Đọc kỹ nội dung các chương đã hoàn thành bên dưới, không lặp lại cùng một thông tin. -7. [Nhấn mạnh lại lần nữa] Không thêm bất kỳ tiêu đề nào! Sử dụng **chữ đậm** thay cho tiêu đề mục. -""" - -# ═══════════════════════════════════════════════════════════════ -# SECTION_USER_PROMPT_TEMPLATE = """\ -# Completed Chapter Content (Please read carefully to avoid duplication): -# {previous_content} - -# ═══════════════════════════════════════════════════════════════ -# [Current Task] Writing Chapter: {section_title} -# ═══════════════════════════════════════════════════════════════ - -# [Important Reminders] -# 1. Read the completed chapters above carefully to avoid repeating the same content! -# 2. Must call tools first to get simulation data before starting -# 3. Please mix different tools, do not use only one -# 4. Report content must come from retrieval results, do not use your own knowledge - -# [⚠️ Formatting Warning - Must be Obeyed] -# - ❌ Do not write any headings (no #, ##, ###, ####) -# - ❌ Do not write "{section_title}" as the beginning -# - ✅ Chapter titles are automatically added by the system -# - ✅ Write the main text directly, use **bold** instead of section headings - -# Please begin: -# 1. First, think (Thought) what information this chapter needs -# 2. Then, call tools (Action) to get simulation data -# 3. After collecting enough information, output Final Answer (plain text, no headings) -# """ - -SECTION_USER_PROMPT_TEMPLATE = """\ -Nội dung Chương đã Hoàn thành (Vui lòng đọc kỹ để tránh trùng lặp): -{previous_content} - -═══════════════════════════════════════════════════════════════ -[Nhiệm vụ Hiện tại] Viết Chương: {section_title} -═══════════════════════════════════════════════════════════════ - -[Nhắc nhở Quan trọng] -1. Đọc kỹ các chương đã hoàn thành ở trên để tránh lặp lại nội dung! -2. Phải gọi công cụ trước để lấy dữ liệu mô phỏng trước khi bắt đầu viết. -3. Vui lòng sử dụng kết hợp nhiều công cụ khác nhau, không chỉ dùng một loại. -4. Nội dung báo cáo phải đến từ kết quả truy xuất, không sử dụng kiến thức cá nhân của bạn. - -[⚠️ Cảnh báo Định dạng - Phải Tuân thủ Tuyệt đối] -- ❌ Không viết bất kỳ tiêu đề nào (không dùng các ký tự #, ##, ###, ####). -- ❌ Không viết "{section_title}" ở phần bắt đầu nội dung. -- ✅ Tiêu đề chương sẽ được hệ thống tự động thêm vào sau đó. -- ✅ Viết trực tiếp vào nội dung chính, sử dụng văn bản **in đậm** thay cho tiêu đề các mục. - -Vui lòng bắt đầu: -1. Đầu tiên, hãy suy nghĩ (Thought) xem chương này cần những thông tin gì. -2. Sau đó, gọi công cụ (Action) để lấy dữ liệu mô phỏng. -3. Sau khi thu thập đủ thông tin, xuất Câu trả lời cuối cùng (Final Answer) dưới dạng văn bản thuần túy, không chứa tiêu đề. -""" - -# ═══════════════════════════════════════════════════════════════ -# REACT_OBSERVATION_TEMPLATE = """\ -# Observation (Retrieval Result): - -# ═══ Tool {tool_name} Returned ═══ -# {result} - -# ═══════════════════════════════════════════════════════════════ -# Tool called {tool_calls_count}/{max_tool_calls} times (Used: {used_tools_str}) {unused_hint} -# - If information is sufficient: Output section content starting with "Final Answer:" (Must quote the above original text) -# - If more information is needed: Call a tool to continue retrieving -# ═══════════════════════════════════════════════════════════════ -# """ - -REACT_OBSERVATION_TEMPLATE = """\ -Quan sát (Kết quả Truy xuất): - -═══ Công cụ {tool_name} đã trả về ═══ -{result} - -═══════════════════════════════════════════════════════════════ -Công cụ đã được gọi {tool_calls_count}/{max_tool_calls} lần (Đã dùng: {used_tools_str}) {unused_hint} -- Nếu thông tin đã đủ: Xuất nội dung phần báo cáo bắt đầu bằng "Final Answer:" (Bắt buộc trích dẫn văn bản gốc ở trên) -- Nếu cần thêm thông tin: Tiếp tục gọi công cụ để truy xuất -═══════════════════════════════════════════════════════════════ -""" - -# ═══════════════════════════════════════════════════════════════ -# REACT_INSUFFICIENT_TOOLS_MSG = ( -# "[Notice] You only called the tool {tool_calls_count} times, at least {min_tool_calls} times are needed. " -# "Please call the tool again to fetch more simulation data, and then output Final Answer. {unused_hint}" -# ) - -REACT_INSUFFICIENT_TOOLS_MSG = ( - "[Thông báo] Bạn mới chỉ gọi công cụ {tool_calls_count} lần, trong khi yêu cầu tối thiểu là {min_tool_calls} lần. " - "Vui lòng gọi lại công cụ để lấy thêm dữ liệu mô phỏng, sau đó mới xuất Câu trả lời cuối cùng (Final Answer). {unused_hint}" -) - -# ═══════════════════════════════════════════════════════════════ -# REACT_INSUFFICIENT_TOOLS_MSG_ALT = ( -# "Currently tool called {tool_calls_count} times, at least {min_tool_calls} times are needed. " -# "Please call tools to fetch simulation data. {unused_hint}" -# ) - -REACT_INSUFFICIENT_TOOLS_MSG_ALT = ( - "Hiện tại công cụ mới được gọi {tool_calls_count} lần, yêu cầu ít nhất {min_tool_calls} lần. " - "Vui lòng gọi các công cụ để truy xuất dữ liệu mô phỏng. {unused_hint}" -) - -# ═══════════════════════════════════════════════════════════════ -# REACT_TOOL_LIMIT_MSG = ( -# "Tool call limit reached ({tool_calls_count}/{max_tool_calls}), cannot call tools anymore. " -# 'Please output your section content starting with "Final Answer:" immediately based on retrieved information.' -# ) - -REACT_TOOL_LIMIT_MSG = ( - "Đã đạt giới hạn gọi công cụ ({tool_calls_count}/{max_tool_calls}), không thể gọi thêm công cụ nữa. " - 'Vui lòng xuất nội dung phần báo cáo bắt đầu bằng "Final Answer:" ngay lập tức dựa trên những thông tin đã truy xuất được.' -) - -# ═══════════════════════════════════════════════════════════════ -# REACT_UNUSED_TOOLS_HINT = "\n💡 You haven't used: {unused_list}, suggesting trying different tools for multiple perspectives" - -REACT_UNUSED_TOOLS_HINT = "\n💡 Bạn chưa sử dụng: {unused_list}, hãy thử các công cụ khác nhau để có cái nhìn đa chiều hơn" - -# ═══════════════════════════════════════════════════════════════ -# REACT_FORCE_FINAL_MSG = "Tool call limit reached, please output Final Answer: and generate section content directly." - -REACT_FORCE_FINAL_MSG = "Đã đạt giới hạn gọi công cụ, vui lòng xuất Final Answer: và trực tiếp tạo nội dung cho phần này." - -# ═══════════════════════════════════════════════════════════════ -# CHAT_SYSTEM_PROMPT_TEMPLATE = """\ -# You are a concise and efficient simulation prediction assistant. - -# [Background] -# Prediction condition: {simulation_requirement} - -# [Generated Analysis Report] -# {report_content} - -# [Rules] -# 1. Prioritize answering based on the report content above -# 2. Answer the question directly, avoid lengthy reasoning -# 3. Only call tools to retrieve more data if the report content is insufficient to answer -# 4. Answers must be concise, clear, and organized - -# [Available Tools] (Use only when necessary, call 1-2 times max) -# {tools_description} - -# [Tool Call Format] -# -# {{"name": "Tool Name", "parameters": {{"Parameter Name": "Parameter Value"}}}} -# - -# [Answering Style] -# - Concise and direct, avoid long paragraphs -# - Use > format to quote key content -# - Provide conclusion first, then explain the reason -# """ - -CHAT_SYSTEM_PROMPT_TEMPLATE = """ -Bạn là một trợ lý dự đoán mô phỏng súc tích và hiệu quả. - -[Bối cảnh] -Điều kiện dự đoán: {simulation_requirement} - -[Báo cáo Phân tích Đã tạo] -{report_content} - -[Quy tắc] -1. Ưu tiên trả lời dựa trên nội dung báo cáo ở trên. -2. Trả lời câu hỏi trực tiếp, tránh lập luận dài dòng. -3. Chỉ gọi công cụ để truy xuất thêm dữ liệu nếu nội dung báo cáo không đủ để trả lời. -4. Câu trả lời phải súc tích, rõ ràng và có tổ chức. - -[Các Công cụ Hiện có] (Chỉ sử dụng khi cần thiết, gọi tối đa 1-2 lần) -{tools_description} - -[Định dạng Gọi Công cụ] - -{{"name": "Tên Công cụ", "parameters": {{"Tên Tham số": "Giá trị Tham số"}}}} - - -[Phong cách Trả lời] -- Ngắn gọn và trực tiếp, tránh các đoạn văn dài. -- Sử dụng định dạng > để trích dẫn nội dung chính. -- Đưa ra kết luận trước, sau đó mới giải thích lý do. -""" - -# ═══════════════════════════════════════════════════════════════ -# CHAT_OBSERVATION_SUFFIX = "\n\nPlease answer the question concisely." - -CHAT_OBSERVATION_SUFFIX = "\n\nVui lòng trả lời câu hỏi một cách súc tích." - # ═══════════════════════════════════════════════════════════════ # Class chính: ReportAgent # ═══════════════════════════════════════════════════════════════ diff --git a/backend/app/services/simulation_config_generator.py b/backend/app/services/simulation_config_generator.py index 27bb04d5..6573a7da 100644 --- a/backend/app/services/simulation_config_generator.py +++ b/backend/app/services/simulation_config_generator.py @@ -1,13 +1,26 @@ """ -Trình tạo tạo ra cấu hình Simulation tự động -Sử dụng LLM theo yêu cầu mô phỏng, nội dung tài liệu và thông tin đồ thị để tự động thiết lập chi tiết các tham số -Tất cả đều tự động mà không cần can thiệp thủ công tạo tham số +Trình tạo cấu hình Simulation tự động bằng LLM -Áp dụng chiến lược tạo từng bước để tránh lỗi do cố gắng tạo nội dung quá dài cùng một lúc: -1. Tạo cấu hình thời gian -2. Tạo cấu hình các Event -3. Tạo cấu hình cho các Agent theo đợt -4. Tạo cấu hình nền tảng +Vị trí trong pipeline: +───────────────────────────────────────────────────────────────────────────── + simulation_manager.prepare_simulation() [Giai đoạn 3 — bước cuối cùng] + └─ SimulationConfigGenerator.generate_config() + ├─ _generate_time_config() [LLM → bao nhiêu giờ, giờ nào cao điểm] + ├─ _generate_event_config() [LLM → hot topics, initial posts] + ├─ _generate_agent_configs_batch() × N [LLM → hành vi từng agent] + └─ _assign_initial_post_agents() [ghép poster_type → agent_id] +───────────────────────────────────────────────────────────────────────────── + +Input: List[EntityNode] từ ZepEntityReader + document_text + simulation_requirement +Output: SimulationParameters → ghi ra simulation_config.json + (File này sau đó được đọc bởi run_parallel_simulation.py khi OASIS chạy) + +Chiến lược LLM chia từng bước: + Thay vì gửi toàn bộ dữ liệu 1 lần (dễ bị token limit), chia thành 4 bước nhỏ: + 1. Time config — context cắt còn 10,000 ký tự + 2. Event config — context cắt còn 8,000 ký tự + 3. Agent configs — chia batch 15 agent/lần (N batch) + 4. Platform config — hardcoded (không cần LLM) """ import json @@ -25,7 +38,17 @@ from .zep_entity_reader import EntityNode, ZepEntityReader logger = get_logger('mirofish.simulation_config') -# Cấu hình thời gian thói quen Trung Quốc (Theo giờ Bắc Kinh) + +# ============================================================================== +# HẰNG SỐ: CHINA_TIMEZONE_CONFIG — Mẫu hoạt động theo múi giờ UTC+7 +# ============================================================================== +# Lưu ý tên biến: được kế thừa từ codebase OASIS gốc của Trung Quốc (UTC+8). +# Thực tế trong MiroFish, hệ thống đã chuyển sang dùng múi giờ Việt Nam (UTC+7) — +# các prompt LLM đều chỉ định "người Việt Nam / giờ Hà Nội". +# Biến này hiện chỉ dùng để tham khảo/documentation, KHÔNG được import trực tiếp +# vào các hàm tạo config (logic thực tế nằm trong prompt LLM và rule fallback). +# ============================================================================== + CHINA_TIMEZONE_CONFIG = { # Khung giờ khuya (Hầu như không có hoạt động) "dead_hours": [0, 1, 2, 3, 4, 5], @@ -48,133 +71,220 @@ CHINA_TIMEZONE_CONFIG = { } +# ============================================================================== +# DATACLASS: AgentActivityConfig — Cấu hình hành vi của một agent trong OASIS +# ============================================================================== +# Mỗi AgentActivityConfig tương ứng với 1 EntityNode đã được map thành agent. +# OASIS đọc các field này để quyết định: +# - Agent này có "thức dậy" trong round này không? (activity_level) +# - Nếu thức, nó làm gì? (posts_per_hour, comments_per_hour, stance) +# - Bài của nó được bao nhiêu agent khác nhìn thấy? (influence_weight) +# ============================================================================== + @dataclass class AgentActivityConfig: """Cấu hình hoạt động cho một Agent""" - agent_id: int - entity_uuid: str - entity_name: str - entity_type: str - - # Mức độ hoạt động (0.0-1.0) - activity_level: float = 0.5 # Hoạt động tổng thể - - # Tần suất phát ngôn (Số lần comment dự kiến mỗi giờ) + + # --- Định danh --- + agent_id: int # Phải khớp với user_id trong reddit_profiles.json / twitter_profiles.csv + entity_uuid: str # UUID trong Zep — dùng để trace ngược lại nguồn gốc + entity_name: str # Tên hiển thị (ví dụ: "Trần Văn An") + entity_type: str # Loại entity (ví dụ: "Student", "MediaOutlet") + + # --- Mức độ hoạt động tổng thể --- + # 0.0 = không bao giờ hoạt động; 1.0 = luôn hoạt động mỗi round + # OASIS dùng giá trị này nhân với activity_multiplier của giờ hiện tại + # để tính xác suất agent được chọn trong round. + activity_level: float = 0.5 + + # --- Tần suất phát ngôn --- + # Số lần trung bình agent đăng bài mới / bình luận trong 1 giờ mô phỏng posts_per_hour: float = 1.0 comments_per_hour: float = 2.0 - - # Khoảng thời gian hoạt động (Hệ 24 giờ, 0-23) + + # --- Khoảng thời gian hoạt động (hệ 24 giờ, 0–23) --- + # Agent chỉ có thể được kích hoạt trong các giờ này. + # Ví dụ: student=[8,9,10,18,19,20,21,22,23], university=[9,10,...,17] active_hours: List[int] = field(default_factory=lambda: list(range(8, 23))) - - # Tốc độ phản hồi (Độ trễ phản ứng với sự kiện nóng, đơn vị: phút mô phỏng) + + # --- Tốc độ phản hồi --- + # Độ trễ (phút mô phỏng) trước khi agent phản ứng với sự kiện nóng. + # Nhỏ = phản ứng nhanh (sinh viên: 1-15 phút) + # Lớn = phản ứng chậm (cơ quan chính phủ: 60-240 phút) response_delay_min: int = 5 response_delay_max: int = 60 - - # Khuynh hướng cảm xúc (-1.0 đến 1.0, từ tiêu cực đến tích cực) + + # --- Khuynh hướng cảm xúc --- + # -1.0 = rất tiêu cực (phản đối mạnh mẽ) + # 0.0 = trung lập + # +1.0 = rất tích cực (ủng hộ nhiệt tình) sentiment_bias: float = 0.0 - - # Lập trường (Thái độ đối với chủ đề cụ thể) - stance: str = "neutral" # supportive, opposing, neutral, observer - - # Trọng số ảnh hưởng (Xác định mức độ bài đăng được Agent khác nhìn thấy) + + # --- Lập trường với chủ đề mô phỏng --- + # "supportive": ủng hộ quyết định/sự kiện được mô phỏng + # "opposing": phản đối + # "neutral": không rõ ràng, phân tích khách quan + # "observer": chỉ theo dõi, ít tham gia tranh luận + stance: str = "neutral" + + # --- Trọng số ảnh hưởng --- + # Mức độ bài đăng của agent này xuất hiện trên "timeline" của agent khác. + # Cao = lan rộng hơn (mediaoutlet: 2.5, university: 3.0) + # Thấp = ít người thấy (student: 0.8) influence_weight: float = 1.0 -@dataclass +# ============================================================================== +# DATACLASS: TimeSimulationConfig — Cấu hình thời gian và nhịp hoạt động +# ============================================================================== +# Định nghĩa "lịch sinh hoạt" của thế giới mô phỏng: +# - Tổng độ dài mô phỏng (total_simulation_hours) +# - Mỗi round đại diện cho bao nhiêu phút thực (minutes_per_round) +# - Khung giờ nào sẽ có nhiều hay ít agent hoạt động (peak/off_peak/morning/work) +# +# OASIS đọc các multiplier để tính số agent active mỗi round: +# active_count = random(min, max) × multiplier[current_hour] +# ============================================================================== + +@dataclass class TimeSimulationConfig: - """Cấu hình thời gian mô phỏng (Dựa trên thói quen sinh hoạt của người Trung)""" - # Tổng thời gian mô phỏng (Giờ) - total_simulation_hours: int = 72 # Mặc định là chạy mô phỏng 72 tiếng (3 ngày) - - # Số phút đại diện cho mỗi vòng - Mặc định 60 phút (1 giờ), đẩy nhanh thời gian + """Cấu hình thời gian mô phỏng""" + + # Tổng số giờ mô phỏng — 72 giờ = 3 ngày thực + # (Với minutes_per_round=60, tổng số round = 72) + total_simulation_hours: int = 72 + + # Mỗi round đại diện cho bao nhiêu phút trong thế giới thực + # 60 = mỗi round là 1 giờ (khuyến nghị để cân bằng tốc độ vs độ chi tiết) minutes_per_round: int = 60 - - # Phạm vi số lượng Agent kích hoạt mỗi giờ + + # Số agent được kích hoạt mỗi giờ — LLM chọn giá trị trong [min, max] + # dựa trên quy mô simulation (số entity) agents_per_hour_min: int = 5 agents_per_hour_max: int = 20 - - # Giờ cao điểm (19-22 giờ tối, thời gian sôi động nhất) + + # Khung giờ cao điểm: 19-22 giờ — đông nhất, nhân với 1.5 peak_hours: List[int] = field(default_factory=lambda: [19, 20, 21, 22]) peak_activity_multiplier: float = 1.5 - - # Khung giờ chết (0-5 giờ, hầu như không ai on) + + # Khung giờ chết: 0-5 giờ sáng — gần như không ai online, nhân với 0.05 off_peak_hours: List[int] = field(default_factory=lambda: [0, 1, 2, 3, 4, 5]) - off_peak_activity_multiplier: float = 0.05 # Rạng sáng gần như bằng không - - # Khung giờ buổi sáng + off_peak_activity_multiplier: float = 0.05 + + # Buổi sáng: 6-8 giờ — hoạt động tăng dần, nhân với 0.4 morning_hours: List[int] = field(default_factory=lambda: [6, 7, 8]) morning_activity_multiplier: float = 0.4 - - # Khung giờ làm việc + + # Giờ làm việc: 9-18 giờ — hoạt động ổn định, nhân với 0.7 work_hours: List[int] = field(default_factory=lambda: [9, 10, 11, 12, 13, 14, 15, 16, 17, 18]) work_activity_multiplier: float = 0.7 +# ============================================================================== +# DATACLASS: EventConfig — Cấu hình "ngòi nổ" khởi động cuộc mô phỏng +# ============================================================================== +# initial_posts là các bài đăng đầu tiên được post ngay khi simulation bắt đầu. +# Chúng tạo ra "sự kiện kích hoạt" để các agent khác phản ứng. +# +# Sau khi LLM sinh initial_posts (có poster_type), hàm _assign_initial_post_agents() +# sẽ map poster_type → agent_id thực để OASIS biết agent nào post bài đó. +# ============================================================================== + @dataclass class EventConfig: - """Cấu hình sự kiện cho Simulation""" - # Các bài Post/Sự kiện khởi đầu (Bắt đầu ngay khi chạy mô phỏng) + """Cấu hình sự kiện cho Simulation gồm initial_posts, scheduled_events, hot_topics, narrative_direction""" + + # Bài đăng khởi đầu — post ngay round 1, tạo "tin nóng" để agent phản ứng + # Mỗi phần tử: {"content": "...", "poster_type": "MediaOutlet", "poster_agent_id": 12} + # poster_agent_id được gán sau bởi _assign_initial_post_agents() initial_posts: List[Dict[str, Any]] = field(default_factory=list) - - # Các sự kiện được lập lịch vào các thời điểm nhất định + + # Sự kiện lập lịch — chưa được implement, dành cho tính năng tương lai scheduled_events: List[Dict[str, Any]] = field(default_factory=list) - - # Từ khóa dành cho các chủ đề đang hot (Hot topics) + + # Từ khóa chủ đề nóng — OASIS dùng để tăng khả năng agent chú ý đến chủ đề này + # Ví dụ: ["học phí", "biểu tình", "giáo dục"] hot_topics: List[str] = field(default_factory=list) - - # Hướng dẫn dư luận / Đường lối thảo luận + + # Hướng dẫn tổng quan diễn biến dư luận — LLM viết ra để định hướng + # Ví dụ: "Thông báo tăng học phí → sinh viên phản đối → leo thang thành phong trào..." narrative_direction: str = "" +# ============================================================================== +# DATACLASS: PlatformConfig — Cấu hình riêng cho từng nền tảng mạng xã hội +# ============================================================================== +# Twitter và Reddit có thuật toán feed khác nhau → cần cấu hình riêng. +# Các giá trị này được OASIS dùng khi quyết định bài nào hiển thị trên timeline agent. +# +# recency_weight + popularity_weight + relevance_weight = 1.0 (không bắt buộc, nhưng +# nên cộng lại = 1 để tránh scale lệch) +# ============================================================================== + @dataclass class PlatformConfig: - """Cấu hình đặc thù dành riêng cho các nền tảng""" - platform: str # twitter or reddit - - # Trọng số cho các thuật toán đề xuất - recency_weight: float = 0.4 # Độ mới của bài - popularity_weight: float = 0.3 # Mức độ phổ biến truyền miệng - relevance_weight: float = 0.3 # Mức độ quan tâm / tương quan - - # Ngưỡng lan truyền virus (Cần bao nhiêu tương tác để nội dung bắt đầu phát tán mạnh) + """Cấu hình đặc thù dành riêng cho các nền tảng gồm platform, recency_weight, popularity_weight, relevance_weight, viral_threshold và echo_chamber_strength""" + platform: str # "twitter" hoặc "reddit" + + # Ba trọng số trong thuật toán đề xuất nội dung: + recency_weight: float = 0.4 # Ưu tiên bài mới (Twitter: cao hơn Reddit) + popularity_weight: float = 0.3 # Ưu tiên bài nhiều like/repost + relevance_weight: float = 0.3 # Ưu tiên bài liên quan đến sở thích agent + + # Ngưỡng lan truyền "viral": + # Twitter: 10 — dễ lan hơn (retweet lan nhanh) + # Reddit: 15 — khó lan hơn (cần nhiều upvote hơn) + # Khi bài đạt ngưỡng này, OASIS tăng xác suất nó xuất hiện trên feed của nhiều agent viral_threshold: int = 10 - - # Độ mạnh của hiệu ứng lan truyền trong nhóm chung chí hướng (buồng phản âm) + + # Hiệu ứng "buồng phản âm" (echo chamber): + # 0.0 = không có — agent thấy đa chiều + # 1.0 = tối đa — agent chỉ thấy bài cùng quan điểm + # Reddit thường cao hơn (0.6) vì cơ chế subreddit tạo bong bóng thông tin echo_chamber_strength: float = 0.5 +# ============================================================================== +# DATACLASS: SimulationParameters — Object tổng hợp toàn bộ cấu hình simulation +# ============================================================================== +# Đây là output cuối cùng của SimulationConfigGenerator.generate_config(). +# Nó được serialize thành simulation_config.json — file trung tâm mà: +# 1. Flask đọc để điền metadata vào state.json +# 2. OASIS scripts đọc để khởi tạo môi trường và chạy simulation +# +# to_json() đảm bảo Unicode (tiếng Việt) không bị escape thành \uXXXX +# ============================================================================== + @dataclass class SimulationParameters: - """完整的模拟参数配置""" - # 基础信息 + """Bộ tham số cấu hình đầy đủ cho một lượt Simulation""" + + # --- Định danh (bắt buộc khi tạo object) --- simulation_id: str project_id: str graph_id: str - simulation_requirement: str - - # Cấu hình thời gian + simulation_requirement: str # Yêu cầu gốc từ người dùng — được lưu để traceback + + # --- Các cấu hình con (có default để tạo object không cần truyền hết) --- time_config: TimeSimulationConfig = field(default_factory=TimeSimulationConfig) - - # Danh sách cấu hình Agent agent_configs: List[AgentActivityConfig] = field(default_factory=list) - - # Cấu hình Event event_config: EventConfig = field(default_factory=EventConfig) - - # Cấu hình nền tảng + + # None nếu platform tương ứng bị tắt (enable_twitter=False / enable_reddit=False) twitter_config: Optional[PlatformConfig] = None reddit_config: Optional[PlatformConfig] = None - - # Cấu hình LLM - llm_model: str = "" - llm_base_url: str = "" - - # Dữ liệu metadata khi tạo + + # --- Metadata LLM --- + llm_model: str = "" # Tên model đã dùng để gen (dùng để debug khi chất lượng kém) + llm_base_url: str = "" # Base URL API (có thể là OpenAI, Azure, local...) + + # --- Metadata tạo --- generated_at: str = field(default_factory=lambda: datetime.now().isoformat()) - generation_reasoning: str = "" # Giải thích suy luận từ LLM - + # Chuỗi reasoning nối từ tất cả các bước: "Time config: ...|Event config: ...|Agent: ..." + generation_reasoning: str = "" + def to_dict(self) -> Dict[str, Any]: - """Convert sang định dạng Dictionary""" + """Serialize toàn bộ object thành dict — dùng để truyền nội bộ.""" time_dict = asdict(self.time_config) return { "simulation_id": self.simulation_id, @@ -191,37 +301,61 @@ class SimulationParameters: "generated_at": self.generated_at, "generation_reasoning": self.generation_reasoning, } - + def to_json(self, indent: int = 2) -> str: - """Convert sang định dạng chuỗi JSON""" + """Serialize thành chuỗi JSON — ensure_ascii=False để giữ nguyên tiếng Việt.""" return json.dumps(self.to_dict(), ensure_ascii=False, indent=indent) +# ============================================================================== +# CLASS: SimulationConfigGenerator — Trình điều phối sinh config bằng LLM +# ============================================================================== +# Được gọi bởi: SimulationManager.prepare_simulation() ở Giai đoạn 3. +# Input đến từ: ZepEntityReader.filter_defined_entities() (danh sách EntityNode) +# +# Sơ đồ luồng gọi LLM trong generate_config(): +# +# generate_config() +# │ +# ├── Bước 1: _generate_time_config() ← 1 LLM call +# │ └── _parse_time_config() ← validate + parse +# │ +# ├── Bước 2: _generate_event_config() ← 1 LLM call +# │ └── _parse_event_config() ← parse đơn giản +# │ +# ├── Bước 3..N: _generate_agent_configs_batch() ← 1 LLM call × N batch +# │ └── _generate_agent_config_by_rule() ← fallback nếu LLM fail/thiếu +# │ +# ├── _assign_initial_post_agents() ← không cần LLM, map bằng alias +# │ +# └── Bước cuối: PlatformConfig hardcoded ← không cần LLM +# +# Mỗi LLM call đi qua _call_llm_with_retry() với retry 3 lần, temperature giảm dần. +# ============================================================================== + class SimulationConfigGenerator: """ Trình tạo cấu hình Simulation tự động bằng LLM - + Sử dụng LLM phân tích yêu cầu mô phỏng, nội dung tài liệu, Entity từ đồ thị, - Tự động xây dựng các thông số cấu trúc tối ưu cho đợt Simulation - - Áp dụng chiến lược tạo từng bước: + tự động xây dựng các thông số cấu trúc tối ưu cho đợt Simulation. + + Chiến lược chia từng bước (step-by-step generation): 1. Tạo cấu hình thời gian và cấu hình Event (Nhẹ, chạy nhanh) - 2. Phân nhỏ đợt tạo cấu hình cho Agent (Khoảng 10-20 agent mỗi đợt) - 3. Tạo cấu hình nền tảng + 2. Phân nhỏ đợt tạo cấu hình cho Agent (15 agent mỗi đợt) + 3. Tạo cấu hình nền tảng (hardcoded, không cần LLM) """ - - # Số lượng ký tự tối đa của bộ context - MAX_CONTEXT_LENGTH = 50000 - # Số lượng Agent để gen cho một lần - AGENTS_PER_BATCH = 15 - - # Số lượng ký tự giới hạn ở các bước để cắt chuỗi (Ký tự đoạn) - TIME_CONFIG_CONTEXT_LENGTH = 10000 # Cấu hình thời gian - EVENT_CONFIG_CONTEXT_LENGTH = 8000 # Cấu hình sự kiện - ENTITY_SUMMARY_LENGTH = 300 # Tóm tắt các thực thể - AGENT_SUMMARY_LENGTH = 300 # Tóm tắt cấu hình Agent - ENTITIES_PER_TYPE_DISPLAY = 20 # Lượng thực thể cho mổi loại để hiển thị - + + MAX_CONTEXT_LENGTH = 50000 # Giới hạn tổng ngữ cảnh gửi cho LLM (ký tự) + AGENTS_PER_BATCH = 15 # Số agent xử lý mỗi lần gọi LLM (tránh token limit) + + # Độ dài context cho từng bước — cắt ngắn để không waste token + TIME_CONFIG_CONTEXT_LENGTH = 10000 # Bước 1: chỉ cần biết chủ đề + entity types + EVENT_CONFIG_CONTEXT_LENGTH = 8000 # Bước 2: cần biết thêm về các entity điển hình + ENTITY_SUMMARY_LENGTH = 300 # Tóm tắt mỗi entity trong context + AGENT_SUMMARY_LENGTH = 300 # Tóm tắt entity trong prompt sinh agent config + ENTITIES_PER_TYPE_DISPLAY = 20 # Tối đa bao nhiêu entity/loại trong context + def __init__( self, api_key: Optional[str] = None, @@ -231,16 +365,17 @@ class SimulationConfigGenerator: self.api_key = api_key or Config.LLM_API_KEY self.base_url = base_url or Config.LLM_BASE_URL self.model_name = model_name or Config.LLM_MODEL_NAME - + if not self.api_key: raise ValueError("LLM_API_KEY has not been configured") - + self.client = OpenAI( api_key=self.api_key, base_url=self.base_url ) + # Metadata runtime được inject vào mỗi LLM call để theo dõi cost/usage self._runtime_metadata: Dict[str, Any] = {} - + def generate_config( self, simulation_id: str, @@ -254,21 +389,30 @@ class SimulationConfigGenerator: progress_callback: Optional[Callable[[int, int, str], None]] = None, ) -> SimulationParameters: """ - Tạo cấu hình Simulation thông minh tự động hoàn chỉnh (Bằng tư duy chia từng bước) - + Pipeline sinh cấu hình simulation hoàn chỉnh theo từng bước. + + Tổng quan luồng: + ┌───────────────────────────────────────────────────────────────┐ + │ Bước 1: _generate_time_config() → TimeSimulationConfig │ + │ Bước 2: _generate_event_config() → EventConfig + initial_posts│ + │ Bước 3..N: _generate_agent_configs_batch() × ceil(N/15) │ + │ _assign_initial_post_agents() → gán poster_agent_id │ + │ Bước cuối: PlatformConfig hardcoded │ + └───────────────────────────────────────────────────────────────┘ + Args: - simulation_id: Nhận dạng quy trình chạy Simulation - project_id: Mã định danh dự án - graph_id: Đồ thị đồ thị - simulation_requirement: Yêu cầu của quá trình mô phỏng - document_text: Nội dung file tài liệu nguồn - entities: Danh sách các thực thể đã được lọc - enable_twitter: Cờ hiệu để bật Twitter - enable_reddit: Cờ hiệu để bật Reddit - progress_callback: Hàm callback lấy trạng thái tiến trình hiện tại (current_step, total_steps, message) - + simulation_id: ID của simulation (dùng để điền vào SimulationParameters) + project_id: ID project + graph_id: ID Zep Graph (dùng để điền metadata, không query thêm ở đây) + simulation_requirement: Yêu cầu người dùng (truyền thẳng vào LLM prompt) + document_text: Văn bản tài liệu gốc (làm ngữ cảnh nền tảng cho LLM) + entities: Danh sách EntityNode từ ZepEntityReader (đã lọc + enrich) + enable_twitter: True → sinh twitter_config + enable_reddit: True → sinh reddit_config + progress_callback: Hàm nhận (current_step, total_steps, message: str) + Returns: - SimulationParameters: Bộ tổng cấu hình thông số đầy đủ + SimulationParameters đầy đủ — gọi .to_json() để ghi ra file. """ logger.info(f"Start generating simulation configuration: simulation_id={simulation_id}, entity_count={len(entities)}") self._runtime_metadata = { @@ -277,52 +421,57 @@ class SimulationConfigGenerator: "component": "simulation_config_generator", "phase": "prepare_simulation_config", } - - # Tính toán tổng số bước + + # Tính tổng số bước để progress_callback hiển thị đúng phần trăm + # total_steps = 1 (time) + 1 (event) + N_batch (agents) + 1 (platform) = 3 + N_batch num_batches = math.ceil(len(entities) / self.AGENTS_PER_BATCH) - total_steps = 3 + num_batches # Cấu hình tgian + Sự kiện + Nx(Agent Batch) + Nền tảng + total_steps = 3 + num_batches current_step = 0 - + def report_progress(step: int, message: str): nonlocal current_step current_step = step if progress_callback: progress_callback(step, total_steps, message) logger.info(f"[{step}/{total_steps}] {message}") - - # 1. Xây dựng thông tin ngữ cảnh cơ bản + + # Xây dựng context chung (tối đa 50,000 ký tự) dùng cho tất cả các bước + # Bao gồm: simulation_requirement + entity summary + document_text (phần còn lại) context = self._build_context( simulation_requirement=simulation_requirement, document_text=document_text, entities=entities ) - + + # Mảng tích lũy reasoning từ mỗi bước — nối lại cuối để lưu vào SimulationParameters reasoning_parts = [] - - # ========== Bước 1: Tạo bộ cấu hình về Thời Gian ========== + + # ========== Bước 1: Cấu hình Thời gian ========== report_progress(1, "Generating time configuration...") num_entities = len(entities) time_config_result = self._generate_time_config(context, num_entities) time_config = self._parse_time_config(time_config_result, num_entities) reasoning_parts.append(f"Time config reasoning: {time_config_result.get('reasoning', 'Success')}") - # ========== Bước 2: Tạo cấu hình Event ========== + + # ========== Bước 2: Cấu hình Sự kiện ========== report_progress(2, "Generating event configuration and hot topics...") event_config_result = self._generate_event_config(context, simulation_requirement, entities) event_config = self._parse_event_config(event_config_result) reasoning_parts.append(f"Event config reasoning: {event_config_result.get('reasoning', 'Success')}") - - # ========== Bước 3-N: Chia thành các đợt để lấy cấu hình Agent ========== + + # ========== Bước 3..N: Cấu hình Agent (chia batch) ========== + # Ví dụ: 47 entity → batch1(0-14) + batch2(15-29) + batch3(30-44) + batch4(45-46) all_agent_configs = [] for batch_idx in range(num_batches): start_idx = batch_idx * self.AGENTS_PER_BATCH end_idx = min(start_idx + self.AGENTS_PER_BATCH, len(entities)) batch_entities = entities[start_idx:end_idx] - + report_progress( 3 + batch_idx, f"Generating agent configuration ({start_idx + 1}-{end_idx}/{len(entities)})..." ) - + batch_configs = self._generate_agent_configs_batch( context=context, entities=batch_entities, @@ -330,41 +479,43 @@ class SimulationConfigGenerator: simulation_requirement=simulation_requirement ) all_agent_configs.extend(batch_configs) - + reasoning_parts.append(f"Agent config reasoning: Successfully generated {len(all_agent_configs)} agents") - - # ========== Tiến hành gán người (Agent) để đăng các bài Initial Post ========== + + # ========== Gán agent cho initial posts ========== + # Sau khi có đủ all_agent_configs, map poster_type → agent_id thực logger.info("Assigning poster agents for initial posts...") event_config = self._assign_initial_post_agents(event_config, all_agent_configs) assigned_count = len([p for p in event_config.initial_posts if p.get("poster_agent_id") is not None]) reasoning_parts.append(f"Initial post assignment: {assigned_count} posts have been assigned to publishers") - - # ========== Bước cuối: Thiết lập nền tảng ========== + + # ========== Bước cuối: Platform Config (hardcoded) ========== + # Không cần LLM — các giá trị này là hằng số đã được tuning thực nghiệm report_progress(total_steps, "Generating platform configuration...") twitter_config = None reddit_config = None - + if enable_twitter: twitter_config = PlatformConfig( platform="twitter", - recency_weight=0.4, + recency_weight=0.4, # Twitter ưu tiên bài mới hơn Reddit popularity_weight=0.3, relevance_weight=0.3, - viral_threshold=10, + viral_threshold=10, # Dễ lan hơn Reddit echo_chamber_strength=0.5 ) - + if enable_reddit: reddit_config = PlatformConfig( platform="reddit", - recency_weight=0.3, - popularity_weight=0.4, + recency_weight=0.3, # Reddit cân bằng hơn giữa mới vs phổ biến + popularity_weight=0.4, # Reddit ưu tiên upvote hơn thời gian relevance_weight=0.3, - viral_threshold=15, - echo_chamber_strength=0.6 + viral_threshold=15, # Ngưỡng cao hơn → khó "bùng" hơn Twitter + echo_chamber_strength=0.6 # Reddit có subreddit → buồng phản âm mạnh hơn ) - - # Xây dựng các tham số cuối cùng kết thúc quy trình + + # Gộp tất cả vào SimulationParameters params = SimulationParameters( simulation_id=simulation_id, project_id=project_id, @@ -379,54 +530,69 @@ class SimulationConfigGenerator: llm_base_url=self.base_url, generation_reasoning=" | ".join(reasoning_parts) ) - + logger.info(f"Simulation configuration generation complete: {len(params.agent_configs)} agent configs created") - + return params - + def _build_context( self, simulation_requirement: str, document_text: str, entities: List[EntityNode] ) -> str: - """Thực hiện xây dựng nội dung Prompt Ngữ cảnh cho LLM, với độ dài có thể bị giới hạn""" - - # Tóm tắt lại Thực thể + """ + Tổng hợp ngữ cảnh cho LLM — tối đa MAX_CONTEXT_LENGTH ký tự. + + Ưu tiên (thứ tự giảm dần): + 1. simulation_requirement — giữ nguyên 100%, không cắt + 2. entity_summary — tóm tắt nhóm theo loại, tối đa ENTITIES_PER_TYPE_DISPLAY/loại + 3. document_text — phần còn lại sau khi đã dùng cho 1+2 + + Document_text bị cắt cuối nếu tổng vượt quá 50,000 ký tự. + """ entity_summary = self._summarize_entities(entities) - - # Xây dựng nội dung + context_parts = [ f"## Simulation Requirements\n{simulation_requirement}", f"\n## Entity Information ({len(entities)} entities)\n{entity_summary}", ] - + current_length = sum(len(p) for p in context_parts) - remaining_length = self.MAX_CONTEXT_LENGTH - current_length - 500 # Dành sẵn 500 ký tự trống - + # Dành 500 ký tự buffer để tránh off-by-one khi LLM đếm token + remaining_length = self.MAX_CONTEXT_LENGTH - current_length - 500 + if remaining_length > 0 and document_text: doc_text = document_text[:remaining_length] if len(document_text) > remaining_length: - doc_text += "\n...(Document Truncated)" + doc_text += "\n...(Document Truncated)" # note: việc cắt bớt đi có bị ảnh hưởng gì nghiêm trọng không? nếu để full thì có tốn context không? context_parts.append(f"\n## Original Document Content\n{doc_text}") - + return "\n".join(context_parts) - + def _summarize_entities(self, entities: List[EntityNode]) -> str: - """Tạo chuỗi văn bản Tóm tắt cho các Thực thể""" + """ + Tạo chuỗi tóm tắt entity theo nhóm loại — dùng trong _build_context(). + + Format output: + ### Student (5 entity) + - Trần Văn An: Sinh viên năm 3 ngành CNTT... + - Nguyễn Thị B: ... + ### University (2 entity) + - Đại học X: ... + """ lines = [] - - # Phân nhóm bằng Loại + + # Nhóm entity theo loại để LLM dễ nắm phân phối nhân vật by_type: Dict[str, List[EntityNode]] = {} for e in entities: t = e.get_entity_type() or "Unknown" if t not in by_type: by_type[t] = [] by_type[t].append(e) - + for entity_type, type_entities in by_type.items(): lines.append(f"\n### {entity_type} ({len(type_entities)} entity)") - # Số lượng đã được thiết lập mặc định và Giới hạn chiều dài của bảng tóm tắt display_count = self.ENTITIES_PER_TYPE_DISPLAY summary_len = self.ENTITY_SUMMARY_LENGTH for e in type_entities[:display_count]: @@ -434,16 +600,40 @@ class SimulationConfigGenerator: lines.append(f"- {e.name}: {summary_preview}") if len(type_entities) > display_count: lines.append(f" ... and {len(type_entities) - display_count} more entities") - + return "\n".join(lines) - + + ''' + ### Student (5 entity) + - Trần Văn An: Sinh viên năm 3 ngành CNTT tại Đại học X... + - Nguyễn Thị B: Sinh viên năm 2, hoạt động phong trào... + + ### University (1 entity) + - Đại học X: Trường đại học công lập lớn tại Hà Nội... + + ### MediaOutlet (2 entity) + - VnExpress: Báo điện tử lớn nhất Việt Nam... + ''' + def _call_llm_with_retry(self, prompt: str, system_prompt: str) -> Dict[str, Any]: - """Tích hợp cơ chế retry mỗi lúc gọi Request LLM bị lỗi và Logic sửa lỗi JSON string""" + """ + Gọi LLM và retry tối đa 3 lần với temperature giảm dần. + + Chiến lược retry: + Attempt 1: temperature=0.7 (sáng tạo, đa dạng hơn) + Attempt 2: temperature=0.6 + sleep 2s (nếu attempt 1 fail) + Attempt 3: temperature=0.5 + sleep 4s (nếu attempt 2 fail) + + Temperature giảm dần → output ổn định hơn, ít hallucination → dễ parse JSON hơn. + Nếu finish_reason="length" (bị cắt do token limit) → _fix_truncated_json() trước khi parse. + Nếu json.loads() lỗi → _try_fix_config_json() để cố sửa. + Nếu cả 3 lần đều fail → raise exception để caller dùng hardcode fallback. + """ import re - + max_attempts = 3 last_error = None - + for attempt in range(max_attempts): try: response = create_tracked_chat_completion( @@ -454,144 +644,138 @@ class SimulationConfigGenerator: {"role": "user", "content": prompt} ], response_format={"type": "json_object"}, - temperature=0.7 - (attempt * 0.1), # Giảm temperature cho mỗi lần retry + temperature=0.7 - (attempt * 0.1), # 0.7 → 0.6 → 0.5 metadata=self._runtime_metadata, ) - + content = response.choices[0].message.content finish_reason = response.choices[0].finish_reason - - # Kiểm tra nội dung trã về xem có phải bị chặn vì thiếu token (Length vượt qua max) hay không + + # Nếu bị cắt giữa chừng do max_tokens → thêm ngoặc đóng trước khi parse if finish_reason == 'length': logger.warning(f"LLM output was truncated (attempt {attempt+1})") content = self._fix_truncated_json(content) - - # Phân tích nội dung JSON + try: return json.loads(content) except json.JSONDecodeError as e: logger.warning(f"Failed to parse JSON (attempt {attempt+1}): {str(e)[:80]}") - - # Tiến hành sửa chữa nội dung JSON nếu bị lỗi + # Thử sửa JSON trước khi bỏ cuộc fixed = self._try_fix_config_json(content) if fixed: return fixed - last_error = e - + except Exception as e: logger.warning(f"Failed to call LLM (attempt {attempt+1}): {str(e)[:80]}") last_error = e import time - time.sleep(2 * (attempt + 1)) - + time.sleep(2 * (attempt + 1)) # Sleep 2s → 4s → (không có attempt 3+) + raise last_error or Exception("LLM connection completely failed") - + def _fix_truncated_json(self, content: str) -> str: - """Đóng dấu ngoặc JSON một cách an toàn cho các string bị cắt ngang""" + """ + Đóng các ngoặc JSON bị bỏ ngỏ khi LLM bị cắt giữa chừng. + + Thuật toán: + 1. Đếm số { chưa có } đóng tương ứng → thêm } vào cuối + 2. Đếm số [ chưa có ] đóng tương ứng → thêm ] vào cuối + 3. Nếu ký tự cuối là dở dang (không phải ",}]) → thêm " để đóng string + + Ví dụ input bị cắt: + '{"agent_configs": [{"agent_id": 0, "stance": "oppos' + Sau fix: + '{"agent_configs": [{"agent_id": 0, "stance": "oppos"}]}' + """ content = content.strip() - - # Đếm các dấu ngoặc mở bị bỏ sót chưa đóng + open_braces = content.count('{') - content.count('}') open_brackets = content.count('[') - content.count(']') - - # Đảm bảo các thuộc tính string đã được bọc đủ dấu ngoặc kép + + # Đóng string bị bỏ ngỏ trước khi đóng object/array if content and content[-1] not in '",}]': content += '"' - - # Thêm ngoặc đóng cho toàn bộ + content += ']' * open_brackets content += '}' * open_braces - + return content - + def _try_fix_config_json(self, content: str) -> Optional[Dict[str, Any]]: - """Cố gắng khôi phục, chắp ghép lại file cấu trúc config JSON""" + """ + Cố sửa JSON bị lỗi format bằng chuỗi bước: + + Bước 1: Gọi _fix_truncated_json() (đóng bracket thiếu) + Bước 2: Regex tìm khối {...} lớn nhất trong chuỗi (loại bỏ text thừa trước/sau) + Bước 3: Clean newline trong string values (LLM hay nhét \n trong string JSON) + Bước 4: Thử json.loads() — nếu OK trả về luôn + Bước 5: Xóa control characters (0x00-0x1F, 0x7F-0x9F) gây lỗi parse + Bước 6: Chuẩn hóa whitespace thừa + Bước 7: Thử json.loads() lần nữa — nếu vẫn fail trả về None + + Trả về None nếu không thể sửa → caller sẽ retry hoặc dùng fallback. + """ import re - - # Điền những dấu ngoặc vào chuỗi bị cắt + content = self._fix_truncated_json(content) - - # Regex ra đúng phần ruột nội dung JSON + json_match = re.search(r'\{[\s\S]*\}', content) if json_match: json_str = json_match.group() - - # Loại bỏ các đoạn tab, ngắt line cho string + + # Clean newline và whitespace thừa bên trong string values def fix_string(match): s = match.group(0) s = s.replace('\n', ' ').replace('\r', ' ') s = re.sub(r'\s+', ' ', s) return s - + json_str = re.sub(r'"[^"\\]*(?:\\.[^"\\]*)*"', fix_string, json_str) - + try: return json.loads(json_str) except: - # Tìm và xóa các control character + # Xóa control characters và thử lần nữa json_str = re.sub(r'[\x00-\x1f\x7f-\x9f]', ' ', json_str) json_str = re.sub(r'\s+', ' ', json_str) try: return json.loads(json_str) except: pass - + return None - + + # -------------------------------------------------------------------------- + # BƯỚC 1: Sinh cấu hình Thời gian + # -------------------------------------------------------------------------- + def _generate_time_config(self, context: str, num_entities: int) -> Dict[str, Any]: - """Tạo cấu hình thời gian (Time config) cho các tiến trình""" - # Áp dụng nội dung ngữ cảnh đã được giới hạn chiều dài + """ + Gọi LLM để sinh cấu hình thời gian mô phỏng. + + LLM nhận context cắt còn TIME_CONFIG_CONTEXT_LENGTH (10,000 ký tự) và trả về: + { + "total_simulation_hours": 72, + "minutes_per_round": 60, + "agents_per_hour_min": 5, + "agents_per_hour_max": 30, + "peak_hours": [19, 20, 21, 22], + "off_peak_hours": [0, 1, 2, 3, 4, 5], + "morning_hours": [6, 7, 8], + "work_hours": [9, 10, ..., 18], + "reasoning": "Giải thích tại sao chọn các thông số này" + } + + Ràng buộc gửi cho LLM: + - Người dùng là người Việt Nam → giờ Hà Nội (UTC+7) + - agents_per_hour tối đa = 90% số entity (max_agents_allowed = num_entities × 0.9) + + Fallback (LLM fail): _get_default_time_config() — hardcoded defaults. + """ context_truncated = context[:self.TIME_CONFIG_CONTEXT_LENGTH] - - # Cắt lấy số lượng Tối đa số lượng (Chiếm 80% từ số lượng lượng Agent thực thể) max_agents_allowed = max(1, int(num_entities * 0.9)) -# prompt = f"""Based on the following simulation requirements, generate a time simulation configuration. - -# {context_truncated} - -# ## Task -# Please generate a time configuration JSON. - -# ### General Principles (For reference only; adjust flexibly based on specific events and participant groups): -# - The user group consists of Vietnamese people; must comply with Hanoi Time (CST) daily routines. -# - 0:00–5:00 AM: Almost no activity (Activity Coefficient: 0.05). -# - 6:00–8:00 AM: Gradual increase in activity (Activity Coefficient: 0.4). -# - 9:00 AM–6:00 PM (Work hours): Moderate activity (Activity Coefficient: 0.7). -# - 7:00 PM–10:00 PM: Peak period (Activity Coefficient: 1.5). -# - After 11:00 PM: Activity declines (Activity Coefficient: 0.5). -# - General Pattern: Low activity in the early morning, gradual increase in the morning, moderate during work hours, and peak in the evening. -# - **Important:**: The example values below are for reference only. You need to adjust specific periods based on the nature of the event and characteristics of the participant group. -# - e.g., The peak for students might be 9:00 PM–11:00 PM; Media groups remain active all day; Official organizations only during work hours. -# - e.g., Breaking news may lead to discussions late at night; off_peak_hours can be shortened accordingly. - -# ### Return JSON Format (Do not use Markdown) - -# Example: -# {{ -# "total_simulation_hours": 72, -# "minutes_per_round": 60, -# "agents_per_hour_min": 5, -# "agents_per_hour_max": 50, -# "peak_hours": [19, 20, 21, 22], -# "off_peak_hours": [0, 1, 2, 3, 4, 5], -# "morning_hours": [6, 7, 8], -# "work_hours": [9, 10, 11, 12, 13, 14, 15, 16, 17, 18], -# "reasoning": "Time configuration explanation for this specific event." -# }} - -# Field Descriptions: -# - total_simulation_hours (int): Total simulation duration, 24–168 hours. Short for breaking news, long for sustained topics. -# - minutes_per_round (int): Duration per round, 30–120 minutes, suggested 60 minutes. -# - agents_per_hour_min (int): Minimum activated agents per hour (Range: 1-{max_agents_allowed}). -# - agents_per_hour_max (int): Maximum activated agents per hour (Range: 1-{max_agents_allowed}). -# - peak_hours (int array): Peak hours, adjusted based on the participant group. -# - off_peak_hours (int array): Off-peak hours, usually late night/early morning. -# - morning_hours (int array): Morning hours. -# - work_hours (int array): Working hours. -# - reasoning (string): Brief explanation of why this configuration was chosen.""" - prompt = f"""Dựa trên các yêu cầu mô phỏng dưới đây, hãy tạo cấu hình mô phỏng thời gian. {context_truncated} @@ -615,7 +799,7 @@ Hãy tạo JSON cấu hình thời gian. Ví dụ Format như sau: {{ - "total_simulation_hours": 72, + "total_simulation_hours": 72, "minutes_per_round": 60, "agents_per_hour_min": 5, "agents_per_hour_max": 50, @@ -636,22 +820,27 @@ Mô tả các trường: - morning_hours (mảng int): Khung giờ buổi sáng. - work_hours (mảng int): Khung giờ làm việc. - reasoning (string): Giải thích ngắn gọn lý do tại sao cấu hình như vậy.""" - - # system_prompt = "You are a social media simulation expert. Return in pure JSON format; time configurations must comply with Vietnamese daily routines." system_prompt = "Bạn là chuyên gia mô phỏng mạng xã hội. Trả về định dạng JSON thuần túy; cấu hình thời gian cần phù hợp với thói quen sinh hoạt của người Việt Nam." - + try: return self._call_llm_with_retry(prompt, system_prompt) except Exception as e: logger.warning(f"Failed to generate Time Config through LLM {e}. Returning the basic default rules...") return self._get_default_time_config(num_entities) - + def _get_default_time_config(self, num_entities: int) -> Dict[str, Any]: - """Tạo sẵn file chuẩn nếu bị đơ để trả ra theo múi giờ chuẩn sinh hoạt China""" + """ + Fallback hardcoded khi LLM fail hoàn toàn. + + agents_per_hour được tính theo công thức: + min = num_entities // 15 (ví dụ: 47 entity → min=3) + max = num_entities // 5 (ví dụ: 47 entity → max=9) + Đảm bảo không vượt quá tổng số agent thực tế. + """ return { "total_simulation_hours": 72, - "minutes_per_round": 60, # 1 Hour / Vòng -> Rút ngắn Time + "minutes_per_round": 60, "agents_per_hour_min": max(1, num_entities // 15), "agents_per_hour_max": max(5, num_entities // 5), "peak_hours": [19, 20, 21, 22], @@ -660,56 +849,87 @@ Mô tả các trường: "work_hours": [9, 10, 11, 12, 13, 14, 15, 16, 17, 18], "reasoning": "Defaults to Vietnamese users' daily routines and working hours (1 hour/round)" } - + def _parse_time_config(self, result: Dict[str, Any], num_entities: int) -> TimeSimulationConfig: - """Phân tích nội dung được định hình của JSON qua hàm parse kiểm tra, Xác nhận nếu lượng agents_per_hour vượt ngưỡng giới hạn """ - # Lấy giá trị chưa chỉnh sửa + """ + Parse dict từ LLM thành TimeSimulationConfig, kèm validation. + + Validation quan trọng: đảm bảo agents_per_hour không vượt tổng số agent thực tế. + Nếu LLM trả về agents_per_hour_max=100 nhưng chỉ có 47 entity, + OASIS sẽ cố chọn 100 agent nhưng không đủ → lỗi. + + Logic điều chỉnh: + agents_per_hour_min > num_entities → reset về num_entities // 10 + agents_per_hour_max > num_entities → reset về num_entities // 2 + min >= max → reset min về max // 2 + """ agents_per_hour_min = result.get("agents_per_hour_min", max(1, num_entities // 15)) agents_per_hour_max = result.get("agents_per_hour_max", max(5, num_entities // 5)) - - # Tiến hành kiểm tra xác minh: Đảm bảo độ lớn không lớn hơn con số Total Agent + if agents_per_hour_min > num_entities: logger.warning(f"agents_per_hour_min ({agents_per_hour_min}) exceeds total number of Agents ({num_entities}), corrected.") agents_per_hour_min = max(1, num_entities // 10) - + if agents_per_hour_max > num_entities: logger.warning(f"agents_per_hour_max ({agents_per_hour_max}) exceeds total number of Agents ({num_entities}), corrected.") agents_per_hour_max = max(agents_per_hour_min + 1, num_entities // 2) - - # Đảm bảo min luôn luôn nhỏ hơn max + if agents_per_hour_min >= agents_per_hour_max: agents_per_hour_min = max(1, agents_per_hour_max // 2) logger.warning(f"agents_per_hour_min >= max, modified to {agents_per_hour_min}") - + return TimeSimulationConfig( total_simulation_hours=result.get("total_simulation_hours", 72), - minutes_per_round=result.get("minutes_per_round", 60), # Mặc định mỗi vòng = 1 giờ + minutes_per_round=result.get("minutes_per_round", 60), agents_per_hour_min=agents_per_hour_min, agents_per_hour_max=agents_per_hour_max, peak_hours=result.get("peak_hours", [19, 20, 21, 22]), off_peak_hours=result.get("off_peak_hours", [0, 1, 2, 3, 4, 5]), - off_peak_activity_multiplier=0.05, # Gần như 0 mạng sáng rạng sáng + off_peak_activity_multiplier=0.05, morning_hours=result.get("morning_hours", [6, 7, 8]), morning_activity_multiplier=0.4, work_hours=result.get("work_hours", list(range(9, 19))), work_activity_multiplier=0.7, peak_activity_multiplier=1.5 ) - + + # -------------------------------------------------------------------------- + # BƯỚC 2: Sinh cấu hình Sự kiện + # -------------------------------------------------------------------------- + def _generate_event_config( - self, - context: str, + self, + context: str, simulation_requirement: str, entities: List[EntityNode] ) -> Dict[str, Any]: - """Tạo ra cho các thông số Event config""" - - # Tự liệt kê các Loại có thể xuất hiện để LLM tham khảo + """ + Gọi LLM để sinh các bài đăng khởi động và hướng phát triển dư luận. + + Ràng buộc quan trọng gửi cho LLM: poster_type phải thuộc danh sách + entity types thực sự có trong simulation. LLM được cung cấp danh sách + type_examples (ví dụ điển hình của mỗi loại) để chọn đúng. + + Ví dụ output: + { + "hot_topics": ["học phí", "biểu tình"], + "narrative_direction": "Thông báo tăng học phí → làn sóng phản đối...", + "initial_posts": [ + {"content": "CHÍNH THỨC: Đại học X tăng học phí...", "poster_type": "University"}, + {"content": "Không thể chấp nhận được! #HọcPhíTăng", "poster_type": "Student"} + ] + } + + Sau bước này, _assign_initial_post_agents() sẽ gán poster_agent_id thực. + Fallback khi LLM fail: hot_topics=[], initial_posts=[] — simulation vẫn chạy + nhưng không có "ngòi nổ" → agent tự tạo bài đăng theo activity_level. + """ + # Thu thập danh sách loại thực thể có thực để LLM chọn poster_type đúng entity_types_available = list(set( e.get_entity_type() or "Unknown" for e in entities )) - - # Ghi các Thực thể điển hình của mổi loại + + # Ví dụ điển hình mỗi loại (tối đa 3) — giúp LLM biết tên thật của entity type_examples = {} for e in entities: etype = e.get_entity_type() or "Unknown" @@ -717,46 +937,14 @@ Mô tả các trường: type_examples[etype] = [] if len(type_examples[etype]) < 3: type_examples[etype].append(e.name) - + type_info = "\n".join([ - f"- {t}: {', '.join(examples)}" + f"- {t}: {', '.join(examples)}" for t, examples in type_examples.items() ]) - - # Có chặn để lấy chuỗi theo cấu hình chiều dài giới hạn + context_truncated = context[:self.EVENT_CONFIG_CONTEXT_LENGTH] -# prompt = f"""Based on the following simulation requirements, generate an event configuration. - -# Simulation Requirements: {simulation_requirement} - -# {context_truncated} - -# ## Available Entity Types and Examples -# {type_info} - -# ## Task -# Please generate an event configuration JSON: -# - Extract key hot topic keywords. -# - Describe the direction of public opinion development. -# - Design initial post content; **each post must specify a poster_type (publisher type)**. - -# **IMPORTANT**: The poster_type must be selected from the "Available Entity Types" above so that initial posts can be assigned to the appropriate Agent for publishing. -# For example: Official statements should be posted by Official/University types, news by MediaOutlet, and student perspectives by Student. - -# Return in JSON format (no markdown): -# {{ -# "hot_topics": ["Keyword1", "Keyword2", ...], -# "narrative_direction": "", -# "initial_posts": [ -# {{"content": "Post Content...", "poster_type": "Entity type (must be selected from available types)"}}, -# ... -# ], -# "reasoning": "" -# }}""" - -# system_prompt = "You are a public opinion analysis expert. Return in pure JSON format. Ensure that poster_type exactly matches the available entity types." - prompt = f"""Dựa trên các yêu cầu mô phỏng sau đây, hãy tạo cấu hình sự kiện. Yêu cầu mô phỏng: {simulation_requirement} @@ -787,7 +975,7 @@ Trả về định dạng JSON (không sử dụng markdown): }}""" system_prompt = "Bạn là chuyên gia phân tích dư luận. Trả về định dạng JSON thuần túy. Lưu ý rằng poster_type phải khớp chính xác với các loại thực thể khả dụng." - + try: return self._call_llm_with_retry(prompt, system_prompt) except Exception as e: @@ -798,38 +986,56 @@ Trả về định dạng JSON (không sử dụng markdown): "initial_posts": [], "reasoning": "Sử dụng Config mặc định do LLM lỗi" } - + def _parse_event_config(self, result: Dict[str, Any]) -> EventConfig: - """Parse lấy các Thuộc Tính cấu hình Event""" + """Parse dict từ LLM thành EventConfig. Đơn giản — không có validation phức tạp.""" return EventConfig( initial_posts=result.get("initial_posts", []), - scheduled_events=[], + scheduled_events=[], # Chưa implement — dành cho tính năng tương lai hot_topics=result.get("hot_topics", []), narrative_direction=result.get("narrative_direction", "") ) - + def _assign_initial_post_agents( self, event_config: EventConfig, agent_configs: List[AgentActivityConfig] ) -> EventConfig: """ - Khớp quyền Agent với loại Poster_type cho các Bài Post đầu - - So sánh cho phù hợp của mỗi post để phân bố Agent id tối ưu nhất + Gán agent_id thực cho mỗi initial_post dựa trên poster_type. + + 3 tầng matching (ưu tiên từ cao đến thấp): + ┌─────────────────────────────────────────────────────────────────┐ + │ Tầng 1: Direct match │ + │ poster_type.lower() khớp chính xác key trong agents_by_type │ + │ Ví dụ: "student" → agents_by_type["student"][0] │ + ├─────────────────────────────────────────────────────────────────┤ + │ Tầng 2: Alias match │ + │ poster_type nằm trong danh sách alias của một key │ + │ Ví dụ: "media" → alias của "mediaoutlet" → dùng mediaoutlet │ + │ Cho phép LLM dùng tên biến thể mà vẫn map được đúng │ + ├─────────────────────────────────────────────────────────────────┤ + │ Tầng 3: Influence fallback │ + │ Không tìm được → lấy agent có influence_weight cao nhất │ + │ (Thường là mediaoutlet hoặc university) │ + └─────────────────────────────────────────────────────────────────┘ + + Lưu ý used_indices: mỗi loại có counter riêng để tránh dùng cùng 1 agent + nhiều lần khi có nhiều post cùng poster_type. + Ví dụ: 3 post "Student" → agent 0, 1, 2 (thay vì 0, 0, 0). """ if not event_config.initial_posts: return event_config - - # Build hệ thống agent index bằng kiểu loại + + # Index agents theo loại để tra cứu nhanh agents_by_type: Dict[str, List[AgentActivityConfig]] = {} for agent in agent_configs: etype = agent.entity_type.lower() if etype not in agents_by_type: agents_by_type[etype] = [] agents_by_type[etype].append(agent) - - # Bảng Alias ánh xạ tương đương (Cho phép LLM sử dụng nhiều quy ước format khác nhau) + + # Bảng alias — LLM đôi khi dùng tên khác nhau cho cùng một loại type_aliases = { "official": ["official", "university", "governmentagency", "government"], "university": ["university", "official"], @@ -840,26 +1046,24 @@ Trả về định dạng JSON (không sử dụng markdown): "organization": ["organization", "ngo", "company", "group"], "person": ["person", "student", "alumni"], } - - # Ghi chú từng loại agent đã dùng index nào, tránh dùng lại cùng 1 agent lặp đi lặp lại + + # Dùng round-robin trong mỗi loại (modulo len) để phân bổ đều used_indices: Dict[str, int] = {} - + updated_posts = [] for post in event_config.initial_posts: poster_type = post.get("poster_type", "").lower() content = post.get("content", "") - - # Khớp tìm agent phù hợp matched_agent_id = None - - # 1. Trùng khớp trực tiếp lấy luôn + + # Tầng 1: Direct match if poster_type in agents_by_type: agents = agents_by_type[poster_type] idx = used_indices.get(poster_type, 0) % len(agents) matched_agent_id = agents[idx].agent_id used_indices[poster_type] = idx + 1 else: - # 2. Sử dụng bí danh alias để khớp nếu dùng sai keyword + # Tầng 2: Alias match for alias_key, aliases in type_aliases.items(): if poster_type in aliases or alias_key == poster_type: for alias in aliases: @@ -871,28 +1075,31 @@ Trả về định dạng JSON (không sử dụng markdown): break if matched_agent_id is not None: break - - # 3. Nếu xui xẻo vẫn không tìm thấy, lấy thẳng Agent có điểm Influence (Sức ảnh hưởng) cao nhất + + # Tầng 3: Influence fallback if matched_agent_id is None: logger.warning(f"Could not find matching Agent type '{poster_type}', assigning to highest influence Agent instead") if agent_configs: - # Sort ảnh hưởng giảm dần, lấy index [0] sorted_agents = sorted(agent_configs, key=lambda a: a.influence_weight, reverse=True) matched_agent_id = sorted_agents[0].agent_id else: matched_agent_id = 0 - + updated_posts.append({ "content": content, "poster_type": post.get("poster_type", "Unknown"), "poster_agent_id": matched_agent_id }) - + logger.info(f"Initial post assignment: poster_type='{poster_type}' -> agent_id={matched_agent_id}") - + event_config.initial_posts = updated_posts return event_config - + + # -------------------------------------------------------------------------- + # BƯỚC 3..N: Sinh cấu hình Agent (chia batch) + # -------------------------------------------------------------------------- + def _generate_agent_configs_batch( self, context: str, @@ -900,9 +1107,22 @@ Trả về định dạng JSON (không sử dụng markdown): start_idx: int, simulation_requirement: str ) -> List[AgentActivityConfig]: - """Chia đợt gửi lên gọi tạo Cấu hình mạng lưới Agents""" - - # Build các node Entity (Dựa trên cấu hình lượng chữ giới hạn) + """ + Gọi LLM để sinh cấu hình hoạt động cho một batch agents (tối đa AGENTS_PER_BATCH=15). + + Lý do chia batch: nếu có 50+ entity, gửi tất cả 1 lần sẽ vượt token limit + và LLM có xu hướng bỏ sót entity cuối. Chia 15 agent/lần đảm bảo đủ. + + Mỗi entity trong batch được format thành: + {"agent_id": 5, "entity_name": "Trần Văn An", "entity_type": "Student", "summary": "..."} + + LLM trả về dict với key "agent_configs": [...] + → extract theo agent_id, map vào AgentActivityConfig object. + → Nếu LLM bỏ sót agent nào → _generate_agent_config_by_rule() bù vào. + + agent_id trong output phải khớp chính xác với start_idx + i + (để OASIS map đúng với user_id trong profile file). + """ entity_list = [] summary_len = self.AGENT_SUMMARY_LENGTH for i, e in enumerate(entities): @@ -912,44 +1132,6 @@ Trả về định dạng JSON (không sử dụng markdown): "entity_type": e.get_entity_type() or "Unknown", "summary": e.summary[:summary_len] if e.summary else "" }) - -# prompt = f"""Based on the following information, generate social media activity configurations for each entity. - -# Simulation Requirements: {simulation_requirement} - -# ## Entity List -# ```json -# {json.dumps(entity_list, ensure_ascii=False, indent=2)} -# ``` - -# ## Task -# Generate activity configurations for each entity, noting: -# - **Time aligns with Vietnamese daily routines**: Almost no activity between 0-5 AM; most active during 7-10 PM (19:00-22:00). -# - **Official Institutions (University/GovernmentAgency)**: Low activity (0.1-0.3), active during work hours (9:00-17:00), slow response (60-240 mins), high influence (2.5-3.0). -# - **Media (MediaOutlet)**: Medium activity (0.4-0.6), active all day (8:00-23:00), fast response (5-30 mins), high influence (2.0-2.5). -# - **Individuals (Student/Person/Alumni)**: High activity (0.6-0.9), active mainly in the evening (18:00-23:00), fast response (1-15 mins), low influence (0.8-1.2). -# - **Public Figures/Experts**: Medium activity (0.4-0.6), medium-high influence (1.5-2.0). - -# Return in JSON format (no markdown): -# {{ -# "agent_configs": [ -# {{ -# "agent_id": , -# "activity_level": <0.0-1.0>, -# "posts_per_hour": , -# "comments_per_hour": , -# "active_hours": [], -# "response_delay_min": , -# "response_delay_max": , -# "sentiment_bias": <-1.0 to 1.0>, -# "stance": "", -# "influence_weight": -# }}, -# ... -# ] -# }}""" - -# system_prompt = "You are a social media behavior analysis expert. Return pure JSON. Configurations must comply with Vietnamese daily routines." prompt = f"""Dựa trên các thông tin sau đây, hãy tạo cấu hình hoạt động trên mạng xã hội cho từng thực thể. @@ -988,24 +1170,25 @@ Trả về định dạng JSON (không sử dụng markdown): }}""" system_prompt = "Bạn là chuyên gia phân tích hành vi mạng xã hội. Trả về JSON thuần túy. Cấu hình phải phù hợp với thói quen sinh hoạt của người Việt Nam." - + try: result = self._call_llm_with_retry(prompt, system_prompt) + # Index theo agent_id để tra cứu O(1) khi fill vào từng entity llm_configs = {cfg["agent_id"]: cfg for cfg in result.get("agent_configs", [])} except Exception as e: logger.warning(f"Failed LLM generating Agent batch configs: {e}, falling back to default manual rules.") llm_configs = {} - - # Tạo object list cho AgentActivityConfig + + # Tạo AgentActivityConfig cho mỗi entity trong batch configs = [] for i, entity in enumerate(entities): agent_id = start_idx + i cfg = llm_configs.get(agent_id, {}) - - # Gán Manual tự động nếu Bot LLM thiếu xót + + # Nếu LLM bỏ sót agent này → dùng rule-based fallback if not cfg: cfg = self._generate_agent_config_by_rule(entity) - + config = AgentActivityConfig( agent_id=agent_id, entity_uuid=entity.uuid, @@ -1022,20 +1205,35 @@ Trả về định dạng JSON (không sử dụng markdown): influence_weight=cfg.get("influence_weight", 1.0) ) configs.append(config) - + return configs - + def _generate_agent_config_by_rule(self, entity: EntityNode) -> Dict[str, Any]: - """Tự động gen cấu hình 1 người (agent) dựa trên bộ rule cứng có sẵn nếu gọi bot LLM bị fail (Luật theo múi giờ sinh học)""" + """ + Sinh cấu hình agent bằng rule cứng khi LLM fail hoặc bỏ sót entity này. + + Rule được xây dựng dựa trên hành vi thực tế trên mạng xã hội Việt Nam: + + ┌──────────────────────┬──────────┬──────────────────┬───────────┬───────────┐ + │ Loại entity │ activity │ active_hours │ delay(ph) │ influence │ + ├──────────────────────┼──────────┼──────────────────┼───────────┼───────────┤ + │ university/gov/ngo │ 0.2 │ 9-17 (hành chính)│ 60-240 │ 3.0 │ + │ mediaoutlet │ 0.5 │ 7-23 (cả ngày) │ 5-30 │ 2.5 │ + │ professor/expert │ 0.4 │ 8-21 │ 15-90 │ 2.0 │ + │ student │ 0.8 │ sáng + tối │ 1-15 │ 0.8 │ + │ alumni │ 0.6 │ trưa + tối │ 5-30 │ 1.0 │ + │ (default) │ 0.7 │ 9-23 │ 2-20 │ 1.0 │ + └──────────────────────┴──────────┴──────────────────┴───────────┴───────────┘ + """ entity_type = (entity.get_entity_type() or "Unknown").lower() - + if entity_type in ["university", "governmentagency", "ngo"]: - # Cơ quan chức năng Nhà nước / Doanh nghiệp: làm việc trong khung giờ chuẩn hành chính, trả lời ít nhưng nặng đô + # Cơ quan nhà nước/tổ chức: hoạt động giờ hành chính, phát ngôn ít nhưng ảnh hưởng lớn return { "activity_level": 0.2, "posts_per_hour": 0.1, "comments_per_hour": 0.05, - "active_hours": list(range(9, 18)), # 9:00-17:59 + "active_hours": list(range(9, 18)), "response_delay_min": 60, "response_delay_max": 240, "sentiment_bias": 0.0, @@ -1043,12 +1241,12 @@ Trả về định dạng JSON (không sử dụng markdown): "influence_weight": 3.0 } elif entity_type in ["mediaoutlet"]: - # Báo đài truyền thông: cả ngày đưa tin, ra bài lẹ giật tít, tốc độ cao + # Báo đài: hoạt động cả ngày, đưa tin nhanh, nhiều người theo dõi return { "activity_level": 0.5, "posts_per_hour": 0.8, "comments_per_hour": 0.3, - "active_hours": list(range(7, 24)), # 7:00-23:59 + "active_hours": list(range(7, 24)), "response_delay_min": 5, "response_delay_max": 30, "sentiment_bias": 0.0, @@ -1056,12 +1254,12 @@ Trả về định dạng JSON (không sử dụng markdown): "influence_weight": 2.5 } elif entity_type in ["professor", "expert", "official"]: - # Giáo sư đại học/Người phát biểu: Chỉ nói ban ngày và tối, ra bài ít + # Chuyên gia/giảng viên: phát biểu có chọn lọc, ban ngày và tối sớm return { "activity_level": 0.4, "posts_per_hour": 0.3, "comments_per_hour": 0.5, - "active_hours": list(range(8, 22)), # 8:00-21:59 + "active_hours": list(range(8, 22)), "response_delay_min": 15, "response_delay_max": 90, "sentiment_bias": 0.0, @@ -1069,12 +1267,12 @@ Trả về định dạng JSON (không sử dụng markdown): "influence_weight": 2.0 } elif entity_type in ["student"]: - # Tần suất cho lứa Sinh viên: hay ra bài / cãi nhau liên tục ban đêm rất nhiều + # Sinh viên: rất tích cực, đặc biệt buổi tối, phản ứng nhanh return { "activity_level": 0.8, "posts_per_hour": 0.6, "comments_per_hour": 1.5, - "active_hours": [8, 9, 10, 11, 12, 13, 18, 19, 20, 21, 22, 23], # Sáng + Đêm Tối + "active_hours": [8, 9, 10, 11, 12, 13, 18, 19, 20, 21, 22, 23], "response_delay_min": 1, "response_delay_max": 15, "sentiment_bias": 0.0, @@ -1082,12 +1280,12 @@ Trả về định dạng JSON (không sử dụng markdown): "influence_weight": 0.8 } elif entity_type in ["alumni"]: - # Cựu sinh viên: Thường online đêm là chính + # Cựu sinh viên: online giờ nghỉ trưa + buổi tối sau giờ làm return { "activity_level": 0.6, "posts_per_hour": 0.4, "comments_per_hour": 0.8, - "active_hours": [12, 13, 19, 20, 21, 22, 23], # Giờ nghỉ trưa + Buổi tối + "active_hours": [12, 13, 19, 20, 21, 22, 23], "response_delay_min": 5, "response_delay_max": 30, "sentiment_bias": 0.0, @@ -1095,17 +1293,15 @@ Trả về định dạng JSON (không sử dụng markdown): "influence_weight": 1.0 } else: - # Thuộc cho số đông (Cư dân mạng / Người Qua Đường): Phấn khích về đêm + # Default — cư dân mạng thông thường: hoạt động ban ngày + tối return { "activity_level": 0.7, "posts_per_hour": 0.5, "comments_per_hour": 1.2, - "active_hours": [9, 10, 11, 12, 13, 18, 19, 20, 21, 22, 23], # Ban Ngày rảnh + Buổi tối rảnh + "active_hours": [9, 10, 11, 12, 13, 18, 19, 20, 21, 22, 23], "response_delay_min": 2, "response_delay_max": 20, "sentiment_bias": 0.0, "stance": "neutral", "influence_weight": 1.0 } - - diff --git a/backend/app/services/simulation_ipc.py b/backend/app/services/simulation_ipc.py index afddff04..4c7b6eed 100644 --- a/backend/app/services/simulation_ipc.py +++ b/backend/app/services/simulation_ipc.py @@ -137,7 +137,7 @@ class SimulationIPCClient: TimeoutError: Lỗi quá thời gian chờ phản hồi """ command_id = str(uuid.uuid4()) - command = IPCCommand( + command = s( command_id=command_id, command_type=command_type, args=args diff --git a/backend/app/services/simulation_runner.py b/backend/app/services/simulation_runner.py index bc5193fd..baf2421e 100644 --- a/backend/app/services/simulation_runner.py +++ b/backend/app/services/simulation_runner.py @@ -1,6 +1,28 @@ """ -OASIS模拟运行器 -在后台运行模拟并记录每个Agent的动作,支持实时状态监控 +Trình chạy và giám sát Simulation OASIS + +Vị trí trong pipeline: +───────────────────────────────────────────────────────────────────────────── + simulation_manager.py → SimulationRunner.start_simulation() + └─ subprocess.Popen(run_parallel_simulation.py --config ...) + │ (OASIS chạy ngầm, ghi log ra actions.jsonl) + │ + └─ Thread(_monitor_simulation) [daemon thread, chạy song song với Flask] + └─ _read_action_log() [đọc actions.jsonl mỗi 2 giây] + └─ _save_run_state() [cập nhật run_state.json liên tục] +───────────────────────────────────────────────────────────────────────────── + +Hai file state riêng biệt (KHÔNG phải một): + state.json — lifecycle state, quản lý bởi SimulationManager + (CREATED → PREPARING → READY → RUNNING → COMPLETED/FAILED) + run_state.json — runtime progress, quản lý bởi SimulationRunner + (round hiện tại, action count, PID, ...) + Cập nhật mỗi 2 giây trong khi OASIS đang chạy. + +Giao tiếp Flask ↔ OASIS subprocess: + File-based: Flask đọc actions.jsonl để biết tiến độ + IPC: Flask ghi lệnh vào ipc_commands/, OASIS trả lời vào ipc_responses/ + (dùng cho tính năng Interview agent đang chạy) """ import os @@ -25,38 +47,61 @@ from .simulation_ipc import SimulationIPCClient, CommandType, IPCResponse logger = get_logger('mirofish.simulation_runner') -# Cờ đánh dấu đã đăng ký hàm dọn dẹp hay chưa +# Cờ đánh dấu đã đăng ký hàm dọn dẹp (atexit) hay chưa — tránh đăng ký nhiều lần _cleanup_registered = False -# Kiểm tra hệ điều hành +# Kiểm tra hệ điều hành — dùng để chọn cách kill process (killpg vs taskkill) IS_WINDOWS = sys.platform == 'win32' +# ============================================================================== +# ENUM: RunnerStatus — Trạng thái của bộ chạy tiến trình (khác với SimulationStatus) +# ============================================================================== +# RunnerStatus theo dõi trạng thái của OASIS subprocess, không phải lifecycle tổng thể. +# Được lưu trong run_state.json — SimulationManager đọc file này để cập nhật state.json. +# +# Luồng trạng thái: +# IDLE → STARTING → RUNNING → COMPLETED +# └──────────→ FAILED (subprocess crash) +# RUNNING → STOPPING → STOPPED (người dùng stop) +# ============================================================================== + class RunnerStatus(str, Enum): """Trạng thái của bộ chạy tiến trình mô phỏng""" - IDLE = "idle" # Rảnh rỗi, chưa chạy - STARTING = "starting" # Đang khởi động - RUNNING = "running" # Đang chạy - PAUSED = "paused" # Đã tạm dừng - STOPPING = "stopping" # Đang dừng lại - STOPPED = "stopped" # Đã dừng - COMPLETED = "completed" # Đã hoàn thành - FAILED = "failed" # Bị lỗi + IDLE = "idle" # Rảnh rỗi, chưa chạy + STARTING = "starting" # Đang khởi động subprocess + RUNNING = "running" # Subprocess đang chạy, agents đang hành động + PAUSED = "paused" # Đã tạm dừng (tính năng tương lai) + STOPPING = "stopping" # Đang gửi SIGTERM / taskkill + STOPPED = "stopped" # Đã dừng theo yêu cầu người dùng + COMPLETED = "completed" # OASIS chạy hết số vòng và thoát thành công (exit_code=0) + FAILED = "failed" # Subprocess crash (exit_code != 0) hoặc exception không xử lý được +# ============================================================================== +# DATACLASS: AgentAction — Bản ghi một hành động đơn lẻ từ actions.jsonl +# ============================================================================== +# Mỗi dòng trong actions.jsonl tương ứng với 1 AgentAction. +# _read_action_log() parse từng dòng JSON thành AgentAction và lưu vào SimulationRunState. +# +# Ví dụ 1 dòng trong twitter/actions.jsonl: +# {"round": 15, "agent_id": 3, "agent_name": "tran_van_an_492", +# "action_type": "CREATE_POST", "action_args": {"content": "..."}, "success": true} +# ============================================================================== + @dataclass class AgentAction: """Bản ghi hành động của Agent""" - round_num: int # Số thứ tự của vòng (round) mô phỏng + round_num: int # Số thứ tự vòng (round) mô phỏng timestamp: str # Dấu thời gian - platform: str # Nền tảng thực hiện: twitter / reddit - agent_id: int # ID của agent - agent_name: str # Tên của agent - action_type: str # Loại hành động: CREATE_POST, LIKE_POST, v.v. - action_args: Dict[str, Any] = field(default_factory=dict) # Tham số của hành động - result: Optional[str] = None # Kết quả thực thi + platform: str # Nền tảng: "twitter" hoặc "reddit" + agent_id: int # ID của agent (khớp với user_id trong profile file) + agent_name: str # Tên tài khoản (ví dụ: "tran_van_an_492") + action_type: str # Loại hành động: CREATE_POST, LIKE_POST, REPOST, FOLLOW, DO_NOTHING... + action_args: Dict[str, Any] = field(default_factory=dict) # Tham số của hành động + result: Optional[str] = None # Kết quả thực thi (ví dụ: "Post created with id 142") success: bool = True # Hành động có thành công hay không - + def to_dict(self) -> Dict[str, Any]: return { "round_num": self.round_num, @@ -71,18 +116,25 @@ class AgentAction: } +# ============================================================================== +# DATACLASS: RoundSummary — Tóm tắt một vòng (round) mô phỏng +# ============================================================================== +# Được tổng hợp từ các AgentAction trong cùng round_num. +# Lưu vào SimulationRunState.rounds để hiển thị timeline trên frontend. +# ============================================================================== + @dataclass class RoundSummary: """Tóm tắt thông tin của mỗi vòng (round)""" - round_num: int # Số thứ tự vòng - start_time: str # Thời gian bắt đầu - end_time: Optional[str] = None # Thời gian kết thúc - simulated_hour: int = 0 # Số giờ đã mô phỏng trong vòng này - twitter_actions: int = 0 # Số hành động trên Twitter - reddit_actions: int = 0 # Số hành động trên Reddit - active_agents: List[int] = field(default_factory=list) # Danh sách ID các agent đang hoạt động - actions: List[AgentAction] = field(default_factory=list) # Danh sách các hành động - + round_num: int # Số thứ tự vòng + start_time: str # Thời gian bắt đầu vòng + end_time: Optional[str] = None # Thời gian kết thúc vòng + simulated_hour: int = 0 # Giờ mô phỏng tương ứng (0–71 nếu total_hours=72) + twitter_actions: int = 0 # Số hành động trên Twitter trong vòng này + reddit_actions: int = 0 # Số hành động trên Reddit trong vòng này + active_agents: List[int] = field(default_factory=list) # Danh sách agent_id đã hành động + actions: List[AgentAction] = field(default_factory=list) # Chi tiết các hành động + def to_dict(self) -> Dict[str, Any]: return { "round_num": self.round_num, @@ -97,66 +149,86 @@ class RoundSummary: } +# ============================================================================== +# DATACLASS: SimulationRunState — Trạng thái runtime của OASIS subprocess +# ============================================================================== +# Đây là object lưu trạng thái "đang chạy" — được cập nhật mỗi 2 giây bởi monitor thread. +# Được ghi ra run_state.json để Frontend có thể poll API và hiển thị tiến độ. +# +# Phân biệt với SimulationState (state.json): +# SimulationState = lifecycle tổng thể (CREATED→READY→RUNNING→COMPLETED) +# SimulationRunState = tiến độ chi tiết (round hiện tại, action count, PID, ...) +# +# recent_actions: chỉ giữ 50 hành động gần nhất (max_recent_actions) +# → Không lưu toàn bộ actions vào RAM để tránh tốn bộ nhớ với simulation dài. +# ============================================================================== + @dataclass class SimulationRunState: """Trạng thái đang thực thi của tiến trình mô phỏng (cập nhật theo thời gian thực)""" simulation_id: str runner_status: RunnerStatus = RunnerStatus.IDLE - - # Thông tin tiến độ - current_round: int = 0 - total_rounds: int = 0 - simulated_hours: int = 0 - total_simulation_hours: int = 0 - - # Các vòng lặp và thời gian độc lập cho từng nền tảng (sử dụng để hiển thị song song hai nền tảng) + + # --- Tiến độ chung --- + current_round: int = 0 # Vòng hiện tại (số lớn nhất của 2 platform) + total_rounds: int = 0 # Tổng số vòng cần chạy + simulated_hours: int = 0 # Số giờ đã mô phỏng + total_simulation_hours: int = 0 # Tổng số giờ cần mô phỏng + + # --- Tiến độ riêng từng platform --- + # Hai platform chạy song song asyncio → hoàn thành không đồng bộ + # Frontend hiển thị 2 progress bar riêng dựa trên các field này twitter_current_round: int = 0 reddit_current_round: int = 0 twitter_simulated_hours: int = 0 reddit_simulated_hours: int = 0 - - # Trạng thái nền tảng đang chạy - twitter_running: bool = False - reddit_running: bool = False - twitter_actions_count: int = 0 - reddit_actions_count: int = 0 - - # Trạng thái hoàn thành chung của nền tảng (phát hiện qua sự kiện simulation_end trong actions.jsonl) + + # --- Trạng thái nền tảng --- + twitter_running: bool = False # True khi OASIS Twitter đang chạy + reddit_running: bool = False # True khi OASIS Reddit đang chạy + twitter_actions_count: int = 0 # Tổng hành động Twitter đã ghi được + reddit_actions_count: int = 0 # Tổng hành động Reddit đã ghi được + + # Phát hiện qua event_type="simulation_end" trong actions.jsonl + # (KHÔNG dựa vào exit_code của subprocess — vì exit_code chỉ biết khi process kết thúc) twitter_completed: bool = False reddit_completed: bool = False - - # Tóm tắt lại ở mỗi vòng + + # --- Lịch sử vòng --- rounds: List[RoundSummary] = field(default_factory=list) - - # Các hành động gần nhất (để hiển thị theo thời gian thực (real-time) trên frontend) + + # --- Hành động gần nhất (tối đa 50) --- + # insert(0, ...) → luôn giữ mới nhất ở đầu list recent_actions: List[AgentAction] = field(default_factory=list) max_recent_actions: int = 50 - - # Dấu thời gian + + # --- Timestamps --- started_at: Optional[str] = None updated_at: str = field(default_factory=lambda: datetime.now().isoformat()) completed_at: Optional[str] = None - - # Thông tin lỗi + + # --- Lỗi --- error: Optional[str] = None - - # ID tiến trình (PID) (để dừng/hủy tiến trình) + + # --- Process ID --- + # Lưu lại để có thể kill process khi cần dừng process_pid: Optional[int] = None - + def add_action(self, action: AgentAction): - """Thêm một hành động vào danh sách các hành động gần nhất""" + """Thêm hành động vào đầu recent_actions, giữ tối đa max_recent_actions.""" self.recent_actions.insert(0, action) if len(self.recent_actions) > self.max_recent_actions: self.recent_actions = self.recent_actions[:self.max_recent_actions] - + if action.platform == "twitter": self.twitter_actions_count += 1 else: self.reddit_actions_count += 1 - + self.updated_at = datetime.now().isoformat() - + def to_dict(self) -> Dict[str, Any]: + """Serialize ra dict gọn — dùng cho API response và ghi run_state.json.""" return { "simulation_id": self.simulation_id, "runner_status": self.runner_status.value, @@ -164,8 +236,8 @@ class SimulationRunState: "total_rounds": self.total_rounds, "simulated_hours": self.simulated_hours, "total_simulation_hours": self.total_simulation_hours, + # Phần trăm hoàn thành: tránh ZeroDivisionError khi total_rounds=0 "progress_percent": round(self.current_round / max(self.total_rounds, 1) * 100, 1), - # Vòng lặp và thời gian độc lập cho mỗi nền tảng "twitter_current_round": self.twitter_current_round, "reddit_current_round": self.reddit_current_round, "twitter_simulated_hours": self.twitter_simulated_hours, @@ -183,72 +255,103 @@ class SimulationRunState: "error": self.error, "process_pid": self.process_pid, } - + def to_detail_dict(self) -> Dict[str, Any]: - """Chi tiết thông tin bao gồm các hành động gần nhất""" + """Serialize đầy đủ bao gồm recent_actions — dùng để ghi run_state.json.""" result = self.to_dict() result["recent_actions"] = [a.to_dict() for a in self.recent_actions] result["rounds_count"] = len(self.rounds) return result +# ============================================================================== +# CLASS: SimulationRunner — Trình chạy và giám sát OASIS subprocess +# ============================================================================== +# SimulationRunner là class-only (tất cả methods đều là @classmethod) — dùng như singleton. +# Không cần khởi tạo instance, gọi trực tiếp: SimulationRunner.start_simulation(...) +# +# Tại sao class-level dict thay vì instance dict? +# Flask chạy nhiều request đồng thời → nhiều thread cùng truy cập runner state. +# Class-level dict được chia sẻ giữa tất cả thread trong cùng Flask process. +# → Không cần singleton pattern phức tạp, class dict là đủ. +# +# _processes: simulation_id → subprocess.Popen (để kill khi cần) +# _run_states: simulation_id → SimulationRunState (RAM cache của run_state.json) +# _monitor_threads: simulation_id → Thread (daemon thread đọc actions.jsonl) +# _stdout_files: simulation_id → file handle (để đóng khi process kết thúc) +# ============================================================================== + class SimulationRunner: """ Trình chạy mô phỏng - + Quy trách nhiệm: 1. Chạy mô phỏng OASIS trong tiến trình nền (background process) 2. Phân tích nhật ký chạy (log), ghi lại hành động của mỗi Agent 3. Cung cấp API truy vấn trạng thái thời gian thực 4. Hỗ trợ thao tác tạm dừng (pause)/dừng (stop)/tiếp tục (resume) """ - - # Thư mục lưu trữ trạng thái chạy + + # Thư mục lưu trữ trạng thái chạy (mỗi simulation có subfolder riêng) RUN_STATE_DIR = os.path.join( os.path.dirname(__file__), '../../uploads/simulations' ) - - # Thư mục chứa các script con (script chạy ứng dụng) + + # Thư mục chứa các OASIS script (run_parallel_simulation.py, ...) + # Scripts không được copy vào thư mục simulation — gọi thẳng từ đây SCRIPTS_DIR = os.path.join( os.path.dirname(__file__), '../../scripts' ) - - # Trạng thái chạy trong bộ nhớ Memory (RAM) + + # --- Class-level state (chia sẻ giữa tất cả request threads) --- _run_states: Dict[str, SimulationRunState] = {} _processes: Dict[str, subprocess.Popen] = {} _action_queues: Dict[str, Queue] = {} _monitor_threads: Dict[str, threading.Thread] = {} - _stdout_files: Dict[str, Any] = {} # Lưu trữ tay cầm file đầu ra chuẩn (stdout) - _stderr_files: Dict[str, Any] = {} # Lưu trữ tay cầm file lỗi chuẩn (stderr) - - # Cấu hình cập nhật bộ nhớ Đồ thị (Graph Memory) - _graph_memory_enabled: Dict[str, bool] = {} # simulation_id -> enabled (Bật/tắt) - + _stdout_files: Dict[str, Any] = {} # Lưu file handle để đóng khi process kết thúc + _stderr_files: Dict[str, Any] = {} # Không dùng riêng (stderr merge vào stdout) + + # Cấu hình Graph Memory Update (tính năng tùy chọn: ghi hành động agent vào Zep) + _graph_memory_enabled: Dict[str, bool] = {} # simulation_id → True/False + @classmethod def get_run_state(cls, simulation_id: str) -> Optional[SimulationRunState]: - """Lấy trạng thái chạy hiện tại""" + """ + Lấy trạng thái runtime hiện tại — cache-aside (RAM trước, disk fallback). + Trả về None nếu simulation chưa được start. + """ if simulation_id in cls._run_states: return cls._run_states[simulation_id] - - # Thử tải từ file nếu không có trong memory + + # Không có trong RAM → thử load từ run_state.json state = cls._load_run_state(simulation_id) if state: cls._run_states[simulation_id] = state return state - + @classmethod def _load_run_state(cls, simulation_id: str) -> Optional[SimulationRunState]: - """Tải trạng thái chạy từ tệp tin (run_state.json)""" + """ + Load SimulationRunState từ run_state.json trên disk. + + Được gọi khi: + 1. Server restart → RAM cache trống → cần load lại từ disk + 2. Lần đầu gọi get_run_state() sau khi start_simulation() + + Lưu ý: nếu run_state.json có status="running" sau server restart, + đó là "false positive" — process đã chết cùng server. Phải gọi + cleanup_simulation_logs() trước khi start lại. + """ state_file = os.path.join(cls.RUN_STATE_DIR, simulation_id, "run_state.json") if not os.path.exists(state_file): return None - + try: with open(state_file, 'r', encoding='utf-8') as f: data = json.load(f) - + state = SimulationRunState( simulation_id=simulation_id, runner_status=RunnerStatus(data.get("runner_status", "idle")), @@ -256,7 +359,6 @@ class SimulationRunner: total_rounds=data.get("total_rounds", 0), simulated_hours=data.get("simulated_hours", 0), total_simulation_hours=data.get("total_simulation_hours", 0), - # Các vòng lặp và thời gian độc lập cho mỗi nền tảng twitter_current_round=data.get("twitter_current_round", 0), reddit_current_round=data.get("reddit_current_round", 0), twitter_simulated_hours=data.get("twitter_simulated_hours", 0), @@ -273,8 +375,8 @@ class SimulationRunner: error=data.get("error"), process_pid=data.get("process_pid"), ) - - # Tải danh sách các hành động gần đây + + # Restore 50 hành động gần nhất (để hiển thị lại sau restart) actions_data = data.get("recent_actions", []) for a in actions_data: state.recent_actions.append(AgentAction( @@ -288,76 +390,99 @@ class SimulationRunner: result=a.get("result"), success=a.get("success", True), )) - + return state except Exception as e: logger.error(f"Failed to load run state: {str(e)}") return None - + @classmethod def _save_run_state(cls, state: SimulationRunState): - """Lưu trạng thái chạy vào file""" + """ + Ghi SimulationRunState ra run_state.json VÀ cập nhật RAM cache. + + Được gọi: + 1. Sau mỗi lần đọc actions.jsonl trong monitor thread (mỗi 2 giây) + 2. Khi trạng thái thay đổi (STARTING → RUNNING → COMPLETED/FAILED) + 3. Khi process kết thúc (cleanup trong finally block) + """ sim_dir = os.path.join(cls.RUN_STATE_DIR, state.simulation_id) os.makedirs(sim_dir, exist_ok=True) state_file = os.path.join(sim_dir, "run_state.json") - + + # to_detail_dict() bao gồm recent_actions — đầy đủ hơn to_dict() data = state.to_detail_dict() - + with open(state_file, 'w', encoding='utf-8') as f: json.dump(data, f, ensure_ascii=False, indent=2) - + cls._run_states[state.simulation_id] = state - + + # -------------------------------------------------------------------------- + # PUBLIC: start_simulation — Khởi động OASIS subprocess + monitor thread + # -------------------------------------------------------------------------- + @classmethod def start_simulation( cls, simulation_id: str, - platform: str = "parallel", # twitter / reddit / parallel - max_rounds: int = None, # Số vòng mô phỏng tối đa (tùy chọn, dùng để cắt ngắn các mô phỏng quá dài) - enable_graph_memory_update: bool = False, # Có liên tục cập nhật hoạt động của Agent vào Zep graph hay không - graph_id: str = None # ID của Zep graph (Bắt buộc nếu bật tính năng cập nhật sơ đồ (graph)) + platform: str = "parallel", # "twitter" / "reddit" / "parallel" + max_rounds: int = None, # Giới hạn số vòng tối đa (None = không giới hạn) + enable_graph_memory_update: bool = False, # Ghi hành động agent vào Zep Graph + graph_id: str = None # Bắt buộc nếu enable_graph_memory_update=True ) -> SimulationRunState: """ - Bắt đầu mô phỏng - + Khởi động simulation: + 1. Đọc simulation_config.json → tính total_rounds + 2. subprocess.Popen(run_parallel_simulation.py --config ...) → OASIS chạy ngầm + 3. Thread(_monitor_simulation) → daemon thread đọc actions.jsonl mỗi 2 giây + + Sau khi hàm này return, OASIS đang chạy trong background. + Flask vẫn nhận request bình thường — không bị block. + + start_new_session=True: tạo process group mới → có thể kill toàn bộ cây process + bằng os.killpg() (thay vì chỉ kill process cha). + Args: - simulation_id: ID mô phỏng - platform: Nền tảng chạy (twitter/reddit/parallel) - max_rounds: Số vòng chạy tối đa (để cắt bớt) - enable_graph_memory_update: Có cập nhật hành vi Agent vào Zep Graph hay không - graph_id: Zep Graph ID - + simulation_id: ID của simulation đã qua prepare (status=READY) + platform: Nền tảng chạy ("twitter"/"reddit"/"parallel") + max_rounds: Cắt ngắn số vòng nếu chỉ muốn chạy thử + enable_graph_memory_update: Ghi lại hành động agent vào Zep Graph theo real-time + graph_id: Zep Graph ID (bắt buộc khi enable_graph_memory_update=True) + Returns: - SimulationRunState (Trạng thái sau khi cấu hình) + SimulationRunState với runner_status=RUNNING """ - # Kiểm tra xem có tiến trình nào đang chạy không + # Kiểm tra không cho start khi đã đang chạy existing = cls.get_run_state(simulation_id) if existing and existing.runner_status in [RunnerStatus.RUNNING, RunnerStatus.STARTING]: raise ValueError(f"Simulation is already running: {simulation_id}") - - # Tải cấu hình mô phỏng + + # Đọc simulation_config.json để tính total_rounds sim_dir = os.path.join(cls.RUN_STATE_DIR, simulation_id) config_path = os.path.join(sim_dir, "simulation_config.json") - + if not os.path.exists(config_path): raise ValueError(f"Simulation configuration not found, please call the /prepare API first") - + with open(config_path, 'r', encoding='utf-8') as f: config = json.load(f) - - # Khởi tạo trạng thái chạy + + # Tính total_rounds từ time_config time_config = config.get("time_config", {}) total_hours = time_config.get("total_simulation_hours", 72) minutes_per_round = time_config.get("minutes_per_round", 30) total_rounds = int(total_hours * 60 / minutes_per_round) - - # Nếu chỉ định maximum rounds, tiến hành việc cắt bớt + # Ví dụ: 72h × 60min / 60min/round = 72 rounds + + # Áp dụng giới hạn max_rounds nếu có (dùng để test chạy nhanh) if max_rounds is not None and max_rounds > 0: original_rounds = total_rounds total_rounds = min(total_rounds, max_rounds) if total_rounds < original_rounds: logger.info(f"Rounds truncated: {original_rounds} -> {total_rounds} (max_rounds={max_rounds})") - + + # Khởi tạo run state ban đầu với status=STARTING state = SimulationRunState( simulation_id=simulation_id, runner_status=RunnerStatus.STARTING, @@ -365,14 +490,14 @@ class SimulationRunner: total_simulation_hours=total_hours, started_at=datetime.now().isoformat(), ) - + cls._save_run_state(state) - - # Nếu tính năng cập nhật bộ nhớ graph được bật, tạo một updater + + # Khởi tạo Graph Memory Updater nếu được bật if enable_graph_memory_update: if not graph_id: raise ValueError("graph_id is required to enable graph memory updates") - + try: ZepGraphMemoryManager.create_updater(simulation_id, graph_id) cls._graph_memory_enabled[simulation_id] = True @@ -382,8 +507,8 @@ class SimulationRunner: cls._graph_memory_enabled[simulation_id] = False else: cls._graph_memory_enabled[simulation_id] = False - - # Xác định script nào sẽ chạy (các script nằm trong thư mục backend/scripts/) + + # Chọn script OASIS phù hợp với platform if platform == "twitter": script_name = "run_twitter_simulation.py" state.twitter_running = True @@ -391,71 +516,63 @@ class SimulationRunner: script_name = "run_reddit_simulation.py" state.reddit_running = True else: + # "parallel" — chạy cả hai platform đồng thời (asyncio bên trong script) script_name = "run_parallel_simulation.py" state.twitter_running = True state.reddit_running = True - + script_path = os.path.join(cls.SCRIPTS_DIR, script_name) - + if not os.path.exists(script_path): raise ValueError(f"Script no longer exists: {script_path}") - - # Tạo hàng đợi các hành động (Queue) + + # Tạo action queue (hiện chưa dùng nhưng giữ lại cho tương lai) action_queue = Queue() cls._action_queues[simulation_id] = action_queue - - # Bắt đầu chạy tiến trình mô phỏng + + # Khởi động OASIS subprocess try: - # Xây dựng lệnh chạy, sử dụng full path - # Cấu trúc log mới: - # twitter/actions.jsonl - Log cho các hành động trên Twitter - # reddit/actions.jsonl - Log cho các hành động trên Reddit - # simulation.log - Log cho tiến trình chính - cmd = [ - sys.executable, # Python Interpreter + sys.executable, # Python interpreter hiện tại script_path, - "--config", config_path, # Use full path to config + "--config", config_path, ] - - # Nếu có thiết lập giới hạn vòng tối đa, hãy truyền nó qua dòng lệnh (command line args) + + # Truyền max_rounds vào script nếu có if max_rounds is not None and max_rounds > 0: cmd.extend(["--max-rounds", str(max_rounds)]) - - # Tạo tệp log chính để tránh bộ đệm ống dẫn (pipe buffer) stdout/stderr của tiến trình đầy + + # stdout/stderr → simulation.log (tránh pipe buffer đầy gây block) main_log_path = os.path.join(sim_dir, "simulation.log") main_log_file = open(main_log_path, 'w', encoding='utf-8') - - # Đặt môi trường cho quy trình con để đảm bảo trên Windows được mã hóa thành UTF-8 - # Điều này sửa lỗi thư viện của bên thứ 3 khi họ gọi file hệ thống nếu không chỉ định rõ encode. + + # Đảm bảo UTF-8 trên mọi OS (đặc biệt Windows mặc định CP1252) env = os.environ.copy() - env['PYTHONUTF8'] = '1' # Python 3.7+ hỗ trợ điều này, giúp mọi hàm open() mặc định theo UTF-8 - env['PYTHONIOENCODING'] = 'utf-8' # Đảm bảo đầu ra có stdout/stderr dưới dạng UTF-8 - - # Đặt thư mục làm việc (CWD - Current Working Directory) thành thư mục nơi mô phỏng - # Thiết lập start_new_session=True sẽ tạo ra nhóm các tiến trình con mới, vì thế thông qua os.killpg có thể hủy toàn bộ những cái đó + env['PYTHONUTF8'] = '1' # Python 3.7+: buộc mọi open() dùng UTF-8 + env['PYTHONIOENCODING'] = 'utf-8' # stdout/stderr cũng UTF-8 + process = subprocess.Popen( cmd, - cwd=sim_dir, + cwd=sim_dir, # Working dir = thư mục simulation (script đọc file tương đối từ đây) stdout=main_log_file, - stderr=subprocess.STDOUT, # Đẩy luồng stderr cũng vào file đó + stderr=subprocess.STDOUT, # Merge stderr vào stdout → 1 file log duy nhất text=True, - encoding='utf-8', # Explicitly specify encoding + encoding='utf-8', bufsize=1, - env=env, # Đi kèm bộ setting Environment có set UTF-8 - start_new_session=True, # Bắt đầu tạo 1 luồng xử lý mới (New process group) + env=env, + start_new_session=True, # Tạo process group mới → os.killpg() kill được toàn cây ) - - # Ghi lại file để cho bước đóng (close) được thực hiện dễ dàng + cls._stdout_files[simulation_id] = main_log_file - cls._stderr_files[simulation_id] = None # Không cần lưu file stderr độc lập nữa - + cls._stderr_files[simulation_id] = None # Không cần file riêng cho stderr + state.process_pid = process.pid state.runner_status = RunnerStatus.RUNNING cls._processes[simulation_id] = process cls._save_run_state(state) - - # Khởi động tiểu trình giám sát (Monitor thread) + + # Khởi động daemon monitor thread — chạy song song với Flask + # daemon=True → thread tự chết khi main process kết thúc monitor_thread = threading.Thread( target=cls._monitor_simulation, args=(simulation_id,), @@ -463,105 +580,125 @@ class SimulationRunner: ) monitor_thread.start() cls._monitor_threads[simulation_id] = monitor_thread - + logger.info(f"Simulation started successfully: {simulation_id}, pid={process.pid}, platform={platform}") - + except Exception as e: state.runner_status = RunnerStatus.FAILED state.error = str(e) cls._save_run_state(state) raise - + return state - + + # -------------------------------------------------------------------------- + # PRIVATE: _monitor_simulation — Daemon thread theo dõi tiến độ real-time + # -------------------------------------------------------------------------- + @classmethod def _monitor_simulation(cls, simulation_id: str): - """Giám sát (Monitor) phân tích nhật ký ghi lại các hành động""" + """ + Daemon thread chạy liên tục, đọc actions.jsonl và cập nhật run_state.json. + + Cấu trúc log (mới — tách theo platform): + uploads/simulations//twitter/actions.jsonl + uploads/simulations//reddit/actions.jsonl + + Cơ chế file seek: + Mỗi lần đọc, hàm _read_action_log() trả về vị trí cuối file (f.tell()). + Lần tiếp theo, seek đến vị trí đó → chỉ đọc các dòng MỚI thêm vào. + → Tránh parse lại toàn bộ file mỗi 2 giây. + + Vòng lặp kết thúc khi process.poll() != None (process đã thoát). + Sau đó: đọc log lần cuối → xử lý exit_code → cleanup resources. + """ sim_dir = os.path.join(cls.RUN_STATE_DIR, simulation_id) - - # 新的日志结构:分平台的动作日志 + + # Cấu trúc log mới: tách log hành động theo từng nền tảng twitter_actions_log = os.path.join(sim_dir, "twitter", "actions.jsonl") reddit_actions_log = os.path.join(sim_dir, "reddit", "actions.jsonl") - + process = cls._processes.get(simulation_id) state = cls.get_run_state(simulation_id) - + if not process or not state: return - + + # Vị trí đọc cuối cùng trong mỗi file (file seek position) twitter_position = 0 reddit_position = 0 - + try: - while process.poll() is None: # 进程仍在运行 - # 读取 Twitter 动作日志 + # Vòng lặp chính: chạy cho đến khi process kết thúc + while process.poll() is None: + # Đọc log Twitter (chỉ đọc phần mới từ twitter_position) if os.path.exists(twitter_actions_log): twitter_position = cls._read_action_log( twitter_actions_log, twitter_position, state, "twitter" ) - - # 读取 Reddit 动作日志 + + # Đọc log Reddit (chỉ đọc phần mới từ reddit_position) if os.path.exists(reddit_actions_log): reddit_position = cls._read_action_log( reddit_actions_log, reddit_position, state, "reddit" ) - - # 更新状态 + + # Lưu trạng thái → Frontend có thể poll API để lấy tiến độ cls._save_run_state(state) - time.sleep(2) - - # 进程结束后,最后读取一次日志 + time.sleep(2) # Poll mỗi 2 giây + + # Process đã kết thúc — đọc log lần cuối để không bỏ sót action cuối if os.path.exists(twitter_actions_log): cls._read_action_log(twitter_actions_log, twitter_position, state, "twitter") if os.path.exists(reddit_actions_log): cls._read_action_log(reddit_actions_log, reddit_position, state, "reddit") - - # 进程结束 + + # Xử lý kết quả dựa trên exit_code exit_code = process.returncode - + if exit_code == 0: state.runner_status = RunnerStatus.COMPLETED state.completed_at = datetime.now().isoformat() - logger.info(f"模拟完成: {simulation_id}") + logger.info(f"Simulation completed: {simulation_id}") else: state.runner_status = RunnerStatus.FAILED - # 从主日志文件读取错误信息 + # Đọc 2000 ký tự cuối của simulation.log làm error message main_log_path = os.path.join(sim_dir, "simulation.log") error_info = "" try: if os.path.exists(main_log_path): with open(main_log_path, 'r', encoding='utf-8') as f: - error_info = f.read()[-2000:] # 取最后2000字符 + error_info = f.read()[-2000:] except Exception: pass - state.error = f"进程退出码: {exit_code}, 错误: {error_info}" - logger.error(f"模拟失败: {simulation_id}, error={state.error}") - + state.error = f"Process exit code: {exit_code}, error: {error_info}" + logger.error(f"Simulation failed: {simulation_id}, error={state.error}") + state.twitter_running = False state.reddit_running = False cls._save_run_state(state) - + except Exception as e: - logger.error(f"监控线程异常: {simulation_id}, error={str(e)}") + logger.error(f"Monitor thread exception: {simulation_id}, error={str(e)}") state.runner_status = RunnerStatus.FAILED state.error = str(e) cls._save_run_state(state) - + finally: - # 停止图谱记忆更新器 + # Dừng Graph Memory Updater nếu đang chạy if cls._graph_memory_enabled.get(simulation_id, False): try: ZepGraphMemoryManager.stop_updater(simulation_id) - logger.info(f"已停止图谱记忆更新: simulation_id={simulation_id}") + logger.info(f"Graph memory update stopped: simulation_id={simulation_id}") except Exception as e: - logger.error(f"停止图谱记忆更新器失败: {e}") + logger.error(f"Failed to stop graph memory updater: {e}") cls._graph_memory_enabled.pop(simulation_id, None) - - # 清理进程资源 + + # Dọn dẹp tài nguyên process cls._processes.pop(simulation_id, None) cls._action_queues.pop(simulation_id, None) - - # 关闭日志文件句柄 + + # Đóng file handle log (stdout) if simulation_id in cls._stdout_files: try: cls._stdout_files[simulation_id].close() @@ -574,48 +711,66 @@ class SimulationRunner: except Exception: pass cls._stderr_files.pop(simulation_id, None) - + + # -------------------------------------------------------------------------- + # PRIVATE: _read_action_log — Parse actions.jsonl từ vị trí đã đọc + # -------------------------------------------------------------------------- + @classmethod def _read_action_log( - cls, - log_path: str, - position: int, + cls, + log_path: str, + position: int, state: SimulationRunState, platform: str ) -> int: """ - Đọc tệp tin nhật ký (log) của hệ thống - + Đọc các dòng mới trong actions.jsonl từ vị trí `position`. + + Cơ chế file seek: + f.seek(position) → bỏ qua phần đã đọc trước + f.tell() → trả về vị trí hiện tại sau khi đọc → dùng cho lần tiếp theo + + Hai loại dòng JSON trong actions.jsonl: + + 1. Sự kiện hệ thống (có field "event_type"): + {"event_type": "round_end", "round": 15, "simulated_hours": 15} + {"event_type": "simulation_end", "total_rounds": 72, "total_actions": 3420} + → Cập nhật round_num, simulated_hours, completed flag + + 2. Hành động agent (không có "event_type", có "agent_id"): + {"round": 15, "agent_id": 3, "action_type": "CREATE_POST", ...} + → Tạo AgentAction object, thêm vào state.recent_actions + Args: - log_path: Đường dẫn tệp nhật ký - position: Vị trí đọc trước đó - state: Đối tượng trạng thái đang chạy - platform: Nền tảng (twitter/reddit) - + log_path: Đường dẫn file actions.jsonl + position: Vị trí byte đã đọc tới lần trước + state: SimulationRunState cần cập nhật + platform: "twitter" hoặc "reddit" + Returns: - Vị trí đọc mới + Vị trí byte mới (dùng cho lần gọi tiếp theo) """ - # Kiểm tra xem có bật tính năng cập nhật bộ nhớ graph hay không graph_memory_enabled = cls._graph_memory_enabled.get(state.simulation_id, False) graph_updater = None if graph_memory_enabled: graph_updater = ZepGraphMemoryManager.get_updater(state.simulation_id) - + try: with open(log_path, 'r', encoding='utf-8') as f: - f.seek(position) + f.seek(position) # Nhảy đến vị trí đã đọc lần trước for line in f: line = line.strip() if line: try: action_data = json.loads(line) - - # Xử lý các mục của loại sự kiện + + # Xử lý sự kiện hệ thống (event_type) if "event_type" in action_data: event_type = action_data.get("event_type") - - # Phát hiện sự kiện simulation_end và đánh dấu nền tảng đã hoàn thành + if event_type == "simulation_end": + # OASIS báo hiệu đã chạy hết số vòng → đánh dấu platform completed if platform == "twitter": state.twitter_completed = True state.twitter_running = False @@ -624,22 +779,20 @@ class SimulationRunner: state.reddit_completed = True state.reddit_running = False logger.info(f"Reddit simulation completed: {state.simulation_id}, total_rounds={action_data.get('total_rounds')}, total_actions={action_data.get('total_actions')}") - - # Kiểm tra xem có phải tất cả các nền tảng được bật đều đã hoàn thành hay không - # Nếu chỉ một nền tảng đang chạy, hãy chỉ kiểm tra nền tảng đó - # Nếu 2 nền tảng đang chạy thì yêu cầu phải hoàn thành cả 2 nền tảng + + # Kiểm tra nếu tất cả platform đã xong → đánh dấu COMPLETED + # Logic: platform nào bật (có actions.jsonl) thì phải xong; phải có ít nhất 1 xong all_completed = cls._check_all_platforms_completed(state) if all_completed: state.runner_status = RunnerStatus.COMPLETED state.completed_at = datetime.now().isoformat() logger.info(f"Simulation completed for all platforms: {state.simulation_id}") - - # Cập nhật thông tin vòng (round_num) (từ sự kiện round_end) + elif event_type == "round_end": + # Cập nhật số vòng và giờ đã mô phỏng (độc lập cho từng platform) round_num = action_data.get("round", 0) simulated_hours = action_data.get("simulated_hours", 0) - - # Cập nhật thời gian và vòng thứ tự độc lập cho nền tảng + if platform == "twitter": if round_num > state.twitter_current_round: state.twitter_current_round = round_num @@ -648,15 +801,15 @@ class SimulationRunner: if round_num > state.reddit_current_round: state.reddit_current_round = round_num state.reddit_simulated_hours = simulated_hours - - # Số vòng chung sẽ là số lớn nhất của hai nền tảng + + # Vòng chung = max của 2 platform (platform nhanh hơn làm mốc) if round_num > state.current_round: state.current_round = round_num - # Thời gian chung sẽ là số lớn nhất của hai nền tảng state.simulated_hours = max(state.twitter_simulated_hours, state.reddit_simulated_hours) - - continue - + + continue # Không tạo AgentAction cho sự kiện hệ thống + + # Xử lý hành động agent (có agent_id) action = AgentAction( round_num=action_data.get("round", 0), timestamp=action_data.get("timestamp", datetime.now().isoformat()), @@ -669,65 +822,83 @@ class SimulationRunner: success=action_data.get("success", True), ) state.add_action(action) - - # Cập nhật thông tin (vòng) round + + # Cập nhật current_round từ action nếu cần if action.round_num and action.round_num > state.current_round: state.current_round = action.round_num - - # Nếu cập nhật bộ nhớ graph được bật, thêm hoạt động vào Zep graph + + # Ghi hành động vào Zep Graph nếu tính năng được bật if graph_updater: graph_updater.add_activity_from_dict(action_data, platform) - + except json.JSONDecodeError: - pass - return f.tell() + pass # Dòng không phải JSON hợp lệ → bỏ qua + + return f.tell() # Trả về vị trí cuối để lần sau tiếp tục từ đây except Exception as e: logger.warning(f"Failed to read action logs: {log_path}, error={e}") - return position - + return position # Giữ nguyên position nếu đọc lỗi + @classmethod def _check_all_platforms_completed(cls, state: SimulationRunState) -> bool: """ - Kiểm tra xem tất cả các nền tảng có hoàn thành quá trình mô phỏng hay chưa? - - Kiểm tra xem nền tảng có được kích hoạt (hay enable) hay không bằng cách xem tệp tin actions.jsonl có tồn tại hay không - + Kiểm tra xem tất cả platform đã hoàn thành chưa. + + Cách xác định platform có được bật: + Kiểm tra file actions.jsonl có tồn tại không (không dùng enable_twitter/enable_reddit flag, + vì flag đó chỉ có trong SimulationState — không truyền sang đây). + + Logic: + Nếu twitter/actions.jsonl tồn tại → twitter được bật → phải twitter_completed=True + Nếu reddit/actions.jsonl tồn tại → reddit được bật → phải reddit_completed=True + Phải có ít nhất 1 platform xong (tránh trả True khi cả 2 chưa tạo file) + Returns: - True Nếu tất cả các nền tảng được bật đều đã hoàn thành + True nếu tất cả platform đã bật đều đã completed """ sim_dir = os.path.join(cls.RUN_STATE_DIR, state.simulation_id) twitter_log = os.path.join(sim_dir, "twitter", "actions.jsonl") reddit_log = os.path.join(sim_dir, "reddit", "actions.jsonl") - - # Kiểm tra xem mình có đang bật nền tảng nào không (Sử dụng cách kiểm tra tệp tin có tồn tại (exist) hay không) + twitter_enabled = os.path.exists(twitter_log) reddit_enabled = os.path.exists(reddit_log) - - # Nền tảng nào chưa xong thì trả về false + + # Platform nào được bật mà chưa xong → False if twitter_enabled and not state.twitter_completed: return False if reddit_enabled and not state.reddit_completed: return False - - # Phải có ít nhất 1 nền tảng chạy xong thì mới là true. (nếu 1 nền tảng không chạy => False. False and False = False. True and False = True) + + # Phải có ít nhất 1 platform chạy xong return twitter_enabled or reddit_enabled - + + # -------------------------------------------------------------------------- + # PRIVATE: _terminate_process — Kill process + toàn bộ process group + # -------------------------------------------------------------------------- + @classmethod def _terminate_process(cls, process: subprocess.Popen, simulation_id: str, timeout: int = 10): """ - Khả năng tương thích nền tảng, dừng một quá trình và các quá trình con (nhánh) - + Dừng process và toàn bộ process group (bao gồm child processes). + + Tại sao phải kill cả process group? + OASIS script có thể spawn thêm asyncio workers hoặc subprocess con. + Kill chỉ process cha sẽ để lại các zombie child processes. + + Unix: os.killpg(pgid, SIGTERM) → chờ 10s → os.killpg(pgid, SIGKILL) nếu không thoát + Windows: taskkill /T /PID → chờ → taskkill /F /T /PID nếu không thoát + + start_new_session=True (trong Popen) đảm bảo pgid == pid của process cha. + Args: - process: Quy trình để chấm dứt (kill) - simulation_id: Ghi log ID - timeout: Thời gian chờ cho phép tiến trình kết thúc tính bằng giây (seconds) + process: Subprocess.Popen object + simulation_id: Chỉ dùng để log + timeout: Giây chờ trước khi SIGKILL (mặc định 10s) """ if IS_WINDOWS: - # Windows: Sử dụng câu lệnh taskkill để xóa cả tiến trình theo cấu trúc branch tree - # /F = Force termination (Xoa bằng mọi giá), /T = Terminate tree (xóa nhánh tiến trình) bao gồm các sub-process - logger.info(f"Đang dừng quá trình (Windows): simulation={simulation_id}, pid={process.pid}") + logger.info(f"Terminating process (Windows): simulation={simulation_id}, pid={process.pid}") try: - # Trước hết hãy cố găng dừng mềm + # Bước 1: Dừng nhẹ nhàng (/T = kill cả tree) subprocess.run( ['taskkill', '/PID', str(process.pid), '/T'], capture_output=True, @@ -736,7 +907,7 @@ class SimulationRunner: try: process.wait(timeout=timeout) except subprocess.TimeoutExpired: - # Nếu không thể dùng dừng mềm (sau timeout), xóa cứng bằng /F + # Bước 2: Kill cưỡng bức (/F = force) logger.warning(f"Process unresponsive, force terminating: {simulation_id}") subprocess.run( ['taskkill', '/F', '/PID', str(process.pid), '/T'], @@ -752,59 +923,72 @@ class SimulationRunner: except subprocess.TimeoutExpired: process.kill() else: - # Unix: Sử dụng process group chấm dứt - # Dùng start_new_session=True, giá trị pgid sẽ bằng đúng với PID gốc của tiến trình + # Unix: dùng process group để kill toàn bộ cây pgid = os.getpgid(process.pid) - logger.info(f"Đang dừng nhóm tiến trình (Unix): simulation={simulation_id}, pgid={pgid}") - - # Gửi SIGTERM tới toàn bộ process group + logger.info(f"Terminating process group (Unix): simulation={simulation_id}, pgid={pgid}") + + # Bước 1: SIGTERM — tín hiệu nhẹ nhàng, cho phép process dọn dẹp os.killpg(pgid, signal.SIGTERM) - + try: process.wait(timeout=timeout) except subprocess.TimeoutExpired: - # Nếu xảy ra hiện tượng chưa tự hủy sau timeout, xóa cưỡng bức bằng SIGKILL + # Bước 2: SIGKILL — kill cưỡng bức, không thể bị bắt hay bỏ qua logger.warning(f"Process group unresponsive to SIGTERM, force terminating: {simulation_id}") os.killpg(pgid, signal.SIGKILL) process.wait(timeout=5) - + + # -------------------------------------------------------------------------- + # PUBLIC: stop_simulation — Dừng simulation theo yêu cầu người dùng + # -------------------------------------------------------------------------- + @classmethod def stop_simulation(cls, simulation_id: str) -> SimulationRunState: - """Dừng lại tiến trình mô phỏng""" + """ + Dừng OASIS subprocess → state = STOPPED. + + Thứ tự: + 1. Chuyển state sang STOPPING (để frontend biết đang xử lý) + 2. _terminate_process() — kill process group + 3. Chuyển state sang STOPPED + 4. Dừng Graph Memory Updater nếu đang chạy + + Khác với FAILED: STOPPED là do người dùng chủ động, FAILED là do crash. + """ state = cls.get_run_state(simulation_id) if not state: raise ValueError(f"Simulation not found: {simulation_id}") - + if state.runner_status not in [RunnerStatus.RUNNING, RunnerStatus.PAUSED]: raise ValueError(f"Simulation is not running: {simulation_id}, status={state.runner_status}") - + state.runner_status = RunnerStatus.STOPPING cls._save_run_state(state) - - # Kết thúc tiến trình con (child process) + + # Kill OASIS subprocess (và toàn bộ process group) process = cls._processes.get(simulation_id) if process and process.poll() is None: try: cls._terminate_process(process, simulation_id) except ProcessLookupError: - # Quá trình không còn ở đây nữa (Đã thoát hoặc bị đóng) + # Process đã tự thoát trước khi kịp kill — OK pass except Exception as e: logger.error(f"Failed to terminate process group: {simulation_id}, error={e}") - # Thử thêm một cách nữa để chăc chắn hủy tiến trình + # Fallback: try terminate() rồi kill() try: process.terminate() process.wait(timeout=5) except Exception: process.kill() - + state.runner_status = RunnerStatus.STOPPED state.twitter_running = False state.reddit_running = False state.completed_at = datetime.now().isoformat() cls._save_run_state(state) - - # Dừng quá trình graph memory updater + + # Dừng Graph Memory Updater if cls._graph_memory_enabled.get(simulation_id, False): try: ZepGraphMemoryManager.stop_updater(simulation_id) @@ -812,10 +996,14 @@ class SimulationRunner: except Exception as e: logger.error(f"Failed to stop graph memory updater: {e}") cls._graph_memory_enabled.pop(simulation_id, None) - + logger.info(f"Simulation stopped: {simulation_id}") return state - + + # -------------------------------------------------------------------------- + # PUBLIC: Đọc lịch sử hành động (dùng sau khi simulation kết thúc) + # -------------------------------------------------------------------------- + @classmethod def _read_actions_from_file( cls, @@ -826,48 +1014,51 @@ class SimulationRunner: round_num: Optional[int] = None ) -> List[AgentAction]: """ - Đọc các hoạt động từ một tệp duy nhất - + Đọc toàn bộ actions từ 1 file actions.jsonl với các bộ lọc tùy chọn. + + Bỏ qua: + - Dòng có "event_type" (sự kiện hệ thống: round_end, simulation_end) + - Dòng không có "agent_id" (không phải hành động của agent) + Args: - file_path: Đường dẫn tệp log của hành động đó - default_platform: Nền tảng mặc định (nếu trong nhật ký không có platform) - platform_filter: Lọc nền tảng (chỉ định platform cần đọc log) - agent_id: Lọc ID của Agent cụ thể - round_num: Lọc số vòng của Agent + file_path: Đường dẫn file actions.jsonl + default_platform: Điền tự động nếu record không có field "platform" + platform_filter: Chỉ lấy record của platform này + agent_id: Chỉ lấy record của agent này + round_num: Chỉ lấy record của vòng này """ if not os.path.exists(file_path): return [] - + actions = [] - + with open(file_path, 'r', encoding='utf-8') as f: for line in f: line = line.strip() if not line: continue - + try: data = json.loads(line) - - # Bỏ qua các bản ghi không phải hành động (chẳng hạn như là sự kiện về hệ thống: simulation_start, round_start, round_end v.v.) + + # Bỏ qua sự kiện hệ thống if "event_type" in data: continue - - # Bỏ lỡ các sự kiện không phải do agent tạo ra (Không có ID đặc trưng của Agent) + + # Bỏ qua record không có agent_id if "agent_id" not in data: continue - - # Lấy nền tảng (Platform): Ưu tiên lấy từ bản ghi nếu có `platform`, nếu không thì dùng `default_platform` + record_platform = data.get("platform") or default_platform or "" - - # Bộ lọc (Filtering) + + # Áp dụng bộ lọc if platform_filter and record_platform != platform_filter: continue if agent_id is not None and data.get("agent_id") != agent_id: continue if round_num is not None and data.get("round") != round_num: continue - + actions.append(AgentAction( round_num=data.get("round", 0), timestamp=data.get("timestamp", ""), @@ -879,12 +1070,12 @@ class SimulationRunner: result=data.get("result"), success=data.get("success", True), )) - + except json.JSONDecodeError: continue - + return actions - + @classmethod def get_all_actions( cls, @@ -894,58 +1085,55 @@ class SimulationRunner: round_num: Optional[int] = None ) -> List[AgentAction]: """ - Lấy thông tin tất cả lịch sử hoạt động của các nền tảng (không giới hạn phân trang) - - Args: - simulation_id: ID mô phỏng - platform: Bộ lọc nền tảng hoạt động (twitter/reddit) - agent_id: Lọc Agent - round_num: Lọc số vòng - - Returns: - Danh sách đầy đủ các actions (sắp xếp theo thời gian mới nhất lên trước) + Lấy toàn bộ lịch sử hành động — đọc từ file (không giới hạn phân trang). + + Thứ tự ưu tiên đọc file: + 1. twitter/actions.jsonl (cấu trúc mới — tách theo platform) + 2. reddit/actions.jsonl + 3. actions.jsonl (cấu trúc cũ — fallback tương thích ngược) + + Kết quả được sort theo timestamp giảm dần (mới nhất lên đầu). """ sim_dir = os.path.join(cls.RUN_STATE_DIR, simulation_id) actions = [] - - # Đọc tệp tin Actions của Twitter (Khai báo tự động điền twitter theo cấu trúc tệp tin log) + + # Đọc Twitter actions twitter_actions_log = os.path.join(sim_dir, "twitter", "actions.jsonl") if not platform or platform == "twitter": actions.extend(cls._read_actions_from_file( twitter_actions_log, - default_platform="twitter", # Điền dữ liệu tự động cho record `platform` + default_platform="twitter", platform_filter=platform, - agent_id=agent_id, + agent_id=agent_id, round_num=round_num )) - - # Đọc tệp tin Actions của Reddit (Tự động điền phần 'reddit' căn cứ thư mục chứa tệp tin) + + # Đọc Reddit actions reddit_actions_log = os.path.join(sim_dir, "reddit", "actions.jsonl") if not platform or platform == "reddit": actions.extend(cls._read_actions_from_file( reddit_actions_log, - default_platform="reddit", # Automatically fill the platform field + default_platform="reddit", platform_filter=platform, agent_id=agent_id, round_num=round_num )) - - # Nếu thư mục chạy các nền tảng chạy parallel này (twitter / reddit) không có ở đó. Hãy thử với các tệp định dạng cũ + + # Fallback: đọc file actions.jsonl cũ nếu không có thư mục twitter/reddit if not actions: actions_log = os.path.join(sim_dir, "actions.jsonl") actions = cls._read_actions_from_file( actions_log, - default_platform=None, # Các file json log định dạng cũ đã có sẵn record về platform nên không điền default + default_platform=None, # File cũ đã có field platform trong record platform_filter=platform, agent_id=agent_id, round_num=round_num ) - - # Sắp xếp lại log theo thời gian timestamp giảm dần (từ mới hơn lên trước) + actions.sort(key=lambda x: x.timestamp, reverse=True) - + return actions - + @classmethod def get_actions( cls, @@ -957,18 +1145,10 @@ class SimulationRunner: round_num: Optional[int] = None ) -> List[AgentAction]: """ - Lấy thông tin lịch sử diễn ra (Hỗ trợ phân trang bằng offset và limit) - - Args: - simulation_id: Simulation ID - limit: Record Count returns limit - offset: Offset - platform: Filter Platform - agent_id: Filter Agent by ID - round_num: Filter round loop - - Returns: - Actions list + Lấy lịch sử hành động có phân trang (offset + limit). + + Gọi get_all_actions() rồi cắt — không tối ưu cho file lớn nhưng đơn giản. + Nếu hiệu năng là vấn đề, cần cải thiện bằng cách chỉ đọc đến offset+limit. """ actions = cls.get_all_actions( simulation_id=simulation_id, @@ -976,10 +1156,9 @@ class SimulationRunner: agent_id=agent_id, round_num=round_num ) - - # Phân trang + return actions[offset:offset + limit] - + @classmethod def get_timeline( cls, @@ -988,29 +1167,27 @@ class SimulationRunner: end_round: Optional[int] = None ) -> List[Dict[str, Any]]: """ - Lấy thời gian (timeline) mô phỏng diễn ra (tóm tắt theo vòng được khai báo) - - Args: - simulation_id: ID Mô phỏng - start_round: Bắt đầu từ một vòng lặp nhất định (First round Number) - end_round: Kết thúc từ vòng ở đó (End round Number) - - Returns: - Cung cấp đầy đủ thông tin về các round bị gói gọn + Lấy timeline mô phỏng — tóm tắt theo từng vòng. + + Nhóm tất cả actions theo round_num, tính: + - Số action twitter/reddit trong vòng + - Danh sách agent_id đã hoạt động + - Phân phối action_type (CREATE_POST, LIKE_POST, ...) + + Dùng để hiển thị biểu đồ hoạt động theo thời gian trên frontend. """ actions = cls.get_actions(simulation_id, limit=10000) - - # Nhóm tiến trình lại vào trong Vòng (round grouping) + rounds: Dict[int, Dict[str, Any]] = {} - + for action in actions: round_num = action.round_num - + if round_num < start_round: continue if end_round is not None and round_num > end_round: continue - + if round_num not in rounds: rounds[round_num] = { "round_num": round_num, @@ -1021,19 +1198,19 @@ class SimulationRunner: "first_action_time": action.timestamp, "last_action_time": action.timestamp, } - + r = rounds[round_num] - + if action.platform == "twitter": r["twitter_actions"] += 1 else: r["reddit_actions"] += 1 - + r["active_agents"].add(action.agent_id) r["action_types"][action.action_type] = r["action_types"].get(action.action_type, 0) + 1 r["last_action_time"] = action.timestamp - - # Chuyển đổi trạng thái về Lists Arrays Data type + + # Chuyển set → list để JSON serializable, sort theo round_num tăng dần result = [] for round_num in sorted(rounds.keys()): r = rounds[round_num] @@ -1048,24 +1225,23 @@ class SimulationRunner: "first_action_time": r["first_action_time"], "last_action_time": r["last_action_time"], }) - + return result - + @classmethod def get_agent_stats(cls, simulation_id: str) -> List[Dict[str, Any]]: """ - Lấy thông kê của mọi agent - - Returns: - Danh sách thống kê Agent + Thống kê hoạt động của từng agent — sort theo tổng số hành động giảm dần. + + Dùng để hiển thị bảng xếp hạng "agent nào hoạt động nhiều nhất". """ actions = cls.get_actions(simulation_id, limit=10000) - + agent_stats: Dict[int, Dict[str, Any]] = {} - + for action in actions: agent_id = action.agent_id - + if agent_id not in agent_stats: agent_stats[agent_id] = { "agent_id": agent_id, @@ -1077,71 +1253,68 @@ class SimulationRunner: "first_action_time": action.timestamp, "last_action_time": action.timestamp, } - + stats = agent_stats[agent_id] stats["total_actions"] += 1 - + if action.platform == "twitter": stats["twitter_actions"] += 1 else: stats["reddit_actions"] += 1 - + stats["action_types"][action.action_type] = stats["action_types"].get(action.action_type, 0) + 1 stats["last_action_time"] = action.timestamp - - # Sắp xếp theo tổng số hành động giảm dần (reverse = true) + result = sorted(agent_stats.values(), key=lambda x: x["total_actions"], reverse=True) - + return result - + + # -------------------------------------------------------------------------- + # PUBLIC: cleanup — Dọn dẹp file log để cho phép chạy lại + # -------------------------------------------------------------------------- + @classmethod def cleanup_simulation_logs(cls, simulation_id: str) -> Dict[str, Any]: """ - Xóa tệp log chạy để buộc mô phỏng được khởi động lại - - Xóa sạch các tệp tin này bao gồm: - - run_state.json - - twitter/actions.jsonl - - reddit/actions.jsonl - - simulation.log - - stdout.log / stderr.log - - twitter_simulation.db(Dữ liệu nền tảng twitter) - - reddit_simulation.db(Dữ liệu nền tảng reddit) - - env_status.json(Trạng thái file Environment status) - - Chú ý: Các file liên kết đến cấu hình mô phỏng hay config thiết lập (như là simulation_config.json) hay Profile đều sẽ KHÔNG bị xóa đi. - - Args: - simulation_id: Simulation ID - + Xóa các file runtime để buộc simulation được khởi động lại từ đầu. + + CÁC FILE BỊ XÓA (runtime logs): + run_state.json, simulation.log, stdout.log, stderr.log, + twitter_simulation.db, reddit_simulation.db, env_status.json, + twitter/actions.jsonl, reddit/actions.jsonl + + CÁC FILE GIỮ LẠI (config + profiles — không xóa): + simulation_config.json, reddit_profiles.json, twitter_profiles.csv, state.json + + Khi nào cần gọi: + Sau server restart nếu run_state.json vẫn hiển thị status="running" + (process đã chết cùng server → không thể tiếp tục, phải cleanup rồi start lại) + Returns: - Kết quả của lệnh xóa sạch (clean up) + {"success": bool, "cleaned_files": [...], "errors": [...]} """ import shutil - + sim_dir = os.path.join(cls.RUN_STATE_DIR, simulation_id) - + if not os.path.exists(sim_dir): return {"success": True, "message": "Simulation directory does not exist, no need to clean."} - + cleaned_files = [] errors = [] - - # Các tệp tin cần bị loại bỏ bao gồm log, database... + files_to_delete = [ "run_state.json", "simulation.log", "stdout.log", "stderr.log", - "twitter_simulation.db", # Twitter Database - "reddit_simulation.db", # Reddit Database - "env_status.json", # Env state status file + "twitter_simulation.db", + "reddit_simulation.db", + "env_status.json", ] - - # Nhưng tệp tin có cấp quyền cần xóa (có liên quan nhật ký hoạt động actions.jsonl) + dirs_to_clean = ["twitter", "reddit"] - - # Loại bỏ các tệp không cần tới (Delete them) + for filename in files_to_delete: file_path = os.path.join(sim_dir, filename) if os.path.exists(file_path): @@ -1150,8 +1323,7 @@ class SimulationRunner: cleaned_files.append(filename) except Exception as e: errors.append(f"Failed to delete {filename}: {str(e)}") - - # Kiểm tra lại các file theo thư mục chứa action history action.jsonl files + for dir_name in dirs_to_clean: dir_path = os.path.join(sim_dir, dir_name) if os.path.exists(dir_path): @@ -1162,70 +1334,78 @@ class SimulationRunner: cleaned_files.append(f"{dir_name}/actions.jsonl") except Exception as e: errors.append(f"Failed to delete {dir_name}/actions.jsonl: {str(e)}") - - # Xóa (clear) cache nhớ run state + + # Xóa RAM cache cho simulation này if simulation_id in cls._run_states: del cls._run_states[simulation_id] - + logger.info(f"Clean up complete for simulation: {simulation_id}, Deleted files: {cleaned_files}") - + return { "success": len(errors) == 0, "cleaned_files": cleaned_files, "errors": errors if errors else None } - - # Flags Ngăn việc phải làm quá nhiều việc cho một action cleanup đã làm từ đầu hay khi gọi lần tới lệnh giống + + # Cờ chống gọi cleanup nhiều lần (atexit có thể gọi nhiều lần trong một số trường hợp) _cleanup_done = False - + @classmethod def cleanup_all_simulations(cls): """ - Dọn dẹp tất cả các tiến trình mô phỏng đang chạy - - Được gọi (call) khi đóng máy chủ, nhằm vào việc muốn các máy chủ con (child processes) bị tắt theo + Dọn dẹp tất cả simulation đang chạy — gọi khi server tắt. + + Được đăng ký bởi: + atexit.register(cls.cleanup_all_simulations) → khi Python interpreter kết thúc + signal.signal(SIGTERM, cleanup_handler) → khi nhận SIGTERM (kill server) + signal.signal(SIGINT, cleanup_handler) → khi Ctrl+C + + Thực hiện: + 1. Dừng tất cả Graph Memory Updaters + 2. Kill toàn bộ OASIS subprocess đang chạy + 3. Cập nhật state.json → status="stopped" (để frontend không hiển thị "running" sau khi restart) + 4. Cập nhật run_state.json → runner_status="stopped" + 5. Đóng tất cả file handle + 6. Clear RAM cache """ - # Nếu đã clean (dọn dẹp) thì không làm nữa if cls._cleanup_done: return cls._cleanup_done = True - - # Kiểm tra xem có gì để dọn dẹp không (tránh log rỗng khi không có tiến trình chạy) + has_processes = bool(cls._processes) has_updaters = bool(cls._graph_memory_enabled) - + if not has_processes and not has_updaters: - return # Không có gì để dọn, kết thúc - + return # Không có gì để dọn + logger.info("Cleaning up all simulation processes...") - - # Dừng tất cả cái update đồ thị nhớ (Bộ nhớ Graph) (Stop_all ghi nhận ở bên trong) + + # Dừng tất cả Graph Memory Updaters try: ZepGraphMemoryManager.stop_all() except Exception as e: logger.error(f"Failed to stop Zep Graph update daemon: {e}") cls._graph_memory_enabled.clear() - - # Tạo bản sao Dictionary dict() từ cls._processes.items để không bị hỏng List lúc lặp (iterating) + + # Tạo bản sao để tránh RuntimeError khi dict thay đổi trong khi lặp processes = list(cls._processes.items()) - + for simulation_id, process in processes: try: - if process.poll() is None: # Process (Tiến trình) Vẫn đang chạy + if process.poll() is None: # Process vẫn đang chạy logger.info(f"Terminating simulation process: {simulation_id}, pid={process.pid}") - + try: - # Áp dụng giải pháp dừng liên nền tảng (Cross-platform termination method) cls._terminate_process(process, simulation_id, timeout=5) except (ProcessLookupError, OSError): - # Trong trường hợp có thể các process này đã biến mất ở đâu đó rồi, xóa một cách bắt buộc + # Process đã biến mất → fallback terminate/kill try: process.terminate() process.wait(timeout=3) except Exception: process.kill() - - # Cập nhật run_state.json + + # Cập nhật run_state.json → status=stopped state = cls.get_run_state(simulation_id) if state: state.runner_status = RunnerStatus.STOPPED @@ -1234,8 +1414,9 @@ class SimulationRunner: state.completed_at = datetime.now().isoformat() state.error = "Server shut down, simulation terminated." cls._save_run_state(state) - - # Đồng thời cập nhật trạng thái `stopped` cho tệp (file) state.json + + # Cập nhật state.json → status="stopped" (SimulationManager's file) + # Lý do: SimulationManager không tự biết server đang tắt → phải cập nhật thủ công try: sim_dir = os.path.join(cls.RUN_STATE_DIR, simulation_id) state_file = os.path.join(sim_dir, "state.json") @@ -1252,11 +1433,11 @@ class SimulationRunner: logger.warning(f"state.json not found: {state_file}") except Exception as state_err: logger.warning(f"Failed to update state.json: {simulation_id}, error={state_err}") - + except Exception as e: logger.error(f"Failed to clean up process: {simulation_id}, error={e}") - - # Đóng tất cả tệp xử lý file handles (Log file, Errors File) + + # Đóng tất cả file handle log for simulation_id, file_handle in list(cls._stdout_files.items()): try: if file_handle: @@ -1264,7 +1445,7 @@ class SimulationRunner: except Exception: pass cls._stdout_files.clear() - + for simulation_id, file_handle in list(cls._stderr_files.items()): try: if file_handle: @@ -1272,109 +1453,105 @@ class SimulationRunner: except Exception: pass cls._stderr_files.clear() - - # Dọn dẹp trạng thái ở trong Ram Memory + + # Clear RAM cache cls._processes.clear() cls._action_queues.clear() - + logger.info("Simulation process clean up completed.") - + @classmethod def register_cleanup(cls): """ - Đăng ký một lệnh Dọn dẹp (Cleanup command) - - Trong lúc chuẩn bị khởi tạo App Flask, mình sẽ thiết lập nó sao cho gọi là máy chủ kết thúc (tắt) mọi quá trình (Simulation Process) + Đăng ký hàm dọn dẹp khi server tắt. + + Gọi 1 lần duy nhất trong app initialization (Flask app factory). + Đăng ký 3 signal handlers + atexit fallback: + SIGTERM: kill server lệnh (Linux/Mac: systemd stop, Docker stop) + SIGINT: Ctrl+C từ terminal + SIGHUP: Terminal bị đóng (Unix only) + atexit: Fallback khi signal handling không khả dụng + + Lưu ý Flask debug mode: + Werkzeug reload spawns 2 processes — chỉ đăng ký trong reloader child process + (WERKZEUG_RUN_MAIN=true). Production luôn đăng ký. """ global _cleanup_registered - + if _cleanup_registered: return - - # Flask ở trong cơ chế gỡ rối `debug`, lúc này chỉ đăng ký ứng dụng để cho thằng app.run làm (Werkzeug) - # WERKZEUG_RUN_MAIN=true Đại diện quá trình tiến trình máy chủ được nạp lại - # Nhưng nó sẽ ko apply điều này ở Production nếu app không debug + is_reloader_process = os.environ.get('WERKZEUG_RUN_MAIN') == 'true' is_debug_mode = os.environ.get('FLASK_DEBUG') == '1' or os.environ.get('WERKZEUG_RUN_MAIN') is not None - - # Trong DebugMode, chúng ta chỉ cho phép re-loader Child-process chạy. Production vẫn luôn phải chạy process này + + # Debug mode: chỉ đăng ký trong child process của Werkzeug reloader if is_debug_mode and not is_reloader_process: - _cleanup_registered = True # Check list đã lưu lại. Hủy quyền yêu cầu thêm + _cleanup_registered = True return - - # Lưu các tín hiệu để trả về Signal handling sau khi dừng Process hoàn tất + + # Lưu signal handlers gốc để gọi lại sau khi cleanup original_sigint = signal.getsignal(signal.SIGINT) original_sigterm = signal.getsignal(signal.SIGTERM) - # SIGHUP Chỉ xuất hiện trong Unix (Mac/Linux), Windows không có cái này original_sighup = None has_sighup = hasattr(signal, 'SIGHUP') if has_sighup: original_sighup = signal.getsignal(signal.SIGHUP) - + def cleanup_handler(signum=None, frame=None): - """Xử lý điều hướng tín hiệu (Signal Routing): Bắt đầu dọn tiến trình xong gởi lệnh báo (Original Processing Router)""" - # Chỉ báo nhật kí (log) nếu có tiến trình (Process) cần xử lý + """Cleanup toàn bộ simulation rồi forward signal về handler gốc.""" if cls._processes or cls._graph_memory_enabled: logger.info(f"Received signal {signum}, starting clean up...") cls.cleanup_all_simulations() - - # Gửi tín hiệu gọi hàm báo (handling functions) lúc đấy của Flask => App được tự do ngắt điện + + # Forward signal về Flask/Werkzeug handler gốc if signum == signal.SIGINT and callable(original_sigint): original_sigint(signum, frame) elif signum == signal.SIGTERM and callable(original_sigterm): original_sigterm(signum, frame) elif has_sighup and signum == signal.SIGHUP: - # SIGHUP: Được trả về khi máy chủ bị dừng (Terminal Closed) if callable(original_sighup): original_sighup(signum, frame) else: - # Mặc định hành vi: Đóng bình thường => sys.exit(0) "Tạm biệt các hành khách" sys.exit(0) else: - # Hành động ở cơ sở gốc (Root) không gọi được (SIG_DFL) => Hãy phát cảnh báo raise KeyboardInterrupt - - # Một phương án khác nếu tín hiệu đăng ký xử lý gặp khó (Fallback Option) + + # Đăng ký atexit làm fallback (nếu signal handler không chạy được) atexit.register(cls.cleanup_all_simulations) - - # Đăng ký quản lý báo tín hiệu (Chỉ riêng trong chủ / main thread có cái luồng) + + # Đăng ký signal handlers — chỉ hoạt động trong main thread try: - # SIGTERM: Tín hiệu gốc của Kill Server (Linux/Mac) signal.signal(signal.SIGTERM, cleanup_handler) - # SIGINT: Bấm lệnh Control + C / Ctrl+C signal.signal(signal.SIGINT, cleanup_handler) - # SIGHUP: Máy bị đóng (Unix OS) if has_sighup: signal.signal(signal.SIGHUP, cleanup_handler) except ValueError: - # Không ở MainThread => Sử dụng được duy nhất fallback Atexit + # Không ở main thread → chỉ dùng atexit fallback logger.warning("Failed to register signal handlers (not in main thread). Falling back to atexit.") - + _cleanup_registered = True - + @classmethod def get_running_simulations(cls) -> List[str]: - """ - Lấy danh sách tất cả các ID của các phiên mô phỏng đang hoạt động - """ + """Lấy danh sách simulation_id đang có process chạy (process.poll() is None).""" running = [] for sim_id, process in cls._processes.items(): if process.poll() is None: running.append(sim_id) return running - - # ============== Tính năng Phỏng vấn (Interview) ============== - + + # -------------------------------------------------------------------------- + # PUBLIC: Interview — Phỏng vấn agent đang chạy qua IPC + # -------------------------------------------------------------------------- + @classmethod def check_env_alive(cls, simulation_id: str) -> bool: """ - Kiểm tra xem environment còn sống không (có thể nhận lệnh Interview) + Kiểm tra xem OASIS environment có còn nhận lệnh IPC không. - Args: - simulation_id: Simulation ID - - Returns: - True Nếu environment còn sống, False nghĩa là đã đóng + Đọc env_status.json → status == "alive". + Phải kiểm tra trước khi gửi bất kỳ lệnh Interview nào. + Nếu False → raise ValueError ngay, không gửi IPC để tránh timeout. """ sim_dir = os.path.join(cls.RUN_STATE_DIR, simulation_id) if not os.path.exists(sim_dir): @@ -1386,27 +1563,27 @@ class SimulationRunner: @classmethod def get_env_status_detail(cls, simulation_id: str) -> Dict[str, Any]: """ - Lấy thông tin chi tiết về trạng thái của environment + Lấy thông tin chi tiết trạng thái IPC environment. - Args: - simulation_id: Mô phỏng ID - - Returns: - Bảng trạng thái chi tiết (Dictionary) bao gồm: status, twitter_available, reddit_available, timestamp + Returns dict: + status: "alive" hoặc "stopped" + twitter_available: True nếu Twitter environment đang active + reddit_available: True nếu Reddit environment đang active + timestamp: Thời điểm cập nhật env_status.json cuối cùng """ sim_dir = os.path.join(cls.RUN_STATE_DIR, simulation_id) status_file = os.path.join(sim_dir, "env_status.json") - + default_status = { "status": "stopped", "twitter_available": False, "reddit_available": False, "timestamp": None } - + if not os.path.exists(status_file): return default_status - + try: with open(status_file, 'r', encoding='utf-8') as f: status = json.load(f) @@ -1429,24 +1606,20 @@ class SimulationRunner: timeout: float = 60.0 ) -> Dict[str, Any]: """ - Phỏng vấn trên 1 Agent + Phỏng vấn 1 agent đang chạy qua IPC. + + Flow: + 1. check_env_alive() → nếu False → raise ValueError ngay + 2. SimulationIPCClient.send_interview() → ghi file ipc_commands/{uuid}.json + 3. Poll ipc_responses/{uuid}.json mỗi 0.5s, timeout 60s + 4. OASIS script phát hiện command → chạy ManualAction INTERVIEW → ghi response + 5. Trả về response về caller Args: - simulation_id: ID mô phỏng - agent_id: Agent ID + agent_id: ID của agent (khớp với user_id trong profile file) prompt: Câu hỏi phỏng vấn - platform: Chỉ định nền tảng (Tùy chọn/Optional) - - "twitter": Chỉ PV trên account Twitter - - "reddit": Chỉ PV trên account Reddit - - None: Phỏng vấn chéo trên cả hai nền tảng, trả về kết quả hợp lại (Nếu chạy mô phỏng nền tảng kép) - timeout: Thời gian chờ tối đa (giây) - - Returns: - Từ điền chứa kết quả PV - - Raises: - ValueError: Không có mô phỏng hoặc environment ko chạy - TimeoutError: Đang chờ phản hồi bị timeout + platform: "twitter"/"reddit"/None (None = phỏng vấn trên platform đang active) + timeout: Timeout tính bằng giây """ sim_dir = os.path.join(cls.RUN_STATE_DIR, simulation_id) if not os.path.exists(sim_dir): @@ -1482,7 +1655,7 @@ class SimulationRunner: "error": response.error, "timestamp": response.timestamp } - + @classmethod def interview_agents_batch( cls, @@ -1492,23 +1665,15 @@ class SimulationRunner: timeout: float = 120.0 ) -> Dict[str, Any]: """ - Phỏng vấn hàng loạt nhiều Agent + Phỏng vấn nhiều agent cùng lúc — gửi 1 batch command duy nhất. + + Hiệu quả hơn gọi interview_agent() N lần vì chỉ cần 1 round-trip IPC. + OASIS xử lý tất cả interviews trong batch rồi trả về kết quả gộp. Args: - simulation_id: ID mô phỏng - interviews: Danh sách nội dung phỏng vấn, mỗi phần tử (element) chứa {"agent_id": int, "prompt": str, "platform": str(tùy chọn)} - platform: Nền tảng mặc định (Nếu không chọn riêng cho từng phần tử) - - "twitter": Mặc định chỉ dùng mạng Twitter - - "reddit": Mặc định chỉ dùng mạng Reddit - - None: Phỏng vấn gộp trên cả hai nền tảng với mỗi Agent - timeout: Hết thời gian chờ (ms) (seconds) - - Returns: - Dict từ điển với các kết quả phỏng vấn hàng loạt - - Raises: - ValueError: Chưa có tiến trình chạy mô phỏng - TimeoutError: Phỏng vấn lâu quá (Timeout timeout timeout) + interviews: List[{"agent_id": int, "prompt": str, "platform": str (optional)}] + platform: Platform mặc định cho tất cả (nếu không chỉ định riêng từng phần tử) + timeout: Timeout tổng cho toàn bộ batch """ sim_dir = os.path.join(cls.RUN_STATE_DIR, simulation_id) if not os.path.exists(sim_dir): @@ -1541,7 +1706,7 @@ class SimulationRunner: "error": response.error, "timestamp": response.timestamp } - + @classmethod def interview_all_agents( cls, @@ -1551,27 +1716,18 @@ class SimulationRunner: timeout: float = 180.0 ) -> Dict[str, Any]: """ - Phỏng vấn TOÀN BỘ Agent (Phỏng vấn tổng quan - Global interview) + Phỏng vấn TẤT CẢ agent với cùng 1 câu hỏi. - Hỏi một câu hỏi với toàn bộ Agent đang có trong phiên mô phỏng hiện tại + Đọc danh sách agent từ simulation_config.json → build interviews list → + gọi interview_agents_batch(). - Args: - simulation_id: ID mô phỏng - prompt: Câu hỏi (Cho tất cả các Agent) - platform: Quyết định nền tảng (Platform decision) - - "twitter": Trỏ tới Twitter Platform - - "reddit": Trỏ tới Reddit Platform - - None: Interview kết hợp trên cả nền tảng của từng agent - timeout: Timeout - - Returns: - Kết quả của toàn thể hội đồng Agents (Lớp/Nhóm) + Dùng khi muốn biết toàn bộ quan điểm của cộng đồng về 1 chủ đề. + Timeout mặc định 180s vì số lượng agent nhiều. """ sim_dir = os.path.join(cls.RUN_STATE_DIR, simulation_id) if not os.path.exists(sim_dir): raise ValueError(f"Simulation does not exist: {simulation_id}") - # Fetch All Agents profile (Lấy thông tin agents từ phần thiết lập) config_path = os.path.join(sim_dir, "simulation_config.json") if not os.path.exists(config_path): raise ValueError(f"Simulation config not found: {simulation_id}") @@ -1583,7 +1739,6 @@ class SimulationRunner: if not agent_configs: raise ValueError(f"No Agent defined under simulation configs: {simulation_id}") - # Tập hợp danh sách phỏng vấn tất cả interviews = [] for agent_config in agent_configs: agent_id = agent_config.get("agent_id") @@ -1601,7 +1756,7 @@ class SimulationRunner: platform=platform, timeout=timeout ) - + @classmethod def close_simulation_env( cls, @@ -1609,34 +1764,32 @@ class SimulationRunner: timeout: float = 30.0 ) -> Dict[str, Any]: """ - Đóng Environment giả lập (Không phải dừng process tắt hẳn nó đi) - - Gửi lệnh ra hiệu cho Simulation ngắt bỏ tiến trình để các processes thoát ra an toàn và êm đẹp về trạng thái đang chờ nhận lệnh - - Args: - simulation_id: ID Simulation - timeout: Timeout chờ kết nối - - Returns: - Kiểu từ điển: Quá trình (Process Status execution) + Gửi lệnh đóng environment qua IPC (không kill process ngay). + + Khác với stop_simulation(): + stop_simulation() → SIGTERM ngay lập tức (brutal) + close_simulation_env() → gửi lệnh IPC để OASIS tự dọn dẹp rồi thoát (graceful) + + Dùng khi muốn kết thúc simulation sớm nhưng để OASIS lưu trạng thái trước. + Nếu timeout → trả về success=True vì environment có thể đang trong quá trình đóng. """ sim_dir = os.path.join(cls.RUN_STATE_DIR, simulation_id) if not os.path.exists(sim_dir): raise ValueError(f"Simulation does not exist: {simulation_id}") - + ipc_client = SimulationIPCClient(sim_dir) - + if not ipc_client.check_env_alive(): return { "success": True, "message": "Environment is already closed" } - + logger.info(f"Sending command to close Environment: simulation_id={simulation_id}") - + try: response = ipc_client.send_close_env(timeout=timeout) - + return { "success": response.status.value == "completed", "message": "Environment close command sent", @@ -1644,7 +1797,7 @@ class SimulationRunner: "timestamp": response.timestamp } except TimeoutError: - # Hết thời gian chờ nguyên nhân lớn nhất là vì Simulation environment đang đóng giữa chừng. + # Timeout thường xảy ra khi environment đang trong quá trình đóng → OK return { "success": True, "message": "Environment close command sent (timeout waiting for response, env might be closing)" @@ -1654,7 +1807,7 @@ class SimulationRunner: "success": False, "message": f"Failed to send close env command: {str(e)}" } - + @classmethod def _get_interview_history_from_db( cls, @@ -1663,18 +1816,24 @@ class SimulationRunner: agent_id: Optional[int] = None, limit: int = 100 ) -> List[Dict[str, Any]]: - """Lấy lịch sử phỏng vấn từ Local Database của nền tảng""" + """ + Đọc lịch sử phỏng vấn từ SQLite database của OASIS. + + OASIS lưu kết quả interview vào bảng `trace` trong database + (twitter_simulation.db hoặc reddit_simulation.db). + Truy vấn WHERE action='interview' để lọc ra các record phỏng vấn. + """ import sqlite3 - + if not os.path.exists(db_path): return [] - + results = [] - + try: conn = sqlite3.connect(db_path) cursor = conn.cursor() - + if agent_id is not None: cursor.execute(""" SELECT user_id, info, created_at @@ -1691,13 +1850,13 @@ class SimulationRunner: ORDER BY created_at DESC LIMIT ? """, (limit,)) - + for user_id, info_json, created_at in cursor.fetchall(): try: info = json.loads(info_json) if info_json else {} except json.JSONDecodeError: info = {"raw": info_json} - + results.append({ "agent_id": user_id, "response": info.get("response", info), @@ -1705,12 +1864,12 @@ class SimulationRunner: "timestamp": created_at, "platform": platform_name }) - + conn.close() - + except Exception as e: logger.error(f"Failed to load Interview history ({platform_name}): {e}") - + return results @classmethod @@ -1722,31 +1881,26 @@ class SimulationRunner: limit: int = 100 ) -> List[Dict[str, Any]]: """ - Lịch sử lấy danh sách câu trả lời của các câu hỏi với agents (Đọc từ DataBase db) - + Lấy lịch sử phỏng vấn từ database — dùng sau khi simulation đã chạy. + + Đọc từ: + twitter_simulation.db (nếu platform="twitter" hoặc None) + reddit_simulation.db (nếu platform="reddit" hoặc None) + + Kết quả sort theo timestamp giảm dần, giới hạn `limit` record. + Khi query kết hợp cả 2 platform, giới hạn tổng = limit (không phải limit×2). + Args: - simulation_id: Nhận dạng ID cho mỗi Simulation - platform: Chị định Nền tảng (reddit/twitter/None) - - "reddit": Chỉ trên reddit - - "twitter": Chỉ lấy records ghi được trên mạng xã hội twitter giả lập - - None: Kết hợp lấy logs của cả hai social network - agent_id: Cung cấp tùy chọn cho loại Agent qua ID - limit: Lượng dữ liệu load tối đa cho 1 request get query trên 1 nền tảng - - Returns: - Danh sách lưu vết record lịch sử phỏng vấn của các agents + platform: "twitter"/"reddit"/None (None = cả hai) + agent_id: Lọc theo agent cụ thể + limit: Số record tối đa trả về """ sim_dir = os.path.join(cls.RUN_STATE_DIR, simulation_id) - + results = [] - - # Xác nhận nền tảng cung cấp cho truy vấn - if platform in ("reddit", "twitter"): - platforms = [platform] - else: - # Nếu người dùng để trống, có nghĩa là gọi tất cả kết quả - platforms = ["twitter", "reddit"] - + + platforms = [platform] if platform in ("reddit", "twitter") else ["twitter", "reddit"] + for p in platforms: db_path = os.path.join(sim_dir, f"{p}_simulation.db") platform_results = cls._get_interview_history_from_db( @@ -1756,13 +1910,12 @@ class SimulationRunner: limit=limit ) results.extend(platform_results) - - # Sắp xếp chúng lại bằng thời gian gần nhất lên trước + + # Sort tổng hợp theo timestamp mới nhất results.sort(key=lambda x: x.get("timestamp", ""), reverse=True) - - # Cho trường hợp query kết hợp nhiều nền tảng, phải tiến hành gọt lấy đúng 1 giới hạn nhất định + + # Cắt về đúng limit khi query kết hợp nhiều platform if len(platforms) > 1 and len(results) > limit: results = results[:limit] - - return results + return results diff --git a/backend/app/services/zep_tools.py b/backend/app/services/zep_tools.py index 27fedc94..db1a35a5 100644 --- a/backend/app/services/zep_tools.py +++ b/backend/app/services/zep_tools.py @@ -1741,7 +1741,7 @@ Trả về định dạng JSON: {"questions": ["Câu hỏi 1", "Câu hỏi 2", . Bối cảnh mô phỏng: {simulation_requirement if simulation_requirement else "Not provided"} -Vai trò của đối tượng phỏng vấ: {', '.join(agent_roles)} +Vai trò của đối tượng phỏng vấn: {', '.join(agent_roles)} Hãy tạo từ 3-5 câu hỏi phỏng vấn.""" diff --git a/backend/app/utils/llm_client.py b/backend/app/utils/llm_client.py index 4d923cad..d1a9eb26 100644 --- a/backend/app/utils/llm_client.py +++ b/backend/app/utils/llm_client.py @@ -41,7 +41,7 @@ class LLMClient: self, messages: List[Dict[str, str]], temperature: float = 0.7, - max_tokens: int = 4096, + max_tokens: int = 16000, response_format: Optional[Dict] = None, metadata: Optional[Dict[str, Any]] = None, ) -> str: diff --git a/backend/app/utils/llm_cost.py b/backend/app/utils/llm_cost.py index 6b2e14dc..e71b3c94 100644 --- a/backend/app/utils/llm_cost.py +++ b/backend/app/utils/llm_cost.py @@ -22,6 +22,7 @@ from typing import Any, Dict, Optional MODEL_COSTS_PER_1M_TOKENS: Dict[str, Dict[str, float]] = { "Qwen/Qwen3.5-27B": {"input": 0.5, "output": 3.0}, + "Qwen/Qwen3.6-27B": {"input": 0.5, "output": 3.0}, "gemini-3-flash": {"input": 0.5, "output": 3.0}, "gemini-3.1-flash-lite": {"input": 0.25, "output": 1.5}, } diff --git a/backend/scripts/run_parallel_simulation.py b/backend/scripts/run_parallel_simulation.py index c0953f6a..ae9aa485 100644 --- a/backend/scripts/run_parallel_simulation.py +++ b/backend/scripts/run_parallel_simulation.py @@ -1,19 +1,56 @@ """ -Kịch bản mô phỏng song song hai nền tảng OASIS -Chạy đồng thời mô phỏng Twitter và Reddit, đọc cùng một tệp cấu hình +OASIS Script — Chạy simulation song song hai nền tảng Twitter + Reddit + +Vị trí trong pipeline: +───────────────────────────────────────────────────────────────────────────── + SimulationRunner.start_simulation() [Flask backend] + └─ subprocess.Popen(run_parallel_simulation.py --config ...) + │ (Script này chạy độc lập — không phải Flask process) + │ + ├── asyncio.gather(run_twitter_simulation, run_reddit_simulation) + │ └─ Hai platform chạy SONG SONG trong cùng event loop + │ + └── Sau khi xong: vào chế độ chờ lệnh IPC (Interview) +───────────────────────────────────────────────────────────────────────────── + +Luồng dữ liệu vào script: + simulation_config.json → thời gian, agent configs, initial posts + twitter_profiles.csv → hồ sơ agent Twitter (user_char = system prompt) + reddit_profiles.json → hồ sơ agent Reddit + +Luồng dữ liệu ra script: + twitter/actions.jsonl → mỗi dòng = 1 hành động của agent Twitter + reddit/actions.jsonl → mỗi dòng = 1 hành động của agent Reddit + simulation.log → stdout của script (được Flask redirect vào đây) + env_status.json → trạng thái IPC (alive/stopped) + twitter_simulation.db → SQLite database của OASIS Twitter + reddit_simulation.db → SQLite database của OASIS Reddit + +Cơ chế đọc action từ DB (quan trọng): + env.step(actions) KHÔNG trả về kết quả hữu ích — OASIS ghi action vào SQLite. + Script dùng rowid để track bản ghi đã xử lý và đọc action MỚI sau mỗi round. + Lý do dùng rowid thay vì created_at: Twitter dùng integer timestamp, + Reddit dùng datetime string — rowid là integer tự tăng, nhất quán trên cả hai. + +Hai LLM riêng biệt (tùy chọn): + Twitter → LLM_API_KEY (general) + Reddit → LLM_BOOST_API_KEY (boost, nếu có) → tăng throughput khi chạy song song + Nếu không có boost config → Reddit fallback về general LLM. Tính năng: -- Mô phỏng song song hai nền tảng (Twitter + Reddit) -- Không đóng môi trường ngay sau khi hoàn tất mô phỏng, chuyển sang chế độ chờ lệnh -- Hỗ trợ nhận lệnh Interview qua IPC -- Hỗ trợ phỏng vấn một Agent và phỏng vấn hàng loạt -- Hỗ trợ lệnh đóng môi trường từ xa +- Chạy song song hai nền tảng (Twitter + Reddit) qua asyncio.gather +- Sau simulation, KHÔNG đóng environment → vào chế độ chờ lệnh IPC +- Hỗ trợ Interview 1 agent, batch, và toàn bộ agent +- Hỗ trợ lệnh đóng environment từ xa (close_env) +- Tham số --no-wait để đóng ngay sau simulation (dùng khi test) +- Tham số --max-rounds để cắt ngắn số vòng (dùng khi test nhanh) Cách dùng: python run_parallel_simulation.py --config simulation_config.json - python run_parallel_simulation.py --config simulation_config.json --no-wait # Đóng ngay sau khi hoàn tất + python run_parallel_simulation.py --config simulation_config.json --no-wait python run_parallel_simulation.py --config simulation_config.json --twitter-only python run_parallel_simulation.py --config simulation_config.json --reddit-only + python run_parallel_simulation.py --config simulation_config.json --max-rounds 5 Cấu trúc log: sim_xxx/ @@ -21,13 +58,14 @@ Cấu trúc log: │ └── actions.jsonl # Log hành động nền tảng Twitter ├── reddit/ │ └── actions.jsonl # Log hành động nền tảng Reddit - ├── simulation.log # Log tiến trình mô phỏng chính - └── run_state.json # Trạng thái chạy (cho API truy vấn) + ├── simulation.log # Log tiến trình mô phỏng chính (stdout của script) + └── run_state.json # Trạng thái chạy (cho API truy vấn từ Flask) """ # ============================================================ -# Khắc phục vấn đề mã hóa trên Windows: đặt UTF-8 trước mọi import -# Mục tiêu là sửa lỗi thư viện OASIS bên thứ ba đọc file mà không chỉ định encoding +# Khắc phục vấn đề mã hóa trên Windows — PHẢI đặt trước mọi import +# Mục tiêu: sửa lỗi thư viện OASIS bên thứ ba đọc file không chỉ định encoding, +# gây UnicodeDecodeError khi có nội dung tiếng Việt/Unicode trong profile file. # ============================================================ import sys import os @@ -37,31 +75,28 @@ if sys.platform == 'win32': # Thiết lập này ảnh hưởng đến mọi lời gọi open() không chỉ định encoding os.environ.setdefault('PYTHONUTF8', '1') os.environ.setdefault('PYTHONIOENCODING', 'utf-8') - - # Cấu hình lại stdout/stderr sang UTF-8 (tránh lỗi hiển thị ký tự tiếng Trung trên console) + + # Cấu hình lại stdout/stderr sang UTF-8 if hasattr(sys.stdout, 'reconfigure'): sys.stdout.reconfigure(encoding='utf-8', errors='replace') if hasattr(sys.stderr, 'reconfigure'): sys.stderr.reconfigure(encoding='utf-8', errors='replace') - - # Ép đặt mã hóa mặc định (ảnh hưởng encoding mặc định của open()) - # Lưu ý: tốt nhất cần thiết lập khi Python khởi động, thiết lập lúc runtime có thể không hiệu quả - # Vì vậy cần monkey-patch thêm hàm open tích hợp + + # Monkey-patch hàm open() tích hợp để mặc định dùng UTF-8 + # Cần thiết vì PYTHONUTF8 không ảnh hưởng đến code đã chạy trước khi đặt env var. + # Lưu ý: thiết lập lúc runtime có thể không hiệu quả với một số thư viện, nhưng + # đây là phương án tốt nhất khi không kiểm soát được source code của OASIS. import builtins _original_open = builtins.open - - def _utf8_open(file, mode='r', buffering=-1, encoding=None, errors=None, + + def _utf8_open(file, mode='r', buffering=-1, encoding=None, errors=None, newline=None, closefd=True, opener=None): - """ - Wrapper cho hàm open(), mặc định dùng UTF-8 cho chế độ văn bản - Điều này giúp sửa lỗi thư viện bên thứ ba (như OASIS) đọc file không chỉ định encoding - """ - # Chỉ đặt encoding mặc định cho chế độ văn bản (không phải binary) khi chưa chỉ định encoding + """Wrapper cho hàm open(), mặc định dùng UTF-8 cho chế độ văn bản.""" if encoding is None and 'b' not in mode: encoding = 'utf-8' - return _original_open(file, mode, buffering, encoding, errors, + return _original_open(file, mode, buffering, encoding, errors, newline, closefd, opener) - + builtins.open = _utf8_open import argparse @@ -77,78 +112,85 @@ from datetime import datetime from typing import Dict, Any, List, Optional, Tuple -# Biến toàn cục dùng cho xử lý tín hiệu -_shutdown_event = None -_cleanup_done = False +# Biến toàn cục cho signal handling và shutdown coordination +_shutdown_event = None # asyncio.Event — được set khi nhận SIGTERM/SIGINT +_cleanup_done = False # Cờ chống gọi cleanup nhiều lần -# Thêm thư mục backend vào sys.path -# Script nằm cố định trong thư mục backend/scripts/ +# Thêm thư mục backend vào sys.path để import được các module nội bộ +# (action_logger, llm_cost_patch từ backend/scripts/) _scripts_dir = os.path.dirname(os.path.abspath(__file__)) _backend_dir = os.path.abspath(os.path.join(_scripts_dir, '..')) _project_root = os.path.abspath(os.path.join(_backend_dir, '..')) sys.path.insert(0, _scripts_dir) sys.path.insert(0, _backend_dir) -# Tải tệp .env ở thư mục gốc dự án (chứa các cấu hình như LLM_API_KEY) +# Tải .env ở thư mục gốc dự án (chứa LLM_API_KEY, LLM_BASE_URL, ...) +# Script chạy độc lập (subprocess) nên không thừa hưởng env từ Flask from dotenv import load_dotenv _env_file = os.path.join(_project_root, '.env') if os.path.exists(_env_file): load_dotenv(_env_file) print(f"Environment config loaded: {_env_file}") else: - # Thử tải backend/.env _backend_env = os.path.join(_backend_dir, '.env') if os.path.exists(_backend_env): load_dotenv(_backend_env) print(f"Environment config loaded: {_backend_env}") +# ============================================================================== +# Lọc cảnh báo thừa từ camel-ai về max_tokens +# ============================================================================== +# camel-ai log cảnh báo "Invalid or missing max_tokens" mỗi khi tạo LLM call. +# Đây là cảnh báo không quan trọng (chủ động không set max_tokens để model tự quyết định). +# Filter này ngăn cảnh báo đó làm ô nhiễm simulation.log. +# ============================================================================== + class MaxTokensWarningFilter(logging.Filter): - """Lọc cảnh báo max_tokens của camel-ai (chủ động không đặt max_tokens để model tự quyết định)""" - + """Lọc cảnh báo max_tokens của camel-ai.""" + def filter(self, record): - # Lọc log cảnh báo liên quan đến max_tokens if "max_tokens" in record.getMessage() and "Invalid or missing" in record.getMessage(): return False return True -# Thêm filter ngay khi module được nạp để bảo đảm có hiệu lực trước khi mã camel chạy +# Thêm filter ngay khi module được nạp — phải trước khi camel-ai chạy logging.getLogger().addFilter(MaxTokensWarningFilter()) def disable_oasis_logging(): """ - Tắt log chi tiết của thư viện OASIS - Log của OASIS quá dài dòng (ghi từng quan sát và hành động của agent), ở đây dùng action_logger riêng + Tắt log chi tiết của thư viện OASIS. + + OASIS ghi log cực kỳ dài dòng (từng quan sát + action của mỗi agent mỗi round). + MiroFish dùng action_logger riêng (ghi ra actions.jsonl) nên không cần OASIS log. + Chỉ để CRITICAL — chỉ ghi lỗi nghiêm trọng không thể bỏ qua. """ - # Tắt toàn bộ logger của OASIS oasis_loggers = [ "social.agent", - "social.twitter", + "social.twitter", "social.rec", "oasis.env", "table", ] - + for logger_name in oasis_loggers: logger = logging.getLogger(logger_name) - logger.setLevel(logging.CRITICAL) # Chỉ ghi lỗi nghiêm trọng + logger.setLevel(logging.CRITICAL) logger.handlers.clear() logger.propagate = False def init_logging_for_simulation(simulation_dir: str): """ - Khởi tạo cấu hình log cho mô phỏng - - Args: - simulation_dir: Đường dẫn thư mục mô phỏng + Khởi tạo cấu hình log cho mô phỏng. + + Dọn thư mục log/ cũ nếu còn từ lần chạy trước (cấu trúc log cũ đã deprecated). + Cấu trúc log hiện tại dùng twitter/actions.jsonl và reddit/actions.jsonl. """ - # Tắt log chi tiết của OASIS disable_oasis_logging() - - # Dọn thư mục log cũ (nếu tồn tại) + old_log_dir = os.path.join(simulation_dir, "log") if os.path.exists(old_log_dir): import shutil @@ -175,7 +217,13 @@ except ImportError as e: sys.exit(1) -# Action khả dụng trên Twitter (không gồm INTERVIEW; INTERVIEW chỉ kích hoạt thủ công qua ManualAction) +# ============================================================================== +# Action types khả dụng trên mỗi platform +# ============================================================================== +# INTERVIEW không có trong danh sách — nó chỉ được kích hoạt thủ công qua +# ManualAction (IPC command), không phải LLMAction trong vòng lặp simulation. +# ============================================================================== + TWITTER_ACTIONS = [ ActionType.CREATE_POST, ActionType.LIKE_POST, @@ -185,7 +233,6 @@ TWITTER_ACTIONS = [ ActionType.QUOTE_POST, ] -# Action khả dụng trên Reddit (không gồm INTERVIEW; INTERVIEW chỉ kích hoạt thủ công qua ManualAction) REDDIT_ACTIONS = [ ActionType.LIKE_POST, ActionType.DISLIKE_POST, @@ -203,25 +250,41 @@ REDDIT_ACTIONS = [ ] -# Hằng số liên quan đến IPC +# Tên file/folder cho cơ chế IPC IPC_COMMANDS_DIR = "ipc_commands" IPC_RESPONSES_DIR = "ipc_responses" ENV_STATUS_FILE = "env_status.json" + class CommandType: - """Hằng số loại lệnh""" + """Hằng số loại lệnh IPC""" INTERVIEW = "interview" BATCH_INTERVIEW = "batch_interview" CLOSE_ENV = "close_env" +# ============================================================================== +# CLASS: ParallelIPCHandler — Xử lý lệnh IPC trong khi environment còn sống +# ============================================================================== +# Sau khi simulation loop kết thúc, OASIS environment KHÔNG bị đóng. +# Script vào chế độ chờ — ParallelIPCHandler poll ipc_commands/ mỗi 0.5 giây. +# +# Cơ chế IPC (file-based): +# Flask ghi: ipc_commands/{uuid}.json +# {"command_id": "abc", "command_type": "interview", "args": {...}} +# Script phát hiện → xử lý → ghi: ipc_responses/{uuid}.json +# {"command_id": "abc", "status": "completed", "result": {...}} +# Flask poll ipc_responses/{uuid}.json cho đến khi có kết quả. +# Script xóa file command sau khi đã xử lý xong. +# +# Tại sao dùng file thay vì socket/queue? +# Subprocess và Flask process không chia sẻ memory. +# File là cách đơn giản nhất để giao tiếp cross-process mà không cần thêm dependency. +# ============================================================================== + class ParallelIPCHandler: - """ - Bộ xử lý lệnh IPC cho hai nền tảng - - Quản lý môi trường của cả hai nền tảng và xử lý lệnh Interview - """ - + """Bộ xử lý lệnh IPC cho hai nền tảng.""" + def __init__( self, simulation_dir: str, @@ -235,17 +298,24 @@ class ParallelIPCHandler: self.twitter_agent_graph = twitter_agent_graph self.reddit_env = reddit_env self.reddit_agent_graph = reddit_agent_graph - + self.commands_dir = os.path.join(simulation_dir, IPC_COMMANDS_DIR) self.responses_dir = os.path.join(simulation_dir, IPC_RESPONSES_DIR) self.status_file = os.path.join(simulation_dir, ENV_STATUS_FILE) - - # Đảm bảo thư mục tồn tại + os.makedirs(self.commands_dir, exist_ok=True) os.makedirs(self.responses_dir, exist_ok=True) - + def update_status(self, status: str): - """Cập nhật trạng thái môi trường""" + """ + Ghi env_status.json — Flask đọc file này để biết environment còn sống không. + + Được gọi tại 2 thời điểm: + 1. Khi vào chế độ chờ: update_status("alive") + 2. Khi thoát chế độ chờ: update_status("stopped") + + Flask's check_env_alive() đọc file này trước khi gửi bất kỳ IPC command nào. + """ with open(self.status_file, 'w', encoding='utf-8') as f: json.dump({ "status": status, @@ -253,32 +323,43 @@ class ParallelIPCHandler: "reddit_available": self.reddit_env is not None, "timestamp": datetime.now().isoformat() }, f, ensure_ascii=False, indent=2) - + def poll_command(self) -> Optional[Dict[str, Any]]: - """Poll để lấy lệnh đang chờ xử lý""" + """ + Kiểm tra xem có lệnh nào đang chờ xử lý không. + + Đọc file JSON cũ nhất trong ipc_commands/ (sort theo mtime). + Trả về None nếu không có lệnh nào, hoặc nếu file không hợp lệ. + + Flask đảm bảo chỉ ghi 1 command tại 1 thời điểm — không cần lock. + """ if not os.path.exists(self.commands_dir): return None - - # Lấy tệp lệnh (sắp xếp theo thời gian) + command_files = [] for filename in os.listdir(self.commands_dir): if filename.endswith('.json'): filepath = os.path.join(self.commands_dir, filename) command_files.append((filepath, os.path.getmtime(filepath))) - - command_files.sort(key=lambda x: x[1]) - + + command_files.sort(key=lambda x: x[1]) # Xử lý lệnh cũ nhất trước (FIFO) + for filepath, _ in command_files: try: with open(filepath, 'r', encoding='utf-8') as f: return json.load(f) except (json.JSONDecodeError, OSError): continue - + return None - + def send_response(self, command_id: str, status: str, result: Dict = None, error: str = None): - """Gửi phản hồi""" + """ + Ghi kết quả vào ipc_responses/{command_id}.json và xóa file lệnh. + + Flask poll file này (mỗi 0.5s, timeout 60s) để lấy kết quả interview. + Sau khi ghi response, xóa file command để tránh xử lý lại. + """ response = { "command_id": command_id, "status": status, @@ -286,47 +367,42 @@ class ParallelIPCHandler: "error": error, "timestamp": datetime.now().isoformat() } - + response_file = os.path.join(self.responses_dir, f"{command_id}.json") with open(response_file, 'w', encoding='utf-8') as f: json.dump(response, f, ensure_ascii=False, indent=2) - - # Xóa tệp lệnh + + # Xóa file command sau khi đã xử lý xong command_file = os.path.join(self.commands_dir, f"{command_id}.json") try: os.remove(command_file) except OSError: pass - + def _get_env_and_graph(self, platform: str): - """ - Lấy env và agent_graph của nền tảng được chỉ định - - Args: - platform: Tên nền tảng ("twitter" hoặc "reddit") - - Returns: - (env, agent_graph, platform_name) hoặc (None, None, None) - """ + """Lấy (env, agent_graph) của platform được chỉ định. Trả về (None, None, None) nếu không có.""" if platform == "twitter" and self.twitter_env: return self.twitter_env, self.twitter_agent_graph, "twitter" elif platform == "reddit" and self.reddit_env: return self.reddit_env, self.reddit_agent_graph, "reddit" else: return None, None, None - + async def _interview_single_platform(self, agent_id: int, prompt: str, platform: str) -> Dict[str, Any]: """ - Thực thi Interview trên một nền tảng - - Returns: - Dictionary chứa kết quả hoặc lỗi + Thực thi Interview trên 1 platform bằng ManualAction. + + ManualAction(INTERVIEW) là cách inject hành động thủ công vào OASIS — + thay vì để LLM quyết định action, ép agent trả lời câu hỏi cụ thể. + + OASIS ghi kết quả vào bảng `trace` trong SQLite database. + _get_interview_result() đọc bản ghi mới nhất từ đó. """ env, agent_graph, actual_platform = self._get_env_and_graph(platform) - + if not env or not agent_graph: return {"platform": platform, "error": f"{platform} platform is unavailable"} - + try: agent = agent_graph.get_agent(agent_id) interview_action = ManualAction( @@ -334,35 +410,30 @@ class ParallelIPCHandler: action_args={"prompt": prompt} ) actions = {agent: interview_action} - await env.step(actions) - + await env.step(actions) # OASIS chạy action → ghi vào DB + result = self._get_interview_result(agent_id, actual_platform) result["platform"] = actual_platform return result - + except Exception as e: return {"platform": platform, "error": str(e)} - + async def handle_interview(self, command_id: str, agent_id: int, prompt: str, platform: str = None) -> bool: """ - Xử lý lệnh phỏng vấn một Agent - - Args: - command_id: ID lệnh - agent_id: Agent ID - prompt: Câu hỏi phỏng vấn - platform: Nền tảng chỉ định (tùy chọn) - - "twitter": Chỉ phỏng vấn trên Twitter - - "reddit": Chỉ phỏng vấn trên Reddit - - None/không chỉ định: Phỏng vấn đồng thời cả hai nền tảng, trả kết quả gộp - + Xử lý lệnh phỏng vấn 1 agent. + + Nếu platform được chỉ định: phỏng vấn trên platform đó thôi. + Nếu platform=None: phỏng vấn song song cả 2 platform (asyncio.gather), + trả về kết quả gộp — hữu ích khi muốn so sánh phản ứng của agent + trên Twitter vs Reddit. + Returns: - True là thành công, False là thất bại + True nếu ít nhất 1 platform thành công, False nếu tất cả fail. """ - # Nếu có chỉ định nền tảng, chỉ phỏng vấn trên nền tảng đó if platform in ("twitter", "reddit"): - result = await self._interview_single_platform(agent_id, prompt, platform) - + result = await self._interview_single_platform(agent_id, prompt, platform) + if "error" in result: self.send_response(command_id, "failed", error=result["error"]) print(f" Interview failed: agent_id={agent_id}, platform={platform}, error={result['error']}") @@ -371,39 +442,38 @@ class ParallelIPCHandler: self.send_response(command_id, "completed", result=result) print(f" Interview completed: agent_id={agent_id}, platform={platform}") return True - - # Không chỉ định nền tảng: phỏng vấn đồng thời hai nền tảng + + # Không chỉ định platform → phỏng vấn cả 2 song song if not self.twitter_env and not self.reddit_env: self.send_response(command_id, "failed", error="No simulation environment available") return False - + results = { "agent_id": agent_id, "prompt": prompt, "platforms": {} } success_count = 0 - - # Phỏng vấn song song hai nền tảng + tasks = [] platforms_to_interview = [] - + if self.twitter_env: tasks.append(self._interview_single_platform(agent_id, prompt, "twitter")) platforms_to_interview.append("twitter") - + if self.reddit_env: tasks.append(self._interview_single_platform(agent_id, prompt, "reddit")) platforms_to_interview.append("reddit") - - # Chạy song song + + # Chạy song song — asyncio.gather không block nhau platform_results = await asyncio.gather(*tasks) - + for platform_name, platform_result in zip(platforms_to_interview, platform_results): results["platforms"][platform_name] = platform_result if "error" not in platform_result: success_count += 1 - + if success_count > 0: self.send_response(command_id, "completed", result=results) print(f" Interview completed: agent_id={agent_id}, successful platforms={success_count}/{len(platforms_to_interview)}") @@ -413,24 +483,26 @@ class ParallelIPCHandler: self.send_response(command_id, "failed", error="; ".join(errors)) print(f" Interview failed: agent_id={agent_id}, all platforms failed") return False - + async def handle_batch_interview(self, command_id: str, interviews: List[Dict], platform: str = None) -> bool: """ - Xử lý lệnh phỏng vấn hàng loạt - - Args: - command_id: ID lệnh - interviews: [{"agent_id": int, "prompt": str, "platform": str(optional)}, ...] - platform: Nền tảng mặc định (có thể bị ghi đè ở từng mục interview) - - "twitter": Chỉ phỏng vấn Twitter - - "reddit": Chỉ phỏng vấn Reddit - - None/không chỉ định: Mỗi Agent được phỏng vấn trên cả hai nền tảng + Xử lý lệnh phỏng vấn hàng loạt (batch) — hiệu quả hơn gọi N lần. + + Chiến lược tối ưu: nhóm tất cả agent cùng platform thành 1 env.step() call, + thay vì gọi env.step() riêng lẻ cho từng agent. + → Giảm số lần call asyncio, tăng throughput. + + Phân nhóm: + - platform="twitter" → twitter_interviews + - platform="reddit" → reddit_interviews + - platform=None → both_platforms_interviews → mở rộng vào cả 2 danh sách + + Mỗi nhóm được thực thi trong 1 env.step() call với dict {agent: ManualAction}. """ - # Nhóm theo nền tảng twitter_interviews = [] reddit_interviews = [] - both_platforms_interviews = [] # Cần phỏng vấn đồng thời hai nền tảng - + both_platforms_interviews = [] + for interview in interviews: item_platform = interview.get("platform", platform) if item_platform == "twitter": @@ -438,19 +510,17 @@ class ParallelIPCHandler: elif item_platform == "reddit": reddit_interviews.append(interview) else: - # Không chỉ định nền tảng: phỏng vấn cả hai nền tảng both_platforms_interviews.append(interview) - - # Tách both_platforms_interviews vào hai nền tảng + if both_platforms_interviews: if self.twitter_env: twitter_interviews.extend(both_platforms_interviews) if self.reddit_env: reddit_interviews.extend(both_platforms_interviews) - + results = {} - - # Xử lý phỏng vấn trên nền tảng Twitter + + # Xử lý Twitter batch — 1 env.step() cho tất cả agent Twitter if twitter_interviews and self.twitter_env: try: twitter_actions = {} @@ -465,10 +535,10 @@ class ParallelIPCHandler: ) except Exception as e: print(f" Warning: Cannot get Twitter Agent {agent_id}: {e}") - + if twitter_actions: await self.twitter_env.step(twitter_actions) - + for interview in twitter_interviews: agent_id = interview.get("agent_id") result = self._get_interview_result(agent_id, "twitter") @@ -476,8 +546,8 @@ class ParallelIPCHandler: results[f"twitter_{agent_id}"] = result except Exception as e: print(f" Twitter batch interview failed: {e}") - - # Xử lý phỏng vấn trên nền tảng Reddit + + # Xử lý Reddit batch — tương tự if reddit_interviews and self.reddit_env: try: reddit_actions = {} @@ -492,10 +562,10 @@ class ParallelIPCHandler: ) except Exception as e: print(f" Warning: Cannot get Reddit Agent {agent_id}: {e}") - + if reddit_actions: await self.reddit_env.step(reddit_actions) - + for interview in reddit_interviews: agent_id = interview.get("agent_id") result = self._get_interview_result(agent_id, "reddit") @@ -503,7 +573,7 @@ class ParallelIPCHandler: results[f"reddit_{agent_id}"] = result except Exception as e: print(f" Reddit batch interview failed: {e}") - + if results: self.send_response(command_id, "completed", result={ "interviews_count": len(results), @@ -514,25 +584,29 @@ class ParallelIPCHandler: else: self.send_response(command_id, "failed", error="No successful interviews") return False - + def _get_interview_result(self, agent_id: int, platform: str) -> Dict[str, Any]: - """Lấy kết quả Interview mới nhất từ cơ sở dữ liệu""" + """ + Đọc kết quả interview mới nhất từ bảng `trace` trong SQLite. + + OASIS ghi kết quả ManualAction(INTERVIEW) vào bảng trace với action='interview'. + Truy vấn ORDER BY created_at DESC LIMIT 1 → lấy bản ghi mới nhất. + """ db_path = os.path.join(self.simulation_dir, f"{platform}_simulation.db") - + result = { "agent_id": agent_id, "response": None, "timestamp": None } - + if not os.path.exists(db_path): return result - + try: conn = sqlite3.connect(db_path) cursor = conn.cursor() - - # Truy vấn bản ghi Interview mới nhất + cursor.execute(""" SELECT user_id, info, created_at FROM trace @@ -540,7 +614,7 @@ class ParallelIPCHandler: ORDER BY created_at DESC LIMIT 1 """, (ActionType.INTERVIEW.value, agent_id)) - + row = cursor.fetchone() if row: user_id, info_json, created_at = row @@ -550,31 +624,37 @@ class ParallelIPCHandler: result["timestamp"] = created_at except json.JSONDecodeError: result["response"] = info_json - + conn.close() - + except Exception as e: print(f" Failed to read interview result: {e}") - + return result - + async def process_commands(self) -> bool: """ - Xử lý tất cả lệnh đang chờ - + Xử lý 1 lệnh đang chờ (nếu có). + + Được gọi trong vòng lặp chờ mỗi 0.5 giây. + Dispatch theo command_type: + "interview" → handle_interview() + "batch_interview" → handle_batch_interview() + "close_env" → ghi response → trả về False (thoát vòng lặp) + Returns: - True để tiếp tục chạy, False để thoát + True để tiếp tục vòng lặp, False để thoát (close_env hoặc unknown) """ command = self.poll_command() if not command: return True - + command_id = command.get("command_id") command_type = command.get("command_type") args = command.get("args", {}) - + print(f"\nReceived IPC command: {command_type}, id={command_id}") - + if command_type == CommandType.INTERVIEW: await self.handle_interview( command_id, @@ -583,7 +663,7 @@ class ParallelIPCHandler: args.get("platform") ) return True - + elif command_type == CommandType.BATCH_INTERVIEW: await self.handle_batch_interview( command_id, @@ -591,27 +671,37 @@ class ParallelIPCHandler: args.get("platform") ) return True - + elif command_type == CommandType.CLOSE_ENV: print("Received close environment command") self.send_response(command_id, "completed", result={"message": "Environment will close soon"}) - return False - + return False # Thoát vòng lặp chờ + else: self.send_response(command_id, "failed", error=f"Unknown command type: {command_type}") return True def load_config(config_path: str) -> Dict[str, Any]: - """Tải tệp cấu hình""" + """Đọc simulation_config.json từ đường dẫn đã chỉ định.""" with open(config_path, 'r', encoding='utf-8') as f: return json.load(f) -# Các loại action không cốt lõi cần lọc (giá trị phân tích thấp) +# ============================================================================== +# Lọc action và ánh xạ tên +# ============================================================================== +# FILTERED_ACTIONS: các loại action không có giá trị phân tích — bỏ qua khi ghi log. +# refresh = agent "tải lại" feed (không tạo nội dung mới) +# sign_up = đăng ký tài khoản (chỉ xảy ra 1 lần lúc khởi tạo) +# ============================================================================== + FILTERED_ACTIONS = {'refresh', 'sign_up'} -# Bảng ánh xạ loại action (tên trong DB -> tên chuẩn) +# ACTION_TYPE_MAP: ánh xạ từ tên trong SQLite database của OASIS +# sang tên chuẩn viết hoa dùng trong actions.jsonl. +# OASIS lưu dạng lowercase trong DB (ví dụ: 'create_post'), +# nhưng actions.jsonl dùng UPPER_SNAKE_CASE ('CREATE_POST') để nhất quán. ACTION_TYPE_MAP = { 'create_post': 'CREATE_POST', 'like_post': 'LIKE_POST', @@ -633,25 +723,24 @@ ACTION_TYPE_MAP = { def get_agent_names_from_config(config: Dict[str, Any]) -> Dict[int, str]: """ - Lấy ánh xạ agent_id -> entity_name từ simulation_config - - Mục tiêu là hiển thị tên thực thể thật trong actions.jsonl thay vì mã như "Agent_0" - - Args: - config: Nội dung của simulation_config.json - - Returns: - Dictionary ánh xạ agent_id -> entity_name + Lấy ánh xạ agent_id → entity_name từ simulation_config.json. + + OASIS mặc định dùng tên "Agent_0", "Agent_1", ... trong DB. + Hàm này map agent_id sang tên thực thể (ví dụ: "Trần Văn An") + để actions.jsonl hiển thị tên thật thay vì "Agent_0". + + Nếu config không có entry cho agent nào, phần code gọi hàm này + sẽ fallback về tên OASIS mặc định của agent đó. """ agent_names = {} agent_configs = config.get("agent_configs", []) - + for agent_config in agent_configs: agent_id = agent_config.get("agent_id") entity_name = agent_config.get("entity_name", f"Agent_{agent_id}") if agent_id is not None: agent_names[agent_id] = entity_name - + return agent_names @@ -661,52 +750,61 @@ def fetch_new_actions_from_db( agent_names: Dict[int, str] ) -> Tuple[List[Dict[str, Any]], int]: """ - Lấy bản ghi action mới từ DB và bổ sung ngữ cảnh đầy đủ - + Đọc các action MỚI từ SQLite database kể từ last_rowid. + + Tại sao dùng rowid thay vì created_at? + Twitter lưu created_at dạng integer timestamp (Unix epoch). + Reddit lưu created_at dạng ISO 8601 string. + → So sánh/sort theo created_at không nhất quán giữa hai platform. + rowid là integer auto-increment của SQLite — đơn giản và nhất quán. + + Chiến lược: + 1. SELECT WHERE rowid > last_rowid → chỉ lấy bản ghi MỚI + 2. Cập nhật last_rowid = rowid lớn nhất trong batch này + 3. Lần tiếp theo gọi lại với last_rowid mới → không đọc lại bản ghi cũ + + Mỗi action được enrich thêm context (nội dung bài viết, tên tác giả) + bằng _enrich_action_context() để actions.jsonl có đủ thông tin. + Args: - db_path: Đường dẫn tệp cơ sở dữ liệu - last_rowid: Giá trị rowid lớn nhất đã đọc trước đó (dùng rowid thay vì created_at vì định dạng created_at khác nhau giữa nền tảng) - agent_names: Ánh xạ agent_id -> agent_name - + db_path: Đường dẫn tệp SQLite + last_rowid: rowid lớn nhất đã đọc lần trước (0 = lần đầu, đọc tất cả) + agent_names: Ánh xạ agent_id → tên thật + Returns: (actions_list, new_last_rowid) - - actions_list: Danh sách action, mỗi phần tử gồm agent_id, agent_name, action_type, action_args (có ngữ cảnh) - - new_last_rowid: Giá trị rowid lớn nhất mới """ actions = [] new_last_rowid = last_rowid - + if not os.path.exists(db_path): return actions, new_last_rowid - + try: conn = sqlite3.connect(db_path) cursor = conn.cursor() - - # Dùng rowid để theo dõi bản ghi đã xử lý (rowid là trường tự tăng tích hợp của SQLite) - # Cách này tránh vấn đề khác biệt định dạng created_at (Twitter dùng số nguyên, Reddit dùng chuỗi datetime) + cursor.execute(""" SELECT rowid, user_id, action, info FROM trace WHERE rowid > ? ORDER BY rowid ASC """, (last_rowid,)) - + for rowid, user_id, action, info_json in cursor.fetchall(): - # Cập nhật rowid lớn nhất - new_last_rowid = rowid - - # Lọc action không cốt lõi + new_last_rowid = rowid # Cập nhật cursor cho lần tiếp theo + + # Bỏ qua action không có giá trị phân tích if action in FILTERED_ACTIONS: continue - - # Parse tham số action + try: action_args = json.loads(info_json) if info_json else {} except json.JSONDecodeError: action_args = {} - - # Tinh gọn action_args, chỉ giữ trường quan trọng (giữ nguyên nội dung, không cắt) + + # Chỉ giữ các field quan trọng của action_args + # (bỏ các field nội bộ của OASIS không cần thiết cho phân tích) simplified_args = {} if 'content' in action_args: simplified_args['content'] = action_args['content'] @@ -726,24 +824,23 @@ def fetch_new_actions_from_db( simplified_args['like_id'] = action_args['like_id'] if 'dislike_id' in action_args: simplified_args['dislike_id'] = action_args['dislike_id'] - - # Chuyển tên loại action + action_type = ACTION_TYPE_MAP.get(action, action.upper()) - - # Bổ sung ngữ cảnh (nội dung bài viết, tên người dùng...) + + # Bổ sung context (nội dung bài/comment, tên tác giả) để log có nghĩa hơn _enrich_action_context(cursor, action_type, simplified_args, agent_names) - + actions.append({ 'agent_id': user_id, 'agent_name': agent_names.get(user_id, f'Agent_{user_id}'), 'action_type': action_type, 'action_args': simplified_args, }) - + conn.close() except Exception as e: print(f"Failed to read actions from database: {e}") - + return actions, new_last_rowid @@ -754,16 +851,24 @@ def _enrich_action_context( agent_names: Dict[int, str] ) -> None: """ - Bổ sung ngữ cảnh cho action (nội dung bài viết, tên người dùng...) - - Args: - cursor: DB cursor - action_type: Loại action - action_args: Tham số action (sẽ bị cập nhật) - agent_names: Ánh xạ agent_id -> agent_name + Bổ sung context vào action_args bằng cách join thêm thông tin từ DB. + + Tại sao cần enrich? + OASIS ghi action_args dạng thô: {"post_id": 42} — không có nội dung bài. + Khi đọc actions.jsonl để phân tích, cần biết agent đã like BÀI NÀO, ai viết. + Enrich thêm post_content và post_author_name để log có ý nghĩa. + + Các loại được enrich: + - LIKE/DISLIKE_POST → thêm nội dung bài + tên tác giả + - REPOST → thêm nội dung + tác giả bài gốc + - QUOTE_POST → thêm nội dung bài gốc + phần quote của agent + - FOLLOW/MUTE → thêm tên người dùng được follow/mute + - LIKE/DISLIKE_COMMENT → thêm nội dung comment + tên tác giả + - CREATE_COMMENT → thêm nội dung bài được comment vào + + Lỗi trong enrich KHÔNG ngăn luồng chính — silently pass. """ try: - # Like/dislike bài viết: bổ sung nội dung bài và tác giả if action_type in ('LIKE_POST', 'DISLIKE_POST'): post_id = action_args.get('post_id') if post_id: @@ -771,12 +876,10 @@ def _enrich_action_context( if post_info: action_args['post_content'] = post_info.get('content', '') action_args['post_author_name'] = post_info.get('author_name', '') - - # Repost: bổ sung nội dung và tác giả bài gốc + elif action_type == 'REPOST': new_post_id = action_args.get('new_post_id') if new_post_id: - # original_post_id của bài repost trỏ đến bài gốc cursor.execute(""" SELECT original_post_id FROM post WHERE post_id = ? """, (new_post_id,)) @@ -787,19 +890,17 @@ def _enrich_action_context( if original_info: action_args['original_content'] = original_info.get('content', '') action_args['original_author_name'] = original_info.get('author_name', '') - - # Quote post: bổ sung nội dung bài gốc, tác giả và phần quote + elif action_type == 'QUOTE_POST': quoted_id = action_args.get('quoted_id') new_post_id = action_args.get('new_post_id') - + if quoted_id: original_info = _get_post_info(cursor, quoted_id, agent_names) if original_info: action_args['original_content'] = original_info.get('content', '') action_args['original_author_name'] = original_info.get('author_name', '') - - # Lấy nội dung quote của bài trích dẫn (quote_content) + if new_post_id: cursor.execute(""" SELECT quote_content FROM post WHERE post_id = ? @@ -807,12 +908,10 @@ def _enrich_action_context( row = cursor.fetchone() if row and row[0]: action_args['quote_content'] = row[0] - - # Follow user: bổ sung tên người dùng được follow + elif action_type == 'FOLLOW': follow_id = action_args.get('follow_id') if follow_id: - # Lấy followee_id từ bảng follow cursor.execute(""" SELECT followee_id FROM follow WHERE follow_id = ? """, (follow_id,)) @@ -822,17 +921,14 @@ def _enrich_action_context( target_name = _get_user_name(cursor, followee_id, agent_names) if target_name: action_args['target_user_name'] = target_name - - # Mute user: bổ sung tên người dùng bị mute + elif action_type == 'MUTE': - # Lấy user_id hoặc target_id từ action_args target_id = action_args.get('user_id') or action_args.get('target_id') if target_id: target_name = _get_user_name(cursor, target_id, agent_names) if target_name: action_args['target_user_name'] = target_name - - # Like/dislike comment: bổ sung nội dung comment và tác giả + elif action_type in ('LIKE_COMMENT', 'DISLIKE_COMMENT'): comment_id = action_args.get('comment_id') if comment_id: @@ -840,8 +936,7 @@ def _enrich_action_context( if comment_info: action_args['comment_content'] = comment_info.get('content', '') action_args['comment_author_name'] = comment_info.get('author_name', '') - - # Create comment: bổ sung thông tin bài viết được bình luận + elif action_type == 'CREATE_COMMENT': post_id = action_args.get('post_id') if post_id: @@ -849,10 +944,9 @@ def _enrich_action_context( if post_info: action_args['post_content'] = post_info.get('content', '') action_args['post_author_name'] = post_info.get('author_name', '') - + except Exception as e: - # Bổ sung ngữ cảnh thất bại không ảnh hưởng luồng chính - print(f"Failed to enrich action context: {e}") + pass # Enrich thất bại không ảnh hưởng luồng chính def _get_post_info( @@ -860,17 +954,7 @@ def _get_post_info( post_id: int, agent_names: Dict[int, str] ) -> Optional[Dict[str, str]]: - """ - Lấy thông tin bài viết - - Args: - cursor: DB cursor - post_id: Post ID - agent_names: Ánh xạ agent_id -> agent_name - - Returns: - Dictionary chứa content và author_name, hoặc None - """ + """Lấy {content, author_name} của một bài viết từ DB.""" try: cursor.execute(""" SELECT p.content, p.user_id, u.agent_id @@ -883,18 +967,16 @@ def _get_post_info( content = row[0] or '' user_id = row[1] agent_id = row[2] - - # Ưu tiên dùng tên từ agent_names + author_name = '' if agent_id is not None and agent_id in agent_names: author_name = agent_names[agent_id] elif user_id: - # Lấy tên từ bảng user cursor.execute("SELECT name, user_name FROM user WHERE user_id = ?", (user_id,)) user_row = cursor.fetchone() if user_row: author_name = user_row[0] or user_row[1] or '' - + return {'content': content, 'author_name': author_name} except Exception: pass @@ -906,17 +988,7 @@ def _get_user_name( user_id: int, agent_names: Dict[int, str] ) -> Optional[str]: - """ - Lấy tên người dùng - - Args: - cursor: DB cursor - user_id: User ID - agent_names: Ánh xạ agent_id -> agent_name - - Returns: - Tên người dùng, hoặc None - """ + """Lấy tên hiển thị của một user từ DB. Ưu tiên entity_name từ agent_names.""" try: cursor.execute(""" SELECT agent_id, name, user_name FROM user WHERE user_id = ? @@ -926,8 +998,7 @@ def _get_user_name( agent_id = row[0] name = row[1] user_name = row[2] - - # Ưu tiên dùng tên từ agent_names + if agent_id is not None and agent_id in agent_names: return agent_names[agent_id] return name or user_name or '' @@ -941,17 +1012,7 @@ def _get_comment_info( comment_id: int, agent_names: Dict[int, str] ) -> Optional[Dict[str, str]]: - """ - Lấy thông tin bình luận - - Args: - cursor: DB cursor - comment_id: Comment ID - agent_names: Ánh xạ agent_id -> agent_name - - Returns: - Dictionary chứa content và author_name, hoặc None - """ + """Lấy {content, author_name} của một comment từ DB.""" try: cursor.execute(""" SELECT c.content, c.user_id, u.agent_id @@ -964,18 +1025,16 @@ def _get_comment_info( content = row[0] or '' user_id = row[1] agent_id = row[2] - - # Ưu tiên dùng tên từ agent_names + author_name = '' if agent_id is not None and agent_id in agent_names: author_name = agent_names[agent_id] elif user_id: - # Lấy tên từ bảng user cursor.execute("SELECT name, user_name FROM user WHERE user_id = ?", (user_id,)) user_row = cursor.fetchone() if user_row: author_name = user_row[0] or user_row[1] or '' - + return {'content': content, 'author_name': author_name} except Exception: pass @@ -984,17 +1043,23 @@ def _get_comment_info( def create_model(config: Dict[str, Any], use_boost: bool = False): """ - Tạo mô hình LLM - - Hỗ trợ cấu hình hai LLM để tăng tốc khi mô phỏng song song: - - Cấu hình chung: LLM_API_KEY, LLM_BASE_URL, LLM_MODEL_NAME - - Cấu hình tăng tốc (tùy chọn): LLM_BOOST_API_KEY, LLM_BOOST_BASE_URL, LLM_BOOST_MODEL_NAME - - Nếu có cấu hình LLM tăng tốc, mỗi nền tảng có thể dùng nhà cung cấp API khác nhau để tăng khả năng song song. - - Args: - config: Dictionary cấu hình mô phỏng - use_boost: Có dùng cấu hình LLM tăng tốc hay không (nếu khả dụng) + Tạo LLM model cho OASIS agent. + + Hỗ trợ 2 cấu hình LLM để tối ưu throughput khi chạy song song: + - General LLM: LLM_API_KEY + LLM_BASE_URL + LLM_MODEL_NAME + - Boost LLM: LLM_BOOST_API_KEY + LLM_BOOST_BASE_URL + LLM_BOOST_MODEL_NAME + + Twitter dùng General LLM (use_boost=False). + Reddit dùng Boost LLM nếu có (use_boost=True), fallback về General nếu không. + + Lý do tách hai LLM: khi Twitter và Reddit chạy asyncio.gather song song, + nếu cùng dùng 1 API key sẽ bị rate limit. Dùng 2 API key khác nhau (hoặc + 2 provider khác nhau) giúp tăng throughput gấp đôi. + + Validation: + - API key không được là placeholder ("your_api_key_here") + - Base URL phải bắt đầu bằng http:// hoặc https:// + - Nếu boost config không hợp lệ → in cảnh báo → fallback về general """ def _is_placeholder(value: str) -> bool: v = (value or "").strip().lower() @@ -1004,7 +1069,6 @@ def create_model(config: Dict[str, Any], use_boost: bool = False): v = (value or "").strip().lower() return v.startswith("http://") or v.startswith("https://") - # Kiểm tra có cấu hình tăng tốc hợp lệ không boost_api_key = os.environ.get("LLM_BOOST_API_KEY", "") boost_base_url = os.environ.get("LLM_BOOST_BASE_URL", "") boost_model = os.environ.get("LLM_BOOST_MODEL_NAME", "") @@ -1017,43 +1081,38 @@ def create_model(config: Dict[str, Any], use_boost: bool = False): and not _is_placeholder(boost_model) and _is_http_url(boost_base_url) ) - - # Chọn LLM theo tham số và trạng thái cấu hình + if use_boost and has_boost_config: - # Dùng cấu hình tăng tốc llm_api_key = boost_api_key llm_base_url = boost_base_url llm_model = boost_model or os.environ.get("LLM_MODEL_NAME", "") config_label = "[Boost LLM]" else: - # Dùng cấu hình chung if use_boost and not has_boost_config: print("[Boost LLM] Invalid or placeholder boost config, fallback to [General LLM].") llm_api_key = os.environ.get("LLM_API_KEY", "") llm_base_url = os.environ.get("LLM_BASE_URL", "") llm_model = os.environ.get("LLM_MODEL_NAME", "") config_label = "[General LLM]" - - # Nếu .env không có model name, dùng config làm phương án dự phòng + if not llm_model: llm_model = config.get("llm_model", "gpt-4o-mini") - - # Thiết lập biến môi trường cần thiết cho camel-ai + if llm_api_key: os.environ["OPENAI_API_KEY"] = llm_api_key - + if not os.environ.get("OPENAI_API_KEY"): raise ValueError("Missing API key config. Please set LLM_API_KEY in the project root .env file") - + if llm_base_url: if not _is_http_url(llm_base_url): raise ValueError( f"Invalid LLM base URL: {llm_base_url}. It must start with http:// or https://" ) os.environ["OPENAI_API_BASE_URL"] = llm_base_url - + print(f"{config_label} model={llm_model}, base_url={llm_base_url[:40] if llm_base_url else 'default'}...") - + return ModelFactory.create( model_platform=ModelPlatformType.OPENAI, model_type=llm_model, @@ -1067,42 +1126,68 @@ def get_active_agents_for_round( current_hour: int, round_num: int ) -> List: - """Quyết định Agent nào được kích hoạt trong round hiện tại dựa trên thời gian và cấu hình""" + """ + Quyết định agent nào được kích hoạt trong round hiện tại. + + Thuật toán 3 bước: + + Bước 1 — Tính số lượng agent cần kích hoạt: + target_count = random(base_min, base_max) × activity_multiplier + Trong đó: + peak_hours → multiplier = 1.5 (đông nhất, ví dụ 19-22h) + off_peak_hours → multiplier = 0.05 (vắng nhất, ví dụ 0-5h) + giờ khác → multiplier = 1.0 + + Bước 2 — Lọc agent ứng cử viên: + Với mỗi agent trong agent_configs: + - current_hour ∈ active_hours? → agent có thể hoạt động giờ này + - random() < activity_level? → agent "thức dậy" trong round này + → Chỉ agent thỏa mãn CẢ HAI điều kiện mới vào candidates + + Bước 3 — Chọn ngẫu nhiên: + random.sample(candidates, min(target_count, len(candidates))) + → Đảm bảo không chọn nhiều hơn số candidates thực tế + + Kết quả: List[(agent_id, agent_object)] — agent_object để truyền vào env.step() + """ time_config = config.get("time_config", {}) agent_configs = config.get("agent_configs", []) - + base_min = time_config.get("agents_per_hour_min", 5) base_max = time_config.get("agents_per_hour_max", 20) - + peak_hours = time_config.get("peak_hours", [9, 10, 11, 14, 15, 20, 21, 22]) off_peak_hours = time_config.get("off_peak_hours", [0, 1, 2, 3, 4, 5]) - + if current_hour in peak_hours: multiplier = time_config.get("peak_activity_multiplier", 1.5) elif current_hour in off_peak_hours: multiplier = time_config.get("off_peak_activity_multiplier", 0.3) else: multiplier = 1.0 - + target_count = int(random.uniform(base_min, base_max) * multiplier) - + + # Lọc ứng cử viên: đúng giờ + random activity_level candidates = [] for cfg in agent_configs: agent_id = cfg.get("agent_id", 0) active_hours = cfg.get("active_hours", list(range(8, 23))) activity_level = cfg.get("activity_level", 0.5) - + if current_hour not in active_hours: continue - + if random.random() < activity_level: candidates.append(agent_id) - + + # Chọn ngẫu nhiên từ danh sách ứng cử viên selected_ids = random.sample( - candidates, + candidates, min(target_count, len(candidates)) ) if candidates else [] - + + # Lấy agent object từ env (để truyền vào env.step() sau đó) active_agents = [] for agent_id in selected_ids: try: @@ -1110,96 +1195,125 @@ def get_active_agents_for_round( active_agents.append((agent_id, agent)) except Exception: pass - + return active_agents class PlatformSimulation: - """Container kết quả mô phỏng theo nền tảng""" + """ + Container kết quả của một platform simulation. + + Được trả về bởi run_twitter_simulation() và run_reddit_simulation(). + env và agent_graph được giữ lại sau khi vòng lặp kết thúc + để ParallelIPCHandler có thể dùng cho Interview. + """ def __init__(self): - self.env = None - self.agent_graph = None - self.total_actions = 0 + self.env = None # OASIS environment object + self.agent_graph = None # Agent graph (dùng để get_agent(id)) + self.total_actions = 0 # Tổng số action đã ghi được async def run_twitter_simulation( - config: Dict[str, Any], + config: Dict[str, Any], simulation_dir: str, action_logger: Optional[PlatformActionLogger] = None, main_logger: Optional[SimulationLogManager] = None, max_rounds: Optional[int] = None ) -> PlatformSimulation: - """Chạy mô phỏng Twitter - + """ + Khởi tạo và chạy simulation Twitter. + + Luồng chính: + 1. Tạo LLM model (General LLM) + 2. generate_twitter_agent_graph() — tải agent từ twitter_profiles.csv + 3. oasis.make() — tạo Twitter environment + SQLite DB + 4. env.reset() — khởi tạo environment + 5. Round 0: đăng initial_posts (ManualAction) + 6. Vòng lặp Round 1..N: + a. get_active_agents_for_round() — chọn agent dựa trên giờ + activity_level + b. env.step({agent: LLMAction()}) — OASIS cho agent LLM chọn action + c. fetch_new_actions_from_db() — đọc action thực tế từ DB + d. action_logger.log_action() — ghi vào twitter/actions.jsonl + 7. KHÔNG đóng env → env được giữ cho Interview + + Cơ chế đọc action từ DB (thay vì từ return value của env.step()): + env.step() không trả về action_args đủ chi tiết. + Sau mỗi round, đọc từ DB với last_rowid để lấy action MỚI. + last_rowid được cập nhật sau mỗi lần đọc → không đọc lại bản ghi cũ. + + semaphore=3 trong oasis.make(): giới hạn số LLM call đồng thời trong 1 round + → tránh bị rate limit API khi nhiều agent cùng gọi LLM. + Args: - config: Cấu hình mô phỏng - simulation_dir: Thư mục mô phỏng - action_logger: Logger hành động - main_logger: Trình quản lý log chính - max_rounds: Số round tối đa (tùy chọn, dùng để cắt ngắn mô phỏng quá dài) - + config: Nội dung simulation_config.json + simulation_dir: Thư mục chứa profile files và sẽ chứa output + action_logger: Logger ghi ra twitter/actions.jsonl + main_logger: Logger ghi ra simulation.log + max_rounds: Giới hạn số round (None = chạy đủ theo config) + Returns: - PlatformSimulation: Đối tượng kết quả chứa env và agent_graph + PlatformSimulation với env và agent_graph còn sống """ result = PlatformSimulation() - + def log_info(msg): if main_logger: main_logger.info(f"[Twitter] {msg}") print(f"[Twitter] {msg}") - + log_info("Initializing...") - - # Twitter dùng cấu hình LLM chung + + # Twitter dùng General LLM (không dùng boost) model = create_model(config, use_boost=False) - - # OASIS Twitter dùng định dạng CSV + profile_path = os.path.join(simulation_dir, "twitter_profiles.csv") if not os.path.exists(profile_path): log_info(f"Error: Profile file not found: {profile_path}") return result - + result.agent_graph = await generate_twitter_agent_graph( profile_path=profile_path, model=model, available_actions=TWITTER_ACTIONS, ) - - # Lấy ánh xạ tên thật của Agent từ config (dùng entity_name thay vì Agent_X mặc định) + + # Xây dựng agent_names map — dùng cho log và enrich context agent_names = get_agent_names_from_config(config) - # Nếu config không có Agent nào đó thì dùng tên mặc định của OASIS for agent_id, agent in result.agent_graph.get_agents(): if agent_id not in agent_names: agent_names[agent_id] = getattr(agent, 'name', f'Agent_{agent_id}') - + + # Xóa DB cũ nếu còn từ lần chạy trước (cleanup_simulation_logs() nên đã xóa rồi, + # nhưng xóa lần nữa ở đây để chắc chắn) db_path = os.path.join(simulation_dir, "twitter_simulation.db") if os.path.exists(db_path): os.remove(db_path) - + result.env = oasis.make( agent_graph=result.agent_graph, platform=oasis.DefaultPlatformType.TWITTER, database_path=db_path, - semaphore=3, # Giới hạn số request LLM đồng thời để tránh quá tải API + semaphore=3, # Tối đa 3 LLM call đồng thời trong cùng 1 round ) - + await result.env.reset() log_info("Environment started") - + if action_logger: action_logger.log_simulation_start(config) - + total_actions = 0 - last_rowid = 0 # Theo dõi row đã xử lý cuối cùng trong DB (dùng rowid để tránh khác biệt định dạng created_at) - - # Thực thi sự kiện khởi tạo + last_rowid = 0 # Theo dõi rowid đã đọc → chỉ đọc bản ghi MỚI mỗi round + + # ── Round 0: Đăng initial_posts ─────────────────────────────────────────── + # Round 0 là round đặc biệt — không phải LLMAction mà là ManualAction. + # Các bài đăng này tạo ra "tin nóng đầu tiên" để agent phản ứng từ round 1. event_config = config.get("event_config", {}) initial_posts = event_config.get("initial_posts", []) - - # Ghi log bắt đầu round 0 (giai đoạn sự kiện khởi tạo) + if action_logger: - action_logger.log_round_start(0, 0) # round 0, simulated_hour 0 - + action_logger.log_round_start(0, 0) + initial_action_count = 0 if initial_posts: initial_actions = {} @@ -1212,7 +1326,7 @@ async def run_twitter_simulation( action_type=ActionType.CREATE_POST, action_args={"content": content} ) - + if action_logger: action_logger.log_action( round_num=0, @@ -1225,63 +1339,61 @@ async def run_twitter_simulation( initial_action_count += 1 except Exception: pass - + if initial_actions: await result.env.step(initial_actions) log_info(f"Published {len(initial_actions)} initial posts") - - # Ghi log kết thúc round 0 + if action_logger: action_logger.log_round_end(0, initial_action_count) - - # Vòng lặp mô phỏng chính + + # ── Vòng lặp Round 1..N ─────────────────────────────────────────────────── time_config = config.get("time_config", {}) total_hours = time_config.get("total_simulation_hours", 72) minutes_per_round = time_config.get("minutes_per_round", 30) total_rounds = (total_hours * 60) // minutes_per_round - - # Nếu chỉ định max rounds thì cắt ngắn + if max_rounds is not None and max_rounds > 0: original_rounds = total_rounds total_rounds = min(total_rounds, max_rounds) if total_rounds < original_rounds: log_info(f"Rounds truncated: {original_rounds} -> {total_rounds} (max_rounds={max_rounds})") - + start_time = datetime.now() - + for round_num in range(total_rounds): - # Kiểm tra có nhận tín hiệu thoát không + # Kiểm tra tín hiệu thoát (SIGTERM/SIGINT) — thoát gracefully if _shutdown_event and _shutdown_event.is_set(): if main_logger: main_logger.info(f"Received shutdown signal, stop simulation at round {round_num + 1}") break - + simulated_minutes = round_num * minutes_per_round - simulated_hour = (simulated_minutes // 60) % 24 + simulated_hour = (simulated_minutes // 60) % 24 # 0-23 (giờ trong ngày) simulated_day = simulated_minutes // (60 * 24) + 1 - + active_agents = get_active_agents_for_round( result.env, config, simulated_hour, round_num ) - - # Dù có Agent hoạt động hay không, vẫn ghi log bắt đầu round + + # Log round_start dù không có agent (giúp monitor thread track round đúng) if action_logger: action_logger.log_round_start(round_num + 1, simulated_hour) - + if not active_agents: - # Không có Agent hoạt động thì vẫn ghi log kết thúc round (actions_count=0) if action_logger: action_logger.log_round_end(round_num + 1, 0) continue - + + # Cho các agent chọn action (LLMAction = OASIS tự quyết định qua LLM) actions = {agent: LLMAction() for _, agent in active_agents} await result.env.step(actions) - - # Lấy action thực tế đã chạy từ DB và ghi log + + # Đọc action thực tế từ DB (KHÔNG dùng return value của env.step()) actual_actions, last_rowid = fetch_new_actions_from_db( db_path, last_rowid, agent_names ) - + round_action_count = 0 for action_data in actual_actions: if action_logger: @@ -1294,103 +1406,102 @@ async def run_twitter_simulation( ) total_actions += 1 round_action_count += 1 - + if action_logger: action_logger.log_round_end(round_num + 1, round_action_count) - + if (round_num + 1) % 20 == 0: progress = (round_num + 1) / total_rounds * 100 log_info(f"Day {simulated_day}, {simulated_hour:02d}:00 - Round {round_num + 1}/{total_rounds} ({progress:.1f}%)") - - # Lưu ý: Không đóng environment, giữ lại để dùng cho Interview - + + # KHÔNG đóng env ở đây — giữ lại cho Interview IPC + if action_logger: action_logger.log_simulation_end(total_rounds, total_actions) - + result.total_actions = total_actions elapsed = (datetime.now() - start_time).total_seconds() log_info(f"Simulation loop completed! Elapsed: {elapsed:.1f}s, total actions: {total_actions}") - + return result async def run_reddit_simulation( - config: Dict[str, Any], + config: Dict[str, Any], simulation_dir: str, action_logger: Optional[PlatformActionLogger] = None, main_logger: Optional[SimulationLogManager] = None, max_rounds: Optional[int] = None ) -> PlatformSimulation: - """Chạy mô phỏng Reddit - - Args: - config: Cấu hình mô phỏng - simulation_dir: Thư mục mô phỏng - action_logger: Logger hành động - main_logger: Trình quản lý log chính - max_rounds: Số round tối đa (tùy chọn, dùng để cắt ngắn mô phỏng quá dài) - - Returns: - PlatformSimulation: Đối tượng kết quả chứa env và agent_graph + """ + Khởi tạo và chạy simulation Reddit. + + Tương tự run_twitter_simulation() với 2 điểm khác biệt: + + 1. Dùng Boost LLM (use_boost=True): + Reddit và Twitter chạy asyncio.gather song song. + Dùng API key khác nhau giúp tránh rate limit và tăng throughput. + + 2. Xử lý initial_posts phức tạp hơn: + OASIS Reddit cho phép 1 agent đăng nhiều bài trong initial_posts. + Nếu cùng agent được assign nhiều initial_posts, initial_actions[agent] + trở thành List[ManualAction] thay vì ManualAction đơn lẻ. + (Twitter không hỗ trợ điều này — chỉ 1 action/agent/step) """ result = PlatformSimulation() - + def log_info(msg): if main_logger: main_logger.info(f"[Reddit] {msg}") print(f"[Reddit] {msg}") - + log_info("Initializing...") - - # Reddit dùng cấu hình LLM tăng tốc (nếu có, nếu không thì fallback về cấu hình chung) + + # Reddit dùng Boost LLM (fallback về General nếu không có config boost) model = create_model(config, use_boost=True) - + profile_path = os.path.join(simulation_dir, "reddit_profiles.json") if not os.path.exists(profile_path): log_info(f"Error: Profile file not found: {profile_path}") return result - + result.agent_graph = await generate_reddit_agent_graph( profile_path=profile_path, model=model, available_actions=REDDIT_ACTIONS, ) - - # Lấy ánh xạ tên thật của Agent từ config (dùng entity_name thay vì Agent_X mặc định) + agent_names = get_agent_names_from_config(config) - # Nếu config không có Agent nào đó thì dùng tên mặc định của OASIS for agent_id, agent in result.agent_graph.get_agents(): if agent_id not in agent_names: agent_names[agent_id] = getattr(agent, 'name', f'Agent_{agent_id}') - + db_path = os.path.join(simulation_dir, "reddit_simulation.db") if os.path.exists(db_path): os.remove(db_path) - + result.env = oasis.make( agent_graph=result.agent_graph, platform=oasis.DefaultPlatformType.REDDIT, database_path=db_path, - semaphore=3, # Giới hạn số request LLM đồng thời để tránh quá tải API + semaphore=3, ) - + await result.env.reset() log_info("Environment started") - + if action_logger: action_logger.log_simulation_start(config) - + total_actions = 0 - last_rowid = 0 # Theo dõi row đã xử lý cuối cùng trong DB (dùng rowid để tránh khác biệt định dạng created_at) - - # Thực thi sự kiện khởi tạo + last_rowid = 0 + event_config = config.get("event_config", {}) initial_posts = event_config.get("initial_posts", []) - - # Ghi log bắt đầu round 0 (giai đoạn sự kiện khởi tạo) + if action_logger: - action_logger.log_round_start(0, 0) # round 0, simulated_hour 0 - + action_logger.log_round_start(0, 0) + initial_action_count = 0 if initial_posts: initial_actions = {} @@ -1399,6 +1510,7 @@ async def run_reddit_simulation( content = post.get("content", "") try: agent = result.env.agent_graph.get_agent(agent_id) + # Reddit cho phép nhiều ManualAction/agent → dùng List nếu cần if agent in initial_actions: if not isinstance(initial_actions[agent], list): initial_actions[agent] = [initial_actions[agent]] @@ -1411,7 +1523,7 @@ async def run_reddit_simulation( action_type=ActionType.CREATE_POST, action_args={"content": content} ) - + if action_logger: action_logger.log_action( round_num=0, @@ -1424,63 +1536,56 @@ async def run_reddit_simulation( initial_action_count += 1 except Exception: pass - + if initial_actions: await result.env.step(initial_actions) log_info(f"Published {len(initial_actions)} initial posts") - - # Ghi log kết thúc round 0 + if action_logger: action_logger.log_round_end(0, initial_action_count) - - # Vòng lặp mô phỏng chính + time_config = config.get("time_config", {}) total_hours = time_config.get("total_simulation_hours", 72) minutes_per_round = time_config.get("minutes_per_round", 30) total_rounds = (total_hours * 60) // minutes_per_round - - # Nếu chỉ định max rounds thì cắt ngắn + if max_rounds is not None and max_rounds > 0: original_rounds = total_rounds total_rounds = min(total_rounds, max_rounds) if total_rounds < original_rounds: log_info(f"Rounds truncated: {original_rounds} -> {total_rounds} (max_rounds={max_rounds})") - + start_time = datetime.now() - + for round_num in range(total_rounds): - # Kiểm tra có nhận tín hiệu thoát không if _shutdown_event and _shutdown_event.is_set(): if main_logger: main_logger.info(f"Received shutdown signal, stop simulation at round {round_num + 1}") break - + simulated_minutes = round_num * minutes_per_round simulated_hour = (simulated_minutes // 60) % 24 simulated_day = simulated_minutes // (60 * 24) + 1 - + active_agents = get_active_agents_for_round( result.env, config, simulated_hour, round_num ) - - # Dù có Agent hoạt động hay không, vẫn ghi log bắt đầu round + if action_logger: action_logger.log_round_start(round_num + 1, simulated_hour) - + if not active_agents: - # Không có Agent hoạt động thì vẫn ghi log kết thúc round (actions_count=0) if action_logger: action_logger.log_round_end(round_num + 1, 0) continue - + actions = {agent: LLMAction() for _, agent in active_agents} await result.env.step(actions) - - # Lấy action thực tế đã chạy từ DB và ghi log + actual_actions, last_rowid = fetch_new_actions_from_db( db_path, last_rowid, agent_names ) - + round_action_count = 0 for action_data in actual_actions: if action_logger: @@ -1493,69 +1598,53 @@ async def run_reddit_simulation( ) total_actions += 1 round_action_count += 1 - + if action_logger: action_logger.log_round_end(round_num + 1, round_action_count) - + if (round_num + 1) % 20 == 0: progress = (round_num + 1) / total_rounds * 100 log_info(f"Day {simulated_day}, {simulated_hour:02d}:00 - Round {round_num + 1}/{total_rounds} ({progress:.1f}%)") - - # Lưu ý: Không đóng environment, giữ lại để dùng cho Interview - + + # KHÔNG đóng env — giữ lại cho Interview + if action_logger: action_logger.log_simulation_end(total_rounds, total_actions) - + result.total_actions = total_actions elapsed = (datetime.now() - start_time).total_seconds() log_info(f"Simulation loop completed! Elapsed: {elapsed:.1f}s, total actions: {total_actions}") - + return result async def main(): parser = argparse.ArgumentParser(description='OASIS dual-platform parallel simulation') - parser.add_argument( - '--config', - type=str, - required=True, - help='Path to config file (simulation_config.json)' - ) - parser.add_argument( - '--twitter-only', - action='store_true', - help='Run Twitter simulation only' - ) - parser.add_argument( - '--reddit-only', - action='store_true', - help='Run Reddit simulation only' - ) - parser.add_argument( - '--max-rounds', - type=int, - default=None, - help='Maximum simulation rounds (optional, used to truncate long simulations)' - ) - parser.add_argument( - '--no-wait', - action='store_true', - default=False, - help='Close environment immediately after simulation, do not enter command wait mode' - ) - + parser.add_argument('--config', type=str, required=True, + help='Path to config file (simulation_config.json)') + parser.add_argument('--twitter-only', action='store_true', + help='Run Twitter simulation only') + parser.add_argument('--reddit-only', action='store_true', + help='Run Reddit simulation only') + parser.add_argument('--max-rounds', type=int, default=None, + help='Maximum simulation rounds (optional, used to truncate long simulations)') + parser.add_argument('--no-wait', action='store_true', default=False, + help='Close environment immediately after simulation, do not enter command wait mode') + args = parser.parse_args() - - # Tạo shutdown event khi vào main để toàn bộ chương trình có thể phản hồi tín hiệu thoát + + # Khởi tạo shutdown event — dùng để phối hợp graceful shutdown global _shutdown_event _shutdown_event = asyncio.Event() - + if not os.path.exists(args.config): print(f"Error: Config file not found: {args.config}") sys.exit(1) - + config = load_config(args.config) simulation_dir = os.path.dirname(args.config) or "." + # wait_for_commands=True: sau simulation, vào chế độ chờ lệnh IPC (Interview) + # wait_for_commands=False (--no-wait): đóng ngay — dùng khi test hoặc không cần Interview wait_for_commands = not args.no_wait install_openai_cost_patch( @@ -1565,27 +1654,25 @@ async def main(): component="scripts.run_parallel_simulation", phase="simulation_run", ) - - # Khởi tạo cấu hình log (tắt log OASIS, dọn file cũ) + init_logging_for_simulation(simulation_dir) - - # Tạo trình quản lý log + log_manager = SimulationLogManager(simulation_dir) twitter_logger = log_manager.get_twitter_logger() reddit_logger = log_manager.get_reddit_logger() - + log_manager.info("=" * 60) log_manager.info("OASIS dual-platform parallel simulation") log_manager.info(f"Config file: {args.config}") log_manager.info(f"Simulation ID: {config.get('simulation_id', 'unknown')}") log_manager.info(f"Command wait mode: {'enabled' if wait_for_commands else 'disabled'}") log_manager.info("=" * 60) - + time_config = config.get("time_config", {}) total_hours = time_config.get('total_simulation_hours', 72) minutes_per_round = time_config.get('minutes_per_round', 30) config_total_rounds = (total_hours * 60) // minutes_per_round - + log_manager.info(f"Simulation parameters:") log_manager.info(f" - Total simulation duration: {total_hours} hours") log_manager.info(f" - Minutes per round: {minutes_per_round}") @@ -1595,44 +1682,47 @@ async def main(): if args.max_rounds < config_total_rounds: log_manager.info(f" - Actual executed rounds: {args.max_rounds} (truncated)") log_manager.info(f" - Agent count: {len(config.get('agent_configs', []))}") - + log_manager.info("Log structure:") log_manager.info(f" - Main log: simulation.log") log_manager.info(f" - Twitter actions: twitter/actions.jsonl") log_manager.info(f" - Reddit actions: reddit/actions.jsonl") log_manager.info("=" * 60) - + start_time = datetime.now() - - # Lưu kết quả mô phỏng của hai nền tảng + twitter_result: Optional[PlatformSimulation] = None reddit_result: Optional[PlatformSimulation] = None - + if args.twitter_only: twitter_result = await run_twitter_simulation(config, simulation_dir, twitter_logger, log_manager, args.max_rounds) elif args.reddit_only: reddit_result = await run_reddit_simulation(config, simulation_dir, reddit_logger, log_manager, args.max_rounds) else: - # Chạy song song (mỗi nền tảng dùng logger riêng) + # Chạy song song — asyncio.gather chạy cả 2 coroutine trong cùng event loop + # Twitter dùng General LLM, Reddit dùng Boost LLM → hai API key khác nhau + # → không tranh nhau rate limit, throughput tăng đôi results = await asyncio.gather( run_twitter_simulation(config, simulation_dir, twitter_logger, log_manager, args.max_rounds), run_reddit_simulation(config, simulation_dir, reddit_logger, log_manager, args.max_rounds), ) twitter_result, reddit_result = results - + total_elapsed = (datetime.now() - start_time).total_seconds() log_manager.info("=" * 60) log_manager.info(f"Simulation loop completed! Total elapsed: {total_elapsed:.1f}s") - - # Có vào chế độ chờ lệnh hay không + + # ── Chế độ chờ lệnh IPC ─────────────────────────────────────────────────── + # Sau khi simulation xong, environment vẫn còn sống. + # Vào vòng lặp poll ipc_commands/ mỗi 0.5 giây. + # Thoát khi: nhận close_env command, SIGTERM/SIGINT, hoặc exception. if wait_for_commands: log_manager.info("") log_manager.info("=" * 60) log_manager.info("Entering command wait mode - environment stays running") log_manager.info("Supported commands: interview, batch_interview, close_env") log_manager.info("=" * 60) - - # Tạo bộ xử lý IPC + ipc_handler = ParallelIPCHandler( simulation_dir=simulation_dir, twitter_env=twitter_result.env if twitter_result else None, @@ -1640,39 +1730,39 @@ async def main(): reddit_env=reddit_result.env if reddit_result else None, reddit_agent_graph=reddit_result.agent_graph if reddit_result else None ) + # Ghi env_status.json → Flask biết có thể gửi IPC command ipc_handler.update_status("alive") - - # Vòng lặp chờ lệnh (dùng _shutdown_event toàn cục) + try: while not _shutdown_event.is_set(): should_continue = await ipc_handler.process_commands() if not should_continue: break - # Dùng wait_for thay cho sleep để có thể phản hồi shutdown_event + # asyncio.wait_for thay vì sleep — có thể interrupt ngay khi nhận signal try: await asyncio.wait_for(_shutdown_event.wait(), timeout=0.5) - break # Đã nhận tín hiệu thoát + break # Nhận signal → thoát except asyncio.TimeoutError: - pass # Timeout thì tiếp tục vòng lặp + pass # Timeout thì tiếp tục poll except KeyboardInterrupt: print("\nInterrupt signal received") except asyncio.CancelledError: print("\nTask cancelled") except Exception as e: print(f"\nCommand processing error: {e}") - + log_manager.info("\nClosing environment...") - ipc_handler.update_status("stopped") - - # Đóng environment + ipc_handler.update_status("stopped") # Flask biết environment đã đóng + + # Đóng environment sau khi thoát chế độ chờ (hoặc --no-wait) if twitter_result and twitter_result.env: await twitter_result.env.close() log_manager.info("[Twitter] Environment closed") - + if reddit_result and reddit_result.env: await reddit_result.env.close() log_manager.info("[Reddit] Environment closed") - + log_manager.info("=" * 60) log_manager.info(f"All done!") log_manager.info(f"Log files:") @@ -1684,31 +1774,38 @@ async def main(): def setup_signal_handlers(loop=None): """ - Thiết lập signal handler để thoát đúng cách khi nhận SIGTERM/SIGINT - - Kịch bản mô phỏng bền vững: không thoát ngay sau khi mô phỏng xong, tiếp tục chờ lệnh interview - Khi nhận tín hiệu dừng, cần: - 1. Thông báo vòng lặp asyncio thoát khỏi trạng thái chờ - 2. Cho chương trình cơ hội dọn tài nguyên đúng cách (đóng DB, environment...) - 3. Sau đó mới thoát + Thiết lập signal handler cho SIGTERM và SIGINT. + + Chiến lược 2 bước (thay vì sys.exit() ngay): + Lần nhận signal đầu tiên: + _shutdown_event.set() → thông báo vòng lặp asyncio thoát + Cho chương trình cơ hội dọn tài nguyên (đóng DB, env...) + Không gọi sys.exit() ngay + + Lần nhận signal thứ hai (người dùng ép thoát): + sys.exit(1) → force quit ngay lập tức + + Tại sao không gọi sys.exit() ngay lần đầu? + asyncio event loop cần được thông báo gracefully để: + 1. Hoàn thành các coroutine đang chạy + 2. Đóng SQLite connections đúng cách + 3. Ghi simulation_end event vào actions.jsonl """ def signal_handler(signum, frame): global _cleanup_done sig_name = "SIGTERM" if signum == signal.SIGTERM else "SIGINT" print(f"\nReceived {sig_name}, shutting down...") - + if not _cleanup_done: _cleanup_done = True - # Set event để thông báo vòng lặp asyncio thoát (để kịp dọn tài nguyên) + # Thông báo asyncio event loop thoát gracefully if _shutdown_event: _shutdown_event.set() - - # Không gọi sys.exit() ngay, để asyncio thoát tự nhiên và dọn tài nguyên - # Nếu nhận tín hiệu lặp lại thì mới ép thoát else: + # Nhận signal lần 2 → force exit print("Force exit...") sys.exit(1) - + signal.signal(signal.SIGTERM, signal_handler) signal.signal(signal.SIGINT, signal_handler) @@ -1722,7 +1819,8 @@ if __name__ == "__main__": except SystemExit: pass finally: - # Dọn resource tracker của multiprocessing (tránh cảnh báo khi thoát) + # Dọn resource tracker của multiprocessing — tránh warning khi thoát + # trên một số hệ thống Python/OS kết hợp nhất định try: from multiprocessing import resource_tracker resource_tracker._resource_tracker._stop() diff --git a/test_code_backend/calculate_cost.ipynb b/test_code_backend/calculate_cost.ipynb new file mode 100644 index 00000000..d9a72ee1 --- /dev/null +++ b/test_code_backend/calculate_cost.ipynb @@ -0,0 +1,107 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": 1, + "id": "1606cd6c", + "metadata": {}, + "outputs": [], + "source": [ + "import json" + ] + }, + { + "cell_type": "code", + "execution_count": 2, + "id": "69ce47c3", + "metadata": {}, + "outputs": [], + "source": [ + "input = \"/home/anman/intern/MiroFish/backend/logs/proj_ac763defa255/cost_Qwen_Qwen3.6-27B-FP8.jsonl\"\n", + "\n", + "data=[]\n", + "with open(input, 'r', encoding='utf-8') as f:\n", + " data = [json.loads(line.strip()) for line in f if line.strip()]" + ] + }, + { + "cell_type": "code", + "execution_count": 3, + "id": "6f1fe377", + "metadata": {}, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "Input Tokens: 49830604\n", + "Output Tokens: 5432209\n" + ] + } + ], + "source": [ + "input_tokens = 0\n", + "output_tokens = 0\n", + "for item in data:\n", + " input_tokens += item['input_tokens']\n", + " output_tokens += item['output_tokens']\n", + "\n", + "print(f\"Input Tokens: {input_tokens}\")\n", + "print(f\"Output Tokens: {output_tokens}\")" + ] + }, + { + "cell_type": "code", + "execution_count": 5, + "id": "95cce20c", + "metadata": {}, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "Total Cost: $41.2119\n", + "Cost Breakdown: Input Cost = $24.9153, Output Cost = $16.2966\n" + ] + } + ], + "source": [ + "cost_1M_input = 0.5\n", + "cost_1M_output = 3.0\n", + "\n", + "total_cost = (input_tokens / 1_000_000) * cost_1M_input + (output_tokens / 1_000_000) * cost_1M_output\n", + "print(f\"Total Cost: ${total_cost:.4f}\")\n", + "print(f\"Cost Breakdown: Input Cost = ${(input_tokens / 1_000_000) * cost_1M_input:.4f}, Output Cost = ${(output_tokens / 1_000_000) * cost_1M_output:.4f}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "273cd657", + "metadata": {}, + "outputs": [], + "source": [] + } + ], + "metadata": { + "kernelspec": { + "display_name": "mirofish_v1", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.11.15" + } + }, + "nbformat": 4, + "nbformat_minor": 5 +} diff --git a/test_code_backend/full_pipeline/README.md b/test_code_backend/full_pipeline/README.md index e9894299..0a2d121c 100644 --- a/test_code_backend/full_pipeline/README.md +++ b/test_code_backend/full_pipeline/README.md @@ -100,6 +100,8 @@ Lưu lại đường dẫn `combined_file` để dùng ở bước tiếp theo. ```bash COMBINED_FILE="/home/anman/intern/MiroFish/test_code_backend/full_pipeline/data/articles_combined.md" +COMBINED_FILE="/home/anman/intern/MiroFish/data/news/2026-02/2026-02-03_oil-prices-add-gains-after-report-says_0178c4.md" + curl -s http://localhost:5001/api/graph/ontology/generate \ -F "files=@${COMBINED_FILE};filename=articles.md" \ -F "simulation_requirement=Analyze how these oil and financial market news articles affect investor sentiment and market dynamics. Predict how different market participants (traders, analysts, retail investors) will react and what the overall price trend will be." \ @@ -248,6 +250,38 @@ curl -s http://localhost:5001/api/report/$REPORT_ID/download -o report.md echo "Saved to report.md" ``` +Nếu sim đã có report rồi mà muốn chạy lại thì dùng như sau + +```bash +SIM_ID="sim_xxxxxxxxxxxx" + +SIM_ID="sim_07ab325b3818" +# 1. Bắt đầu generate → lấy task_id và report_id +RESP=$(curl -s http://localhost:5001/api/report/generate \ + -H "Content-Type: application/json" \ + -d "{\"simulation_id\": \"$SIM_ID\", \"force_regenerate\": true}") +TASK_ID=$(echo "$RESP" | jq -r '.data.task_id') +REPORT_ID=$(echo "$RESP" | jq -r '.data.report_id') +echo "task_id=$TASK_ID report_id=$REPORT_ID" + +# 2. Poll đến khi completed (chạy lặp lại, ~5–15 phút) +while true; do + R=$(curl -s http://localhost:5001/api/report/generate/status \ + -H "Content-Type: application/json" \ + -d "{\"task_id\": \"$TASK_ID\"}") + ST=$(echo "$R" | jq -r '.data.status') + PG=$(echo "$R" | jq -r '.data.progress') + echo "$ST ${PG}%" + [[ "$ST" == "completed" ]] && break + [[ "$ST" == "failed" ]] && { echo "FAILED"; break; } + sleep 15 +done + +# 3. Download file Markdown +curl -s http://localhost:5001/api/report/$REPORT_ID/download -o report.md +echo "Saved to report.md" +``` + --- ## Output files diff --git a/test_code_backend/full_pipeline/config_articles.env b/test_code_backend/full_pipeline/config_articles.env index 30cd28cc..71cf6278 100644 --- a/test_code_backend/full_pipeline/config_articles.env +++ b/test_code_backend/full_pipeline/config_articles.env @@ -10,7 +10,7 @@ full_content = False # start_time = 2026-02-09 # end_time = 2026-02-12 start_time = 2026-04-13 -end_time = 2026-04-16 +end_time = 2026-04-14 # ─── Simulation settings ────────────────────────────────────────────────────── # What the simulation should analyse / predict (sent to the LLM) diff --git a/test_code_backend/gen_ontogogy_and_graph/build_graph.py b/test_code_backend/gen_ontogogy_and_graph/build_graph.py deleted file mode 100644 index 9d914b16..00000000 --- a/test_code_backend/gen_ontogogy_and_graph/build_graph.py +++ /dev/null @@ -1,197 +0,0 @@ -""" -Test: Knowledge Graph build on Zep Cloud -Input : - output/ontology/ontology_*.json (latest, or pass path as arg) - - news articles (same date range as ontology) -Output: output/graph/ - graph_.json — build summary (node/edge count, timing) - entities_.json — entity detail: raw nodes + filtered result - -Run: - python build_graph.py # auto-picks latest ontology - python build_graph.py output/ontology/ontology_xyz.json # specific ontology file -""" - -import sys -import json -import time -import argparse -from datetime import datetime, date -from pathlib import Path - -# ── Path setup ─────────────────────────────────────────────────────────────── -SCRIPT_DIR = Path(__file__).parent -PROJECT_ROOT = SCRIPT_DIR.parent.parent -BACKEND_DIR = PROJECT_ROOT / "backend" -ONTOLOGY_DIR = SCRIPT_DIR.parent / "output" / "ontology" -GRAPH_DIR = SCRIPT_DIR.parent / "output" / "graph" -NEWS_DIR = PROJECT_ROOT / "data" / "news" / "2026-04" - -sys.path.insert(0, str(BACKEND_DIR)) -GRAPH_DIR.mkdir(parents=True, exist_ok=True) - -# ── Config ─────────────────────────────────────────────────────────────────── -START_DATE = date(2026, 4, 20) -END_DATE = date(2026, 4, 22) -GRAPH_NAME = "MiroFish_Test_Apr2026" -CHUNK_SIZE = 500 -CHUNK_OVERLAP = 50 -BATCH_SIZE = 3 -POLL_TIMEOUT = 600 - - -# ── Helpers ────────────────────────────────────────────────────────────────── -def log(msg: str): - print(f"[{datetime.now().strftime('%H:%M:%S')}] {msg}") - - -def extract_body(md_text: str) -> str: - s = md_text.strip() - if s.startswith("---"): - end = s.find("---", 3) - if end != -1: - return s[end + 3:].strip() - return s - - -def load_combined_text(news_dir: Path, start: date, end: date) -> str: - parts = [] - for md_file in sorted(news_dir.glob("*.md")): - try: - file_date = date.fromisoformat(md_file.stem.split("_")[0]) - except ValueError: - continue - if not (start <= file_date <= end): - continue - body = extract_body(md_file.read_text(encoding="utf-8")) - if body: - parts.append(body) - return "\n\n---\n\n".join(parts) - - -def pick_ontology_file(arg_path: str | None) -> Path: - if arg_path: - p = Path(arg_path) - if not p.is_absolute(): - p = ONTOLOGY_DIR / p - if not p.exists(): - raise FileNotFoundError(f"Ontology file not found: {p}") - return p - - candidates = sorted(ONTOLOGY_DIR.glob("ontology_*.json"), key=lambda f: f.stat().st_mtime) - if not candidates: - raise FileNotFoundError( - f"No ontology_*.json found in {ONTOLOGY_DIR}. Run gen_ontology.py first." - ) - return candidates[-1] # latest - - -# ── Main ───────────────────────────────────────────────────────────────────── -def main(): - parser = argparse.ArgumentParser(description="Build Zep Knowledge Graph from ontology") - parser.add_argument("ontology_file", nargs="?", help="Path to ontology JSON (default: latest in output/)") - args = parser.parse_args() - - from app.services.graph_builder import GraphBuilderService - from app.services.text_processor import TextProcessor - from app.services.zep_entity_reader import ZepEntityReader - - # Load ontology - ontology_path = pick_ontology_file(args.ontology_file) - log(f"Using ontology: {ontology_path.name}") - with open(ontology_path, encoding="utf-8") as f: - payload = json.load(f) - ontology = payload["ontology"] - - # Load news text - log(f"Loading news text: {START_DATE} → {END_DATE}") - combined_text = load_combined_text(NEWS_DIR, START_DATE, END_DATE) - log(f" Text length: {len(combined_text):,} chars") - - svc = GraphBuilderService() - t_total = time.time() - - # Create graph - log("Creating Zep graph...") - graph_id = svc.create_graph(GRAPH_NAME) - log(f" graph_id = {graph_id}") - - # Apply ontology - log("Applying ontology schema...") - svc.set_ontology(graph_id, ontology) - - # Chunk & upload - chunks = TextProcessor.split_text(combined_text, CHUNK_SIZE, CHUNK_OVERLAP) - log(f"Uploading {len(chunks)} chunks (batch_size={BATCH_SIZE})...") - - episode_uuids = svc.add_text_batches( - graph_id, chunks, BATCH_SIZE, - progress_callback=lambda msg, _: log(f" {msg}"), - ) - log(f" {len(episode_uuids)} episodes uploaded") - - # Wait for processing - log("Waiting for Zep to process episodes...") - svc._wait_for_episodes( - episode_uuids, - progress_callback=lambda msg, _: log(f" {msg}"), - timeout=POLL_TIMEOUT, - ) - - # Fetch result - log("Fetching graph stats...") - graph_info = svc._get_graph_info(graph_id) - elapsed = round(time.time() - t_total, 2) - - log(f"Done in {elapsed}s — nodes={graph_info.node_count}, edges={graph_info.edge_count}") - log(f" Entity types: {graph_info.entity_types}") - - out_path = GRAPH_DIR / f"graph_{graph_id}.json" - with open(out_path, "w", encoding="utf-8") as f: - json.dump({ - "graph_id": graph_id, - "graph_name": GRAPH_NAME, - "built_at": datetime.now().isoformat(), - "elapsed_seconds": elapsed, - "ontology_source": ontology_path.name, - "date_range": {"start": str(START_DATE), "end": str(END_DATE)}, - "chunks_uploaded": len(chunks), - "node_count": graph_info.node_count, - "edge_count": graph_info.edge_count, - "entity_types": graph_info.entity_types, - }, f, ensure_ascii=False, indent=2) - - log(f"Saved → {out_path.name}") - - # ── Entity detail ───────────────────────────────────────────────────────── - # Raw graph data — lưu nguyên output của get_graph_data() - log("Fetching raw graph data...") - graph_data = svc.get_graph_data(graph_id) - log(f" nodes={graph_data['node_count']}, edges={graph_data['edge_count']}") - - # Filtered entities — lưu nguyên output của filter_defined_entities().to_dict() - log("Running entity filter...") - defined_types = [e["name"] for e in ontology.get("entity_types", [])] - reader = ZepEntityReader() - filtered = reader.filter_defined_entities( - graph_id=graph_id, - defined_entity_types=defined_types, - enrich_with_edges=True, - ) - log(f" total={filtered.total_count}, matched={filtered.filtered_count}") - log(f" types found: {sorted(filtered.entity_types)}") - - entities_path = GRAPH_DIR / f"entities_{graph_id}.json" - with open(entities_path, "w", encoding="utf-8") as f: - json.dump({ - "graph_id": graph_id, - "fetched_at": datetime.now().isoformat(), - "ontology_entity_types": defined_types, - "raw_graph_data": graph_data, - "filtered_entities": filtered.to_dict(), - }, f, ensure_ascii=False, indent=2) - - log(f"Saved → {entities_path.name}") - - -if __name__ == "__main__": - main() diff --git a/test_code_backend/gen_ontogogy_and_graph/gen_ontology.py b/test_code_backend/gen_ontogogy_and_graph/gen_ontology.py deleted file mode 100644 index 16b3e577..00000000 --- a/test_code_backend/gen_ontogogy_and_graph/gen_ontology.py +++ /dev/null @@ -1,109 +0,0 @@ -""" -Test: Ontology generation from news articles -Input : data/news/2026-04/ (date range configured below) -Output: test_code_backend/output/ontology/ontology__.json - -Run: - python gen_ontology.py -""" - -import sys -import json -import time -from datetime import datetime, date -from pathlib import Path - -# ── Path setup ─────────────────────────────────────────────────────────────── -SCRIPT_DIR = Path(__file__).parent -PROJECT_ROOT = SCRIPT_DIR.parent.parent -BACKEND_DIR = PROJECT_ROOT / "backend" -ONTOLOGY_DIR = SCRIPT_DIR.parent / "output" / "ontology" -NEWS_DIR = PROJECT_ROOT / "data" / "news" / "2026-04" - -sys.path.insert(0, str(BACKEND_DIR)) -ONTOLOGY_DIR.mkdir(parents=True, exist_ok=True) - -# ── Config ─────────────────────────────────────────────────────────────────── -START_DATE = date(2026, 4, 20) -END_DATE = date(2026, 4, 22) - -SIMULATION_REQUIREMENT = ( - "Simulate social media public opinion dynamics around global oil/energy markets " - "and US-Iran geopolitical tensions during April 2026. " - "Focus on how governments, corporations, media outlets, financial analysts, and " - "ordinary citizens react and interact on platforms like Twitter and Reddit." -) - - -# ── Helpers ────────────────────────────────────────────────────────────────── -def log(msg: str): - print(f"[{datetime.now().strftime('%H:%M:%S')}] {msg}") - - -def extract_body(md_text: str) -> str: - """Strip YAML frontmatter (---...---), return only article body.""" - s = md_text.strip() - if s.startswith("---"): - end = s.find("---", 3) - if end != -1: - return s[end + 3:].strip() - return s - - -def load_articles(news_dir: Path, start: date, end: date) -> list[dict]: - articles = [] - for md_file in sorted(news_dir.glob("*.md")): - try: - file_date = date.fromisoformat(md_file.stem.split("_")[0]) - except ValueError: - continue - if not (start <= file_date <= end): - continue - body = extract_body(md_file.read_text(encoding="utf-8")) - if body: - articles.append({"filename": md_file.name, "date": str(file_date), "body": body}) - return articles - - -# ── Main ───────────────────────────────────────────────────────────────────── -def main(): - from app.services.ontology_generator import OntologyGenerator - - log(f"Loading news: {START_DATE} → {END_DATE}") - articles = load_articles(NEWS_DIR, START_DATE, END_DATE) - log(f" {len(articles)} articles loaded") - - log("Calling LLM to generate ontology...") - t0 = time.time() - ontology = OntologyGenerator().generate( - document_texts=[a["body"] for a in articles], - simulation_requirement=SIMULATION_REQUIREMENT, - ) - elapsed = round(time.time() - t0, 2) - - entity_names = [e["name"] for e in ontology.get("entity_types", [])] - edge_names = [e["name"] for e in ontology.get("edge_types", [])] - log(f"Done in {elapsed}s") - log(f" Entities ({len(entity_names)}): {entity_names}") - log(f" Edges ({len(edge_names)}): {edge_names}") - - date_range = f"{START_DATE.strftime('%Y%m%d')}-{END_DATE.strftime('%Y%m%d')}" - ts = datetime.now().strftime("%H%M%S") - out_path = ONTOLOGY_DIR / f"ontology_{date_range}_{ts}.json" - - with open(out_path, "w", encoding="utf-8") as f: - json.dump({ - "meta": { - "generated_at": datetime.now().isoformat(), - "elapsed_seconds": elapsed, - "date_range": {"start": str(START_DATE), "end": str(END_DATE)}, - "article_count": len(articles), - }, - "ontology": ontology, - }, f, ensure_ascii=False, indent=2) - - log(f"Saved → {out_path.name}") - - -if __name__ == "__main__": - main()