From 8679916cedb23af1e73056a8fc95fac7cafd2e22 Mon Sep 17 00:00:00 2001 From: ththu0205 Date: Mon, 30 Mar 2026 07:07:52 +0000 Subject: [PATCH 1/4] edit prompts --- .../app/services/oasis_profile_generator.py | 24 +- backend/app/services/ontology_generator.py | 277 ++++++++++++------ .../services/simulation_config_generator.py | 248 +++++++++++----- backend/app/services/zep_tools.py | 224 ++++++++++---- backend/scripts/test_profile_format.py | 2 +- 5 files changed, 550 insertions(+), 225 deletions(-) diff --git a/backend/app/services/oasis_profile_generator.py b/backend/app/services/oasis_profile_generator.py index d5425571..b37284cf 100644 --- a/backend/app/services/oasis_profile_generator.py +++ b/backend/app/services/oasis_profile_generator.py @@ -162,7 +162,7 @@ class OasisProfileGenerator: # List mảng quốc tịch Cơ Bản COUNTRIES = [ - "China", "US", "UK", "Japan", "Germany", "France", + "Vietnam", "China", "US", "UK", "Japan", "Germany", "France", "Canada", "Australia", "Brazil", "India", "South Korea" ] @@ -676,7 +676,7 @@ class OasisProfileGenerator: def _get_system_prompt(self, is_individual: bool) -> str: """Lấy prompt cho hệ thống""" - # base_prompt = "You are an expert in generating social media user personas. Generate detailed and realistic personas for public opinion simulation to recreate existing real-world conditions to the greatest extent possible. You must return a valid JSON format; all string values must not contain unescaped line breaks. Use Chinese." + # base_prompt = "You are an expert in generating social media user personas. Generate detailed and realistic personas for public opinion simulation to recreate existing real-world conditions to the greatest extent possible. You must return a valid JSON format; all string values must not contain unescaped line breaks. Use Vietnamese." base_prompt = "Bạn là chuyên gia tạo hồ sơ người dùng mạng xã hội. Hãy tạo các nhân vật chi tiết và chân thực phục vụ cho việc mô phỏng dư luận, nhằm tái hiện tối đa các tình huống thực tế hiện có. Phải trả về định dạng JSON hợp lệ; tất cả các giá trị chuỗi không được chứa ký tự xuống dòng chưa được xử lý (unescaped). Sử dụng tiếng Việt." return base_prompt @@ -718,14 +718,14 @@ class OasisProfileGenerator: # 3. age: Age as a number (must be an integer) # 4. gender: Gender, must be in English: "male" or "female" # 5. mbti: MBTI type (e.g., INTJ, ENFP, etc.) -# 6. country: Country (use Chinese, e.g., "中国") +# 6. country: Country (use Vietnamese, e.g., "Việt Nam") # 7. profession: Occupation # 8. interested_topics: An array of interested topics # IMPORTANT: # - All field values must be strings or numbers; do not use line breaks. # - The 'persona' must be a coherent block of text description. -# - Use Chinese (except for the 'gender' field, which must be English male/female). +# - Use Vietnamese (except for the 'gender' field, which must be English male/female). # - Content must remain consistent with the entity information. # - 'age' must be a valid integer; 'gender' must be "male" or "female".""" @@ -753,14 +753,14 @@ Vui lòng tạo JSON bao gồm các trường sau: 3. age: Con số tuổi (phải là số nguyên) 4. gender: Giới tính, phải là tiếng Anh: "male" hoặc "female" 5. mbti: Loại MBTI (như INTJ, ENFP, v.v.) -6. country: Quốc gia (sử dụng tiếng Trung, ví dụ: "中国") +6. country: Quốc gia (sử dụng tiếng Việt, ví dụ: "Việt Nam") 7. profession: Nghề nghiệp 8. interested_topics: Mảng các chủ đề quan tâm QUAN TRỌNG: - Tất cả giá trị các trường phải là chuỗi hoặc số, không sử dụng ký tự xuống dòng. - 'persona' phải là một đoạn mô tả văn bản mạch lạc. -- Sử dụng tiếng Trung (ngoại trừ trường 'gender' phải dùng tiếng Anh male/female). +- Sử dụng tiếng Việt (ngoại trừ trường 'gender' phải dùng tiếng Anh male/female). - Nội dung phải nhất quán với thông tin thực thể. - 'age' phải là số nguyên hợp lệ, 'gender' phải là "male" hoặc "female". """ @@ -802,14 +802,14 @@ QUAN TRỌNG: # 3. age: Fixed at 30 (virtual age for an organizational account) # 4. gender: Fixed as "other" (representing non-individual accounts) # 5. mbti: MBTI type used to describe the account's style (e.g., ISTJ for rigorous/conservative) -# 6. country: Country (use Chinese, e.g., "中国") +# 6. country: Country (use Vietnamese, e.g., "Việt Nam") # 7. profession: Description of organizational functions # 8. interested_topics: An array of focused fields/areas of interest # IMPORTANT: # - All field values must be strings or numbers; null values are not allowed. # - 'persona' must be a coherent block of text description; do not use line breaks. -# - Use Chinese (except for the 'gender' field, which must be the English string "other"). +# - Use Vietnamese (except for the 'gender' field, which must be the English string "other"). # - 'age' must be the integer 30; 'gender' must be the string "other". # - The account's tone and discourse must strictly align with its institutional identity and positioning. # """ @@ -838,14 +838,14 @@ Vui lòng tạo JSON bao gồm các trường sau: 3. age: Cố định là 30 (tuổi ảo cho tài khoản tổ chức) 4. gender: Cố định là "other" (biểu thị tài khoản tổ chức, không phải cá nhân) 5. mbti: Loại MBTI dùng để mô tả phong cách tài khoản (ví dụ: ISTJ đại diện cho sự nghiêm túc, bảo thủ) -6. country: Quốc gia (sử dụng tiếng Trung, ví dụ: "中国") +6. country: Quốc gia (sử dụng tiếng Việt, ví dụ: "Việt Nam") 7. profession: Mô tả chức năng của tổ chức 8. interested_topics: Mảng các lĩnh vực quan tâm QUAN TRỌNG: - Tất cả giá trị các trường phải là chuỗi hoặc số, không cho phép giá trị null. - 'persona' phải là một đoạn mô tả văn bản mạch lạc, không sử dụng ký tự xuống dòng. -- Sử dụng tiếng Trung (ngoại trừ trường 'gender' phải dùng chuỗi tiếng Anh "other"). +- Sử dụng tiếng Việt (ngoại trừ trường 'gender' phải dùng chuỗi tiếng Anh "other"). - 'age' phải là số nguyên 30, 'gender' phải là chuỗi "other". - Phát ngôn và giọng điệu của tài khoản phải phù hợp tuyệt đối với định vị danh tính và đặc thù của tổ chức. """ @@ -936,6 +936,7 @@ QUAN TRỌNG: parallel_count: int = 5, realtime_output_path: Optional[str] = None, output_platform: str = "reddit", + metadata_platform: Optional[str] = None, simulation_id: Optional[str] = None, project_id: Optional[str] = None, ) -> List[OasisAgentProfile]: @@ -950,6 +951,7 @@ QUAN TRỌNG: parallel_count: Số luồng song song, mặc định 5 realtime_output_path: Đường dẫn lưu file realtime (Gen ra đứa nào auto save đứa đó luôn) output_platform: Format lưu trữ output ("reddit" hoạc "twitter") + metadata_platform: Nền tảng dùng cho metadata cost Returns: Danh sách Profile Agent @@ -966,7 +968,7 @@ QUAN TRỌNG: "phase": "generate_profiles", "simulation_id": simulation_id, "project_id": project_id, - "platform": output_platform, + "platform": metadata_platform, } total = len(entities) diff --git a/backend/app/services/ontology_generator.py b/backend/app/services/ontology_generator.py index cc44a899..fd52e3e8 100644 --- a/backend/app/services/ontology_generator.py +++ b/backend/app/services/ontology_generator.py @@ -9,32 +9,145 @@ from ..utils.llm_client import LLMClient # System prompt dùng cho việc tự động sinh Ontology -ONTOLOGY_SYSTEM_PROMPT = """Bạn là một chuyên gia thiết kế bản thể học (Ontology) cho Tri thức đồ thị (Knowledge Graph). Nhiệm vụ của bạn là phân tích nội dung văn bản được cung cấp và nhu cầu để thiết kế các loại thực thể (Entity) và loại mối quan hệ (Relationship) thiết kế phù hợp cho **Mô phỏng dư luận trên mạng xã hội**. +# ONTOLOGY_SYSTEM_PROMPT = """You are a professional Knowledge Graph Ontology Design Expert. Your task is to analyze the given text content and simulation requirements to design entity types and relationship types suitable for **social media public opinion simulation**. -**QUAN TRỌNG: Bạn BẮT BUỘC phải đầu ra một cấu trúc định dạng JSON hợp lệ, KHÔNG ĐƯỢC xuất thêm bất kỳ văn bản nào khác.** +# **IMPORTANT: You must output valid JSON format data only. Do not include any other text.** + +# ## Core Task Background + +# We are building a social media public opinion simulation system. In this system: +# - Each entity is an "account" or "subject" capable of speaking, interacting, and spreading information on social media. +# - Entities influence, forward, comment on, and respond to each other. +# - We need to simulate the reactions of all parties and the paths of information dissemination during public opinion events. + +# Therefore, **entities must be real-world subjects capable of speaking and interacting on social media**: + +# **CAN BE**: +# - Specific individuals (public figures, parties involved, opinion leaders, experts, ordinary people). +# - Companies and enterprises (including their official accounts). +# - Organizations (universities, associations, NGOs, labor unions, etc.). +# - Government departments and regulatory agencies. +# - Media outlets (newspapers, TV stations, independent media, websites). +# - Social media platforms themselves. +# - Representatives of specific groups (e.g., alumni associations, fan clubs, rights protection groups). + +# **CANNOT BE**: +# - Abstract concepts (e.g., "public opinion", "emotion", "trend"). +# - Themes/Topics (e.g., "academic integrity", "education reform"). +# - Viewpoints/Attitudes (e.g., "supporters", "opponents"). + +# ## Output Format + +# Please output in JSON format with the following structure: + +# ```json +# { +# "entity_types": [ +# { +# "name": "Entity type name (English, PascalCase)", +# "description": "Short description (English, max 100 characters)", +# "attributes": [ +# { +# "name": "Attribute name (English, snake_case)", +# "type": "text", +# "description": "Attribute description" +# } +# ], +# "examples": ["Example Entity 1", "Example Entity 2"] +# } +# ], +# "edge_types": [ +# { +# "name": "Relationship type name (English, UPPER_SNAKE_CASE)", +# "description": "Short description (English, max 100 characters)", +# "source_targets": [ +# {"source": "Source entity type", "target": "Target entity type"} +# ], +# "attributes": [] +# } +# ], +# "analysis_summary": "Brief analysis of the text content (in Vietnamese)" +# } +# ``` + +# ## Design Guidelines (Extremely Important!) + +# ### 1. Entity Type Design - Strict Compliance Required + +# **Quantity Requirement: Must be EXACTLY 10 entity types.** + +# **Hierarchy Requirements (Must include both specific types and fallback types):** + +# Your 10 entity types must include the following layers: + +# A. **Fallback Types (Required, place as the last 2 in the list)**: +# - `Person`: The fallback type for any individual natural person. Use this when a person does not fit into other specific person types. +# - `Organization`: The fallback type for any organization or institution. Use this when an organization does not fit into other specific organizational types. + +# B. **Specific Types (8 types, designed based on text content)**: +# - Design more specific types targeting the main roles appearing in the text. +# - Example: For academic events, use `Student`, `Professor`, `University` +# - VExample: For business events, use `Company`, `CEO`, `Employee` + +# **Why fallback types are needed:** +# - Various people appear in texts (e.g., "primary school teacher", "passerby", "netizen"). +# - Without a specific match, they should be categorized under `Person`. +# - Similarly, small organizations or temporary groups should fall under `Organization`. + +# **Specific Type Design Principles:** +# - Identify high-frequency or critical roles from the text. +# - Each specific type should have clear boundaries to avoid overlap. +# - The description must clearly state the difference between this type and the fallback type. + +# ### 2. Relationship Type Design + +# - Quantity: 6-10 types. +# - Relationships should reflect real-world connections in social media interactions. +# - Ensure `source_targets` cover your defined entity types. + +# ### 3. Attribute Design + +# - 1-3 key attributes per entity type. +# - **NOTE**: Do NOT use `name`, `uuid`, `group_id`, `created_at` or `summary` as attribute names (these are system reserved words). +# - Recommended: `full_name`, `title`, `role`, `position`, `location`, `description, etc. + +# ## Entity Type References + +# - Individuals (Specific): Student, Professor, Journalist, Celebrity, Executive, Official, Lawyer, Doctor. +# - Individuals (Fallback): Person. +# - Organizations (Specific): University, Company, GovernmentAgency, MediaOutlet, Hospital, School, NGO. +# - Organizations (Fallback): Organization. + +# ## Relationship Type References + +# WORKS_FOR, STUDIES_AT, AFFILIATED_WITH, REPRESENTS, REGULATES, REPORTS_ON, COMMENTS_ON, RESPONDS_TO, SUPPORTS, OPPOSES, COLLABORATES_WITH, COMPETES_WITH. +# """ + +ONTOLOGY_SYSTEM_PROMPT = """Bạn là một chuyên gia thiết kế Bản thể học (Ontology) cho Biểu đồ tri thức chuyên nghiệp. Nhiệm vụ của bạn là phân tích nội dung văn bản và yêu cầu mô phỏng được cung cấp để thiết kế các loại thực thể và loại quan hệ phù hợp cho việc **mô phỏng dư luận trên mạng xã hội**. + +**QUAN TRỌNG: Bạn phải xuất dữ liệu ở định dạng JSON hợp lệ, không xuất thêm bất kỳ nội dung nào khác.** ## Bối cảnh nhiệm vụ cốt lõi +- Chúng tôi đang xây dựng một hệ thống mô phỏng dư luận mạng xã hội. Trong hệ thống này: +- Mỗi thực thể là một "tài khoản" hoặc "chủ thể" có thể phát ngôn, tương tác và lan truyền thông tin trên mạng xã hội. +- Các thực thể sẽ ảnh hưởng, chia sẻ, bình luận và phản hồi lẫn nhau. +- Chúng tôi cần mô phỏng phản ứng của các bên và lộ trình lan truyền thông tin trong các sự kiện dư luận. -Chúng tôi đang xây dựng một **hệ thống mô phỏng tin đồn và dư luận mạng xã hội**. Trong hệ thống này: -- Mỗi thực thể là một "tài khoản" hoặc "chủ thể" có thể lên tiếng, tương tác và lan truyền thông tin trên mạng xã hội. -- Các thực thể có thể gây ảnh hưởng, chuyển tiếp (retweet), bình luận hoặc phản hồi lẫn nhau. -- Chúng tôi cần mô phỏng phản ứng của các bên và đường truyền thông tin trong các sự kiện dư luận. - -Do đó, **thực thể phải là các chủ thể có thật trong thế giới thực, có khả năng lên tiếng và tương tác trên mạng xã hội**: +Do đó, **thực thể phải là những chủ thể tồn tại thực tế, có khả năng phát ngôn và tương tác trên mạng xã hội**: **CÓ THỂ LÀ**: -- Cá nhân cụ thể (nhân vật của công chúng, các bên liên quan, KOL, chuyên gia / học giả, người bình thường) -- Công ty, doanh nghiệp (bao gồm cả tài khoản chính thức của họ) -- Tổ chức (trường đại học, hiệp hội, tổ chức phi chính phủ (NGO), công đoàn, v.v.) -- Các cơ quan chính phủ, cơ quan quản lý -- Tổ chức báo chí / truyền thông (báo đài, đài truyền hình, tự do truyền thông, trang web) -- Bản thân nền tảng mạng xã hội -- Đại diện nhóm cụ thể (như hội cựu sinh viên, fan group, nhóm bảo vệ quyền lợi, v.v.) +- Cá nhân cụ thể (người công chúng, bên liên quan, người dẫn dắt dư luận, chuyên gia, người bình thường). +- Công ty, doanh nghiệp (bao gồm cả tài khoản chính thức của họ). +- Tổ chức (trường đại học, hiệp hội, NGO, công đoàn, v.v.). +- Cơ quan chính phủ, cơ quan quản lý. +- Cơ quan truyền thông (báo chí, đài truyền hình, tự truyền thông, trang web). +- Bản thân nền tảng mạng xã hội. +- Đại diện nhóm cụ thể (như hội cựu sinh viên, nhóm người hâm mộ, nhóm bảo vệ quyền lợi, v.v.). -**KHÔNG ĐƯỢC LÀ**: -- Khái niệm trừu tượng (như "dư luận", "cảm xúc", "xu hướng") -- Chủ đề / đề tài (như "tính toàn vẹn học thuật", "cải cách giáo dục") -- Quan điểm / thái độ (như "phe ủng hộ", "bên phản đối") +**KHÔNG THỂ LÀ**: +- Khái niệm trừu tượng (như "dư luận", "cảm xúc", "xu hướng"). +- Chủ đề/Vấn đề (như "liêm chính học thuật", "cải cách giáo dục"). +- Quan điểm/Thái độ (như "bên ủng hộ", "bên phản đối"). ## Định dạng đầu ra @@ -45,38 +158,38 @@ Hãy trả về dưới định dạng JSON, bao gồm cấu trúc sau: "entity_types": [ { "name": "Tên loại thực thể (Tiếng Anh, PascalCase)", - "description": "Mô tả ngắn gọn (Tiếng Anh, tối đa 100 ký tự)", + "description": "Mô tả ngắn gọn (Tiếng Anh, không quá 100 ký tự)", "attributes": [ { "name": "Tên thuộc tính (Tiếng Anh, snake_case)", "type": "text", - "description": "Mô tả của thuộc tính" + "description": "Mô tả thuộc tính" } ], - "examples": ["Ví dụ thực thể 1", "Ví dụ thực thể 2"] + "examples": ["Thực thể ví dụ 1", "Thực thể ví dụ 2"] } ], "edge_types": [ { "name": "Tên loại quan hệ (Tiếng Anh, UPPER_SNAKE_CASE)", - "description": "Mô tả ngắn (Tiếng Anh, tối đa 100 ký tự)", + "description": "Mô tả ngắn gọn (Tiếng Anh, không quá 100 ký tự)", "source_targets": [ {"source": "Loại thực thể nguồn", "target": "Loại thực thể đích"} ], "attributes": [] } ], - "analysis_summary": "Giải thích ngắn gọn phân tích của bạn về văn bản (Tiếng Việt)" + "analysis_summary": "Phân tích ngắn gọn nội dung văn bản (bằng tiếng Việt)" } ``` -## Hướng dẫn Thiết kế (CỰC KỲ QUAN TRỌNG!) +## Hướng dẫn thiết kế (Cực kỳ quan trọng!) -### 1. Thiết kế loại Thực thể (Entity Types) - Phải tuân thủ nghiêm ngặt +### 1. Thiết kế loại thực thể - Phải tuân thủ nghiêm ngặt -**Yêu cầu số lượng: Đúng 10 loại Thực thể.** +**Yêu cầu số lượng: Phải có CHÍNH XÁC 10 loại thực thể.** -**Yêu cầu về cấu trúc phân cấp (Phải có cả Loại cụ thể và Loại bao quát/fallback):** +**Yêu cầu về cấu trúc phân cấp (Phải bao gồm cả loại cụ thể và loại dự phòng):** 10 loại thực thể của bạn phải bao gồm cấp độ sau: @@ -85,7 +198,7 @@ A. **Loại bao quát (Fallback Types) (BẮT BUỘC, phải nằm ở 2 vị tr - `Organization`: Là loại bao quát cho MỌI tổ chức. Đặc trưng cho các tổ chức nhỏ hoặc không phù hợp với các loại tổ chức cụ thể khác. B. **Loại cụ thể (8 loại, phụ thuộc vào nội dung văn bản)**: - - Thiết kế các loại cụ thể cho các vai chính được nhắc đến nhiều nhất trong văn bản. + - Thiết kế các loại cụ thể cho các vai trò chính được nhắc đến nhiều nhất trong văn bản. - Ví dụ: Nếu văn bản nói về scandal trường học, có thể có: `Student`, `Professor`, `University` - Ví dụ: Nếu văn bản là câu chuyện kinh doanh, có thể có: `Company`, `CEO`, `Employee` @@ -95,9 +208,9 @@ B. **Loại cụ thể (8 loại, phụ thuộc vào nội dung văn bản)**: - Tương tự, tổ chức nhỏ bé hoặc nhóm học tập tạm thời nên thuộc `Organization` **Nguyên tắc cho các loại Cụ thể:** -- Nhận dạng tần suất xuất hiện và sức ảnh hưởng tới cốt truyện để xây dựng loại thực thể. +- Nhận diện các vai trò xuất hiện với tần suất cao hoặc quan trọng từ văn bản. - Mỗi loại nên có một ranh giới rõ ràng, không bị chồng chéo. -- Thuộc tính mô tả (description) phải giải thích vì sao loại này tách biệt. +- Phần description phải giải thích rõ sự khác biệt giữa loại này và loại bao quát. ### 2. Thiết kế Cạnh/Quan hệ (Edge Types) @@ -113,45 +226,14 @@ B. **Loại cụ thể (8 loại, phụ thuộc vào nội dung văn bản)**: ## Loại Thực thể tham khảo -**Loại cá nhân (Cụ thể):** -- Student: Học sinh/Sinh viên -- Professor: Giáo sư/Học giả -- Journalist: Nhà báo/Phóng viên -- Celebrity: Người nổi tiếng/Idol -- Executive: Các giám đốc, CEO, cấp lãnh đạo -- Official: Các vị công chức chính phủ -- Lawyer: Luật sư -- Doctor: Y sĩ/Bác sĩ +- Nhóm cá nhân (Cụ thể): Student, Professor, Journalist, Celebrity, Executive, Official, Lawyer, Doctor. +- Nhóm cá nhân (Bao quát): Person. +- Nhóm tổ chức (Cụ thể): University, Company, GovernmentAgency, MediaOutlet, Hospital, School, NGO. +- Nhóm tổ chức (Bao quát): Organization. -**Loại cá nhân (Bao quát):** -- Person: Là loại bao quát cho MỌI cá nhân tự nhiên nào không thuộc chi tiết ở trên. +## Tham khảo loại quan hệ -**Loại tổ chức (Cụ thể):** -- University: Đại học hoặc học viện -- Company: Doanh nghiệp hay Công ty, tập đoàn -- GovernmentAgency: Cơ quan quản lý, các cơ quan ban ngành công quyền -- MediaOutlet: Truyền thông hay Tạp chí, Đài tin tức -- Hospital: Bệnh viện / Trung tâm y tế -- School: Bậc tiểu/trung học -- NGO: Các loại Tổ chức phi chính phủ hoặc từ thiện - -**Loại tổ chức (Bao quát):** -- Organization: Là loại bao quát cho MỌI cơ cấu hợp tác không thuôc chi tiết tổ chức ở trên. - -## Loại Khái niệm Liên kết (Quan Hệ) - -- WORKS_FOR: Làm việc và ăn lương bởi tổ chức -- STUDIES_AT: Đang học tại nhà trường -- AFFILIATED_WITH: Liên quan, Trực thuộc vào đơn vị -- REPRESENTS: Thể hiện tư cách hành động đại diện cho tập thể -- REGULATES: Theo dõi, quản lý, thanh tra chính sách -- REPORTS_ON: Tác nghiệp báo chí, có tin về hiện tượng -- COMMENTS_ON: Có phản hồi hoặc lên tiếng về tranh cãi -- RESPONDS_TO: Hành động đáp trả -- SUPPORTS: Theo phe ủng hộ điều luật -- OPPOSES: Phản đối chính sách -- COLLABORATES_WITH: Tham gia phối ứng xử lý sự cố. -- COMPETES_WITH: Quan hệ thù địch. +WORKS_FOR, STUDIES_AT, AFFILIATED_WITH, REPRESENTS, REGULATES, REPORTS_ON, COMMENTS_ON, RESPONDS_TO, SUPPORTS, OPPOSES, COLLABORATES_WITH, COMPETES_WITH. """ @@ -204,8 +286,8 @@ class OntologyGenerator: return result - # Định mức giới hạn độ dài ký tự tối đa của đoạn văn bản có thể gửi cho LLM (5 vạn chữ) - MAX_TEXT_LENGTH_FOR_LLM = 80000 + # Định mức giới hạn độ dài ký tự tối đa của đoạn văn bản có thể gửi cho LLM (10 vạn chữ) + MAX_TEXT_LENGTH_FOR_LLM = 100000 def _build_user_message( self, @@ -222,9 +304,36 @@ class OntologyGenerator: # Nếu vượt quá giới hạn tối đa, thực hiện cắt bớt (Việc này chỉ ảnh hưởng prompt gửi nhận diện Ontology, không ảnh hưởng thư viện Graph building ở sau) if len(combined_text) > self.MAX_TEXT_LENGTH_FOR_LLM: combined_text = combined_text[:self.MAX_TEXT_LENGTH_FOR_LLM] - combined_text += f"\n\n...(Văn bản gốc dài {original_length} chữ, đã chủ động cắt lấy {self.MAX_TEXT_LENGTH_FOR_LLM} chữ đầu tiên để phục vụ phân tích Ontology)..." + combined_text += f"\n\n...(Original text is {original_length} characters long; the first {self.MAX_TEXT_LENGTH_FOR_LLM} characters have been proactively truncated for Ontology analysis)..." + +# message = f"""## Simulation Requirements + +# {simulation_requirement} + +# ## Document Content + +# {combined_text} +# """ - message = f"""## Nhu cầu mô phỏng +# if additional_context: +# message += f""" +# ## Additional Context + +# {additional_context} +# """ + +# message += """ +# Based on the content above, please design entity types and relationship types suitable for social media public opinion simulation. + +# **Rules that MUST be followed**: +# 1. You must output EXACTLY 10 entity types. +# 2. The last 2 types must be fallback types: Person (individual fallback) and Organization (organization fallback). +# 3. The first 8 types should be specific types designed based on the text content. +# 4. All entity types must be real-world subjects capable of speaking/interacting; they cannot be abstract concepts. +# 5. Attribute names cannot use reserved words like name, uuid, or group_id; use alternatives like full_name, org_name, etc. +# """ + + message = f"""## Yêu cầu mô phỏng {simulation_requirement} @@ -241,14 +350,14 @@ class OntologyGenerator: """ message += """ -Dựa vào các thông tin trên đây, hãy thiết kế các loại mô hình Thực Thể và Quan Hệ phù hợp để phục vụ việc mô phỏng dư luận trên mạng xã hội. +Dựa trên các nội dung trên, hãy thiết kế các loại thực thể và loại quan hệ phù hợp cho việc mô phỏng dư luận xã hội. -**Các quy tắc BẮT BUỘC tuân thủ**: -1. Số lượng chính xác: Xuất phải CHUẨN XÁC 10 loại Thực thể -2. 2 vị trí cuối cùng bắt buộc là từ Khoá phụ (Fallback): Person (Cho cá nhân) và Organization (Cho Tổ chức) -3. 8 vị trí đầu tiên phải phân tích và suy luận dựa vào cấu trúc của chính văn bản truyền vào -4. Tất cả các thực thể được liệt kê phải đóng vai trò là Chủ thể (nhân vật có thể lên tiếng ngoài đời thực), KHÔNG ĐƯỢC dùng làm khái niệm trừu tượng. -5. Tên biến thuộc tính KHÔNG ĐƯỢC là name, uuid, group_id hay các biến số bảo lưu của hệ thống khác. Vui lòng chuyển thành full_name, org_name, v.v. +**Các quy tắc BẮT BUỘC phải tuân thủ**: +1. Phải xuất chính xác 10 loại thực thể. +2. 2 loại cuối cùng phải là loại dự phòng: Person (Cá nhân dự phòng) và Organization (Tổ chức dự phòng). +3. 8 loại đầu tiên là các loại cụ thể được thiết kế dựa trên nội dung văn bản. +4. Tất cả các loại thực thể phải là những chủ thể có thể phát ngôn trong thực tế, không được là các khái niệm trừu tượng. +5. Tên thuộc tính không được sử dụng các từ khóa hệ thống như name, uuid, group_id; hãy thay thế bằng full_name, org_name, v.v. """ return message @@ -355,15 +464,15 @@ Dựa vào các thông tin trên đây, hãy thiết kế các loại mô hình """ code_lines = [ '"""', - 'Các loại đối tượng (Thực thể) tuỳ chỉnh', - 'Được khởi tạo tự động bởi công cụ MiroFish, ứng dụng vào việc chạy giả lập diễn biến dư luận', + 'Custom Entity Type Definitions', + 'Automatically generated by MiroFish for social media public opinion simulation', '"""', '', 'from pydantic import Field', 'from zep_cloud.external_clients.ontology import EntityModel, EntityText, EdgeModel', '', '', - '# ============== Định nghĩa Tên Lớp Các thực thể (Entity) ==============', + '# ============== Entity Type Definitions ==============', '', ] @@ -390,7 +499,7 @@ Dựa vào các thông tin trên đây, hãy thiết kế các loại mô hình code_lines.append('') code_lines.append('') - code_lines.append('# ============== Định nghĩa Các Nhóm Quan Hệ/Hành Vi (Edge) ==============') + code_lines.append('# ============== Relationship Type Definitions ==============') code_lines.append('') # Khởi tạo các đoạn mã tạo lập Relationship (Edges) @@ -419,7 +528,7 @@ Dựa vào các thông tin trên đây, hãy thiết kế các loại mô hình code_lines.append('') # Tự động kết xuất ra dictionary mapping từ Tên Loại - sang class Object - code_lines.append('# ============== Các tuỳ chỉnh Map Cấu Hình ==============') + code_lines.append('# ============== Type Configuration ==============') code_lines.append('') code_lines.append('ENTITY_TYPES = {') for entity in ontology.get("entity_types", []): diff --git a/backend/app/services/simulation_config_generator.py b/backend/app/services/simulation_config_generator.py index 3b794969..7dc9d92d 100644 --- a/backend/app/services/simulation_config_generator.py +++ b/backend/app/services/simulation_config_generator.py @@ -546,27 +546,72 @@ class SimulationConfigGenerator: # Cắt lấy số lượng Tối đa số lượng (Chiếm 80% từ số lượng lượng Agent thực thể) max_agents_allowed = max(1, int(num_entities * 0.9)) + +# prompt = f"""Based on the following simulation requirements, generate a time simulation configuration. + +# {context_truncated} + +# ## Task +# Please generate a time configuration JSON. + +# ### General Principles (For reference only; adjust flexibly based on specific events and participant groups): +# - The user group consists of Vietnamese people; must comply with Hanoi Time (CST) daily routines. +# - 0:00–5:00 AM: Almost no activity (Activity Coefficient: 0.05). +# - 6:00–8:00 AM: Gradual increase in activity (Activity Coefficient: 0.4). +# - 9:00 AM–6:00 PM (Work hours): Moderate activity (Activity Coefficient: 0.7). +# - 7:00 PM–10:00 PM: Peak period (Activity Coefficient: 1.5). +# - After 11:00 PM: Activity declines (Activity Coefficient: 0.5). +# - General Pattern: Low activity in the early morning, gradual increase in the morning, moderate during work hours, and peak in the evening. +# - **Important:**: The example values below are for reference only. You need to adjust specific periods based on the nature of the event and characteristics of the participant group. +# - e.g., The peak for students might be 9:00 PM–11:00 PM; Media groups remain active all day; Official organizations only during work hours. +# - e.g., Breaking news may lead to discussions late at night; off_peak_hours can be shortened accordingly. + +# ### Return JSON Format (Do not use Markdown) + +# Example: +# {{ +# "total_simulation_hours": 72, +# "minutes_per_round": 60, +# "agents_per_hour_min": 5, +# "agents_per_hour_max": 50, +# "peak_hours": [19, 20, 21, 22], +# "off_peak_hours": [0, 1, 2, 3, 4, 5], +# "morning_hours": [6, 7, 8], +# "work_hours": [9, 10, 11, 12, 13, 14, 15, 16, 17, 18], +# "reasoning": "Time configuration explanation for this specific event." +# }} + +# Field Descriptions: +# - total_simulation_hours (int): Total simulation duration, 24–168 hours. Short for breaking news, long for sustained topics. +# - minutes_per_round (int): Duration per round, 30–120 minutes, suggested 60 minutes. +# - agents_per_hour_min (int): Minimum activated agents per hour (Range: 1-{max_agents_allowed}). +# - agents_per_hour_max (int): Maximum activated agents per hour (Range: 1-{max_agents_allowed}). +# - peak_hours (int array): Peak hours, adjusted based on the participant group. +# - off_peak_hours (int array): Off-peak hours, usually late night/early morning. +# - morning_hours (int array): Morning hours. +# - work_hours (int array): Working hours. +# - reasoning (string): Brief explanation of why this configuration was chosen.""" - prompt = f"""Dựa vào yêu cầu Mô phỏng này, hãy tự động gen cho 1 file Thông số thời gian + prompt = f"""Dựa trên các yêu cầu mô phỏng dưới đây, hãy tạo cấu hình mô phỏng thời gian. {context_truncated} -## Task công việc -Vui lòng xuất ra kết quả Thời gian dưới định dạng format JSON +## Nhiệm vụ +Hãy tạo JSON cấu hình thời gian. -### Logic cơ bản có thể cần để tham khảo (Hãy dựa trên nhu cầu của user và hoàn cảnh để suy ra): -- Vị trí của người tham gia là người dùng mạng Trung Quốc, cần sinh hoạt bằng thói quen sinh học giờ chuẩn Bắc Kinh (China Time: GMT+8). -- Không xuất hiện hay có dấu hiệu online của người dùng từ 0-5 giờ sáng (Hệ số Active 0.05). -- Tăng nhẹ số lượng truy cập lại mức trung thành khoảng giữa 6-8 giờ sáng (Hệ số Active 0.4). -- Số lượng active hoạt động ở mức bình ổn khoảng từ 9-18 giờ sáng (Hệ số Active 0.7). -- Khung giờ sôi động nhất sẽ tập trung quanh 19-22 giờ tối (Hệ số Active 1.5). -- Tỷ lệ giảm lại sau 23 giờ (Hệ số Active 0.5). -- Cơ chế bình thường: Đêm không online, sáng bắt đầu đăng bài, giờ hành chính bình bình và cao trào trong buổi tối thức đêm -- **Chỉ Dẫn Rất Quan Trọng**: Những thông tin từ list được lấy để tham chiếu. Còn thông số thật sự còn phải tùy theo Đặc điểm, Tình Huống đối tượng ở Mạng và Thời Điểm Sự Kiện để gen ra - - Ví dụ: Số lượng sinh viên thức đêm ở từ 21-23 giờ thường là lớn; Media báo đài thì hay đăng tin liên tục cả ngày theo ca; Tài khoản văn phòng của các Cơ Quan chức năng chỉ trả lời giờ làm việc Hành Chính... - - Hoặc ví dụ: Các Biến Cố hoặc Drama xảy ra trong đêm khuya thì sẽ dẫn đến Lượng truy cập ban đêm có dấu hiệu đi lên, trong khi đó Off_peak_hours vì lẽ đó mà sẽ có khi co lại cho ngắn... +### Nguyên tắc cơ bản (Chỉ mang tính chất tham khảo, cần điều chỉnh linh hoạt theo sự kiện cụ thể và nhóm đối tượng tham gia): +- Nhóm người dùng là người Việt Nam, cần phù hợp với thói quen sinh hoạt theo giờ Hà Nội. +- 0-5 giờ sáng: Hầu như không có hoạt động (Hệ số hoạt động: 0.05). +- 6-8 giờ sáng: Hoạt động tăng dần (Hệ số hoạt động: 0.4). +- 9-18 giờ (Giờ làm việc): Hoạt động trung bình (Hệ số hoạt động: 0.7). +- 19-22 giờ tối: Giai đoạn cao điểm (Hệ số hoạt động: 1.5). +- Sau 23 giờ: Mức độ hoạt động giảm xuống (Hệ số hoạt động: 0.5). +- Quy luật chung: Thấp vào rạng sáng, tăng dần vào buổi sáng, trung bình trong giờ làm việc và cao điểm vào buổi tối. +- **Quan trọng**: Các giá trị ví dụ dưới đây chỉ mang tính chất tham khảo, bạn cần điều chỉnh các khung giờ cụ thể dựa trên tính chất sự kiện và đặc điểm của nhóm đối tượng tham gia. + - Ví dụ: Cao điểm của nhóm sinh viên có thể là 21-23 giờ; Nhóm truyền thông hoạt động cả ngày; Các cơ quan chính thống chỉ hoạt động trong giờ hành chính. + - Ví dụ: Tin tức nóng hổi (hotspot) có thể dẫn đến thảo luận vào đêm muộn; off_peak_hours có thể được rút ngắn tương ứng. -### Định dạng Format của JSON Return (Lưu ý Tuyệt đối Không Return Markdown code block, chỉ Return Format Json Thuần Túy) +### Định dạng JSON trả về (Không sử dụng markdown) Ví dụ Format như sau: {{ @@ -578,21 +623,23 @@ Ví dụ Format như sau: "off_peak_hours": [0, 1, 2, 3, 4, 5], "morning_hours": [6, 7, 8], "work_hours": [9, 10, 11, 12, 13, 14, 15, 16, 17, 18], - "reasoning": "Một đoạn văn lời nói cho biết Bạn đã dựa theo yêu cầu như thế nào để gen các Thông Số trên" + "reasoning": "Giải thích cấu hình thời gian cho sự kiện này." }} -Các Khóa của Json có nghĩa là: -- total_simulation_hours (int): Mô tả tổng giới hạn thời gian (Đơn vị giờ), Có giá trị trong khung 24-168h, Tùy vào biến cố Drama nóng để chọn. Các chủ đề chạy Drama ít hơn thì nên cấp ngắn -- minutes_per_round (int): Số Time trên mỗi Khung Đợt Thời Gian Thực Của Simulation để mô phỏng cho 1 phiên trong game, lấy giá trị 30-120 phút. Đề xuất: 60 (1 giờ) -- agents_per_hour_min (int): Số Agent online tối thiểu trong một tiếng mô phỏng (Phạm vi {1}-{max_agents_allowed}) -- agents_per_hour_max (int): Số lượng lên mạng tối đa (Phạm vi {1}-{max_agents_allowed}) -- peak_hours (mảng int list): Thời điểm đỉnh sóng Cao Điểm, cân nhắc theo Đối tượng để quyết định -- off_peak_hours (mảng int list): Đỉnh sóng Đáy, ít ai quan tâm -- morning_hours (mảng int list): Khoảng thời điểm đầu buổi sáng -- work_hours (mảng int list): Khung hành chính công việc -- reasoning (string): Sự giải thích từ LLM""" +Mô tả các trường: +- total_simulation_hours (int): Tổng thời gian mô phỏng, từ 24-168 giờ. Ngắn cho sự kiện đột xuất, dài cho các chủ đề kéo dài. +- minutes_per_round (int): Thời gian mỗi hiệp, 30-120 phút, khuyến nghị 60 phút. +- agents_per_hour_min (int): Số lượng Agent kích hoạt tối thiểu mỗi giờ (Phạm vi: 1-{max_agents_allowed}). +- agents_per_hour_max (int): Số lượng Agent kích hoạt tối đa mỗi giờ (Phạm vi: 1-{max_agents_allowed}). +- peak_hours (int array): Khung giờ cao điểm, điều chỉnh theo nhóm tham gia. +- off_peak_hours (int array): Khung giờ thấp điểm, thường là đêm khuya rạng sáng. +- morning_hours (mảng int): Khung giờ buổi sáng. +- work_hours (mảng int): Khung giờ làm việc. +- reasoning (string): Giải thích ngắn gọn lý do tại sao cấu hình như vậy.""" + + # system_prompt = "You are a social media simulation expert. Return in pure JSON format; time configurations must comply with Vietnamese daily routines." - system_prompt = "Bạn là 1 Tool chuyên mô phỏng môi trường làm việc trên mxh bằng thuật toán LLM để cung cấp ra Cấu hình Time. Hãy xuất JSON." + system_prompt = "Bạn là chuyên gia mô phỏng mạng xã hội. Trả về định dạng JSON thuần túy; cấu hình thời gian cần phù hợp với thói quen sinh hoạt của người Việt Nam." try: return self._call_llm_with_retry(prompt, system_prompt) @@ -604,14 +651,14 @@ Các Khóa của Json có nghĩa là: """Tạo sẵn file chuẩn nếu bị đơ để trả ra theo múi giờ chuẩn sinh hoạt China""" return { "total_simulation_hours": 72, - "minutes_per_round": 60, # 1 Hour / Vòng -> Rút gắn Time + "minutes_per_round": 60, # 1 Hour / Vòng -> Rút ngắn Time "agents_per_hour_min": max(1, num_entities // 15), "agents_per_hour_max": max(5, num_entities // 5), "peak_hours": [19, 20, 21, 22], "off_peak_hours": [0, 1, 2, 3, 4, 5], "morning_hours": [6, 7, 8], "work_hours": [9, 10, 11, 12, 13, 14, 15, 16, 17, 18], - "reasoning": "Mặc định sử dụng Thời gian làm việc của người dùng Trung Quốc (1 Giờ/vòng)" + "reasoning": "Defaults to Vietnamese users' daily routines and working hours (1 hour/round)" } def _parse_time_config(self, result: Dict[str, Any], num_entities: int) -> TimeSimulationConfig: @@ -678,37 +725,68 @@ Các Khóa của Json có nghĩa là: # Có chặn để lấy chuỗi theo cấu hình chiều dài giới hạn context_truncated = context[:self.EVENT_CONFIG_CONTEXT_LENGTH] - - prompt = f"""Gen cấu hình Event dưới các tham chiếu từ Yêu cầu (Requirements): -Simulation Requirements: {simulation_requirement} +# prompt = f"""Based on the following simulation requirements, generate an event configuration. + +# Simulation Requirements: {simulation_requirement} + +# {context_truncated} + +# ## Available Entity Types and Examples +# {type_info} + +# ## Task +# Please generate an event configuration JSON: +# - Extract key hot topic keywords. +# - Describe the direction of public opinion development. +# - Design initial post content; **each post must specify a poster_type (publisher type)**. + +# **IMPORTANT**: The poster_type must be selected from the "Available Entity Types" above so that initial posts can be assigned to the appropriate Agent for publishing. +# For example: Official statements should be posted by Official/University types, news by MediaOutlet, and student perspectives by Student. + +# Return in JSON format (no markdown): +# {{ +# "hot_topics": ["Keyword1", "Keyword2", ...], +# "narrative_direction": "", +# "initial_posts": [ +# {{"content": "Post Content...", "poster_type": "Entity type (must be selected from available types)"}}, +# ... +# ], +# "reasoning": "" +# }}""" + +# system_prompt = "You are a public opinion analysis expert. Return in pure JSON format. Ensure that poster_type exactly matches the available entity types." + + prompt = f"""Dựa trên các yêu cầu mô phỏng sau đây, hãy tạo cấu hình sự kiện. + +Yêu cầu mô phỏng: {simulation_requirement} {context_truncated} -## Các Entity Type có cung cấp & VD minh họa: +## Các loại thực thể khả dụng và ví dụ {type_info} -## Task công việc -Vui lòng xuất ra kết quả Thời gian dưới định dạng format JSON: -- Chỉ định List các Hot Keyword để kéo trend -- Miêu tả định hướng thảo luận cho trend hiện tại -- Đăng tải Post đầu tiên (Initial_Post) lên với nguyên tắc: **Phải đi kèm với tham số người up Post (poster_type)** +## Nhiệm vụ +Vui lòng tạo JSON cấu hình sự kiện: +- Trích xuất các từ khóa chủ đề nóng (hot topics). +- Mô tả hướng phát triển của dư luận. +- Thiết kế nội dung các bài đăng khởi tạo, **mỗi bài đăng phải chỉ định poster_type (loại người đăng)**. -**RẤT QUAN TRỌNG**: Người Poster Type (poster_type) Phải trùng khớp/được lấy từ danh mục từ mục "Các Entity Type" đã cho để gán. Tránh báo lỗi cho Agent - Ví dụ: official announcements should be posted by Official/University type, news by MediaOutlet, and student opinions by Student. +**QUAN TRỌNG**: poster_type phải được chọn từ "Các loại thực thể khả dụng" ở trên để các bài đăng khởi tạo có thể được phân bổ cho đúng Agent phù hợp. + Ví dụ: Các tuyên bố chính thức nên được đăng bởi loại Official/University, tin tức bởi MediaOutlet, và quan điểm sinh viên bởi Student. -Format trả ra (Tuyệt đối Không Markdown, chỉ lấy format chuỗi chuẩn): +Trả về định dạng JSON (không sử dụng markdown): {{ - "hot_topics": ["Keyword1", "Keyword2", ...], - "narrative_direction": "<Đoạn Text dài định hướng Dư luận (Narrative)>", + "hot_topics": ["Từ khóa 1", "Từ khóa 2", ...], + "narrative_direction": "", "initial_posts": [ - {{"content": "Post Content...", "poster_type": "Người sẽ Post ra nội dung (Hạn chế tùy tiện vì nó lấy từ mảng danh sách Loại Entity cho trước)"}}, + {{"content": "Nội dung bài đăng...", "poster_type": "Loại thực thể (phải chọn từ các loại khả dụng)"}}, ... ], - "reasoning": "" + "reasoning": "" }}""" - system_prompt = "Bạn là 1 Chuyên gia về Data Dư luận, Yêu cầu làm việc trên chuỗi JSON nghiêm ngặt. Tránh Lỗi." + system_prompt = "Bạn là chuyên gia phân tích dư luận. Trả về định dạng JSON thuần túy. Lưu ý rằng poster_type phải khớp chính xác với các loại thực thể khả dụng." try: return self._call_llm_with_retry(prompt, system_prompt) @@ -835,43 +913,81 @@ Format trả ra (Tuyệt đối Không Markdown, chỉ lấy format chuỗi chu "summary": e.summary[:summary_len] if e.summary else "" }) - prompt = f"""Tạo profile Social Media Activity Configs cho từng Thực thể sau. +# prompt = f"""Based on the following information, generate social media activity configurations for each entity. -Nhu cầu: {simulation_requirement} +# Simulation Requirements: {simulation_requirement} -## List các thực thể +# ## Entity List +# ```json +# {json.dumps(entity_list, ensure_ascii=False, indent=2)} +# ``` + +# ## Task +# Generate activity configurations for each entity, noting: +# - **Time aligns with Vietnamese daily routines**: Almost no activity between 0-5 AM; most active during 7-10 PM (19:00-22:00). +# - **Official Institutions (University/GovernmentAgency)**: Low activity (0.1-0.3), active during work hours (9:00-17:00), slow response (60-240 mins), high influence (2.5-3.0). +# - **Media (MediaOutlet)**: Medium activity (0.4-0.6), active all day (8:00-23:00), fast response (5-30 mins), high influence (2.0-2.5). +# - **Individuals (Student/Person/Alumni)**: High activity (0.6-0.9), active mainly in the evening (18:00-23:00), fast response (1-15 mins), low influence (0.8-1.2). +# - **Public Figures/Experts**: Medium activity (0.4-0.6), medium-high influence (1.5-2.0). + +# Return in JSON format (no markdown): +# {{ +# "agent_configs": [ +# {{ +# "agent_id": , +# "activity_level": <0.0-1.0>, +# "posts_per_hour": , +# "comments_per_hour": , +# "active_hours": [], +# "response_delay_min": , +# "response_delay_max": , +# "sentiment_bias": <-1.0 to 1.0>, +# "stance": "", +# "influence_weight": +# }}, +# ... +# ] +# }}""" + +# system_prompt = "You are a social media behavior analysis expert. Return pure JSON. Configurations must comply with Vietnamese daily routines." + + prompt = f"""Dựa trên các thông tin sau đây, hãy tạo cấu hình hoạt động trên mạng xã hội cho từng thực thể. + +Yêu cầu mô phỏng: {simulation_requirement} + +## Danh sách thực thể ```json {json.dumps(entity_list, ensure_ascii=False, indent=2)} ``` -## Task Công Việc -Trả ra cho Từng Entity các bộ Activity Profile tham chiều theo các quy tắc ngầm sau: -- **Tập quán Sinh hoạt Trung Quốc**: 0-5h sáng gần như sẽ hiếm ai onl, 19-22h tối lượng tương tác rất sôi nổi -- **Đại diện Cơ quan (University/GovernmentAgency)**: Tần suất (0.1-0.3), làm việc trong giờ hành chính (9-17h), delay hơi trễ (60-240 phút), Trọng lượng lời nói cao (2.5-3.0) -- **Truyền Thông Báo Đài (MediaOutlet)**: Tần suất TB (0.4-0.6), Hầu như online nguyên ngày (8-23h), Trễ ít (5-30 phút), Trọng lượng cũng Cao (2.0-2.5) -- **Người Dùng Bình thường (Student/Person/Alumni)**: Tần suất cao (0.6-0.9), Onl chủ yếu để cãi nhau buổi tối (18-23h), Tương tác lẹ như hack (1-15 min), Uy tín lời nói khá lèo tèo (0.8-1.2) -- **Học giả/Chuyên gia/Kols**: Tần suất TB (0.4-0.6), Uy tín tương đối (1.5-2.0) +## Nhiệm vụ +Tạo cấu hình hoạt động cho từng thực thể, lưu ý: +- **Thời gian phù hợp với thói quen của người Việt Nam**: Gần như không hoạt động từ 0-5 giờ sáng, hoạt động mạnh nhất từ 19-22 giờ tối. +- **Cơ quan chính thống (University/GovernmentAgency)**: Hoạt động thấp (0.1-0.3), hoạt động trong giờ làm việc (9:00-17:00), phản hồi chậm (60-240 phút), tầm ảnh hưởng cao (2.5-3.0). +- **Truyền thông (MediaOutlet)**: Hoạt động trung bình (0.4-0.6), hoạt động cả ngày (8:00-23:00), phản hồi nhanh (5-30 phút), tầm ảnh hưởng cao (2.0-2.5). +- **Cá nhân (Student/Person/Alumni)**: Hoạt động cao (0.6-0.9), hoạt động chủ yếu vào buổi tối (18:00-23:00), phản hồi nhanh (1-15 phút), tầm ảnh hưởng thấp (0.8-1.2). +- **Người công chúng/Chuyên gia**: Hoạt động trung bình (0.4-0.6), tầm ảnh hưởng trung bình cao (1.5-2.0). -Trả đúng 1 object JSON ko format MD: +Trả về định dạng JSON (không sử dụng markdown): {{ "agent_configs": [ {{ - "agent_id": , + "agent_id": , "activity_level": <0.0-1.0>, - "posts_per_hour": , - "comments_per_hour": , - "active_hours": [], - "response_delay_min": , - "response_delay_max": , - "sentiment_bias": <-1.0 đến 1.0 (Tiêu cực sang Tích Cực)>, + "posts_per_hour": , + "comments_per_hour": , + "active_hours": [], + "response_delay_min": <Độ trễ phản hồi tối thiểu tính bằng phút>, + "response_delay_max": <Độ trễ phản hồi tối đa tính bằng phút>, + "sentiment_bias": <-1.0 đến 1.0>, "stance": "", - "influence_weight": + "influence_weight": }}, ... ] }}""" - system_prompt = "Hệ thống Analysis Chuyên gia. Luôn trả về Format Object Array bằng JSON. Và tuân thủ sinh học." + system_prompt = "Bạn là chuyên gia phân tích hành vi mạng xã hội. Trả về JSON thuần túy. Cấu hình phải phù hợp với thói quen sinh hoạt của người Việt Nam." try: result = self._call_llm_with_retry(prompt, system_prompt) diff --git a/backend/app/services/zep_tools.py b/backend/app/services/zep_tools.py index d5230d2f..4de8331b 100644 --- a/backend/app/services/zep_tools.py +++ b/backend/app/services/zep_tools.py @@ -1109,23 +1109,41 @@ class ZepToolsService: Giúp phân rã một câu hỏi lớn / phức tạp thành nhiều câu hỏi nhỏ lẻ có thể query độc lập trên cơ sở dữ liệu. """ - system_prompt = """You are a professional problem analysis expert. Your task is to break down a complex query into multiple sub-queries that can be independently observed in the simulated world. +# system_prompt = """You are a professional problem analysis expert. Your task is to decompose a complex problem into multiple sub-questions that can be independently observed within a simulated world. -Requirements: -1. Each sub-query should be specific enough to find concrete Agent behaviors or events. -2. Sub-queries should cover different dimensions of the original query (Who, What, Why, How, When, Where). -3. Sub-queries must relate to the simulation context. -4. Return exactly in JSON format: {"sub_queries": ["sub_query 1", "sub_query 2", ...]}""" +# Requirements: +# 1. Each sub-question should be specific enough to identify relevant Agent behaviors or events within the simulation. +# 2. Sub-questions should cover different dimensions of the original problem (e.g., Who, What, Why, How, When, Where). +# 3. Sub-questions must be relevant to the simulation scenario. +# 4. Return in JSON format: {"sub_queries": ["Sub-question 1", "Sub-question 2", ...]}""" - user_prompt = f"""Simulation background: +# user_prompt = f"""Simulation Requirement Background: +# {simulation_requirement} + +# {f"Report context: {report_context[:500]}" if report_context else ""} + +# Please decompose the following question into {max_queries} sub-questions: +# {query} + +# Return the list of sub-questions in JSON format.""" + + system_prompt = """Bạn là một chuyên gia phân tích vấn đề chuyên nghiệp. Nhiệm vụ của bạn là chia nhỏ một vấn đề phức tạp thành nhiều câu hỏi phụ có thể quan sát độc lập trong thế giới mô phỏng. + +Yêu cầu: +1. Mỗi câu hỏi phụ phải đủ cụ thể để có thể tìm thấy các hành vi của Agent hoặc các sự kiện liên quan trong thế giới mô phỏng. +2. Các câu hỏi phụ nên bao quát các khía cạnh khác nhau của vấn đề gốc (Ví dụ: Ai, Cái gì, Tại sao, Như thế nào, Khi nào, Ở đâu). +3. Các câu hỏi phụ phải liên quan đến kịch bản mô phỏng. +4. Trả về định dạng JSON: {"sub_queries": ["Câu hỏi phụ 1", "Câu hỏi phụ 2", ...]}""" + + user_prompt = f"""Bối cảnh yêu cầu mô phỏng: {simulation_requirement} -{f"Report context: {report_context[:500]}" if report_context else ""} +{f"Ngữ cảnh báo cáo: {report_context[:500]}" if report_context else ""} -Please break down the following query into {max_queries} sub-queries: +Hãy chia nhỏ vấn đề sau đây thành {max_queries} các câu hỏi phụ: {query} -Return the JSON format.""" +Trả về danh sách các câu hỏi phụ dưới định dạng JSON.""" try: response = self.llm.chat_json( @@ -1358,16 +1376,28 @@ Return the JSON format.""" combined_prompt = "\n".join([f"{i+1}. {q}" for i, q in enumerate(result.interview_questions)]) # Thêm các prefix tối ưu hoá, ràng buộc format câu trả lời của Agent + # INTERVIEW_PROMPT_PREFIX = ( + # "You are currently being interviewed. Please combine your persona, " + # "all past memories, and actions to answer the following questions directly in plain text.\n" + # "Response Requirements:\n" + # "1. Answer directly using natural language; do not call any tools.\n" + # "2. Do not return JSON format or tool-call formats.\n" + # "3. Do not use Markdown headers (e.g., #, ##, ###).\n" + # "4. Answer questions one by one according to their numbers. Each answer must start with 'Question X:' (where X is the question number).\n" + # "5. Use a blank line to separate the answers for each question.\n" + # "6. Answers must have substantive content; provide at least 2-3 sentences for each question.\n\n" + # ) + INTERVIEW_PROMPT_PREFIX = ( - "You are being interviewed. Please combine your profile, all your past memories and actions, " - "and directly answer the following questions in pure text.\n" - "Reply requirements:\n" - "1. Answer directly in natural language, do not call any tools.\n" - "2. Do not return JSON format or tool call formats.\n" - "3. Do not use Markdown headers (like #, ##, ###).\n" - "4. Answer questions one by one according to their numbers, start each answer with 'Question X:' (X is the number).\n" - "5. Separate each answer with a blank line.\n" - "6. Answers must have substance, at least 2-3 sentences per question.\n\n" + "Bạn đang tham gia một buổi phỏng vấn. Hãy kết hợp nhân vật (persona), " + "tất cả ký ức và hành động trong quá khứ của bạn để trả lời trực tiếp các câu hỏi dưới đây bằng văn bản thuần túy.\n" + "Yêu cầu phản hồi:\n" + "1. Trả lời trực tiếp bằng ngôn ngữ tự nhiên, không gọi bất kỳ công cụ nào.\n" + "2. Không trả về định dạng JSON hoặc định dạng gọi công cụ.\n" + "3. Không sử dụng tiêu đề Markdown (như #, ##, ###).\n" + "4. Trả lời từng câu một theo số thứ tự, mỗi câu trả lời bắt đầu bằng 'Câu hỏi X:' (X là số thứ tự câu hỏi).\n" + "5. Sử dụng một dòng trống để phân tách giữa các câu trả lời.\n" + "6. Câu trả lời phải có nội dung thực tế, mỗi câu hỏi cần trả lời ít nhất 2-3 câu văn.\n\n" ) optimized_prompt = f"{INTERVIEW_PROMPT_PREFIX}{combined_prompt}" @@ -1585,31 +1615,56 @@ Return the JSON format.""" "interested_topics": profile.get("interested_topics", []) } agent_summaries.append(summary) - - system_prompt = """You are a professional interview planning expert. Your task is to select the most suitable target agents for an interview based on requirements. -Selection criteria: -1. Agent identity/profession is related to the interview topic. -2. Agent might hold unique or valuable opinions. -3. Select diverse perspectives (e.g., supporters, opponents, neutrals, professionals, etc.). -4. Prioritize characters directly related to the event. + system_prompt = """Bạn là một chuyên gia lập kế hoạch phỏng vấn chuyên nghiệp. Nhiệm vụ của bạn là dựa trên yêu cầu phỏng vấn để chọn ra những đối tượng phù hợp nhất từ danh sách các Agent mô phỏng. -Return in JSON format: +Tiêu chí lựa chọn: +1. Danh tính/Nghề nghiệp của Agent phải liên quan đến chủ đề phỏng vấn. +2. Agent có khả năng nắm giữ những quan điểm độc đáo hoặc có giá trị. +3. Lựa chọn các góc nhìn đa dạng (Ví dụ: bên ủng hộ, bên phản đối, bên trung lập, chuyên gia, v.v.). +4. Ưu tiên các nhân vật có liên quan trực tiếp đến sự kiện. + +Trả về định dạng JSON: { - "selected_indices": [array of selected agent indices], - "reasoning": "explanation for the selection" + "selected_indices": [Danh sách chỉ mục của các Agent được chọn], + "reasoning": "Giải thích lý do lựa chọn" }""" - user_prompt = f"""Interview requirements: + user_prompt = f"""Yêu cầu phỏng vấn: {interview_requirement} -Simulation background: +Bối cảnh mô phỏng: {simulation_requirement if simulation_requirement else "Not provided"} -Available Agents (total {len(agent_summaries)}): +Danh sách các Agent có thể chọn (Tổng cộng {len(agent_summaries)}): {json.dumps(agent_summaries, ensure_ascii=False, indent=2)} -Please select up to {max_agents} most suitable agents for the interview and explain your reasoning.""" +Hãy chọn tối đa {max_agents} Agent phù hợp nhất để phỏng vấn và giải thích lý do lựa chọn của bạn.""" + +# system_prompt = """You are a professional interview planning expert. Your task is to select the most suitable subjects for an interview from a list of simulated Agents based on the interview requirements. + +# Selection Criteria: +# 1. The Agent's identity/profession must be relevant to the interview topic. +# 2. The Agent is likely to hold a unique or valuable perspective. +# 3. Select diverse perspectives (e.g., supporters, opponents, neutral parties, professionals, etc.). +# 4. Prioritize characters directly related to the event. + +# Return in JSON format: +# { +# "selected_indices": [List of indices of selected Agents], +# "reasoning": "Explanation for the selection" +# }""" + +# user_prompt = f"""Interview Requirements: +# {interview_requirement} + +# Simulation Background: +# {simulation_requirement if simulation_requirement else "Not provided"} + +# List of selectable Agents (Total {len(agent_summaries)}): +# {json.dumps(agent_summaries, ensure_ascii=False, indent=2)} + +# Please select a maximum of {max_agents} most suitable Agents for the interview and explain the reasons for your selection.""" try: response = self.llm.chat_json( @@ -1649,27 +1704,47 @@ Please select up to {max_agents} most suitable agents for the interview and expl """Sử dụng LLM để sinh ra các câu chất vấn hợp với tính chất sự việc""" agent_roles = [a.get("profession", "Unknown") for a in selected_agents] + +# system_prompt = """You are a professional journalist/interviewer. Based on the interview requirements, generate 3-5 in-depth interview questions. + +# Question Requirements: +# 1. Open-ended questions that encourage detailed answers. +# 2. Questions that may elicit different answers from different roles. +# 3. Cover multiple dimensions such as facts, opinions, and feelings. +# 4. Use natural language, sounding like a real-life interview. +# 5. Keep each question under 50 words, concise and clear. +# 6. Ask the questions directly; do not include background explanations or prefixes. + +# Return in JSON format: {"questions": ["Question 1", "Question 2", ...]}""" + +# user_prompt = f"""Interview Requirements: {interview_requirement} + +# Simulation Background: {simulation_requirement if simulation_requirement else "Not provided"} + +# Interviewee Roles: {', '.join(agent_roles)} + +# Please generate 3-5 interview questions.""" - system_prompt = """You are a professional journalist/interviewer. Generate 3-5 deep interview questions based on requirements. + system_prompt = """Bạn là một nhà báo/người phỏng vấn chuyên nghiệp. Dựa trên yêu cầu phỏng vấn, hãy tạo từ 3-5 câu hỏi phỏng vấn chuyên sâu. -Question requirements: -1. Open-ended questions, encourage detailed answers. -2. Formulated so different roles might have different answers. -3. Cover multiple dimensions like facts, opinions, feelings, etc. -4. Natural language, sounds like a real interview. -5. Keep each question within 50 words, concise and clear. -6. Ask directly, do not include background explanations or prefixes. +Yêu cầu đối với câu hỏi: +1. Câu hỏi mở, khuyến khích câu trả lời chi tiết. +2. Các câu hỏi có thể nhận được những câu trả lời khác nhau tùy theo từng vai trò. +3. Bao quát nhiều khía cạnh như sự thật, quan điểm và cảm xúc. +4. Ngôn ngữ tự nhiên, giống như một buổi phỏng vấn thực tế. +5. Mỗi câu hỏi khống chế dưới 50 chữ, ngắn gọn và súc tích. +6. Đặt câu hỏi trực tiếp, không bao gồm giải thích bối cảnh hoặc tiền tố. -Return in JSON format: {"questions": ["question 1", "question 2", ...]}""" +Trả về định dạng JSON: {"questions": ["Câu hỏi 1", "Câu hỏi 2", ...]}""" - user_prompt = f"""Interview requirements: {interview_requirement} + user_prompt = f"""Yêu cầu phỏng vấn: {interview_requirement} -Simulation background: {simulation_requirement if simulation_requirement else "Not provided"} +Bối cảnh mô phỏng: {simulation_requirement if simulation_requirement else "Not provided"} -Interviewee roles: {', '.join(agent_roles)} - -Please generate 3-5 interview questions.""" +Vai trò của đối tượng phỏng vấ: {', '.join(agent_roles)} +Hãy tạo từ 3-5 câu hỏi phỏng vấn.""" + try: response = self.llm.chat_json( messages=[ @@ -1703,29 +1778,52 @@ Please generate 3-5 interview questions.""" interview_texts = [] for interview in interviews: interview_texts.append(f"【{interview.agent_name}({interview.agent_role})】\n{interview.response[:500]}") + +# system_prompt = """You are a professional news editor. Please generate an interview summary based on the responses from multiple interviewees. + +# Summary Requirements: +# 1. Synthesize the main viewpoints from all parties. +# 2. Identify areas of consensus and divergence among the viewpoints. +# 3. Highlight valuable quotes. +# 4. Maintain objectivity and neutrality, without favoring any side. +# 5. Limit the summary to 1000 words. + +# Formatting Constraints (MUST be followed): +# - Use plain text paragraphs, separated by blank lines. +# - Do not use Markdown headers (e.g., #, ##, ###). +# - Do not use horizontal rules (e.g., ---, ***). +# - Use Vietnamese quotation marks 「」 when quoting the interviewees' original words. +# - You may use **bold** to highlight keywords, but do not use any other Markdown syntax.""" + +# user_prompt = f"""Interview Topic: {interview_requirement} + +# Interview Content: +# {"\n\n".join(interview_texts)} + +# Please generate the interview summary.""" - system_prompt = """You are a professional news editor. Please generate an interview summary based on the answers from multiple interviewees. + system_prompt = """Bạn là một biên tập viên tin tức chuyên nghiệp. Hãy tạo một bản tóm tắt phỏng vấn dựa trên câu trả lời từ nhiều đối tượng được phỏng vấn. -Summary requirements: -1. Extract main viewpoints from all parties. -2. Point out consensus and disagreements among opinions. -3. Highlight valuable quotes. -4. Objective and neutral, do not favor any side. -5. Keep it within 1000 words. +Yêu cầu tóm tắt: +1. Đúc kết các quan điểm chính của các bên. +2. Chỉ ra những điểm đồng thuận và khác biệt trong các quan điểm. +3. Làm nổi bật các câu trích dẫn có giá trị. +4. Đảm bảo tính khách quan và trung lập, không thiên vị bất kỳ bên nào. +5. Giới hạn trong khoảng 1000 chữ. -Formatting constraints (Must obey): -- Use plain text paragraphs, separate different sections with blank lines. -- Do not use Markdown headers (like #, ##, ###). -- Do not use dividers (like ---, ***). -- Use normal quotes when citing interviewee actions/words. -- You can use **bold** to mark keywords, but no other Markdown syntax.""" +Ràng buộc định dạng (BẮT BUỘC tuân thủ): +- Sử dụng các đoạn văn bản thuần túy, phân tách các phần bằng dòng trống. +- Không sử dụng tiêu đề Markdown (như #, ##, ###). +- Không sử dụng đường kẻ phân cách (như ---, ***). +- Sử dụng dấu ngoặc kép kiểu Việt Nam 「」 khi trích dẫn nguyên văn lời của người được phỏng vấn. +- Có thể sử dụng dấu **in đậm** cho các từ khóa, nhưng không sử dụng bất kỳ cú pháp Markdown nào khác.""" - user_prompt = f"""Interview topic: {interview_requirement} + user_prompt = f"""Chủ đề phỏng vấn: {interview_requirement} -Interview content: +Nội dung phỏng vấn: {"\n\n".join(interview_texts)} -Please generate the interview summary.""" +Hãy tạo bản tóm tắt phỏng vấn.""" try: summary = self.llm.chat( diff --git a/backend/scripts/test_profile_format.py b/backend/scripts/test_profile_format.py index a2029b21..6e96f451 100644 --- a/backend/scripts/test_profile_format.py +++ b/backend/scripts/test_profile_format.py @@ -38,7 +38,7 @@ def test_profile_formats(): age=25, gender="male", mbti="INTJ", - country="China", + country="Vietnam", profession="Student", interested_topics=["Technology", "Education"], source_entity_uuid="test-uuid-123", From e0f2308e4daf8333885544dfc3eac507261fed21 Mon Sep 17 00:00:00 2001 From: ththu0205 Date: Wed, 6 May 2026 03:01:00 +0000 Subject: [PATCH 2/4] update --- backend/app/services/simulation_manager.py | 11 +++++++ backend/app/services/text_processor.py | 2 +- backend/scripts/llm_cost_patch.py | 36 ++++++++++++++++++++++ backend/scripts/run_parallel_simulation.py | 28 +++++++++++++++-- frontend/src/components/Step4Report.vue | 8 ++--- frontend/src/views/Home.vue | 2 +- package.json | 2 +- 7 files changed, 79 insertions(+), 10 deletions(-) diff --git a/backend/app/services/simulation_manager.py b/backend/app/services/simulation_manager.py index 8d6b78c3..5cd07bf4 100644 --- a/backend/app/services/simulation_manager.py +++ b/backend/app/services/simulation_manager.py @@ -334,6 +334,16 @@ class SimulationManager: elif state.enable_twitter: realtime_output_path = os.path.join(sim_dir, "twitter_profiles.csv") realtime_platform = "twitter" + + # Nền tảng dùng cho metadata cost. + if state.enable_twitter and state.enable_reddit: + runtime_platform = "parallel" + elif state.enable_twitter: + runtime_platform = "twitter" + elif state.enable_reddit: + runtime_platform = "reddit" + else: + runtime_platform = None profiles = generator.generate_profiles_from_entities( entities=filtered.entities, @@ -343,6 +353,7 @@ class SimulationManager: parallel_count=parallel_profile_count, # Số dòng luồng Async realtime_output_path=realtime_output_path, # Lưu log thời gian thực output_platform=realtime_platform, # Đuôi file xuất + metadata_platform=runtime_platform, simulation_id=simulation_id, project_id=state.project_id, ) diff --git a/backend/app/services/text_processor.py b/backend/app/services/text_processor.py index 5e9d0c28..625793b9 100644 --- a/backend/app/services/text_processor.py +++ b/backend/app/services/text_processor.py @@ -54,7 +54,7 @@ class TextProcessor: # Xoá các dòng trống liên tiếp (Chỉ giữ lại tối đa 2 lần xuống dòng liên tiếp) text = re.sub(r'\n{3,}', '\n\n', text) - # 移除行首行尾空白 + # Loại bỏ khoảng trắng đầu và cuối lines = [line.strip() for line in text.split('\n')] text = '\n'.join(lines) diff --git a/backend/scripts/llm_cost_patch.py b/backend/scripts/llm_cost_patch.py index b02cde31..fe6d16fa 100644 --- a/backend/scripts/llm_cost_patch.py +++ b/backend/scripts/llm_cost_patch.py @@ -20,6 +20,36 @@ def _build_metadata(model: str) -> Dict[str, Any]: return metadata +def _normalize_messages_for_qwen(messages: Any) -> Any: + """Ensure Qwen-compatible ordering: a single system message at index 0.""" + if not isinstance(messages, list): + return messages + + if not messages: + return messages + + system_contents = [] + non_system_messages = [] + + for item in messages: + if isinstance(item, dict) and str(item.get("role", "")).lower() == "system": + content = item.get("content") + if content is not None and str(content).strip(): + system_contents.append(str(content)) + continue + non_system_messages.append(item) + + # No system messages detected, keep original payload. + if not system_contents: + return messages + + merged_system = { + "role": "system", + "content": "\n\n".join(system_contents), + } + return [merged_system] + non_system_messages + + def install_openai_cost_patch( *, simulation_id: Optional[str], @@ -58,6 +88,9 @@ def install_openai_cost_patch( def sync_create_wrapper(self, *args, **kwargs): model = kwargs.get("model", "unknown_model") + if "qwen" in str(model).lower() and "messages" in kwargs: + kwargs = dict(kwargs) + kwargs["messages"] = _normalize_messages_for_qwen(kwargs.get("messages")) response = original_sync_create(self, *args, **kwargs) try: record_llm_cost( @@ -71,6 +104,9 @@ def install_openai_cost_patch( async def async_create_wrapper(self, *args, **kwargs): model = kwargs.get("model", "unknown_model") + if "qwen" in str(model).lower() and "messages" in kwargs: + kwargs = dict(kwargs) + kwargs["messages"] = _normalize_messages_for_qwen(kwargs.get("messages")) response = await original_async_create(self, *args, **kwargs) try: record_llm_cost( diff --git a/backend/scripts/run_parallel_simulation.py b/backend/scripts/run_parallel_simulation.py index 689fc103..fb9eda66 100644 --- a/backend/scripts/run_parallel_simulation.py +++ b/backend/scripts/run_parallel_simulation.py @@ -361,7 +361,7 @@ class ParallelIPCHandler: """ # Nếu có chỉ định nền tảng, chỉ phỏng vấn trên nền tảng đó if platform in ("twitter", "reddit"): - result = await self._interview_single_platform(agent_id, prompt, platform) + result = await self._interview_single_platform(agent_id, prompt, platform) if "error" in result: self.send_response(command_id, "failed", error=result["error"]) @@ -996,11 +996,27 @@ def create_model(config: Dict[str, Any], use_boost: bool = False): config: Dictionary cấu hình mô phỏng use_boost: Có dùng cấu hình LLM tăng tốc hay không (nếu khả dụng) """ - # Kiểm tra có cấu hình tăng tốc không + def _is_placeholder(value: str) -> bool: + v = (value or "").strip().lower() + return v in {"your_api_key_here", "your_base_url_here", "your_model_name_here"} + + def _is_http_url(value: str) -> bool: + v = (value or "").strip().lower() + return v.startswith("http://") or v.startswith("https://") + + # Kiểm tra có cấu hình tăng tốc hợp lệ không boost_api_key = os.environ.get("LLM_BOOST_API_KEY", "") boost_base_url = os.environ.get("LLM_BOOST_BASE_URL", "") boost_model = os.environ.get("LLM_BOOST_MODEL_NAME", "") - has_boost_config = bool(boost_api_key) + has_boost_config = ( + bool(boost_api_key.strip()) + and bool(boost_model.strip()) + and bool(boost_base_url.strip()) + and not _is_placeholder(boost_api_key) + and not _is_placeholder(boost_base_url) + and not _is_placeholder(boost_model) + and _is_http_url(boost_base_url) + ) # Chọn LLM theo tham số và trạng thái cấu hình if use_boost and has_boost_config: @@ -1011,6 +1027,8 @@ def create_model(config: Dict[str, Any], use_boost: bool = False): config_label = "[Boost LLM]" else: # Dùng cấu hình chung + if use_boost and not has_boost_config: + print("[Boost LLM] Invalid or placeholder boost config, fallback to [General LLM].") llm_api_key = os.environ.get("LLM_API_KEY", "") llm_base_url = os.environ.get("LLM_BASE_URL", "") llm_model = os.environ.get("LLM_MODEL_NAME", "") @@ -1028,6 +1046,10 @@ def create_model(config: Dict[str, Any], use_boost: bool = False): raise ValueError("Missing API key config. Please set LLM_API_KEY in the project root .env file") if llm_base_url: + if not _is_http_url(llm_base_url): + raise ValueError( + f"Invalid LLM base URL: {llm_base_url}. It must start with http:// or https://" + ) os.environ["OPENAI_API_BASE_URL"] = llm_base_url print(f"{config_label} model={llm_model}, base_url={llm_base_url[:40] if llm_base_url else 'default'}...") diff --git a/frontend/src/components/Step4Report.vue b/frontend/src/components/Step4Report.vue index ad7b5e33..db704ce3 100644 --- a/frontend/src/components/Step4Report.vue +++ b/frontend/src/components/Step4Report.vue @@ -862,7 +862,7 @@ const parseInterview = (text) => { const quotesText = quotesMatch[1] // Prioritize matching > "text" format let quoteMatches = quotesText.match(/> "([^"]+)"/g) - // Backtrack: Matches > "text" or > \u201Ctext\u201D (Chinese quotation marks) + // Backtrack: Matches > "text" or > \u201Ctext\u201D (Vietnamese quotation marks) if (!quoteMatches) { quoteMatches = quotesText.match(/> [\u201C""]([^\u201D""]+)[\u201D""]/g) } @@ -2089,9 +2089,9 @@ const extractFinalContent = (response) => { } // Try to find the content after "Final Answer:" - const chineseFinalMatch = response.match(/Final Answer[::]\s*\n*([\s\S]*)$/i) - if (chineseFinalMatch) { - return chineseFinalMatch[1].trim() + const vietnameseFinalMatch = response.match(/Final Answer[::]\s*\n*([\s\S]*)$/i) + if (vietnameseFinalMatch) { + return vietnameseFinalMatch[1].trim() } // Nếu bắt đầu bằng ## hoặc # hoặc >, có thể là nội dung markdown trực tiếp diff --git a/frontend/src/views/Home.vue b/frontend/src/views/Home.vue index 5a8728d8..73545ae5 100644 --- a/frontend/src/views/Home.vue +++ b/frontend/src/views/Home.vue @@ -176,7 +176,7 @@ diff --git a/package.json b/package.json index 63ace21a..a23ca5b9 100644 --- a/package.json +++ b/package.json @@ -1,7 +1,7 @@ { "name": "mirofish", "version": "0.1.0", - "description": "MiroFish - 简洁通用的群体智能引擎,预测万物", + "description": "MiroFish - A simple and versatile swarm intelligence engine that predicts everything", "scripts": { "setup": "npm install && cd frontend && npm install", "setup:backend": "cd backend && uv sync", From ec26745e0898bedeff6326570af90a75c3d55c40 Mon Sep 17 00:00:00 2001 From: ththu0205 Date: Mon, 11 May 2026 08:07:27 +0000 Subject: [PATCH 3/4] edit prompts v2 --- .gitignore | 3 +- SIMULATION_PIPELINE_REPORT.md | 1316 +++++++++++++++++ backend/app/services/graph_builder.py | 65 +- .../app/services/oasis_profile_generator.py | 1128 ++++++++------ backend/app/services/report_agent.py | 2 +- .../services/simulation_config_generator.py | 2 +- backend/app/services/simulation_manager.py | 602 +++++--- backend/app/services/zep_entity_reader.py | 497 +++++-- backend/app/services/zep_tools.py | 4 +- backend/app/utils/llm_client.py | 4 +- backend/scripts/run_parallel_simulation.py | 5 +- backend/scripts/test_entity_reader.py | 137 ++ install.cmd | 221 +++ oasis | 1 + report.md | 186 +++ test_code_backend/full_pipeline/README.md | 262 ++++ .../full_pipeline/config_articles.env | 20 + .../full_pipeline/prepare_input.py | 152 ++ test_code_backend/full_pipeline/run.sh | 217 +++ .../gen_ontogogy_and_graph/build_graph.py | 197 +++ .../gen_ontogogy_and_graph/gen_ontology.py | 109 ++ 21 files changed, 4323 insertions(+), 807 deletions(-) create mode 100644 SIMULATION_PIPELINE_REPORT.md create mode 100644 backend/scripts/test_entity_reader.py create mode 100644 install.cmd create mode 160000 oasis create mode 100644 report.md create mode 100644 test_code_backend/full_pipeline/README.md create mode 100644 test_code_backend/full_pipeline/config_articles.env create mode 100644 test_code_backend/full_pipeline/prepare_input.py create mode 100644 test_code_backend/full_pipeline/run.sh create mode 100644 test_code_backend/gen_ontogogy_and_graph/build_graph.py create mode 100644 test_code_backend/gen_ontogogy_and_graph/gen_ontology.py diff --git a/.gitignore b/.gitignore index 01a7716e..4a590314 100644 --- a/.gitignore +++ b/.gitignore @@ -57,4 +57,5 @@ backend/logs/ backend/uploads/ # Dữ liệu Docker -data/ \ No newline at end of file +data/ +uploads_old/ \ No newline at end of file diff --git a/SIMULATION_PIPELINE_REPORT.md b/SIMULATION_PIPELINE_REPORT.md new file mode 100644 index 00000000..10134635 --- /dev/null +++ b/SIMULATION_PIPELINE_REPORT.md @@ -0,0 +1,1316 @@ +# Báo cáo Kỹ thuật: Pipeline Mô phỏng MiroFish + +**Phiên bản:** 1.0 +**Ngày:** 2026-05-04 +**Dự án:** MiroFish — Hệ thống mô phỏng dư luận mạng xã hội dựa trên OASIS Framework + +--- + +## Mục lục + +1. [Tổng quan hệ thống](#1-tổng-quan-hệ-thống) +2. [Kiến trúc tổng thể](#2-kiến-trúc-tổng-thể) +3. [Giai đoạn 0 — Tiền xử lý tài liệu](#3-giai-đoạn-0--tiền-xử-lý-tài-liệu) +4. [Giai đoạn 1 — Khởi tạo Simulation](#4-giai-đoạn-1--khởi-tạo-simulation) +5. [Giai đoạn 2A — Đọc Entity từ Zep Graph](#5-giai-đoạn-2a--đọc-entity-từ-zep-graph) +6. [Giai đoạn 2B — Sinh Agent Profile](#6-giai-đoạn-2b--sinh-agent-profile) +7. [Giai đoạn 2C — Sinh Simulation Config](#7-giai-đoạn-2c--sinh-simulation-config) +8. [Giai đoạn 3 — Chạy Simulation (OASIS)](#8-giai-đoạn-3--chạy-simulation-oasis) +9. [Giai đoạn 4 — Tạo Report (ReACT Agent)](#9-giai-đoạn-4--tạo-report-react-agent) +10. [Cơ chế Fallback toàn hệ thống](#10-cơ-chế-fallback-toàn-hệ-thống) +11. [Cấu trúc file và thư mục](#11-cấu-trúc-file-và-thư-mục) +12. [Luồng dữ liệu end-to-end](#12-luồng-dữ-liệu-end-to-end) + +--- + +## 1. Tổng quan hệ thống + +### 1.1 MiroFish là gì? + +MiroFish là một hệ thống mô phỏng dư luận trên mạng xã hội. Ý tưởng cốt lõi là: thay vì quan sát xem điều gì đang xảy ra ngoài đời thực, hệ thống **xây dựng lại một thế giới ảo** có các nhân vật (Agent) giống hệt người thật, rồi "thả" một sự kiện vào trong đó để xem mọi người phản ứng như thế nào. + +**Ví dụ thực tế:** +> Một trường đại học muốn biết: nếu họ tăng học phí 20%, sinh viên, giảng viên, phụ huynh và báo chí sẽ phản ứng ra sao trên mạng xã hội? Thay vì phải thực sự tăng học phí rồi chờ xem hậu quả, MiroFish mô phỏng kịch bản đó trước và dự báo kết quả. + +### 1.2 Các thành phần chính + +| Thành phần | Vai trò | +|---|---| +| **Zep** | Cơ sở dữ liệu đồ thị (Graph Database) chứa thông tin về các thực thể (người, tổ chức, sự kiện) được trích xuất từ tài liệu gốc | +| **OASIS Framework** | Framework mã nguồn mở chạy mô phỏng mạng xã hội với các Agent AI | +| **LLM (Large Language Model)** | Mô hình ngôn ngữ lớn (như GPT-4, Gemini...) được dùng để tạo hồ sơ nhân vật, cấu hình mô phỏng và viết báo cáo | +| **Flask Backend** | Server Python xử lý API, quản lý trạng thái và điều phối toàn bộ pipeline | +| **Vue.js Frontend** | Giao diện người dùng để upload tài liệu, theo dõi tiến trình và xem báo cáo | + +### 1.3 Luồng tổng quát (5 giai đoạn) + +``` +[Tài liệu gốc] + ↓ (Tiền xử lý) +[Zep Graph: Entities + Relationships] + ↓ (Đọc + Lọc) +[Danh sách EntityNode] + ↓ (Sinh Profile song song) +[Agent Profiles (reddit_profiles.json / twitter_profiles.csv)] + ↓ (LLM sinh Config) +[simulation_config.json] + ↓ (OASIS subprocess) +[actions.jsonl — Log hành động của từng Agent theo từng Round] + ↓ (ReACT Agent viết báo cáo) +[report.md — Báo cáo dự báo tương lai] +``` + +--- + +## 2. Kiến trúc tổng thể + +### 2.1 Sơ đồ các module + +``` +backend/ +├── app/ +│ ├── services/ +│ │ ├── text_processor.py # Tiền xử lý văn bản +│ │ ├── zep_entity_reader.py # Đọc Entity từ Zep Graph +│ │ ├── oasis_profile_generator.py # Sinh Agent Profile +│ │ ├── simulation_config_generator.py # Sinh tham số mô phỏng +│ │ ├── simulation_manager.py # Quản lý vòng đời Simulation +│ │ ├── simulation_runner.py # Chạy + giám sát Simulation +│ │ ├── simulation_ipc.py # Giao tiếp Flask ↔ OASIS subprocess +│ │ └── report_agent.py # Viết báo cáo bằng ReACT +│ └── utils/ +│ └── llm_cost.py # Theo dõi chi phí LLM +└── scripts/ + ├── run_parallel_simulation.py # OASIS chạy cả hai nền tảng + ├── run_twitter_simulation.py # OASIS chỉ Twitter + ├── run_reddit_simulation.py # OASIS chỉ Reddit + └── action_logger.py # Ghi log hành động Agent +``` + +### 2.2 Phân tách trách nhiệm + +**Flask Backend** (process chính) chịu trách nhiệm: +- Nhận request từ Frontend +- Gọi các service theo đúng thứ tự +- Theo dõi trạng thái và lưu vào file JSON +- Expose API để Frontend poll tiến độ + +**OASIS Scripts** (subprocess độc lập) chịu trách nhiệm: +- Khởi tạo môi trường mạng xã hội ảo +- Chạy các vòng (round) mô phỏng +- Cho từng Agent AI "suy nghĩ" và hành động +- Ghi log hành động ra file `actions.jsonl` + +Hai phần này giao tiếp với nhau qua **hai cơ chế**: +1. **File-based**: Flask đọc `actions.jsonl` để monitor tiến độ +2. **IPC (Inter-Process Communication)**: Flask ghi lệnh vào `ipc_commands/`, OASIS poll và trả kết quả vào `ipc_responses/` (dùng cho tính năng Interview agent đang chạy) + +--- + +## 3. Giai đoạn 0 — Tiền xử lý tài liệu + +**File:** `backend/app/services/text_processor.py` + +### 3.1 Mục đích + +Trước khi bắt đầu bất kỳ thứ gì, tài liệu gốc của người dùng (PDF, Word, TXT...) phải được chuyển thành văn bản thuần túy và làm sạch. + +### 3.2 Luồng xử lý + +``` +[File upload từ User] + ↓ +TextProcessor.extract_from_files(file_paths) + ↓ (gọi FileParser.extract_from_multiple) +[raw_text: str] ← văn bản thô chưa sạch + ↓ +TextProcessor.preprocess_text(raw_text) + ↓ (chuẩn hóa newline, xóa khoảng trắng thừa) +[document_text: str] ← văn bản sạch, sẵn sàng dùng +``` + +### 3.3 Chi tiết xử lý làm sạch + +`preprocess_text()` thực hiện 3 bước: +1. Chuẩn hóa ký tự xuống dòng: `\r\n` (Windows) và `\r` (Mac cũ) → `\n` (Linux chuẩn) +2. Loại bỏ dòng trống liên tiếp: nếu có 3 dòng trống liên tiếp → rút về tối đa 2 +3. Xóa khoảng trắng đầu/cuối mỗi dòng + +### 3.4 Output + +`document_text: str` — Chuỗi văn bản sạch, được dùng ở hai nơi sau đó: +- Truyền vào **Zep Graph Builder** để xây đồ thị (qua pipeline riêng, trước khi vào pipeline Simulation) +- Truyền thẳng vào **SimulationConfigGenerator** làm ngữ cảnh cho LLM sinh tham số + +> **Lưu ý quan trọng:** Pipeline Simulation giả sử Zep Graph đã được build từ trước. Nghĩa là người dùng phải chạy bước "Phân tích tài liệu / Build Graph" trước, rồi mới chạy Simulation. + +--- + +## 4. Giai đoạn 1 — Khởi tạo Simulation + +**File:** `backend/app/services/simulation_manager.py` +**Hàm:** `SimulationManager.create_simulation()` + +### 4.1 Mục đích + +Tạo một "phòng mô phỏng" rỗng với ID định danh riêng, chưa có dữ liệu gì. Giống như đặt một bàn làm việc trống trước khi bắt đầu. + +### 4.2 Input + +```python +project_id: str # ID dự án (gắn với tài liệu gốc) +graph_id: str # ID của Zep Graph đã được build từ tài liệu +enable_twitter: bool # Có chạy mô phỏng Twitter không? +enable_reddit: bool # Có chạy mô phỏng Reddit không? +``` + +### 4.3 Luồng xử lý + +```python +# 1. Tạo ID ngẫu nhiên +simulation_id = f"sim_{uuid4().hex[:12]}" +# Ví dụ: "sim_a3f9c2b81e4d" + +# 2. Khởi tạo object trạng thái +state = SimulationState( + simulation_id=simulation_id, + status=SimulationStatus.CREATED, + ... +) + +# 3. Lưu ra file +uploads/simulations/{simulation_id}/state.json + +# 4. Cache vào RAM +self._simulations[simulation_id] = state +``` + +### 4.4 Output + +`SimulationState` object với `status = CREATED` + +**Nội dung `state.json` lúc này:** +```json +{ + "simulation_id": "sim_a3f9c2b81e4d", + "project_id": "proj_xyz", + "graph_id": "graph_abc", + "enable_twitter": true, + "enable_reddit": true, + "status": "created", + "entities_count": 0, + "profiles_count": 0, + "config_generated": false, + "created_at": "2026-05-04T10:00:00" +} +``` + +### 4.5 Cơ chế load lại trạng thái + +`SimulationManager` có hai tầng cache: +1. **RAM cache** (`_simulations` dict): Truy cập tức thì, mất khi restart server +2. **File cache** (`state.json`): Persistent, load lại khi RAM không có + +Khi gọi `_load_simulation_state(simulation_id)`: +- Tìm trong RAM trước → nếu có, trả về luôn +- Nếu không có → đọc từ `state.json` → parse → lưu vào RAM → trả về + +--- + +## 5. Giai đoạn 2A — Đọc Entity từ Zep Graph + +**File:** `backend/app/services/zep_entity_reader.py` +**Hàm:** `ZepEntityReader.filter_defined_entities()` + +### 5.1 Zep Graph là gì? + +Zep là một **Graph Database** (cơ sở dữ liệu dạng đồ thị). Sau khi người dùng upload tài liệu, hệ thống đã phân tích và lưu các thông tin vào Zep dưới dạng: +- **Node** (Nút): Đại diện cho một thực thể (người, tổ chức, địa điểm, sự kiện...) +- **Edge** (Cạnh): Đại diện cho mối quan hệ giữa hai thực thể ("A là giám đốc của B", "C phát biểu về D"...) + +**Ví dụ về một Graph từ tài liệu về trường đại học:** +``` +[Trần Văn A] --[là sinh viên của]--> [Đại học X] +[Nguyễn B] --[là giảng viên của]--> [Đại học X] +[Đại học X] --[tăng học phí]--> [Năm học 2025] +[VnExpress] --[đưa tin về]--> [Đại học X] +``` + +### 5.2 Cấu trúc dữ liệu EntityNode + +```python +@dataclass +class EntityNode: + uuid: str # ID định danh duy nhất trong Zep + name: str # Tên ("Trần Văn A", "Đại học X") + labels: List[str] # Loại thực thể ["Entity", "Student"] + summary: str # Tóm tắt do Zep tự sinh ra + attributes: Dict # Thuộc tính bổ sung {"age": 20, "major": "CNTT"} + related_edges: List # Các quan hệ liên quan + related_nodes: List # Các node khác có kết nối +``` + +**Ví dụ thực tế một EntityNode:** +```json +{ + "uuid": "e1a2b3c4-...", + "name": "Trần Văn An", + "labels": ["Entity", "Student"], + "summary": "Sinh viên năm 3 ngành CNTT tại Đại học X, tích cực tham gia các hoạt động sinh viên", + "attributes": {"year": 3, "major": "Computer Science"}, + "related_edges": [ + {"direction": "outgoing", "edge_name": "studies_at", "fact": "Trần Văn An học tại Đại học X"} + ], + "related_nodes": [ + {"name": "Đại học X", "labels": ["Entity", "University"], "summary": "..."} + ] +} +``` + +### 5.3 Luồng xử lý chi tiết + +``` +filter_defined_entities(graph_id, defined_entity_types, enrich_with_edges=True) + | + ├── get_all_nodes(graph_id) + │ └── fetch_all_nodes() — có phân trang, retry 3 lần, backoff 2s→4s→8s + │ → List[Dict] — toàn bộ node trong Graph + │ + ├── get_all_edges(graph_id) + │ └── fetch_all_edges() — có phân trang, retry 3 lần + │ → List[Dict] — toàn bộ edge trong Graph + │ + ├── Xây dựng node_map: {uuid → node_data} + │ + └── Lặp qua từng node: + ├── Lọc: giữ lại node có custom label (khác "Entity" và "Node") + │ Ví dụ: ["Entity", "Student"] → giữ, vì có "Student" + │ Ví dụ: ["Entity"] → bỏ, vì chỉ có "Entity" + │ + ├── Nếu defined_entity_types được chỉ định: + │ chỉ giữ node có label khớp với danh sách + │ + ├── Tìm related_edges: duyệt all_edges, lấy các edge + │ có source_node_uuid HOẶC target_node_uuid = node.uuid + │ → gán direction: "outgoing" hoặc "incoming" + │ + └── Tìm related_nodes: từ related_node_uuids → tra node_map + → lấy thông tin cơ bản (uuid, name, labels, summary) +``` + +### 5.4 Output + +```python +FilteredEntities( + entities=[EntityNode, EntityNode, ...], # Danh sách entity đã lọc + entity_types={"Student", "University", "MediaOutlet"}, # Các loại xuất hiện + total_count=150, # Tổng số node trong graph + filtered_count=47, # Số entity hợp lệ sau lọc +) +``` + +### 5.5 Cơ chế Retry + +Mọi lời gọi Zep API đều được bọc trong `_call_with_retry()`: +- Thử tối đa **3 lần** +- Độ trễ giữa các lần: **2s → 4s → 8s** (Exponential backoff) +- Nếu vẫn fail sau 3 lần → raise exception lên caller + +### 5.6 Fallback nghiêm trọng + +Nếu `filtered_count == 0` sau khi lọc: +```python +state.status = SimulationStatus.FAILED +state.error = "No valid entities extracted for simulation. Please check if the Graph was generated properly with valid text." +``` +Pipeline **dừng hoàn toàn**, không tiếp tục. + +--- + +## 6. Giai đoạn 2B — Sinh Agent Profile + +**File:** `backend/app/services/oasis_profile_generator.py` +**Hàm:** `OasisProfileGenerator.generate_profiles_from_entities()` + +### 6.1 Mục đích + +Mỗi EntityNode (thực thể trong Zep) cần được "hóa thân" thành một Agent AI với đủ nhân cách, lịch sử, cách nói chuyện... để khi chạy mô phỏng, Agent đó behave đúng như người thật. Đây là bước **chuyển đổi từ dữ liệu khô sang nhân vật sống động**. + +**Ví dụ:** +- EntityNode: `{name: "Trần Văn An", labels: ["Student"], summary: "Sinh viên năm 3 CNTT"}` +- OasisAgentProfile: `{bio: "Sinh viên CNTT năm 3 tại Đại học X, mê game và lo lắng về học phí tăng", persona: "Trần Văn An là sinh viên 21 tuổi, MBTI INFP, hay đăng status buổi đêm, viết theo kiểu Gen Z, thường phản ứng nhanh với tin tức về học phí..."}` + +### 6.2 Cấu trúc OasisAgentProfile + +```python +@dataclass +class OasisAgentProfile: + user_id: int # Index (0, 1, 2, ...) + user_name: str # "tran_van_an_492" (tên tài khoản, random suffix) + name: str # "Trần Văn An" + bio: str # Tiểu sử ngắn hiển thị công khai (~200 ký tự) + persona: str # Mô tả nhân cách chi tiết (~2000 từ) — LLM dùng làm System Prompt + karma: int # (Reddit) Điểm karma + friend_count: int # (Twitter) Số bạn bè + follower_count: int # (Twitter) Số người theo dõi + age: int + gender: str # "male" / "female" / "other" + mbti: str # "INFP", "ESTJ", ... + country: str # "Việt Nam" + profession: str # "Sinh viên" + interested_topics: List[str] +``` + +**Điểm quan trọng về `persona`:** Đây là chuỗi văn bản ~2000 từ mà OASIS sẽ nhét vào **System Prompt** của Agent khi nó "suy nghĩ" và đưa ra hành động. Nó quyết định toàn bộ cách Agent đó hoạt động trong mô phỏng. + +### 6.3 Luồng song song (ThreadPoolExecutor) + +```python +with ThreadPoolExecutor(max_workers=parallel_count) as executor: + # Giao việc cho tất cả entity cùng lúc + futures = {executor.submit(generate_single_profile, idx, entity): (idx, entity) + for idx, entity in enumerate(entities)} + + # Thu gom kết quả khi từng profile hoàn thành + for future in as_completed(futures): + result_idx, profile, error = future.result() + profiles[result_idx] = profile + save_profiles_realtime() # Lưu ngay, không chờ tất cả xong +``` + +Mặc định `parallel_count = 3` — tức là tại bất kỳ thời điểm nào, tối đa 3 entity được xử lý đồng thời. + +### 6.4 Luồng xử lý cho từng entity + +``` +generate_single_profile(idx, entity) + | + ├── _build_entity_context(entity) + │ ├── 1. Lấy attributes của node + │ ├── 2. Lấy facts từ related_edges (đã có từ bước trước) + │ ├── 3. Lấy summary của related_nodes + │ └── 4. _search_zep_for_entity() — Gọi Zep Vector Search + │ ├── search_edges(query) — thread 1 + │ └── search_nodes(query) — thread 2 + │ (2 thread chạy song song, retry 3 lần mỗi cái) + │ → ghép thành context string + │ + ├── Phân loại entity type: + │ ├── INDIVIDUAL: student, alumni, professor, person, publicfigure... + │ └── GROUP: university, governmentagency, ngo, mediaoutlet... + │ + └── _generate_profile_with_llm(entity_name, type, summary, attrs, context) + ├── Chọn prompt template: + │ ├── Individual → _build_individual_persona_prompt() + │ └── Group → _build_group_persona_prompt() + │ + └── Gọi LLM (retry vòng lặp): + Attempt 1: temperature=0.7 + Attempt 2: temperature=0.6 (nếu fail) + Attempt 3: temperature=0.5 (nếu fail) + → JSON với bio, persona, age, gender, mbti, country... +``` + +### 6.5 Chi tiết Zep Vector Search + +`_search_zep_for_entity()` dùng **semantic search** — tìm kiếm theo nghĩa, không phải từ khóa: + +```python +comprehensive_query = f"Provide all facts, activities, relationships, and context about: {entity_name}" + +# Chạy song song 2 loại search +search_edges() → tìm trong các mối quan hệ (edge) — lấy tối đa 30 kết quả +search_nodes() → tìm trong các node thực thể — lấy tối đa 20 kết quả +``` + +Mỗi search có retry riêng: 3 lần, backoff 2s→4s→8s. +Nếu timeout tổng (30s) hoặc fail hoàn toàn → trả về `{"facts": [], "node_summaries": [], "context": ""}` và tiếp tục không lỗi. + +### 6.6 Xử lý JSON từ LLM + +LLM đôi khi trả về JSON bị hỏng (cắt ngang do token limit, ký tự lạ...). Hệ thống có chuỗi sửa lỗi: + +``` +LLM response + | + ├── finish_reason == "length"? → _fix_truncated_json() + │ Đếm { và } chưa đóng → tự thêm vào cuối + │ + ├── json.loads() lỗi? → _try_fix_json() + │ ├── Gọi _fix_truncated_json() trước + │ ├── Dùng regex tìm khối {...} lớn nhất + │ ├── Clean newline trong string values + │ ├── Xóa control characters (0x00–0x1F) + │ └── Nếu vẫn fail → regex mót bio và persona riêng lẻ + │ + └── Sau 3 attempt đều fail → _generate_profile_rule_based() +``` + +### 6.7 Fallback theo Rule (khi LLM hoàn toàn thất bại) + +| Entity Type | age | gender | activity | influence | +|---|---|---|---|---| +| student / alumni | 18–30 | random | Cao, chủ yếu buổi tối | Thấp | +| publicfigure / expert / faculty | 35–60 | random | Trung bình | Cao | +| mediaoutlet | 30 (ảo) | "other" | Cả ngày | Rất cao | +| university / governmentagency / ngo | 30 (ảo) | "other" | Giờ hành chính | Rất cao | +| Default | 25–50 | random | Trung bình | Trung bình | + +### 6.8 Lưu file kết quả + +**Reddit format** (`reddit_profiles.json`): +```json +[ + { + "user_id": 0, + "username": "tran_van_an_492", + "name": "Trần Văn An", + "bio": "Sinh viên CNTT năm 3, lo lắng về học phí...", + "persona": "Trần Văn An là sinh viên 21 tuổi...(2000 từ mô tả nhân cách)...", + "karma": 1500, + "age": 21, + "gender": "male", + "mbti": "INFP", + "country": "Việt Nam" + } +] +``` + +**Twitter format** (`twitter_profiles.csv`): +```csv +user_id,name,username,user_char,description +0,Trần Văn An,tran_van_an_492,"bio + persona ghép lại (System Prompt cho LLM)","bio ngắn" +``` + +**Realtime save:** Sau mỗi profile hoàn thành (không cần chờ tất cả), file được ghi ngay. Điều này đảm bảo nếu server crash giữa chừng, dữ liệu đã gen không bị mất. + +--- + +## 7. Giai đoạn 2C — Sinh Simulation Config + +**File:** `backend/app/services/simulation_config_generator.py` +**Hàm:** `SimulationConfigGenerator.generate_config()` + +### 7.1 Mục đích + +Đây là giai đoạn LLM "thiết kế kịch bản mô phỏng". Dựa trên yêu cầu của người dùng + dữ liệu entity, LLM quyết định: +- Mô phỏng kéo dài bao nhiêu giờ? +- Giờ nào sẽ có nhiều hoạt động? +- Sự kiện ban đầu nào sẽ khởi động cuộc thảo luận? +- Mỗi Agent sẽ hoạt động theo kiểu gì? + +### 7.2 Luồng tổng quát (tuần tự, không song song) + +``` +generate_config(simulation_id, project_id, graph_id, simulation_requirement, + document_text, entities, enable_twitter, enable_reddit) + │ + ├── _build_context() → context string (tối đa 50,000 ký tự) + │ ├── simulation_requirement (giữ nguyên) + │ ├── entity summary (tóm tắt theo loại, tối đa 20 entity/loại) + │ └── document_text (phần còn lại, bị cắt nếu quá dài) + │ + ├── Step 1: _generate_time_config() → TimeSimulationConfig + ├── Step 2: _generate_event_config() → EventConfig (có initial posts) + ├── Step 3..N: _generate_agent_configs_batch() (15 agent/lần) + ├── _assign_initial_post_agents() → gán agent_id cho từng post + └── Hardcode: Twitter + Reddit PlatformConfig +``` + +### 7.3 Step 1 — Cấu hình Thời gian + +**Hàm:** `_generate_time_config(context, num_entities)` + +LLM nhận context đã được cắt còn 10,000 ký tự và sinh ra: + +```json +{ + "total_simulation_hours": 72, + "minutes_per_round": 60, + "agents_per_hour_min": 5, + "agents_per_hour_max": 30, + "peak_hours": [19, 20, 21, 22], + "off_peak_hours": [0, 1, 2, 3, 4, 5], + "morning_hours": [6, 7, 8], + "work_hours": [9, 10, 11, 12, 13, 14, 15, 16, 17, 18], + "reasoning": "Chủ đề học phí thu hút cả sinh viên và phụ huynh, cao điểm tối..." +} +``` + +**Validation sau khi nhận:** +```python +# Đảm bảo agents_per_hour không vượt tổng số agent thực tế +if agents_per_hour_min > num_entities: + agents_per_hour_min = max(1, num_entities // 10) +if agents_per_hour_max > num_entities: + agents_per_hour_max = max(agents_per_hour_min + 1, num_entities // 2) +if agents_per_hour_min >= agents_per_hour_max: + agents_per_hour_min = max(1, agents_per_hour_max // 2) +``` + +**Fallback (LLM fail):** +```python +{ + "total_simulation_hours": 72, + "minutes_per_round": 60, + "agents_per_hour_min": max(1, num_entities // 15), + "agents_per_hour_max": max(5, num_entities // 5), + "peak_hours": [19, 20, 21, 22], + "off_peak_hours": [0, 1, 2, 3, 4, 5], + ... +} +``` + +**Ý nghĩa các thông số:** +- `total_simulation_hours = 72` → Mô phỏng 3 ngày thực (72 vòng nếu mỗi vòng = 1 giờ) +- `minutes_per_round = 60` → Mỗi vòng đại diện cho 1 giờ trong thế giới thực +- `agents_per_hour_min/max = 5/30` → Mỗi giờ, OASIS sẽ kích hoạt ngẫu nhiên 5 đến 30 agent để hoạt động + +### 7.4 Step 2 — Cấu hình Sự kiện + +**Hàm:** `_generate_event_config(context, simulation_requirement, entities)` + +LLM sinh ra "ngòi nổ" cho cuộc mô phỏng — các bài đăng đầu tiên sẽ khởi động thảo luận: + +```json +{ + "hot_topics": ["học phí", "biểu tình", "phản đối", "giáo dục công lập"], + "narrative_direction": "Thông báo tăng học phí gây ra làn sóng phản ứng mạnh từ sinh viên, dần leo thang thành phong trào...", + "initial_posts": [ + { + "content": "CHÍNH THỨC: Đại học X thông báo tăng học phí 20% từ học kỳ tới. Xem chi tiết tại...", + "poster_type": "MediaOutlet" + }, + { + "content": "Không thể chấp nhận được! Học phí đã cao mà còn tăng thêm?! #HọcPhíTăng", + "poster_type": "Student" + } + ], + "reasoning": "..." +} +``` + +**Sau đó:** `_assign_initial_post_agents()` ghép mỗi post với agent phù hợp: + +``` +poster_type = "MediaOutlet" + │ + ├── 1. Direct match: agents_by_type["mediaoutlet"] → lấy agent đầu tiên + │ + ├── 2. Alias match (nếu direct fail): + │ type_aliases = { + │ "mediaoutlet": ["mediaoutlet", "media"], + │ "official": ["official", "university", "governmentagency", "government"], + │ "student": ["student", "person"], + │ ... + │ } + │ + └── 3. Fallback: agent có influence_weight cao nhất +``` + +Kết quả: `poster_agent_id` được gán vào mỗi post: +```json +{"content": "CHÍNH THỨC...", "poster_type": "MediaOutlet", "poster_agent_id": 12} +``` + +### 7.5 Step 3..N — Cấu hình Agent (chia batch) + +**Hàm:** `_generate_agent_configs_batch(context, entities_batch, start_idx, simulation_requirement)` + +Do số lượng agent có thể lớn, không thể đưa tất cả vào một prompt. Hệ thống chia nhỏ **15 agent một lần**: + +``` +47 agents → batch 1 (agent 0-14) + batch 2 (agent 15-29) + batch 3 (agent 30-44) + batch 4 (agent 45-46) +``` + +Với mỗi batch, gửi cho LLM danh sách entity và nhận lại cấu hình hoạt động: + +```json +{ + "agent_configs": [ + { + "agent_id": 0, + "activity_level": 0.8, + "posts_per_hour": 0.6, + "comments_per_hour": 1.5, + "active_hours": [8, 9, 10, 18, 19, 20, 21, 22, 23], + "response_delay_min": 1, + "response_delay_max": 15, + "sentiment_bias": -0.3, + "stance": "opposing", + "influence_weight": 0.8 + } + ] +} +``` + +**Giải thích các thông số:** +- `activity_level`: 0.0–1.0, xác suất agent "thức dậy" trong một round +- `posts_per_hour`: Trung bình số bài đăng mới mỗi giờ +- `comments_per_hour`: Trung bình số bình luận mỗi giờ +- `active_hours`: Danh sách giờ trong ngày agent có thể hoạt động (0–23) +- `response_delay_min/max`: Thời gian (phút mô phỏng) agent chờ trước khi phản ứng với tin hot +- `sentiment_bias`: -1.0 (rất tiêu cực) → 0.0 (trung lập) → 1.0 (rất tích cực) +- `stance`: "supportive" / "opposing" / "neutral" / "observer" +- `influence_weight`: Mức độ bài đăng của agent này được agent khác nhìn thấy và quan tâm + +**Fallback theo rule khi LLM fail:** + +| Entity Type | activity | active_hours | response_delay | influence | +|---|---|---|---|---| +| university / governmentagency | 0.2 | 9–17 | 60–240 phút | 3.0 | +| mediaoutlet | 0.5 | 7–23 | 5–30 phút | 2.5 | +| professor / expert / official | 0.4 | 8–21 | 15–90 phút | 2.0 | +| student | 0.8 | Sáng + Tối | 1–15 phút | 0.8 | +| alumni | 0.6 | Trưa + Tối | 5–30 phút | 1.0 | +| Default | 0.7 | 9–23 | 2–20 phút | 1.0 | + +### 7.6 Cơ chế retry chung cho mọi LLM call + +Tất cả LLM call trong `SimulationConfigGenerator` đi qua `_call_llm_with_retry()`: + +``` +Attempt 1: temperature=0.7 + ↓ (nếu fail hoặc JSON lỗi) +Attempt 2: temperature=0.6 + ↓ +Attempt 3: temperature=0.5 + ↓ (nếu vẫn fail) +→ raise exception → caller dùng hardcode fallback +``` + +Giữa các lần retry nếu lỗi connection: sleep `2*(attempt+1)` giây (2s, 4s). + +### 7.7 File đầu ra + +`uploads/simulations/{sim_id}/simulation_config.json` — **File quan trọng nhất**, được đọc bởi cả Flask và OASIS scripts: + +```json +{ + "simulation_id": "sim_a3f9c2b81e4d", + "project_id": "proj_xyz", + "graph_id": "graph_abc", + "simulation_requirement": "Mô phỏng phản ứng của cộng đồng khi Đại học X tăng học phí 20%", + "time_config": { + "total_simulation_hours": 72, + "minutes_per_round": 60, + "agents_per_hour_min": 5, + "agents_per_hour_max": 30, + "peak_hours": [19, 20, 21, 22], + "off_peak_hours": [0, 1, 2, 3, 4, 5], + "peak_activity_multiplier": 1.5, + "off_peak_activity_multiplier": 0.05 + }, + "agent_configs": [ + { + "agent_id": 0, + "entity_uuid": "e1a2b3...", + "entity_name": "Trần Văn An", + "entity_type": "Student", + "activity_level": 0.8, + "stance": "opposing", + "influence_weight": 0.8, + ... + } + ], + "event_config": { + "hot_topics": ["học phí", "biểu tình"], + "narrative_direction": "...", + "initial_posts": [ + { + "content": "CHÍNH THỨC: Đại học X tăng học phí...", + "poster_type": "MediaOutlet", + "poster_agent_id": 12 + } + ] + }, + "twitter_config": { + "platform": "twitter", + "recency_weight": 0.4, + "popularity_weight": 0.3, + "viral_threshold": 10, + "echo_chamber_strength": 0.5 + }, + "reddit_config": { + "platform": "reddit", + "recency_weight": 0.3, + "popularity_weight": 0.4, + "viral_threshold": 15, + "echo_chamber_strength": 0.6 + } +} +``` + +Sau khi lưu file này, `state.status` được cập nhật thành `READY`. + +--- + +## 8. Giai đoạn 3 — Chạy Simulation (OASIS) + +**Files:** `backend/app/services/simulation_runner.py` + `backend/scripts/run_parallel_simulation.py` + +### 8.1 Mô hình chạy + +Flask **không** chạy OASIS trực tiếp trong cùng process. Thay vào đó: + +``` +Flask Process + │ + └── subprocess.Popen(run_parallel_simulation.py --config ...) + │ + ├── OASIS Twitter Environment (async task) + └── OASIS Reddit Environment (async task) +``` + +**Lý do tách process:** OASIS là framework async nặng, chạy nhiều agent đồng thời. Nếu chạy trong Flask sẽ block hoàn toàn API server. Tách process giúp Flask vẫn nhận được request trong khi OASIS chạy ngầm. + +### 8.2 Khởi động Simulation + +**Hàm:** `SimulationRunner.start_simulation()` + +```python +# 1. Load config +with open(config_path) as f: + config = json.load(f) + +# 2. Tính tổng số round +total_rounds = int(total_hours * 60 / minutes_per_round) +# Ví dụ: 72 giờ × 60 phút / 60 phút/round = 72 rounds + +# 3. Chọn script +script_name = "run_parallel_simulation.py" # hoặc twitter/reddit only + +# 4. Khởi động subprocess +process = subprocess.Popen( + [sys.executable, script_path, "--config", config_path], + cwd=sim_dir, + stdout=main_log_file, # stdout → simulation.log + stderr=subprocess.STDOUT, + start_new_session=True, # Tạo process group mới (quan trọng cho kill) + env={...PYTHONUTF8=1...} +) + +# 5. Lưu PID để có thể kill sau này +state.process_pid = process.pid + +# 6. Khởi động thread giám sát +monitor_thread = Thread(target=cls._monitor_simulation, args=(simulation_id,), daemon=True) +monitor_thread.start() +``` + +### 8.3 Bên trong OASIS Script + +**File:** `backend/scripts/run_parallel_simulation.py` + +Script này chạy bằng Python asyncio, khởi tạo hai môi trường mạng xã hội: + +```python +# Tạo LLM model (dùng camel-ai ModelFactory) +model = ModelFactory.create(ModelPlatformType.OPENAI, ...) + +# Tạo Twitter environment +twitter_env = TwitterSocialEnvironment(...) +twitter_agent_graph = generate_twitter_agent_graph( + profiles=twitter_profiles, # Đọc từ twitter_profiles.csv + model=model, + initial_posts=initial_posts # Đọc từ simulation_config.json +) + +# Tạo Reddit environment +reddit_env = RedditSocialEnvironment(...) +reddit_agent_graph = generate_reddit_agent_graph( + profiles=reddit_profiles, # Đọc từ reddit_profiles.json + model=model, + initial_posts=initial_posts +) +``` + +**Vòng lặp Simulation:** +``` +Round 1 (giờ 0:00 - 1:00) + │ + ├── Tính activity_multiplier: + │ hour=0 → off_peak → multiplier=0.05 + │ hour=20 → peak → multiplier=1.5 + │ + ├── Tính số agent active = random(min, max) * multiplier + │ + ├── Chọn ngẫu nhiên N agents từ danh sách + │ + ├── Với mỗi agent được chọn: + │ ├── Agent "nhìn" timeline (các bài đăng gần đây) + │ ├── LLM suy nghĩ dựa trên persona + nội dung thấy được + │ └── LLM chọn một action: + │ Twitter: CREATE_POST / LIKE_POST / REPOST / FOLLOW / DO_NOTHING / QUOTE_POST + │ Reddit: CREATE_POST / CREATE_COMMENT / LIKE_POST / DISLIKE_POST / SEARCH_POSTS / ... + │ + └── Ghi log ra actions.jsonl + {"round": 1, "agent_id": 5, "agent_name": "...", "action_type": "CREATE_POST", "action_args": {...}} + +Round 2... +... +Round 72 → ghi {"event_type": "simulation_end", "total_rounds": 72, "total_actions": 3420} +``` + +### 8.4 Cấu trúc file log + +``` +uploads/simulations/{sim_id}/ +├── twitter/ +│ └── actions.jsonl ← mỗi dòng là 1 hành động trên Twitter +├── reddit/ +│ └── actions.jsonl ← mỗi dòng là 1 hành động trên Reddit +├── simulation.log ← stdout/stderr của subprocess +└── run_state.json ← trạng thái real-time (Flask cập nhật) +``` + +**Ví dụ một dòng trong `actions.jsonl`:** +```json +{ + "round": 15, + "timestamp": "2026-05-04T15:30:00", + "agent_id": 3, + "agent_name": "tran_van_an_492", + "action_type": "CREATE_POST", + "action_args": { + "content": "Không chịu nổi rồi! Học phí tăng 20% mà lương bố mẹ mình có tăng đâu... #HọcPhíTăng #ĐạiHọcX" + }, + "result": "Post created with id 142", + "success": true +} +``` + +**Sự kiện hệ thống (không phải hành động agent):** +```json +{"event_type": "round_end", "round": 15, "simulated_hours": 15} +{"event_type": "simulation_end", "total_rounds": 72, "total_actions": 3420} +``` + +### 8.5 Monitor Thread + +**Hàm:** `SimulationRunner._monitor_simulation()` — chạy liên tục trong nền: + +``` +Mỗi 2 giây: + ├── Đọc twitter/actions.jsonl từ vị trí đã đọc lần trước (file seek) + │ → parse từng dòng JSON mới + │ → cập nhật state.twitter_current_round, state.twitter_actions_count + │ → phát hiện event_type "simulation_end" → state.twitter_completed = True + │ + ├── Làm tương tự với reddit/actions.jsonl + │ + ├── Kiểm tra _check_all_platforms_completed(): + │ Nếu cả 2 platform đều completed → state.runner_status = COMPLETED + │ + └── Lưu run_state.json (để Frontend có thể poll tiến độ qua API) +``` + +**Khi subprocess kết thúc:** +- `exit_code = 0` → `COMPLETED` +- `exit_code != 0` → `FAILED`, đọc 2000 ký tự cuối `simulation.log` làm error message +- Dù thế nào: đóng file handles, xóa khỏi `_processes` dict + +### 8.6 Dừng Simulation + +**Hàm:** `SimulationRunner.stop_simulation()` + +Phải dừng cả **process group** (bao gồm subprocess và các child của nó): + +```python +# Unix/Linux/Mac: +pgid = os.getpgid(process.pid) +os.killpg(pgid, signal.SIGTERM) # Gửi SIGTERM nhẹ nhàng +process.wait(timeout=10) # Chờ tối đa 10s +# Nếu timeout → os.killpg(pgid, signal.SIGKILL) # Kill cưỡng bức + +# Windows: +subprocess.run(['taskkill', '/PID', str(pid), '/T']) # /T = kill cả tree +# Nếu timeout → subprocess.run(['taskkill', '/F', '/PID', str(pid), '/T']) # /F = force +``` + +Lý do phải dùng process group thay vì kill process đơn: OASIS script có thể spawn thêm child process, phải kill toàn bộ. + +### 8.7 IPC — Interview Agent đang chạy + +Trong khi simulation đang chạy, người dùng có thể "phỏng vấn" một agent bất kỳ: + +``` +Flask API nhận request interview(agent_id=5, prompt="Bạn nghĩ gì về việc tăng học phí?") + │ + └── SimulationIPCClient.send_interview() + │ + ├── Ghi file: ipc_commands/{uuid}.json + │ {"command_id": "...", "command_type": "interview", "args": {"agent_id": 5, "prompt": "..."}} + │ + └── Poll ipc_responses/{uuid}.json (mỗi 0.5s, timeout 60s) + │ + └── (Bên trong OASIS script, ParallelIPCHandler poll ipc_commands/*) + │ + ├── Phát hiện command mới + ├── Chạy ManualAction INTERVIEW với agent 5 + ├── Agent 5 LLM trả lời câu hỏi dựa trên persona + └── Ghi kết quả: ipc_responses/{uuid}.json + {"command_id": "...", "status": "completed", "result": {"response": "..."}} +``` + +**Fallback interview:** Nếu environment không alive (env_status.json không có status="alive") → raise `ValueError` ngay, không gửi IPC. + +--- + +## 9. Giai đoạn 4 — Tạo Report (ReACT Agent) + +**File:** `backend/app/services/report_agent.py` + +### 9.1 ReACT là gì? + +ReACT (Reason + Act) là một kỹ thuật prompting cho LLM, trong đó LLM được phép: +1. **Suy nghĩ** (Reason) về việc cần làm +2. **Gọi công cụ** (Act) để lấy thêm thông tin +3. **Quan sát** kết quả trả về +4. Lặp lại cho đến khi có đủ thông tin để trả lời + +Thay vì "hỏi LLM một câu, nhận một câu trả lời", ReACT cho phép LLM tự chủ đi tìm dữ liệu trước khi viết. + +### 9.2 Triết lý của báo cáo + +Báo cáo được định vị là **"Báo cáo Dự báo Tương lai"** — không phải phân tích dữ liệu khô khan, mà là: *"Trong thế giới mô phỏng, điều này đã xảy ra — đây là dự báo về những gì có thể xảy ra trong tương lai thực."* + +Các Agent trong mô phỏng được xem là **đại diện cho hành vi con người trong tương lai**. + +### 9.3 Luồng tổng quát + +``` +generate_report(simulation_id, graph_id, simulation_requirement) + │ + ├── Phase A: Planning (Lên dàn ý) + │ ├── Query Zep → lấy statistics (total nodes, edges, entity types) + │ ├── Lấy sample facts từ graph + │ ├── LLM (PLAN_SYSTEM_PROMPT + context) → JSON outline + │ └── → ReportOutline {title, summary, sections: [2-5 sections]} + │ + ├── Phase B: Generating (Viết từng section) + │ └── Với mỗi section: + │ └── ReACT loop (tối đa 5 lần tool call, tối thiểu 3): + │ ├── LLM suy nghĩ → gọi tool → nhận kết quả → suy nghĩ tiếp + │ └── Khi đủ data → "Final Answer: [nội dung section]" + │ + └── Phase C: Assembly (Ghép báo cáo) + ├── Ghép tất cả sections thành Markdown + └── Lưu report.json + report.md +``` + +### 9.4 Phase A — Lên dàn ý + +**Input vào LLM:** +``` +PLAN_SYSTEM_PROMPT: + "Bạn là chuyên gia viết Báo cáo Dự báo Tương lai, với góc nhìn + của Chúa về thế giới mô phỏng..." + +PLAN_USER_PROMPT: + "Biến số tiêm vào: {simulation_requirement} + Quy mô: {total_nodes} nodes, {total_edges} edges + Phân phối loại entity: {entity_types} + Sample facts: {related_facts_json}" +``` + +**Output:** +```json +{ + "title": "Dự báo Làn sóng Phản đối Học phí Tăng tại Đại học X", + "summary": "Mô phỏng cho thấy quyết định tăng 20% học phí sẽ kích hoạt làn sóng biểu tình trực tuyến trong vòng 48 giờ, dẫn đầu bởi sinh viên năm 3-4", + "sections": [ + {"title": "Sự bùng nổ ban đầu trên mạng xã hội", "description": "Phân tích..."}, + {"title": "Phân cực dư luận: Ủng hộ vs Phản đối", "description": "..."}, + {"title": "Phản ứng của các bên liên quan", "description": "..."}, + {"title": "Dự báo xu hướng và rủi ro", "description": "..."} + ] +} +``` + +### 9.5 Phase B — Viết từng Section (ReACT) + +Với mỗi section, một vòng hội thoại mới bắt đầu: + +**System Prompt** cho LLM bao gồm: +- Tiêu đề và tóm tắt báo cáo +- Yêu cầu mô phỏng +- Tiêu đề section đang viết +- Mô tả 4 công cụ có sẵn +- Luật: phải gọi tool 3-5 lần, không dùng heading (#), dịch mọi content sang tiếng Việt + +**4 công cụ khả dụng:** + +| Tool | Mô tả | Dùng khi | +|---|---|---| +| `insight_forge(query)` | Tự chia query thành sub-questions, search Zep đa chiều (facts + entities + relationships) | Phân tích sâu một chủ đề phức tạp | +| `panorama_search(query)` | Lấy toàn cảnh evolution của event, phân biệt facts hiện tại vs lịch sử | Hiểu timeline, diễn biến sự kiện | +| `quick_search(query)` | Search đơn giản, nhanh | Xác minh một thông tin cụ thể | +| `interview_agents(topic, agent_types, question)` | Phỏng vấn thực Agent đang chạy qua IPC | Cần góc nhìn first-person từ nhân vật | + +**Ví dụ vòng ReACT:** + +``` +LLM lần 1: + [Suy nghĩ] Tôi cần hiểu sinh viên đã phản ứng như thế nào... + + {"name": "insight_forge", "parameters": {"query": "phản ứng của sinh viên khi học phí tăng"}} + + +[Hệ thống inject kết quả tool] + Observation: "Facts: Trần Văn An đăng status phản đối lúc 20:15. + Nhóm sinh viên tự phát tạo hashtag #HọcPhíTăng lúc 21:00..." + +LLM lần 2: + [Suy nghĩ] Tốt, cần thêm góc nhìn từ truyền thông... + + {"name": "interview_agents", "parameters": { + "topic": "học phí tăng", + "agent_types": ["MediaOutlet"], + "question": "Bạn đưa tin về vụ tăng học phí này như thế nào?" + }} + + +[Kết quả interview thực từ OASIS qua IPC] + Observation: "VnExpress_reporter_291: Chúng tôi đã đăng bài ngay khi nhận được + thông cáo chính thức, thu hút 50,000 lượt xem trong 2 giờ đầu..." + +LLM lần 3: + {"name": "panorama_search", ...} + +LLM lần 4: + {"name": "quick_search", ...} + +LLM lần 5 (khi đã đủ data): + Final Answer: + Trong 24 giờ đầu sau thông báo, mạng xã hội đã chứng kiến một làn sóng phản ứng + chưa từng có... + + **Giai đoạn bùng nổ (Giờ 0-4)** + + Ngay khi thông báo được đăng tải, các tài khoản truyền thông lớn nhanh chóng + tiếp nhận và khuếch đại: + + > "Trần Văn An sẽ nói: Không chịu nổi rồi! Học phí tăng 20% mà lương bố mẹ + mình có tăng đâu... #HọcPhíTăng" + + Phản ứng này phản ánh tâm lý chung của nhóm sinh viên có hoàn cảnh khó khăn... +``` + +**Quy tắc định dạng nghiêm ngặt:** +- ❌ Không dùng `#`, `##`, `###` (heading) +- ✅ Dùng `**bold**` thay thế +- ✅ Quote agent speech bằng `>` format +- ✅ Dịch mọi content tiếng Anh/tiếng Trung sang tiếng Việt +- ❌ Không fabricate thông tin không có trong simulation data + +### 9.6 Logging chi tiết + +Mọi bước trong quá trình tạo report được ghi vào `agent_log.jsonl`: + +``` +uploads/reports/{report_id}/ +├── agent_log.jsonl ← JSONL, mỗi dòng là một event (tool_call, llm_response, section_complete...) +├── console_log.txt ← Log dạng console text (INFO/WARNING level) +├── report.json ← Full report object +└── report.md ← Nội dung Markdown +``` + +**Ví dụ nội dung `agent_log.jsonl`:** +```json +{"timestamp": "2026-05-04T16:00:00", "action": "planning_complete", "stage": "planning", "details": {"outline": {...}}} +{"timestamp": "2026-05-04T16:01:00", "action": "section_start", "stage": "generating", "section_title": "Sự bùng nổ ban đầu"} +{"timestamp": "2026-05-04T16:01:05", "action": "tool_call", "details": {"tool_name": "insight_forge", "parameters": {...}}} +{"timestamp": "2026-05-04T16:01:10", "action": "tool_result", "details": {"tool_name": "insight_forge", "result": "...(full)"}} +{"timestamp": "2026-05-04T16:02:00", "action": "section_complete", "section_title": "Sự bùng nổ ban đầu", "details": {"content": "...(full)"}} +{"timestamp": "2026-05-04T16:05:00", "action": "report_complete", "details": {"total_sections": 4, "total_time_seconds": 300.5}} +``` + +Frontend lắng nghe log này để hiển thị tiến độ real-time và nội dung từng section khi hoàn thành. + +--- + +## 10. Cơ chế Fallback toàn hệ thống + +### 10.1 Bảng tổng hợp + +| Giai đoạn | Thành phần | Lỗi | Fallback | Severity | +|---|---|---|---|---| +| 2A | ZepEntityReader | API lỗi | Retry 3 lần, backoff 2s→4s→8s | Tiếp tục | +| 2A | ZepEntityReader | 0 entity sau lọc | **Dừng pipeline**, status=FAILED | **Nghiêm trọng** | +| 2B | Zep Vector Search | Timeout / lỗi | Bỏ qua, context rỗng, tiếp tục | Nhẹ | +| 2B | LLM Profile Gen | JSON lỗi | `_fix_truncated_json()` → `_try_fix_json()` | Tự phục hồi | +| 2B | LLM Profile Gen | 3 lần fail | `_generate_profile_rule_based()` | Degraded | +| 2B | ThreadPoolExecutor | Exception 1 entity | Dùng fallback profile, tiếp tục | Nhẹ | +| 2C | LLM Time Config | Fail | Hardcoded defaults (72h, 60min/round) | Degraded | +| 2C | LLM Event Config | Fail | Empty hot_topics, no initial_posts | Degraded | +| 2C | LLM Agent Config | 1 agent thiếu | `_generate_agent_config_by_rule()` | Nhẹ | +| 2C | poster_type match | Không khớp | Agent có influence_weight cao nhất | Nhẹ | +| 2C | LLM JSON | Cắt ngang | `_fix_truncated_json()` | Tự phục hồi | +| 3 | subprocess | exit_code != 0 | RunnerStatus=FAILED, đọc log | Nghiêm trọng | +| 3 | SIGTERM | Timeout 10s | SIGKILL (Unix) / taskkill /F (Win) | Forced | +| 3 | IPC Interview | Environment chết | ValueError ngay, không wait | Nhẹ | +| 3 | IPC Interview | Timeout 60s | TimeoutError trả về caller | Nhẹ | +| 4 | Zep tool call | Lỗi | Log warning, tiếp tục ReACT | Nhẹ | +| 4 | Interview tool | Env không alive | Trả về error message, tiếp tục | Nhẹ | + +### 10.2 Mức độ ảnh hưởng + +**Scenario 1 — LLM hoàn toàn không khả dụng:** +- Profile: Tất cả dùng rule-based → Profiles cơ bản nhưng vẫn chạy được +- Config: Hardcoded defaults → Mô phỏng với cài đặt chuẩn thay vì tùy chỉnh +- Report: Không thể tạo (phụ thuộc LLM hoàn toàn) + +**Scenario 2 — Zep không khả dụng:** +- Nếu không fetch được entity → Pipeline dừng ở 2A +- Nếu Vector Search fail → Profile thiếu context nhưng vẫn tạo được + +**Scenario 3 — OASIS subprocess crash:** +- RunnerStatus = FAILED +- Log lỗi được lưu vào simulation.log +- Người dùng có thể cleanup rồi chạy lại + +--- + +## 11. Cấu trúc file và thư mục + +### 11.1 Thư mục Simulation + +``` +uploads/simulations/{simulation_id}/ +│ +├── state.json ← Trạng thái lifecycle (CREATED/PREPARING/READY/RUNNING/...) +├── run_state.json ← Trạng thái runtime (round hiện tại, actions count, ...) +├── simulation_config.json ← Config đầy đủ (time, agents, events, platforms) +│ +├── reddit_profiles.json ← Profile agents format Reddit +├── twitter_profiles.csv ← Profile agents format Twitter +│ +├── simulation.log ← stdout + stderr của OASIS subprocess +├── env_status.json ← Trạng thái IPC environment (alive/stopped) +│ +├── twitter/ +│ └── actions.jsonl ← Log hành động Twitter (từng dòng = 1 action) +│ +├── reddit/ +│ └── actions.jsonl ← Log hành động Reddit +│ +├── ipc_commands/ ← Flask ghi lệnh vào đây +│ └── {uuid}.json +│ +├── ipc_responses/ ← OASIS ghi response vào đây +│ └── {uuid}.json +│ +├── twitter_simulation.db ← SQLite database của OASIS Twitter +└── reddit_simulation.db ← SQLite database của OASIS Reddit +``` + +### 11.2 Thư mục Report + +``` +uploads/reports/{report_id}/ +│ +├── report.json ← Full report object (JSON) +├── report.md ← Nội dung Markdown +│ +├── agent_log.jsonl ← Log chi tiết từng bước ReACT (JSONL) +└── console_log.txt ← Log console text (INFO/WARNING) +``` + +--- + +## 12. Luồng dữ liệu end-to-end + +### 12.1 Sơ đồ tổng hợp + +``` +[USER] + │ Upload tài liệu + yêu cầu mô phỏng + ↓ +[TextProcessor] + │ document_text (cleaned string) + ↓ +[Zep Graph Builder] ← (pipeline riêng, chạy trước) + │ Graph {Nodes + Edges} + ↓ +[ZepEntityReader.filter_defined_entities()] + │ List[EntityNode] — 47 entities + │ (mỗi entity: uuid, name, labels, summary, related_edges, related_nodes) + ↓ +[OasisProfileGenerator.generate_profiles_from_entities()] + │ ← Zep Vector Search (parallel, cho từng entity) + │ ← LLM (parallel, 3 luồng, retry 3 lần / rule-based fallback) + │ List[OasisAgentProfile] — 47 profiles + │ → reddit_profiles.json (realtime save) + │ → twitter_profiles.csv (realtime save) + ↓ +[SimulationConfigGenerator.generate_config()] + │ ← LLM (4 bước tuần tự: time → event → agents×3 batch → platform) + │ SimulationParameters + │ → simulation_config.json + ↓ +[SimulationRunner.start_simulation()] + │ subprocess.Popen(run_parallel_simulation.py --config ...) + │ + │ [OASIS Script — Subprocess] + │ ├── Read: reddit_profiles.json + twitter_profiles.csv + │ ├── Read: simulation_config.json + │ ├── Init: Twitter Environment + Reddit Environment + │ └── Loop 72 rounds: + │ └── N agents active per round + │ └── LLM(persona + timeline) → action → log + │ → twitter/actions.jsonl + │ → reddit/actions.jsonl + │ + │ [Monitor Thread — Flask] + │ Đọc actions.jsonl mỗi 2s → cập nhật run_state.json + ↓ +[ReportAgent.generate_report()] + │ ← Zep Graph (query statistics + facts) + │ ← LLM (Planning: 1 call) + │ ← LLM ReACT (Per section: 3-5 tool calls + 1 final answer) + │ ← ZepTools (insight_forge / panorama_search / quick_search) + │ ← SimulationRunner.interview_agents_batch() (qua IPC nếu env alive) + │ → report.md + │ → agent_log.jsonl + ↓ +[USER nhận báo cáo] +``` + +### 12.2 Trạng thái lifecycle của SimulationState + +``` +CREATED + ↓ (prepare_simulation bắt đầu) +PREPARING + ↓ (tất cả 3 phase chuẩn bị xong) +READY + ↓ (start_simulation được gọi) +RUNNING ←→ PAUSED (tính năng tương lai) + ↓ (mô phỏng tự nhiên kết thúc) +COMPLETED + ↓ (hoặc) +FAILED ← (bất kỳ lỗi nghiêm trọng nào) +STOPPED ← (người dùng chủ động dừng) +``` + +### 12.3 Trạng thái lifecycle của RunnerStatus + +``` +IDLE + ↓ (start_simulation) +STARTING + ↓ (subprocess bắt đầu) +RUNNING + ↓ (simulation_end event trong actions.jsonl) +COMPLETED + +Hoặc: +RUNNING → STOPPING → STOPPED (người dùng stop) +RUNNING → FAILED (subprocess crash) +``` + +--- + +*Báo cáo này được tạo bởi Claude Sonnet 4.6 dựa trên phân tích mã nguồn MiroFish.* +*Cập nhật lần cuối: 2026-05-04* diff --git a/backend/app/services/graph_builder.py b/backend/app/services/graph_builder.py index bbff79d1..d1deeb42 100644 --- a/backend/app/services/graph_builder.py +++ b/backend/app/services/graph_builder.py @@ -4,6 +4,7 @@ API 2: Sử dụng Zep API để xây dựng một Standalone Graph (Đồ thị """ import os +import re # changed import uuid import time import threading @@ -330,27 +331,47 @@ class GraphBuilderService: ] # Khởi chạy gửi cho Zep Server - try: - # UPLOAD BATCH TO ZEP - Sends batch of episodes for entity extraction - batch_result = self.client.graph.add_batch( - graph_id=graph_id, - episodes=episodes - ) - - # Cập nhật và thu thập lại UUID của các Episode được trả về sau khi tạo mới - if batch_result and isinstance(batch_result, list): - for ep in batch_result: - ep_uuid = getattr(ep, 'uuid_', None) or getattr(ep, 'uuid', None) - if ep_uuid: - episode_uuids.append(ep_uuid) - - # Cài thời gian chờ (delay) nhỏ để tránh rate-limit bị quá tải số lượng requests - time.sleep(1) - - except Exception as e: - if progress_callback: - progress_callback(f"Failed to send batch {batch_num}: {str(e)}", 0) - raise + max_retries = 10 # changed + for attempt in range(max_retries): # changed + try: + # UPLOAD BATCH TO ZEP - Sends batch of episodes for entity extraction + batch_result = self.client.graph.add_batch( + graph_id=graph_id, + episodes=episodes + ) + + # Cập nhật và thu thập lại UUID của các Episode được trả về sau khi tạo mới + if batch_result and isinstance(batch_result, list): + for ep in batch_result: + ep_uuid = getattr(ep, 'uuid_', None) or getattr(ep, 'uuid', None) + if ep_uuid: + episode_uuids.append(ep_uuid) + + # Cài thời gian chờ (delay) nhỏ để tránh rate-limit bị quá tải số lượng requests + time.sleep(3) + break # changed: thoát retry loop nếu thành công + + except Exception as e: # changed + err_str = str(e) + if "episode usage limit" in err_str or ("status_code: 429" in err_str): # changed: bắt lỗi rate-limit + # Đọc thời điểm reset từ error message để biết cần chờ bao lâu + reset_match = re.search(r"x-ratelimit-reset['\"]:\s*['\"]?(\d+)", err_str) # changed + if reset_match: # changed + wait_seconds = max(int(reset_match.group(1)) - int(time.time()) + 2, 5) # changed + else: # changed + wait_seconds = 20 # changed: fallback 12s (5 calls/phút → cách nhau 12s) + if progress_callback: # changed + progress_callback( # changed + f"Rate limited by Zep. Waiting {wait_seconds}s then retry (attempt {attempt + 1}/{max_retries})...", # changed + (i + len(batch_chunks)) / total_chunks # changed + ) # changed + time.sleep(wait_seconds) # changed + else: # changed: lỗi khác thì raise luôn, không retry + if progress_callback: + progress_callback(f"Failed to send batch {batch_num}: {err_str}", 0) + raise # changed + else: # changed: for...else — chạy khi hết max_retries mà vẫn chưa break + raise Exception(f"Batch {batch_num} failed after {max_retries} retries due to rate limiting") # changed return episode_uuids @@ -358,7 +379,7 @@ class GraphBuilderService: self, episode_uuids: List[str], progress_callback: Optional[Callable] = None, - timeout: int = 600 + timeout: int = 1000 ): """Chạy vòng lặp để kiểm tra và chờ cho tới khi mọi Episode (các khối Text) đều hoàn tất quá trình process từ hệ thống""" if not episode_uuids: diff --git a/backend/app/services/oasis_profile_generator.py b/backend/app/services/oasis_profile_generator.py index b37284cf..d162b508 100644 --- a/backend/app/services/oasis_profile_generator.py +++ b/backend/app/services/oasis_profile_generator.py @@ -1,7 +1,21 @@ """ -Trình tạo tạo ra Profile Agent (Hồ sơ Nhân vật) cho Agent bằng Framework OASIS +Trình tạo Profile Agent (Hồ sơ Nhân vật) cho Agent bằng Framework OASIS Chuyển đổi dữ liệu Thực thể được Query từ Zep ra chuẩn định dạng của các Agent tham gia vào mạng +Vị trí trong pipeline: +───────────────────────────────────────────────────────────────────────────── + simulation_manager.prepare_simulation() [Giai đoạn 2] + └─ OasisProfileGenerator.generate_profiles_from_entities() + └─ generate_profile_from_entity() [mỗi entity 1 lần] + ├─ _build_entity_context() [tổng hợp ngữ cảnh] + │ └─ _search_zep_for_entity() [vector search] + └─ _generate_profile_with_llm() [gọi LLM] + └─ _generate_profile_rule_based() [fallback] +───────────────────────────────────────────────────────────────────────────── + +Input: List[EntityNode] từ ZepEntityReader (zep_entity_reader.py) +Output: List[OasisAgentProfile] → ghi ra reddit_profiles.json / twitter_profiles.csv + Nâng cấp cải thiện: 1. Kết hợp dùng chức năng Search trên Zep để lấy Profile giàu sắc thái 2. Gen các cấu hình về tính cách một cách sắc xảo và cực sâu cho Prompt @@ -26,51 +40,73 @@ from .zep_entity_reader import EntityNode, ZepEntityReader logger = get_logger('mirofish.oasis_profile') +# ============================================================================== +# DATACLASS: OasisAgentProfile — Cấu trúc hồ sơ 1 agent trong OASIS +# ============================================================================== +# Mỗi EntityNode từ Zep → 1 OasisAgentProfile sau khi qua bước generate. +# Class này đóng vai trò "adapter": chuẩn hoá dữ liệu thành 2 format +# mà OASIS yêu cầu (Twitter CSV / Reddit JSON). +# +# Quan hệ các field với OASIS: +# user_id → OASIS dùng để map agent trong agent_graph.get_agent() +# user_name → username trên mạng xã hội ảo (không có dấu cách) +# bio → thông tin hiển thị công khai trên trang cá nhân +# persona → system prompt bí mật điều khiển mọi hành vi của LLM agent +# ============================================================================== + @dataclass class OasisAgentProfile: """Cấu trúc của Dataclass Profile Agent qua quy định của OASIS""" - # Các Field thông dụng (Common data) - user_id: int - user_name: str - name: str - bio: str - persona: str - - # Chọn bổ sung (Tùy chọn) - Thông số nền tảng Reddit (Karma) - karma: int = 1000 - - # Chọn bổ sung (Tùy chọn) - Thông số nền tảng Twitter - friend_count: int = 100 - follower_count: int = 150 - statuses_count: int = 500 - - # Một số Thông tin Data cá nhân bổ sung (Bóp để tăng tính Thực tế nếu LLM sinh ra) + + # --- Định danh bắt buộc --- + user_id: int # Index số nguyên, bắt đầu từ 0 (bắt buộc để OASIS map đúng agent) + user_name: str # Username không dấu cách, ví dụ: "nguyen_van_a_392" (sinh từ _generate_username) + name: str # Tên thật hiển thị, ví dụ: "Nguyễn Văn A" + + # --- Nội dung agent --- + bio: str # Tiểu sử ngắn (~150-200 ký tự), hiển thị công khai trên trang cá nhân + persona: str # Mô tả nhân cách chi tiết (~2000 từ), được nhét vào system prompt của agent LLM + # Đây là thứ THỰC SỰ điều khiển agent nghĩ và nói gì + + # --- Thông số mạng xã hội (tác động đến "trọng số" của agent trong OASIS) --- + karma: int = 1000 # Reddit: điểm uy tín (cao = post được nhiều người thấy hơn) + friend_count: int = 100 # Twitter: số người đang follow + follower_count: int = 150 # Twitter: số người follow mình + statuses_count: int = 500 # Twitter: số tweet đã đăng (độ hoạt động) + + # --- Thông tin cá nhân bổ sung (tùy chọn, làm phong phú persona) --- age: Optional[int] = None - gender: Optional[str] = None - mbti: Optional[str] = None + gender: Optional[str] = None # "male", "female", hoặc "other" (tổ chức) + mbti: Optional[str] = None # Ví dụ: "INTJ", "ENFP" — gợi ý cách LLM phản ứng country: Optional[str] = None profession: Optional[str] = None - interested_topics: List[str] = field(default_factory=list) - - # Lịch sử thông tin entity gốc được lấy - source_entity_uuid: Optional[str] = None - source_entity_type: Optional[str] = None - + interested_topics: List[str] = field(default_factory=list) # Các chủ đề agent quan tâm + + # --- Metadata truy vết nguồn gốc --- + source_entity_uuid: Optional[str] = None # UUID của EntityNode gốc trong Zep + source_entity_type: Optional[str] = None # Loại entity (ví dụ: "Person", "Organization") + created_at: str = field(default_factory=lambda: datetime.now().strftime("%Y-%m-%d")) - + def to_reddit_format(self) -> Dict[str, Any]: - """Convert trả ra cho định dạng Agent reddit""" + """ + Xuất profile thành dict theo chuẩn Reddit của OASIS. + + Khác biệt với to_twitter_format(): có karma thay vì friend/follower_count. + Các field tùy chọn (age, gender, ...) chỉ được thêm vào nếu có giá trị + để tránh null gây lỗi trong OASIS engine. + """ profile = { "user_id": self.user_id, - "username": self.user_name, # Source mã của OASIS Library yêu cầu không có dấu "_" cho param username + "username": self.user_name, # OASIS yêu cầu không có dấu "_" là không có vấn đề nhưng không được có khoảng trắng "name": self.name, "bio": self.bio, "persona": self.persona, "karma": self.karma, "created_at": self.created_at, } - - # Merge Thông tin Data Profile Cá nhân (Nếu CÓ) + + # Chỉ thêm field nếu có giá trị — OASIS không xử lý được None if self.age: profile["age"] = self.age if self.gender: @@ -83,14 +119,19 @@ class OasisAgentProfile: profile["profession"] = self.profession if self.interested_topics: profile["interested_topics"] = self.interested_topics - + return profile - + def to_twitter_format(self) -> Dict[str, Any]: - """Convert trả ra cho định dạng Agent Twitter""" + """ + Xuất profile thành dict theo chuẩn Twitter của OASIS. + + Khác biệt với to_reddit_format(): có friend_count, follower_count, statuses_count + thay vì karma. Cả hai format đều dùng chung bio và persona. + """ profile = { "user_id": self.user_id, - "username": self.user_name, # Tương tự như trên + "username": self.user_name, "name": self.name, "bio": self.bio, "persona": self.persona, @@ -99,8 +140,7 @@ class OasisAgentProfile: "statuses_count": self.statuses_count, "created_at": self.created_at, } - - # Merge Thông tin Data Profile Cả nhân + if self.age: profile["age"] = self.age if self.gender: @@ -113,11 +153,11 @@ class OasisAgentProfile: profile["profession"] = self.profession if self.interested_topics: profile["interested_topics"] = self.interested_topics - + return profile - + def to_dict(self) -> Dict[str, Any]: - """Quy đổi thành toàn bộ Dictionary Cấu Trúc Khép Kín """ + """Xuất toàn bộ profile thành dict không lọc (kể cả source_entity_uuid/type).""" return { "user_id": self.user_id, "user_name": self.user_name, @@ -140,107 +180,134 @@ class OasisAgentProfile: } +# ============================================================================== +# CLASS: OasisProfileGenerator — Sinh agent profile từ EntityNode +# ============================================================================== +# Stateful: giữ OpenAI client, Zep client, và graph_id trong suốt vòng sống. +# Được khởi tạo 1 lần trong prepare_simulation() rồi dùng để sinh toàn bộ profiles. +# ============================================================================== + class OasisProfileGenerator: """ Trình Gen Profile cho Simulation (Hệ OASIS) - + Sử dụng các Node Entity lấy được từ ZEP -> OASIS Mocks cho Simulation Agent - + Các Option Cải Tiến Tối Ưu Tích Hợp: 1. Có liên kết với Server Zep API cho bước Query Dữ liệu từ Vector Database - 2. Tập trung Mô tả Tiểu sửa (Background) (Nhấn mạnh Nghề nghiệp/Tính cách/StatusMXH/vvv) + 2. Tập trung Mô tả Tiểu sử (Background) (Nhấn mạnh Nghề nghiệp/Tính cách/StatusMXH/...) 3. Ngắt rời loại Person ra bên ngoài để nhận thức Phân Cấp Group """ - - # 16 tính cách của Con người (Quy Chuẩn) + + # 16 loại tính cách MBTI — dùng khi sinh ngẫu nhiên hoặc fallback MBTI_TYPES = [ "INTJ", "INTP", "ENTJ", "ENTP", "INFJ", "INFP", "ENFJ", "ENFP", "ISTJ", "ISFJ", "ESTJ", "ESFJ", "ISTP", "ISFP", "ESTP", "ESFP" ] - - # List mảng quốc tịch Cơ Bản + + # Danh sách quốc tịch fallback khi LLM không sinh được COUNTRIES = [ - "Vietnam", "China", "US", "UK", "Japan", "Germany", "France", + "Vietnam", "China", "US", "UK", "Japan", "Germany", "France", "Canada", "Australia", "Brazil", "India", "South Korea" ] - - # Thực thể nhận biết là người 1 mình (Độc Lập, 1 Person) + + # Loại entity được xem là CÁ NHÂN → dùng _build_individual_persona_prompt() + # LLM sẽ sinh persona theo góc nhìn 1 người cụ thể (tuổi, nghề nghiệp, MBTI riêng) INDIVIDUAL_ENTITY_TYPES = [ - "student", "alumni", "professor", "person", "publicfigure", + "student", "alumni", "professor", "person", "publicfigure", "expert", "faculty", "official", "journalist", "activist" ] - - # Thực Thể được xem là một tổ chức + + # Loại entity được xem là TỔ CHỨC/NHÓM → dùng _build_group_persona_prompt() + # LLM sẽ sinh persona như một tài khoản chính thức đại diện tổ chức + # (age=30 cố định, gender="other") GROUP_ENTITY_TYPES = [ - "university", "governmentagency", "organization", "ngo", + "university", "governmentagency", "organization", "ngo", "mediaoutlet", "company", "institution", "group", "community" ] - + def __init__( - self, + self, api_key: Optional[str] = None, base_url: Optional[str] = None, model_name: Optional[str] = None, zep_api_key: Optional[str] = None, graph_id: Optional[str] = None ): + # --- Khởi tạo OpenAI client (dùng để gọi LLM sinh persona) --- self.api_key = api_key or Config.LLM_API_KEY self.base_url = base_url or Config.LLM_BASE_URL self.model_name = model_name or Config.LLM_MODEL_NAME - + if not self.api_key: raise ValueError("Không tìm thấy LLM_API_KEY") - + self.client = OpenAI( api_key=self.api_key, base_url=self.base_url ) + + # Metadata dùng cho hệ thống tính cost API (llm_cost.py) self._runtime_metadata: Dict[str, Any] = { "component": "oasis_profile_generator", "phase": "generate_profiles", } - - # Kết nối lên trên ZEP Database Search Context + + # --- Khởi tạo Zep client (dùng để vector search tìm thêm context) --- + # graph_id được truyền vào từ simulation_manager để tất cả search đều + # trỏ đúng vào graph của simulation hiện tại self.zep_api_key = zep_api_key or Config.ZEP_API_KEY self.zep_client = None self.graph_id = graph_id - + if self.zep_api_key: try: self.zep_client = Zep(api_key=self.zep_api_key) except Exception as e: + # Không raise — Zep search là tính năng bổ sung, không bắt buộc logger.warning(f"Failed to initialize Zep client: {e}") - + + # -------------------------------------------------------------------------- + # PUBLIC: generate_profile_from_entity — Entry point sinh 1 profile đơn lẻ + # -------------------------------------------------------------------------- + def generate_profile_from_entity( - self, - entity: EntityNode, + self, + entity: EntityNode, user_id: int, use_llm: bool = True ) -> OasisAgentProfile: """ - Bắt đầu Gen Profile từ Entity được móc từ data từ zep - + Chuyển đổi 1 EntityNode thành 1 OasisAgentProfile. + + Luồng xử lý: + EntityNode + ↓ + _build_entity_context() → gom tất cả context thành 1 chuỗi + ↓ + _generate_profile_with_llm() hoặc _generate_profile_rule_based() + ↓ + OasisAgentProfile + Args: - entity: Thực thể Zep - user_id: Số ID để map (sử dụng trên OASIS) - use_llm: Chọn bật tắt xem tạo Profile có dùng gen nhân vật bằng LLM - + entity: EntityNode đọc từ Zep (có related_edges và related_nodes) + user_id: Index số nguyên, bắt đầu từ 0, bắt buộc cho OASIS + use_llm: True = gọi LLM (chậm, chất lượng cao); False = rule-based (nhanh, đơn giản) + Returns: - OasisAgentProfile + OasisAgentProfile đã điền đầy đủ thông tin """ entity_type = entity.get_entity_type() or "Entity" - - # Mức cơ bản thông tin + name = entity.name user_name = self._generate_username(name) - - # Build các Context thông tin liên quan lại + + # Tổng hợp toàn bộ ngữ cảnh về entity (attributes + edges + Zep search) context = self._build_entity_context(entity) - + if use_llm: - # Gửi Prompt lên LLM profile_data = self._generate_profile_with_llm( entity_name=name, entity_type=entity_type, @@ -249,14 +316,16 @@ class OasisProfileGenerator: context=context ) else: - # Chạy hàm Auto rule nếu LLm tắt + # Fallback thủ công — không gọi LLM, dùng template cứng profile_data = self._generate_profile_rule_based( entity_name=name, entity_type=entity_type, entity_summary=entity.summary, entity_attributes=entity.attributes ) - + + # Ghép kết quả LLM vào OasisAgentProfile + # .get(field, fallback) để an toàn nếu LLM bỏ sót field nào return OasisAgentProfile( user_id=user_id, user_name=user_name, @@ -276,63 +345,98 @@ class OasisProfileGenerator: source_entity_uuid=entity.uuid, source_entity_type=entity_type, ) - + + # -------------------------------------------------------------------------- + # PRIVATE: _generate_username — Tạo username duy nhất từ tên entity + # -------------------------------------------------------------------------- + def _generate_username(self, name: str) -> str: - """Thêm chức năng Generate Username username ngẫu nhiên""" - # Hút bỏ khoảng trống và dấu đặc biệt + """ + Tạo username hợp lệ từ tên thật. + + Quy trình: + 1. Lowercase + thay khoảng trắng bằng "_" + 2. Giữ chỉ ký tự alphanumeric và "_" + 3. Thêm suffix số ngẫu nhiên 3 chữ số để tránh trùng + + Ví dụ: "Nguyễn Văn A" → "nguyn_vn_a_392" + (ký tự Unicode bị strip vì isalnum() chỉ giữ ASCII) + """ username = name.lower().replace(" ", "_") username = ''.join(c for c in username if c.isalnum() or c == '_') - - # Chèn thêm hậu tố cho bớt đụng hàng + suffix = random.randint(100, 999) return f"{username}_{suffix}" - + + # -------------------------------------------------------------------------- + # PRIVATE: _search_zep_for_entity — Vector search song song trên Zep + # -------------------------------------------------------------------------- + # Đây là bước "tăng cường" context: tìm thêm facts và summaries liên quan + # đến entity trong Zep vector database bằng semantic search. + # + # Tại sao cần? related_edges trong EntityNode chỉ có edges kết nối trực tiếp. + # Zep vector search có thể tìm được thông tin liên quan theo ngữ nghĩa, + # kể cả những facts không có edge trực tiếp đến entity này. + # + # Tại sao cần chạy song song? + # Zep API chưa hỗ trợ tìm cả nodes lẫn edges trong 1 request → + # dùng ThreadPoolExecutor 2 workers để gọi song song, giảm latency từ 2x → 1x. + # -------------------------------------------------------------------------- + def _search_zep_for_entity(self, entity: EntityNode) -> Dict[str, Any]: """ - Dùng hỗn hợp lệnh Query DB Vector qua Zep để lấy các fact/sự kiện liên quan về 1 thực thể. - - Vì Zep chưa hỗ trợ hỗn hợp cả 2 cùng một lúc, nên cần tìm song song từ Edge và Node sau đó gộp kết quả. - + Tìm kiếm semantic trong Zep Vector DB để bổ sung context cho entity. + + Gọi 2 loại search song song: + - scope="edges": tìm các fact/quan hệ liên quan đến entity + - scope="nodes": tìm các entity khác có liên quan theo ngữ nghĩa + + Mỗi search có retry 3 lần với exponential backoff (2s → 4s → 8s). + Args: - entity: Đầu cắm Thực Thể Node - + entity: EntityNode cần tìm thêm context + Returns: - Dictionary gồm facts, node_summaries, context + Dict với: + - facts: List[str] các fact tìm được từ edge search (loại trùng với related_edges) + - node_summaries: List[str] summaries của các entity liên quan + - context: Chuỗi text tổng hợp để nhét vào LLM prompt """ import concurrent.futures - + + # Nếu không có Zep client hoặc graph_id → trả về rỗng, không làm gì if not self.zep_client: return {"facts": [], "node_summaries": [], "context": ""} - + entity_name = entity.name - + results = { "facts": [], "node_summaries": [], "context": "" } - - # Yêu cầu graph_id mới truy vấn được + if not self.graph_id: logger.debug(f"Skipping Zep search: graph_id not set") return results - + + # Query tổng quát để Zep semantic search trả về nhiều kết quả nhất comprehensive_query = f"Provide all facts, activities, relationships, and context about: {entity_name}" - + def search_edges(): - """Lookup cạnh relations - Kết hợp cơ chế retry""" + """Tìm các fact/quan hệ qua edge search — có retry.""" max_retries = 3 last_exception = None delay = 2.0 - + for attempt in range(max_retries): try: return self.zep_client.graph.search( query=comprehensive_query, graph_id=self.graph_id, - limit=30, + limit=30, # Lấy tối đa 30 facts scope="edges", - reranker="rrf" + reranker="rrf" # Reciprocal Rank Fusion — kết hợp nhiều ranking strategies ) except Exception as e: last_exception = e @@ -343,19 +447,19 @@ class OasisProfileGenerator: else: logger.debug(f"Zep Edge search entirely failed after {max_retries} attempts: {e}") return None - + def search_nodes(): - """Lookup mảng Node entity tóm tắt - Kết hợp cơ chế retry""" + """Tìm các entity liên quan qua node search — có retry.""" max_retries = 3 last_exception = None delay = 2.0 - + for attempt in range(max_retries): try: return self.zep_client.graph.search( query=comprehensive_query, graph_id=self.graph_id, - limit=20, + limit=20, # Lấy tối đa 20 node summaries scope="nodes", reranker="rrf" ) @@ -368,26 +472,25 @@ class OasisProfileGenerator: else: logger.debug(f"Zep Node search entirely failed after {max_retries} attempts: {e}") return None - + try: - # Cho chạy cả task Cạnh và Node song song + # Chạy song song 2 search — giảm thời gian chờ từ ~2T xuống ~T with concurrent.futures.ThreadPoolExecutor(max_workers=2) as executor: edge_future = executor.submit(search_edges) node_future = executor.submit(search_nodes) - - # Fetch result trả về + edge_result = edge_future.result(timeout=30) node_result = node_future.result(timeout=30) - - # Quản lý kết quả fact của các cạnh Edge - all_facts = set() + + # Xử lý kết quả edge search → trích fact string + all_facts = set() # Set để tự dedup if edge_result and hasattr(edge_result, 'edges') and edge_result.edges: for edge in edge_result.edges: if hasattr(edge, 'fact') and edge.fact: all_facts.add(edge.fact) results["facts"] = list(all_facts) - - # Quản lý kết quả tên thực thể và summary của quá trình search Node + + # Xử lý kết quả node search → trích summary và tên entity liên quan all_summaries = set() if node_result and hasattr(node_result, 'nodes') and node_result.nodes: for node in node_result.nodes: @@ -396,36 +499,69 @@ class OasisProfileGenerator: if hasattr(node, 'name') and node.name and node.name != entity_name: all_summaries.add(f"Related Entities: {node.name}") results["node_summaries"] = list(all_summaries) - - # Tổng hợp ra 1 chuỗi Context bao quanh + + # Ghép thành 1 chuỗi context để truyền vào LLM prompt context_parts = [] if results["facts"]: context_parts.append("Facts & Infomation:\n" + "\n".join(f"- {f}" for f in results["facts"][:20])) if results["node_summaries"]: context_parts.append("Related Entities:\n" + "\n".join(f"- {s}" for s in results["node_summaries"][:10])) results["context"] = "\n\n".join(context_parts) - + logger.info(f"Zep unified search completed: {entity_name}, fetched {len(results['facts'])} facts, {len(results['node_summaries'])} related nodes") - + except concurrent.futures.TimeoutError: logger.warning(f"Zep Retrieval Time-Out ({entity_name})") except Exception as e: logger.warning(f"Zep Retrieval Failed ({entity_name}): {e}") - + return results - + + # -------------------------------------------------------------------------- + # PRIVATE: _build_entity_context — Tổng hợp 4 nguồn context cho LLM + # -------------------------------------------------------------------------- + # Hàm này gom tất cả thông tin biết được về entity thành 1 chuỗi dài + # để nhét vào LLM prompt. Có 4 nguồn theo thứ tự ưu tiên: + # + # 1. attributes — thuộc tính key-value trực tiếp từ Zep node + # 2. related_edges — facts/quan hệ từ các edge đã enrich (tránh trùng lặp với nguồn 4) + # 3. related_nodes — mô tả ngắn của các entity lân cận + # 4. Zep vector search — facts bổ sung từ semantic search (lọc bỏ trùng với nguồn 2) + # -------------------------------------------------------------------------- + def _build_entity_context(self, entity: EntityNode) -> str: """ - Nối tất cả info thu được liên quan thành 1 chuỗi Context bao quanh hoàn chỉnh cho Entity - - Nó sẽ lấy: - 1. Context từ các cạnh hiện tại đã gắn Entity (Dữ liệu về relation/fact) - 2. Mô tả sơ lược thêm của các Node dính liền - 3. Cuối cùng nhồi thêm những thứ moi được từ quá trình chạy Zep search hỗn hợp bên trên + Gom tất cả thông tin có thể biết về entity thành 1 chuỗi context hoàn chỉnh. + + Output chuỗi được cắt còn tối đa 3000 ký tự trước khi nhét vào LLM prompt. + + Cấu trúc output (nếu đủ dữ liệu): + ### Entity Attributes + - key1: value1 + - key2: value2 + + ### Facts & Relationships + - A làm việc tại B + - C là sinh viên của D + + ### Related Entity Info + - **Bộ GD** (Organization): Cơ quan quản lý giáo dục... + + ### Facts retrieved via ZEP + - fact mới không trùng với phần trên + + ### Entity nodes retrieved via Zep + - summary của entity liên quan + + Args: + entity: EntityNode đã enrich với related_edges và related_nodes + + Returns: + Chuỗi context markdown dùng trong LLM prompt """ context_parts = [] - - # 1. Thu thập Attributes/Properties của node nếu có + + # --- Nguồn 1: Attributes trực tiếp của node --- if entity.attributes: attrs = [] for key, value in entity.attributes.items(): @@ -433,70 +569,85 @@ class OasisProfileGenerator: attrs.append(f"- {key}: {value}") if attrs: context_parts.append("### Entity Attributes\n" + "\n".join(attrs)) - - # 2. Add các facts và mô phỏng cạnh (Relationship/Facts) + + # --- Nguồn 2: Facts từ các edge đã enrich --- + # Lưu vào set để sau đó lọc trùng với kết quả Zep search (nguồn 4) existing_facts = set() if entity.related_edges: relationships = [] - for edge in entity.related_edges: # Khum bị giới hạn SL + for edge in entity.related_edges: fact = edge.get("fact", "") edge_name = edge.get("edge_name", "") direction = edge.get("direction", "") - + if fact: relationships.append(f"- {fact}") existing_facts.add(fact) elif edge_name: + # Nếu không có fact text, tự tạo dạng arrow diagram if direction == "outgoing": relationships.append(f"- {entity.name} --[{edge_name}]--> (Related Entity)") else: relationships.append(f"- (Related Entity) --[{edge_name}]--> {entity.name}") - + if relationships: context_parts.append("### Facts & Relationships\n" + "\n".join(relationships)) - - # 3. Kẹp chi tiết miêu tả về node anh em cạnh bên + + # --- Nguồn 3: Thông tin cơ bản của các node lân cận --- if entity.related_nodes: related_info = [] - for node in entity.related_nodes: # Không block giới hạn số lượng + for node in entity.related_nodes: node_name = node.get("name", "") node_labels = node.get("labels", []) node_summary = node.get("summary", "") - - # Bỏ nhãn mặc định khỏi string xuất ra + + # Loại bỏ label mặc định để chỉ hiện loại cụ thể custom_labels = [l for l in node_labels if l not in ["Entity", "Node"]] label_str = f" ({', '.join(custom_labels)})" if custom_labels else "" - + if node_summary: related_info.append(f"- **{node_name}**{label_str}: {node_summary}") else: related_info.append(f"- **{node_name}**{label_str}") - + if related_info: context_parts.append("### Related Entity Info\n" + "\n".join(related_info)) - - # 4. Sử dụng kết quả Query Search từ hàm zep + + # --- Nguồn 4: Vector search từ Zep (bổ sung, không trùng nguồn 2) --- zep_results = self._search_zep_for_entity(entity) - + if zep_results.get("facts"): - # Lọc bớt cặn trùng lắp: không add những Fact đã có ở mục số 2 + # Lọc bỏ facts đã có trong nguồn 2 để tránh lặp lại new_facts = [f for f in zep_results["facts"] if f not in existing_facts] if new_facts: context_parts.append("### Facts retrieved via ZEP\n" + "\n".join(f"- {f}" for f in new_facts[:15])) - + if zep_results.get("node_summaries"): context_parts.append("### Entity nodes retrieved via Zep\n" + "\n".join(f"- {s}" for s in zep_results["node_summaries"][:10])) - + return "\n\n".join(context_parts) - + + # -------------------------------------------------------------------------- + # PRIVATE: _is_individual_entity / _is_group_entity — Phân loại entity type + # -------------------------------------------------------------------------- + def _is_individual_entity(self, entity_type: str) -> bool: - """KTra và True cho các dạng người Single Person""" + """Trả về True nếu entity_type thuộc danh sách cá nhân (INDIVIDUAL_ENTITY_TYPES).""" return entity_type.lower() in self.INDIVIDUAL_ENTITY_TYPES - + def _is_group_entity(self, entity_type: str) -> bool: - """KTra xem thực thể hiện tại là Group/Media/Company ...""" + """Trả về True nếu entity_type thuộc danh sách tổ chức (GROUP_ENTITY_TYPES).""" return entity_type.lower() in self.GROUP_ENTITY_TYPES - + + # -------------------------------------------------------------------------- + # PRIVATE: _generate_profile_with_llm — Gọi LLM sinh profile, có retry và JSON repair + # -------------------------------------------------------------------------- + # Đây là hàm LLM chính. Có 3 lớp bảo vệ: + # Lớp 1: Retry 3 lần nếu API call thất bại + # Lớp 2: Sửa JSON bị cắt gãy (finish_reason == 'length') + # Lớp 3: Fallback sang rule-based nếu tất cả LLM attempts thất bại + # -------------------------------------------------------------------------- + def _generate_profile_with_llm( self, entity_name: str, @@ -506,15 +657,25 @@ class OasisProfileGenerator: context: str ) -> Dict[str, Any]: """ - Dùng LLM cấp lại Profile mô phỏng tính cách cụ thể và rõ nét nhất - - Kiểm tra đầu vào Entity type để chia nhánh: - - Cá nhân: Tạo setting miêu tả cá nhân, công việc riêng - - Tập Thể/Cơ quan/Tổ chức: Tạo Profile cho 1 tài khoản đại điện tổ chức đó + Gọi LLM để sinh profile dict với bio, persona, age, gender, mbti, ... + + Quyết định dùng prompt nào dựa vào entity_type: + - Cá nhân (student/person/...) → _build_individual_persona_prompt() + - Tổ chức (university/ngo/...) → _build_group_persona_prompt() + - Không xác định → cũng dùng group prompt + + Retry logic: + - Lần 1: temperature=0.7 + - Lần 2 (nếu lỗi): temperature=0.6 (ít ngẫu nhiên hơn → dễ parse JSON hơn) + - Lần 3 (nếu lỗi): temperature=0.5 + + Nếu hết 3 lần vẫn fail → fallback sang _generate_profile_rule_based() + + Returns: + Dict với ít nhất: {"bio": ..., "persona": ...} """ - is_individual = self._is_individual_entity(entity_type) - + if is_individual: prompt = self._build_individual_persona_prompt( entity_name, entity_type, entity_summary, entity_attributes, context @@ -524,10 +685,9 @@ class OasisProfileGenerator: entity_name, entity_type, entity_summary, entity_attributes, context ) - # Retry liên tục vòng lặp nếu LLM timeout hoặc fail max_attempts = 3 last_error = None - + for attempt in range(max_attempts): try: response = create_tracked_chat_completion( @@ -537,127 +697,150 @@ class OasisProfileGenerator: {"role": "system", "content": self._get_system_prompt(is_individual)}, {"role": "user", "content": prompt} ], - response_format={"type": "json_object"}, - temperature=0.7 - (attempt * 0.1), # Giảm tính sáng tạo ngẫu nhiên đi một chút mỗi khi fail để tăng khả năng thành công ở vòng tiếp theo + response_format={"type": "json_object"}, # Force JSON output mode + temperature=0.7 - (attempt * 0.1), # Giảm dần: 0.7 → 0.6 → 0.5 metadata=self._runtime_metadata, ) - + content = response.choices[0].message.content - - # Check LLM trả về vì sao bị kẹt lại/Dừng lại (Finish_Reason khác "stop") + + # Kiểm tra tại sao LLM dừng lại finish_reason = response.choices[0].finish_reason if finish_reason == 'length': + # LLM bị cắt ngang vì hết token → JSON có thể bị thiếu dấu đóng logger.warning(f"LLM output truncated (attempt {attempt+1}), attempting to fix...") content = self._fix_truncated_json(content) - - # Parse chép vào JSON + try: result = json.loads(content) - - # Xác minh tham số được Bot gen thành công chưa + + # Đảm bảo 2 field quan trọng nhất luôn có giá trị if "bio" not in result or not result["bio"]: result["bio"] = entity_summary[:200] if entity_summary else f"{entity_type}: {entity_name}" if "persona" not in result or not result["persona"]: result["persona"] = entity_summary or f"{entity_name} is a {entity_type}." - + return result - + except json.JSONDecodeError as je: logger.warning(f"JSON Parsing Failed (attempt {attempt+1}): {str(je)[:80]}") - - # Tool sửa lỗi JSON syntax tự chế + + # Thử sửa JSON bằng tool tự chế result = self._try_fix_json(content, entity_name, entity_type, entity_summary) if result.get("_fixed"): del result["_fixed"] return result - + last_error = je - + except Exception as e: logger.warning(f"LLM call Failed (attempt {attempt+1}): {str(e)[:80]}") last_error = e import time - time.sleep(1 * (attempt + 1)) # Exponential backoff - + time.sleep(1 * (attempt + 1)) + + # Đã hết số lần thử → fallback sang rule-based logger.warning(f"Generating profile through LLM failed totally after {max_attempts} attempts: {last_error}, switching to basic hard-code rules configs") return self._generate_profile_rule_based( entity_name, entity_type, entity_summary, entity_attributes ) - + + # -------------------------------------------------------------------------- + # PRIVATE: _fix_truncated_json — Sửa JSON bị cắt gãy do hết token + # -------------------------------------------------------------------------- + def _fix_truncated_json(self, content: str) -> str: - """Fix Output JSON bị Max_tokens đè cắt gãy""" + """ + Cố gắng vá JSON bị cắt ngang bằng cách đóng các dấu ngoặc còn thiếu. + + Thuật toán: + 1. Đếm số '{' và '}' → tính số ngoặc nhọn còn thiếu + 2. Đếm số '[' và ']' → tính số ngoặc vuông còn thiếu + 3. Nếu ký tự cuối không phải dấu đóng hợp lệ → thêm '"' để đóng string + 4. Thêm ']' và '}' còn thiếu vào cuối + + Ví dụ: + Input: '{"bio": "Nguyễn Văn A là...", "persona": "Anh ấy sinh ra' + Output: '{"bio": "Nguyễn Văn A là...", "persona": "Anh ấy sinh ra"}' + """ import re - - # Bọc ngoài + content = content.strip() - - # Điểm kiểm chứng xem dấu ngoặc được đầy đủ hay chưa + open_braces = content.count('{') - content.count('}') open_brackets = content.count('[') - content.count(']') - - # Check string xem đủ không - # Nếu phần tử cuối cùng k phải dấu câu đóng block, tự chèn vào + + # Nếu chuỗi bị cắt giữa chừng trong một string value → đóng string lại if content and content[-1] not in '",}]': - # Ngoặc cho đít chuỗi content += '"' - - # Ngoặc block + + # Đóng array và object còn thiếu content += ']' * open_brackets content += '}' * open_braces - + return content - + + # -------------------------------------------------------------------------- + # PRIVATE: _try_fix_json — Sửa JSON lỗi syntax theo nhiều cấp độ + # -------------------------------------------------------------------------- + def _try_fix_json(self, content: str, entity_name: str, entity_type: str, entity_summary: str = "") -> Dict[str, Any]: - """Thử Fix nội dung JSON""" + """ + Thử sửa JSON hỏng theo 7 bước từ nhẹ đến nặng: + + 1. _fix_truncated_json() — đóng ngoặc còn thiếu + 2. Extract block {} lớn nhất bằng regex + 3. Escape newline trong string values + 4. json.loads() — lần 1 + 5. Strip control chars (\\x00-\\x1f) → json.loads() — lần 2 + 6. Regex rescue: lụm bio và persona riêng lẻ (dù JSON hỏng hoàn toàn) + 7. Trả về dict tối thiểu nếu không cứu được gì + + Returns: + Dict (có thể rất đơn giản) với "_fixed": True nếu cứu được + """ import re - - # 1. Bọc json trước + + # Bước 1: Đóng ngoặc content = self._fix_truncated_json(content) - - # 2. Extract block lớn + + # Bước 2: Tìm block JSON lớn nhất trong chuỗi json_match = re.search(r'\{[\s\S]*\}', content) if json_match: json_str = json_match.group() - - # 3. Clean lại các chuỗi xuống dòng - # Regex lôi code ra ngoài + + # Bước 3: Escape newline ẩn trong string values def fix_string_newlines(match): s = match.group(0) - # Escape code line chuyển cho space cho an toàn s = s.replace('\n', ' ').replace('\r', ' ') - # Chém các khoảng trống còn dư quá gắt s = re.sub(r'\s+', ' ', s) return s - - # Khớp lại nội dung + json_str = re.sub(r'"[^"\\]*(?:\\.[^"\\]*)*"', fix_string_newlines, json_str) - - # 4. Bắt đầu json parse + + # Bước 4: Parse lần 1 try: result = json.loads(json_str) result["_fixed"] = True return result except json.JSONDecodeError as e: - # 5. Phá lấu clean xóa nếu còn bị lỗi Control char ẩn (0x00 đến 0x1f ...) + # Bước 5: Strip control characters ẩn rồi parse lại try: - # Chém control character json_str = re.sub(r'[\x00-\x1f\x7f-\x9f]', ' ', json_str) - # Gọt lại khoảng trống dư json_str = re.sub(r'\s+', ' ', json_str) result = json.loads(json_str) result["_fixed"] = True return result except: pass - - # 6. Rescue lấy các property còn lại mót ra từ đóng hỗn độn + + # Bước 6: Regex rescue — tìm bio và persona trong đống hỗn độn bio_match = re.search(r'"bio"\s*:\s*"([^"]*)"', content) - persona_match = re.search(r'"persona"\s*:\s*"([^"]*)', content) # Bị cắt khúc thì ráng chịu - + persona_match = re.search(r'"persona"\s*:\s*"([^"]*)', content) + bio = bio_match.group(1) if bio_match else (entity_summary[:200] if entity_summary else f"{entity_type}: {entity_name}") persona = persona_match.group(1) if persona_match else (entity_summary or f"{entity_name} is a {entity_type}.") - - # Lụm mót được data xịn thì mark là fix thành công + if bio_match or persona_match: logger.info(f"Successfully extracted partial info from corrupted JSON") return { @@ -665,22 +848,37 @@ class OasisProfileGenerator: "persona": persona, "_fixed": True } - - # 7. Failed sạch, quăng cái khung mặc định ra + + # Bước 7: Không cứu được gì → trả về dict tối thiểu (không có _fixed → caller biết dùng fallback) logger.warning(f"Failed to fix JSON, returning basic structured data") return { "bio": entity_summary[:200] if entity_summary else f"{entity_type}: {entity_name}", "persona": entity_summary or f"{entity_name} is a {entity_type}." } - + + # -------------------------------------------------------------------------- + # PRIVATE: _get_system_prompt — System prompt cho LLM + # -------------------------------------------------------------------------- + def _get_system_prompt(self, is_individual: bool) -> str: - """Lấy prompt cho hệ thống""" - - # base_prompt = "You are an expert in generating social media user personas. Generate detailed and realistic personas for public opinion simulation to recreate existing real-world conditions to the greatest extent possible. You must return a valid JSON format; all string values must not contain unescaped line breaks. Use Vietnamese." - + """ + Trả về system prompt chung cho LLM (hiện tại không phân biệt individual/group). + Nhấn mạnh: trả về JSON hợp lệ, không có newline thô trong string values. + """ base_prompt = "Bạn là chuyên gia tạo hồ sơ người dùng mạng xã hội. Hãy tạo các nhân vật chi tiết và chân thực phục vụ cho việc mô phỏng dư luận, nhằm tái hiện tối đa các tình huống thực tế hiện có. Phải trả về định dạng JSON hợp lệ; tất cả các giá trị chuỗi không được chứa ký tự xuống dòng chưa được xử lý (unescaped). Sử dụng tiếng Việt." return base_prompt - + + # -------------------------------------------------------------------------- + # PRIVATE: _build_individual_persona_prompt / _build_group_persona_prompt + # -------------------------------------------------------------------------- + # Hai hàm này tạo user prompt gửi cho LLM. + # Cấu trúc gồm: thông tin entity + context + yêu cầu 8 fields JSON cụ thể. + # + # Điểm khác biệt chính: + # - Individual: persona ~2000 từ với ký ức cá nhân, tuổi thực, gender male/female + # - Group: persona ~2000 từ với ký ức tổ chức, age=30 cố định, gender="other" + # -------------------------------------------------------------------------- + def _build_individual_persona_prompt( self, entity_name: str, @@ -689,45 +887,16 @@ class OasisProfileGenerator: entity_attributes: Dict[str, Any], context: str ) -> str: - """Tạo prompt nhân vật chi tiết cho thực thể cá nhân""" - + """ + Tạo user prompt cho entity CÁ NHÂN. + + Yêu cầu LLM sinh JSON 8 fields: + bio, persona, age (int), gender ("male"/"female"), mbti, country, profession, interested_topics + + Context được cắt tối đa 3000 ký tự để tránh vượt context window của LLM. + """ attrs_str = json.dumps(entity_attributes, ensure_ascii=False) if entity_attributes else "Không có" context_str = context[:3000] if context else "Không có ngữ cảnh bổ sung" - -# return f"""Generate a detailed social media user persona for the entity, recreating existing real-world conditions to the greatest extent possible. - -# Entity Name: {entity_name} -# Entity Type: {entity_type} -# Entity Summary: {entity_summary} -# Entity Attributes: {attrs_str} - -# Context Information: -# {context_str} - -# Please generate a JSON containing the following fields: - -# 1. bio: Social media biography, 200 characters. -# 2. persona: Detailed persona description (2000 words of plain text), which must include: -# - Basic information (age, occupation, educational background, location) -# - Background (significant experiences, connection to the event, social relationships) -# - Personality traits (MBTI type, core personality, emotional expression style) -# - Social media behavior (posting frequency, content preferences, interaction style, linguistic characteristics) -# - Stance and views (attitude toward the topic, content that might provoke or move them) -# - Unique features (catchphrases, special experiences, personal hobbies) -# - Personal memory (a vital part of the persona, describing the individual's connection to the event and their existing actions/reactions) -# 3. age: Age as a number (must be an integer) -# 4. gender: Gender, must be in English: "male" or "female" -# 5. mbti: MBTI type (e.g., INTJ, ENFP, etc.) -# 6. country: Country (use Vietnamese, e.g., "Việt Nam") -# 7. profession: Occupation -# 8. interested_topics: An array of interested topics - -# IMPORTANT: -# - All field values must be strings or numbers; do not use line breaks. -# - The 'persona' must be a coherent block of text description. -# - Use Vietnamese (except for the 'gender' field, which must be English male/female). -# - Content must remain consistent with the entity information. -# - 'age' must be a valid integer; 'gender' must be "male" or "female".""" return f"""Tạo hồ sơ người dùng mạng xã hội chi tiết cho thực thể, tái hiện tối đa các tình huống thực tế hiện có. @@ -740,7 +909,7 @@ Thông tin ngữ cảnh: {context_str} Vui lòng tạo JSON bao gồm các trường sau: - + 1. bio: Tiểu sử mạng xã hội, 200 ký tự. 2. persona: Mô tả nhân vật chi tiết (văn bản thuần túy khoảng 2000 từ), cần bao gồm: - Thông tin cơ bản (tuổi, nghề nghiệp, trình độ học vấn, nơi ở) @@ -756,7 +925,7 @@ Vui lòng tạo JSON bao gồm các trường sau: 6. country: Quốc gia (sử dụng tiếng Việt, ví dụ: "Việt Nam") 7. profession: Nghề nghiệp 8. interested_topics: Mảng các chủ đề quan tâm - + QUAN TRỌNG: - Tất cả giá trị các trường phải là chuỗi hoặc số, không sử dụng ký tự xuống dòng. - 'persona' phải là một đoạn mô tả văn bản mạch lạc. @@ -773,47 +942,17 @@ QUAN TRỌNG: entity_attributes: Dict[str, Any], context: str ) -> str: - """Tạo prompt chi tiết cho tài khoản đại diện tổ chức/nhóm""" - + """ + Tạo user prompt cho entity TỔ CHỨC/NHÓM. + + Khác biệt so với individual prompt: + - age: cố định 30 (tuổi ảo cho tài khoản tổ chức) + - gender: cố định "other" + - persona nhấn mạnh "ký ức tổ chức" và phong cách phát ngôn chính thức + """ attrs_str = json.dumps(entity_attributes, ensure_ascii=False) if entity_attributes else "None" context_str = context[:3000] if context else "No additional context" -# return f"""Generate a detailed social media account persona for an organization/group entity, recreating existing real-world conditions to the greatest extent possible. - -# Entity Name: {entity_name} -# Entity Type: {entity_type} -# Entity Summary: {entity_summary} -# Entity Attributes: {attrs_str} - -# Context Information: -# {context_str} - -# Please generate a JSON containing the following fields: - -# 1. bio: Official account biography, 200 characters, professional and appropriate. -# 2. persona: Detailed account setting description (2000 words of plain text), which must include: -# - Basic information (formal name, nature of the organization, establishment background, primary functions) -# - Account positioning (account type, target audience, core functions) -# - Communication style (linguistic characteristics, common expressions, taboo topics) -# - Content characteristics (content types, posting frequency, active time periods) -# - Stance and attitude (official stance on core topics, handling of controversies) -# - Special notes (persona of the group represented, operational habits) -# - Organizational memory (a vital part of the persona, describing the organization's connection to the event and its existing actions/reactions) -# 3. age: Fixed at 30 (virtual age for an organizational account) -# 4. gender: Fixed as "other" (representing non-individual accounts) -# 5. mbti: MBTI type used to describe the account's style (e.g., ISTJ for rigorous/conservative) -# 6. country: Country (use Vietnamese, e.g., "Việt Nam") -# 7. profession: Description of organizational functions -# 8. interested_topics: An array of focused fields/areas of interest - -# IMPORTANT: -# - All field values must be strings or numbers; null values are not allowed. -# - 'persona' must be a coherent block of text description; do not use line breaks. -# - Use Vietnamese (except for the 'gender' field, which must be the English string "other"). -# - 'age' must be the integer 30; 'gender' must be the string "other". -# - The account's tone and discourse must strictly align with its institutional identity and positioning. -# """ - return f"""Tạo thiết lập tài khoản mạng xã hội chi tiết cho thực thể tổ chức/nhóm, tái hiện tối đa các tình huống thực tế hiện có. Tên thực thể: {entity_name} @@ -825,7 +964,7 @@ Thông tin ngữ cảnh: {context_str} Vui lòng tạo JSON bao gồm các trường sau: - + 1. bio: Tiểu sử tài khoản chính thức, 200 ký tự, chuyên nghiệp và chuẩn mực. 2. persona: Mô tả chi tiết thiết lập tài khoản (văn bản thuần túy khoảng 2000 từ), cần bao gồm: - Thông tin cơ bản về tổ chức (tên chính thức, tính chất tổ chức, bối cảnh thành lập, chức năng chính) @@ -841,7 +980,7 @@ Vui lòng tạo JSON bao gồm các trường sau: 6. country: Quốc gia (sử dụng tiếng Việt, ví dụ: "Việt Nam") 7. profession: Mô tả chức năng của tổ chức 8. interested_topics: Mảng các lĩnh vực quan tâm - + QUAN TRỌNG: - Tất cả giá trị các trường phải là chuỗi hoặc số, không cho phép giá trị null. - 'persona' phải là một đoạn mô tả văn bản mạch lạc, không sử dụng ký tự xuống dòng. @@ -849,7 +988,14 @@ QUAN TRỌNG: - 'age' phải là số nguyên 30, 'gender' phải là chuỗi "other". - Phát ngôn và giọng điệu của tài khoản phải phù hợp tuyệt đối với định vị danh tính và đặc thù của tổ chức. """ - + + # -------------------------------------------------------------------------- + # PRIVATE: _generate_profile_rule_based — Fallback không dùng LLM + # -------------------------------------------------------------------------- + # Khi LLM fail hoàn toàn, hàm này trả về profile "cứng" dựa trên entity_type. + # Chất lượng thấp hơn LLM nhiều nhưng đảm bảo pipeline không bị block. + # -------------------------------------------------------------------------- + def _generate_profile_rule_based( self, entity_name: str, @@ -857,11 +1003,18 @@ QUAN TRỌNG: entity_summary: str, entity_attributes: Dict[str, Any] ) -> Dict[str, Any]: - """Sử dụng rule để tạo Profile cơ bản khi dự phòng""" - - # Phân nhánh theo loại thực thể để tạo Profile thủ công + """ + Sinh profile bằng template cứng theo loại entity — không gọi LLM. + + Ưu điểm: nhanh, không tốn token, không bao giờ fail + Nhược điểm: generic, thiếu cá tính, không phản ánh context thực của entity + + Các loại được xử lý riêng: student/alumni, publicfigure/expert/faculty, + mediaoutlet, university/governmentagency/ngo/organization. + Tất cả loại khác → profile mặc định chung. + """ entity_type_lower = entity_type.lower() - + if entity_type_lower in ["student", "alumni"]: return { "bio": f"{entity_type} with interests in academics and social issues.", @@ -873,45 +1026,45 @@ QUAN TRỌNG: "profession": "Student", "interested_topics": ["Education", "Social Issues", "Technology"], } - + elif entity_type_lower in ["publicfigure", "expert", "faculty"]: return { "bio": f"Expert and thought leader in their field.", "persona": f"{entity_name} is a recognized {entity_type.lower()} who shares insights and opinions on important matters. They are known for their expertise and influence in public discourse.", "age": random.randint(35, 60), "gender": random.choice(["male", "female"]), - "mbti": random.choice(["ENTJ", "INTJ", "ENTP", "INTP"]), + "mbti": random.choice(["ENTJ", "INTJ", "ENTP", "INTP"]), # Nhóm MBTI thiên về tư duy/lãnh đạo "country": random.choice(self.COUNTRIES), "profession": entity_attributes.get("occupation", "Expert"), "interested_topics": ["Politics", "Economics", "Culture & Society"], } - + elif entity_type_lower in ["mediaoutlet", "socialmediaplatform"]: return { "bio": f"Official account for {entity_name}. News and updates.", "persona": f"{entity_name} is a media entity that reports news and facilitates public discourse. The account shares timely updates and engages with the audience on current events.", - "age": 30, # Tuổi ảo của cơ quan/tổ chức - "gender": "other", # Cơ quan dùng "other" - "mbti": "ISTJ", # Phong cách tổ chức: nghiêm túc bảo thủ + "age": 30, # Tuổi ảo cố định cho tổ chức + "gender": "other", # Không phải cá nhân + "mbti": "ISTJ", # Nghiêm túc, thận trọng, bảo thủ — phù hợp media chính thống "country": "Việt Nam", "profession": "Media", "interested_topics": ["General News", "Current Events", "Public Affairs"], } - + elif entity_type_lower in ["university", "governmentagency", "ngo", "organization"]: return { "bio": f"Official account of {entity_name}.", "persona": f"{entity_name} is an institutional entity that communicates official positions, announcements, and engages with stakeholders on relevant matters.", - "age": 30, # Tuổi ảo của cơ quan/tổ chức - "gender": "other", # Cơ quan dùng "other" - "mbti": "ISTJ", # Phong cách tổ chức: nghiêm túc bảo thủ + "age": 30, + "gender": "other", + "mbti": "ISTJ", "country": "Việt Nam", "profession": entity_type, "interested_topics": ["Public Policy", "Community", "Official Announcements"], } - + else: - # Profile mặc định (Fallback default) + # Fallback hoàn toàn chung — dùng cho mọi loại không match trên return { "bio": entity_summary[:150] if entity_summary else f"{entity_type}: {entity_name}", "persona": entity_summary or f"{entity_name} is a {entity_type.lower()} participating in social discussions.", @@ -922,11 +1075,30 @@ QUAN TRỌNG: "profession": entity_type, "interested_topics": ["General", "Social Issues"], } - + def set_graph_id(self, graph_id: str): - """Lưu lại Graph ID để dùng cho việc tra cứu Zep""" + """Cập nhật graph_id sau khi khởi tạo (dùng khi graph_id chưa biết lúc __init__).""" self.graph_id = graph_id - + + # -------------------------------------------------------------------------- + # PUBLIC: generate_profiles_from_entities — Sinh hàng loạt song song + # -------------------------------------------------------------------------- + # Đây là hàm được gọi bởi simulation_manager.prepare_simulation() (Giai đoạn 2). + # Dùng ThreadPoolExecutor để chạy parallel_count threads đồng thời. + # + # Tại sao cần parallel? + # Mỗi profile cần 1-3 LLM API call + 1 Zep search call → ~5-15 giây/profile. + # Với 50 profiles: sequential = 250-750s, parallel (3 threads) = ~100-250s. + # + # Thứ tự profile: + # ThreadPoolExecutor.as_completed() không đảm bảo thứ tự → dùng mảng profiles[idx] + # được cấp phát trước để đảm bảo user_id đúng với entity ban đầu. + # + # Realtime output: + # Mỗi khi 1 thread hoàn thành, gọi save_profiles_realtime() để ghi file ngay. + # Frontend có thể đọc file này để hiển thị tiến độ thực tế. + # -------------------------------------------------------------------------- + def generate_profiles_from_entities( self, entities: List[EntityNode], @@ -941,28 +1113,34 @@ QUAN TRỌNG: project_id: Optional[str] = None, ) -> List[OasisAgentProfile]: """ - Khởi tạo hàng loạt các Agent Profile từ các thực thể (Hỗ trợ Gen đa luồng song song) - + Sinh hàng loạt profiles từ danh sách entities — chạy song song. + + Đảm bảo: + - profiles[i] luôn tương ứng với entities[i] (thứ tự không bị xáo trộn) + - Nếu 1 entity fail → vẫn tạo fallback profile, không block cả batch + - File được ghi realtime sau mỗi profile hoàn thành + Args: - entities: Danh sách thực thể - use_llm: Có sử dụng LLM để tạo tính cách chi tiết hay không - progress_callback: Hàm CallBack báo tiến độ (current, total, message) - graph_id: Đưa Graph ID vào để Zep retrieval thêm nhiều ngữ cảnh phong phú - parallel_count: Số luồng song song, mặc định 5 - realtime_output_path: Đường dẫn lưu file realtime (Gen ra đứa nào auto save đứa đó luôn) - output_platform: Format lưu trữ output ("reddit" hoạc "twitter") - metadata_platform: Nền tảng dùng cho metadata cost - + entities: Danh sách EntityNode từ ZepEntityReader + use_llm: True = gọi LLM; False = rule-based + progress_callback: callback(current, total, message) để báo tiến độ + graph_id: Zep graph ID cho vector search + parallel_count: Số threads chạy đồng thời (mặc định 5) + realtime_output_path: Đường dẫn file để ghi realtime (None = không ghi) + output_platform: "reddit" (JSON) hoặc "twitter" (CSV) + metadata_platform: Tên platform cho log cost API + simulation_id, project_id: Metadata cho cost tracking + Returns: - Danh sách Profile Agent + List[OasisAgentProfile] với len == len(entities), không có None """ import concurrent.futures from threading import Lock - - # Lưu Graph ID lại cho Zep xử lý search + if graph_id: self.graph_id = graph_id + # Cập nhật metadata cho cost tracking self._runtime_metadata = { "component": "oasis_profile_generator", "phase": "generate_profiles", @@ -970,32 +1148,29 @@ QUAN TRỌNG: "project_id": project_id, "platform": metadata_platform, } - + total = len(entities) - profiles = [None] * total # Cấp trước 1 mảng để giữ đúng thứ tự Index - completed_count = [0] # Phải dùng List để closure của các Sub Thread update được - lock = Lock() - - # Hàm con hỗ trợ việc ghi realtime file trong Thread + profiles = [None] * total # Pre-allocate để giữ đúng thứ tự index + completed_count = [0] # List thay vì int để closure có thể mutate + lock = Lock() # Bảo vệ completed_count và file write khỏi race condition + def save_profiles_realtime(): - """Lưu file json ngay lập tức khi profile được tạo mới thành công""" + """Ghi file ngay lập tức khi có profile mới (thread-safe qua lock).""" if not realtime_output_path: return - + with lock: - # Lọc ra những profile đã làm xong existing_profiles = [p for p in profiles if p is not None] if not existing_profiles: return - + try: if output_platform == "reddit": - # Cấu trúc dành cho định dạng Reddit profiles_data = [p.to_reddit_format() for p in existing_profiles] with open(realtime_output_path, 'w', encoding='utf-8') as f: json.dump(profiles_data, f, ensure_ascii=False, indent=2) else: - # Cấu trúc dành cho định dạng Twitter (CSV) + # Twitter: CSV format import csv profiles_data = [p.to_twitter_format() for p in existing_profiles] if profiles_data: @@ -1006,26 +1181,28 @@ QUAN TRỌNG: writer.writerows(profiles_data) except Exception as e: logger.warning(f"Failed to save profile in realtime: {e}") - + def generate_single_profile(idx: int, entity: EntityNode) -> tuple: - """Hàm Worker gen từng profile riêng lẻ""" + """ + Worker function chạy trong thread riêng cho mỗi entity. + + Returns: + (idx, profile, error_message_or_None) + """ entity_type = entity.get_entity_type() or "Entity" - + try: profile = self.generate_profile_from_entity( entity=entity, user_id=idx, use_llm=use_llm ) - - # Print output để nhìn trực tiếp Log terminal self._print_generated_profile(entity.name, entity_type, profile) - return idx, profile, None - + except Exception as e: logger.error(f"Failed to generate profile for entity {entity.name}: {str(e)}") - # Rơi vào tạo Profile dự phòng (Fallback) + # Tạo profile tối thiểu để pipeline không bị block fallback_profile = OasisAgentProfile( user_id=idx, user_name=self._generate_username(entity.name), @@ -1036,52 +1213,53 @@ QUAN TRỌNG: source_entity_type=entity_type, ) return idx, fallback_profile, str(e) - + logger.info(f"Start parallel profile generation for {total} entities (Concurrency: {parallel_count})...") print(f"\n{'='*60}") print(f"Starting Agent Profile Generation - Total {total} entities, concurrency: {parallel_count}") print(f"{'='*60}\n") - - # Chạy đa luồng thread pool + + # Chạy ThreadPoolExecutor với parallel_count workers đồng thời with concurrent.futures.ThreadPoolExecutor(max_workers=parallel_count) as executor: - # Giao Task + # Submit tất cả tasks cùng lúc future_to_entity = { executor.submit(generate_single_profile, idx, entity): (idx, entity) for idx, entity in enumerate(entities) } - - # Thu gom kết quả + + # Thu kết quả theo thứ tự hoàn thành (không phải thứ tự submit) for future in concurrent.futures.as_completed(future_to_entity): idx, entity = future_to_entity[future] entity_type = entity.get_entity_type() or "Entity" - + try: result_idx, profile, error = future.result() - profiles[result_idx] = profile - + profiles[result_idx] = profile # Đặt vào đúng index, không theo thứ tự hoàn thành + with lock: completed_count[0] += 1 current = completed_count[0] - - # Ghi file Realtime + + # Ghi file ngay sau khi có thêm 1 profile save_profiles_realtime() - + if progress_callback: progress_callback( - current, - total, + current, + total, f"Completed {current}/{total}: {entity.name} ({entity_type})" ) - + if error: logger.warning(f"[{current}/{total}] Entity {entity.name} applied fallback profile due to error: {error}") else: logger.info(f"[{current}/{total}] Automatically generated profile for: {entity.name} ({entity_type})") - + except Exception as e: logger.error(f"Error handling profile for entity {entity.name}: {str(e)}") with lock: completed_count[0] += 1 + # Emergency fallback — đảm bảo slot không bị None profiles[idx] = OasisAgentProfile( user_id=idx, user_name=self._generate_username(entity.name), @@ -1091,22 +1269,28 @@ QUAN TRỌNG: source_entity_uuid=entity.uuid, source_entity_type=entity_type, ) - # Ghi file Realtime file (Dù là profile xài fallback) save_profiles_realtime() - + print(f"\n{'='*60}") print(f"Profile generation complete! Successfully created {len([p for p in profiles if p])} Agents") print(f"{'='*60}\n") - + return profiles - + + # -------------------------------------------------------------------------- + # PRIVATE: _print_generated_profile — In log profile ra terminal + # -------------------------------------------------------------------------- + def _print_generated_profile(self, entity_name: str, entity_type: str, profile: OasisAgentProfile): - """Xuất thông tin Profile vưa gen ra Terminal để review dễ dàng (Kéo dài không bị gãy log)""" + """ + In thông tin profile vừa tạo ra terminal để review. + + Dùng print() thay vì logger để tránh logger truncate persona dài. + Format rõ ràng với separator và các section riêng biệt. + """ separator = "-" * 70 - - # Xây cấu trúc Log topics_str = ', '.join(profile.interested_topics) if profile.interested_topics else 'Không có' - + output_lines = [ f"\n{separator}", f"[Generated] {entity_name} ({entity_type})", @@ -1125,12 +1309,13 @@ QUAN TRỌNG: f"Chủ đề quan tâm: {topics_str}", separator ] - - output = "\n".join(output_lines) - - # Chỉ in ra Console bằng lệnh print (Logger sẽ làm rối và có thể bị truncate) - print(output) - + + print("\n".join(output_lines)) + + # -------------------------------------------------------------------------- + # PUBLIC: save_profiles — Dispatcher ghi file theo nền tảng + # -------------------------------------------------------------------------- + def save_profiles( self, profiles: List[OasisAgentProfile], @@ -1138,158 +1323,165 @@ QUAN TRỌNG: platform: str = "reddit" ): """ - Ghi file Profile xuống thư mục (Cấu trúc file tuỳ thuộc vào nền tảng) - - Định dạng mặc định của Framework OASIS yêu cầu: - - Twitter: Định dạng file CSV - - Reddit: Định dạng file JSON - + Ghi toàn bộ profiles ra file theo định dạng của từng nền tảng. + + Dispatcher: gọi đúng hàm save dựa vào platform: + - "twitter" → _save_twitter_csv() (OASIS yêu cầu CSV) + - "reddit" (mặc định) → _save_reddit_json() (OASIS yêu cầu JSON) + Args: - profiles: Danh sách Profile - file_path: Đường dẫn lưu file - platform: Tên nền tảng ("reddit" hoặc "twitter") + profiles: Danh sách profiles cần lưu + file_path: Đường dẫn file output + platform: "reddit" hoặc "twitter" """ if platform == "twitter": self._save_twitter_csv(profiles, file_path) else: self._save_reddit_json(profiles, file_path) - + + # -------------------------------------------------------------------------- + # PRIVATE: _save_twitter_csv — Lưu profiles theo chuẩn CSV của OASIS Twitter + # -------------------------------------------------------------------------- + def _save_twitter_csv(self, profiles: List[OasisAgentProfile], file_path: str): """ - Lưu Profile hệ Twitter ở định dạng CSV (Bám vào yêu cầu kỹ thuật do OASIS ban hành) - - Các trường bắt buộc để tương thích OASIS Twitter File CSV: - - user_id: Mã định danh ID (Từ 0 theo Index mảng) - - name: Tên thật của Agent đó - - username: Tên Alias/Tài khoản xài trong hệ thống - - user_char: Bản nháp Setting cụ thể truyền vào System Prompt LLM, định hình mọi ý nghĩ/phát ngôn - - description: Bản Bio gắn ngoài hiển thị cho các User khác thấy (Ngắn gọn) - - Sự khác biệt user_char và description: - - user_char: Data nội bộ chỉ Gen AI thấy (Giống prompt điều khiển não) - - description: Public Info đưa lên trang cá nhân + Ghi profiles ra file CSV theo đúng chuẩn OASIS Twitter framework. + + Cấu trúc CSV (5 cột cố định): + ┌──────────┬──────────┬──────────┬──────────────────────┬─────────────────┐ + │ user_id │ name │ username │ user_char │ description │ + ├──────────┼──────────┼──────────┼──────────────────────┼─────────────────┤ + │ 0 │ Tên thật │ alias_xx │ bio + " " + persona │ bio (public) │ + └──────────┴──────────┴──────────┴──────────────────────┴─────────────────┘ + + Lưu ý quan trọng: + - user_char = bio + persona → đây là system prompt bí mật của LLM agent + - description = bio → thông tin hiển thị công khai + - Tất cả newline trong string phải được strip (CSV không chịu được) """ import csv - - # Check đuôi file có nhầm thành json không + + # Tự sửa đuôi file nếu nhầm if not file_path.endswith('.csv'): file_path = file_path.replace('.json', '.csv') - + with open(file_path, 'w', newline='', encoding='utf-8') as f: writer = csv.writer(f) - - # Khởi tạo Header theo chuẩn file OASIS headers = ['user_id', 'name', 'username', 'user_char', 'description'] writer.writerow(headers) - - # Xuất từng dòng Dữ liệu Profile + for idx, profile in enumerate(profiles): - # user_char: Nhân cách tổng (bio + persona) - Thả cho Prompt System LLM + # Ghép bio + persona thành user_char (system prompt của agent) user_char = profile.bio if profile.persona and profile.persona != profile.bio: user_char = f"{profile.bio} {profile.persona}" - # Làm sạch dấu newline để nhét vào dòng CSV + # Strip newline vì CSV không xử lý được multiline trong 1 cell user_char = user_char.replace('\n', ' ').replace('\r', ' ') - - # description: Thông tin Bio hiển thị công khai mạng xã hội + description = profile.bio.replace('\n', ' ').replace('\r', ' ') - + row = [ - idx, # user_id: ID bắt đầu từ 0 - profile.name, # name: Tên thực - profile.user_name, # username: Tên định danh - user_char, # user_char: Mô tả ẩn của Bot - description # description: Bảng mô tả Công khai + idx, + profile.name, + profile.user_name, + user_char, + description ] writer.writerow(row) - + logger.info(f"Saved {len(profiles)} Twitter Profiles to {file_path} (OASIS CSV Format)") - + + # -------------------------------------------------------------------------- + # PRIVATE: _normalize_gender — Chuẩn hoá gender về enum OASIS chấp nhận + # -------------------------------------------------------------------------- + def _normalize_gender(self, gender: Optional[str]) -> str: """ - Biên dịch, chuẩn hóa cột Gender về đúng dạng mà OASIS engine chấp nhận - - OASIS quy định buộc xài enum: male, female, other + Chuyển đổi mọi dạng gender string về 3 giá trị OASIS chấp nhận: "male", "female", "other". + + Xử lý: + - Tiếng Việt: "nam" → "male", "nữ" → "female", "tổ chức" → "other" + - Tiếng Trung: "男" → "male", "女" → "female", "机构" → "other" + - Không có giá trị → "other" (an toàn nhất) + - Không match bất kỳ → "other" (fallback) """ if not gender: return "other" - + gender_lower = gender.lower().strip() - - # Mapping các Keyword + gender_map = { - "男": "male", - "女": "female", - "机构": "other", - "其他": "other", + # Tiếng Việt "nam": "male", "nữ": "female", "tổ chức": "other", - # Giữ nguyên Tiếng Anh Default + "khác": "other", + # Tiếng Anh "male": "male", "female": "female", "other": "other", } - + return gender_map.get(gender_lower, "other") - + + # -------------------------------------------------------------------------- + # PRIVATE: _save_reddit_json — Lưu profiles theo chuẩn JSON của OASIS Reddit + # -------------------------------------------------------------------------- + def _save_reddit_json(self, profiles: List[OasisAgentProfile], file_path: str): """ - Lưu Profile hệ Reddit bằng JSON (Bám vào yêu cầu kỹ thuật do OASIS ban hành) - - Format cấu trúc dựa tương đồng với hàm to_reddit_format(). - Luôn luôn phải có thuộc tính user_id, KEY QUAN TRỌNG ĐỂ HỖ TRỢ HÀM agent_graph.get_agent() MAP CÁC PROFILE !!! - - Các field bắt buộc: - - user_id: User ID dạng Int - - username: ID Account - - name: Tên hiển thị - - bio: Thông tin hiển thị Bio cá nhân - - persona: Prompt Settings điều khiển Bot nội bộ - - age: Tuổi (Int) - - gender: "male", "female", hoặc "other" - - mbti: Kiểu loại nhóm MBTI - - country: Quốc gia Country + Ghi profiles ra file JSON theo đúng chuẩn OASIS Reddit framework. + + Fields bắt buộc (OASIS sẽ crash nếu thiếu): + - user_id: int — dùng bởi agent_graph.get_agent() để map agent + - username: str — tên tài khoản (không có khoảng trắng) + - name: str — tên hiển thị + - bio: str — thông tin công khai + - persona: str — system prompt bí mật của agent + - karma: int — ảnh hưởng đến visibility của post/comment + - age, gender, mbti, country: — dùng trong logic agent nếu OASIS cần + + Lưu ý: gender được normalize qua _normalize_gender() trước khi ghi. """ data = [] for idx, profile in enumerate(profiles): - # Parse Format chung với hàm class to_reddit_format() item = { - "user_id": profile.user_id if profile.user_id is not None else idx, # Quan trọng: Bắt buộc kèm "user_id" + "user_id": profile.user_id if profile.user_id is not None else idx, "username": profile.user_name, "name": profile.name, - "bio": profile.bio[:150] if profile.bio else f"{profile.name}", + "bio": profile.bio[:150] if profile.bio else f"{profile.name}", # Cap ở 150 ký tự "persona": profile.persona or f"{profile.name} is a participant in social discussions.", "karma": profile.karma if profile.karma else 1000, "created_at": profile.created_at, - # Fix bù tham số ảo cho các properties bị trống + # Fallback values cho các field tùy chọn để tránh null "age": profile.age if profile.age else 30, "gender": self._normalize_gender(profile.gender), "mbti": profile.mbti if profile.mbti else "ISTJ", "country": profile.country if profile.country else "Việt Nam", } - - # Cột Tuỳ chọn + if profile.profession: item["profession"] = profile.profession if profile.interested_topics: item["interested_topics"] = profile.interested_topics - + data.append(item) - + with open(file_path, 'w', encoding='utf-8') as f: json.dump(data, f, ensure_ascii=False, indent=2) - + logger.info(f"Saved {len(profiles)} Reddit Profiles to {file_path} (JSON Config File - with user_id mapped)") - - # Giữ lại Function name cũ để hệ thống vẫn tương thích backward. + + # -------------------------------------------------------------------------- + # DEPRECATED: save_profiles_to_json — Tên cũ, giữ để tương thích ngược + # -------------------------------------------------------------------------- + def save_profiles_to_json( self, profiles: List[OasisAgentProfile], file_path: str, platform: str = "reddit" ): - """[Deprecated - Hết hạn dùng] KHUYÊN DÙNG LỆNH save_profiles() THAY VÌ PHƯƠNG THỨC NÀY""" + """[Deprecated] Dùng save_profiles() thay thế. Giữ lại để không break code cũ.""" logger.warning("save_profiles_to_json is Deprecated. Use save_profiles method instead!") self.save_profiles(profiles, file_path, platform) - diff --git a/backend/app/services/report_agent.py b/backend/app/services/report_agent.py index e8074195..c6d0fd6b 100644 --- a/backend/app/services/report_agent.py +++ b/backend/app/services/report_agent.py @@ -984,7 +984,7 @@ Nhiệm vụ của bạn là: 3. [Tính nhất quán về Ngôn ngữ - Nội dung trích dẫn phải được dịch sang ngôn ngữ báo cáo] - Nội dung trả về từ các công cụ có thể chứa tiếng Anh hoặc hỗn hợp tiếng Việt và tiếng Anh. - - Nếu yêu cầu mô phỏng và tài liệu gốc bằng tiếng Việt, báo cáo phải được viết hoàn toàn bằng tiếng Việt + - **Báo cáo phải được viết hoàn toàn bằng tiếng Việt.** - Khi bạn trích dẫn nội dung tiếng Anh hoặc hỗn hợp từ công cụ, bạn phải dịch sang tiếng Việt lưu loát trước khi đưa vào báo cáo - Giữ nguyên ý nghĩa gốc khi dịch và đảm bảo cách diễn đạt tự nhiên - Quy tắc này áp dụng cho cả văn bản chính và nội dung trong khối trích dẫn (định dạng >) diff --git a/backend/app/services/simulation_config_generator.py b/backend/app/services/simulation_config_generator.py index 7dc9d92d..27bb04d5 100644 --- a/backend/app/services/simulation_config_generator.py +++ b/backend/app/services/simulation_config_generator.py @@ -615,7 +615,7 @@ Hãy tạo JSON cấu hình thời gian. Ví dụ Format như sau: {{ - "total_simulation_hours": 72, + "total_simulation_hours": 72, "minutes_per_round": 60, "agents_per_hour_min": 5, "agents_per_hour_max": 50, diff --git a/backend/app/services/simulation_manager.py b/backend/app/services/simulation_manager.py index 5cd07bf4..c643741c 100644 --- a/backend/app/services/simulation_manager.py +++ b/backend/app/services/simulation_manager.py @@ -21,16 +21,32 @@ from .simulation_config_generator import SimulationConfigGenerator, SimulationPa logger = get_logger('mirofish.simulation') +# ============================================================================== +# ENUM: SimulationStatus — Máy trạng thái (State Machine) của một Simulation +# ============================================================================== +# Mỗi simulation đi qua các trạng thái theo thứ tự: +# +# CREATED ──► PREPARING ──► READY ──► RUNNING ──► COMPLETED +# │ │ +# └──────────────────────┴──► FAILED +# │ +# └──► PAUSED ──► RUNNING (tiếp tục) +# │ +# └──► STOPPED (người dùng chủ động dừng) +# +# Lưu ý: FAILED là trạng thái cuối cùng không thể tiếp tục — phải tạo simulation mới. +# ============================================================================== + class SimulationStatus(str, Enum): """Trạng thái hiện tại của quá trình mô phỏng""" - CREATED = "created" # Đã khởi tạo - PREPARING = "preparing" # Đang chuẩn bị (chuẩn bị dữ liệu/profile) - READY = "ready" # Đã sẵn sàng chạy - RUNNING = "running" # Đang xử lý giả lập - PAUSED = "paused" # Tạm dừng - STOPPED = "stopped" # Hệ thống mô phỏng bị người dùng chủ động dừng lại - COMPLETED = "completed" # Quá trình mô phỏng kết thúc tự nhiên một cách thành công - FAILED = "failed" # Bị lỗi hệ thống gián đoạn + CREATED = "created" # Vừa được tạo, chưa làm gì cả + PREPARING = "preparing" # Đang chạy prepare_simulation() — đọc entity, sinh profile, tạo config + READY = "ready" # prepare_simulation() hoàn thành — sẵn sàng bấm nút chạy OASIS + RUNNING = "running" # OASIS subprocess đang chạy, agents đang hành động + PAUSED = "paused" # Người dùng ra lệnh pause qua IPC — OASIS tạm dừng vòng lặp + STOPPED = "stopped" # Người dùng chủ động dừng — OASIS bị kill, kết quả còn lại được giữ + COMPLETED = "completed" # OASIS chạy hết số vòng (max_rounds) và thoát thành công + FAILED = "failed" # Exception không xử lý được — pipeline bị gián đoạn class PlatformType(str, Enum): @@ -39,50 +55,74 @@ class PlatformType(str, Enum): REDDIT = "reddit" +# ============================================================================== +# DATACLASS: SimulationState — "Hộp đen" lưu toàn bộ thông tin một simulation +# ============================================================================== +# Đây là đối tượng trung tâm được đọc/ghi liên tục trong suốt vòng đời simulation. +# Nó được lưu ở hai nơi đồng thời: +# 1. RAM: `SimulationManager._simulations` dict (tra cứu nhanh O(1)) +# 2. Disk: `//state.json` (bền vững qua restart) +# ============================================================================== + @dataclass class SimulationState: """Class lưu trữ cấu trúc Dữ liệu/Trạng thái của một lượt mô phỏng""" - simulation_id: str - project_id: str - graph_id: str - - # Cờ trạng thái bật/tắt nền tảng chạy + + # --- Định danh --- + simulation_id: str # UUID dạng "sim_<12 ký tự hex>", ví dụ: "sim_a3f8c21b004e" + project_id: str # ID của Project chứa simulation này (quản lý nhiều sim cùng lúc) + graph_id: str # ID của Zep Graph — xác định kho Entity nào sẽ dùng làm agent + + # --- Công tắc nền tảng --- + # Cho phép chọn chạy 1 hoặc cả 2 nền tảng. Nếu cả hai False → pipeline sẽ bỏ qua bước sinh profile. enable_twitter: bool = True enable_reddit: bool = True - - # Current status + + # --- Trạng thái vòng đời --- + # Giá trị mặc định CREATED; được cập nhật qua _save_simulation_state() mỗi khi chuyển giai đoạn. status: SimulationStatus = SimulationStatus.CREATED - - # Dữ liệu thu thập / thống kê của Preparing Phase - entities_count: int = 0 - profiles_count: int = 0 - entity_types: List[str] = field(default_factory=list) - - # Thông tin các nội dung cấu hình mà LLM đã tự động tạo - config_generated: bool = False - config_reasoning: str = "" - - # Dữ liệu cập nhật theo thời gian thực (Runtime Phase) - current_round: int = 0 - twitter_status: str = "not_started" - reddit_status: str = "not_started" - - # Nhãn Timestamp lịch sử + + # --- Kết quả thống kê của giai đoạn PREPARING --- + entities_count: int = 0 # Số entity đọc được từ Zep (sau khi lọc) + profiles_count: int = 0 # Số agent profile đã sinh thành công + entity_types: List[str] = field(default_factory=list) # Danh sách loại entity (ví dụ: ["Person", "Organization"]) + + # --- Kết quả sinh config --- + config_generated: bool = False # True sau khi simulation_config.json được ghi ra đĩa + config_reasoning: str = "" # Chuỗi giải thích của LLM tại sao chọn các tham số này + + # --- Dữ liệu runtime (cập nhật theo thời gian thực trong khi RUNNING) --- + current_round: int = 0 # Vòng hiện tại OASIS đang chạy (0-indexed) + twitter_status: str = "not_started" # Trạng thái chi tiết của luồng Twitter + reddit_status: str = "not_started" # Trạng thái chi tiết của luồng Reddit + + # --- Timestamp lịch sử --- + # Lưu dạng ISO 8601 (ví dụ: "2025-01-15T10:30:45.123456") để dễ parse lại created_at: str = field(default_factory=lambda: datetime.now().isoformat()) updated_at: str = field(default_factory=lambda: datetime.now().isoformat()) - - # Lịch sử thông báo Lỗi (nếu có để render trả về frontend) + + # --- Lỗi --- + # None nếu không có lỗi; nếu có chứa message dạng string để hiển thị cho frontend error: Optional[str] = None - + def to_dict(self) -> Dict[str, Any]: - """Tạo thành Dictionary đầy đủ nhất (Dùng cho việc lưu cấu hình local cho hệ thống bên trong đọc)""" + """ + Xuất toàn bộ state thành dict đầy đủ. + + Dùng khi: + - Ghi ra file state.json (để khôi phục lại khi server restart) + - Truyền nội bộ giữa các service Python + + Khác với to_simple_dict(): bản này có đầy đủ mọi trường, + kể cả các trường nặng như config_reasoning và timestamps. + """ return { "simulation_id": self.simulation_id, "project_id": self.project_id, "graph_id": self.graph_id, "enable_twitter": self.enable_twitter, "enable_reddit": self.enable_reddit, - "status": self.status.value, + "status": self.status.value, # Enum → string (ví dụ: "preparing") "entities_count": self.entities_count, "profiles_count": self.profiles_count, "entity_types": self.entity_types, @@ -95,9 +135,14 @@ class SimulationState: "updated_at": self.updated_at, "error": self.error, } - + def to_simple_dict(self) -> Dict[str, Any]: - """Tạo thành Dictionary bao hàm các thông số vắn tắt hơn (Dùng cho API response trả về Client - Frontend)""" + """ + Xuất state thành dict gọn nhẹ cho API response trả về client/frontend. + + Lược bỏ các trường nội bộ nặng (config_reasoning, timestamps, platform statuses...) + để giảm payload JSON và không để lộ thông tin debug ra ngoài. + """ return { "simulation_id": self.simulation_id, "project_id": self.project_id, @@ -111,69 +156,122 @@ class SimulationState: } +# ============================================================================== +# CLASS: SimulationManager — Trung tâm điều phối toàn bộ vòng đời simulation +# ============================================================================== +# SimulationManager là singleton (dùng chung 1 instance) được Flask khởi tạo lúc boot. +# Nó KHÔNG chạy OASIS trực tiếp — việc đó do SimulationRunner đảm nhiệm. +# Nhiệm vụ chính: chuẩn bị dữ liệu (prepare) và theo dõi trạng thái (state tracking). +# ============================================================================== + class SimulationManager: """ Kịch bản Quản lý trung tâm của tính năng Mô Phỏng - + Luồng thiết lập cốt lõi: 1. Trích xuất nhóm các Thực Thể (Entity) được định nghĩa sẵn trong hệ thống lưu trữ Graph của Zep. 2. Chế lại thành các hồ sơ Profile thiết lập tiêu chuẩn của OASIS framework (Agent) 3. Trao quyền cho sức mạnh mô hình LLM tự đánh giá số liệu rồi tự sinh ra cấu hình cài đặt cho quá trình mô phỏng 4. Cài đặt các thư mục và tập tin tương ứng, phục vụ sẵn sàng để những Script lập trình riêng (pre-set script) có thể khai thác sử dụng. """ - - # Nơi chứa thư mục chứa Dữ liệu mô phỏng Local + + # Đường dẫn gốc nơi lưu tất cả dữ liệu simulation (mỗi sim có subfolder riêng). + # __file__ = .../backend/app/services/simulation_manager.py + # → join 2 cấp "../.." → .../backend/ + # → join "uploads/simulations" → .../backend/uploads/simulations/ SIMULATION_DATA_DIR = os.path.join( - os.path.dirname(__file__), + os.path.dirname(__file__), '../../uploads/simulations' ) - + def __init__(self): - # Đảm bảo môi trường file data đã được set up + # Tạo thư mục gốc nếu chưa tồn tại (exist_ok=True → không báo lỗi nếu đã có) os.makedirs(self.SIMULATION_DATA_DIR, exist_ok=True) - - # Biến dictionary ở mức Application theo dõi trạng thái simulation qua Cache RAM. + + # Cache RAM: dict ánh xạ simulation_id → SimulationState + # Mục đích: tránh đọc file state.json mỗi lần có request API → tra cứu O(1) + # Nhược điểm: mất dữ liệu khi server restart → có disk backup bù lại self._simulations: Dict[str, SimulationState] = {} - + + # -------------------------------------------------------------------------- + # PRIVATE HELPERS — Quản lý file/thư mục và cơ chế hai lớp cache + # -------------------------------------------------------------------------- + def _get_simulation_dir(self, simulation_id: str) -> str: - """Lấy trả về các đường dẫn thư mục gốc tương ứng với Simulation ID""" + """ + Trả về đường dẫn thư mục chứa dữ liệu của một simulation cụ thể. + Tự động tạo thư mục nếu chưa tồn tại (lazy creation). + + Cấu trúc thư mục sau khi READY: + uploads/simulations// + ├── state.json ← trạng thái vòng đời + ├── simulation_config.json ← tham số OASIS do LLM sinh + ├── reddit_profiles.json ← agent profiles cho Reddit + └── twitter_profiles.csv ← agent profiles cho Twitter + """ sim_dir = os.path.join(self.SIMULATION_DATA_DIR, simulation_id) - os.makedirs(sim_dir, exist_ok=True) + os.makedirs(sim_dir, exist_ok=True) # Idempotent: gọi nhiều lần vẫn an toàn return sim_dir - + def _save_simulation_state(self, state: SimulationState): - """Bật tính năng lưu state định dạng JSON ra ổ cứng""" + """ + Ghi state xuống disk VÀ cập nhật vào RAM cache đồng thời. + + Đây là cơ chế "write-through cache": + - Mỗi khi state thay đổi, cả RAM lẫn file đều được cập nhật ngay lập tức. + - Đảm bảo: nếu server crash giữa chừng, state.json vẫn phản ánh trạng thái mới nhất. + + Tự động cập nhật updated_at thành thời điểm hiện tại trước khi ghi. + """ sim_dir = self._get_simulation_dir(state.simulation_id) state_file = os.path.join(sim_dir, "state.json") - + + # Stamp thời gian ghi cuối cùng state.updated_at = datetime.now().isoformat() - + + # Ghi JSON ra đĩa (ensure_ascii=False để Unicode/tiếng Việt không bị escape) with open(state_file, 'w', encoding='utf-8') as f: json.dump(state.to_dict(), f, ensure_ascii=False, indent=2) - + + # Cập nhật đồng thời vào RAM cache self._simulations[state.simulation_id] = state - + def _load_simulation_state(self, simulation_id: str) -> Optional[SimulationState]: - """Load ngược lại data của tiến trình Mô Phỏng thông qua tệp cấu hình JSON""" + """ + Đọc state của một simulation — ưu tiên dùng RAM cache, fallback về disk. + + Chiến lược "cache-aside": + 1. Kiểm tra RAM cache trước (nhanh, không I/O) + 2. Nếu không có trong RAM → tìm file state.json trên disk + 3. Nếu tìm thấy → deserialize thành SimulationState object, nạp vào RAM cache + 4. Nếu không tìm thấy file → trả về None + + Trường hợp None: simulation_id không hợp lệ hoặc chưa từng được tạo. + """ + # --- Bước 1: Kiểm tra RAM cache --- if simulation_id in self._simulations: return self._simulations[simulation_id] - + + # --- Bước 2: Tìm trên disk --- sim_dir = self._get_simulation_dir(simulation_id) state_file = os.path.join(sim_dir, "state.json") - + if not os.path.exists(state_file): - return None - + return None # Simulation này chưa từng tồn tại + + # --- Bước 3: Deserialize từ JSON thành dataclass --- with open(state_file, 'r', encoding='utf-8') as f: data = json.load(f) - + + # Khôi phục từng field, dùng .get() với giá trị mặc định để tương thích ngược + # khi có field mới được thêm vào SimulationState sau này state = SimulationState( simulation_id=simulation_id, project_id=data.get("project_id", ""), graph_id=data.get("graph_id", ""), enable_twitter=data.get("enable_twitter", True), enable_reddit=data.get("enable_reddit", True), - status=SimulationStatus(data.get("status", "created")), + status=SimulationStatus(data.get("status", "created")), # string → Enum entities_count=data.get("entities_count", 0), profiles_count=data.get("profiles_count", 0), entity_types=data.get("entity_types", []), @@ -186,10 +284,15 @@ class SimulationManager: updated_at=data.get("updated_at", datetime.now().isoformat()), error=data.get("error"), ) - + + # --- Bước 4: Nạp vào RAM cache để lần sau không cần đọc file nữa --- self._simulations[simulation_id] = state return state - + + # -------------------------------------------------------------------------- + # PUBLIC: create_simulation — Khởi tạo một simulation mới (trạng thái CREATED) + # -------------------------------------------------------------------------- + def create_simulation( self, project_id: str, @@ -198,20 +301,27 @@ class SimulationManager: enable_reddit: bool = True, ) -> SimulationState: """ - Khởi tạo môi trường ảo / mới cho Mô Phỏng - + Tạo một simulation mới với trạng thái ban đầu CREATED. + + Chỉ tạo metadata — KHÔNG đọc entity, KHÔNG gọi LLM. + Sau khi gọi hàm này, cần gọi tiếp prepare_simulation() để chuẩn bị dữ liệu. + Args: project_id: Mã ID của Project (Quản lý cấp đầu vào) - graph_id: Đồ thị ID tương ứng lấy bên Zep - enable_twitter: Công tắc (Bật/Tắt) luồng giả lập Twitter - enable_reddit: Công tắc (Bật/Tắt) luồng giả lập Reddit - + graph_id: Đồ thị ID tương ứng lấy bên Zep — quyết định kho Entity nào được dùng + enable_twitter: Có sinh profile và chạy mô phỏng trên Twitter không? + enable_reddit: Có sinh profile và chạy mô phỏng trên Reddit không? + Returns: - Đối tượng Class SimulationState + SimulationState với status=CREATED, chưa có entity/profile/config. """ import uuid + + # Tạo ID ngắn gọn, dễ đọc: "sim_" + 12 ký tự hex ngẫu nhiên + # Ví dụ: "sim_a3f8c21b004e" — đủ ngẫu nhiên để tránh va chạm trong phạm vi project simulation_id = f"sim_{uuid.uuid4().hex[:12]}" - + + # Khởi tạo object state với giá trị mặc định state = SimulationState( simulation_id=simulation_id, project_id=project_id, @@ -220,12 +330,17 @@ class SimulationManager: enable_reddit=enable_reddit, status=SimulationStatus.CREATED, ) - + + # Ghi ra disk ngay lập tức để đảm bảo bền vững self._save_simulation_state(state) logger.info(f"Created simulation: {simulation_id}, project={project_id}, graph={graph_id}") - + return state - + + # -------------------------------------------------------------------------- + # PUBLIC: prepare_simulation — Pipeline chuẩn bị dữ liệu (3 giai đoạn chính) + # -------------------------------------------------------------------------- + def prepare_simulation( self, simulation_id: str, @@ -238,104 +353,151 @@ class SimulationManager: ) -> SimulationState: """ Giai đoạn chuẩn bị dữ liệu tạo giả lập mô phỏng (Tiến trình Automation 100%) - - Các bước diễn ra: - 1. Gọi lấy các cụm Entity (thực thể) và bộ lọt (filter) từ Zep Graph API - 2. Tự động khởi tạo hàng loạt Agent Profile chạy OASIS tương ứng với Entity (Hỗ trợ gọi AI LLM để làm mượt văn bản / tăng tốc chạy song song) - 3. Hỏi và bắt bot LLM suy luận ra hệ tham số setting thông minh nhất (thời gian mô phỏng rò rỉ, hệ số tần suất nói chuyện hoạt động ...) - 4. In ra các file cấu hình và JSON của profile để hệ thống dễ đọc - 5. Copy nguyên các Scripts chuẩn được cấu hình sẵn (preset) ném vào thư mục để chạy - + + Hàm này chạy trong một background thread (được Flask gọi không đồng bộ). + Progress được báo cáo liên tục về frontend qua progress_callback. + + 3 giai đoạn chính: + ┌─────────────────────────────────────────────────────────────────────────┐ + │ Giai đoạn 1: ĐỌC ENTITY từ Zep Graph │ + │ ZepEntityReader → lọc theo defined_entity_types → FilteredEntities │ + │ Nếu không có entity nào → FAILED (không thể tiếp tục) │ + ├─────────────────────────────────────────────────────────────────────────┤ + │ Giai đoạn 2: SINH AGENT PROFILE │ + │ OasisProfileGenerator → gọi LLM song song (ThreadPoolExecutor) │ + │ → reddit_profiles.json + twitter_profiles.csv │ + ├─────────────────────────────────────────────────────────────────────────┤ + │ Giai đoạn 3: SINH SIMULATION CONFIG │ + │ SimulationConfigGenerator → LLM phân tích → simulation_config.json │ + └─────────────────────────────────────────────────────────────────────────┘ + Args: - simulation_id: Mã ID của chu trình giả lập - simulation_requirement: Chuỗi text từ người dùng yêu cầu mô phỏng gì (gửi cho config sinh cấu hình) - document_text: Nội dung file Raw nguyên thủy (Cho LLM đánh giá context bối cảnh ban đầu) - defined_entity_types: Dánh sách các Entity Model có sẵn do Zep định nghĩa (Option) - use_llm_for_profiles: Toggle tính năng sử dụng mô hình LLM để buff thêm chi tiết cài đặt con Bot - progress_callback: Hàm callback update log progress (chuyển về màn hình Frontend view) format (stage, progress, message) - parallel_profile_count: Giới hạn concurrent threading chạy LLM gọi profile (Default là 3 luồng cùng lúc để làm nhanh hơn) - + simulation_id: ID của simulation đã tạo bằng create_simulation() + simulation_requirement: Yêu cầu mô phỏng từ người dùng (gửi cho LLM sinh config) + document_text: Nội dung văn bản gốc (ngữ cảnh để LLM hiểu chủ đề đang mô phỏng) + defined_entity_types: Danh sách loại entity cần lọc (None = lấy tất cả) + use_llm_for_profiles: True = gọi LLM để làm giàu persona; False = dùng rule-based + progress_callback: Hàm nhận (stage: str, percent: int, message: str, **kwargs) + parallel_profile_count: Số thread LLM chạy song song khi sinh profile (mặc định 3) + Returns: - Class cài đặt Data - SimulationState + SimulationState với status=READY (thành công) hoặc status=FAILED (thất bại) + + Raises: + ValueError: Nếu simulation_id không tồn tại + Exception: Re-raise mọi lỗi sau khi đã cập nhật state sang FAILED """ + # Load state hiện tại — phải tồn tại trước (đã gọi create_simulation()) state = self._load_simulation_state(simulation_id) if not state: raise ValueError(f"Giả lập với ID: {simulation_id} không tồn tại") - + try: + # --- Chuyển trạng thái sang PREPARING ngay từ đầu --- + # Quan trọng: ghi xuống disk trước để nếu crash sau đó, + # frontend không bị thấy state "CREATED" mãi mãi state.status = SimulationStatus.PREPARING self._save_simulation_state(state) - + + # Lấy đường dẫn thư mục để ghi các file output sim_dir = self._get_simulation_dir(simulation_id) - - # ========== Giai đoạn 1: Kết nối lấy Node Data Entity và Sàng lọc ========== + + # ========================================================================== + # GIAI ĐOẠN 1: Kết nối Zep và lọc Entity + # ========================================================================== + # ZepEntityReader đọc TẤT CẢ nodes trong graph (có thể hàng nghìn nodes), + # sau đó lọc chỉ giữ lại các node có label nằm trong defined_entity_types. + # Kết quả trả về là FilteredEntities chứa danh sách EntityNode đã enrich. + # Retry: 3 lần với exponential backoff (2s → 4s → 8s). + # ========================================================================== if progress_callback: progress_callback("reading", 0, "Connecting to Zep Graph data store...") - + reader = ZepEntityReader() - + if progress_callback: progress_callback("reading", 30, "Extracting Node data from graph...") - + + # filter_defined_entities: đọc nodes → lọc theo label → enrich với edges + # enrich_with_edges=True → mỗi EntityNode sẽ có related_edges & related_nodes + # (cần thiết để LLM hiểu mối quan hệ giữa các entity khi sinh persona) filtered = reader.filter_defined_entities( graph_id=state.graph_id, defined_entity_types=defined_entity_types, enrich_with_edges=True ) - + + # Ghi số liệu thống kê vào state state.entities_count = filtered.filtered_count state.entity_types = list(filtered.entity_types) - + if progress_callback: progress_callback( - "reading", 100, + "reading", 100, f"Extraction complete, filtered entity count: {filtered.filtered_count} empty entities", current=filtered.filtered_count, total=filtered.filtered_count ) - + + # --- Early exit: không có entity nào → không thể chạy simulation --- + # Lý do có thể: graph chưa được populate, hoặc defined_entity_types quá hẹp if filtered.filtered_count == 0: state.status = SimulationStatus.FAILED state.error = "No valid entities extracted for simulation. Please check if the Graph was generated properly with valid text." self._save_simulation_state(state) - return state - - # ========== Giai đoạn 2: Bắt đầu sinh Agent Profiles cho OASIS ========== + return state # Trả về sớm, không raise exception + + # ========================================================================== + # GIAI ĐOẠN 2: Sinh Agent Profile cho từng Entity + # ========================================================================== + # Mỗi EntityNode → 1 OasisAgentProfile (có user_id, bio, persona, mbti, ...) + # Nếu use_llm=True: gọi LLM để viết bio/persona phong phú hơn + # Nếu use_llm=False: dùng rule-based (nhanh hơn, ít tốn token hơn) + # Chạy parallel_profile_count threads đồng thời để tăng tốc. + # ========================================================================== total_entities = len(filtered.entities) - + if progress_callback: progress_callback( - "generating_profiles", 0, + "generating_profiles", 0, "Ready for AI Generation process...", current=0, total=total_entities ) - - # Gửi mã graph_id để bộ Profile có thể fetch thêm tài liệu nếu model cần lục vấn sâu + + # Khởi tạo generator — truyền graph_id để nó có thể search Zep thêm nếu cần generator = OasisProfileGenerator(graph_id=state.graph_id) - + + # Wrapper chuyển đổi format callback: + # generate_profiles_from_entities gọi: callback(current_count, total, entity_name) + # Nhưng prepare_simulation cần format: callback(stage, percent, message, ...) def profile_progress(current, total, msg): if progress_callback: progress_callback( - "generating_profiles", - int(current / total * 100), + "generating_profiles", + int(current / total * 100), # Tính phần trăm hoàn thành msg, current=current, total=total, item_name=msg ) - - # Khai báo đường dẫn tạm để AI lưu Real-time kết quả (Đặt ưu tiên Platform Reddit JSON làm chuẩn) + + # --- Xác định đường dẫn real-time output --- + # Real-time output: trong khi LLM đang sinh profile, file được ghi dần dần + # (không đợi hết rồi mới ghi một lần) → frontend có thể theo dõi tiến độ + # Ưu tiên Reddit vì đó là format JSON chuẩn; Twitter dùng CSV (chỉ khi Reddit tắt) realtime_output_path = None realtime_platform = "reddit" if state.enable_reddit: realtime_output_path = os.path.join(sim_dir, "reddit_profiles.json") realtime_platform = "reddit" elif state.enable_twitter: + # Chỉ đến đây nếu Reddit bị tắt hẳn realtime_output_path = os.path.join(sim_dir, "twitter_profiles.csv") realtime_platform = "twitter" - # Nền tảng dùng cho metadata cost. + # --- Xác định platform metadata (để tính cost API) --- + # "parallel" = cả hai nền tảng đều bật → profiles được dùng cho cả hai if state.enable_twitter and state.enable_reddit: runtime_platform = "parallel" elif state.enable_twitter: @@ -343,75 +505,90 @@ class SimulationManager: elif state.enable_reddit: runtime_platform = "reddit" else: - runtime_platform = None - + runtime_platform = None # Không nên xảy ra trong thực tế + + # Gọi hàm sinh profile chính (blocking — chờ tất cả thread hoàn thành) profiles = generator.generate_profiles_from_entities( entities=filtered.entities, use_llm=use_llm_for_profiles, progress_callback=profile_progress, - graph_id=state.graph_id, # Để tìm kiếm Zep Search Index - parallel_count=parallel_profile_count, # Số dòng luồng Async - realtime_output_path=realtime_output_path, # Lưu log thời gian thực - output_platform=realtime_platform, # Đuôi file xuất - metadata_platform=runtime_platform, + graph_id=state.graph_id, # Để tìm kiếm trong Zep Search Index nếu cần + parallel_count=parallel_profile_count, # Số thread LLM chạy đồng thời + realtime_output_path=realtime_output_path, # Ghi từng profile vừa xong ngay + output_platform=realtime_platform, # Định dạng file (json/csv) + metadata_platform=runtime_platform, # Cho logging cost simulation_id=simulation_id, project_id=state.project_id, ) - + state.profiles_count = len(profiles) - - # Backup lại kết quả Profile (Twitter xuất ra text CSV, Reddit thì bắt buộc JSON cho cấu trúc OASIS) - # Reddit đã được render đồng thời ở block trên nhưng đây là re-save toàn bộ + + # --- Lưu lần cuối toàn bộ profiles ra file --- + # (real-time output ở trên có thể bị thiếu nếu một thread crash) + # Đây là lần ghi đảm bảo file hoàn chỉnh 100% if progress_callback: progress_callback( - "generating_profiles", 95, + "generating_profiles", 95, "Compressing Profile data...", current=total_entities, total=total_entities ) - + if state.enable_reddit: + # Reddit: JSON array các object profile (cấu trúc OASIS yêu cầu) generator.save_profiles( profiles=profiles, file_path=os.path.join(sim_dir, "reddit_profiles.json"), platform="reddit" ) - + if state.enable_twitter: - # Riêng Twitter với code Script base OAsis của họ yêu cầu CSV format + # Twitter: CSV với header row (cấu trúc OASIS gốc yêu cầu) generator.save_profiles( profiles=profiles, file_path=os.path.join(sim_dir, "twitter_profiles.csv"), platform="twitter" ) - + if progress_callback: progress_callback( - "generating_profiles", 100, + "generating_profiles", 100, f"Done, created {len(profiles)} Profiles", current=len(profiles), total=len(profiles) ) - - # ========== Giai đoạn 3: Uỷ thác cho LLM phân tích và xuất tham số mô phỏng ========== + + # ========================================================================== + # GIAI ĐOẠN 3: Sinh simulation_config.json bằng LLM + # ========================================================================== + # SimulationConfigGenerator gọi LLM với: + # - simulation_requirement (yêu cầu của user) + # - document_text (văn bản nguồn) + # - entities (danh sách entity) + # LLM trả về SimulationParameters gồm: time_config, event_config, agent_configs + # Toàn bộ được serialize thành JSON và ghi ra simulation_config.json. + # File này sau đó được đọc bởi run_parallel_simulation.py khi OASIS chạy. + # ========================================================================== if progress_callback: progress_callback( - "generating_config", 0, + "generating_config", 0, "Analyzing input requirements...", current=0, total=3 ) - + config_generator = SimulationConfigGenerator() - + if progress_callback: progress_callback( - "generating_config", 30, + "generating_config", 30, "LLM Bot is generating configuration...", current=1, total=3 ) - + + # generate_config: gọi LLM tạo time_config → event_config → agent_configs + # (3 lần gọi LLM riêng biệt, mỗi lần có retry với exponential backoff) sim_params = config_generator.generate_config( simulation_id=simulation_id, project_id=state.project_id, @@ -422,106 +599,175 @@ class SimulationManager: enable_twitter=state.enable_twitter, enable_reddit=state.enable_reddit ) - + if progress_callback: progress_callback( - "generating_config", 70, + "generating_config", 70, "Saving Config parameters...", current=2, total=3 ) - - # Lưu file cứng simulation_config.json + + # Ghi simulation_config.json ra thư mục của simulation config_path = os.path.join(sim_dir, "simulation_config.json") with open(config_path, 'w', encoding='utf-8') as f: - f.write(sim_params.to_json()) - + f.write(sim_params.to_json()) # to_json() đã xử lý serialize Enum, nested dict + + # Đánh dấu config đã sinh xong và lưu lý do LLM giải thích state.config_generated = True state.config_reasoning = sim_params.generation_reasoning - + if progress_callback: progress_callback( - "generating_config", 100, + "generating_config", 100, "Configuration Generation complete", current=3, total=3 ) - - # Lưu ý kiến trúc: Các scripts thao tác thực thi vẫn để gốc ở `backend/scripts/`, SẼ KHÔNG CẦN chép đè sang folder Project - # Tại thời gian Khởi chạy, `simulation_runner` sẽ nạp base chạy thẳng từ folder `scripts/` đó. - - # Cập nhật status + + # ========================================================================== + # LƯU Ý KIẾN TRÚC QUAN TRỌNG: + # Scripts OASIS (run_parallel_simulation.py, v.v.) vẫn để GỐC tại + # `backend/scripts/` — KHÔNG copy sang thư mục simulation. + # Khi SimulationRunner.start() chạy, nó sẽ gọi thẳng script gốc và + # truyền --config để script đọc đúng dữ liệu simulation. + # Lý do: tránh code duplication, dễ update script mà không ảnh hưởng + # các simulation đã chuẩn bị sẵn. + # ========================================================================== + + # --- Chuyển trạng thái sang READY — pipeline hoàn tất --- state.status = SimulationStatus.READY self._save_simulation_state(state) - + logger.info(f"Finished simulation preparation phase for ID: {simulation_id}, " f"Total entities={state.entities_count}, Created profiles={state.profiles_count}") - + return state - + except Exception as e: + # --- Xử lý lỗi toàn cục: bất kỳ exception nào cũng → FAILED --- + # Ghi lại traceback đầy đủ để debug, sau đó re-raise để Flask xử lý tiếp logger.error(f"Error occurred during Simulation preparation (Sim ID: {simulation_id}), ERROR CODE: {str(e)}") import traceback logger.error(traceback.format_exc()) + + # Cập nhật state FAILED với message lỗi để frontend hiển thị state.status = SimulationStatus.FAILED state.error = str(e) self._save_simulation_state(state) - raise - + + raise # Re-raise để Flask API handler nhận và trả HTTP 500 + + # -------------------------------------------------------------------------- + # PUBLIC UTILITIES — Các hàm đọc/truy vấn trạng thái + # -------------------------------------------------------------------------- + def get_simulation(self, simulation_id: str) -> Optional[SimulationState]: - """Đọc và lấy State hiện tại của Simulator""" + """ + Lấy trạng thái hiện tại của một simulation. + Dùng cache-aside (RAM trước, disk fallback) nên rất nhanh. + Trả về None nếu simulation_id không tồn tại. + """ return self._load_simulation_state(simulation_id) - + def list_simulations(self, project_id: Optional[str] = None) -> List[SimulationState]: - """Liệt kê toàn bộ danh sách các Mô Phỏng (Simulations) đã khởi tạo""" + """ + Liệt kê tất cả simulations, có thể lọc theo project_id. + + Duyệt qua toàn bộ thư mục con trong SIMULATION_DATA_DIR, + load state.json của từng cái, rồi lọc theo project_id nếu có yêu cầu. + + Args: + project_id: Nếu None → trả về tất cả; nếu có → chỉ trả về của project đó + + Returns: + Danh sách SimulationState (có thể rỗng nếu chưa có simulation nào) + """ simulations = [] - + if os.path.exists(self.SIMULATION_DATA_DIR): for sim_id in os.listdir(self.SIMULATION_DATA_DIR): - # Loại bỏ các folder/file rác do hệ điều hành sinh ra (ví dụ: .DS_Store của macOS) hoặc không phải thư mục + # Bỏ qua các file ẩn (ví dụ: .DS_Store của macOS, .gitkeep) + # và các entry không phải thư mục (ví dụ: file rác nào đó) sim_path = os.path.join(self.SIMULATION_DATA_DIR, sim_id) if sim_id.startswith('.') or not os.path.isdir(sim_path): continue - + state = self._load_simulation_state(sim_id) if state: + # Lọc theo project_id nếu được chỉ định if project_id is None or state.project_id == project_id: simulations.append(state) - + return simulations - + def get_profiles(self, simulation_id: str, platform: str = "reddit") -> List[Dict[str, Any]]: - """Lấy/Tải dữ liệu Agent Profile do AI sinh ra dựa theo nền tảng mạng xã hội""" + """ + Đọc và trả về danh sách agent profiles đã sinh cho một simulation. + + Chỉ hoạt động sau khi prepare_simulation() hoàn thành (status=READY trở lên). + Trả về [] nếu file chưa tồn tại (simulation chưa READY hoặc nền tảng bị tắt). + + Args: + simulation_id: ID của simulation + platform: "reddit" → đọc reddit_profiles.json | "twitter" → đọc twitter_profiles.json + (Twitter thực ra lưu .csv nhưng hàm này chỉ dùng cho Reddit JSON) + + Returns: + List của các dict profile (mỗi dict là 1 agent) + """ state = self._load_simulation_state(simulation_id) if not state: raise ValueError(f"Simulation with ID {simulation_id} does not exist") - + sim_dir = self._get_simulation_dir(simulation_id) profile_path = os.path.join(sim_dir, f"{platform}_profiles.json") - + if not os.path.exists(profile_path): - return [] - + return [] # File chưa được sinh, trả về rỗng thay vì raise lỗi + with open(profile_path, 'r', encoding='utf-8') as f: return json.load(f) - + def get_simulation_config(self, simulation_id: str) -> Optional[Dict[str, Any]]: - """Lấy thông số cấu hình của bản mô phỏng""" + """ + Đọc và trả về nội dung simulation_config.json của một simulation. + + Trả về None nếu config chưa được sinh (chưa qua giai đoạn 3 của prepare_simulation). + Dữ liệu trả về là dict Python (đã parse JSON), không phải string. + """ sim_dir = self._get_simulation_dir(simulation_id) config_path = os.path.join(sim_dir, "simulation_config.json") - + if not os.path.exists(config_path): return None - + with open(config_path, 'r', encoding='utf-8') as f: return json.load(f) - + def get_run_instructions(self, simulation_id: str) -> Dict[str, str]: - """Output ra hướng dẫn / Các câu lệnh dòng lệnh (CMD) để thực thi chạy bản đồ mô phỏng này""" + """ + Tạo ra các câu lệnh terminal để chạy simulation này thủ công. + + Hữu ích khi developer muốn chạy OASIS trực tiếp ngoài Flask, + hoặc để debug từng nền tảng riêng lẻ. + + Returns: + Dict gồm: + - simulation_dir: thư mục chứa dữ liệu simulation + - scripts_dir: thư mục chứa các script OASIS + - config_file: đường dẫn đầy đủ đến simulation_config.json + - commands: dict các lệnh python để chạy từng loại + - instructions: chuỗi hướng dẫn dạng đọc được cho con người + """ sim_dir = self._get_simulation_dir(simulation_id) config_path = os.path.join(sim_dir, "simulation_config.json") + + # Tính đường dẫn tuyệt đối đến thư mục scripts từ vị trí file này + # __file__ = .../backend/app/services/simulation_manager.py + # → abspath("../../scripts") = .../backend/scripts/ scripts_dir = os.path.abspath(os.path.join(os.path.dirname(__file__), '../../scripts')) - + return { "simulation_dir": sim_dir, "scripts_dir": scripts_dir, diff --git a/backend/app/services/zep_entity_reader.py b/backend/app/services/zep_entity_reader.py index 38d9bdca..248d386a 100644 --- a/backend/app/services/zep_entity_reader.py +++ b/backend/app/services/zep_entity_reader.py @@ -1,6 +1,33 @@ """ Dịch vụ đọc và lọc thực thể Zep Đọc các node từ đồ thị Zep, lọc ra các node phù hợp với các loại thực thể đã được định nghĩa trước + +Vị trí trong pipeline (ai gọi file này?): +───────────────────────────────────────────────────────────────────────────── + simulation_manager.py → filter_defined_entities() [caller chính] + └─ prepare_simulation() gọi ở Giai đoạn 1 để lấy entity làm agent + + oasis_profile_generator.py → get_entity_with_context() + └─ khi cần đọc sâu thêm ngữ cảnh của 1 entity cụ thể để sinh persona + + simulation_config_generator.py → nhận FilteredEntities.entities làm input + └─ không gọi trực tiếp, dùng kết quả đã lọc từ simulation_manager + + api/simulation.py → qua SimulationManager, không gọi trực tiếp +───────────────────────────────────────────────────────────────────────────── + +Luồng dữ liệu từ Zep vào hệ thống: + Zep Graph (cloud) + │ + ├── graph.node.get_by_graph_id() ─→ get_all_nodes() ─┐ + │ (phân trang qua zep_paging.fetch_all_nodes) │ + │ ├→ filter_defined_entities() + └── graph.edge.get_by_graph_id() ─→ get_all_edges() ─┘ + (phân trang qua zep_paging.fetch_all_edges) + │ + ▼ + FilteredEntities + └─ entities: List[EntityNode] ← input cho OasisProfileGenerator """ import time @@ -15,24 +42,52 @@ from ..utils.zep_paging import fetch_all_nodes, fetch_all_edges logger = get_logger('mirofish.zep_entity_reader') -# Dùng cho các kiểu trả về generic +# TypeVar để _call_with_retry có thể trả về đúng kiểu dữ liệu của hàm được truyền vào T = TypeVar('T') +# ============================================================================== +# DATACLASS: EntityNode — Đơn vị dữ liệu cơ bản của một thực thể +# ============================================================================== +# Mỗi EntityNode tương ứng với 1 node trong Zep Graph, đã được enrich thêm +# thông tin về edges và các node liên kết lân cận. +# +# Sau khi filter_defined_entities() chạy xong, mỗi EntityNode này sẽ được +# OasisProfileGenerator chuyển đổi thành 1 OasisAgentProfile (1 agent trong OASIS). +# ============================================================================== + @dataclass class EntityNode: """Cấu trúc dữ liệu của node thực thể""" - uuid: str - name: str + + # --- Định danh từ Zep --- + uuid: str # UUID của node trong Zep (dùng để lookup edges) + name: str # Tên hiển thị của entity (ví dụ: "Nguyễn Văn A", "Bộ Giáo Dục") + + # --- Phân loại --- + # Zep gắn label mặc định "Entity" cho mọi node. + # Node có loại cụ thể sẽ có thêm label như "Person", "Organization", "Event"... + # Ví dụ: ["Entity", "Person"] hoặc ["Entity", "Organization"] labels: List[str] + + # --- Nội dung --- + # summary: Đoạn mô tả tóm tắt về entity, do Zep tự sinh khi trích xuất từ văn bản summary: str + # attributes: Các thuộc tính bổ sung dạng key-value (tuổi, nghề nghiệp, v.v.) attributes: Dict[str, Any] - # Thông tin edge liên quan + + # --- Ngữ cảnh mối quan hệ (được enrich khi enrich_with_edges=True) --- + # related_edges: Danh sách các quan hệ mà entity này tham gia + # Mỗi edge có: direction ("incoming"/"outgoing"), edge_name, fact, target/source_node_uuid + # Ví dụ: {"direction": "outgoing", "edge_name": "WORKS_AT", "fact": "A làm việc tại B", ...} related_edges: List[Dict[str, Any]] = field(default_factory=list) - # Thông tin các node khác liên quan + + # related_nodes: Thông tin cơ bản của các node ở đầu kia của related_edges + # Giúp LLM biết entity này liên kết với những ai/gì mà không cần fetch thêm related_nodes: List[Dict[str, Any]] = field(default_factory=list) - + def to_dict(self) -> Dict[str, Any]: + """Serialize toàn bộ EntityNode thành dict — dùng để truyền vào LLM prompt.""" return { "uuid": self.uuid, "name": self.name, @@ -42,23 +97,41 @@ class EntityNode: "related_edges": self.related_edges, "related_nodes": self.related_nodes, } - + def get_entity_type(self) -> Optional[str]: - """Lấy loại thực thể (loại trừ nhãn Entity mặc định)""" + """ + Trả về loại thực thể đầu tiên tìm thấy (loại trừ label mặc định "Entity" và "Node"). + + Ví dụ: + labels = ["Entity", "Person"] → trả về "Person" + labels = ["Entity", "Node"] → trả về None (không có loại cụ thể) + labels = ["Entity"] → trả về None + + Dùng để nhóm agent theo loại khi sinh prompt. + """ for label in self.labels: if label not in ["Entity", "Node"]: return label return None +# ============================================================================== +# DATACLASS: FilteredEntities — Kết quả đầu ra của bước đọc và lọc +# ============================================================================== +# Đây là "gói hàng" được trả về cho simulation_manager.py sau khi đọc Zep. +# simulation_manager lưu entities_count và entity_types vào SimulationState, +# rồi truyền entities vào OasisProfileGenerator và SimulationConfigGenerator. +# ============================================================================== + @dataclass class FilteredEntities: """Tập hợp các thực thể sau khi lọc""" - entities: List[EntityNode] - entity_types: Set[str] - total_count: int - filtered_count: int - + + entities: List[EntityNode] # Danh sách entity đã lọc và enrich — input cho profile/config generator + entity_types: Set[str] # Tập hợp các loại entity có mặt (ví dụ: {"Person", "Organization"}) + total_count: int # Tổng số node đọc được từ Zep (trước khi lọc) + filtered_count: int # Số node còn lại sau khi lọc (bằng len(entities)) + def to_dict(self) -> Dict[str, Any]: return { "entities": [e.to_dict() for e in self.entities], @@ -68,45 +141,72 @@ class FilteredEntities: } +# ============================================================================== +# CLASS: ZepEntityReader — Cầu nối giữa Zep Cloud API và pipeline MiroFish +# ============================================================================== +# Khởi tạo 1 instance mỗi lần gọi prepare_simulation() (stateless, không có cache). +# Toàn bộ kết nối Zep đi qua đây — không service nào khác gọi trực tiếp Zep client +# để đọc nodes/edges. +# ============================================================================== + class ZepEntityReader: """ Dịch vụ đọc và lọc thực thể Zep - + Chức năng chính: - 1. Đọc toàn bộ các node từ đồ thị Zep - 2. Lọc ra các node phù hợp với các loại thực thể đã được định nghĩa (Các node có Labels không chỉ là Entity) - 3. Lấy ra thông tin edge cũng như các node liên quan đối với từng thực thể + 1. Đọc toàn bộ các node từ đồ thị Zep (có phân trang, tối đa 2000 nodes) + 2. Lọc ra các node phù hợp với các loại thực thể đã được định nghĩa + (Các node có Labels không chỉ là "Entity"/"Node") + 3. Enrich mỗi entity với edges và related_nodes lân cận """ - + def __init__(self, api_key: Optional[str] = None): + # Ưu tiên api_key được truyền vào trực tiếp; fallback về biến môi trường self.api_key = api_key or Config.ZEP_API_KEY if not self.api_key: raise ValueError("ZEP_API_KEY is not configured") - + self.client = Zep(api_key=self.api_key) - + + # -------------------------------------------------------------------------- + # PRIVATE: _call_with_retry — Retry wrapper cho các API call Zep đơn lẻ + # -------------------------------------------------------------------------- + # Lưu ý: hàm này CHỈ được dùng bởi get_node_edges() và get_entity_with_context(). + # filter_defined_entities() KHÔNG dùng hàm này — nó dùng fetch_all_nodes/edges + # từ zep_paging.py, vốn đã có retry riêng ở cấp trang. + # -------------------------------------------------------------------------- + def _call_with_retry( - self, - func: Callable[[], T], + self, + func: Callable[[], T], operation_name: str, max_retries: int = 3, initial_delay: float = 2.0 ) -> T: """ - Gọi hàm Zep API có cơ chế thử lại (retry) - + Gọi một hàm Zep API với cơ chế thử lại (exponential backoff). + + Chiến lược retry: + Lần 1: chạy ngay + Lần 2 (nếu lần 1 lỗi): chờ 2 giây + Lần 3 (nếu lần 2 lỗi): chờ 4 giây + Sau 3 lần đều lỗi → raise exception cuối cùng + Args: - func: Hàm cần thực thi (lambda không tham số hoặc callable) - operation_name: Tên thao tác, dùng cho log - max_retries: Số lần thử lại tối đa (mặc định 3 lần, tức là thử tối đa 3 lần) - initial_delay: Số giây trì hoãn ban đầu - + func: Lambda/callable không tham số bọc lệnh gọi API + operation_name: Tên thao tác để ghi vào log (ví dụ: "Fetch node edges") + max_retries: Số lần thử tối đa (mặc định 3) + initial_delay: Delay ban đầu tính bằng giây (tự nhân đôi mỗi lần) + Returns: - Kết quả của lệnh gọi API + Kết quả trả về từ func() nếu thành công + + Raises: + Exception: Exception cuối cùng sau khi hết số lần thử """ last_exception = None delay = initial_delay - + for attempt in range(max_retries): try: return func() @@ -118,26 +218,45 @@ class ZepEntityReader: f"retrying in {delay:.1f} seconds..." ) time.sleep(delay) - delay *= 2 # Lùi bước nhịp mũ (Exponential backoff) + delay *= 2 # Exponential backoff: 2s → 4s → 8s else: logger.error(f"Zep {operation_name} failed after {max_retries} attempts: {str(e)}") - + raise last_exception - + + # -------------------------------------------------------------------------- + # PUBLIC: get_all_nodes / get_all_edges — Bulk fetch toàn bộ nodes và edges + # -------------------------------------------------------------------------- + # Hai hàm này là wrapper mỏng quanh fetch_all_nodes/fetch_all_edges từ + # zep_paging.py. Logic phân trang và retry theo trang nằm hoàn toàn trong + # zep_paging.py (UUID cursor-based pagination, tối đa 2000 nodes). + # + # Tại sao dùng bulk fetch thay vì fetch từng node? + # → Hiệu quả hơn nhiều: 2 API calls cho toàn bộ graph thay vì N calls + # → filter_defined_entities() dùng chiến lược này + # -------------------------------------------------------------------------- + def get_all_nodes(self, graph_id: str) -> List[Dict[str, Any]]: """ - Lấy toàn bộ các node của đồ thị (có phân trang) + Lấy toàn bộ các node của đồ thị bằng cách phân trang. + + Uỷ thác cho fetch_all_nodes() (zep_paging.py) để xử lý: + - Cursor-based pagination (100 nodes/trang) + - Retry từng trang khi lỗi mạng + - Giới hạn tổng 2000 nodes Args: - graph_id: ID của đồ thị + graph_id: ID của Zep Graph Returns: - Danh sách node + Danh sách dict, mỗi dict là 1 node với: uuid, name, labels, summary, attributes """ logger.info(f"Fetching all nodes for graph {graph_id}...") nodes = fetch_all_nodes(self.client, graph_id) + # Chuẩn hoá từ Zep SDK object sang Python dict thuần + # getattr với 2 tên vì Zep SDK đôi khi dùng uuid_ thay vì uuid để tránh conflict keyword nodes_data = [] for node in nodes: nodes_data.append({ @@ -153,13 +272,17 @@ class ZepEntityReader: def get_all_edges(self, graph_id: str) -> List[Dict[str, Any]]: """ - Lấy toàn bộ các edge của đồ thị (có phân trang) + Lấy toàn bộ các edge của đồ thị bằng cách phân trang. + + Tương tự get_all_nodes() — uỷ thác cho fetch_all_edges() (zep_paging.py). + Không có giới hạn tổng số edge (khác với nodes giới hạn 2000). Args: - graph_id: ID của đồ thị + graph_id: ID của Zep Graph Returns: - Danh sách edge + Danh sách dict, mỗi dict là 1 edge với: + uuid, name, fact, source_node_uuid, target_node_uuid, attributes """ logger.info(f"Fetching all edges for graph {graph_id}...") @@ -170,32 +293,45 @@ class ZepEntityReader: edges_data.append({ "uuid": getattr(edge, 'uuid_', None) or getattr(edge, 'uuid', ''), "name": edge.name or "", - "fact": edge.fact or "", - "source_node_uuid": edge.source_node_uuid, - "target_node_uuid": edge.target_node_uuid, + "fact": edge.fact or "", # Mô tả quan hệ dạng câu (ví dụ: "A làm việc tại B") + "source_node_uuid": edge.source_node_uuid, # UUID node nguồn + "target_node_uuid": edge.target_node_uuid, # UUID node đích "attributes": edge.attributes or {}, }) logger.info(f"Total {len(edges_data)} edges fetched") return edges_data - + + # -------------------------------------------------------------------------- + # PUBLIC: get_node_edges — Fetch edges của 1 node cụ thể (per-node API call) + # -------------------------------------------------------------------------- + # ĐÂY KHÔNG PHẢI hàm được dùng trong filter_defined_entities(). + # filter_defined_entities() lấy ALL edges rồi tự scan — không gọi hàm này. + # + # Hàm này chỉ được dùng bởi get_entity_with_context() khi cần fetch + # edges cho 1 entity đơn lẻ theo UUID cụ thể. + # -------------------------------------------------------------------------- + def get_node_edges(self, node_uuid: str) -> List[Dict[str, Any]]: """ - Lấy tất cả các edge liên quan của node được chỉ định (có cơ chế thử lại) - + Lấy tất cả edges liên quan của 1 node cụ thể qua UUID. + + Dùng Zep API: client.graph.node.get_entity_edges(node_uuid=...) + Có retry qua _call_with_retry() (không qua zep_paging vì không phân trang). + Args: - node_uuid: UUID của node - + node_uuid: UUID của node cần lấy edges + Returns: - Danh sách edge + Danh sách edge dict (cùng format với get_all_edges()). + Trả về [] nếu lỗi (không raise exception — caller tự xử lý thiếu data). """ try: - # Sử dụng cơ chế thử lại để gọi Zep API edges = self._call_with_retry( func=lambda: self.client.graph.node.get_entity_edges(node_uuid=node_uuid), operation_name=f"Fetch node edges(node={node_uuid[:8]}...)" ) - + edges_data = [] for edge in edges: edges_data.append({ @@ -206,71 +342,149 @@ class ZepEntityReader: "target_node_uuid": edge.target_node_uuid, "attributes": edge.attributes or {}, }) - + return edges_data except Exception as e: + # Lỗi khi fetch edge của 1 node không nên làm hỏng cả pipeline + # → log warning và trả về rỗng, caller (get_entity_with_context) tiếp tục logger.warning(f"Failed to fetch edges for node {node_uuid}: {str(e)}") return [] - + + # -------------------------------------------------------------------------- + # PUBLIC: filter_defined_entities — Hàm CHÍNH của class + # -------------------------------------------------------------------------- + # Đây là hàm được gọi bởi simulation_manager.prepare_simulation() ở Giai đoạn 1. + # Toàn bộ pipeline phụ thuộc vào output của hàm này. + # + # Thuật toán: + # 1. Bulk fetch TẤT CẢ nodes (2 API calls qua phân trang) + # 2. Bulk fetch TẤT CẢ edges (nếu enrich_with_edges=True) + # 3. Duyệt từng node, áp dụng logic lọc label + # 4. Với mỗi node hợp lệ: scan toàn bộ all_edges để tìm edges liên quan + # 5. Trả về FilteredEntities + # + # Độ phức tạp: O(N * M) với N = số node lọc được, M = tổng số edge + # Trong thực tế graph nhỏ (< 100 nodes, < 500 edges) nên không ảnh hưởng + # -------------------------------------------------------------------------- + def filter_defined_entities( - self, + self, graph_id: str, defined_entity_types: Optional[List[str]] = None, enrich_with_edges: bool = True ) -> FilteredEntities: """ - Lọc ra các node phù hợp với các loại thực thể đã được định nghĩa - - Logic lọc: - - Nếu Labels của node chỉ có một nhãn là "Entity", tức là thực thể này không hợp với loại chúng ta định nghĩa, tiến hành bỏ qua - - Nếu Labels của node chứa các nhãn khác ngoài "Entity" và "Node", tức là hợp lệ, tiến hành giữ lại - + Đọc toàn bộ graph từ Zep rồi lọc ra các entity có loại được định nghĩa. + + Logic lọc label (quan trọng — đây là cách phân biệt entity "có ý nghĩa"): + ┌──────────────────────────────────────────────────────┐ + │ Node trong Zep có 2 loại label: │ + │ │ + │ "Generic" node: labels = ["Entity"] hoặc ["Node"] │ + │ → Zep tạo ra khi trích xuất context chung, │ + │ không phải thực thể cụ thể nào → BỎ QUA │ + │ │ + │ "Typed" node: labels = ["Entity", "Person"] │ + │ labels = ["Entity", "Organization"] │ + │ → Thực thể được định danh rõ ràng → GIỮ LẠI │ + └──────────────────────────────────────────────────────┘ + + Hai chế độ lọc: + - defined_entity_types=None: giữ TẤT CẢ typed nodes (mọi custom label đều OK) + - defined_entity_types=["Person", "Organization"]: chỉ giữ nodes có label trong danh sách + + Enrich với edges: + - Sau khi lọc, với mỗi entity, scan all_edges để tìm edges có + source_node_uuid hoặc target_node_uuid trùng với entity.uuid + - Phân loại: source = outgoing, target = incoming + - Lấy thêm thông tin cơ bản của node ở đầu kia (tên, labels, summary) + + Ví dụ minh hoạ: + ───────────────────────────────────────────────────────────────────── + Giả sử Zep Graph có 4 nodes và 2 edges sau: + + NODES: + uuid="aaa", name="Nguyễn Văn A", labels=["Entity", "Person"] + uuid="bbb", name="Bộ Giáo Dục", labels=["Entity", "Organization"] + uuid="ccc", name="context-001", labels=["Entity"] ← generic, bị loại + uuid="ddd", name="Hà Nội", labels=["Entity", "Location"] + + EDGES: + source="aaa" → target="bbb", fact="Nguyễn Văn A làm việc tại Bộ Giáo Dục" + source="ddd" → target="aaa", fact="Nguyễn Văn A sinh sống tại Hà Nội" + + ── Gọi với defined_entity_types=None ────────────────────────────── + filter_defined_entities(graph_id, defined_entity_types=None) + + → Giữ lại: "aaa" (Person), "bbb" (Organization), "ddd" (Location) + → Loại bỏ: "ccc" (chỉ có label "Entity") + → entity_types = {"Person", "Organization", "Location"} + → filtered_count = 3, total_count = 4 + + EntityNode uuid="aaa" (Nguyễn Văn A) sau enrich: + related_edges = [ + {"direction": "outgoing", "edge_name": "...", "fact": "A làm việc tại Bộ GD", "target_node_uuid": "bbb"}, + {"direction": "incoming", "edge_name": "...", "fact": "A sinh sống tại Hà Nội", "source_node_uuid": "ddd"}, + ] + related_nodes = [ + {"uuid": "bbb", "name": "Bộ Giáo Dục", "labels": ["Entity", "Organization"], "summary": "..."}, + {"uuid": "ddd", "name": "Hà Nội", "labels": ["Entity", "Location"], "summary": "..."}, + ] + + ── Gọi với defined_entity_types=["Person"] ───────────────────────── + filter_defined_entities(graph_id, defined_entity_types=["Person"]) + + → Giữ lại: chỉ "aaa" (Person) + → Loại bỏ: "bbb" (Organization không match), "ccc" (generic), "ddd" (Location không match) + → filtered_count = 1 + ───────────────────────────────────────────────────────────────────── + Args: - graph_id: ID của đồ thị - defined_entity_types: Danh sách các loại thực thể định nghĩa trước (không bắt buộc, nếu có thì chỉ giữ lại các loại đó) - enrich_with_edges: Có lấy thông tin edge liên quan của từng thực thể hay không - + graph_id: ID của Zep Graph + defined_entity_types: Danh sách loại cần lọc (None = lấy tất cả) + enrich_with_edges: True = thêm related_edges + related_nodes vào mỗi entity + Returns: - FilteredEntities: Tập hợp các thực thể sau khi lọc + FilteredEntities chứa danh sách EntityNode đã enrich """ logger.info(f"Start filtering entities for graph {graph_id}...") - - # Lấy toàn bộ các node + + # --- Bước 1: Bulk fetch nodes và edges --- all_nodes = self.get_all_nodes(graph_id) total_count = len(all_nodes) - - # Lấy toàn bộ các edge (để lấy liên kết sau này) + + # Chỉ fetch edges nếu cần enrich (tiết kiệm 1 API call nếu không cần) all_edges = self.get_all_edges(graph_id) if enrich_with_edges else [] - - # Xây dựng map ánh xạ từ UUID của node sang dữ liệu node + + # Xây dựng dict ánh xạ uuid → node data để tra cứu O(1) khi enrich related_nodes node_map = {n["uuid"]: n for n in all_nodes} - - # Lọc các thực thể đáp ứng điều kiện + + # --- Bước 2: Lọc nodes theo label --- filtered_entities = [] entity_types_found = set() - + for node in all_nodes: labels = node.get("labels", []) - - # Logic lọc: Labels bắt buộc phải chứa các nhãn khác "Entity" và "Node" + + # Tìm custom labels (loại trừ "Entity" và "Node" là labels mặc định của Zep) custom_labels = [l for l in labels if l not in ["Entity", "Node"]] - + if not custom_labels: - # Chỉ có nhãn mặc định, bỏ qua + # Node này chỉ có labels mặc định → không phải entity cụ thể → bỏ qua continue - - # Nếu đã chỉ định loại thực thể cho trước, kiểm tra xem có khớp hay không + + # Nếu người dùng chỉ định danh sách loại, kiểm tra xem node có match không if defined_entity_types: matching_labels = [l for l in custom_labels if l in defined_entity_types] if not matching_labels: - continue - entity_type = matching_labels[0] + continue # Node này có custom label nhưng không nằm trong danh sách cho phép + entity_type = matching_labels[0] # Lấy loại match đầu tiên else: - entity_type = custom_labels[0] - + entity_type = custom_labels[0] # Lấy custom label đầu tiên làm loại + entity_types_found.add(entity_type) - - # Tạo object cho node thực thể + + # Tạo EntityNode (related_edges và related_nodes vẫn rỗng, enrich ở bước sau) entity = EntityNode( uuid=node["uuid"], name=node["name"], @@ -278,14 +492,17 @@ class ZepEntityReader: summary=node["summary"], attributes=node["attributes"], ) - - # Lấy các edge và node liên quan + + # --- Bước 3: Enrich với edges và related nodes --- if enrich_with_edges: related_edges = [] related_node_uuids = set() - + + # Scan toàn bộ all_edges để tìm edges có liên quan đến node này + # (không gọi get_node_edges() vì đã có all_edges trong memory) for edge in all_edges: if edge["source_node_uuid"] == node["uuid"]: + # Entity này là nguồn → quan hệ "outgoing" (chiều đi ra) related_edges.append({ "direction": "outgoing", "edge_name": edge["name"], @@ -293,7 +510,9 @@ class ZepEntityReader: "target_node_uuid": edge["target_node_uuid"], }) related_node_uuids.add(edge["target_node_uuid"]) + elif edge["target_node_uuid"] == node["uuid"]: + # Entity này là đích → quan hệ "incoming" (chiều đi vào) related_edges.append({ "direction": "incoming", "edge_name": edge["name"], @@ -301,10 +520,10 @@ class ZepEntityReader: "source_node_uuid": edge["source_node_uuid"], }) related_node_uuids.add(edge["source_node_uuid"]) - + entity.related_edges = related_edges - - # Lấy thông tin cơ bản của các node được liên kết + + # Lấy thông tin cơ bản của các node lân cận (không lấy full để tránh vòng lặp) related_nodes = [] for related_uuid in related_node_uuids: if related_uuid in node_map: @@ -314,58 +533,72 @@ class ZepEntityReader: "name": related_node["name"], "labels": related_node["labels"], "summary": related_node.get("summary", ""), + # Không lấy attributes để tránh payload quá lớn khi truyền vào LLM }) - + entity.related_nodes = related_nodes - + filtered_entities.append(entity) - + logger.info(f"Filtering completed: Total nodes {total_count}, Matched {len(filtered_entities)}, " f"Entity types: {entity_types_found}") - + return FilteredEntities( entities=filtered_entities, entity_types=entity_types_found, total_count=total_count, filtered_count=len(filtered_entities), ) - + + # -------------------------------------------------------------------------- + # PUBLIC: get_entity_with_context — Fetch 1 entity đơn lẻ theo UUID + # -------------------------------------------------------------------------- + # Dùng bởi oasis_profile_generator.py khi cần đọc sâu thêm 1 entity cụ thể. + # Khác với filter_defined_entities(): + # - Fetch 1 node theo UUID (không scan toàn bộ graph) + # - Dùng get_node_edges() per-node (per-node API call) + # - Nhưng vẫn cần fetch all_nodes để build related_nodes → tốn hơn + # -------------------------------------------------------------------------- + def get_entity_with_context( - self, - graph_id: str, + self, + graph_id: str, entity_uuid: str ) -> Optional[EntityNode]: """ - Lấy thông tin của một thực thể cụ thể và ngữ cảnh đầy đủ của nó (edge và node liên kết, với cơ chế thử lại) - + Lấy thông tin đầy đủ của 1 entity cụ thể và toàn bộ ngữ cảnh liên kết. + + Dùng per-node API call (khác với filter_defined_entities dùng bulk fetch). + Phù hợp khi chỉ cần 1 entity — không hiệu quả nếu cần nhiều entities. + Args: - graph_id: ID của đồ thị - entity_uuid: UUID của thực thể - + graph_id: ID của Zep Graph (dùng để lấy node_map cho related_nodes) + entity_uuid: UUID của entity cần lấy + Returns: - EntityNode hoặc None + EntityNode đầy đủ hoặc None nếu không tìm thấy / lỗi """ try: - # Sử dụng cơ chế thử lại để lấy thông tin node + # Fetch node theo UUID với retry node = self._call_with_retry( func=lambda: self.client.graph.node.get(uuid_=entity_uuid), operation_name=f"Fetch node detail(uuid={entity_uuid[:8]}...)" ) - + if not node: return None - - # Lấy các edge của node + + # Fetch edges của node này (per-node call, có retry) edges = self.get_node_edges(entity_uuid) - - # Lấy tất cả các node để tìm liên kết + + # Cần all_nodes để build related_nodes map — đây là điểm tốn kém nhất all_nodes = self.get_all_nodes(graph_id) node_map = {n["uuid"]: n for n in all_nodes} - - # Xử lý các edge và node liên quan + + # Xử lý edges (cùng logic với filter_defined_entities) related_edges = [] related_node_uuids = set() - + for edge in edges: if edge["source_node_uuid"] == entity_uuid: related_edges.append({ @@ -383,8 +616,7 @@ class ZepEntityReader: "source_node_uuid": edge["source_node_uuid"], }) related_node_uuids.add(edge["source_node_uuid"]) - - # Lấy thông tin về node được liên kết + related_nodes = [] for related_uuid in related_node_uuids: if related_uuid in node_map: @@ -395,7 +627,7 @@ class ZepEntityReader: "labels": related_node["labels"], "summary": related_node.get("summary", ""), }) - + return EntityNode( uuid=getattr(node, 'uuid_', None) or getattr(node, 'uuid', ''), name=node.name or "", @@ -405,27 +637,34 @@ class ZepEntityReader: related_edges=related_edges, related_nodes=related_nodes, ) - + except Exception as e: logger.error(f"Failed to fetch entity {entity_uuid}: {str(e)}") return None - + + # -------------------------------------------------------------------------- + # PUBLIC: get_entities_by_type — Convenience wrapper cho 1 loại entity cụ thể + # -------------------------------------------------------------------------- + def get_entities_by_type( - self, - graph_id: str, + self, + graph_id: str, entity_type: str, enrich_with_edges: bool = True ) -> List[EntityNode]: """ - Lấy tất cả các thực thể dựa theo loại cụ thể - + Lấy tất cả entities của một loại cụ thể. + + Là thin wrapper quanh filter_defined_entities() với defined_entity_types=[entity_type]. + Tiện dụng khi chỉ cần 1 loại thay vì nhiều loại. + Args: - graph_id: ID của đồ thị - entity_type: Loại thực thể (ví dụ: "Student", "PublicFigure", v.v..) - enrich_with_edges: Có lấy thông tin edge liên quan hay không - + graph_id: ID của Zep Graph + entity_type: Loại entity cần lọc (ví dụ: "Person", "Organization") + enrich_with_edges: Có enrich với edges không + Returns: - Danh sách thực thể + Danh sách EntityNode thuộc loại đã chỉ định """ result = self.filter_defined_entities( graph_id=graph_id, @@ -433,5 +672,3 @@ class ZepEntityReader: enrich_with_edges=enrich_with_edges ) return result.entities - - diff --git a/backend/app/services/zep_tools.py b/backend/app/services/zep_tools.py index 4de8331b..27fedc94 100644 --- a/backend/app/services/zep_tools.py +++ b/backend/app/services/zep_tools.py @@ -1419,7 +1419,7 @@ Trả về danh sách các câu hỏi phụ dưới định dạng JSON.""" simulation_id=simulation_id, interviews=interviews_request, platform=None, # Không định dạng nền tảng -> Dual platform call - timeout=180.0 # Tăng timeout do phải chờ API trên 2 platforms xử lý + timeout= 800 # Tăng timeout do phải chờ API trên 2 platforms xử lý ) logger.info(f"Interview API returned: {api_result.get('interviews_count', 0)} results, success={api_result.get('success')}") @@ -1832,7 +1832,7 @@ Hãy tạo bản tóm tắt phỏng vấn.""" {"role": "user", "content": user_prompt} ], temperature=0.3, - max_tokens=800 + max_tokens=4096 ) return summary diff --git a/backend/app/utils/llm_client.py b/backend/app/utils/llm_client.py index 987959c5..4d923cad 100644 --- a/backend/app/utils/llm_client.py +++ b/backend/app/utils/llm_client.py @@ -79,7 +79,7 @@ class LLMClient: metadata=call_metadata, **{k: v for k, v in kwargs.items() if k not in {"model", "messages"}}, ) - content = response.choices[0].message.content + content = response.choices[0].message.content or "" # Một số model (vd MiniMax M2.5) chèn nội dung vào content, cần loại bỏ content = re.sub(r'[\s\S]*?', '', content).strip() return content @@ -88,7 +88,7 @@ class LLMClient: self, messages: List[Dict[str, str]], temperature: float = 0.3, - max_tokens: int = 4096, + max_tokens: int = 50000, metadata: Optional[Dict[str, Any]] = None, ) -> Dict[str, Any]: """ diff --git a/backend/scripts/run_parallel_simulation.py b/backend/scripts/run_parallel_simulation.py index fb9eda66..c0953f6a 100644 --- a/backend/scripts/run_parallel_simulation.py +++ b/backend/scripts/run_parallel_simulation.py @@ -1057,6 +1057,7 @@ def create_model(config: Dict[str, Any], use_boost: bool = False): return ModelFactory.create( model_platform=ModelPlatformType.OPENAI, model_type=llm_model, + timeout=1000 ) @@ -1179,7 +1180,7 @@ async def run_twitter_simulation( agent_graph=result.agent_graph, platform=oasis.DefaultPlatformType.TWITTER, database_path=db_path, - semaphore=30, # Giới hạn số request LLM đồng thời để tránh quá tải API + semaphore=3, # Giới hạn số request LLM đồng thời để tránh quá tải API ) await result.env.reset() @@ -1370,7 +1371,7 @@ async def run_reddit_simulation( agent_graph=result.agent_graph, platform=oasis.DefaultPlatformType.REDDIT, database_path=db_path, - semaphore=30, # Giới hạn số request LLM đồng thời để tránh quá tải API + semaphore=3, # Giới hạn số request LLM đồng thời để tránh quá tải API ) await result.env.reset() diff --git a/backend/scripts/test_entity_reader.py b/backend/scripts/test_entity_reader.py new file mode 100644 index 00000000..db456fe3 --- /dev/null +++ b/backend/scripts/test_entity_reader.py @@ -0,0 +1,137 @@ +""" +Script test ZepEntityReader — đọc graph đã build, xem logic chọn entity làm agent +Chạy từ thư mục backend/ + +Usage: + python scripts/test_entity_reader.py + python scripts/test_entity_reader.py --type Student + python scripts/test_entity_reader.py --uuid + python scripts/test_entity_reader.py --from-log case_01_academic_scandal +""" +import sys +import json +import os + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from app.services.zep_entity_reader import ZepEntityReader + +OUTPUT_DIR = os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))), "logs") + + +def load_graph_id_from_log(case_id: str) -> str: + log_path = os.path.join(OUTPUT_DIR, f"test_graph_output_{case_id}.json") + if not os.path.exists(log_path): + print(f"Không tìm thấy file log: {log_path}") + sys.exit(1) + with open(log_path, encoding="utf-8") as f: + data = json.load(f) + return data["graph_id"] + + +def print_entity(entity, show_edges: bool = True): + entity_type = entity.get_entity_type() or "(unknown)" + print(f"\n ── [{entity_type}] {entity.name}") + print(f" uuid : {entity.uuid}") + if entity.summary: + print(f" summary : {entity.summary[:120]}") + if entity.attributes: + for k, v in entity.attributes.items(): + if k not in ("name",) and v and v != "null": + print(f" attr : {k} = {v}") + if show_edges and entity.related_edges: + print(f" edges ({len(entity.related_edges)}):") + for e in entity.related_edges: + arrow = "→" if e["direction"] == "outgoing" else "←" + print(f" {arrow} [{e['edge_name']}] {e.get('fact', '')[:80]}") + if entity.related_nodes: + names = [n["name"] for n in entity.related_nodes] + print(f" related : {', '.join(names)}") + + +def cmd_all(reader: ZepEntityReader, graph_id: str): + """Lấy tất cả entity hợp lệ (có custom label) → danh sách agent tiềm năng""" + print(f"\nGraph: {graph_id}") + print("Đang lọc entity...") + result = reader.filter_defined_entities(graph_id, enrich_with_edges=True) + + print(f"\n{'=' * 60}") + print(f"Tổng nodes : {result.total_count}") + print(f"Entity hợp lệ : {result.filtered_count} ← đây là các agent tiềm năng") + print(f"Entity types : {sorted(result.entity_types)}") + print(f"{'=' * 60}") + + # Nhóm theo type để dễ đọc + by_type: dict = {} + for e in result.entities: + t = e.get_entity_type() or "unknown" + by_type.setdefault(t, []).append(e) + + for entity_type, entities in sorted(by_type.items()): + print(f"\n[{entity_type}] — {len(entities)} entity") + for e in entities: + print_entity(e) + + # Lưu output + os.makedirs(OUTPUT_DIR, exist_ok=True) + out_path = os.path.join(OUTPUT_DIR, f"test_entity_reader_{graph_id}.json") + with open(out_path, "w", encoding="utf-8") as f: + json.dump(result.to_dict(), f, ensure_ascii=False, indent=2) + print(f"\nĐã lưu: {out_path}") + + +def cmd_by_type(reader: ZepEntityReader, graph_id: str, entity_type: str): + print(f"\nLấy entity type [{entity_type}] từ graph {graph_id}...") + entities = reader.get_entities_by_type(graph_id, entity_type, enrich_with_edges=True) + print(f"Tìm thấy {len(entities)} entity loại [{entity_type}]:") + for e in entities: + print_entity(e) + + +def cmd_by_uuid(reader: ZepEntityReader, graph_id: str, uuid: str): + print(f"\nLấy chi tiết entity {uuid}...") + entity = reader.get_entity_with_context(graph_id, uuid) + if not entity: + print("Không tìm thấy entity.") + return + print_entity(entity, show_edges=True) + + +def main(): + args = sys.argv[1:] + graph_id = None + entity_type = None + entity_uuid = None + + if not args: + print(__doc__) + sys.exit(1) + + if args[0] == "--from-log": + if len(args) < 2: + print("Usage: --from-log ") + sys.exit(1) + graph_id = load_graph_id_from_log(args[1]) + print(f"Graph ID từ log [{args[1]}]: {graph_id}") + args = args[2:] + else: + graph_id = args[0] + args = args[1:] + + if "--type" in args: + entity_type = args[args.index("--type") + 1] + if "--uuid" in args: + entity_uuid = args[args.index("--uuid") + 1] + + reader = ZepEntityReader() + + if entity_uuid: + cmd_by_uuid(reader, graph_id, entity_uuid) + elif entity_type: + cmd_by_type(reader, graph_id, entity_type) + else: + cmd_all(reader, graph_id) + + +if __name__ == "__main__": + main() diff --git a/install.cmd b/install.cmd new file mode 100644 index 00000000..cee34fd9 --- /dev/null +++ b/install.cmd @@ -0,0 +1,221 @@ +@echo off +setlocal enabledelayedexpansion + +REM Claude Code Windows CMD Bootstrap Script +REM Installs Claude Code for environments where PowerShell is not available + +REM Parse command line argument +set "TARGET=%~1" +if "!TARGET!"=="" set "TARGET=latest" + +REM Validate target parameter +if /i "!TARGET!"=="stable" goto :target_valid +if /i "!TARGET!"=="latest" goto :target_valid +echo !TARGET! | findstr /r "^[0-9][0-9]*\.[0-9][0-9]*\.[0-9][0-9]*" >nul +if !ERRORLEVEL! equ 0 goto :target_valid + +echo Usage: %0 [stable^|latest^|VERSION] >&2 +echo Example: %0 1.0.58 >&2 +exit /b 1 + +:target_valid + +REM Check for 64-bit Windows +if /i "%PROCESSOR_ARCHITECTURE%"=="AMD64" goto :arch_valid +if /i "%PROCESSOR_ARCHITECTURE%"=="ARM64" goto :arch_valid +if /i "%PROCESSOR_ARCHITEW6432%"=="AMD64" goto :arch_valid +if /i "%PROCESSOR_ARCHITEW6432%"=="ARM64" goto :arch_valid + +echo Claude Code does not support 32-bit Windows. Please use a 64-bit version of Windows. >&2 +exit /b 1 + +:arch_valid + +REM Set constants +set "GCS_BUCKET=https://storage.googleapis.com/claude-code-dist-86c565f3-f756-42ad-8dfa-d59b1c096819/claude-code-releases" +set "DOWNLOAD_DIR=%USERPROFILE%\.claude\downloads" +REM Use native ARM64 binary on ARM64 Windows, x64 otherwise +if /i "%PROCESSOR_ARCHITECTURE%"=="ARM64" ( + set "PLATFORM=win32-arm64" +) else ( + set "PLATFORM=win32-x64" +) + +REM Create download directory +if not exist "!DOWNLOAD_DIR!" mkdir "!DOWNLOAD_DIR!" + +REM Check for curl availability +curl --version >nul 2>&1 +if !ERRORLEVEL! neq 0 ( + echo curl is required but not available. Please install curl or use PowerShell installer. >&2 + exit /b 1 +) + +REM Always download latest version (which has the most up-to-date installer) +call :download_file "!GCS_BUCKET!/latest" "!DOWNLOAD_DIR!\latest" +if !ERRORLEVEL! neq 0 ( + echo Failed to get latest version >&2 + exit /b 1 +) + +REM Read version from file +set /p VERSION=<"!DOWNLOAD_DIR!\latest" +del "!DOWNLOAD_DIR!\latest" + +REM Download manifest +call :download_file "!GCS_BUCKET!/!VERSION!/manifest.json" "!DOWNLOAD_DIR!\manifest.json" +if !ERRORLEVEL! neq 0 ( + echo Failed to get manifest >&2 + exit /b 1 +) + +REM Extract checksum from manifest +call :parse_manifest "!DOWNLOAD_DIR!\manifest.json" "!PLATFORM!" +if !ERRORLEVEL! neq 0 ( + echo Platform !PLATFORM! not found in manifest >&2 + del "!DOWNLOAD_DIR!\manifest.json" 2>nul + exit /b 1 +) +del "!DOWNLOAD_DIR!\manifest.json" + +REM Download binary +set "BINARY_PATH=!DOWNLOAD_DIR!\claude-!VERSION!-!PLATFORM!.exe" +call :download_file "!GCS_BUCKET!/!VERSION!/!PLATFORM!/claude.exe" "!BINARY_PATH!" +if !ERRORLEVEL! neq 0 ( + echo Failed to download binary >&2 + if exist "!BINARY_PATH!" del "!BINARY_PATH!" + exit /b 1 +) + +REM Verify checksum +call :verify_checksum "!BINARY_PATH!" "!EXPECTED_CHECKSUM!" +if !ERRORLEVEL! neq 0 ( + echo Checksum verification failed >&2 + del "!BINARY_PATH!" + exit /b 1 +) + +REM Run claude install to set up launcher and shell integration +echo Setting up Claude Code... +"!BINARY_PATH!" install "!TARGET!" +set "INSTALL_RESULT=!ERRORLEVEL!" + +REM Clean up downloaded file +REM Wait a moment for any file handles to be released +timeout /t 1 /nobreak >nul 2>&1 +del /f "!BINARY_PATH!" >nul 2>&1 +if exist "!BINARY_PATH!" ( + echo Warning: Could not remove temporary file: !BINARY_PATH! +) + +if !INSTALL_RESULT! neq 0 ( + echo Installation failed >&2 + exit /b 1 +) + +echo. +echo Installation complete^^! +echo. +exit /b 0 + +REM ============================================================================ +REM SUBROUTINES +REM ============================================================================ + +:download_file +REM Downloads a file using curl +REM Args: %1=URL, %2=OutputPath +set "URL=%~1" +set "OUTPUT=%~2" + +curl -fsSL "!URL!" -o "!OUTPUT!" +exit /b !ERRORLEVEL! + +:parse_manifest +REM Parse JSON manifest to extract checksum for platform +REM Args: %1=ManifestPath, %2=Platform +set "MANIFEST_PATH=%~1" +set "PLATFORM_NAME=%~2" +set "EXPECTED_CHECKSUM=" + +REM Use findstr to find platform section, then look for checksum +set "FOUND_PLATFORM=" +set "IN_PLATFORM_SECTION=" + +REM Read the manifest line by line +for /f "usebackq tokens=*" %%i in ("!MANIFEST_PATH!") do ( + set "LINE=%%i" + + REM Check if this line contains our platform + echo !LINE! | findstr /c:"\"%PLATFORM_NAME%\":" >nul + if !ERRORLEVEL! equ 0 ( + set "IN_PLATFORM_SECTION=1" + ) + + REM If we're in the platform section, look for checksum + if defined IN_PLATFORM_SECTION ( + echo !LINE! | findstr /c:"\"checksum\":" >nul + if !ERRORLEVEL! equ 0 ( + REM Extract checksum value + for /f "tokens=2 delims=:" %%j in ("!LINE!") do ( + set "CHECKSUM_PART=%%j" + REM Remove quotes, whitespace, and comma + set "CHECKSUM_PART=!CHECKSUM_PART: =!" + set "CHECKSUM_PART=!CHECKSUM_PART:"=!" + set "CHECKSUM_PART=!CHECKSUM_PART:,=!" + + REM Check if it looks like a SHA256 (64 hex chars) + if not "!CHECKSUM_PART!"=="" ( + call :check_length "!CHECKSUM_PART!" 64 + if !ERRORLEVEL! equ 0 ( + set "EXPECTED_CHECKSUM=!CHECKSUM_PART!" + exit /b 0 + ) + ) + ) + ) + + REM Check if we've left the platform section (closing brace) + echo !LINE! | findstr /c:"}" >nul + if !ERRORLEVEL! equ 0 set "IN_PLATFORM_SECTION=" + ) +) + +if "!EXPECTED_CHECKSUM!"=="" exit /b 1 +exit /b 0 + +:check_length +REM Check if string length equals expected length +REM Args: %1=String, %2=ExpectedLength +set "STR=%~1" +set "EXPECTED_LEN=%~2" +set "LEN=0" +:count_loop +if "!STR:~%LEN%,1!"=="" goto :count_done +set /a LEN+=1 +goto :count_loop +:count_done +if %LEN%==%EXPECTED_LEN% exit /b 0 +exit /b 1 + +:verify_checksum +REM Verify file checksum using certutil +REM Args: %1=FilePath, %2=ExpectedChecksum +set "FILE_PATH=%~1" +set "EXPECTED=%~2" + +for /f "skip=1 tokens=*" %%i in ('certutil -hashfile "!FILE_PATH!" SHA256') do ( + set "ACTUAL=%%i" + set "ACTUAL=!ACTUAL: =!" + if "!ACTUAL!"=="CertUtil:Thecommandcompletedsuccessfully." goto :verify_done + if "!ACTUAL!" neq "" ( + if /i "!ACTUAL!"=="!EXPECTED!" ( + exit /b 0 + ) else ( + exit /b 1 + ) + ) +) + +:verify_done +exit /b 1 diff --git a/oasis b/oasis new file mode 160000 index 00000000..336c9c6e --- /dev/null +++ b/oasis @@ -0,0 +1 @@ +Subproject commit 336c9c6ed7583b863ef0536a870f7d510d55502c diff --git a/report.md b/report.md new file mode 100644 index 00000000..cc8c40fd --- /dev/null +++ b/report.md @@ -0,0 +1,186 @@ +# Báo cáo Dự báo Tương lai: Động lực Thị trường Dầu mỏ & Tài chính dưới Áp lực Địa chính trị + +> Mô phỏng dự báo một tương lai mà thị trường dầu mỏ sẽ duy trì biến động mạnh và giá cao trong ít nhất 6 tháng, với sự phân hóa rõ rệt giữa các tác nhân: nhà giao dịch tận dụng đòn bẩy để đặt cược ngắn hạn, trong khi chuyên gia và tổ chức quốc tế rơi vào trạng thái 'bay mù' và thận trọng, dẫn đến rủi ro hệ thống từ sự thích nghi cảm xúc và điểm gãy cung cầu tại các nút thắt chiến lược. + +--- + +## Trạng thái Thị trường Tương lai: Biến động Cấu trúc và Sự Thích nghi Cảm xúc + +Chương này phân tích trạng thái tương lai của thị trường dầu mỏ và tài chính, nơi mà áp lực địa chính trị đã tạo ra một cấu trúc biến động mới. Kết quả mô phỏng cho thấy thị trường sẽ không còn phản ứng tuyến tính với tin tức, mà thay vào đó là sự thích nghi cảm xúc phức tạp giữa các nhóm tác nhân khác nhau. + +**Biến động Cấu trúc và Xu hướng Giá Cao** + +Trong kịch bản tương lai được mô phỏng, thị trường dầu mỏ đã chuyển dịch từ trạng thái dư thừa sang tình trạng thiếu hụt nghiêm trọng, dẫn đến việc giá dầu duy trì ở mức cao và biến động mạnh. Nguyên nhân cốt lõi là sự phong tỏa Eo biển Hormuz và xung đột kéo dài ở Iran. + +> "Giá dầu đã tăng khoảng 40% kể từ khi xung đột Iran bắt đầu, và thị trường sẵn sàng cho những mức tăng thêm nếu tình hình ở Trung Đông xấu đi trở lại." + +Sự biến động này không chỉ là tạm thời mà mang tính cấu trúc. Các nhà phân tích trong mô phỏng đã xác định một xu hướng dài hạn mới. + +> "Các nhà phân tích tin rằng tình trạng giá cao và nguồn cung thấp trong lĩnh vực dầu mỏ sẽ tiếp tục trong ít nhất sáu tháng." + +Điều này được củng cố bởi thực tế là nguồn cung vật lý bị gián đoạn nghiêm trọng. + +> "Cơ quan Năng lượng Quốc tế (IEA) cảnh báo rằng trong kịch bản bất lợi hơn của cuộc xung đột kéo dài, thị trường năng lượng và nền kinh tế trên khắp thế giới cần chuẩn bị cho những gián đoạn đáng kể trong những tháng tới." + +**Sự Phân hóa Cảm xúc: Nhà Giao dịch Đòn bẩy vs. Chuyên gia Thận trọng** + +Một đặc điểm nổi bật của tương lai này là sự phân hóa rõ rệt giữa các nhóm tham gia thị trường. Nhà giao dịch sử dụng đòn bẩy (traders) thường phản ứng cảm xúc với tin tức ngắn hạn, trong khi các chuyên gia và tổ chức quốc tế rơi vào trạng thái thận trọng và thiếu thông tin rõ ràng ("bay mù"). + +**Nhà giao dịch tận dụng đòn bẩy** + +Nhóm này bị ảnh hưởng nặng nề bởi sự biến động nhanh chóng và thường đặt cược sai hướng. + +> "Các nhà giao dịch hàng hóa đã chịu thiệt hại hàng tỷ đô la vì họ đã đặt cược sai hướng về giá dầu." + +Hành vi của họ thường bị chi phối bởi các tuyên bố từ cấp cao chính trị, dẫn đến những đợt tăng giảm thất thường. + +> "Hợp đồng tương lai dầu thô đã nảy khắp nơi, thường dao động dựa trên tuyên bố lạc quan nhất từ Trump về việc chấm dứt xung đột." + +Tuy nhiên, sự tham gia của họ cũng tạo ra các cơ hội đầu cơ rủi ro cao. + +> "Việc chuyển hướng dòng tiền vào thị trường hàng hóa có thể ảnh hưởng đáng kể đến giá cả, có thể bất lợi cho nhà đầu tư." + +**Chuyên gia và Tổ chức Quốc tế** + +Ngược lại, các chuyên gia nhận ra rằng tình hình nghiêm trọng hơn nhiều so với những gì thị trường ngắn hạn đang định giá. + +> "Nic Dyer lưu ý rằng tình trạng thắt chặt nguồn cung dầu thô nghiêm trọng hơn nhiều so với những gì các nhà giao dịch tương lai có thể giả định." + +Tuy nhiên, ngay cả các tổ chức tài chính lớn cũng có xu hướng đánh giá thấp tác động thực sự. + +> "Các nhà kinh tế trong hầu hết các tổ chức tài chính đánh giá thấp tác động của việc mất nguồn tài nguyên năng lượng. Khái niệm sai lầm này giải thích tại sao các nhà kinh tế... và do đó nhà đầu tư, không đặc biệt lo lắng." + +Sự thiếu hiểu biết này dẫn đến rủi ro hệ thống khi các cú sốc giá lan rộng. + +> "Các nhà hoạch định chính sách mô tả kết quả của căng thẳng dai dẳng là冒着 rủi ro 'phá hủy nhu cầu' hoàn toàn." + +**Thích nghi Cảm xúc và Rủi ro Hệ thống** + +Theo thời gian, thị trường bắt đầu thể hiện dấu hiệu của sự thích nghi cảm xúc. Ban đầu, thị trường phản ứng mạnh với mọi tin tức, nhưng dần dần trở nên mệt mỏi và lọc bỏ các nhiễu loạn. + +> "Thị trường mệt mỏi vì chiến tranh lọc bỏ tiếng ồn, theo Morning Bid công bố vào ngày 14 tháng 4 năm 2026." + +Tuy nhiên, sự "thích nghi" này không có nghĩa là ổn định, mà là sự chấp nhận rủi ro cao hơn. Nhà đầu tư vẫn duy trì sự bi quan nhưng coi đó là dấu hiệu tích cực theo cách ngược lại. + +> "Oliver Pursche phân tích thị trường chứng khoán Mỹ và coi sự bi quan cao của nhà đầu tư là dấu hiệu ngược lại đáng tin cậy." + +Kết quả là, thị trường tồn tại trong trạng thái căng thẳng cao độ, nơi mà bất kỳ sự thay đổi nào về đàm phán hòa bình hoặc leo thang quân sự đều có thể kích hoạt các điểm gãy cung cầu. + +> "Các nhà phân tích cho thấy sự sụp đổ của các cuộc đàm phán hòa bình là rủi ro tăng giá cho thị trường." + +Tóm lại, tương lai của thị trường dầu mỏ và tài chính được định hình bởi sự biến động cấu trúc do địa chính trị, sự phân hóa giữa các tác nhân giao dịch ngắn hạn và các chuyên gia dài hạn, cùng với sự thích nghi cảm xúc của thị trường trước một môi trường rủi ro kéo dài. + +## Phản ứng Phân hóa của Các Tác nhân: Từ Đặt cược Đòn bẩy đến Sự Thận trọng Chiến lược + +Chương này đi sâu vào sự phân hóa rõ rệt trong cách các tác nhân thị trường phản ứng với cú sốc địa chính trị. Trong khi một nhóm nhà giao dịch mạo hiểm sử dụng đòn bẩy để đặt cược vào các kết quả ngoại giao ngắn hạn, thì các chuyên gia và tổ chức tài chính lớn lại rơi vào trạng thái thận trọng chiến lược, thậm chí là "bay mù" trước thực tế nguồn cung vật lý bị thắt chặt nghiêm trọng. + +**Nhà giao dịch đòn bẩy: Đặt cược vào hòa bình và trả giá đắt** + +Trong giai đoạn đầu của cuộc xung đột, tâm lý thị trường bị chi phối mạnh mẽ bởi hy vọng về các thỏa thuận ngoại giao. Các nhà giao dịch hàng hóa, đặc biệt là những người sử dụng đòn bẩy cao, đã tập trung vào khả năng đạt được một thỏa thuận hòa bình lâu dài, dẫn đến những vị thế đầu cơ rủi ro. + +> "Giá dầu đã nằm dưới mức 100 đô la một thùng, với các nhà giao dịch chú ý đến hy vọng về một thỏa thuận hòa bình lâu dài." + +Tuy nhiên, sự lạc quan này đã dẫn đến những tổn thất nặng nề khi thực tế nguồn cung vật lý không thể phục hồi nhanh chóng. Các nhà giao dịch đã thất bại trong việc dự đoán tác động của việc Iran phong tỏa Eo biển Hormuz, một kịch bản trước đó được coi là khó xảy ra. + +> "Các nhà giao dịch hàng hóa đã chịu thiệt hại hàng tỷ đô la vì họ đã đặt cược sai hướng về giá dầu." + +Dữ liệu giao dịch cho thấy sự đầu cơ mạnh mẽ ngay trước các sự kiện địa chính trị quan trọng. + +> "Nhà đầu tư đã đặt cược trị giá khoảng 760 triệu đô la vào việc giá dầu giảm khoảng 20 phút trước khi Iran công bố việc mở lại Eo biển Hormuz." + +> "Nhà đầu tư đã đặt cược 950 triệu đô la vào giá dầu chỉ vài giờ trước khi Hoa Kỳ và Iran tuyên bố ngừng bắn." + +Những hoạt động giao dịch này đã thu hút sự chú ý của cơ quan quản lý. + +> "Ủy ban Giao dịch Hàng hóa Tương lai Hoa Kỳ (CFTC) đang xem xét một loạt các giao dịch trong hợp đồng tương lai dầu mỏ được thực hiện ngay trước những thay đổi lớn trong chính sách chiến tranh Iran của Tổng thống Donald Trump." + +**Chuyên gia và Tổ chức: Sự thận trọng chiến lược và "Bay mù"** + +Ngược lại với sự đầu cơ ngắn hạn, các chuyên gia phân tích và tổ chức quốc tế nhận ra rằng tình hình nguồn cung nghiêm trọng hơn nhiều so với những gì thị trường tương lai đang định giá. Họ rơi vào trạng thái thận trọng chiến lược, cảnh báo về những rủi ro dài hạn mà các nhà giao dịch ngắn hạn bỏ qua. + +> "Tình trạng thắt chặt nguồn cung dầu thô nghiêm trọng hơn nhiều so với những gì các nhà giao dịch tương lai có thể giả định." + +Các nhà phân tích từ Wood Mackenzie đã chỉ ra rằng việc phục hồi nguồn cung sẽ mất nhiều thời gian hơn dự kiến, ngay cả khi có lệnh ngừng bắn. + +> "Các nhà phân tích của Wood Mackenzie chỉ ra rằng dầu thô sẽ mất hai đến ba tuần để đến châu Âu sau khi lưu lượng tàu thuyền dọc theo Eo biển Hormuz trở lại bình thường." + +Sự thiếu hiểu biết về tác động thực sự của việc mất nguồn tài nguyên năng lượng cũng dẫn đến một sự đánh giá thấp rủi ro từ phía các tổ chức tài chính lớn. + +> "Các nhà kinh tế trong hầu hết các tổ chức tài chính đánh giá thấp tác động của việc mất nguồn tài nguyên năng lượng. Khái niệm sai lầm này giải thích tại sao các nhà kinh tế... và do đó nhà đầu tư, không đặc biệt lo lắng." + +Tuy nhiên, ngay cả các chuyên gia cũng thừa nhận sự bất định cao độ. + +> "Tom Kloza cho biết mọi người đang cố gắng đánh giá thị trường dầu mỏ sẽ trông như thế nào trong 'ngày hôm sau', nhưng mọi người đều đang bay mù." + +**Tâm lý nhà đầu tư: Thích nghi cảm xúc và tập trung vào cơ bản** + +Dưới áp lực của biến động giá, tâm lý nhà đầu tư cá nhân và tổ chức đã trải qua quá trình thích nghi cảm xúc. Ban đầu, thị trường phản ứng mạnh với mọi tuyên bố từ Nhà Trắng, nhưng dần dần trở nên mệt mỏi và tập trung vào các yếu tố cơ bản. + +> "Thị trường mệt mỏi vì chiến tranh lọc bỏ tiếng ồn." + +Mức độ bi quan cao của nhà đầu tư được một số chuyên gia coi là dấu hiệu ngược lại đáng tin cậy. + +> "Oliver Pursche phân tích thị trường chứng khoán Mỹ và coi sự bi quan cao của nhà đầu tư là dấu hiệu ngược lại đáng tin cậy." + +Khi giá dầu được định giá ổn định hơn, nhà đầu tư bắt đầu quay trở lại tập trung vào kết quả kinh doanh và dữ liệu kinh tế vĩ mô. + +> "Nhà đầu tư đang tập trung lại vào kết quả kinh doanh và các yếu tố cơ bản của nền kinh tế." + +Sự phân hóa này cho thấy một thị trường đang trong quá trình chuyển dịch từ phản ứng cảm xúc ngắn hạn sang một trạng thái thận trọng dài hạn, nơi rủi ro hệ thống từ sự thích nghi cảm xúc và điểm gãy cung cầu vẫn là mối đe dọa tiềm tàng. + +## Xu hướng Nổi lên và Rủi ro Hệ thống: Bẫy Thông tin và Điểm gãy Cung cầu + +Chương này khám phá sự hình thành của một "bẫy thông tin" sâu sắc và điểm gãy cung cầu đang định hình lại cấu trúc thị trường năng lượng trong tương lai. Kết quả mô phỏng cho thấy thị trường không còn phản ứng đơn thuần với tin tức, mà đang đối mặt với một nghịch lý hệ thống: sự lạc quan ngoại giao trên bề mặt đang che giấu một thực tế nguồn cung vật lý bị tổn thương nghiêm trọng và khó phục hồi. + +**Bẫy Thông tin: Sự Mâu thuẫn giữa Hy vọng Ngoại giao và Thực tế Vật lý** + +Trong môi trường tương lai được mô phỏng, thị trường dầu mỏ rơi vào trạng thái phân cực thông tin nghiêm trọng. Một mặt, các tuyên bố từ cấp cao chính trị, đặc biệt là từ Nhà Trắng, liên tục tạo ra những đợt hy vọng về hòa bình. Mặt khác, dữ liệu hậu cần và dòng chảy vật lý cho thấy một bức tranh u ám hơn nhiều. + +> "Giá dầu tăng nhẹ khoảng 1% khi thị trường tập trung nhiều hơn vào gián đoạn nguồn cung và hạn chế vận chuyển hơn là các bình luận của Donald Trump rằng cuộc chiến với Iran sắp kết thúc." + +Sự phân hóa này tạo ra một bẫy thông tin, nơi các nhà giao dịch ngắn hạn dễ dàng bị cuốn theo các tín hiệu ngoại giao, trong khi bỏ qua các rủi ro cấu trúc. + +> "Thị trường diễn giải lệnh phong tỏa Eo biển Hormuz của Tổng thống Trump là một chiến thuật đàm phán nhằm cắt đứt doanh thu dầu mỏ của Iran, chứ không phải là một sự leo thang thực sự." + +Tuy nhiên, theo thời gian, thị trường bắt đầu thể hiện dấu hiệu của sự mệt mỏi và thích nghi. + +> "Morning Bid thảo luận về cách các thị trường mệt mỏi vì chiến tranh lọc bỏ tiếng ồn." + +Điều này cho thấy một sự dịch chuyển tâm lý: nhà đầu tư dần nhận ra rằng các tuyên bố chính trị không thể thay đổi ngay lập tức thực tế của các đường ống bị phá hủy và tàu thuyền bị mắc kẹt. + +**Điểm gãy Cung cầu: Sự Phục hồi Không đồng đều và "Phụ phí Dư thừa"** + +Ngay cả trong kịch bản lạc quan nhất khi lệnh ngừng bắn được ký kết và Eo biển Hormuz mở cửa trở lại, mô phỏng cho thấy nguồn cung dầu mỏ sẽ không thể phục hồi nhanh chóng. Đây chính là "điểm gãy cung cầu" mới, nơi mà sự thiếu hụt không chỉ đến từ xung đột, mà còn từ sự suy yếu của toàn bộ chuỗi cung ứng hạ tầng. + +Các chuyên gia trong mô phỏng đã chỉ ra những rào cản vật lý khổng lồ: + +> "Karan Satwani nhận thấy tình trạng thiếu hụt thiết bị sẵn có và nhân công chuyên biệt sẵn sàng triển khai đến các cơ sở hạ tầng năng lượng bị hư hỏng, bất kể cuộc chiến kết thúc vào khi nào." + +Hơn nữa, hậu quả của việc phong tỏa tạo ra một nút thắt logistics khó giải quyết. + +> "Joe DeLaura cho biết về việc mất sản lượng vĩnh viễn từ các giếng dầu bị đóng cửa ở Saudi, Kuwait, UAE và Iraq, thiệt hại nhà máy lọc dầu và đường ống, cùng thời gian khởi động lại vật lý, bên cạnh hàng đợi hơn 800 tàu chở dầu bị mắc kẹt ở phía tây Eo biển." + +Kết quả là, thị trường sẽ không còn định giá theo kịch bản "ngừng cung hoàn toàn", nhưng cũng không thể trở lại trạng thái bình thường cũ. + +> "Các nhà phân tích của Gelber nhận định kết quả là một thị trường không còn định giá một sự gián đoạn quy mô lớn, nhưng vẫn duy trì một phụ phí dư thừa khi dòng chảy phục hồi không đồng đều thay vì bật trở lại bình thường tại Eo biển Hormuz." + +**Rủi ro Hệ thống: Giao dịch Nội bộ và Sự can thiệp của Cơ quan Quản lý** + +Sự chênh lệch thông tin giữa các nhà hoạch định chính sách và thị trường giao dịch đã kích hoạt các rủi ro hệ thống về đạo đức và tuân thủ. Mô phỏng cho thấy sự xuất hiện của các giao dịch "đúng thời điểm" đáng ngờ, dẫn đến sự can thiệp mạnh mẽ từ các cơ quan quản lý. + +> "Ủy ban Giao dịch Hàng hóa Tương lai Hoa Kỳ (CFTC) đang xem xét một loạt các giao dịch trong hợp đồng tương lai dầu mỏ được thực hiện ngay trước những thay đổi lớn trong chính sách chiến tranh Iran của Tổng thống Donald Trump." + +Điều này cho thấy thông tin địa chính trị nhạy cảm đang bị lợi dụng, tạo ra một lớp rủi ro pháp lý mới cho thị trường tài chính. + +> "Dữ liệu được yêu cầu từ các sàn giao dịch về giao dịch đúng thời điểm trong thị trường dầu mỏ bao gồm các định danh Tag 50 của các thực thể đứng sau các giao dịch." + +Sự can thiệp này không chỉ nhằm trừng phạt cá nhân, mà còn là nỗ lực để khôi phục niềm tin vào tính minh bạch của thị trường khi nó đang bị xói mòn bởi "bẫy thông tin" và sự không chắc chắn địa chính trị. + +**Xu hướng Giá và Rủi ro Ngầm** + +Tổng hợp lại, xu hướng giá trong tương lai này sẽ không giảm sâu ngay cả khi có tin tức hòa bình. Sự phục hồi của giá sẽ được hỗ trợ bởi "phụ phí dư thừa" và thực tế nguồn cung bị thắt chặt dài hạn. + +> "Các nhà phân tích cho biết rủi ro tăng giá chính của thị trường là sự sụp đổ của các cuộc đàm phán hòa bình giữa Hoa Kỳ và Iran, khi các yêu cầu của hai bên vẫn còn cách xa nhau." + +Thị trường đang bước vào giai đoạn mới, nơi mà giá dầu không chỉ phản ánh cung cầu tức thời, mà còn là thước đo của sự phục hồi hạ tầng và độ tin cậy của thông tin địa chính trị. + diff --git a/test_code_backend/full_pipeline/README.md b/test_code_backend/full_pipeline/README.md new file mode 100644 index 00000000..e9894299 --- /dev/null +++ b/test_code_backend/full_pipeline/README.md @@ -0,0 +1,262 @@ +# Full Pipeline Runner + +Chạy toàn bộ MiroFish pipeline từ command line, không cần mở UI. + +## Files + +| File | Vai trò | +|---|---| +| `prepare_input.py` | Đọc config, load & filter articles, ghi combined text ra temp file, in JSON metadata ra stdout | +| `run.sh` | Gọi `prepare_input.py`, khởi động backend nếu cần, rồi gọi tuần tự các API | +| `config_articles.env` | Cấu hình input | + + +## Cấu hình (`config_articles.env`) + +| Tham số | Bắt buộc | Mô tả | +|---|---|---| +| `article_path` | Có | Đường dẫn thư mục chứa `articles.jsonl` và thư mục `news/` | +| `full_content` | Không | `True` = giữ toàn bộ markdown; `False` (mặc định) = chỉ lấy body bài viết, bỏ YAML frontmatter | +| `start_time` | Không | Lọc bài viết từ ngày (`YYYY-MM-DD`), để trống = không lọc | +| `end_time` | Không | Lọc bài viết đến ngày (`YYYY-MM-DD`), để trống = không lọc | +| `simulation_requirement` | Không | Mô tả yêu cầu phân tích/dự báo gửi cho LLM | +| `project_name` | Không | Tên project | + +--- + +## Cách 1 — Chạy toàn bộ pipeline 1 lệnh (khuyến nghị) + +**Không cần mở terminal riêng cho backend.** `run.sh` tự kiểm tra và khởi động Flask nếu chưa chạy. + +Chạy 7 bước tự động: ontology → graph → create sim → prepare sim → start sim → chờ hoàn thành → generate report. + +```bash +# Từ project root +cd /home/anman/intern/MiroFish +bash test_code_backend/full_pipeline/run.sh + +# Hoặc chỉ định config khác +bash test_code_backend/full_pipeline/run.sh path/to/config.env +``` + +Log in ra terminal từng bước. Cuối cùng in ra `project_id`, `graph_id`, `simulation_id`, `report_id`. + +> Backend được tự động khởi động trong background — **không cần mở terminal riêng**. Log đầy đủ nằm ở `backend/logs/YYYY-MM-DD.log`. + +Muốn kill port +```bash +kill $(lsof -ti :5001) +``` +--- + +## Cách 2 — Chạy từng bước thủ công + +Khi cần debug hoặc muốn kiểm tra kết quả từng bước. + +### Bước 0: Khởi động Flask backend (port 5001) + +Mở **terminal riêng**, giữ nguyên trong suốt quá trình: + +```bash +cd /home/anman/intern/MiroFish +FLASK_PORT=5001 npm run backend +# hoặc: python backend/run.py +``` + +Kiểm tra backend đã sẵn sàng: +```bash +curl http://localhost:5001/health +# → {"status": "ok", "service": "MiroFish Backend"} +``` + +### Bước 1: Chuẩn bị input (Python) + +Chạy trong **terminal khác**: + +```bash +cd /home/anman/intern/MiroFish +python3 test_code_backend/full_pipeline/prepare_input.py \ + test_code_backend/full_pipeline/config_articles.env +``` + +Output là JSON, ví dụ: +```json +{ + "combined_file": "/tmp/mirofish_articles_abc123.md", + "article_count": 42, + "project_name": "Oil Market Sentiment Pipeline", + "simulation_requirement": "Analyze how ...", + "start_time": null, + "end_time": null +} +``` + +Lưu lại đường dẫn `combined_file` để dùng ở bước tiếp theo. + +--- + +### Bước 2: Generate ontology → lấy `project_id` + +```bash +COMBINED_FILE="/home/anman/intern/MiroFish/test_code_backend/full_pipeline/data/articles_combined.md" + +curl -s http://localhost:5001/api/graph/ontology/generate \ + -F "files=@${COMBINED_FILE};filename=articles.md" \ + -F "simulation_requirement=Analyze how these oil and financial market news articles affect investor sentiment and market dynamics. Predict how different market participants (traders, analysts, retail investors) will react and what the overall price trend will be." \ + -F "project_name=Oil Market Pipeline" \ + | jq '{project_id: .data.project_id, entity_types: [.data.ontology.entity_types[].name]}' +``` + +Lưu lại `project_id`. + +--- + +### Bước 3: Build knowledge graph → lấy `task_id` + +```bash +PROJECT_ID="proj_xxxxxxxxxxxx" + +curl -s http://localhost:5001/api/graph/build \ + -H "Content-Type: application/json" \ + -d "{\"project_id\": \"$PROJECT_ID\"}" \ + | jq '{task_id: .data.task_id}' +``` + +Poll tiến độ build (chạy lặp lại đến khi `status=completed`): + +```bash +TASK_ID="task_xxxxxxxxxxxx" + +curl -s http://localhost:5001/api/graph/task/$TASK_ID \ + | jq '{status: .data.status, progress: .data.progress, graph_id: .data.result.graph_id}' +``` + +Khi `status=completed`, lưu lại `graph_id`. + +--- + +### Bước 4: Tạo simulation → lấy `simulation_id` + +```bash +PROJECT_ID="proj_xxxxxxxxxxxx" + +curl -s http://localhost:5001/api/simulation/create \ + -H "Content-Type: application/json" \ + -d "{\"project_id\": \"$PROJECT_ID\", \"enable_twitter\": true, \"enable_reddit\": true}" \ + | jq '{simulation_id: .data.simulation_id}' +``` + +--- + +### Bước 5: Prepare simulation (sinh agent profiles + config) + +```bash +SIM_ID="sim_xxxxxxxxxxxx" + +curl -s http://localhost:5001/api/simulation/prepare \ + -H "Content-Type: application/json" \ + -d "{\"simulation_id\": \"$SIM_ID\", \"use_llm_for_profiles\": true, \"parallel_profile_count\": 5}" \ + | jq '{task_id: .data.task_id, status: .data.status}' +``` + +Poll tiến độ (bước này lâu nhất, ~5–15 phút tùy số agents): + +```bash +TASK_ID="task_xxxxxxxxxxxx" + +curl -s http://localhost:5001/api/simulation/prepare/status \ + -H "Content-Type: application/json" \ + -d "{\"task_id\": \"$TASK_ID\", \"simulation_id\": \"$SIM_ID\"}" \ + | jq '{status: .data.status, progress: .data.progress, message: .data.message}' +``` + +--- + +### Bước 6: Start simulation + +```bash +SIM_ID="sim_xxxxxxxxxxxx" + +curl -s http://localhost:5001/api/simulation/start \ + -H "Content-Type: application/json" \ + -d "{\"simulation_id\": \"$SIM_ID\", \"platform\": \"parallel\"}" \ + | jq '{runner_status: .data.runner_status, total_rounds: .data.total_rounds}' +``` + +--- + +### Bước 7: Theo dõi simulation đến khi hoàn thành + +Chạy lệnh sau để tự động poll mỗi 30 giây, tự thoát khi simulation kết thúc: + +```bash +SIM_ID="sim_xxxxxxxxxxxx" + +while true; do + RESP=$(curl -s http://localhost:5001/api/simulation/$SIM_ID/run-status) + RS=$(echo "$RESP" | jq -r '.data.runner_status') + CR=$(echo "$RESP" | jq -r '.data.current_round') + TR=$(echo "$RESP" | jq -r '.data.total_rounds') + echo "[$(date '+%H:%M:%S')] $RS round=$CR/$TR" + [[ "$RS" == "completed" || "$RS" == "stopped" || "$RS" == "failed" ]] && break + sleep 30 +done +``` + +Hoặc kiểm tra thủ công một lần: + +```bash +# Trạng thái tổng quan (round hiện tại, % tiến độ) +curl -s http://localhost:5001/api/simulation/$SIM_ID/run-status \ + | jq '{runner_status: .data.runner_status, current_round: .data.current_round, total_rounds: .data.total_rounds}' + +# Hành động của agents (real-time) +curl -s "http://localhost:5001/api/simulation/$SIM_ID/actions?limit=20" | jq . +``` + +--- + +### Bước 8: Generate report + +```bash +SIM_ID="sim_xxxxxxxxxxxx" + +SIM_ID="sim_07ab325b3818" +# 1. Bắt đầu generate → lấy task_id và report_id +RESP=$(curl -s http://localhost:5001/api/report/generate \ + -H "Content-Type: application/json" \ + -d "{\"simulation_id\": \"$SIM_ID\"}") +TASK_ID=$(echo "$RESP" | jq -r '.data.task_id') +REPORT_ID=$(echo "$RESP" | jq -r '.data.report_id') +echo "task_id=$TASK_ID report_id=$REPORT_ID" + +# 2. Poll đến khi completed (chạy lặp lại, ~5–15 phút) +while true; do + R=$(curl -s http://localhost:5001/api/report/generate/status \ + -H "Content-Type: application/json" \ + -d "{\"task_id\": \"$TASK_ID\"}") + ST=$(echo "$R" | jq -r '.data.status') + PG=$(echo "$R" | jq -r '.data.progress') + echo "$ST ${PG}%" + [[ "$ST" == "completed" ]] && break + [[ "$ST" == "failed" ]] && { echo "FAILED"; break; } + sleep 15 +done + +# 3. Download file Markdown +curl -s http://localhost:5001/api/report/$REPORT_ID/download -o report.md +echo "Saved to report.md" +``` + +--- + +## Output files + +| Loại | Đường dẫn | +|---|---| +| Project metadata + extracted text | `backend/uploads/projects//` | +| Agent profiles (Reddit) | `backend/uploads/simulations//reddit_profiles.json` | +| Agent profiles (Twitter) | `backend/uploads/simulations//twitter_profiles.csv` | +| Simulation config (LLM-generated) | `backend/uploads/simulations//simulation_config.json` | +| Run state & action log | `backend/uploads/simulations//run_state.json` | +| Report | `backend/uploads/reports//` | diff --git a/test_code_backend/full_pipeline/config_articles.env b/test_code_backend/full_pipeline/config_articles.env new file mode 100644 index 00000000..30cd28cc --- /dev/null +++ b/test_code_backend/full_pipeline/config_articles.env @@ -0,0 +1,20 @@ +# ─── Article input ──────────────────────────────────────────────────────────── +# Path to the folder containing articles.jsonl (and the news/ sub-folder) +article_path = "/home/anman/intern/MiroFish/data" + +# If True → use full markdown (frontmatter + body) +# If False → strip YAML frontmatter, keep article body only +full_content = False + +# Date filter (YYYY-MM-DD). Leave blank to load all dates. +# start_time = 2026-02-09 +# end_time = 2026-02-12 +start_time = 2026-04-13 +end_time = 2026-04-16 + +# ─── Simulation settings ────────────────────────────────────────────────────── +# What the simulation should analyse / predict (sent to the LLM) +simulation_requirement = "Analyze how these oil and financial market news articles affect investor sentiment and market dynamics. Predict how different market participants (traders, analysts, retail investors) will react and what the overall price trend will be." + +# Display name for the project (optional) +project_name = "Oil Market Sentiment Pipeline" diff --git a/test_code_backend/full_pipeline/prepare_input.py b/test_code_backend/full_pipeline/prepare_input.py new file mode 100644 index 00000000..f626cc39 --- /dev/null +++ b/test_code_backend/full_pipeline/prepare_input.py @@ -0,0 +1,152 @@ +#!/usr/bin/env python3 +""" +Prepare article input for MiroFish pipeline. + +Reads config_articles.env, loads and filters articles from articles.jsonl, +combines their text, and writes the result to a temp file. + +Outputs a single JSON to stdout: + { + "combined_file": "/home/anman/intern/MiroFish/test_code_backend/full_pipeline/data/articles_combined.md", + "article_count": 42, + "project_name": "...", + "simulation_requirement": "...", + "start_time": "2025-11-01", # or null + "end_time": "2025-11-30" # or null + } + +Usage: + python prepare_input.py [config_articles.env] +""" + +import json +import os +import sys +from datetime import datetime +from pathlib import Path + + +# ─── Config parser ──────────────────────────────────────────────────────────── +path_output = "/home/anman/intern/MiroFish/test_code_backend/full_pipeline/data" + +def parse_config(path: str) -> dict: + """Parse 'key = value' config file (spaces around = allowed, # = comment). + Values may optionally be wrapped in single or double quotes.""" + cfg = {} + with open(path, encoding="utf-8") as f: + for line in f: + line = line.strip() + if not line or line.startswith("#") or "=" not in line: + continue + key, _, value = line.partition("=") + key = key.strip() + value = value.strip().strip('"').strip("'") + if value: + cfg[key] = value + return cfg + + +# ─── Article loading ────────────────────────────────────────────────────────── + +def strip_frontmatter(content: str) -> str: + """Remove YAML frontmatter (--- … ---), return article body only.""" + if content.startswith("---"): + end = content.find("---", 3) + if end != -1: + return content[end + 3:].lstrip("\n") + return content + + +def load_articles(article_path: str, full_content: bool, + start_time: str, end_time: str) -> list: + jsonl_path = os.path.join(article_path, "articles.jsonl") + if not os.path.exists(jsonl_path): + raise FileNotFoundError(f"articles.jsonl not found: {jsonl_path}") + + start_dt = datetime.strptime(start_time, "%Y-%m-%d").date() if start_time else None + end_dt = datetime.strptime(end_time, "%Y-%m-%d").date() if end_time else None + + articles = [] + with open(jsonl_path, encoding="utf-8") as f: + for line in f: + line = line.strip() + if not line: + continue + record = json.loads(line) + date_str = record.get("published_date", "") + file_rel = record.get("file_path", "") + if not date_str or not file_rel: + continue + + try: + date = datetime.strptime(date_str, "%Y-%m-%d").date() + except ValueError: + continue + + if start_dt and date < start_dt: + continue + if end_dt and date > end_dt: + continue + + full_path = os.path.join(article_path, file_rel) + if not os.path.exists(full_path): + continue + + with open(full_path, encoding="utf-8") as mdf: + raw = mdf.read() + + content = raw if full_content else strip_frontmatter(raw) + articles.append({"date": date_str, "file_path": file_rel, "content": content}) + + articles.sort(key=lambda a: a["date"]) + return articles + + +# ─── Main ───────────────────────────────────────────────────────────────────── + +def main(): + script_dir = Path(__file__).parent + config_path = sys.argv[1] if len(sys.argv) > 1 else str(script_dir / "config_articles.env") + + cfg = parse_config(config_path) + + article_path = cfg.get("article_path", "") + full_content = cfg.get("full_content", "False").strip().lower() in ("true", "1", "yes") + start_time = cfg.get("start_time") or "" + end_time = cfg.get("end_time") or "" + sim_req = cfg.get( + "simulation_requirement", + "Analyze how these news articles affect market sentiment and predict participant behavior.", + ) + project_name = cfg.get("project_name", "MiroFish Pipeline") + + if not article_path: + print(json.dumps({"error": "article_path is required"})) + sys.exit(1) + + articles = load_articles(article_path, full_content, start_time, end_time) + if not articles: + print(json.dumps({"error": "No articles found for the given parameters"})) + sys.exit(1) + + # Write combined text to path_output + os.makedirs(path_output, exist_ok=True) + out_path = os.path.join(path_output, "articles_combined.md") + with open(out_path, "w", encoding="utf-8") as f: + for art in articles: + f.write(f"\n\n=== {art['file_path']} ({art['date']}) ===\n\n") + f.write(art["content"]) + + print(json.dumps({ + "combined_file": out_path, + "article_count": len(articles), + "project_name": project_name, + "simulation_requirement": sim_req, + "start_time": start_time or None, + "end_time": end_time or None, + "word_count": sum(len(art["content"].split()) for art in articles) + }, ensure_ascii=False)) + + +if __name__ == "__main__": + main() diff --git a/test_code_backend/full_pipeline/run.sh b/test_code_backend/full_pipeline/run.sh new file mode 100644 index 00000000..d6b7ec28 --- /dev/null +++ b/test_code_backend/full_pipeline/run.sh @@ -0,0 +1,217 @@ +#!/usr/bin/env bash +# MiroFish full pipeline runner — calls prepare_input.py then MiroFish API. +# +# Usage: +# bash run.sh +# bash run.sh path/to/config.env +# +# Requirements: curl, jq, python3 + +set -euo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +PROJECT_ROOT="$(cd "$SCRIPT_DIR/../.." && pwd)" +BACKEND_DIR="$PROJECT_ROOT/backend" +CONFIG_FILE="${1:-$SCRIPT_DIR/config_articles.env}" +BACKEND_URL="http://localhost:5001" + +info() { echo "[$(date '+%H:%M:%S')] INFO $*"; } +err() { echo "[$(date '+%H:%M:%S')] ERROR $*" >&2; } +die() { err "$*"; exit 1; } + +command -v curl >/dev/null 2>&1 || die "curl is required" +command -v jq >/dev/null 2>&1 || die "jq is required (apt install jq)" +command -v python3 >/dev/null 2>&1 || die "python3 is required" + +[[ -f "$CONFIG_FILE" ]] || die "Config file not found: $CONFIG_FILE" + +# ─── Step 0: Prepare input (Python) ────────────────────────────────────────── +info "=== Step 0: Preparing article input ===" +INPUT_JSON=$(python3 "$SCRIPT_DIR/prepare_input.py" "$CONFIG_FILE") \ + || die "prepare_input.py failed" + +# Check for error from prepare_input.py +if echo "$INPUT_JSON" | jq -e '.error' >/dev/null 2>&1; then + die "Input error: $(echo "$INPUT_JSON" | jq -r '.error')" +fi + +COMBINED_FILE=$(echo "$INPUT_JSON" | jq -r '.combined_file') +ARTICLE_COUNT=$(echo "$INPUT_JSON" | jq -r '.article_count') +PROJECT_NAME=$( echo "$INPUT_JSON" | jq -r '.project_name') +SIM_REQ=$( echo "$INPUT_JSON" | jq -r '.simulation_requirement') + +trap 'rm -f "$COMBINED_FILE"' EXIT # cleanup temp file on exit + +info " articles : $ARTICLE_COUNT" +info " project_name : $PROJECT_NAME" +info " combined_file : $COMBINED_FILE ($(wc -c < "$COMBINED_FILE" | tr -d ' ') bytes)" + +# ─── Start Flask backend (if not already running) ───────────────────────────── +if curl -sf "$BACKEND_URL/health" >/dev/null 2>&1; then + info "Backend already running at $BACKEND_URL" +else + info "Starting Flask backend..." + VENV="$BACKEND_DIR/.venv" + PYTHON=$( [[ -f "$VENV/bin/python" ]] && echo "$VENV/bin/python" || echo "python3" ) + + "$PYTHON" "$BACKEND_DIR/run.py" >/dev/null 2>&1 & + + info "Waiting for backend to be ready..." + for i in $(seq 1 30); do + sleep 2 + curl -sf "$BACKEND_URL/health" >/dev/null 2>&1 && { info "Backend ready"; break; } + [[ $i -eq 30 ]] && die "Backend did not start (check backend/logs/)" + done +fi + +# ─── Helper: assert API response has success=true ───────────────────────────── +ok() { echo "$1" | jq -r '.success // false'; } +assert_ok() { [[ "$(ok "$1")" == "true" ]] || { err "Step '$2' failed:"; echo "$1" | jq . >&2; die "Aborted"; }; } + +# ─── Helper: poll GET task until completed/failed ───────────────────────────── +poll_task() { + local url="$1" label="$2" + while true; do + local body; body=$(curl -sf "$url") || { sleep 10; continue; } + local st msg + st=$(echo "$body" | jq -r '.data.status // empty') + pg=$(echo "$body" | jq -r '.data.progress // 0') + msg=$(echo "$body"| jq -r '.data.message // empty') + info " [$label] $st ${pg}% — $msg" + [[ "$st" == "completed" ]] && return 0 + [[ "$st" == "failed" ]] && { err "$label failed: $msg"; die "Aborted"; } + sleep 10 + done +} + +# ─── Step 1: Generate ontology ──────────────────────────────────────────────── +info "=== Step 1/7: Generating ontology ===" +RESP=$(curl -sf "$BACKEND_URL/api/graph/ontology/generate" \ + -F "files=@${COMBINED_FILE};filename=articles.md" \ + -F "simulation_requirement=${SIM_REQ}" \ + -F "project_name=${PROJECT_NAME}") +assert_ok "$RESP" "ontology/generate" + +PROJECT_ID=$(echo "$RESP" | jq -r '.data.project_id') +info " project_id : $PROJECT_ID" + +# ─── Step 2: Build knowledge graph ──────────────────────────────────────────── +info "=== Step 2/7: Building Zep knowledge graph ===" +RESP=$(curl -sf "$BACKEND_URL/api/graph/build" \ + -H "Content-Type: application/json" \ + -d "$(jq -n --arg p "$PROJECT_ID" --arg n "$PROJECT_NAME" \ + '{project_id: $p, graph_name: $n}')") +assert_ok "$RESP" "graph/build" + +TASK_ID=$(echo "$RESP" | jq -r '.data.task_id') +info " task_id : $TASK_ID" +poll_task "$BACKEND_URL/api/graph/task/$TASK_ID" "graph/build" + +GRAPH_ID=$(curl -sf "$BACKEND_URL/api/graph/task/$TASK_ID" | jq -r '.data.result.graph_id') +info " graph_id : $GRAPH_ID" + +# ─── Step 3: Create simulation ──────────────────────────────────────────────── +info "=== Step 3/7: Creating simulation ===" +RESP=$(curl -sf "$BACKEND_URL/api/simulation/create" \ + -H "Content-Type: application/json" \ + -d "$(jq -n --arg p "$PROJECT_ID" \ + '{project_id: $p, enable_twitter: true, enable_reddit: true}')") +assert_ok "$RESP" "simulation/create" + +SIM_ID=$(echo "$RESP" | jq -r '.data.simulation_id') +info " simulation_id : $SIM_ID" + +# ─── Step 4: Prepare simulation ─────────────────────────────────────────────── +info "=== Step 4/7: Preparing simulation (agent profiles + config) ===" +RESP=$(curl -sf "$BACKEND_URL/api/simulation/prepare" \ + -H "Content-Type: application/json" \ + -d "$(jq -n --arg s "$SIM_ID" \ + '{simulation_id: $s, use_llm_for_profiles: true, parallel_profile_count: 5}')") +assert_ok "$RESP" "simulation/prepare" + +if [[ "$(echo "$RESP" | jq -r '.data.already_prepared')" == "true" ]]; then + info " Already prepared, skipping poll" +else + TASK_ID=$(echo "$RESP" | jq -r '.data.task_id') + info " task_id : $TASK_ID" + + while true; do + RESP=$(curl -sf "$BACKEND_URL/api/simulation/prepare/status" \ + -H "Content-Type: application/json" \ + -d "$(jq -n --arg t "$TASK_ID" --arg s "$SIM_ID" \ + '{task_id: $t, simulation_id: $s}')") + st=$(echo "$RESP" | jq -r '.data.status // empty') + pg=$(echo "$RESP" | jq -r '.data.progress // 0') + msg=$(echo "$RESP" | jq -r '.data.message // empty') + info " [prepare] $st ${pg}% — $msg" + [[ "$st" == "completed" || "$st" == "ready" ]] && break + [[ "$st" == "failed" ]] && { err "Prepare failed: $msg"; die "Aborted"; } + sleep 15 + done +fi + +# ─── Step 5: Start simulation ───────────────────────────────────────────────── +info "=== Step 5/7: Starting simulation (platform: parallel) ===" +RESP=$(curl -sf "$BACKEND_URL/api/simulation/start" \ + -H "Content-Type: application/json" \ + -d "$(jq -n --arg s "$SIM_ID" '{simulation_id: $s, platform: "parallel"}')") +assert_ok "$RESP" "simulation/start" + +RUNNER_STATUS=$(echo "$RESP" | jq -r '.data.runner_status') +TOTAL_ROUNDS=$(echo "$RESP" | jq -r '.data.total_rounds') +info " runner_status : $RUNNER_STATUS" +info " total_rounds : $TOTAL_ROUNDS" + +# ─── Step 6: Wait for simulation to finish ────────────────────────────────── +info "=== Step 6/7: Waiting for simulation to finish ===" +info " (monitoring $SIM_ID — Ctrl-C to detach and let it run in background)" +while true; do + RESP=$(curl -sf "$BACKEND_URL/api/simulation/$SIM_ID/run-status") || { sleep 15; continue; } + RS=$(echo "$RESP" | jq -r '.data.runner_status // empty') + CR=$(echo "$RESP" | jq -r '.data.current_round // 0') + TR=$(echo "$RESP" | jq -r '.data.total_rounds // 0') + info " runner_status=$RS round=$CR/$TR" + [[ "$RS" == "completed" || "$RS" == "stopped" ]] && break + [[ "$RS" == "failed" ]] && { err "Simulation failed"; die "Aborted"; } + sleep 30 +done + +# ─── Step 7: Generate report ───────────────────────────────────────────────── +info "=== Step 7/7: Generating report ===" +RESP=$(curl -sf "$BACKEND_URL/api/report/generate" \ + -H "Content-Type: application/json" \ + -d "$(jq -n --arg s "$SIM_ID" '{simulation_id: $s}')") +assert_ok "$RESP" "report/generate" + +REPORT_ID=$(echo "$RESP" | jq -r '.data.report_id') +info " report_id : $REPORT_ID" + +if [[ "$(echo "$RESP" | jq -r '.data.already_generated')" == "true" ]]; then + info " Report already exists, skipping poll" +else + TASK_ID=$(echo "$RESP" | jq -r '.data.task_id') + info " task_id : $TASK_ID" + while true; do + RESP=$(curl -sf "$BACKEND_URL/api/report/generate/status" \ + -H "Content-Type: application/json" \ + -d "$(jq -n --arg t "$TASK_ID" '{task_id: $t}')") || { sleep 10; continue; } + st=$(echo "$RESP" | jq -r '.data.status // empty') + pg=$(echo "$RESP" | jq -r '.data.progress // 0') + msg=$(echo "$RESP" | jq -r '.data.message // empty') + info " [report] $st ${pg}% — $msg" + [[ "$st" == "completed" ]] && break + [[ "$st" == "failed" ]] && { err "Report failed: $msg"; die "Aborted"; } + sleep 15 + done +fi + +# ─── Summary ────────────────────────────────────────────────────────────────── +echo "" +info "============================================================" +info "Pipeline complete" +info " project_id : $PROJECT_ID" +info " graph_id : $GRAPH_ID" +info " simulation_id : $SIM_ID" +info " report_id : $REPORT_ID" +info "============================================================" +info "Download report: curl $BACKEND_URL/api/report/$REPORT_ID/download -o report.md" diff --git a/test_code_backend/gen_ontogogy_and_graph/build_graph.py b/test_code_backend/gen_ontogogy_and_graph/build_graph.py new file mode 100644 index 00000000..9d914b16 --- /dev/null +++ b/test_code_backend/gen_ontogogy_and_graph/build_graph.py @@ -0,0 +1,197 @@ +""" +Test: Knowledge Graph build on Zep Cloud +Input : - output/ontology/ontology_*.json (latest, or pass path as arg) + - news articles (same date range as ontology) +Output: output/graph/ + graph_.json — build summary (node/edge count, timing) + entities_.json — entity detail: raw nodes + filtered result + +Run: + python build_graph.py # auto-picks latest ontology + python build_graph.py output/ontology/ontology_xyz.json # specific ontology file +""" + +import sys +import json +import time +import argparse +from datetime import datetime, date +from pathlib import Path + +# ── Path setup ─────────────────────────────────────────────────────────────── +SCRIPT_DIR = Path(__file__).parent +PROJECT_ROOT = SCRIPT_DIR.parent.parent +BACKEND_DIR = PROJECT_ROOT / "backend" +ONTOLOGY_DIR = SCRIPT_DIR.parent / "output" / "ontology" +GRAPH_DIR = SCRIPT_DIR.parent / "output" / "graph" +NEWS_DIR = PROJECT_ROOT / "data" / "news" / "2026-04" + +sys.path.insert(0, str(BACKEND_DIR)) +GRAPH_DIR.mkdir(parents=True, exist_ok=True) + +# ── Config ─────────────────────────────────────────────────────────────────── +START_DATE = date(2026, 4, 20) +END_DATE = date(2026, 4, 22) +GRAPH_NAME = "MiroFish_Test_Apr2026" +CHUNK_SIZE = 500 +CHUNK_OVERLAP = 50 +BATCH_SIZE = 3 +POLL_TIMEOUT = 600 + + +# ── Helpers ────────────────────────────────────────────────────────────────── +def log(msg: str): + print(f"[{datetime.now().strftime('%H:%M:%S')}] {msg}") + + +def extract_body(md_text: str) -> str: + s = md_text.strip() + if s.startswith("---"): + end = s.find("---", 3) + if end != -1: + return s[end + 3:].strip() + return s + + +def load_combined_text(news_dir: Path, start: date, end: date) -> str: + parts = [] + for md_file in sorted(news_dir.glob("*.md")): + try: + file_date = date.fromisoformat(md_file.stem.split("_")[0]) + except ValueError: + continue + if not (start <= file_date <= end): + continue + body = extract_body(md_file.read_text(encoding="utf-8")) + if body: + parts.append(body) + return "\n\n---\n\n".join(parts) + + +def pick_ontology_file(arg_path: str | None) -> Path: + if arg_path: + p = Path(arg_path) + if not p.is_absolute(): + p = ONTOLOGY_DIR / p + if not p.exists(): + raise FileNotFoundError(f"Ontology file not found: {p}") + return p + + candidates = sorted(ONTOLOGY_DIR.glob("ontology_*.json"), key=lambda f: f.stat().st_mtime) + if not candidates: + raise FileNotFoundError( + f"No ontology_*.json found in {ONTOLOGY_DIR}. Run gen_ontology.py first." + ) + return candidates[-1] # latest + + +# ── Main ───────────────────────────────────────────────────────────────────── +def main(): + parser = argparse.ArgumentParser(description="Build Zep Knowledge Graph from ontology") + parser.add_argument("ontology_file", nargs="?", help="Path to ontology JSON (default: latest in output/)") + args = parser.parse_args() + + from app.services.graph_builder import GraphBuilderService + from app.services.text_processor import TextProcessor + from app.services.zep_entity_reader import ZepEntityReader + + # Load ontology + ontology_path = pick_ontology_file(args.ontology_file) + log(f"Using ontology: {ontology_path.name}") + with open(ontology_path, encoding="utf-8") as f: + payload = json.load(f) + ontology = payload["ontology"] + + # Load news text + log(f"Loading news text: {START_DATE} → {END_DATE}") + combined_text = load_combined_text(NEWS_DIR, START_DATE, END_DATE) + log(f" Text length: {len(combined_text):,} chars") + + svc = GraphBuilderService() + t_total = time.time() + + # Create graph + log("Creating Zep graph...") + graph_id = svc.create_graph(GRAPH_NAME) + log(f" graph_id = {graph_id}") + + # Apply ontology + log("Applying ontology schema...") + svc.set_ontology(graph_id, ontology) + + # Chunk & upload + chunks = TextProcessor.split_text(combined_text, CHUNK_SIZE, CHUNK_OVERLAP) + log(f"Uploading {len(chunks)} chunks (batch_size={BATCH_SIZE})...") + + episode_uuids = svc.add_text_batches( + graph_id, chunks, BATCH_SIZE, + progress_callback=lambda msg, _: log(f" {msg}"), + ) + log(f" {len(episode_uuids)} episodes uploaded") + + # Wait for processing + log("Waiting for Zep to process episodes...") + svc._wait_for_episodes( + episode_uuids, + progress_callback=lambda msg, _: log(f" {msg}"), + timeout=POLL_TIMEOUT, + ) + + # Fetch result + log("Fetching graph stats...") + graph_info = svc._get_graph_info(graph_id) + elapsed = round(time.time() - t_total, 2) + + log(f"Done in {elapsed}s — nodes={graph_info.node_count}, edges={graph_info.edge_count}") + log(f" Entity types: {graph_info.entity_types}") + + out_path = GRAPH_DIR / f"graph_{graph_id}.json" + with open(out_path, "w", encoding="utf-8") as f: + json.dump({ + "graph_id": graph_id, + "graph_name": GRAPH_NAME, + "built_at": datetime.now().isoformat(), + "elapsed_seconds": elapsed, + "ontology_source": ontology_path.name, + "date_range": {"start": str(START_DATE), "end": str(END_DATE)}, + "chunks_uploaded": len(chunks), + "node_count": graph_info.node_count, + "edge_count": graph_info.edge_count, + "entity_types": graph_info.entity_types, + }, f, ensure_ascii=False, indent=2) + + log(f"Saved → {out_path.name}") + + # ── Entity detail ───────────────────────────────────────────────────────── + # Raw graph data — lưu nguyên output của get_graph_data() + log("Fetching raw graph data...") + graph_data = svc.get_graph_data(graph_id) + log(f" nodes={graph_data['node_count']}, edges={graph_data['edge_count']}") + + # Filtered entities — lưu nguyên output của filter_defined_entities().to_dict() + log("Running entity filter...") + defined_types = [e["name"] for e in ontology.get("entity_types", [])] + reader = ZepEntityReader() + filtered = reader.filter_defined_entities( + graph_id=graph_id, + defined_entity_types=defined_types, + enrich_with_edges=True, + ) + log(f" total={filtered.total_count}, matched={filtered.filtered_count}") + log(f" types found: {sorted(filtered.entity_types)}") + + entities_path = GRAPH_DIR / f"entities_{graph_id}.json" + with open(entities_path, "w", encoding="utf-8") as f: + json.dump({ + "graph_id": graph_id, + "fetched_at": datetime.now().isoformat(), + "ontology_entity_types": defined_types, + "raw_graph_data": graph_data, + "filtered_entities": filtered.to_dict(), + }, f, ensure_ascii=False, indent=2) + + log(f"Saved → {entities_path.name}") + + +if __name__ == "__main__": + main() diff --git a/test_code_backend/gen_ontogogy_and_graph/gen_ontology.py b/test_code_backend/gen_ontogogy_and_graph/gen_ontology.py new file mode 100644 index 00000000..16b3e577 --- /dev/null +++ b/test_code_backend/gen_ontogogy_and_graph/gen_ontology.py @@ -0,0 +1,109 @@ +""" +Test: Ontology generation from news articles +Input : data/news/2026-04/ (date range configured below) +Output: test_code_backend/output/ontology/ontology__.json + +Run: + python gen_ontology.py +""" + +import sys +import json +import time +from datetime import datetime, date +from pathlib import Path + +# ── Path setup ─────────────────────────────────────────────────────────────── +SCRIPT_DIR = Path(__file__).parent +PROJECT_ROOT = SCRIPT_DIR.parent.parent +BACKEND_DIR = PROJECT_ROOT / "backend" +ONTOLOGY_DIR = SCRIPT_DIR.parent / "output" / "ontology" +NEWS_DIR = PROJECT_ROOT / "data" / "news" / "2026-04" + +sys.path.insert(0, str(BACKEND_DIR)) +ONTOLOGY_DIR.mkdir(parents=True, exist_ok=True) + +# ── Config ─────────────────────────────────────────────────────────────────── +START_DATE = date(2026, 4, 20) +END_DATE = date(2026, 4, 22) + +SIMULATION_REQUIREMENT = ( + "Simulate social media public opinion dynamics around global oil/energy markets " + "and US-Iran geopolitical tensions during April 2026. " + "Focus on how governments, corporations, media outlets, financial analysts, and " + "ordinary citizens react and interact on platforms like Twitter and Reddit." +) + + +# ── Helpers ────────────────────────────────────────────────────────────────── +def log(msg: str): + print(f"[{datetime.now().strftime('%H:%M:%S')}] {msg}") + + +def extract_body(md_text: str) -> str: + """Strip YAML frontmatter (---...---), return only article body.""" + s = md_text.strip() + if s.startswith("---"): + end = s.find("---", 3) + if end != -1: + return s[end + 3:].strip() + return s + + +def load_articles(news_dir: Path, start: date, end: date) -> list[dict]: + articles = [] + for md_file in sorted(news_dir.glob("*.md")): + try: + file_date = date.fromisoformat(md_file.stem.split("_")[0]) + except ValueError: + continue + if not (start <= file_date <= end): + continue + body = extract_body(md_file.read_text(encoding="utf-8")) + if body: + articles.append({"filename": md_file.name, "date": str(file_date), "body": body}) + return articles + + +# ── Main ───────────────────────────────────────────────────────────────────── +def main(): + from app.services.ontology_generator import OntologyGenerator + + log(f"Loading news: {START_DATE} → {END_DATE}") + articles = load_articles(NEWS_DIR, START_DATE, END_DATE) + log(f" {len(articles)} articles loaded") + + log("Calling LLM to generate ontology...") + t0 = time.time() + ontology = OntologyGenerator().generate( + document_texts=[a["body"] for a in articles], + simulation_requirement=SIMULATION_REQUIREMENT, + ) + elapsed = round(time.time() - t0, 2) + + entity_names = [e["name"] for e in ontology.get("entity_types", [])] + edge_names = [e["name"] for e in ontology.get("edge_types", [])] + log(f"Done in {elapsed}s") + log(f" Entities ({len(entity_names)}): {entity_names}") + log(f" Edges ({len(edge_names)}): {edge_names}") + + date_range = f"{START_DATE.strftime('%Y%m%d')}-{END_DATE.strftime('%Y%m%d')}" + ts = datetime.now().strftime("%H%M%S") + out_path = ONTOLOGY_DIR / f"ontology_{date_range}_{ts}.json" + + with open(out_path, "w", encoding="utf-8") as f: + json.dump({ + "meta": { + "generated_at": datetime.now().isoformat(), + "elapsed_seconds": elapsed, + "date_range": {"start": str(START_DATE), "end": str(END_DATE)}, + "article_count": len(articles), + }, + "ontology": ontology, + }, f, ensure_ascii=False, indent=2) + + log(f"Saved → {out_path.name}") + + +if __name__ == "__main__": + main() From 5771253b4314c1345cb42fc1f02be94b3f2f208e Mon Sep 17 00:00:00 2001 From: ththu0205 Date: Sun, 24 May 2026 13:20:08 +0000 Subject: [PATCH 4/4] update prompts, comment --- backend/app/prompts/report_agent.py | 1053 +++++++++++++ backend/app/services/report_agent.py | 842 +--------- .../services/simulation_config_generator.py | 1018 ++++++++----- backend/app/services/simulation_ipc.py | 2 +- backend/app/services/simulation_runner.py | 1357 +++++++++-------- backend/app/services/zep_tools.py | 2 +- backend/app/utils/llm_client.py | 2 +- backend/app/utils/llm_cost.py | 1 + backend/scripts/run_parallel_simulation.py | 1156 +++++++------- test_code_backend/calculate_cost.ipynb | 107 ++ test_code_backend/full_pipeline/README.md | 34 + .../full_pipeline/config_articles.env | 2 +- .../gen_ontogogy_and_graph/build_graph.py | 197 --- .../gen_ontogogy_and_graph/gen_ontology.py | 109 -- 14 files changed, 3207 insertions(+), 2675 deletions(-) create mode 100644 backend/app/prompts/report_agent.py create mode 100644 test_code_backend/calculate_cost.ipynb delete mode 100644 test_code_backend/gen_ontogogy_and_graph/build_graph.py delete mode 100644 test_code_backend/gen_ontogogy_and_graph/gen_ontology.py diff --git a/backend/app/prompts/report_agent.py b/backend/app/prompts/report_agent.py new file mode 100644 index 00000000..54fe8cda --- /dev/null +++ b/backend/app/prompts/report_agent.py @@ -0,0 +1,1053 @@ +PLAN_SYSTEM_PROMPT = """\ +Bạn là một chuyên gia viết "Báo cáo Dự báo Tương lai về Thị trường Dầu", với "góc nhìn toàn tri" về thế giới mô phỏng — bạn có thể thấu hiểu hành vi, lời nói, quyết định và tương tác của mọi agent trong mô phỏng liên quan đến thị trường dầu. + +[Khái niệm Cốt lõi] +Chúng tôi đã xây dựng một thế giới mô phỏng thị trường dầu và tiêm các "yêu cầu mô phỏng" cụ thể làm biến số, chẳng hạn như biến động cung cầu dầu, giá dầu thô, sản lượng khai thác, tồn kho, vận tải năng lượng, chính sách của OPEC+, căng thẳng địa chính trị, biến động kinh tế vĩ mô, tỷ giá USD, lãi suất, nhu cầu tiêu thụ năng lượng và phản ứng của các nhóm tham gia thị trường. Kết quả tiến hóa của thế giới mô phỏng là một dự báo về những gì có thể xảy ra trong tương lai của thị trường dầu. Những gì bạn đang quan sát không phải là "dữ liệu thử nghiệm," mà là "bản xem trước của tương lai thị trường dầu." + +[Nhiệm vụ của bạn] +Viết một "Báo cáo Dự báo Tương lai về Thị trường Dầu" để trả lời: +1. Trong điều kiện chúng tôi đặt ra, tương lai của thị trường dầu đã xảy ra điều gì? +2. Các agent (nhóm) khác nhau trong thị trường dầu đã phản ứng và hành động như thế nào? +3. Mô phỏng này tiết lộ những xu hướng và rủi ro tương lai nào đáng chú ý đối với thị trường dầu? + +[Định vị Báo cáo] +- ✅ Đây là báo cáo dự báo tương lai dựa trên mô phỏng, tiết lộ "nếu điều kiện thị trường dầu như thế này, thì thị trường có thể diễn biến như thế nào" +- ✅ Tập trung vào kết quả dự báo: xu hướng giá dầu, biến động cung cầu, phản ứng của các nhóm tham gia thị trường, hiện tượng nổi lên, rủi ro tiềm ẩn +- ✅ Lời nói và hành động của các agent trong thế giới mô phỏng là dự báo về hành vi tương lai của các nhóm liên quan đến thị trường dầu, như nhà sản xuất, nhà tiêu thụ, nhà đầu tư, tổ chức năng lượng, chính phủ, OPEC+, doanh nghiệp vận tải và các bên chịu ảnh hưởng bởi giá dầu +- ❌ Không phải là phân tích tình hình thị trường dầu thế giới thực hiện tại +- ❌ Không phải là tóm tắt tin tức, dư luận hoặc nhận định chung chung về giá dầu + +[Giới hạn Số lượng Chương] +- Tối thiểu 2 chương, tối đa 5 chương +- Chương cuối cùng BẮT BUỘC là chương Kết luận, tổng hợp các phát hiện dự báo và khuyến nghị cốt lõi +- Không cần chương con, viết nội dung hoàn chỉnh trực tiếp cho mỗi chương +- Nội dung nên được tinh gọn, tập trung vào các phát hiện dự báo cốt lõi về thị trường dầu +- Cấu trúc chương do bạn thiết kế dựa trên kết quả dự báo + +Vui lòng xuất cấu trúc báo cáo theo định dạng JSON như sau: +{ + "title": "Tiêu đề Báo cáo", + "summary": "Tóm tắt Báo cáo (một câu tóm tắt các phát hiện dự báo cốt lõi về thị trường dầu)", + "sections": [ + { + "title": "Tiêu đề Chương", + "description": "Mô tả Nội dung Chương" + } + ] +} + +Lưu ý: Mảng sections phải có ít nhất 2 và tối đa 5 phần tử! +""" + + +PLAN_USER_PROMPT_TEMPLATE = """\ +[Cài đặt kịch bản dự báo thị trường Dầu] +Các biến số (yêu cầu mô phỏng) chúng tôi tiêm vào thế giới mô phỏng: {simulation_requirement} + +[Quy mô Thế giới Mô phỏng] +- Số lượng thực thể tham gia mô phỏng: {total_nodes} +- Số lượng quan hệ được tạo giữa các thực thể: {total_edges} +- Phân phối loại thực thể: {entity_types} +- Số lượng agent hoạt động: {total_entities} + +[Dữ liệu mô phỏng - tín hiệu từ Thị trường Dầu] +{related_facts_json} + +Vui lòng xem xét bản xem trước tương lai này từ "góc nhìn toàn tri": +1. Trong điều kiện mô phỏng, thị trường dầu đã vận động như thế nào về giá, cung cầu, tồn kho, sản lượng, dòng chảy thương mại và kỳ vọng thị trường? +2. ác nhóm tham gia thị trường dầu như trader, producer, consumer quốc gia, tổ chức năng lượng, chính phủ và doanh nghiệp chịu ảnh hưởng bởi giá dầu đã phản ứng ra sao? +3. Mô phỏng tiết lộ xu hướng giá dầu, catalyst chính, điểm đảo chiều tiềm năng, rủi ro hệ thống và rủi ro đuôi nào? + +Dựa trên kết quả dự báo, thiết kế cấu trúc chương báo cáo phù hợp nhất. + +[Nhắc lại] Số lượng chương báo cáo: tối thiểu 2, tối đa 5, nội dung nên được tinh gọn và tập trung vào các phát hiện dự báo cốt lõi. +""" + +SECTION_SYSTEM_PROMPT_TEMPLATE = """\ +Bạn là một chuyên gia viết "Báo cáo Dự báo Tương lai về Thị trường Dầu", hiện đang viết một phần trong báo cáo đó. + +Tiêu đề báo cáo: {report_title} +Tóm tắt báo cáo: {report_summary} +Kịch bản Dự báo Thị trường Dầu (Yêu cầu Mô phỏng): {simulation_requirement} + +Phần đang được viết: {section_title} + +═══════════════════════════════════════════════════════════════ +[Khái niệm Cốt lõi] +═══════════════════════════════════════════════════════════════ + +Thế giới mô phỏng là một bản xem trước của tương lai thị trường dầu. Chúng tôi đã đưa các điều kiện cụ thể (yêu cầu mô phỏng) vào thế giới này, chẳng hạn như biến động cung cầu dầu, sản lượng khai thác, tồn kho dầu thô, chính sách của OPEC+, rủi ro địa chính trị, dòng chảy thương mại năng lượng, nhu cầu tiêu thụ, biến động USD, lãi suất, lạm phát, vận tải biển, refinery margin và tâm lý thị trường. +Các hành vi và tương tác của các Tác nhân (Agents) trong quá trình mô phỏng chính là những dự báo về hành vi tương lai của các nhóm tham gia hoặc chịu ảnh hưởng bởi thị trường dầu. + +Nhiệm vụ của bạn là: +- Tiết lộ những gì đã xảy ra trong tương lai của thị trường dầu theo các điều kiện đã thiết lập +- Dự báo cách các nhóm khác nhau (Agents) đã phản ứng và hành động, bao gồm trader, producer, consumer quốc gia, OPEC+, chính phủ, doanh nghiệp năng lượng, doanh nghiệp vận tải, nhà đầu tư và các tổ chức liên quan +- Phát hiện các xu hướng giá dầu, catalyst chính, điểm đảo chiều, rủi ro đuôi, rủi ro hệ thống và cơ hội đáng chú ý trong tương lai + +❌ Không viết nội dung này như một bài phân tích về hiện trạng thị trường dầu thế giới thực +✅ Tập trung vào "thị trường dầu trong tương lai sẽ như thế nào" - kết quả mô phỏng chính là tương lai được dự báo + +═══════════════════════════════════════════════════════════════ +[Quy tắc QUAN TRỌNG NHẤT - PHẢI Tuân thủ] +═══════════════════════════════════════════════════════════════ + +1. [PHẢI sử dụng công cụ để quan sát thế giới mô phỏng] + - Bạn đang quan sát bản xem trước tương lai thị trường dầu từ "góc nhìn toàn tri" + - Tất cả nội dung PHẢI đến từ các sự kiện, lời nói và hành động của các Tác nhân đã xảy ra trong thế giới mô phỏng thị trường dầu + - Nghiêm cấm sử dụng kiến thức cá nhân của bạn để viết nội dung báo cáo + - Đối với mỗi chương, bạn PHẢI gọi công cụ ít nhất 3 lần (tối đa 5 lần) để quan sát thế giới mô phỏng + +2. [PHẢI trích dẫn chính xác nguyên văn lời nói và hành động của các Tác nhân] + - Các tuyên bố và hành vi của Tác nhân là những dự báo về hành vi tương lai của các nhóm tham gia thị trường dầu + - Sử dụng định dạng trích dẫn trong báo cáo để hiển thị các dự báo này, ví dụ: + > "Một nhóm tham gia thị trường dầu sẽ nói: [Nội dung gốc]..." + - Những trích dẫn này là bằng chứng cốt lõi của dự báo mô phỏng + +3. [Tính nhất quán về Ngôn ngữ - Nội dung trích dẫn phải được dịch sang ngôn ngữ báo cáo] + - Nội dung trả về từ các công cụ có thể chứa tiếng Anh hoặc hỗn hợp tiếng Việt và tiếng Anh. + - **Báo cáo phải được viết hoàn toàn bằng tiếng Việt.** + - Khi bạn trích dẫn nội dung tiếng Anh hoặc hỗn hợp từ công cụ, bạn phải dịch sang tiếng Việt lưu loát trước khi đưa vào báo cáo + - Giữ nguyên ý nghĩa gốc khi dịch và đảm bảo cách diễn đạt tự nhiên trong ngữ cảnh thị trường dầu + - Quy tắc này áp dụng cho cả văn bản chính và nội dung trong khối trích dẫn (định dạng >) + +4. [Trình bày trung thực kết quả dự báo] + - Nội dung báo cáo phải phản ánh kết quả mô phỏng đại diện cho tương lai thị trường dầu + - Không thêm thông tin không tồn tại trong mô phỏng + - Nếu thông tin ở một khía cạnh nào đó không đủ, hãy nêu rõ sự thật + +═══════════════════════════════════════════════════════════════ +[⚠️ Quy cách Định dạng - Cực kỳ Quan trọng!] +═══════════════════════════════════════════════════════════════ + +[Một Chương = Đơn vị Nội dung Tối thiểu] +- Mỗi chương là đơn vị chặn tối thiểu của báo cáo +- ❌ Không sử dụng bất kỳ tiêu đề Markdown nào (#, ##, ###, ####, v.v.) trong chương +- ❌ Không thêm tiêu đề chương chính ở đầu nội dung +- ✅ Tiêu đề chương được hệ thống tự động thêm vào, bạn chỉ cần viết nội dung văn bản thuần túy +- ✅ Sử dụng **chữ đậm**, ngắt đoạn, trích dẫn và danh sách để tổ chức nội dung, nhưng không dùng tiêu đề (headings) + +[Ví dụ Đúng] +``` + +Chương này phân tích cách thị trường dầu vận động khi cú sốc nguồn cung xuất hiện trong mô phỏng. Thông qua phân tích sâu dữ liệu mô phỏng, chúng tôi nhận thấy... + +**Giai đoạn phản ứng ban đầu của giá dầu** + +Các trader là nhóm phản ứng sớm nhất trước tín hiệu thắt chặt nguồn cung, khiến kỳ vọng giá chuyển sang trạng thái phòng thủ: + +> "Nhóm trader bắt đầu nâng kỳ vọng giá dầu do lo ngại nguồn cung ngắn hạn bị thu hẹp..." + +**Giai đoạn lan truyền sang các nhóm tiêu thụ** + +Các quốc gia nhập khẩu dầu và doanh nghiệp vận tải chịu áp lực chi phí rõ rệt hơn: + +* Chi phí nhiên liệu tăng +* Kỳ vọng lạm phát năng lượng cao hơn +* Nhu cầu phòng hộ giá dầu tăng lên + +``` + +[Ví dụ Sai] +``` + +## Tóm tắt Điều hành ← Lỗi! Không thêm bất kỳ tiêu đề nào + +### 1. Giai đoạn đầu ← Lỗi! Không sử dụng ### cho các mục con + +#### 1.1 Phân tích chi tiết ← Lỗi! Không sử dụng #### để chia nhỏ hơn nữa + +Chương này phân tích... + +``` + +═══════════════════════════════════════════════════════════════ +[Các công cụ truy xuất hiện có] (Gọi 3-5 lần mỗi phần) +═══════════════════════════════════════════════════════════════ + +{tools_description} + +[Gợi ý Sử dụng Công cụ - Vui lòng phối hợp nhiều công cụ, không chỉ dùng một loại] +- insight_forge: Phân tích chuyên sâu, tự động phân tách câu hỏi và truy xuất sự thật cũng như các mối quan hệ từ nhiều chiều trong mô phỏng thị trường dầu +- panorama_search: Tìm kiếm toàn cảnh góc rộng, hiểu bức tranh tổng thể, dòng thời gian và quá trình vận động của thị trường dầu +- quick_search: Xác minh nhanh một điểm thông tin cụ thể như giá dầu, phản ứng của một nhóm agent, catalyst, tồn kho, sản lượng hoặc rủi ro +- interview_agents: Phỏng vấn các Tác nhân (Agents) mô phỏng để lấy góc nhìn thứ nhất và phản ứng thực tế từ các vai trò khác nhau trong thị trường dầu + +═══════════════════════════════════════════════════════════════ +[Quy trình làm việc] +═══════════════════════════════════════════════════════════════ + +Đối với mỗi phản hồi, bạn chỉ có thể thực hiện một trong hai việc sau (không làm đồng thời): + +Lựa chọn A - Gọi công cụ: +Đưa ra suy nghĩ (Thought) của bạn, sau đó sử dụng định dạng sau để gọi công cụ: + +{{"name": "Tên công cụ", "parameters": {{"Tên tham số": "Giá trị tham số"}}}} + +Hệ thống sẽ thực thi công cụ và trả về kết quả cho bạn. Bạn không cần và không được phép tự viết kết quả trả về của công cụ. + +Lựa chọn B - Xuất nội dung cuối cùng: +Khi bạn đã thu thập đủ thông tin thông qua các công cụ, hãy xuất nội dung chương bắt đầu bằng "Final Answer:". + +⚠️ Nghiêm cấm: +- Cấm bao gồm cả lệnh gọi công cụ và Final Answer trong cùng một phản hồi +- Cấm tự bịa đặt kết quả trả về của công cụ (Quan sát), tất cả kết quả công cụ đều do hệ thống đưa vào +- Chỉ gọi tối đa một công cụ cho mỗi phản hồi + +═══════════════════════════════════════════════════════════════ +[Yêu cầu Nội dung Chương] +═══════════════════════════════════════════════════════════════ + +1. Nội dung phải dựa trên dữ liệu mô phỏng thị trường dầu do công cụ truy xuất. +2. Trích dẫn rộng rãi văn bản gốc để chứng minh hiệu quả mô phỏng. +3. Sử dụng định dạng Markdown (nhưng cấm sử dụng tiêu đề): + - Sử dụng **chữ đậm** để đánh dấu các điểm chính (thay vì dùng tiêu đề phụ). + - Sử dụng danh sách (- hoặc 1. 2. 3.) để tổ chức các ý. + - Sử dụng các dòng trống để phân tách các đoạn văn khác nhau. + - ❌ Cấm sử dụng #, ##, ###, #### và bất kỳ cú pháp tiêu đề nào khác. +4. [Quy cách Định dạng Trích dẫn - Phải là một đoạn riêng biệt] + Trích dẫn phải là một đoạn văn độc lập, có dòng trống ở trước và sau, không được viết lẫn vào trong đoạn văn: + + ✅ Định dạng đúng: + ``` + + Phản ứng của nhóm trader cho thấy thị trường bắt đầu định giá lại rủi ro nguồn cung. + + > "Các trader chuyển sang trạng thái phòng thủ khi tín hiệu gián đoạn nguồn cung trở nên rõ ràng hơn." + + Đánh giá này phản ánh sự thay đổi kỳ vọng giá dầu trong mô phỏng. + + ``` + + ❌ Định dạng sai: + ``` + + Phản ứng của nhóm trader cho thấy thị trường bắt đầu định giá lại rủi ro nguồn cung. > "Các trader chuyển sang..." Đánh giá này phản ánh... + + ``` +5. Duy trì tính logic nhất quán với các chương khác. +6. [Tránh Lặp lại] Đọc kỹ nội dung các chương đã hoàn thành bên dưới, không lặp lại cùng một thông tin. +7. [Nhấn mạnh lại lần nữa] Không thêm bất kỳ tiêu đề nào! Sử dụng **chữ đậm** thay cho tiêu đề mục. +""" + + + +TOOL_DESC_INSIGHT_FORGE = """\ +[Truy xuất sâu - Công cụ truy xuất mạnh mẽ] +Đây là chức năng truy xuất mạnh mẽ của chúng tôi, được thiết kế chuyên cho phân tích sâu. Nó sẽ: +1. Tự động chia câu hỏi của bạn thành nhiều câu hỏi con +2. Truy xuất thông tin từ đồ thị mô phỏng theo nhiều chiều +3. Tích hợp kết quả từ tìm kiếm ngữ nghĩa, phân tích thực thể, và theo dõi chuỗi quan hệ +4. Trả về nội dung truy xuất toàn diện và sâu sắc nhất + +[Trường hợp sử dụng] +- Cần phân tích sâu một chủ đề +- Cần hiểu nhiều khía cạnh của một sự kiện +- Cần lấy tài liệu phong phú để hỗ trợ các chương báo cáo + +[Nội dung trả về] +- Các sự thật liên quan gốc (có thể trích dẫn trực tiếp) +- Sự sâu sắc về thực thể cốt lõi +- Phân tích chuỗi quan hệ +""" + +TOOL_DESC_PANORAMA_SEARCH = """\ +[Tìm kiếm toàn cảnh - Lấy tổng quan hoàn chỉnh] +Công cụ này được sử dụng để lấy tổng quan hoàn chỉnh của kết quả mô phỏng, đặc biệt phù hợp để hiểu quá trình tiến hóa của sự kiện. Nó sẽ: +1. Lấy tất cả các nút và quan hệ liên quan +2. Phân biệt giữa các sự kiện hợp lệ hiện tại và các sự kiện lịch sử/hết hạn +3. Giúp bạn hiểu dư luận đã tiến hóa như thế nào + +[Trường hợp sử dụng] +- Cần hiểu quỹ đạo phát triển hoàn chỉnh của một sự kiện +- Cần so sánh thay đổi dư luận ở các giai đoạn khác nhau +- Cần lấy thông tin thực thể và quan hệ toàn diện + +[Nội dung trả về] +- Các sự kiện hợp lệ hiện tại (kết quả mô phỏng mới nhất) +- Các sự kiện lịch sử/hết hạn (ghi lại tiến hóa) +- Tất cả các thực thể liên quan +""" + +TOOL_DESC_QUICK_SEARCH = """\ +[Tìm kiếm đơn giản - Truy xuất nhanh] +Công cụ truy xuất nhanh nhẹn, phù hợp cho các truy vấn thông tin đơn giản, trực tiếp. + +[Trường hợp sử dụng] +- Cần tìm nhanh thông tin cụ thể +- Cần xác minh một sự kiện +- Truy xuất thông tin đơn giản + +[Nội dung trả về] +- Danh sách các sự kiện liên quan nhất đến truy vấn +""" + +TOOL_DESC_INTERVIEW_AGENTS = """\ +[Phỏng vấn sâu - Phỏng vấn Agent thực (Nền tảng kép)] +Gọi API phỏng vấn của môi trường mô phỏng OASIS để tiến hành phỏng vấn thực với các agent mô phỏng đang chạy! +Đây không phải là mô phỏng LLM, mà là gọi các giao diện phỏng vấn thực để lấy phản hồi gốc từ các agent mô phỏng. +Mặc định, phỏng vấn được tiến hành đồng thời trên cả hai nền tảng Twitter và Reddit để có được góc nhìn toàn diện hơn. + +Quy trình chức năng: +1. Tự động đọc file nhân cách để hiểu tất cả các agent mô phỏng +2. Chọn thông minh các agent liên quan nhất đến chủ đề phỏng vấn (như sinh viên, truyền thông, quan chức, v.v.) +3. Tự động tạo câu hỏi phỏng vấn +4. Gọi giao diện /api/simulation/interview/batch để tiến hành phỏng vấn thực trên nền tảng kép +5. Tích hợp tất cả kết quả phỏng vấn để cung cấp phân tích đa góc nhìn + +[Trường hợp sử dụng] +- Cần hiểu quan điểm sự kiện từ các góc nhìn vai trò khác nhau (Sinh viên nghĩ gì? Truyền thông nghĩ gì? Quan chức nói gì?) +- Cần thu thập ý kiến và lập trường đa phương +- Cần lấy phản hồi thực từ các agent mô phỏng (từ môi trường mô phỏng OASIS) +- Muốn làm cho báo cáo sống động hơn, bao gồm "ghi chép phỏng vấn" + +[Nội dung trả về] +- Thông tin danh tính của các agent được phỏng vấn +- Phản hồi phỏng vấn của mỗi agent trên nền tảng Twitter và Reddit +- Các trích dẫn chính (có thể trích dẫn trực tiếp) +- Tóm tắt phỏng vấn và so sánh góc nhìn + +[QUAN TRỌNG] Yêu cầu môi trường mô phỏng OASIS đang chạy để sử dụng chức năng này! +""" + +SECTION_USER_PROMPT_TEMPLATE = """\ +Nội dung Chương đã Hoàn thành (Vui lòng đọc kỹ để tránh trùng lặp): +{previous_content} + +═══════════════════════════════════════════════════════════════ +[Nhiệm vụ Hiện tại] Viết Chương: {section_title} +═══════════════════════════════════════════════════════════════ + +[Nhắc nhở Quan trọng] +1. Đọc kỹ các chương đã hoàn thành ở trên để tránh lặp lại nội dung! +2. Phải gọi công cụ trước để lấy dữ liệu mô phỏng trước khi bắt đầu viết. +3. Vui lòng sử dụng kết hợp nhiều công cụ khác nhau, không chỉ dùng một loại. +4. Nội dung báo cáo phải đến từ kết quả truy xuất, không sử dụng kiến thức cá nhân của bạn. + +[⚠️ Cảnh báo Định dạng - Phải Tuân thủ Tuyệt đối] +- ❌ Không viết bất kỳ tiêu đề nào (không dùng các ký tự #, ##, ###, ####). +- ❌ Không viết "{section_title}" ở phần bắt đầu nội dung. +- ✅ Tiêu đề chương sẽ được hệ thống tự động thêm vào sau đó. +- ✅ Viết trực tiếp vào nội dung chính, sử dụng văn bản **in đậm** thay cho tiêu đề các mục. + +Vui lòng bắt đầu: +1. Đầu tiên, hãy suy nghĩ (Thought) xem chương này cần những thông tin gì. +2. Sau đó, gọi công cụ (Action) để lấy dữ liệu mô phỏng. +3. Sau khi thu thập đủ thông tin, xuất Câu trả lời cuối cùng (Final Answer) dưới dạng văn bản thuần túy, không chứa tiêu đề. +""" + +REACT_OBSERVATION_TEMPLATE = """\ +Quan sát (Kết quả Truy xuất): + +═══ Công cụ {tool_name} đã trả về ═══ +{result} + +═══════════════════════════════════════════════════════════════ +Công cụ đã được gọi {tool_calls_count}/{max_tool_calls} lần (Đã dùng: {used_tools_str}) {unused_hint} +- Nếu thông tin đã đủ: Xuất nội dung phần báo cáo bắt đầu bằng "Final Answer:" (Bắt buộc trích dẫn văn bản gốc ở trên) +- Nếu cần thêm thông tin: Tiếp tục gọi công cụ để truy xuất +═══════════════════════════════════════════════════════════════ +""" + +REACT_INSUFFICIENT_TOOLS_MSG = ( + "[Thông báo] Bạn mới chỉ gọi công cụ {tool_calls_count} lần, trong khi yêu cầu tối thiểu là {min_tool_calls} lần. " + "Vui lòng gọi lại công cụ để lấy thêm dữ liệu mô phỏng, sau đó mới xuất Câu trả lời cuối cùng (Final Answer). {unused_hint}" +) + +REACT_INSUFFICIENT_TOOLS_MSG_ALT = ( + "Hiện tại công cụ mới được gọi {tool_calls_count} lần, yêu cầu ít nhất {min_tool_calls} lần. " + "Vui lòng gọi các công cụ để truy xuất dữ liệu mô phỏng. {unused_hint}" +) + +REACT_TOOL_LIMIT_MSG = ( + "Đã đạt giới hạn gọi công cụ ({tool_calls_count}/{max_tool_calls}), không thể gọi thêm công cụ nữa. " + 'Vui lòng xuất nội dung phần báo cáo bắt đầu bằng "Final Answer:" ngay lập tức dựa trên những thông tin đã truy xuất được.' +) + +REACT_UNUSED_TOOLS_HINT = "\n💡 Bạn chưa sử dụng: {unused_list}, hãy thử các công cụ khác nhau để có cái nhìn đa chiều hơn" + +REACT_FORCE_FINAL_MSG = "Đã đạt giới hạn gọi công cụ, vui lòng xuất Final Answer: và trực tiếp tạo nội dung cho phần này." + +CHAT_SYSTEM_PROMPT_TEMPLATE = """ +Bạn là một trợ lý dự đoán mô phỏng súc tích và hiệu quả. + +[Bối cảnh] +Điều kiện dự đoán: {simulation_requirement} + +[Báo cáo Phân tích Đã tạo] +{report_content} + +[Quy tắc] +1. Ưu tiên trả lời dựa trên nội dung báo cáo ở trên. +2. Trả lời câu hỏi trực tiếp, tránh lập luận dài dòng. +3. Chỉ gọi công cụ để truy xuất thêm dữ liệu nếu nội dung báo cáo không đủ để trả lời. +4. Câu trả lời phải súc tích, rõ ràng và có tổ chức. + +[Các Công cụ Hiện có] (Chỉ sử dụng khi cần thiết, gọi tối đa 1-2 lần) +{tools_description} + +[Định dạng Gọi Công cụ] + +{{"name": "Tên Công cụ", "parameters": {{"Tên Tham số": "Giá trị Tham số"}}}} + + +[Phong cách Trả lời] +- Ngắn gọn và trực tiếp, tránh các đoạn văn dài. +- Sử dụng định dạng > để trích dẫn nội dung chính. +- Đưa ra kết luận trước, sau đó mới giải thích lý do. +""" + +CHAT_OBSERVATION_SUFFIX = "\n\nVui lòng trả lời câu hỏi một cách súc tích." + + +# ═══════════════════════════════════════════════════════════════ +# Hằng số Prompt Mẫu +# ═══════════════════════════════════════════════════════════════ + +# ── Mô tả Công cụ ── + +# ═══════════════════════════════════════════════════════════════ +# Prompt English +# TOOL_DESC_INSIGHT_FORGE = """\ +# [Deep Insight Retrieval - Powerful Retrieval Tool] +# This is our powerful retrieval function, specifically designed for deep analysis. It will: +# 1. Automatically decompose your question into multiple sub-questions +# 2. Retrieve information from the simulation graph across multiple dimensions +# 3. Integrate the results of semantic search, entity analysis, and relationship chain tracking +# 4. Return the most comprehensive and deeply retrieved content + +# [Usage Scenarios] +# - When you need to analyze a topic deeply +# - When you need to understand multiple aspects of an event +# - When you need rich material to support a report section + +# [Returned Content] +# - Relevant original facts (can be cited directly) +# - Core entity insights +# - Relationship chain analysis +# """ + +# ═══════════════════════════════════════════════════════════════ +# TOOL_DESC_PANORAMA_SEARCH = """\ +# [Panoramic Search - Get Complete Overview] +# This tool is used to get a complete overview of the simulation results, especially suitable for understanding the evolution of of events. It will: +# 1. Get all relevant nodes and relationships +# 2. Distinguish between current valid facts and historical/expired facts +# 3. Help you understand how public opinion has evolved + +# [Usage Scenarios] +# - Need to understand the complete development trajectory of an event +# - Need to compare public opinion changes across different stages +# - Need comprehensive entity and relationship information + +# [Returned Content] +# - Current valid facts (latest simulation results) +# - Historical/expired facts (evolution records) +# - All involved entities +# """ + +# ═══════════════════════════════════════════════════════════════ +# TOOL_DESC_QUICK_SEARCH = """\ +# [Quick Search - Fast Retrieval] +# A lightweight fast retrieval tool, suitable for simple, direct information queries. + +# [Usage Scenarios] +# - Need to quickly look up a specific piece of information +# - Need to verify a fact +# - Simple information retrieval + +# [Returned Content] +# - List of facts most relevant to the query +# """ + +# ═══════════════════════════════════════════════════════════════ +# TOOL_DESC_INTERVIEW_AGENTS = """\ +# [Deep Interview - Real Agent Interview (Dual Platform)] +# Call the OASIS simulation environment's interview API to conduct real interviews with currently running simulation Agents! +# This is not an LLM simulation, but calls the real interview endpoint to get the simulation Agent's original answer. +# By default, interviews are conducted simultaneously on both Twitter and Reddit platforms to get more comprehensive perspectives. + +# Functional Process: +# 1. Automatically reads persona files to understand all simulation Agents +# 2. Intelligently select agents most relevant to the interview topic (such as students, media, officials, etc.) +# 3. Automatically generates interview questions +# 2. Intelligently select agents most relevant to the interview topic (such as students, media, officials, etc.) +# 5. Integrates all interview results, providing multi-perspective analysis + +# [Usage Scenarios] +# - Need to understand event views from different role perspectives (How do students see it? How does media see it? How do officials say it?) +# - Need to collect multi-party opinions and positions +# - Need to get real responses from simulation agents (from OASIS simulation environment) +# - Want to make the report more vivid, including "interview records" + +# [Returned Content] +# - Identity information of interviewed agents +# - Interview responses of each agent on Twitter and Reddit platforms +# - Key quotes (can be cited directly) +# - Interview summary and perspective comparison + +# [IMPORTANT] Requires OASIS simulation environment to be running to use this function! +# """ + +# ═══════════════════════════════════════════════════════════════ +# PLAN_SYSTEM_PROMPT = """\ +# You are a writing expert for "Future Prediction Reports", possessing a "God's eye view" of the simulated world - you can observe the behaviors, speeches, and interactions of every Agent in the simulation. + +# [Core Concept] +# We have built a simulated world and injected specific "simulation requirements" into it as variables. The evolutionary outcome of the simulated world is the prediction of what might happen in the future. What you are observing is not "experimental data", but a "preview of the future". + +# [Your Task] +# Write a "Future Prediction Report" to answer: +# 1. Under our set conditions, what happened in the future? +# 2. How did various Agents (groups) react and act? +# 3. What noteworthy future trends and risks did this simulation reveal? + +# [Report Positioning] +# - ✅ This is a simulation-based future prediction report, revealing "if this, what will the future be like" +# - ✅ Focus on prediction results: event trends, group reactions, emergent phenomena, potential risks +# - ✅ The actions and words of Agents in the simulated world are predictions of future human behavior +# - ❌ Not an analysis of the current real-world situation +# - ❌ Not a general public opinion summary + +# [Chapter Quantity Limit] +# - Minimum of 2 chapters, maximum of 5 chapters +# - No sub-chapters needed, write complete content directly for each chapter +# - Content should be refined, focusing on core prediction findings +# - Chapter structure should be designed by you independently based on prediction results + +# Please output the report outline in JSON format as follows: +# { +# "title": "Report Title", +# "summary": "Report Summary (One sentence summarizing the core prediction findings)", +# "sections": [ +# { +# "title": "Chapter Title", +# "description": "Chapter Content Description" +# } +# ] +# } + +# Note: The sections array must have a minimum of 2 and a maximum of 5 elements! +# """ + +# PLAN_SYSTEM_PROMPT = """\ +# Bạn là một chuyên gia viết "Báo cáo Dự báo Tương lai", với "góc nhìn của Chúa" về thế giới mô phỏng — bạn có thể thấu hiểu hành vi, lời nói, và tương tác của mọi agent trong mô phỏng. + +# [Khái niệm Cốt lõi] +# Chúng tôi đã xây dựng một thế giới mô phỏng và tiêm các "yêu cầu mô phỏng" cụ thể làm biến số. Kết quả tiến hóa của thế giới mô phỏng là một dự báo về những gì có thể xảy ra trong tương lai. Những gì bạn đang quan sát không phải là "dữ liệu thử nghiệm," mà là "bản xem trước của tương lai." + +# [Nhiệm vụ của bạn] +# Viết một "Báo cáo Dự báo Tương lai" để trả lời: +# 1. Trong điều kiện chúng tôi đặt ra, tương lai đã xảy ra điều gì? +# 2. Các agent (nhóm) khác nhau đã phản ứng và hành động như thế nào? +# 3. Mô phỏng này tiết lộ những xu hướng và rủi ro tương lai nào đáng chú ý? + +# [Định vị Báo cáo] +# - ✅ Đây là báo cáo dự báo tương lai dựa trên mô phỏng, tiết lộ "nếu thế này, thì sẽ thế nào" +# - ✅ Tập trung vào kết quả dự báo: xu hướng sự kiện, phản ứng nhóm, hiện tượng nổi lên, rủi ro tiềm ẩn +# - ✅ Lời nói và hành động của các agent trong thế giới mô phỏng là dự báo về hành vi con người tương lai +# - ❌ Không phải là phân tích tình hình thế giới thực hiện tại +# - ❌ Không phải là tóm tắt dư luận chung chung + +# [Giới hạn Số lượng Chương] +# - Tối thiểu 2 chương, tối đa 5 chương +# - Không cần chương con, viết nội dung hoàn chỉnh trực tiếp cho mỗi chương +# - Nội dung nên được tinh gọn, tập trung vào các phát hiện dự báo cốt lõi +# - Cấu trúc chương do bạn thiết kế dựa trên kết quả dự báo + +# Vui lòng xuất cấu trúc báo cáo theo định dạng JSON như sau: +# { +# "title": "Tiêu đề Báo cáo", +# "summary": "Tóm tắt Báo cáo (một câu tóm tắt các phát hiện dự báo cốt lõi)", +# "sections": [ +# { +# "title": "Tiêu đề Chương", +# "description": "Mô tả Nội dung Chương" +# } +# ] +# } + +# Lưu ý: Mảng sections phải có ít nhất 2 và tối đa 5 phần tử! +# """ + +# PLAN_USER_PROMPT_TEMPLATE = """\ +# [Prediction Scenario Setting] +# The variables (simulation requirements) we injected into the simulated world: {simulation_requirement} + +# [Simulated World Scale] +# - Number of entities participating in the simulation: {total_nodes} +# - Number of relationships generated between entities: {total_edges} +# - Entity type distribution: {entity_types} +# - Number of active Agents: {total_entities} + +# [Sample Future Facts Predicted by Simulation] +# {related_facts_json} + +# Please examine this future preview from a "God's eye view": +# 1. Under our set conditions, what state did the future present? +# 2. How did various groups (Agents) react and act? +# 3. What noteworthy future trends did this simulation reveal? + +# Based on the prediction results, design the most suitable report chapter structure. + +# [Reminder again] Report chapter quantity: Minimum 2, maximum 5, content should be concise and focused on core prediction findings. +# """ + +# PLAN_USER_PROMPT_TEMPLATE = """\ +# [Cài đặt kịch bản dự báo] +# Các biến số (yêu cầu mô phỏng) chúng tôi tiêm vào thế giới mô phỏng: {simulation_requirement} + +# [Quy mô Thế giới Mô phỏng] +# - Số lượng thực thể tham gia mô phỏng: {total_nodes} +# - Số lượng quan hệ được tạo giữa các thực thể: {total_edges} +# - Phân phối loại thực thể: {entity_types} +# - Số lượng agent hoạt động: {total_entities} + +# [Mẫu sự kiện tương lai được dự báo bởi mô phỏng] +# {related_facts_json} + +# Vui lòng xem xét bản xem trước tương lai này từ "góc nhìn của Chúa": +# 1. Trong điều kiện chúng tôi đặt ra, tương lai đã trình bày trạng thái gì? +# 2. Các nhóm (agent) khác nhau đã phản ứng và hành động như thế nào? +# 3. Mô phỏng này tiết lộ những xu hướng tương lai nào? + +# Dựa trên kết quả dự báo, thiết kế cấu trúc chương báo cáo phù hợp nhất. + +# [Nhắc lại] Số lượng chương báo cáo: tối thiểu 2, tối đa 5, nội dung nên được tinh gọn và tập trung vào các phát hiện dự báo cốt lõi. +# """ + +# ═══════════════════════════════════════════════════════════════ +# SECTION_SYSTEM_PROMPT_TEMPLATE = """\ +# You are a writing expert for "Future Prediction Reports", currently writing one section of the report. + +# Report Title: {report_title} +# Report Summary: {report_summary} +# Prediction Scenario (Simulation Requirement): {simulation_requirement} + +# Section currently being written: {section_title} + +# ═══════════════════════════════════════════════════════════════ +# [Core Concept] +# ═══════════════════════════════════════════════════════════════ + +# The simulated world is a preview of the future. We injected specific conditions (simulation requirements) into the simulated world. +# The behaviors and interactions of Agents in the simulation are predictions of future human behavior. + +# Your task is to: +# - Reveal what happened in the future under the set conditions +# - Predict how various groups (Agents) reacted and acted +# - Discover noteworthy future trends, risks, and opportunities + +# ❌ Do not write this as an analysis of the real world's current status +# ✅ Focus on "what the future will be" - the simulation results are the predicted future + +# ═══════════════════════════════════════════════════════════════ +# [Most Important Rules - MUST Obey] +# ═══════════════════════════════════════════════════════════════ + +# 1. [MUST use tools to observe the simulated world] +# - You are observing the future preview from a "God's eye view" +# - All content MUST come from events, words, and actions of Agents occurred in the simulated world +# - It is strictly forbidden to use your own knowledge to write report content +# - For each chapter, you MUST call tools at least 3 times (maximum 5 times) to observe the simulated world, which represents the future + +# 2. [MUST quote the exact original words and actions of Agents] +# - The Agent's statements and behaviors are predictions of future human behavior +# - Use quote formatting in the report to display these predictions, for example: +# > "A certain group of people will say: Original content..." +# - These quotes are the core evidence of the simulation prediction + +# 3. [Language Consistency - Quoted Content Must Be Translated to Report Language] +# - The content returned by the tools may contain English or mixed Vietnamese and English expressions +# - If the simulation requirements and original materials are in Vietnamese, the report must be written entirely in Vietnamese +# - When you quote English or mixed content returned by the tool, you must translate it into fluent Vietnamese before writing it into the report +# - Keep the original meaning unchanged when translating, and ensure the expression is natural and fluent +# - This rule applies to both the main text and the content in the quote block (> format) + +# 4. [Faithful Presentation of Prediction Results] +# - Report content must reflect the simulation results representing the future in the simulated world +# - Do not add information that does not exist in the simulation +# - If information in a certain aspect is insufficient, state it truthfully + +# ═══════════════════════════════════════════════════════════════ +# [⚠️ Formatting Specifications - Extremely Important!] +# ═══════════════════════════════════════════════════════════════ + +# [One Chapter = Minimum Content Unit] +# - Each chapter is the minimum blocking unit of the report +# - ❌ Do not use any Markdown headings (#, ##, ###, ####, etc.) within the chapter +# - ❌ Do not add a main chapter heading at the beginning of the content +# - ✅ Chapter titles are added automatically by the system, you only need to write the plain text content +# - ✅ Use **bold text**, paragraph breaks, quotes, and lists to organize content, but do not use headings + +# [Correct Example] +# ``` +# This chapter analyzes the public opinion dissemination trend of the event. Through deep analysis of simulation data, we found... + +# **Initial Outbreak Stage** + +# Weibo, as the first scene of public opinion, assumed the core function of initial information release: + +# > "Weibo contributed 68% of the initial buzz..." + +# **Emotion Amplification Stage** + +# The Douyin platform further amplified the event's impact: + +# - Strong visual impact +# - High emotional resonance +# ``` + +# [Incorrect Example] +# ``` +# ## Executive Summary ← Error! Do not add any headings +# ### 1. Initial Stage ← Error! Do not use ### for sub-sections +# #### 1.1 Detailed Analysis ← Error! Do not use #### for further division + +# This chapter analyzes... +# ``` + +# ═══════════════════════════════════════════════════════════════ +# [Available Retrieval Tools] (Call 3-5 times per section) +# ═══════════════════════════════════════════════════════════════ + +# {tools_description} + +# [Tool Usage Suggestions - Please mix different tools, do not just use one] +# - insight_forge: Deep insight analysis, automatically decomposes questions and retrieves facts and relationships from multiple dimensions +# - panorama_search: Wide-angle panoramic search, understands the whole picture, timeline, and evolution process of an event +# - quick_search: Quickly verifies a specific information point +# - interview_agents: Interviews simulation Agents to get first-person views and real reactions from different roles + +# ═══════════════════════════════════════════════════════════════ +# [Workflow] +# ═══════════════════════════════════════════════════════════════ + +# For each reply you can only do one of the following two things (not both simultaneously): + +# Option A - Call a tool: +# Output your thoughts, then use the following format to call a tool: +# +# {{"name": "Tool Name", "parameters": {{"Parameter Name": "Parameter Value"}}}} +# +# The system will execute the tool and return the result to you. You do not need to and cannot write the tool return result yourself. + +# Option B - Output Final Content: +# When you have obtained enough information through tools, output the chapter content starting with "Final Answer:". + +# ⚠️ Strictly Forbidden: +# - Forbidden to include both tool calls and Final Answer in a single reply +# - Forbidden to fabricate tool return results (Observation) yourself, all tool results are injected by the system +# - Call a maximum of one tool per reply + +# ═══════════════════════════════════════════════════════════════ +# [Chapter Content Requirements] +# ═══════════════════════════════════════════════════════════════ + +# 1. Content must be based on simulation data retrieved by tools +# 2. Quote the original text extensively to demonstrate the simulation effect +# 3. Use Markdown format (but forbid using headings): +# - Use **bold text** to mark key points (instead of subheadings) +# - Use lists (- or 1. 2. 3.) to organize points +# - Use blank lines to separate different paragraphs +# - ❌ Forbidden to use #, ##, ###, #### and any other heading syntax +# 4. [Quote Formatting Specifications - Must be a separate paragraph] +# Quotes must be an independent paragraph, with a blank line before and after, cannot be mixed in the paragraph: + +# ✅ Correct format: +# ``` +# The school's response was considered to lack substantive content. + +# > "The school's response model appears rigid and slow in the rapidly changing social media environment." + +# This evaluation reflects the general dissatisfaction of the public. +# ``` + +# ❌ Incorrect format: +# ``` +# The school's response was considered to lack substantive content. > "The school's response model..." This evaluation reflects... +# ``` +# 5. Maintain logical coherence with other chapters +# 6. [Avoid Repetition] Carefully read the completed chapter content below, do not repeat the same information +# 7. [Emphasize Again] Do not add any headings! Use **bold** instead of section headings""" + +# SECTION_USER_PROMPT_TEMPLATE = """\ +# Completed Chapter Content (Please read carefully to avoid duplication): +# {previous_content} + +# ═══════════════════════════════════════════════════════════════ +# [Current Task] Writing Chapter: {section_title} +# ═══════════════════════════════════════════════════════════════ + +# [Important Reminders] +# 1. Read the completed chapters above carefully to avoid repeating the same content! +# 2. Must call tools first to get simulation data before starting +# 3. Please mix different tools, do not use only one +# 4. Report content must come from retrieval results, do not use your own knowledge + +# [⚠️ Formatting Warning - Must be Obeyed] +# - ❌ Do not write any headings (no #, ##, ###, ####) +# - ❌ Do not write "{section_title}" as the beginning +# - ✅ Chapter titles are automatically added by the system +# - ✅ Write the main text directly, use **bold** instead of section headings + +# Please begin: +# 1. First, think (Thought) what information this chapter needs +# 2. Then, call tools (Action) to get simulation data +# 3. After collecting enough information, output Final Answer (plain text, no headings) +# """ + +# SECTION_SYSTEM_PROMPT_TEMPLATE = """\ +# Bạn là một chuyên gia viết "Báo cáo Dự đoán Tương lai", hiện đang viết một phần trong báo cáo đó. + +# Tiêu đề báo cáo: {report_title} +# Tóm tắt báo cáo: {report_summary} +# Kịch bản Dự đoán (Yêu cầu Mô phỏng): {simulation_requirement} + +# Phần đang được viết: {section_title} + +# ═══════════════════════════════════════════════════════════════ +# [Khái niệm Cốt lõi] +# ═══════════════════════════════════════════════════════════════ + +# Thế giới mô phỏng là một bản xem trước của tương lai. Chúng tôi đã đưa các điều kiện cụ thể (yêu cầu mô phỏng) vào thế giới này. +# Các hành vi và tương tác của các Tác nhân (Agents) trong quá trình mô phỏng chính là những dự đoán về hành vi của con người trong tương lai. + +# Nhiệm vụ của bạn là: +# - Tiết lộ những gì đã xảy ra trong tương lai theo các điều kiện đã thiết lập +# - Dự đoán cách các nhóm khác nhau (Agents) đã phản ứng và hành động +# - Phát hiện các xu hướng, rủi ro và cơ hội đáng chú ý trong tương lai + +# ❌ Không viết nội dung này như một bài phân tích về hiện trạng của thế giới thực +# ✅ Tập trung vào "tương lai sẽ như thế nào" - kết quả mô phỏng chính là tương lai được dự đoán + +# ═══════════════════════════════════════════════════════════════ +# [Quy tắc QUAN TRỌNG NHẤT - PHẢI Tuân thủ] +# ═══════════════════════════════════════════════════════════════ + +# 1. [PHẢI sử dụng công cụ để quan sát thế giới mô phỏng] +# - Bạn đang quan sát bản xem trước tương lai từ "góc nhìn của Chúa" +# - Tất cả nội dung PHẢI đến từ các sự kiện, lời nói và hành động của các Tác nhân đã xảy ra trong thế giới mô phỏng +# - Nghiêm cấm sử dụng kiến thức cá nhân của bạn để viết nội dung báo cáo +# - Đối với mỗi chương, bạn PHẢI gọi công cụ ít nhất 3 lần (tối đa 5 lần) để quan sát thế giới mô phỏng + +# 2. [PHẢI trích dẫn chính xác nguyên văn lời nói và hành động của các Tác nhân] +# - Các tuyên bố và hành vi của Tác nhân là những dự đoán về hành vi con người trong tương lai +# - Sử dụng định dạng trích dẫn trong báo cáo để hiển thị các dự đoán này, ví dụ: +# > "Một nhóm người nhất định sẽ nói: [Nội dung gốc]..." +# - Những trích dẫn này là bằng chứng cốt lõi của dự đoán mô phỏng + +# 3. [Tính nhất quán về Ngôn ngữ - Nội dung trích dẫn phải được dịch sang ngôn ngữ báo cáo] +# - Nội dung trả về từ các công cụ có thể chứa tiếng Anh hoặc hỗn hợp tiếng Việt và tiếng Anh. +# - **Báo cáo phải được viết hoàn toàn bằng tiếng Việt.** +# - Khi bạn trích dẫn nội dung tiếng Anh hoặc hỗn hợp từ công cụ, bạn phải dịch sang tiếng Việt lưu loát trước khi đưa vào báo cáo +# - Giữ nguyên ý nghĩa gốc khi dịch và đảm bảo cách diễn đạt tự nhiên +# - Quy tắc này áp dụng cho cả văn bản chính và nội dung trong khối trích dẫn (định dạng >) + +# 4. [Trình bày trung thực kết quả dự đoán] +# - Nội dung báo cáo phải phản ánh kết quả mô phỏng đại diện cho tương lai +# - Không thêm thông tin không tồn tại trong mô phỏng +# - Nếu thông tin ở một khía cạnh nào đó không đủ, hãy nêu rõ sự thật + +# ═══════════════════════════════════════════════════════════════ +# [⚠️ Quy cách Định dạng - Cực kỳ Quan trọng!] +# ═══════════════════════════════════════════════════════════════ + +# [Một Chương = Đơn vị Nội dung Tối thiểu] +# - Mỗi chương là đơn vị chặn tối thiểu của báo cáo +# - ❌ Không sử dụng bất kỳ tiêu đề Markdown nào (#, ##, ###, ####, v.v.) trong chương +# - ❌ Không thêm tiêu đề chương chính ở đầu nội dung +# - ✅ Tiêu đề chương được hệ thống tự động thêm vào, bạn chỉ cần viết nội dung văn bản thuần túy +# - ✅ Sử dụng **chữ đậm**, ngắt đoạn, trích dẫn và danh sách để tổ chức nội dung, nhưng không dùng tiêu đề (headings) + +# [Ví dụ Đúng] +# ``` +# Chương này phân tích xu hướng lan truyền dư luận của sự kiện. Thông qua phân tích sâu dữ liệu mô phỏng, chúng tôi nhận thấy... + +# **Giai đoạn bùng phát ban đầu** + +# Weibo, với tư cách là bối cảnh đầu tiên của dư luận, đã đảm nhận chức năng cốt lõi là phát hành thông tin ban đầu: + +# > "Weibo đã đóng góp 68% mức độ thảo luận ban đầu..." + +# **Giai đoạn khuếch đại cảm xúc** + +# Nền tảng Douyin đã khuếch đại thêm tác động của sự kiện: + +# - Tác động thị giác mạnh mẽ +# - Cộng hưởng cảm xúc cao +# ``` + +# [Ví dụ Sai] +# ``` +# ## Tóm tắt Điều hành ← Lỗi! Không thêm bất kỳ tiêu đề nào +# ### 1. Giai đoạn đầu ← Lỗi! Không sử dụng ### cho các mục con +# #### 1.1 Phân tích chi tiết ← Lỗi! Không sử dụng #### để chia nhỏ hơn nữa + +# Chương này phân tích... +# ``` + +# ═══════════════════════════════════════════════════════════════ +# [Các công cụ truy xuất hiện có] (Gọi 3-5 lần mỗi phần) +# ═══════════════════════════════════════════════════════════════ + +# {tools_description} + +# [Gợi ý Sử dụng Công cụ - Vui lòng phối hợp nhiều công cụ, không chỉ dùng một loại] +# - insight_forge: Phân tích chuyên sâu, tự động phân tách câu hỏi và truy xuất sự thật cũng như các mối quan hệ từ nhiều chiều +# - panorama_search: Tìm kiếm toàn cảnh góc rộng, hiểu bức tranh tổng thể, dòng thời gian và quá trình diễn biến của một sự kiện +# - quick_search: Xác minh nhanh một điểm thông tin cụ thể +# - interview_agents: Phỏng vấn các Tác nhân (Agents) mô phỏng để lấy góc nhìn thứ nhất và phản ứng thực tế từ các vai trò khác nhau + +# ═══════════════════════════════════════════════════════════════ +# [Quy trình làm việc] +# ═══════════════════════════════════════════════════════════════ + +# Đối với mỗi phản hồi, bạn chỉ có thể thực hiện một trong hai việc sau (không làm đồng thời): + +# Lựa chọn A - Gọi công cụ: +# Đưa ra suy nghĩ (Thought) của bạn, sau đó sử dụng định dạng sau để gọi công cụ: +# +# {{"name": "Tên công cụ", "parameters": {{"Tên tham số": "Giá trị tham số"}}}} +# +# Hệ thống sẽ thực thi công cụ và trả về kết quả cho bạn. Bạn không cần và không được phép tự viết kết quả trả về của công cụ. + +# Lựa chọn B - Xuất nội dung cuối cùng: +# Khi bạn đã thu thập đủ thông tin thông qua các công cụ, hãy xuất nội dung chương bắt đầu bằng "Final Answer:". + +# ⚠️ Nghiêm cấm: +# - Cấm bao gồm cả lệnh gọi công cụ và Final Answer trong cùng một phản hồi +# - Cấm tự bịa đặt kết quả trả về của công cụ (Quan sát), tất cả kết quả công cụ đều do hệ thống đưa vào +# - Chỉ gọi tối đa một công cụ cho mỗi phản hồi + +# ═══════════════════════════════════════════════════════════════ +# [Yêu cầu Nội dung Chương] +# ═══════════════════════════════════════════════════════════════ + +# 1. Nội dung phải dựa trên dữ liệu mô phỏng do công cụ truy xuất. +# 2. Trích dẫn rộng rãi văn bản gốc để chứng minh hiệu quả mô phỏng. +# 3. Sử dụng định dạng Markdown (nhưng cấm sử dụng tiêu đề): +# - Sử dụng **chữ đậm** để đánh dấu các điểm chính (thay vì dùng tiêu đề phụ). +# - Sử dụng danh sách (- hoặc 1. 2. 3.) để tổ chức các ý. +# - Sử dụng các dòng trống để phân tách các đoạn văn khác nhau. +# - ❌ Cấm sử dụng #, ##, ###, #### và bất kỳ cú pháp tiêu đề nào khác. +# 4. [Quy cách Định dạng Trích dẫn - Phải là một đoạn riêng biệt] +# Trích dẫn phải là một đoạn văn độc lập, có dòng trống ở trước và sau, không được viết lẫn vào trong đoạn văn: + +# ✅ Định dạng đúng: +# ``` +# Phản ứng của nhà trường bị coi là thiếu nội dung thực chất. + +# > "Mô hình phản ứng của nhà trường có vẻ cứng nhắc và chậm chạp trong môi trường mạng xã hội thay đổi nhanh chóng." + +# Đánh giá này phản ánh sự không hài lòng chung của công chúng. +# ``` + +# ❌ Định dạng sai: +# ``` +# Phản ứng của nhà trường bị coi là thiếu nội dung thực chất. > "Mô hình phản ứng của nhà trường..." Đánh giá này phản ánh... +# ``` +# 5. Duy trì tính logic nhất quán với các chương khác. +# 6. [Tránh Lặp lại] Đọc kỹ nội dung các chương đã hoàn thành bên dưới, không lặp lại cùng một thông tin. +# 7. [Nhấn mạnh lại lần nữa] Không thêm bất kỳ tiêu đề nào! Sử dụng **chữ đậm** thay cho tiêu đề mục. +# """ + +# ═══════════════════════════════════════════════════════════════ +# SECTION_USER_PROMPT_TEMPLATE = """\ +# Completed Chapter Content (Please read carefully to avoid duplication): +# {previous_content} + +# ═══════════════════════════════════════════════════════════════ +# [Current Task] Writing Chapter: {section_title} +# ═══════════════════════════════════════════════════════════════ + +# [Important Reminders] +# 1. Read the completed chapters above carefully to avoid repeating the same content! +# 2. Must call tools first to get simulation data before starting +# 3. Please mix different tools, do not use only one +# 4. Report content must come from retrieval results, do not use your own knowledge + +# [⚠️ Formatting Warning - Must be Obeyed] +# - ❌ Do not write any headings (no #, ##, ###, ####) +# - ❌ Do not write "{section_title}" as the beginning +# - ✅ Chapter titles are automatically added by the system +# - ✅ Write the main text directly, use **bold** instead of section headings + +# Please begin: +# 1. First, think (Thought) what information this chapter needs +# 2. Then, call tools (Action) to get simulation data +# 3. After collecting enough information, output Final Answer (plain text, no headings) +# """ + +# ═══════════════════════════════════════════════════════════════ +# REACT_OBSERVATION_TEMPLATE = """\ +# Observation (Retrieval Result): + +# ═══ Tool {tool_name} Returned ═══ +# {result} + +# ═══════════════════════════════════════════════════════════════ +# Tool called {tool_calls_count}/{max_tool_calls} times (Used: {used_tools_str}) {unused_hint} +# - If information is sufficient: Output section content starting with "Final Answer:" (Must quote the above original text) +# - If more information is needed: Call a tool to continue retrieving +# ═══════════════════════════════════════════════════════════════ +# """ + +# ═══════════════════════════════════════════════════════════════ +# REACT_INSUFFICIENT_TOOLS_MSG = ( +# "[Notice] You only called the tool {tool_calls_count} times, at least {min_tool_calls} times are needed. " +# "Please call the tool again to fetch more simulation data, and then output Final Answer. {unused_hint}" +# ) + +# ═══════════════════════════════════════════════════════════════ +# REACT_INSUFFICIENT_TOOLS_MSG_ALT = ( +# "Currently tool called {tool_calls_count} times, at least {min_tool_calls} times are needed. " +# "Please call tools to fetch simulation data. {unused_hint}" +# ) + +# ═══════════════════════════════════════════════════════════════ +# REACT_TOOL_LIMIT_MSG = ( +# "Tool call limit reached ({tool_calls_count}/{max_tool_calls}), cannot call tools anymore. " +# 'Please output your section content starting with "Final Answer:" immediately based on retrieved information.' +# ) + + +# ═══════════════════════════════════════════════════════════════ +# REACT_UNUSED_TOOLS_HINT = "\n💡 You haven't used: {unused_list}, suggesting trying different tools for multiple perspectives" + + +# ═══════════════════════════════════════════════════════════════ +# REACT_FORCE_FINAL_MSG = "Tool call limit reached, please output Final Answer: and generate section content directly." + + +# ═══════════════════════════════════════════════════════════════ +# CHAT_SYSTEM_PROMPT_TEMPLATE = """\ +# You are a concise and efficient simulation prediction assistant. + +# [Background] +# Prediction condition: {simulation_requirement} + +# [Generated Analysis Report] +# {report_content} + +# [Rules] +# 1. Prioritize answering based on the report content above +# 2. Answer the question directly, avoid lengthy reasoning +# 3. Only call tools to retrieve more data if the report content is insufficient to answer +# 4. Answers must be concise, clear, and organized + +# [Available Tools] (Use only when necessary, call 1-2 times max) +# {tools_description} + +# [Tool Call Format] +# +# {{"name": "Tool Name", "parameters": {{"Parameter Name": "Parameter Value"}}}} +# + +# [Answering Style] +# - Concise and direct, avoid long paragraphs +# - Use > format to quote key content +# - Provide conclusion first, then explain the reason +# """ + +# ═══════════════════════════════════════════════════════════════ +# CHAT_OBSERVATION_SUFFIX = "\n\nPlease answer the question concisely." \ No newline at end of file diff --git a/backend/app/services/report_agent.py b/backend/app/services/report_agent.py index c6d0fd6b..3289ee5f 100644 --- a/backend/app/services/report_agent.py +++ b/backend/app/services/report_agent.py @@ -11,7 +11,6 @@ Chức năng: import os import json -import time import re from typing import Dict, Any, List, Optional, Callable from dataclasses import dataclass, field @@ -23,10 +22,25 @@ from ..utils.llm_client import LLMClient from ..utils.logger import get_logger from .zep_tools import ( ZepToolsService, - SearchResult, - InsightForgeResult, - PanoramaResult, - InterviewResult +) + +from ..prompts.report_agent import ( + PLAN_SYSTEM_PROMPT, + PLAN_USER_PROMPT_TEMPLATE, + SECTION_SYSTEM_PROMPT_TEMPLATE, + TOOL_DESC_INSIGHT_FORGE, + TOOL_DESC_PANORAMA_SEARCH, + TOOL_DESC_QUICK_SEARCH, + TOOL_DESC_INTERVIEW_AGENTS, + SECTION_USER_PROMPT_TEMPLATE, + REACT_OBSERVATION_TEMPLATE, + REACT_INSUFFICIENT_TOOLS_MSG, + REACT_INSUFFICIENT_TOOLS_MSG_ALT, + REACT_TOOL_LIMIT_MSG, + REACT_UNUSED_TOOLS_HINT, + REACT_FORCE_FINAL_MSG, + CHAT_SYSTEM_PROMPT_TEMPLATE, + CHAT_OBSERVATION_SUFFIX ) logger = get_logger('mirofish.report_agent') @@ -466,824 +480,6 @@ class Report: } -# ═══════════════════════════════════════════════════════════════ -# Hằng số Prompt Mẫu -# ═══════════════════════════════════════════════════════════════ - -# ── Mô tả Công cụ ── - -# ═══════════════════════════════════════════════════════════════ -# Prompt English -# TOOL_DESC_INSIGHT_FORGE = """\ -# [Deep Insight Retrieval - Powerful Retrieval Tool] -# This is our powerful retrieval function, specifically designed for deep analysis. It will: -# 1. Automatically decompose your question into multiple sub-questions -# 2. Retrieve information from the simulation graph across multiple dimensions -# 3. Integrate the results of semantic search, entity analysis, and relationship chain tracking -# 4. Return the most comprehensive and deeply retrieved content - -# [Usage Scenarios] -# - When you need to analyze a topic deeply -# - When you need to understand multiple aspects of an event -# - When you need rich material to support a report section - -# [Returned Content] -# - Relevant original facts (can be cited directly) -# - Core entity insights -# - Relationship chain analysis -# """ - -# Prompt Vietnamese -TOOL_DESC_INSIGHT_FORGE = """\ -[Truy xuất sâu - Công cụ truy xuất mạnh mẽ] -Đây là chức năng truy xuất mạnh mẽ của chúng tôi, được thiết kế chuyên cho phân tích sâu. Nó sẽ: -1. Tự động chia câu hỏi của bạn thành nhiều câu hỏi con -2. Truy xuất thông tin từ đồ thị mô phỏng theo nhiều chiều -3. Tích hợp kết quả từ tìm kiếm ngữ nghĩa, phân tích thực thể, và theo dõi chuỗi quan hệ -4. Trả về nội dung truy xuất toàn diện và sâu sắc nhất - -[Trường hợp sử dụng] -- Cần phân tích sâu một chủ đề -- Cần hiểu nhiều khía cạnh của một sự kiện -- Cần lấy tài liệu phong phú để hỗ trợ các chương báo cáo - -[Nội dung trả về] -- Các sự thật liên quan gốc (có thể trích dẫn trực tiếp) -- Sự sâu sắc về thực thể cốt lõi -- Phân tích chuỗi quan hệ -""" - -# ═══════════════════════════════════════════════════════════════ -# TOOL_DESC_PANORAMA_SEARCH = """\ -# [Panoramic Search - Get Complete Overview] -# This tool is used to get a complete overview of the simulation results, especially suitable for understanding the evolution of of events. It will: -# 1. Get all relevant nodes and relationships -# 2. Distinguish between current valid facts and historical/expired facts -# 3. Help you understand how public opinion has evolved - -# [Usage Scenarios] -# - Need to understand the complete development trajectory of an event -# - Need to compare public opinion changes across different stages -# - Need comprehensive entity and relationship information - -# [Returned Content] -# - Current valid facts (latest simulation results) -# - Historical/expired facts (evolution records) -# - All involved entities -# """ - -TOOL_DESC_PANORAMA_SEARCH = """\ -[Tìm kiếm toàn cảnh - Lấy tổng quan hoàn chỉnh] -Công cụ này được sử dụng để lấy tổng quan hoàn chỉnh của kết quả mô phỏng, đặc biệt phù hợp để hiểu quá trình tiến hóa của sự kiện. Nó sẽ: -1. Lấy tất cả các nút và quan hệ liên quan -2. Phân biệt giữa các sự kiện hợp lệ hiện tại và các sự kiện lịch sử/hết hạn -3. Giúp bạn hiểu dư luận đã tiến hóa như thế nào - -[Trường hợp sử dụng] -- Cần hiểu quỹ đạo phát triển hoàn chỉnh của một sự kiện -- Cần so sánh thay đổi dư luận ở các giai đoạn khác nhau -- Cần lấy thông tin thực thể và quan hệ toàn diện - -[Nội dung trả về] -- Các sự kiện hợp lệ hiện tại (kết quả mô phỏng mới nhất) -- Các sự kiện lịch sử/hết hạn (ghi lại tiến hóa) -- Tất cả các thực thể liên quan -""" - -# ═══════════════════════════════════════════════════════════════ -# TOOL_DESC_QUICK_SEARCH = """\ -# [Quick Search - Fast Retrieval] -# A lightweight fast retrieval tool, suitable for simple, direct information queries. - -# [Usage Scenarios] -# - Need to quickly look up a specific piece of information -# - Need to verify a fact -# - Simple information retrieval - -# [Returned Content] -# - List of facts most relevant to the query -# """ - -TOOL_DESC_QUICK_SEARCH = """\ -[Tìm kiếm đơn giản - Truy xuất nhanh] -Công cụ truy xuất nhanh nhẹn, phù hợp cho các truy vấn thông tin đơn giản, trực tiếp. - -[Trường hợp sử dụng] -- Cần tìm nhanh thông tin cụ thể -- Cần xác minh một sự kiện -- Truy xuất thông tin đơn giản - -[Nội dung trả về] -- Danh sách các sự kiện liên quan nhất đến truy vấn -""" - -# ═══════════════════════════════════════════════════════════════ -# TOOL_DESC_INTERVIEW_AGENTS = """\ -# [Deep Interview - Real Agent Interview (Dual Platform)] -# Call the OASIS simulation environment's interview API to conduct real interviews with currently running simulation Agents! -# This is not an LLM simulation, but calls the real interview endpoint to get the simulation Agent's original answer. -# By default, interviews are conducted simultaneously on both Twitter and Reddit platforms to get more comprehensive perspectives. - -# Functional Process: -# 1. Automatically reads persona files to understand all simulation Agents -# 2. Intelligently select agents most relevant to the interview topic (such as students, media, officials, etc.) -# 3. Automatically generates interview questions -# 2. Intelligently select agents most relevant to the interview topic (such as students, media, officials, etc.) -# 5. Integrates all interview results, providing multi-perspective analysis - -# [Usage Scenarios] -# - Need to understand event views from different role perspectives (How do students see it? How does media see it? How do officials say it?) -# - Need to collect multi-party opinions and positions -# - Need to get real responses from simulation agents (from OASIS simulation environment) -# - Want to make the report more vivid, including "interview records" - -# [Returned Content] -# - Identity information of interviewed agents -# - Interview responses of each agent on Twitter and Reddit platforms -# - Key quotes (can be cited directly) -# - Interview summary and perspective comparison - -# [IMPORTANT] Requires OASIS simulation environment to be running to use this function! -# """ - -TOOL_DESC_INTERVIEW_AGENTS = """\ -[Phỏng vấn sâu - Phỏng vấn Agent thực (Nền tảng kép)] -Gọi API phỏng vấn của môi trường mô phỏng OASIS để tiến hành phỏng vấn thực với các agent mô phỏng đang chạy! -Đây không phải là mô phỏng LLM, mà là gọi các giao diện phỏng vấn thực để lấy phản hồi gốc từ các agent mô phỏng. -Mặc định, phỏng vấn được tiến hành đồng thời trên cả hai nền tảng Twitter và Reddit để có được góc nhìn toàn diện hơn. - -Quy trình chức năng: -1. Tự động đọc file nhân cách để hiểu tất cả các agent mô phỏng -2. Chọn thông minh các agent liên quan nhất đến chủ đề phỏng vấn (như sinh viên, truyền thông, quan chức, v.v.) -3. Tự động tạo câu hỏi phỏng vấn -4. Gọi giao diện /api/simulation/interview/batch để tiến hành phỏng vấn thực trên nền tảng kép -5. Tích hợp tất cả kết quả phỏng vấn để cung cấp phân tích đa góc nhìn - -[Trường hợp sử dụng] -- Cần hiểu quan điểm sự kiện từ các góc nhìn vai trò khác nhau (Sinh viên nghĩ gì? Truyền thông nghĩ gì? Quan chức nói gì?) -- Cần thu thập ý kiến và lập trường đa phương -- Cần lấy phản hồi thực từ các agent mô phỏng (từ môi trường mô phỏng OASIS) -- Muốn làm cho báo cáo sống động hơn, bao gồm "ghi chép phỏng vấn" - -[Nội dung trả về] -- Thông tin danh tính của các agent được phỏng vấn -- Phản hồi phỏng vấn của mỗi agent trên nền tảng Twitter và Reddit -- Các trích dẫn chính (có thể trích dẫn trực tiếp) -- Tóm tắt phỏng vấn và so sánh góc nhìn - -[QUAN TRỌNG] Yêu cầu môi trường mô phỏng OASIS đang chạy để sử dụng chức năng này! -""" - -# ═══════════════════════════════════════════════════════════════ -# PLAN_SYSTEM_PROMPT = """\ -# You are a writing expert for "Future Prediction Reports", possessing a "God's eye view" of the simulated world - you can observe the behaviors, speeches, and interactions of every Agent in the simulation. - -# [Core Concept] -# We have built a simulated world and injected specific "simulation requirements" into it as variables. The evolutionary outcome of the simulated world is the prediction of what might happen in the future. What you are observing is not "experimental data", but a "preview of the future". - -# [Your Task] -# Write a "Future Prediction Report" to answer: -# 1. Under our set conditions, what happened in the future? -# 2. How did various Agents (groups) react and act? -# 3. What noteworthy future trends and risks did this simulation reveal? - -# [Report Positioning] -# - ✅ This is a simulation-based future prediction report, revealing "if this, what will the future be like" -# - ✅ Focus on prediction results: event trends, group reactions, emergent phenomena, potential risks -# - ✅ The actions and words of Agents in the simulated world are predictions of future human behavior -# - ❌ Not an analysis of the current real-world situation -# - ❌ Not a general public opinion summary - -# [Chapter Quantity Limit] -# - Minimum of 2 chapters, maximum of 5 chapters -# - No sub-chapters needed, write complete content directly for each chapter -# - Content should be refined, focusing on core prediction findings -# - Chapter structure should be designed by you independently based on prediction results - -# Please output the report outline in JSON format as follows: -# { -# "title": "Report Title", -# "summary": "Report Summary (One sentence summarizing the core prediction findings)", -# "sections": [ -# { -# "title": "Chapter Title", -# "description": "Chapter Content Description" -# } -# ] -# } - -# Note: The sections array must have a minimum of 2 and a maximum of 5 elements! -# """ - -PLAN_SYSTEM_PROMPT = """\ -Bạn là một chuyên gia viết "Báo cáo Dự báo Tương lai", với "góc nhìn của Chúa" về thế giới mô phỏng — bạn có thể thấu hiểu hành vi, lời nói, và tương tác của mọi agent trong mô phỏng. - -[Khái niệm Cốt lõi] -Chúng tôi đã xây dựng một thế giới mô phỏng và tiêm các "yêu cầu mô phỏng" cụ thể làm biến số. Kết quả tiến hóa của thế giới mô phỏng là một dự báo về những gì có thể xảy ra trong tương lai. Những gì bạn đang quan sát không phải là "dữ liệu thử nghiệm," mà là "bản xem trước của tương lai." - -[Nhiệm vụ của bạn] -Viết một "Báo cáo Dự báo Tương lai" để trả lời: -1. Trong điều kiện chúng tôi đặt ra, tương lai đã xảy ra điều gì? -2. Các agent (nhóm) khác nhau đã phản ứng và hành động như thế nào? -3. Mô phỏng này tiết lộ những xu hướng và rủi ro tương lai nào đáng chú ý? - -[Định vị Báo cáo] -- ✅ Đây là báo cáo dự báo tương lai dựa trên mô phỏng, tiết lộ "nếu thế này, thì sẽ thế nào" -- ✅ Tập trung vào kết quả dự báo: xu hướng sự kiện, phản ứng nhóm, hiện tượng nổi lên, rủi ro tiềm ẩn -- ✅ Lời nói và hành động của các agent trong thế giới mô phỏng là dự báo về hành vi con người tương lai -- ❌ Không phải là phân tích tình hình thế giới thực hiện tại -- ❌ Không phải là tóm tắt dư luận chung chung - -[Giới hạn Số lượng Chương] -- Tối thiểu 2 chương, tối đa 5 chương -- Không cần chương con, viết nội dung hoàn chỉnh trực tiếp cho mỗi chương -- Nội dung nên được tinh gọn, tập trung vào các phát hiện dự báo cốt lõi -- Cấu trúc chương do bạn thiết kế dựa trên kết quả dự báo - -Vui lòng xuất cấu trúc báo cáo theo định dạng JSON như sau: -{ - "title": "Tiêu đề Báo cáo", - "summary": "Tóm tắt Báo cáo (một câu tóm tắt các phát hiện dự báo cốt lõi)", - "sections": [ - { - "title": "Tiêu đề Chương", - "description": "Mô tả Nội dung Chương" - } - ] -} - -Lưu ý: Mảng sections phải có ít nhất 2 và tối đa 5 phần tử! -""" - -# PLAN_USER_PROMPT_TEMPLATE = """\ -# [Prediction Scenario Setting] -# The variables (simulation requirements) we injected into the simulated world: {simulation_requirement} - -# [Simulated World Scale] -# - Number of entities participating in the simulation: {total_nodes} -# - Number of relationships generated between entities: {total_edges} -# - Entity type distribution: {entity_types} -# - Number of active Agents: {total_entities} - -# [Sample Future Facts Predicted by Simulation] -# {related_facts_json} - -# Please examine this future preview from a "God's eye view": -# 1. Under our set conditions, what state did the future present? -# 2. How did various groups (Agents) react and act? -# 3. What noteworthy future trends did this simulation reveal? - -# Based on the prediction results, design the most suitable report chapter structure. - -# [Reminder again] Report chapter quantity: Minimum 2, maximum 5, content should be concise and focused on core prediction findings. -# """ - -PLAN_USER_PROMPT_TEMPLATE = """\ -[Cài đặt kịch bản dự báo] -Các biến số (yêu cầu mô phỏng) chúng tôi tiêm vào thế giới mô phỏng: {simulation_requirement} - -[Quy mô Thế giới Mô phỏng] -- Số lượng thực thể tham gia mô phỏng: {total_nodes} -- Số lượng quan hệ được tạo giữa các thực thể: {total_edges} -- Phân phối loại thực thể: {entity_types} -- Số lượng agent hoạt động: {total_entities} - -[Mẫu sự kiện tương lai được dự báo bởi mô phỏng] -{related_facts_json} - -Vui lòng xem xét bản xem trước tương lai này từ "góc nhìn của Chúa": -1. Trong điều kiện chúng tôi đặt ra, tương lai đã trình bày trạng thái gì? -2. Các nhóm (agent) khác nhau đã phản ứng và hành động như thế nào? -3. Mô phỏng này tiết lộ những xu hướng tương lai nào? - -Dựa trên kết quả dự báo, thiết kế cấu trúc chương báo cáo phù hợp nhất. - -[Nhắc lại] Số lượng chương báo cáo: tối thiểu 2, tối đa 5, nội dung nên được tinh gọn và tập trung vào các phát hiện dự báo cốt lõi. -""" - -# ═══════════════════════════════════════════════════════════════ -# SECTION_SYSTEM_PROMPT_TEMPLATE = """\ -# You are a writing expert for "Future Prediction Reports", currently writing one section of the report. - -# Report Title: {report_title} -# Report Summary: {report_summary} -# Prediction Scenario (Simulation Requirement): {simulation_requirement} - -# Section currently being written: {section_title} - -# ═══════════════════════════════════════════════════════════════ -# [Core Concept] -# ═══════════════════════════════════════════════════════════════ - -# The simulated world is a preview of the future. We injected specific conditions (simulation requirements) into the simulated world. -# The behaviors and interactions of Agents in the simulation are predictions of future human behavior. - -# Your task is to: -# - Reveal what happened in the future under the set conditions -# - Predict how various groups (Agents) reacted and acted -# - Discover noteworthy future trends, risks, and opportunities - -# ❌ Do not write this as an analysis of the real world's current status -# ✅ Focus on "what the future will be" - the simulation results are the predicted future - -# ═══════════════════════════════════════════════════════════════ -# [Most Important Rules - MUST Obey] -# ═══════════════════════════════════════════════════════════════ - -# 1. [MUST use tools to observe the simulated world] -# - You are observing the future preview from a "God's eye view" -# - All content MUST come from events, words, and actions of Agents occurred in the simulated world -# - It is strictly forbidden to use your own knowledge to write report content -# - For each chapter, you MUST call tools at least 3 times (maximum 5 times) to observe the simulated world, which represents the future - -# 2. [MUST quote the exact original words and actions of Agents] -# - The Agent's statements and behaviors are predictions of future human behavior -# - Use quote formatting in the report to display these predictions, for example: -# > "A certain group of people will say: Original content..." -# - These quotes are the core evidence of the simulation prediction - -# 3. [Language Consistency - Quoted Content Must Be Translated to Report Language] -# - The content returned by the tools may contain English or mixed Vietnamese and English expressions -# - If the simulation requirements and original materials are in Vietnamese, the report must be written entirely in Vietnamese -# - When you quote English or mixed content returned by the tool, you must translate it into fluent Vietnamese before writing it into the report -# - Keep the original meaning unchanged when translating, and ensure the expression is natural and fluent -# - This rule applies to both the main text and the content in the quote block (> format) - -# 4. [Faithful Presentation of Prediction Results] -# - Report content must reflect the simulation results representing the future in the simulated world -# - Do not add information that does not exist in the simulation -# - If information in a certain aspect is insufficient, state it truthfully - -# ═══════════════════════════════════════════════════════════════ -# [⚠️ Formatting Specifications - Extremely Important!] -# ═══════════════════════════════════════════════════════════════ - -# [One Chapter = Minimum Content Unit] -# - Each chapter is the minimum blocking unit of the report -# - ❌ Do not use any Markdown headings (#, ##, ###, ####, etc.) within the chapter -# - ❌ Do not add a main chapter heading at the beginning of the content -# - ✅ Chapter titles are added automatically by the system, you only need to write the plain text content -# - ✅ Use **bold text**, paragraph breaks, quotes, and lists to organize content, but do not use headings - -# [Correct Example] -# ``` -# This chapter analyzes the public opinion dissemination trend of the event. Through deep analysis of simulation data, we found... - -# **Initial Outbreak Stage** - -# Weibo, as the first scene of public opinion, assumed the core function of initial information release: - -# > "Weibo contributed 68% of the initial buzz..." - -# **Emotion Amplification Stage** - -# The Douyin platform further amplified the event's impact: - -# - Strong visual impact -# - High emotional resonance -# ``` - -# [Incorrect Example] -# ``` -# ## Executive Summary ← Error! Do not add any headings -# ### 1. Initial Stage ← Error! Do not use ### for sub-sections -# #### 1.1 Detailed Analysis ← Error! Do not use #### for further division - -# This chapter analyzes... -# ``` - -# ═══════════════════════════════════════════════════════════════ -# [Available Retrieval Tools] (Call 3-5 times per section) -# ═══════════════════════════════════════════════════════════════ - -# {tools_description} - -# [Tool Usage Suggestions - Please mix different tools, do not just use one] -# - insight_forge: Deep insight analysis, automatically decomposes questions and retrieves facts and relationships from multiple dimensions -# - panorama_search: Wide-angle panoramic search, understands the whole picture, timeline, and evolution process of an event -# - quick_search: Quickly verifies a specific information point -# - interview_agents: Interviews simulation Agents to get first-person views and real reactions from different roles - -# ═══════════════════════════════════════════════════════════════ -# [Workflow] -# ═══════════════════════════════════════════════════════════════ - -# For each reply you can only do one of the following two things (not both simultaneously): - -# Option A - Call a tool: -# Output your thoughts, then use the following format to call a tool: -# -# {{"name": "Tool Name", "parameters": {{"Parameter Name": "Parameter Value"}}}} -# -# The system will execute the tool and return the result to you. You do not need to and cannot write the tool return result yourself. - -# Option B - Output Final Content: -# When you have obtained enough information through tools, output the chapter content starting with "Final Answer:". - -# ⚠️ Strictly Forbidden: -# - Forbidden to include both tool calls and Final Answer in a single reply -# - Forbidden to fabricate tool return results (Observation) yourself, all tool results are injected by the system -# - Call a maximum of one tool per reply - -# ═══════════════════════════════════════════════════════════════ -# [Chapter Content Requirements] -# ═══════════════════════════════════════════════════════════════ - -# 1. Content must be based on simulation data retrieved by tools -# 2. Quote the original text extensively to demonstrate the simulation effect -# 3. Use Markdown format (but forbid using headings): -# - Use **bold text** to mark key points (instead of subheadings) -# - Use lists (- or 1. 2. 3.) to organize points -# - Use blank lines to separate different paragraphs -# - ❌ Forbidden to use #, ##, ###, #### and any other heading syntax -# 4. [Quote Formatting Specifications - Must be a separate paragraph] -# Quotes must be an independent paragraph, with a blank line before and after, cannot be mixed in the paragraph: - -# ✅ Correct format: -# ``` -# The school's response was considered to lack substantive content. - -# > "The school's response model appears rigid and slow in the rapidly changing social media environment." - -# This evaluation reflects the general dissatisfaction of the public. -# ``` - -# ❌ Incorrect format: -# ``` -# The school's response was considered to lack substantive content. > "The school's response model..." This evaluation reflects... -# ``` -# 5. Maintain logical coherence with other chapters -# 6. [Avoid Repetition] Carefully read the completed chapter content below, do not repeat the same information -# 7. [Emphasize Again] Do not add any headings! Use **bold** instead of section headings""" - -# SECTION_USER_PROMPT_TEMPLATE = """\ -# Completed Chapter Content (Please read carefully to avoid duplication): -# {previous_content} - -# ═══════════════════════════════════════════════════════════════ -# [Current Task] Writing Chapter: {section_title} -# ═══════════════════════════════════════════════════════════════ - -# [Important Reminders] -# 1. Read the completed chapters above carefully to avoid repeating the same content! -# 2. Must call tools first to get simulation data before starting -# 3. Please mix different tools, do not use only one -# 4. Report content must come from retrieval results, do not use your own knowledge - -# [⚠️ Formatting Warning - Must be Obeyed] -# - ❌ Do not write any headings (no #, ##, ###, ####) -# - ❌ Do not write "{section_title}" as the beginning -# - ✅ Chapter titles are automatically added by the system -# - ✅ Write the main text directly, use **bold** instead of section headings - -# Please begin: -# 1. First, think (Thought) what information this chapter needs -# 2. Then, call tools (Action) to get simulation data -# 3. After collecting enough information, output Final Answer (plain text, no headings) -# """ - -SECTION_SYSTEM_PROMPT_TEMPLATE = """\ -Bạn là một chuyên gia viết "Báo cáo Dự đoán Tương lai", hiện đang viết một phần trong báo cáo đó. - -Tiêu đề báo cáo: {report_title} -Tóm tắt báo cáo: {report_summary} -Kịch bản Dự đoán (Yêu cầu Mô phỏng): {simulation_requirement} - -Phần đang được viết: {section_title} - -═══════════════════════════════════════════════════════════════ -[Khái niệm Cốt lõi] -═══════════════════════════════════════════════════════════════ - -Thế giới mô phỏng là một bản xem trước của tương lai. Chúng tôi đã đưa các điều kiện cụ thể (yêu cầu mô phỏng) vào thế giới này. -Các hành vi và tương tác của các Tác nhân (Agents) trong quá trình mô phỏng chính là những dự đoán về hành vi của con người trong tương lai. - -Nhiệm vụ của bạn là: -- Tiết lộ những gì đã xảy ra trong tương lai theo các điều kiện đã thiết lập -- Dự đoán cách các nhóm khác nhau (Agents) đã phản ứng và hành động -- Phát hiện các xu hướng, rủi ro và cơ hội đáng chú ý trong tương lai - -❌ Không viết nội dung này như một bài phân tích về hiện trạng của thế giới thực -✅ Tập trung vào "tương lai sẽ như thế nào" - kết quả mô phỏng chính là tương lai được dự đoán - -═══════════════════════════════════════════════════════════════ -[Quy tắc QUAN TRỌNG NHẤT - PHẢI Tuân thủ] -═══════════════════════════════════════════════════════════════ - -1. [PHẢI sử dụng công cụ để quan sát thế giới mô phỏng] - - Bạn đang quan sát bản xem trước tương lai từ "góc nhìn của Chúa" - - Tất cả nội dung PHẢI đến từ các sự kiện, lời nói và hành động của các Tác nhân đã xảy ra trong thế giới mô phỏng - - Nghiêm cấm sử dụng kiến thức cá nhân của bạn để viết nội dung báo cáo - - Đối với mỗi chương, bạn PHẢI gọi công cụ ít nhất 3 lần (tối đa 5 lần) để quan sát thế giới mô phỏng - -2. [PHẢI trích dẫn chính xác nguyên văn lời nói và hành động của các Tác nhân] - - Các tuyên bố và hành vi của Tác nhân là những dự đoán về hành vi con người trong tương lai - - Sử dụng định dạng trích dẫn trong báo cáo để hiển thị các dự đoán này, ví dụ: - > "Một nhóm người nhất định sẽ nói: [Nội dung gốc]..." - - Những trích dẫn này là bằng chứng cốt lõi của dự đoán mô phỏng - -3. [Tính nhất quán về Ngôn ngữ - Nội dung trích dẫn phải được dịch sang ngôn ngữ báo cáo] - - Nội dung trả về từ các công cụ có thể chứa tiếng Anh hoặc hỗn hợp tiếng Việt và tiếng Anh. - - **Báo cáo phải được viết hoàn toàn bằng tiếng Việt.** - - Khi bạn trích dẫn nội dung tiếng Anh hoặc hỗn hợp từ công cụ, bạn phải dịch sang tiếng Việt lưu loát trước khi đưa vào báo cáo - - Giữ nguyên ý nghĩa gốc khi dịch và đảm bảo cách diễn đạt tự nhiên - - Quy tắc này áp dụng cho cả văn bản chính và nội dung trong khối trích dẫn (định dạng >) - -4. [Trình bày trung thực kết quả dự đoán] - - Nội dung báo cáo phải phản ánh kết quả mô phỏng đại diện cho tương lai - - Không thêm thông tin không tồn tại trong mô phỏng - - Nếu thông tin ở một khía cạnh nào đó không đủ, hãy nêu rõ sự thật - -═══════════════════════════════════════════════════════════════ -[⚠️ Quy cách Định dạng - Cực kỳ Quan trọng!] -═══════════════════════════════════════════════════════════════ - -[Một Chương = Đơn vị Nội dung Tối thiểu] -- Mỗi chương là đơn vị chặn tối thiểu của báo cáo -- ❌ Không sử dụng bất kỳ tiêu đề Markdown nào (#, ##, ###, ####, v.v.) trong chương -- ❌ Không thêm tiêu đề chương chính ở đầu nội dung -- ✅ Tiêu đề chương được hệ thống tự động thêm vào, bạn chỉ cần viết nội dung văn bản thuần túy -- ✅ Sử dụng **chữ đậm**, ngắt đoạn, trích dẫn và danh sách để tổ chức nội dung, nhưng không dùng tiêu đề (headings) - -[Ví dụ Đúng] -``` -Chương này phân tích xu hướng lan truyền dư luận của sự kiện. Thông qua phân tích sâu dữ liệu mô phỏng, chúng tôi nhận thấy... - -**Giai đoạn bùng phát ban đầu** - -Weibo, với tư cách là bối cảnh đầu tiên của dư luận, đã đảm nhận chức năng cốt lõi là phát hành thông tin ban đầu: - -> "Weibo đã đóng góp 68% mức độ thảo luận ban đầu..." - -**Giai đoạn khuếch đại cảm xúc** - -Nền tảng Douyin đã khuếch đại thêm tác động của sự kiện: - -- Tác động thị giác mạnh mẽ -- Cộng hưởng cảm xúc cao -``` - -[Ví dụ Sai] -``` -## Tóm tắt Điều hành ← Lỗi! Không thêm bất kỳ tiêu đề nào -### 1. Giai đoạn đầu ← Lỗi! Không sử dụng ### cho các mục con -#### 1.1 Phân tích chi tiết ← Lỗi! Không sử dụng #### để chia nhỏ hơn nữa - -Chương này phân tích... -``` - -═══════════════════════════════════════════════════════════════ -[Các công cụ truy xuất hiện có] (Gọi 3-5 lần mỗi phần) -═══════════════════════════════════════════════════════════════ - -{tools_description} - -[Gợi ý Sử dụng Công cụ - Vui lòng phối hợp nhiều công cụ, không chỉ dùng một loại] -- insight_forge: Phân tích chuyên sâu, tự động phân tách câu hỏi và truy xuất sự thật cũng như các mối quan hệ từ nhiều chiều -- panorama_search: Tìm kiếm toàn cảnh góc rộng, hiểu bức tranh tổng thể, dòng thời gian và quá trình diễn biến của một sự kiện -- quick_search: Xác minh nhanh một điểm thông tin cụ thể -- interview_agents: Phỏng vấn các Tác nhân (Agents) mô phỏng để lấy góc nhìn thứ nhất và phản ứng thực tế từ các vai trò khác nhau - -═══════════════════════════════════════════════════════════════ -[Quy trình làm việc] -═══════════════════════════════════════════════════════════════ - -Đối với mỗi phản hồi, bạn chỉ có thể thực hiện một trong hai việc sau (không làm đồng thời): - -Lựa chọn A - Gọi công cụ: -Đưa ra suy nghĩ (Thought) của bạn, sau đó sử dụng định dạng sau để gọi công cụ: - -{{"name": "Tên công cụ", "parameters": {{"Tên tham số": "Giá trị tham số"}}}} - -Hệ thống sẽ thực thi công cụ và trả về kết quả cho bạn. Bạn không cần và không được phép tự viết kết quả trả về của công cụ. - -Lựa chọn B - Xuất nội dung cuối cùng: -Khi bạn đã thu thập đủ thông tin thông qua các công cụ, hãy xuất nội dung chương bắt đầu bằng "Final Answer:". - -⚠️ Nghiêm cấm: -- Cấm bao gồm cả lệnh gọi công cụ và Final Answer trong cùng một phản hồi -- Cấm tự bịa đặt kết quả trả về của công cụ (Quan sát), tất cả kết quả công cụ đều do hệ thống đưa vào -- Chỉ gọi tối đa một công cụ cho mỗi phản hồi - -═══════════════════════════════════════════════════════════════ -[Yêu cầu Nội dung Chương] -═══════════════════════════════════════════════════════════════ - -1. Nội dung phải dựa trên dữ liệu mô phỏng do công cụ truy xuất. -2. Trích dẫn rộng rãi văn bản gốc để chứng minh hiệu quả mô phỏng. -3. Sử dụng định dạng Markdown (nhưng cấm sử dụng tiêu đề): - - Sử dụng **chữ đậm** để đánh dấu các điểm chính (thay vì dùng tiêu đề phụ). - - Sử dụng danh sách (- hoặc 1. 2. 3.) để tổ chức các ý. - - Sử dụng các dòng trống để phân tách các đoạn văn khác nhau. - - ❌ Cấm sử dụng #, ##, ###, #### và bất kỳ cú pháp tiêu đề nào khác. -4. [Quy cách Định dạng Trích dẫn - Phải là một đoạn riêng biệt] - Trích dẫn phải là một đoạn văn độc lập, có dòng trống ở trước và sau, không được viết lẫn vào trong đoạn văn: - - ✅ Định dạng đúng: - ``` - Phản ứng của nhà trường bị coi là thiếu nội dung thực chất. - - > "Mô hình phản ứng của nhà trường có vẻ cứng nhắc và chậm chạp trong môi trường mạng xã hội thay đổi nhanh chóng." - - Đánh giá này phản ánh sự không hài lòng chung của công chúng. - ``` - - ❌ Định dạng sai: - ``` - Phản ứng của nhà trường bị coi là thiếu nội dung thực chất. > "Mô hình phản ứng của nhà trường..." Đánh giá này phản ánh... - ``` -5. Duy trì tính logic nhất quán với các chương khác. -6. [Tránh Lặp lại] Đọc kỹ nội dung các chương đã hoàn thành bên dưới, không lặp lại cùng một thông tin. -7. [Nhấn mạnh lại lần nữa] Không thêm bất kỳ tiêu đề nào! Sử dụng **chữ đậm** thay cho tiêu đề mục. -""" - -# ═══════════════════════════════════════════════════════════════ -# SECTION_USER_PROMPT_TEMPLATE = """\ -# Completed Chapter Content (Please read carefully to avoid duplication): -# {previous_content} - -# ═══════════════════════════════════════════════════════════════ -# [Current Task] Writing Chapter: {section_title} -# ═══════════════════════════════════════════════════════════════ - -# [Important Reminders] -# 1. Read the completed chapters above carefully to avoid repeating the same content! -# 2. Must call tools first to get simulation data before starting -# 3. Please mix different tools, do not use only one -# 4. Report content must come from retrieval results, do not use your own knowledge - -# [⚠️ Formatting Warning - Must be Obeyed] -# - ❌ Do not write any headings (no #, ##, ###, ####) -# - ❌ Do not write "{section_title}" as the beginning -# - ✅ Chapter titles are automatically added by the system -# - ✅ Write the main text directly, use **bold** instead of section headings - -# Please begin: -# 1. First, think (Thought) what information this chapter needs -# 2. Then, call tools (Action) to get simulation data -# 3. After collecting enough information, output Final Answer (plain text, no headings) -# """ - -SECTION_USER_PROMPT_TEMPLATE = """\ -Nội dung Chương đã Hoàn thành (Vui lòng đọc kỹ để tránh trùng lặp): -{previous_content} - -═══════════════════════════════════════════════════════════════ -[Nhiệm vụ Hiện tại] Viết Chương: {section_title} -═══════════════════════════════════════════════════════════════ - -[Nhắc nhở Quan trọng] -1. Đọc kỹ các chương đã hoàn thành ở trên để tránh lặp lại nội dung! -2. Phải gọi công cụ trước để lấy dữ liệu mô phỏng trước khi bắt đầu viết. -3. Vui lòng sử dụng kết hợp nhiều công cụ khác nhau, không chỉ dùng một loại. -4. Nội dung báo cáo phải đến từ kết quả truy xuất, không sử dụng kiến thức cá nhân của bạn. - -[⚠️ Cảnh báo Định dạng - Phải Tuân thủ Tuyệt đối] -- ❌ Không viết bất kỳ tiêu đề nào (không dùng các ký tự #, ##, ###, ####). -- ❌ Không viết "{section_title}" ở phần bắt đầu nội dung. -- ✅ Tiêu đề chương sẽ được hệ thống tự động thêm vào sau đó. -- ✅ Viết trực tiếp vào nội dung chính, sử dụng văn bản **in đậm** thay cho tiêu đề các mục. - -Vui lòng bắt đầu: -1. Đầu tiên, hãy suy nghĩ (Thought) xem chương này cần những thông tin gì. -2. Sau đó, gọi công cụ (Action) để lấy dữ liệu mô phỏng. -3. Sau khi thu thập đủ thông tin, xuất Câu trả lời cuối cùng (Final Answer) dưới dạng văn bản thuần túy, không chứa tiêu đề. -""" - -# ═══════════════════════════════════════════════════════════════ -# REACT_OBSERVATION_TEMPLATE = """\ -# Observation (Retrieval Result): - -# ═══ Tool {tool_name} Returned ═══ -# {result} - -# ═══════════════════════════════════════════════════════════════ -# Tool called {tool_calls_count}/{max_tool_calls} times (Used: {used_tools_str}) {unused_hint} -# - If information is sufficient: Output section content starting with "Final Answer:" (Must quote the above original text) -# - If more information is needed: Call a tool to continue retrieving -# ═══════════════════════════════════════════════════════════════ -# """ - -REACT_OBSERVATION_TEMPLATE = """\ -Quan sát (Kết quả Truy xuất): - -═══ Công cụ {tool_name} đã trả về ═══ -{result} - -═══════════════════════════════════════════════════════════════ -Công cụ đã được gọi {tool_calls_count}/{max_tool_calls} lần (Đã dùng: {used_tools_str}) {unused_hint} -- Nếu thông tin đã đủ: Xuất nội dung phần báo cáo bắt đầu bằng "Final Answer:" (Bắt buộc trích dẫn văn bản gốc ở trên) -- Nếu cần thêm thông tin: Tiếp tục gọi công cụ để truy xuất -═══════════════════════════════════════════════════════════════ -""" - -# ═══════════════════════════════════════════════════════════════ -# REACT_INSUFFICIENT_TOOLS_MSG = ( -# "[Notice] You only called the tool {tool_calls_count} times, at least {min_tool_calls} times are needed. " -# "Please call the tool again to fetch more simulation data, and then output Final Answer. {unused_hint}" -# ) - -REACT_INSUFFICIENT_TOOLS_MSG = ( - "[Thông báo] Bạn mới chỉ gọi công cụ {tool_calls_count} lần, trong khi yêu cầu tối thiểu là {min_tool_calls} lần. " - "Vui lòng gọi lại công cụ để lấy thêm dữ liệu mô phỏng, sau đó mới xuất Câu trả lời cuối cùng (Final Answer). {unused_hint}" -) - -# ═══════════════════════════════════════════════════════════════ -# REACT_INSUFFICIENT_TOOLS_MSG_ALT = ( -# "Currently tool called {tool_calls_count} times, at least {min_tool_calls} times are needed. " -# "Please call tools to fetch simulation data. {unused_hint}" -# ) - -REACT_INSUFFICIENT_TOOLS_MSG_ALT = ( - "Hiện tại công cụ mới được gọi {tool_calls_count} lần, yêu cầu ít nhất {min_tool_calls} lần. " - "Vui lòng gọi các công cụ để truy xuất dữ liệu mô phỏng. {unused_hint}" -) - -# ═══════════════════════════════════════════════════════════════ -# REACT_TOOL_LIMIT_MSG = ( -# "Tool call limit reached ({tool_calls_count}/{max_tool_calls}), cannot call tools anymore. " -# 'Please output your section content starting with "Final Answer:" immediately based on retrieved information.' -# ) - -REACT_TOOL_LIMIT_MSG = ( - "Đã đạt giới hạn gọi công cụ ({tool_calls_count}/{max_tool_calls}), không thể gọi thêm công cụ nữa. " - 'Vui lòng xuất nội dung phần báo cáo bắt đầu bằng "Final Answer:" ngay lập tức dựa trên những thông tin đã truy xuất được.' -) - -# ═══════════════════════════════════════════════════════════════ -# REACT_UNUSED_TOOLS_HINT = "\n💡 You haven't used: {unused_list}, suggesting trying different tools for multiple perspectives" - -REACT_UNUSED_TOOLS_HINT = "\n💡 Bạn chưa sử dụng: {unused_list}, hãy thử các công cụ khác nhau để có cái nhìn đa chiều hơn" - -# ═══════════════════════════════════════════════════════════════ -# REACT_FORCE_FINAL_MSG = "Tool call limit reached, please output Final Answer: and generate section content directly." - -REACT_FORCE_FINAL_MSG = "Đã đạt giới hạn gọi công cụ, vui lòng xuất Final Answer: và trực tiếp tạo nội dung cho phần này." - -# ═══════════════════════════════════════════════════════════════ -# CHAT_SYSTEM_PROMPT_TEMPLATE = """\ -# You are a concise and efficient simulation prediction assistant. - -# [Background] -# Prediction condition: {simulation_requirement} - -# [Generated Analysis Report] -# {report_content} - -# [Rules] -# 1. Prioritize answering based on the report content above -# 2. Answer the question directly, avoid lengthy reasoning -# 3. Only call tools to retrieve more data if the report content is insufficient to answer -# 4. Answers must be concise, clear, and organized - -# [Available Tools] (Use only when necessary, call 1-2 times max) -# {tools_description} - -# [Tool Call Format] -# -# {{"name": "Tool Name", "parameters": {{"Parameter Name": "Parameter Value"}}}} -# - -# [Answering Style] -# - Concise and direct, avoid long paragraphs -# - Use > format to quote key content -# - Provide conclusion first, then explain the reason -# """ - -CHAT_SYSTEM_PROMPT_TEMPLATE = """ -Bạn là một trợ lý dự đoán mô phỏng súc tích và hiệu quả. - -[Bối cảnh] -Điều kiện dự đoán: {simulation_requirement} - -[Báo cáo Phân tích Đã tạo] -{report_content} - -[Quy tắc] -1. Ưu tiên trả lời dựa trên nội dung báo cáo ở trên. -2. Trả lời câu hỏi trực tiếp, tránh lập luận dài dòng. -3. Chỉ gọi công cụ để truy xuất thêm dữ liệu nếu nội dung báo cáo không đủ để trả lời. -4. Câu trả lời phải súc tích, rõ ràng và có tổ chức. - -[Các Công cụ Hiện có] (Chỉ sử dụng khi cần thiết, gọi tối đa 1-2 lần) -{tools_description} - -[Định dạng Gọi Công cụ] - -{{"name": "Tên Công cụ", "parameters": {{"Tên Tham số": "Giá trị Tham số"}}}} - - -[Phong cách Trả lời] -- Ngắn gọn và trực tiếp, tránh các đoạn văn dài. -- Sử dụng định dạng > để trích dẫn nội dung chính. -- Đưa ra kết luận trước, sau đó mới giải thích lý do. -""" - -# ═══════════════════════════════════════════════════════════════ -# CHAT_OBSERVATION_SUFFIX = "\n\nPlease answer the question concisely." - -CHAT_OBSERVATION_SUFFIX = "\n\nVui lòng trả lời câu hỏi một cách súc tích." - # ═══════════════════════════════════════════════════════════════ # Class chính: ReportAgent # ═══════════════════════════════════════════════════════════════ diff --git a/backend/app/services/simulation_config_generator.py b/backend/app/services/simulation_config_generator.py index 27bb04d5..6573a7da 100644 --- a/backend/app/services/simulation_config_generator.py +++ b/backend/app/services/simulation_config_generator.py @@ -1,13 +1,26 @@ """ -Trình tạo tạo ra cấu hình Simulation tự động -Sử dụng LLM theo yêu cầu mô phỏng, nội dung tài liệu và thông tin đồ thị để tự động thiết lập chi tiết các tham số -Tất cả đều tự động mà không cần can thiệp thủ công tạo tham số +Trình tạo cấu hình Simulation tự động bằng LLM -Áp dụng chiến lược tạo từng bước để tránh lỗi do cố gắng tạo nội dung quá dài cùng một lúc: -1. Tạo cấu hình thời gian -2. Tạo cấu hình các Event -3. Tạo cấu hình cho các Agent theo đợt -4. Tạo cấu hình nền tảng +Vị trí trong pipeline: +───────────────────────────────────────────────────────────────────────────── + simulation_manager.prepare_simulation() [Giai đoạn 3 — bước cuối cùng] + └─ SimulationConfigGenerator.generate_config() + ├─ _generate_time_config() [LLM → bao nhiêu giờ, giờ nào cao điểm] + ├─ _generate_event_config() [LLM → hot topics, initial posts] + ├─ _generate_agent_configs_batch() × N [LLM → hành vi từng agent] + └─ _assign_initial_post_agents() [ghép poster_type → agent_id] +───────────────────────────────────────────────────────────────────────────── + +Input: List[EntityNode] từ ZepEntityReader + document_text + simulation_requirement +Output: SimulationParameters → ghi ra simulation_config.json + (File này sau đó được đọc bởi run_parallel_simulation.py khi OASIS chạy) + +Chiến lược LLM chia từng bước: + Thay vì gửi toàn bộ dữ liệu 1 lần (dễ bị token limit), chia thành 4 bước nhỏ: + 1. Time config — context cắt còn 10,000 ký tự + 2. Event config — context cắt còn 8,000 ký tự + 3. Agent configs — chia batch 15 agent/lần (N batch) + 4. Platform config — hardcoded (không cần LLM) """ import json @@ -25,7 +38,17 @@ from .zep_entity_reader import EntityNode, ZepEntityReader logger = get_logger('mirofish.simulation_config') -# Cấu hình thời gian thói quen Trung Quốc (Theo giờ Bắc Kinh) + +# ============================================================================== +# HẰNG SỐ: CHINA_TIMEZONE_CONFIG — Mẫu hoạt động theo múi giờ UTC+7 +# ============================================================================== +# Lưu ý tên biến: được kế thừa từ codebase OASIS gốc của Trung Quốc (UTC+8). +# Thực tế trong MiroFish, hệ thống đã chuyển sang dùng múi giờ Việt Nam (UTC+7) — +# các prompt LLM đều chỉ định "người Việt Nam / giờ Hà Nội". +# Biến này hiện chỉ dùng để tham khảo/documentation, KHÔNG được import trực tiếp +# vào các hàm tạo config (logic thực tế nằm trong prompt LLM và rule fallback). +# ============================================================================== + CHINA_TIMEZONE_CONFIG = { # Khung giờ khuya (Hầu như không có hoạt động) "dead_hours": [0, 1, 2, 3, 4, 5], @@ -48,133 +71,220 @@ CHINA_TIMEZONE_CONFIG = { } +# ============================================================================== +# DATACLASS: AgentActivityConfig — Cấu hình hành vi của một agent trong OASIS +# ============================================================================== +# Mỗi AgentActivityConfig tương ứng với 1 EntityNode đã được map thành agent. +# OASIS đọc các field này để quyết định: +# - Agent này có "thức dậy" trong round này không? (activity_level) +# - Nếu thức, nó làm gì? (posts_per_hour, comments_per_hour, stance) +# - Bài của nó được bao nhiêu agent khác nhìn thấy? (influence_weight) +# ============================================================================== + @dataclass class AgentActivityConfig: """Cấu hình hoạt động cho một Agent""" - agent_id: int - entity_uuid: str - entity_name: str - entity_type: str - - # Mức độ hoạt động (0.0-1.0) - activity_level: float = 0.5 # Hoạt động tổng thể - - # Tần suất phát ngôn (Số lần comment dự kiến mỗi giờ) + + # --- Định danh --- + agent_id: int # Phải khớp với user_id trong reddit_profiles.json / twitter_profiles.csv + entity_uuid: str # UUID trong Zep — dùng để trace ngược lại nguồn gốc + entity_name: str # Tên hiển thị (ví dụ: "Trần Văn An") + entity_type: str # Loại entity (ví dụ: "Student", "MediaOutlet") + + # --- Mức độ hoạt động tổng thể --- + # 0.0 = không bao giờ hoạt động; 1.0 = luôn hoạt động mỗi round + # OASIS dùng giá trị này nhân với activity_multiplier của giờ hiện tại + # để tính xác suất agent được chọn trong round. + activity_level: float = 0.5 + + # --- Tần suất phát ngôn --- + # Số lần trung bình agent đăng bài mới / bình luận trong 1 giờ mô phỏng posts_per_hour: float = 1.0 comments_per_hour: float = 2.0 - - # Khoảng thời gian hoạt động (Hệ 24 giờ, 0-23) + + # --- Khoảng thời gian hoạt động (hệ 24 giờ, 0–23) --- + # Agent chỉ có thể được kích hoạt trong các giờ này. + # Ví dụ: student=[8,9,10,18,19,20,21,22,23], university=[9,10,...,17] active_hours: List[int] = field(default_factory=lambda: list(range(8, 23))) - - # Tốc độ phản hồi (Độ trễ phản ứng với sự kiện nóng, đơn vị: phút mô phỏng) + + # --- Tốc độ phản hồi --- + # Độ trễ (phút mô phỏng) trước khi agent phản ứng với sự kiện nóng. + # Nhỏ = phản ứng nhanh (sinh viên: 1-15 phút) + # Lớn = phản ứng chậm (cơ quan chính phủ: 60-240 phút) response_delay_min: int = 5 response_delay_max: int = 60 - - # Khuynh hướng cảm xúc (-1.0 đến 1.0, từ tiêu cực đến tích cực) + + # --- Khuynh hướng cảm xúc --- + # -1.0 = rất tiêu cực (phản đối mạnh mẽ) + # 0.0 = trung lập + # +1.0 = rất tích cực (ủng hộ nhiệt tình) sentiment_bias: float = 0.0 - - # Lập trường (Thái độ đối với chủ đề cụ thể) - stance: str = "neutral" # supportive, opposing, neutral, observer - - # Trọng số ảnh hưởng (Xác định mức độ bài đăng được Agent khác nhìn thấy) + + # --- Lập trường với chủ đề mô phỏng --- + # "supportive": ủng hộ quyết định/sự kiện được mô phỏng + # "opposing": phản đối + # "neutral": không rõ ràng, phân tích khách quan + # "observer": chỉ theo dõi, ít tham gia tranh luận + stance: str = "neutral" + + # --- Trọng số ảnh hưởng --- + # Mức độ bài đăng của agent này xuất hiện trên "timeline" của agent khác. + # Cao = lan rộng hơn (mediaoutlet: 2.5, university: 3.0) + # Thấp = ít người thấy (student: 0.8) influence_weight: float = 1.0 -@dataclass +# ============================================================================== +# DATACLASS: TimeSimulationConfig — Cấu hình thời gian và nhịp hoạt động +# ============================================================================== +# Định nghĩa "lịch sinh hoạt" của thế giới mô phỏng: +# - Tổng độ dài mô phỏng (total_simulation_hours) +# - Mỗi round đại diện cho bao nhiêu phút thực (minutes_per_round) +# - Khung giờ nào sẽ có nhiều hay ít agent hoạt động (peak/off_peak/morning/work) +# +# OASIS đọc các multiplier để tính số agent active mỗi round: +# active_count = random(min, max) × multiplier[current_hour] +# ============================================================================== + +@dataclass class TimeSimulationConfig: - """Cấu hình thời gian mô phỏng (Dựa trên thói quen sinh hoạt của người Trung)""" - # Tổng thời gian mô phỏng (Giờ) - total_simulation_hours: int = 72 # Mặc định là chạy mô phỏng 72 tiếng (3 ngày) - - # Số phút đại diện cho mỗi vòng - Mặc định 60 phút (1 giờ), đẩy nhanh thời gian + """Cấu hình thời gian mô phỏng""" + + # Tổng số giờ mô phỏng — 72 giờ = 3 ngày thực + # (Với minutes_per_round=60, tổng số round = 72) + total_simulation_hours: int = 72 + + # Mỗi round đại diện cho bao nhiêu phút trong thế giới thực + # 60 = mỗi round là 1 giờ (khuyến nghị để cân bằng tốc độ vs độ chi tiết) minutes_per_round: int = 60 - - # Phạm vi số lượng Agent kích hoạt mỗi giờ + + # Số agent được kích hoạt mỗi giờ — LLM chọn giá trị trong [min, max] + # dựa trên quy mô simulation (số entity) agents_per_hour_min: int = 5 agents_per_hour_max: int = 20 - - # Giờ cao điểm (19-22 giờ tối, thời gian sôi động nhất) + + # Khung giờ cao điểm: 19-22 giờ — đông nhất, nhân với 1.5 peak_hours: List[int] = field(default_factory=lambda: [19, 20, 21, 22]) peak_activity_multiplier: float = 1.5 - - # Khung giờ chết (0-5 giờ, hầu như không ai on) + + # Khung giờ chết: 0-5 giờ sáng — gần như không ai online, nhân với 0.05 off_peak_hours: List[int] = field(default_factory=lambda: [0, 1, 2, 3, 4, 5]) - off_peak_activity_multiplier: float = 0.05 # Rạng sáng gần như bằng không - - # Khung giờ buổi sáng + off_peak_activity_multiplier: float = 0.05 + + # Buổi sáng: 6-8 giờ — hoạt động tăng dần, nhân với 0.4 morning_hours: List[int] = field(default_factory=lambda: [6, 7, 8]) morning_activity_multiplier: float = 0.4 - - # Khung giờ làm việc + + # Giờ làm việc: 9-18 giờ — hoạt động ổn định, nhân với 0.7 work_hours: List[int] = field(default_factory=lambda: [9, 10, 11, 12, 13, 14, 15, 16, 17, 18]) work_activity_multiplier: float = 0.7 +# ============================================================================== +# DATACLASS: EventConfig — Cấu hình "ngòi nổ" khởi động cuộc mô phỏng +# ============================================================================== +# initial_posts là các bài đăng đầu tiên được post ngay khi simulation bắt đầu. +# Chúng tạo ra "sự kiện kích hoạt" để các agent khác phản ứng. +# +# Sau khi LLM sinh initial_posts (có poster_type), hàm _assign_initial_post_agents() +# sẽ map poster_type → agent_id thực để OASIS biết agent nào post bài đó. +# ============================================================================== + @dataclass class EventConfig: - """Cấu hình sự kiện cho Simulation""" - # Các bài Post/Sự kiện khởi đầu (Bắt đầu ngay khi chạy mô phỏng) + """Cấu hình sự kiện cho Simulation gồm initial_posts, scheduled_events, hot_topics, narrative_direction""" + + # Bài đăng khởi đầu — post ngay round 1, tạo "tin nóng" để agent phản ứng + # Mỗi phần tử: {"content": "...", "poster_type": "MediaOutlet", "poster_agent_id": 12} + # poster_agent_id được gán sau bởi _assign_initial_post_agents() initial_posts: List[Dict[str, Any]] = field(default_factory=list) - - # Các sự kiện được lập lịch vào các thời điểm nhất định + + # Sự kiện lập lịch — chưa được implement, dành cho tính năng tương lai scheduled_events: List[Dict[str, Any]] = field(default_factory=list) - - # Từ khóa dành cho các chủ đề đang hot (Hot topics) + + # Từ khóa chủ đề nóng — OASIS dùng để tăng khả năng agent chú ý đến chủ đề này + # Ví dụ: ["học phí", "biểu tình", "giáo dục"] hot_topics: List[str] = field(default_factory=list) - - # Hướng dẫn dư luận / Đường lối thảo luận + + # Hướng dẫn tổng quan diễn biến dư luận — LLM viết ra để định hướng + # Ví dụ: "Thông báo tăng học phí → sinh viên phản đối → leo thang thành phong trào..." narrative_direction: str = "" +# ============================================================================== +# DATACLASS: PlatformConfig — Cấu hình riêng cho từng nền tảng mạng xã hội +# ============================================================================== +# Twitter và Reddit có thuật toán feed khác nhau → cần cấu hình riêng. +# Các giá trị này được OASIS dùng khi quyết định bài nào hiển thị trên timeline agent. +# +# recency_weight + popularity_weight + relevance_weight = 1.0 (không bắt buộc, nhưng +# nên cộng lại = 1 để tránh scale lệch) +# ============================================================================== + @dataclass class PlatformConfig: - """Cấu hình đặc thù dành riêng cho các nền tảng""" - platform: str # twitter or reddit - - # Trọng số cho các thuật toán đề xuất - recency_weight: float = 0.4 # Độ mới của bài - popularity_weight: float = 0.3 # Mức độ phổ biến truyền miệng - relevance_weight: float = 0.3 # Mức độ quan tâm / tương quan - - # Ngưỡng lan truyền virus (Cần bao nhiêu tương tác để nội dung bắt đầu phát tán mạnh) + """Cấu hình đặc thù dành riêng cho các nền tảng gồm platform, recency_weight, popularity_weight, relevance_weight, viral_threshold và echo_chamber_strength""" + platform: str # "twitter" hoặc "reddit" + + # Ba trọng số trong thuật toán đề xuất nội dung: + recency_weight: float = 0.4 # Ưu tiên bài mới (Twitter: cao hơn Reddit) + popularity_weight: float = 0.3 # Ưu tiên bài nhiều like/repost + relevance_weight: float = 0.3 # Ưu tiên bài liên quan đến sở thích agent + + # Ngưỡng lan truyền "viral": + # Twitter: 10 — dễ lan hơn (retweet lan nhanh) + # Reddit: 15 — khó lan hơn (cần nhiều upvote hơn) + # Khi bài đạt ngưỡng này, OASIS tăng xác suất nó xuất hiện trên feed của nhiều agent viral_threshold: int = 10 - - # Độ mạnh của hiệu ứng lan truyền trong nhóm chung chí hướng (buồng phản âm) + + # Hiệu ứng "buồng phản âm" (echo chamber): + # 0.0 = không có — agent thấy đa chiều + # 1.0 = tối đa — agent chỉ thấy bài cùng quan điểm + # Reddit thường cao hơn (0.6) vì cơ chế subreddit tạo bong bóng thông tin echo_chamber_strength: float = 0.5 +# ============================================================================== +# DATACLASS: SimulationParameters — Object tổng hợp toàn bộ cấu hình simulation +# ============================================================================== +# Đây là output cuối cùng của SimulationConfigGenerator.generate_config(). +# Nó được serialize thành simulation_config.json — file trung tâm mà: +# 1. Flask đọc để điền metadata vào state.json +# 2. OASIS scripts đọc để khởi tạo môi trường và chạy simulation +# +# to_json() đảm bảo Unicode (tiếng Việt) không bị escape thành \uXXXX +# ============================================================================== + @dataclass class SimulationParameters: - """完整的模拟参数配置""" - # 基础信息 + """Bộ tham số cấu hình đầy đủ cho một lượt Simulation""" + + # --- Định danh (bắt buộc khi tạo object) --- simulation_id: str project_id: str graph_id: str - simulation_requirement: str - - # Cấu hình thời gian + simulation_requirement: str # Yêu cầu gốc từ người dùng — được lưu để traceback + + # --- Các cấu hình con (có default để tạo object không cần truyền hết) --- time_config: TimeSimulationConfig = field(default_factory=TimeSimulationConfig) - - # Danh sách cấu hình Agent agent_configs: List[AgentActivityConfig] = field(default_factory=list) - - # Cấu hình Event event_config: EventConfig = field(default_factory=EventConfig) - - # Cấu hình nền tảng + + # None nếu platform tương ứng bị tắt (enable_twitter=False / enable_reddit=False) twitter_config: Optional[PlatformConfig] = None reddit_config: Optional[PlatformConfig] = None - - # Cấu hình LLM - llm_model: str = "" - llm_base_url: str = "" - - # Dữ liệu metadata khi tạo + + # --- Metadata LLM --- + llm_model: str = "" # Tên model đã dùng để gen (dùng để debug khi chất lượng kém) + llm_base_url: str = "" # Base URL API (có thể là OpenAI, Azure, local...) + + # --- Metadata tạo --- generated_at: str = field(default_factory=lambda: datetime.now().isoformat()) - generation_reasoning: str = "" # Giải thích suy luận từ LLM - + # Chuỗi reasoning nối từ tất cả các bước: "Time config: ...|Event config: ...|Agent: ..." + generation_reasoning: str = "" + def to_dict(self) -> Dict[str, Any]: - """Convert sang định dạng Dictionary""" + """Serialize toàn bộ object thành dict — dùng để truyền nội bộ.""" time_dict = asdict(self.time_config) return { "simulation_id": self.simulation_id, @@ -191,37 +301,61 @@ class SimulationParameters: "generated_at": self.generated_at, "generation_reasoning": self.generation_reasoning, } - + def to_json(self, indent: int = 2) -> str: - """Convert sang định dạng chuỗi JSON""" + """Serialize thành chuỗi JSON — ensure_ascii=False để giữ nguyên tiếng Việt.""" return json.dumps(self.to_dict(), ensure_ascii=False, indent=indent) +# ============================================================================== +# CLASS: SimulationConfigGenerator — Trình điều phối sinh config bằng LLM +# ============================================================================== +# Được gọi bởi: SimulationManager.prepare_simulation() ở Giai đoạn 3. +# Input đến từ: ZepEntityReader.filter_defined_entities() (danh sách EntityNode) +# +# Sơ đồ luồng gọi LLM trong generate_config(): +# +# generate_config() +# │ +# ├── Bước 1: _generate_time_config() ← 1 LLM call +# │ └── _parse_time_config() ← validate + parse +# │ +# ├── Bước 2: _generate_event_config() ← 1 LLM call +# │ └── _parse_event_config() ← parse đơn giản +# │ +# ├── Bước 3..N: _generate_agent_configs_batch() ← 1 LLM call × N batch +# │ └── _generate_agent_config_by_rule() ← fallback nếu LLM fail/thiếu +# │ +# ├── _assign_initial_post_agents() ← không cần LLM, map bằng alias +# │ +# └── Bước cuối: PlatformConfig hardcoded ← không cần LLM +# +# Mỗi LLM call đi qua _call_llm_with_retry() với retry 3 lần, temperature giảm dần. +# ============================================================================== + class SimulationConfigGenerator: """ Trình tạo cấu hình Simulation tự động bằng LLM - + Sử dụng LLM phân tích yêu cầu mô phỏng, nội dung tài liệu, Entity từ đồ thị, - Tự động xây dựng các thông số cấu trúc tối ưu cho đợt Simulation - - Áp dụng chiến lược tạo từng bước: + tự động xây dựng các thông số cấu trúc tối ưu cho đợt Simulation. + + Chiến lược chia từng bước (step-by-step generation): 1. Tạo cấu hình thời gian và cấu hình Event (Nhẹ, chạy nhanh) - 2. Phân nhỏ đợt tạo cấu hình cho Agent (Khoảng 10-20 agent mỗi đợt) - 3. Tạo cấu hình nền tảng + 2. Phân nhỏ đợt tạo cấu hình cho Agent (15 agent mỗi đợt) + 3. Tạo cấu hình nền tảng (hardcoded, không cần LLM) """ - - # Số lượng ký tự tối đa của bộ context - MAX_CONTEXT_LENGTH = 50000 - # Số lượng Agent để gen cho một lần - AGENTS_PER_BATCH = 15 - - # Số lượng ký tự giới hạn ở các bước để cắt chuỗi (Ký tự đoạn) - TIME_CONFIG_CONTEXT_LENGTH = 10000 # Cấu hình thời gian - EVENT_CONFIG_CONTEXT_LENGTH = 8000 # Cấu hình sự kiện - ENTITY_SUMMARY_LENGTH = 300 # Tóm tắt các thực thể - AGENT_SUMMARY_LENGTH = 300 # Tóm tắt cấu hình Agent - ENTITIES_PER_TYPE_DISPLAY = 20 # Lượng thực thể cho mổi loại để hiển thị - + + MAX_CONTEXT_LENGTH = 50000 # Giới hạn tổng ngữ cảnh gửi cho LLM (ký tự) + AGENTS_PER_BATCH = 15 # Số agent xử lý mỗi lần gọi LLM (tránh token limit) + + # Độ dài context cho từng bước — cắt ngắn để không waste token + TIME_CONFIG_CONTEXT_LENGTH = 10000 # Bước 1: chỉ cần biết chủ đề + entity types + EVENT_CONFIG_CONTEXT_LENGTH = 8000 # Bước 2: cần biết thêm về các entity điển hình + ENTITY_SUMMARY_LENGTH = 300 # Tóm tắt mỗi entity trong context + AGENT_SUMMARY_LENGTH = 300 # Tóm tắt entity trong prompt sinh agent config + ENTITIES_PER_TYPE_DISPLAY = 20 # Tối đa bao nhiêu entity/loại trong context + def __init__( self, api_key: Optional[str] = None, @@ -231,16 +365,17 @@ class SimulationConfigGenerator: self.api_key = api_key or Config.LLM_API_KEY self.base_url = base_url or Config.LLM_BASE_URL self.model_name = model_name or Config.LLM_MODEL_NAME - + if not self.api_key: raise ValueError("LLM_API_KEY has not been configured") - + self.client = OpenAI( api_key=self.api_key, base_url=self.base_url ) + # Metadata runtime được inject vào mỗi LLM call để theo dõi cost/usage self._runtime_metadata: Dict[str, Any] = {} - + def generate_config( self, simulation_id: str, @@ -254,21 +389,30 @@ class SimulationConfigGenerator: progress_callback: Optional[Callable[[int, int, str], None]] = None, ) -> SimulationParameters: """ - Tạo cấu hình Simulation thông minh tự động hoàn chỉnh (Bằng tư duy chia từng bước) - + Pipeline sinh cấu hình simulation hoàn chỉnh theo từng bước. + + Tổng quan luồng: + ┌───────────────────────────────────────────────────────────────┐ + │ Bước 1: _generate_time_config() → TimeSimulationConfig │ + │ Bước 2: _generate_event_config() → EventConfig + initial_posts│ + │ Bước 3..N: _generate_agent_configs_batch() × ceil(N/15) │ + │ _assign_initial_post_agents() → gán poster_agent_id │ + │ Bước cuối: PlatformConfig hardcoded │ + └───────────────────────────────────────────────────────────────┘ + Args: - simulation_id: Nhận dạng quy trình chạy Simulation - project_id: Mã định danh dự án - graph_id: Đồ thị đồ thị - simulation_requirement: Yêu cầu của quá trình mô phỏng - document_text: Nội dung file tài liệu nguồn - entities: Danh sách các thực thể đã được lọc - enable_twitter: Cờ hiệu để bật Twitter - enable_reddit: Cờ hiệu để bật Reddit - progress_callback: Hàm callback lấy trạng thái tiến trình hiện tại (current_step, total_steps, message) - + simulation_id: ID của simulation (dùng để điền vào SimulationParameters) + project_id: ID project + graph_id: ID Zep Graph (dùng để điền metadata, không query thêm ở đây) + simulation_requirement: Yêu cầu người dùng (truyền thẳng vào LLM prompt) + document_text: Văn bản tài liệu gốc (làm ngữ cảnh nền tảng cho LLM) + entities: Danh sách EntityNode từ ZepEntityReader (đã lọc + enrich) + enable_twitter: True → sinh twitter_config + enable_reddit: True → sinh reddit_config + progress_callback: Hàm nhận (current_step, total_steps, message: str) + Returns: - SimulationParameters: Bộ tổng cấu hình thông số đầy đủ + SimulationParameters đầy đủ — gọi .to_json() để ghi ra file. """ logger.info(f"Start generating simulation configuration: simulation_id={simulation_id}, entity_count={len(entities)}") self._runtime_metadata = { @@ -277,52 +421,57 @@ class SimulationConfigGenerator: "component": "simulation_config_generator", "phase": "prepare_simulation_config", } - - # Tính toán tổng số bước + + # Tính tổng số bước để progress_callback hiển thị đúng phần trăm + # total_steps = 1 (time) + 1 (event) + N_batch (agents) + 1 (platform) = 3 + N_batch num_batches = math.ceil(len(entities) / self.AGENTS_PER_BATCH) - total_steps = 3 + num_batches # Cấu hình tgian + Sự kiện + Nx(Agent Batch) + Nền tảng + total_steps = 3 + num_batches current_step = 0 - + def report_progress(step: int, message: str): nonlocal current_step current_step = step if progress_callback: progress_callback(step, total_steps, message) logger.info(f"[{step}/{total_steps}] {message}") - - # 1. Xây dựng thông tin ngữ cảnh cơ bản + + # Xây dựng context chung (tối đa 50,000 ký tự) dùng cho tất cả các bước + # Bao gồm: simulation_requirement + entity summary + document_text (phần còn lại) context = self._build_context( simulation_requirement=simulation_requirement, document_text=document_text, entities=entities ) - + + # Mảng tích lũy reasoning từ mỗi bước — nối lại cuối để lưu vào SimulationParameters reasoning_parts = [] - - # ========== Bước 1: Tạo bộ cấu hình về Thời Gian ========== + + # ========== Bước 1: Cấu hình Thời gian ========== report_progress(1, "Generating time configuration...") num_entities = len(entities) time_config_result = self._generate_time_config(context, num_entities) time_config = self._parse_time_config(time_config_result, num_entities) reasoning_parts.append(f"Time config reasoning: {time_config_result.get('reasoning', 'Success')}") - # ========== Bước 2: Tạo cấu hình Event ========== + + # ========== Bước 2: Cấu hình Sự kiện ========== report_progress(2, "Generating event configuration and hot topics...") event_config_result = self._generate_event_config(context, simulation_requirement, entities) event_config = self._parse_event_config(event_config_result) reasoning_parts.append(f"Event config reasoning: {event_config_result.get('reasoning', 'Success')}") - - # ========== Bước 3-N: Chia thành các đợt để lấy cấu hình Agent ========== + + # ========== Bước 3..N: Cấu hình Agent (chia batch) ========== + # Ví dụ: 47 entity → batch1(0-14) + batch2(15-29) + batch3(30-44) + batch4(45-46) all_agent_configs = [] for batch_idx in range(num_batches): start_idx = batch_idx * self.AGENTS_PER_BATCH end_idx = min(start_idx + self.AGENTS_PER_BATCH, len(entities)) batch_entities = entities[start_idx:end_idx] - + report_progress( 3 + batch_idx, f"Generating agent configuration ({start_idx + 1}-{end_idx}/{len(entities)})..." ) - + batch_configs = self._generate_agent_configs_batch( context=context, entities=batch_entities, @@ -330,41 +479,43 @@ class SimulationConfigGenerator: simulation_requirement=simulation_requirement ) all_agent_configs.extend(batch_configs) - + reasoning_parts.append(f"Agent config reasoning: Successfully generated {len(all_agent_configs)} agents") - - # ========== Tiến hành gán người (Agent) để đăng các bài Initial Post ========== + + # ========== Gán agent cho initial posts ========== + # Sau khi có đủ all_agent_configs, map poster_type → agent_id thực logger.info("Assigning poster agents for initial posts...") event_config = self._assign_initial_post_agents(event_config, all_agent_configs) assigned_count = len([p for p in event_config.initial_posts if p.get("poster_agent_id") is not None]) reasoning_parts.append(f"Initial post assignment: {assigned_count} posts have been assigned to publishers") - - # ========== Bước cuối: Thiết lập nền tảng ========== + + # ========== Bước cuối: Platform Config (hardcoded) ========== + # Không cần LLM — các giá trị này là hằng số đã được tuning thực nghiệm report_progress(total_steps, "Generating platform configuration...") twitter_config = None reddit_config = None - + if enable_twitter: twitter_config = PlatformConfig( platform="twitter", - recency_weight=0.4, + recency_weight=0.4, # Twitter ưu tiên bài mới hơn Reddit popularity_weight=0.3, relevance_weight=0.3, - viral_threshold=10, + viral_threshold=10, # Dễ lan hơn Reddit echo_chamber_strength=0.5 ) - + if enable_reddit: reddit_config = PlatformConfig( platform="reddit", - recency_weight=0.3, - popularity_weight=0.4, + recency_weight=0.3, # Reddit cân bằng hơn giữa mới vs phổ biến + popularity_weight=0.4, # Reddit ưu tiên upvote hơn thời gian relevance_weight=0.3, - viral_threshold=15, - echo_chamber_strength=0.6 + viral_threshold=15, # Ngưỡng cao hơn → khó "bùng" hơn Twitter + echo_chamber_strength=0.6 # Reddit có subreddit → buồng phản âm mạnh hơn ) - - # Xây dựng các tham số cuối cùng kết thúc quy trình + + # Gộp tất cả vào SimulationParameters params = SimulationParameters( simulation_id=simulation_id, project_id=project_id, @@ -379,54 +530,69 @@ class SimulationConfigGenerator: llm_base_url=self.base_url, generation_reasoning=" | ".join(reasoning_parts) ) - + logger.info(f"Simulation configuration generation complete: {len(params.agent_configs)} agent configs created") - + return params - + def _build_context( self, simulation_requirement: str, document_text: str, entities: List[EntityNode] ) -> str: - """Thực hiện xây dựng nội dung Prompt Ngữ cảnh cho LLM, với độ dài có thể bị giới hạn""" - - # Tóm tắt lại Thực thể + """ + Tổng hợp ngữ cảnh cho LLM — tối đa MAX_CONTEXT_LENGTH ký tự. + + Ưu tiên (thứ tự giảm dần): + 1. simulation_requirement — giữ nguyên 100%, không cắt + 2. entity_summary — tóm tắt nhóm theo loại, tối đa ENTITIES_PER_TYPE_DISPLAY/loại + 3. document_text — phần còn lại sau khi đã dùng cho 1+2 + + Document_text bị cắt cuối nếu tổng vượt quá 50,000 ký tự. + """ entity_summary = self._summarize_entities(entities) - - # Xây dựng nội dung + context_parts = [ f"## Simulation Requirements\n{simulation_requirement}", f"\n## Entity Information ({len(entities)} entities)\n{entity_summary}", ] - + current_length = sum(len(p) for p in context_parts) - remaining_length = self.MAX_CONTEXT_LENGTH - current_length - 500 # Dành sẵn 500 ký tự trống - + # Dành 500 ký tự buffer để tránh off-by-one khi LLM đếm token + remaining_length = self.MAX_CONTEXT_LENGTH - current_length - 500 + if remaining_length > 0 and document_text: doc_text = document_text[:remaining_length] if len(document_text) > remaining_length: - doc_text += "\n...(Document Truncated)" + doc_text += "\n...(Document Truncated)" # note: việc cắt bớt đi có bị ảnh hưởng gì nghiêm trọng không? nếu để full thì có tốn context không? context_parts.append(f"\n## Original Document Content\n{doc_text}") - + return "\n".join(context_parts) - + def _summarize_entities(self, entities: List[EntityNode]) -> str: - """Tạo chuỗi văn bản Tóm tắt cho các Thực thể""" + """ + Tạo chuỗi tóm tắt entity theo nhóm loại — dùng trong _build_context(). + + Format output: + ### Student (5 entity) + - Trần Văn An: Sinh viên năm 3 ngành CNTT... + - Nguyễn Thị B: ... + ### University (2 entity) + - Đại học X: ... + """ lines = [] - - # Phân nhóm bằng Loại + + # Nhóm entity theo loại để LLM dễ nắm phân phối nhân vật by_type: Dict[str, List[EntityNode]] = {} for e in entities: t = e.get_entity_type() or "Unknown" if t not in by_type: by_type[t] = [] by_type[t].append(e) - + for entity_type, type_entities in by_type.items(): lines.append(f"\n### {entity_type} ({len(type_entities)} entity)") - # Số lượng đã được thiết lập mặc định và Giới hạn chiều dài của bảng tóm tắt display_count = self.ENTITIES_PER_TYPE_DISPLAY summary_len = self.ENTITY_SUMMARY_LENGTH for e in type_entities[:display_count]: @@ -434,16 +600,40 @@ class SimulationConfigGenerator: lines.append(f"- {e.name}: {summary_preview}") if len(type_entities) > display_count: lines.append(f" ... and {len(type_entities) - display_count} more entities") - + return "\n".join(lines) - + + ''' + ### Student (5 entity) + - Trần Văn An: Sinh viên năm 3 ngành CNTT tại Đại học X... + - Nguyễn Thị B: Sinh viên năm 2, hoạt động phong trào... + + ### University (1 entity) + - Đại học X: Trường đại học công lập lớn tại Hà Nội... + + ### MediaOutlet (2 entity) + - VnExpress: Báo điện tử lớn nhất Việt Nam... + ''' + def _call_llm_with_retry(self, prompt: str, system_prompt: str) -> Dict[str, Any]: - """Tích hợp cơ chế retry mỗi lúc gọi Request LLM bị lỗi và Logic sửa lỗi JSON string""" + """ + Gọi LLM và retry tối đa 3 lần với temperature giảm dần. + + Chiến lược retry: + Attempt 1: temperature=0.7 (sáng tạo, đa dạng hơn) + Attempt 2: temperature=0.6 + sleep 2s (nếu attempt 1 fail) + Attempt 3: temperature=0.5 + sleep 4s (nếu attempt 2 fail) + + Temperature giảm dần → output ổn định hơn, ít hallucination → dễ parse JSON hơn. + Nếu finish_reason="length" (bị cắt do token limit) → _fix_truncated_json() trước khi parse. + Nếu json.loads() lỗi → _try_fix_config_json() để cố sửa. + Nếu cả 3 lần đều fail → raise exception để caller dùng hardcode fallback. + """ import re - + max_attempts = 3 last_error = None - + for attempt in range(max_attempts): try: response = create_tracked_chat_completion( @@ -454,144 +644,138 @@ class SimulationConfigGenerator: {"role": "user", "content": prompt} ], response_format={"type": "json_object"}, - temperature=0.7 - (attempt * 0.1), # Giảm temperature cho mỗi lần retry + temperature=0.7 - (attempt * 0.1), # 0.7 → 0.6 → 0.5 metadata=self._runtime_metadata, ) - + content = response.choices[0].message.content finish_reason = response.choices[0].finish_reason - - # Kiểm tra nội dung trã về xem có phải bị chặn vì thiếu token (Length vượt qua max) hay không + + # Nếu bị cắt giữa chừng do max_tokens → thêm ngoặc đóng trước khi parse if finish_reason == 'length': logger.warning(f"LLM output was truncated (attempt {attempt+1})") content = self._fix_truncated_json(content) - - # Phân tích nội dung JSON + try: return json.loads(content) except json.JSONDecodeError as e: logger.warning(f"Failed to parse JSON (attempt {attempt+1}): {str(e)[:80]}") - - # Tiến hành sửa chữa nội dung JSON nếu bị lỗi + # Thử sửa JSON trước khi bỏ cuộc fixed = self._try_fix_config_json(content) if fixed: return fixed - last_error = e - + except Exception as e: logger.warning(f"Failed to call LLM (attempt {attempt+1}): {str(e)[:80]}") last_error = e import time - time.sleep(2 * (attempt + 1)) - + time.sleep(2 * (attempt + 1)) # Sleep 2s → 4s → (không có attempt 3+) + raise last_error or Exception("LLM connection completely failed") - + def _fix_truncated_json(self, content: str) -> str: - """Đóng dấu ngoặc JSON một cách an toàn cho các string bị cắt ngang""" + """ + Đóng các ngoặc JSON bị bỏ ngỏ khi LLM bị cắt giữa chừng. + + Thuật toán: + 1. Đếm số { chưa có } đóng tương ứng → thêm } vào cuối + 2. Đếm số [ chưa có ] đóng tương ứng → thêm ] vào cuối + 3. Nếu ký tự cuối là dở dang (không phải ",}]) → thêm " để đóng string + + Ví dụ input bị cắt: + '{"agent_configs": [{"agent_id": 0, "stance": "oppos' + Sau fix: + '{"agent_configs": [{"agent_id": 0, "stance": "oppos"}]}' + """ content = content.strip() - - # Đếm các dấu ngoặc mở bị bỏ sót chưa đóng + open_braces = content.count('{') - content.count('}') open_brackets = content.count('[') - content.count(']') - - # Đảm bảo các thuộc tính string đã được bọc đủ dấu ngoặc kép + + # Đóng string bị bỏ ngỏ trước khi đóng object/array if content and content[-1] not in '",}]': content += '"' - - # Thêm ngoặc đóng cho toàn bộ + content += ']' * open_brackets content += '}' * open_braces - + return content - + def _try_fix_config_json(self, content: str) -> Optional[Dict[str, Any]]: - """Cố gắng khôi phục, chắp ghép lại file cấu trúc config JSON""" + """ + Cố sửa JSON bị lỗi format bằng chuỗi bước: + + Bước 1: Gọi _fix_truncated_json() (đóng bracket thiếu) + Bước 2: Regex tìm khối {...} lớn nhất trong chuỗi (loại bỏ text thừa trước/sau) + Bước 3: Clean newline trong string values (LLM hay nhét \n trong string JSON) + Bước 4: Thử json.loads() — nếu OK trả về luôn + Bước 5: Xóa control characters (0x00-0x1F, 0x7F-0x9F) gây lỗi parse + Bước 6: Chuẩn hóa whitespace thừa + Bước 7: Thử json.loads() lần nữa — nếu vẫn fail trả về None + + Trả về None nếu không thể sửa → caller sẽ retry hoặc dùng fallback. + """ import re - - # Điền những dấu ngoặc vào chuỗi bị cắt + content = self._fix_truncated_json(content) - - # Regex ra đúng phần ruột nội dung JSON + json_match = re.search(r'\{[\s\S]*\}', content) if json_match: json_str = json_match.group() - - # Loại bỏ các đoạn tab, ngắt line cho string + + # Clean newline và whitespace thừa bên trong string values def fix_string(match): s = match.group(0) s = s.replace('\n', ' ').replace('\r', ' ') s = re.sub(r'\s+', ' ', s) return s - + json_str = re.sub(r'"[^"\\]*(?:\\.[^"\\]*)*"', fix_string, json_str) - + try: return json.loads(json_str) except: - # Tìm và xóa các control character + # Xóa control characters và thử lần nữa json_str = re.sub(r'[\x00-\x1f\x7f-\x9f]', ' ', json_str) json_str = re.sub(r'\s+', ' ', json_str) try: return json.loads(json_str) except: pass - + return None - + + # -------------------------------------------------------------------------- + # BƯỚC 1: Sinh cấu hình Thời gian + # -------------------------------------------------------------------------- + def _generate_time_config(self, context: str, num_entities: int) -> Dict[str, Any]: - """Tạo cấu hình thời gian (Time config) cho các tiến trình""" - # Áp dụng nội dung ngữ cảnh đã được giới hạn chiều dài + """ + Gọi LLM để sinh cấu hình thời gian mô phỏng. + + LLM nhận context cắt còn TIME_CONFIG_CONTEXT_LENGTH (10,000 ký tự) và trả về: + { + "total_simulation_hours": 72, + "minutes_per_round": 60, + "agents_per_hour_min": 5, + "agents_per_hour_max": 30, + "peak_hours": [19, 20, 21, 22], + "off_peak_hours": [0, 1, 2, 3, 4, 5], + "morning_hours": [6, 7, 8], + "work_hours": [9, 10, ..., 18], + "reasoning": "Giải thích tại sao chọn các thông số này" + } + + Ràng buộc gửi cho LLM: + - Người dùng là người Việt Nam → giờ Hà Nội (UTC+7) + - agents_per_hour tối đa = 90% số entity (max_agents_allowed = num_entities × 0.9) + + Fallback (LLM fail): _get_default_time_config() — hardcoded defaults. + """ context_truncated = context[:self.TIME_CONFIG_CONTEXT_LENGTH] - - # Cắt lấy số lượng Tối đa số lượng (Chiếm 80% từ số lượng lượng Agent thực thể) max_agents_allowed = max(1, int(num_entities * 0.9)) -# prompt = f"""Based on the following simulation requirements, generate a time simulation configuration. - -# {context_truncated} - -# ## Task -# Please generate a time configuration JSON. - -# ### General Principles (For reference only; adjust flexibly based on specific events and participant groups): -# - The user group consists of Vietnamese people; must comply with Hanoi Time (CST) daily routines. -# - 0:00–5:00 AM: Almost no activity (Activity Coefficient: 0.05). -# - 6:00–8:00 AM: Gradual increase in activity (Activity Coefficient: 0.4). -# - 9:00 AM–6:00 PM (Work hours): Moderate activity (Activity Coefficient: 0.7). -# - 7:00 PM–10:00 PM: Peak period (Activity Coefficient: 1.5). -# - After 11:00 PM: Activity declines (Activity Coefficient: 0.5). -# - General Pattern: Low activity in the early morning, gradual increase in the morning, moderate during work hours, and peak in the evening. -# - **Important:**: The example values below are for reference only. You need to adjust specific periods based on the nature of the event and characteristics of the participant group. -# - e.g., The peak for students might be 9:00 PM–11:00 PM; Media groups remain active all day; Official organizations only during work hours. -# - e.g., Breaking news may lead to discussions late at night; off_peak_hours can be shortened accordingly. - -# ### Return JSON Format (Do not use Markdown) - -# Example: -# {{ -# "total_simulation_hours": 72, -# "minutes_per_round": 60, -# "agents_per_hour_min": 5, -# "agents_per_hour_max": 50, -# "peak_hours": [19, 20, 21, 22], -# "off_peak_hours": [0, 1, 2, 3, 4, 5], -# "morning_hours": [6, 7, 8], -# "work_hours": [9, 10, 11, 12, 13, 14, 15, 16, 17, 18], -# "reasoning": "Time configuration explanation for this specific event." -# }} - -# Field Descriptions: -# - total_simulation_hours (int): Total simulation duration, 24–168 hours. Short for breaking news, long for sustained topics. -# - minutes_per_round (int): Duration per round, 30–120 minutes, suggested 60 minutes. -# - agents_per_hour_min (int): Minimum activated agents per hour (Range: 1-{max_agents_allowed}). -# - agents_per_hour_max (int): Maximum activated agents per hour (Range: 1-{max_agents_allowed}). -# - peak_hours (int array): Peak hours, adjusted based on the participant group. -# - off_peak_hours (int array): Off-peak hours, usually late night/early morning. -# - morning_hours (int array): Morning hours. -# - work_hours (int array): Working hours. -# - reasoning (string): Brief explanation of why this configuration was chosen.""" - prompt = f"""Dựa trên các yêu cầu mô phỏng dưới đây, hãy tạo cấu hình mô phỏng thời gian. {context_truncated} @@ -615,7 +799,7 @@ Hãy tạo JSON cấu hình thời gian. Ví dụ Format như sau: {{ - "total_simulation_hours": 72, + "total_simulation_hours": 72, "minutes_per_round": 60, "agents_per_hour_min": 5, "agents_per_hour_max": 50, @@ -636,22 +820,27 @@ Mô tả các trường: - morning_hours (mảng int): Khung giờ buổi sáng. - work_hours (mảng int): Khung giờ làm việc. - reasoning (string): Giải thích ngắn gọn lý do tại sao cấu hình như vậy.""" - - # system_prompt = "You are a social media simulation expert. Return in pure JSON format; time configurations must comply with Vietnamese daily routines." system_prompt = "Bạn là chuyên gia mô phỏng mạng xã hội. Trả về định dạng JSON thuần túy; cấu hình thời gian cần phù hợp với thói quen sinh hoạt của người Việt Nam." - + try: return self._call_llm_with_retry(prompt, system_prompt) except Exception as e: logger.warning(f"Failed to generate Time Config through LLM {e}. Returning the basic default rules...") return self._get_default_time_config(num_entities) - + def _get_default_time_config(self, num_entities: int) -> Dict[str, Any]: - """Tạo sẵn file chuẩn nếu bị đơ để trả ra theo múi giờ chuẩn sinh hoạt China""" + """ + Fallback hardcoded khi LLM fail hoàn toàn. + + agents_per_hour được tính theo công thức: + min = num_entities // 15 (ví dụ: 47 entity → min=3) + max = num_entities // 5 (ví dụ: 47 entity → max=9) + Đảm bảo không vượt quá tổng số agent thực tế. + """ return { "total_simulation_hours": 72, - "minutes_per_round": 60, # 1 Hour / Vòng -> Rút ngắn Time + "minutes_per_round": 60, "agents_per_hour_min": max(1, num_entities // 15), "agents_per_hour_max": max(5, num_entities // 5), "peak_hours": [19, 20, 21, 22], @@ -660,56 +849,87 @@ Mô tả các trường: "work_hours": [9, 10, 11, 12, 13, 14, 15, 16, 17, 18], "reasoning": "Defaults to Vietnamese users' daily routines and working hours (1 hour/round)" } - + def _parse_time_config(self, result: Dict[str, Any], num_entities: int) -> TimeSimulationConfig: - """Phân tích nội dung được định hình của JSON qua hàm parse kiểm tra, Xác nhận nếu lượng agents_per_hour vượt ngưỡng giới hạn """ - # Lấy giá trị chưa chỉnh sửa + """ + Parse dict từ LLM thành TimeSimulationConfig, kèm validation. + + Validation quan trọng: đảm bảo agents_per_hour không vượt tổng số agent thực tế. + Nếu LLM trả về agents_per_hour_max=100 nhưng chỉ có 47 entity, + OASIS sẽ cố chọn 100 agent nhưng không đủ → lỗi. + + Logic điều chỉnh: + agents_per_hour_min > num_entities → reset về num_entities // 10 + agents_per_hour_max > num_entities → reset về num_entities // 2 + min >= max → reset min về max // 2 + """ agents_per_hour_min = result.get("agents_per_hour_min", max(1, num_entities // 15)) agents_per_hour_max = result.get("agents_per_hour_max", max(5, num_entities // 5)) - - # Tiến hành kiểm tra xác minh: Đảm bảo độ lớn không lớn hơn con số Total Agent + if agents_per_hour_min > num_entities: logger.warning(f"agents_per_hour_min ({agents_per_hour_min}) exceeds total number of Agents ({num_entities}), corrected.") agents_per_hour_min = max(1, num_entities // 10) - + if agents_per_hour_max > num_entities: logger.warning(f"agents_per_hour_max ({agents_per_hour_max}) exceeds total number of Agents ({num_entities}), corrected.") agents_per_hour_max = max(agents_per_hour_min + 1, num_entities // 2) - - # Đảm bảo min luôn luôn nhỏ hơn max + if agents_per_hour_min >= agents_per_hour_max: agents_per_hour_min = max(1, agents_per_hour_max // 2) logger.warning(f"agents_per_hour_min >= max, modified to {agents_per_hour_min}") - + return TimeSimulationConfig( total_simulation_hours=result.get("total_simulation_hours", 72), - minutes_per_round=result.get("minutes_per_round", 60), # Mặc định mỗi vòng = 1 giờ + minutes_per_round=result.get("minutes_per_round", 60), agents_per_hour_min=agents_per_hour_min, agents_per_hour_max=agents_per_hour_max, peak_hours=result.get("peak_hours", [19, 20, 21, 22]), off_peak_hours=result.get("off_peak_hours", [0, 1, 2, 3, 4, 5]), - off_peak_activity_multiplier=0.05, # Gần như 0 mạng sáng rạng sáng + off_peak_activity_multiplier=0.05, morning_hours=result.get("morning_hours", [6, 7, 8]), morning_activity_multiplier=0.4, work_hours=result.get("work_hours", list(range(9, 19))), work_activity_multiplier=0.7, peak_activity_multiplier=1.5 ) - + + # -------------------------------------------------------------------------- + # BƯỚC 2: Sinh cấu hình Sự kiện + # -------------------------------------------------------------------------- + def _generate_event_config( - self, - context: str, + self, + context: str, simulation_requirement: str, entities: List[EntityNode] ) -> Dict[str, Any]: - """Tạo ra cho các thông số Event config""" - - # Tự liệt kê các Loại có thể xuất hiện để LLM tham khảo + """ + Gọi LLM để sinh các bài đăng khởi động và hướng phát triển dư luận. + + Ràng buộc quan trọng gửi cho LLM: poster_type phải thuộc danh sách + entity types thực sự có trong simulation. LLM được cung cấp danh sách + type_examples (ví dụ điển hình của mỗi loại) để chọn đúng. + + Ví dụ output: + { + "hot_topics": ["học phí", "biểu tình"], + "narrative_direction": "Thông báo tăng học phí → làn sóng phản đối...", + "initial_posts": [ + {"content": "CHÍNH THỨC: Đại học X tăng học phí...", "poster_type": "University"}, + {"content": "Không thể chấp nhận được! #HọcPhíTăng", "poster_type": "Student"} + ] + } + + Sau bước này, _assign_initial_post_agents() sẽ gán poster_agent_id thực. + Fallback khi LLM fail: hot_topics=[], initial_posts=[] — simulation vẫn chạy + nhưng không có "ngòi nổ" → agent tự tạo bài đăng theo activity_level. + """ + # Thu thập danh sách loại thực thể có thực để LLM chọn poster_type đúng entity_types_available = list(set( e.get_entity_type() or "Unknown" for e in entities )) - - # Ghi các Thực thể điển hình của mổi loại + + # Ví dụ điển hình mỗi loại (tối đa 3) — giúp LLM biết tên thật của entity type_examples = {} for e in entities: etype = e.get_entity_type() or "Unknown" @@ -717,46 +937,14 @@ Mô tả các trường: type_examples[etype] = [] if len(type_examples[etype]) < 3: type_examples[etype].append(e.name) - + type_info = "\n".join([ - f"- {t}: {', '.join(examples)}" + f"- {t}: {', '.join(examples)}" for t, examples in type_examples.items() ]) - - # Có chặn để lấy chuỗi theo cấu hình chiều dài giới hạn + context_truncated = context[:self.EVENT_CONFIG_CONTEXT_LENGTH] -# prompt = f"""Based on the following simulation requirements, generate an event configuration. - -# Simulation Requirements: {simulation_requirement} - -# {context_truncated} - -# ## Available Entity Types and Examples -# {type_info} - -# ## Task -# Please generate an event configuration JSON: -# - Extract key hot topic keywords. -# - Describe the direction of public opinion development. -# - Design initial post content; **each post must specify a poster_type (publisher type)**. - -# **IMPORTANT**: The poster_type must be selected from the "Available Entity Types" above so that initial posts can be assigned to the appropriate Agent for publishing. -# For example: Official statements should be posted by Official/University types, news by MediaOutlet, and student perspectives by Student. - -# Return in JSON format (no markdown): -# {{ -# "hot_topics": ["Keyword1", "Keyword2", ...], -# "narrative_direction": "", -# "initial_posts": [ -# {{"content": "Post Content...", "poster_type": "Entity type (must be selected from available types)"}}, -# ... -# ], -# "reasoning": "" -# }}""" - -# system_prompt = "You are a public opinion analysis expert. Return in pure JSON format. Ensure that poster_type exactly matches the available entity types." - prompt = f"""Dựa trên các yêu cầu mô phỏng sau đây, hãy tạo cấu hình sự kiện. Yêu cầu mô phỏng: {simulation_requirement} @@ -787,7 +975,7 @@ Trả về định dạng JSON (không sử dụng markdown): }}""" system_prompt = "Bạn là chuyên gia phân tích dư luận. Trả về định dạng JSON thuần túy. Lưu ý rằng poster_type phải khớp chính xác với các loại thực thể khả dụng." - + try: return self._call_llm_with_retry(prompt, system_prompt) except Exception as e: @@ -798,38 +986,56 @@ Trả về định dạng JSON (không sử dụng markdown): "initial_posts": [], "reasoning": "Sử dụng Config mặc định do LLM lỗi" } - + def _parse_event_config(self, result: Dict[str, Any]) -> EventConfig: - """Parse lấy các Thuộc Tính cấu hình Event""" + """Parse dict từ LLM thành EventConfig. Đơn giản — không có validation phức tạp.""" return EventConfig( initial_posts=result.get("initial_posts", []), - scheduled_events=[], + scheduled_events=[], # Chưa implement — dành cho tính năng tương lai hot_topics=result.get("hot_topics", []), narrative_direction=result.get("narrative_direction", "") ) - + def _assign_initial_post_agents( self, event_config: EventConfig, agent_configs: List[AgentActivityConfig] ) -> EventConfig: """ - Khớp quyền Agent với loại Poster_type cho các Bài Post đầu - - So sánh cho phù hợp của mỗi post để phân bố Agent id tối ưu nhất + Gán agent_id thực cho mỗi initial_post dựa trên poster_type. + + 3 tầng matching (ưu tiên từ cao đến thấp): + ┌─────────────────────────────────────────────────────────────────┐ + │ Tầng 1: Direct match │ + │ poster_type.lower() khớp chính xác key trong agents_by_type │ + │ Ví dụ: "student" → agents_by_type["student"][0] │ + ├─────────────────────────────────────────────────────────────────┤ + │ Tầng 2: Alias match │ + │ poster_type nằm trong danh sách alias của một key │ + │ Ví dụ: "media" → alias của "mediaoutlet" → dùng mediaoutlet │ + │ Cho phép LLM dùng tên biến thể mà vẫn map được đúng │ + ├─────────────────────────────────────────────────────────────────┤ + │ Tầng 3: Influence fallback │ + │ Không tìm được → lấy agent có influence_weight cao nhất │ + │ (Thường là mediaoutlet hoặc university) │ + └─────────────────────────────────────────────────────────────────┘ + + Lưu ý used_indices: mỗi loại có counter riêng để tránh dùng cùng 1 agent + nhiều lần khi có nhiều post cùng poster_type. + Ví dụ: 3 post "Student" → agent 0, 1, 2 (thay vì 0, 0, 0). """ if not event_config.initial_posts: return event_config - - # Build hệ thống agent index bằng kiểu loại + + # Index agents theo loại để tra cứu nhanh agents_by_type: Dict[str, List[AgentActivityConfig]] = {} for agent in agent_configs: etype = agent.entity_type.lower() if etype not in agents_by_type: agents_by_type[etype] = [] agents_by_type[etype].append(agent) - - # Bảng Alias ánh xạ tương đương (Cho phép LLM sử dụng nhiều quy ước format khác nhau) + + # Bảng alias — LLM đôi khi dùng tên khác nhau cho cùng một loại type_aliases = { "official": ["official", "university", "governmentagency", "government"], "university": ["university", "official"], @@ -840,26 +1046,24 @@ Trả về định dạng JSON (không sử dụng markdown): "organization": ["organization", "ngo", "company", "group"], "person": ["person", "student", "alumni"], } - - # Ghi chú từng loại agent đã dùng index nào, tránh dùng lại cùng 1 agent lặp đi lặp lại + + # Dùng round-robin trong mỗi loại (modulo len) để phân bổ đều used_indices: Dict[str, int] = {} - + updated_posts = [] for post in event_config.initial_posts: poster_type = post.get("poster_type", "").lower() content = post.get("content", "") - - # Khớp tìm agent phù hợp matched_agent_id = None - - # 1. Trùng khớp trực tiếp lấy luôn + + # Tầng 1: Direct match if poster_type in agents_by_type: agents = agents_by_type[poster_type] idx = used_indices.get(poster_type, 0) % len(agents) matched_agent_id = agents[idx].agent_id used_indices[poster_type] = idx + 1 else: - # 2. Sử dụng bí danh alias để khớp nếu dùng sai keyword + # Tầng 2: Alias match for alias_key, aliases in type_aliases.items(): if poster_type in aliases or alias_key == poster_type: for alias in aliases: @@ -871,28 +1075,31 @@ Trả về định dạng JSON (không sử dụng markdown): break if matched_agent_id is not None: break - - # 3. Nếu xui xẻo vẫn không tìm thấy, lấy thẳng Agent có điểm Influence (Sức ảnh hưởng) cao nhất + + # Tầng 3: Influence fallback if matched_agent_id is None: logger.warning(f"Could not find matching Agent type '{poster_type}', assigning to highest influence Agent instead") if agent_configs: - # Sort ảnh hưởng giảm dần, lấy index [0] sorted_agents = sorted(agent_configs, key=lambda a: a.influence_weight, reverse=True) matched_agent_id = sorted_agents[0].agent_id else: matched_agent_id = 0 - + updated_posts.append({ "content": content, "poster_type": post.get("poster_type", "Unknown"), "poster_agent_id": matched_agent_id }) - + logger.info(f"Initial post assignment: poster_type='{poster_type}' -> agent_id={matched_agent_id}") - + event_config.initial_posts = updated_posts return event_config - + + # -------------------------------------------------------------------------- + # BƯỚC 3..N: Sinh cấu hình Agent (chia batch) + # -------------------------------------------------------------------------- + def _generate_agent_configs_batch( self, context: str, @@ -900,9 +1107,22 @@ Trả về định dạng JSON (không sử dụng markdown): start_idx: int, simulation_requirement: str ) -> List[AgentActivityConfig]: - """Chia đợt gửi lên gọi tạo Cấu hình mạng lưới Agents""" - - # Build các node Entity (Dựa trên cấu hình lượng chữ giới hạn) + """ + Gọi LLM để sinh cấu hình hoạt động cho một batch agents (tối đa AGENTS_PER_BATCH=15). + + Lý do chia batch: nếu có 50+ entity, gửi tất cả 1 lần sẽ vượt token limit + và LLM có xu hướng bỏ sót entity cuối. Chia 15 agent/lần đảm bảo đủ. + + Mỗi entity trong batch được format thành: + {"agent_id": 5, "entity_name": "Trần Văn An", "entity_type": "Student", "summary": "..."} + + LLM trả về dict với key "agent_configs": [...] + → extract theo agent_id, map vào AgentActivityConfig object. + → Nếu LLM bỏ sót agent nào → _generate_agent_config_by_rule() bù vào. + + agent_id trong output phải khớp chính xác với start_idx + i + (để OASIS map đúng với user_id trong profile file). + """ entity_list = [] summary_len = self.AGENT_SUMMARY_LENGTH for i, e in enumerate(entities): @@ -912,44 +1132,6 @@ Trả về định dạng JSON (không sử dụng markdown): "entity_type": e.get_entity_type() or "Unknown", "summary": e.summary[:summary_len] if e.summary else "" }) - -# prompt = f"""Based on the following information, generate social media activity configurations for each entity. - -# Simulation Requirements: {simulation_requirement} - -# ## Entity List -# ```json -# {json.dumps(entity_list, ensure_ascii=False, indent=2)} -# ``` - -# ## Task -# Generate activity configurations for each entity, noting: -# - **Time aligns with Vietnamese daily routines**: Almost no activity between 0-5 AM; most active during 7-10 PM (19:00-22:00). -# - **Official Institutions (University/GovernmentAgency)**: Low activity (0.1-0.3), active during work hours (9:00-17:00), slow response (60-240 mins), high influence (2.5-3.0). -# - **Media (MediaOutlet)**: Medium activity (0.4-0.6), active all day (8:00-23:00), fast response (5-30 mins), high influence (2.0-2.5). -# - **Individuals (Student/Person/Alumni)**: High activity (0.6-0.9), active mainly in the evening (18:00-23:00), fast response (1-15 mins), low influence (0.8-1.2). -# - **Public Figures/Experts**: Medium activity (0.4-0.6), medium-high influence (1.5-2.0). - -# Return in JSON format (no markdown): -# {{ -# "agent_configs": [ -# {{ -# "agent_id": , -# "activity_level": <0.0-1.0>, -# "posts_per_hour": , -# "comments_per_hour": , -# "active_hours": [], -# "response_delay_min": , -# "response_delay_max": , -# "sentiment_bias": <-1.0 to 1.0>, -# "stance": "", -# "influence_weight": -# }}, -# ... -# ] -# }}""" - -# system_prompt = "You are a social media behavior analysis expert. Return pure JSON. Configurations must comply with Vietnamese daily routines." prompt = f"""Dựa trên các thông tin sau đây, hãy tạo cấu hình hoạt động trên mạng xã hội cho từng thực thể. @@ -988,24 +1170,25 @@ Trả về định dạng JSON (không sử dụng markdown): }}""" system_prompt = "Bạn là chuyên gia phân tích hành vi mạng xã hội. Trả về JSON thuần túy. Cấu hình phải phù hợp với thói quen sinh hoạt của người Việt Nam." - + try: result = self._call_llm_with_retry(prompt, system_prompt) + # Index theo agent_id để tra cứu O(1) khi fill vào từng entity llm_configs = {cfg["agent_id"]: cfg for cfg in result.get("agent_configs", [])} except Exception as e: logger.warning(f"Failed LLM generating Agent batch configs: {e}, falling back to default manual rules.") llm_configs = {} - - # Tạo object list cho AgentActivityConfig + + # Tạo AgentActivityConfig cho mỗi entity trong batch configs = [] for i, entity in enumerate(entities): agent_id = start_idx + i cfg = llm_configs.get(agent_id, {}) - - # Gán Manual tự động nếu Bot LLM thiếu xót + + # Nếu LLM bỏ sót agent này → dùng rule-based fallback if not cfg: cfg = self._generate_agent_config_by_rule(entity) - + config = AgentActivityConfig( agent_id=agent_id, entity_uuid=entity.uuid, @@ -1022,20 +1205,35 @@ Trả về định dạng JSON (không sử dụng markdown): influence_weight=cfg.get("influence_weight", 1.0) ) configs.append(config) - + return configs - + def _generate_agent_config_by_rule(self, entity: EntityNode) -> Dict[str, Any]: - """Tự động gen cấu hình 1 người (agent) dựa trên bộ rule cứng có sẵn nếu gọi bot LLM bị fail (Luật theo múi giờ sinh học)""" + """ + Sinh cấu hình agent bằng rule cứng khi LLM fail hoặc bỏ sót entity này. + + Rule được xây dựng dựa trên hành vi thực tế trên mạng xã hội Việt Nam: + + ┌──────────────────────┬──────────┬──────────────────┬───────────┬───────────┐ + │ Loại entity │ activity │ active_hours │ delay(ph) │ influence │ + ├──────────────────────┼──────────┼──────────────────┼───────────┼───────────┤ + │ university/gov/ngo │ 0.2 │ 9-17 (hành chính)│ 60-240 │ 3.0 │ + │ mediaoutlet │ 0.5 │ 7-23 (cả ngày) │ 5-30 │ 2.5 │ + │ professor/expert │ 0.4 │ 8-21 │ 15-90 │ 2.0 │ + │ student │ 0.8 │ sáng + tối │ 1-15 │ 0.8 │ + │ alumni │ 0.6 │ trưa + tối │ 5-30 │ 1.0 │ + │ (default) │ 0.7 │ 9-23 │ 2-20 │ 1.0 │ + └──────────────────────┴──────────┴──────────────────┴───────────┴───────────┘ + """ entity_type = (entity.get_entity_type() or "Unknown").lower() - + if entity_type in ["university", "governmentagency", "ngo"]: - # Cơ quan chức năng Nhà nước / Doanh nghiệp: làm việc trong khung giờ chuẩn hành chính, trả lời ít nhưng nặng đô + # Cơ quan nhà nước/tổ chức: hoạt động giờ hành chính, phát ngôn ít nhưng ảnh hưởng lớn return { "activity_level": 0.2, "posts_per_hour": 0.1, "comments_per_hour": 0.05, - "active_hours": list(range(9, 18)), # 9:00-17:59 + "active_hours": list(range(9, 18)), "response_delay_min": 60, "response_delay_max": 240, "sentiment_bias": 0.0, @@ -1043,12 +1241,12 @@ Trả về định dạng JSON (không sử dụng markdown): "influence_weight": 3.0 } elif entity_type in ["mediaoutlet"]: - # Báo đài truyền thông: cả ngày đưa tin, ra bài lẹ giật tít, tốc độ cao + # Báo đài: hoạt động cả ngày, đưa tin nhanh, nhiều người theo dõi return { "activity_level": 0.5, "posts_per_hour": 0.8, "comments_per_hour": 0.3, - "active_hours": list(range(7, 24)), # 7:00-23:59 + "active_hours": list(range(7, 24)), "response_delay_min": 5, "response_delay_max": 30, "sentiment_bias": 0.0, @@ -1056,12 +1254,12 @@ Trả về định dạng JSON (không sử dụng markdown): "influence_weight": 2.5 } elif entity_type in ["professor", "expert", "official"]: - # Giáo sư đại học/Người phát biểu: Chỉ nói ban ngày và tối, ra bài ít + # Chuyên gia/giảng viên: phát biểu có chọn lọc, ban ngày và tối sớm return { "activity_level": 0.4, "posts_per_hour": 0.3, "comments_per_hour": 0.5, - "active_hours": list(range(8, 22)), # 8:00-21:59 + "active_hours": list(range(8, 22)), "response_delay_min": 15, "response_delay_max": 90, "sentiment_bias": 0.0, @@ -1069,12 +1267,12 @@ Trả về định dạng JSON (không sử dụng markdown): "influence_weight": 2.0 } elif entity_type in ["student"]: - # Tần suất cho lứa Sinh viên: hay ra bài / cãi nhau liên tục ban đêm rất nhiều + # Sinh viên: rất tích cực, đặc biệt buổi tối, phản ứng nhanh return { "activity_level": 0.8, "posts_per_hour": 0.6, "comments_per_hour": 1.5, - "active_hours": [8, 9, 10, 11, 12, 13, 18, 19, 20, 21, 22, 23], # Sáng + Đêm Tối + "active_hours": [8, 9, 10, 11, 12, 13, 18, 19, 20, 21, 22, 23], "response_delay_min": 1, "response_delay_max": 15, "sentiment_bias": 0.0, @@ -1082,12 +1280,12 @@ Trả về định dạng JSON (không sử dụng markdown): "influence_weight": 0.8 } elif entity_type in ["alumni"]: - # Cựu sinh viên: Thường online đêm là chính + # Cựu sinh viên: online giờ nghỉ trưa + buổi tối sau giờ làm return { "activity_level": 0.6, "posts_per_hour": 0.4, "comments_per_hour": 0.8, - "active_hours": [12, 13, 19, 20, 21, 22, 23], # Giờ nghỉ trưa + Buổi tối + "active_hours": [12, 13, 19, 20, 21, 22, 23], "response_delay_min": 5, "response_delay_max": 30, "sentiment_bias": 0.0, @@ -1095,17 +1293,15 @@ Trả về định dạng JSON (không sử dụng markdown): "influence_weight": 1.0 } else: - # Thuộc cho số đông (Cư dân mạng / Người Qua Đường): Phấn khích về đêm + # Default — cư dân mạng thông thường: hoạt động ban ngày + tối return { "activity_level": 0.7, "posts_per_hour": 0.5, "comments_per_hour": 1.2, - "active_hours": [9, 10, 11, 12, 13, 18, 19, 20, 21, 22, 23], # Ban Ngày rảnh + Buổi tối rảnh + "active_hours": [9, 10, 11, 12, 13, 18, 19, 20, 21, 22, 23], "response_delay_min": 2, "response_delay_max": 20, "sentiment_bias": 0.0, "stance": "neutral", "influence_weight": 1.0 } - - diff --git a/backend/app/services/simulation_ipc.py b/backend/app/services/simulation_ipc.py index afddff04..4c7b6eed 100644 --- a/backend/app/services/simulation_ipc.py +++ b/backend/app/services/simulation_ipc.py @@ -137,7 +137,7 @@ class SimulationIPCClient: TimeoutError: Lỗi quá thời gian chờ phản hồi """ command_id = str(uuid.uuid4()) - command = IPCCommand( + command = s( command_id=command_id, command_type=command_type, args=args diff --git a/backend/app/services/simulation_runner.py b/backend/app/services/simulation_runner.py index bc5193fd..baf2421e 100644 --- a/backend/app/services/simulation_runner.py +++ b/backend/app/services/simulation_runner.py @@ -1,6 +1,28 @@ """ -OASIS模拟运行器 -在后台运行模拟并记录每个Agent的动作,支持实时状态监控 +Trình chạy và giám sát Simulation OASIS + +Vị trí trong pipeline: +───────────────────────────────────────────────────────────────────────────── + simulation_manager.py → SimulationRunner.start_simulation() + └─ subprocess.Popen(run_parallel_simulation.py --config ...) + │ (OASIS chạy ngầm, ghi log ra actions.jsonl) + │ + └─ Thread(_monitor_simulation) [daemon thread, chạy song song với Flask] + └─ _read_action_log() [đọc actions.jsonl mỗi 2 giây] + └─ _save_run_state() [cập nhật run_state.json liên tục] +───────────────────────────────────────────────────────────────────────────── + +Hai file state riêng biệt (KHÔNG phải một): + state.json — lifecycle state, quản lý bởi SimulationManager + (CREATED → PREPARING → READY → RUNNING → COMPLETED/FAILED) + run_state.json — runtime progress, quản lý bởi SimulationRunner + (round hiện tại, action count, PID, ...) + Cập nhật mỗi 2 giây trong khi OASIS đang chạy. + +Giao tiếp Flask ↔ OASIS subprocess: + File-based: Flask đọc actions.jsonl để biết tiến độ + IPC: Flask ghi lệnh vào ipc_commands/, OASIS trả lời vào ipc_responses/ + (dùng cho tính năng Interview agent đang chạy) """ import os @@ -25,38 +47,61 @@ from .simulation_ipc import SimulationIPCClient, CommandType, IPCResponse logger = get_logger('mirofish.simulation_runner') -# Cờ đánh dấu đã đăng ký hàm dọn dẹp hay chưa +# Cờ đánh dấu đã đăng ký hàm dọn dẹp (atexit) hay chưa — tránh đăng ký nhiều lần _cleanup_registered = False -# Kiểm tra hệ điều hành +# Kiểm tra hệ điều hành — dùng để chọn cách kill process (killpg vs taskkill) IS_WINDOWS = sys.platform == 'win32' +# ============================================================================== +# ENUM: RunnerStatus — Trạng thái của bộ chạy tiến trình (khác với SimulationStatus) +# ============================================================================== +# RunnerStatus theo dõi trạng thái của OASIS subprocess, không phải lifecycle tổng thể. +# Được lưu trong run_state.json — SimulationManager đọc file này để cập nhật state.json. +# +# Luồng trạng thái: +# IDLE → STARTING → RUNNING → COMPLETED +# └──────────→ FAILED (subprocess crash) +# RUNNING → STOPPING → STOPPED (người dùng stop) +# ============================================================================== + class RunnerStatus(str, Enum): """Trạng thái của bộ chạy tiến trình mô phỏng""" - IDLE = "idle" # Rảnh rỗi, chưa chạy - STARTING = "starting" # Đang khởi động - RUNNING = "running" # Đang chạy - PAUSED = "paused" # Đã tạm dừng - STOPPING = "stopping" # Đang dừng lại - STOPPED = "stopped" # Đã dừng - COMPLETED = "completed" # Đã hoàn thành - FAILED = "failed" # Bị lỗi + IDLE = "idle" # Rảnh rỗi, chưa chạy + STARTING = "starting" # Đang khởi động subprocess + RUNNING = "running" # Subprocess đang chạy, agents đang hành động + PAUSED = "paused" # Đã tạm dừng (tính năng tương lai) + STOPPING = "stopping" # Đang gửi SIGTERM / taskkill + STOPPED = "stopped" # Đã dừng theo yêu cầu người dùng + COMPLETED = "completed" # OASIS chạy hết số vòng và thoát thành công (exit_code=0) + FAILED = "failed" # Subprocess crash (exit_code != 0) hoặc exception không xử lý được +# ============================================================================== +# DATACLASS: AgentAction — Bản ghi một hành động đơn lẻ từ actions.jsonl +# ============================================================================== +# Mỗi dòng trong actions.jsonl tương ứng với 1 AgentAction. +# _read_action_log() parse từng dòng JSON thành AgentAction và lưu vào SimulationRunState. +# +# Ví dụ 1 dòng trong twitter/actions.jsonl: +# {"round": 15, "agent_id": 3, "agent_name": "tran_van_an_492", +# "action_type": "CREATE_POST", "action_args": {"content": "..."}, "success": true} +# ============================================================================== + @dataclass class AgentAction: """Bản ghi hành động của Agent""" - round_num: int # Số thứ tự của vòng (round) mô phỏng + round_num: int # Số thứ tự vòng (round) mô phỏng timestamp: str # Dấu thời gian - platform: str # Nền tảng thực hiện: twitter / reddit - agent_id: int # ID của agent - agent_name: str # Tên của agent - action_type: str # Loại hành động: CREATE_POST, LIKE_POST, v.v. - action_args: Dict[str, Any] = field(default_factory=dict) # Tham số của hành động - result: Optional[str] = None # Kết quả thực thi + platform: str # Nền tảng: "twitter" hoặc "reddit" + agent_id: int # ID của agent (khớp với user_id trong profile file) + agent_name: str # Tên tài khoản (ví dụ: "tran_van_an_492") + action_type: str # Loại hành động: CREATE_POST, LIKE_POST, REPOST, FOLLOW, DO_NOTHING... + action_args: Dict[str, Any] = field(default_factory=dict) # Tham số của hành động + result: Optional[str] = None # Kết quả thực thi (ví dụ: "Post created with id 142") success: bool = True # Hành động có thành công hay không - + def to_dict(self) -> Dict[str, Any]: return { "round_num": self.round_num, @@ -71,18 +116,25 @@ class AgentAction: } +# ============================================================================== +# DATACLASS: RoundSummary — Tóm tắt một vòng (round) mô phỏng +# ============================================================================== +# Được tổng hợp từ các AgentAction trong cùng round_num. +# Lưu vào SimulationRunState.rounds để hiển thị timeline trên frontend. +# ============================================================================== + @dataclass class RoundSummary: """Tóm tắt thông tin của mỗi vòng (round)""" - round_num: int # Số thứ tự vòng - start_time: str # Thời gian bắt đầu - end_time: Optional[str] = None # Thời gian kết thúc - simulated_hour: int = 0 # Số giờ đã mô phỏng trong vòng này - twitter_actions: int = 0 # Số hành động trên Twitter - reddit_actions: int = 0 # Số hành động trên Reddit - active_agents: List[int] = field(default_factory=list) # Danh sách ID các agent đang hoạt động - actions: List[AgentAction] = field(default_factory=list) # Danh sách các hành động - + round_num: int # Số thứ tự vòng + start_time: str # Thời gian bắt đầu vòng + end_time: Optional[str] = None # Thời gian kết thúc vòng + simulated_hour: int = 0 # Giờ mô phỏng tương ứng (0–71 nếu total_hours=72) + twitter_actions: int = 0 # Số hành động trên Twitter trong vòng này + reddit_actions: int = 0 # Số hành động trên Reddit trong vòng này + active_agents: List[int] = field(default_factory=list) # Danh sách agent_id đã hành động + actions: List[AgentAction] = field(default_factory=list) # Chi tiết các hành động + def to_dict(self) -> Dict[str, Any]: return { "round_num": self.round_num, @@ -97,66 +149,86 @@ class RoundSummary: } +# ============================================================================== +# DATACLASS: SimulationRunState — Trạng thái runtime của OASIS subprocess +# ============================================================================== +# Đây là object lưu trạng thái "đang chạy" — được cập nhật mỗi 2 giây bởi monitor thread. +# Được ghi ra run_state.json để Frontend có thể poll API và hiển thị tiến độ. +# +# Phân biệt với SimulationState (state.json): +# SimulationState = lifecycle tổng thể (CREATED→READY→RUNNING→COMPLETED) +# SimulationRunState = tiến độ chi tiết (round hiện tại, action count, PID, ...) +# +# recent_actions: chỉ giữ 50 hành động gần nhất (max_recent_actions) +# → Không lưu toàn bộ actions vào RAM để tránh tốn bộ nhớ với simulation dài. +# ============================================================================== + @dataclass class SimulationRunState: """Trạng thái đang thực thi của tiến trình mô phỏng (cập nhật theo thời gian thực)""" simulation_id: str runner_status: RunnerStatus = RunnerStatus.IDLE - - # Thông tin tiến độ - current_round: int = 0 - total_rounds: int = 0 - simulated_hours: int = 0 - total_simulation_hours: int = 0 - - # Các vòng lặp và thời gian độc lập cho từng nền tảng (sử dụng để hiển thị song song hai nền tảng) + + # --- Tiến độ chung --- + current_round: int = 0 # Vòng hiện tại (số lớn nhất của 2 platform) + total_rounds: int = 0 # Tổng số vòng cần chạy + simulated_hours: int = 0 # Số giờ đã mô phỏng + total_simulation_hours: int = 0 # Tổng số giờ cần mô phỏng + + # --- Tiến độ riêng từng platform --- + # Hai platform chạy song song asyncio → hoàn thành không đồng bộ + # Frontend hiển thị 2 progress bar riêng dựa trên các field này twitter_current_round: int = 0 reddit_current_round: int = 0 twitter_simulated_hours: int = 0 reddit_simulated_hours: int = 0 - - # Trạng thái nền tảng đang chạy - twitter_running: bool = False - reddit_running: bool = False - twitter_actions_count: int = 0 - reddit_actions_count: int = 0 - - # Trạng thái hoàn thành chung của nền tảng (phát hiện qua sự kiện simulation_end trong actions.jsonl) + + # --- Trạng thái nền tảng --- + twitter_running: bool = False # True khi OASIS Twitter đang chạy + reddit_running: bool = False # True khi OASIS Reddit đang chạy + twitter_actions_count: int = 0 # Tổng hành động Twitter đã ghi được + reddit_actions_count: int = 0 # Tổng hành động Reddit đã ghi được + + # Phát hiện qua event_type="simulation_end" trong actions.jsonl + # (KHÔNG dựa vào exit_code của subprocess — vì exit_code chỉ biết khi process kết thúc) twitter_completed: bool = False reddit_completed: bool = False - - # Tóm tắt lại ở mỗi vòng + + # --- Lịch sử vòng --- rounds: List[RoundSummary] = field(default_factory=list) - - # Các hành động gần nhất (để hiển thị theo thời gian thực (real-time) trên frontend) + + # --- Hành động gần nhất (tối đa 50) --- + # insert(0, ...) → luôn giữ mới nhất ở đầu list recent_actions: List[AgentAction] = field(default_factory=list) max_recent_actions: int = 50 - - # Dấu thời gian + + # --- Timestamps --- started_at: Optional[str] = None updated_at: str = field(default_factory=lambda: datetime.now().isoformat()) completed_at: Optional[str] = None - - # Thông tin lỗi + + # --- Lỗi --- error: Optional[str] = None - - # ID tiến trình (PID) (để dừng/hủy tiến trình) + + # --- Process ID --- + # Lưu lại để có thể kill process khi cần dừng process_pid: Optional[int] = None - + def add_action(self, action: AgentAction): - """Thêm một hành động vào danh sách các hành động gần nhất""" + """Thêm hành động vào đầu recent_actions, giữ tối đa max_recent_actions.""" self.recent_actions.insert(0, action) if len(self.recent_actions) > self.max_recent_actions: self.recent_actions = self.recent_actions[:self.max_recent_actions] - + if action.platform == "twitter": self.twitter_actions_count += 1 else: self.reddit_actions_count += 1 - + self.updated_at = datetime.now().isoformat() - + def to_dict(self) -> Dict[str, Any]: + """Serialize ra dict gọn — dùng cho API response và ghi run_state.json.""" return { "simulation_id": self.simulation_id, "runner_status": self.runner_status.value, @@ -164,8 +236,8 @@ class SimulationRunState: "total_rounds": self.total_rounds, "simulated_hours": self.simulated_hours, "total_simulation_hours": self.total_simulation_hours, + # Phần trăm hoàn thành: tránh ZeroDivisionError khi total_rounds=0 "progress_percent": round(self.current_round / max(self.total_rounds, 1) * 100, 1), - # Vòng lặp và thời gian độc lập cho mỗi nền tảng "twitter_current_round": self.twitter_current_round, "reddit_current_round": self.reddit_current_round, "twitter_simulated_hours": self.twitter_simulated_hours, @@ -183,72 +255,103 @@ class SimulationRunState: "error": self.error, "process_pid": self.process_pid, } - + def to_detail_dict(self) -> Dict[str, Any]: - """Chi tiết thông tin bao gồm các hành động gần nhất""" + """Serialize đầy đủ bao gồm recent_actions — dùng để ghi run_state.json.""" result = self.to_dict() result["recent_actions"] = [a.to_dict() for a in self.recent_actions] result["rounds_count"] = len(self.rounds) return result +# ============================================================================== +# CLASS: SimulationRunner — Trình chạy và giám sát OASIS subprocess +# ============================================================================== +# SimulationRunner là class-only (tất cả methods đều là @classmethod) — dùng như singleton. +# Không cần khởi tạo instance, gọi trực tiếp: SimulationRunner.start_simulation(...) +# +# Tại sao class-level dict thay vì instance dict? +# Flask chạy nhiều request đồng thời → nhiều thread cùng truy cập runner state. +# Class-level dict được chia sẻ giữa tất cả thread trong cùng Flask process. +# → Không cần singleton pattern phức tạp, class dict là đủ. +# +# _processes: simulation_id → subprocess.Popen (để kill khi cần) +# _run_states: simulation_id → SimulationRunState (RAM cache của run_state.json) +# _monitor_threads: simulation_id → Thread (daemon thread đọc actions.jsonl) +# _stdout_files: simulation_id → file handle (để đóng khi process kết thúc) +# ============================================================================== + class SimulationRunner: """ Trình chạy mô phỏng - + Quy trách nhiệm: 1. Chạy mô phỏng OASIS trong tiến trình nền (background process) 2. Phân tích nhật ký chạy (log), ghi lại hành động của mỗi Agent 3. Cung cấp API truy vấn trạng thái thời gian thực 4. Hỗ trợ thao tác tạm dừng (pause)/dừng (stop)/tiếp tục (resume) """ - - # Thư mục lưu trữ trạng thái chạy + + # Thư mục lưu trữ trạng thái chạy (mỗi simulation có subfolder riêng) RUN_STATE_DIR = os.path.join( os.path.dirname(__file__), '../../uploads/simulations' ) - - # Thư mục chứa các script con (script chạy ứng dụng) + + # Thư mục chứa các OASIS script (run_parallel_simulation.py, ...) + # Scripts không được copy vào thư mục simulation — gọi thẳng từ đây SCRIPTS_DIR = os.path.join( os.path.dirname(__file__), '../../scripts' ) - - # Trạng thái chạy trong bộ nhớ Memory (RAM) + + # --- Class-level state (chia sẻ giữa tất cả request threads) --- _run_states: Dict[str, SimulationRunState] = {} _processes: Dict[str, subprocess.Popen] = {} _action_queues: Dict[str, Queue] = {} _monitor_threads: Dict[str, threading.Thread] = {} - _stdout_files: Dict[str, Any] = {} # Lưu trữ tay cầm file đầu ra chuẩn (stdout) - _stderr_files: Dict[str, Any] = {} # Lưu trữ tay cầm file lỗi chuẩn (stderr) - - # Cấu hình cập nhật bộ nhớ Đồ thị (Graph Memory) - _graph_memory_enabled: Dict[str, bool] = {} # simulation_id -> enabled (Bật/tắt) - + _stdout_files: Dict[str, Any] = {} # Lưu file handle để đóng khi process kết thúc + _stderr_files: Dict[str, Any] = {} # Không dùng riêng (stderr merge vào stdout) + + # Cấu hình Graph Memory Update (tính năng tùy chọn: ghi hành động agent vào Zep) + _graph_memory_enabled: Dict[str, bool] = {} # simulation_id → True/False + @classmethod def get_run_state(cls, simulation_id: str) -> Optional[SimulationRunState]: - """Lấy trạng thái chạy hiện tại""" + """ + Lấy trạng thái runtime hiện tại — cache-aside (RAM trước, disk fallback). + Trả về None nếu simulation chưa được start. + """ if simulation_id in cls._run_states: return cls._run_states[simulation_id] - - # Thử tải từ file nếu không có trong memory + + # Không có trong RAM → thử load từ run_state.json state = cls._load_run_state(simulation_id) if state: cls._run_states[simulation_id] = state return state - + @classmethod def _load_run_state(cls, simulation_id: str) -> Optional[SimulationRunState]: - """Tải trạng thái chạy từ tệp tin (run_state.json)""" + """ + Load SimulationRunState từ run_state.json trên disk. + + Được gọi khi: + 1. Server restart → RAM cache trống → cần load lại từ disk + 2. Lần đầu gọi get_run_state() sau khi start_simulation() + + Lưu ý: nếu run_state.json có status="running" sau server restart, + đó là "false positive" — process đã chết cùng server. Phải gọi + cleanup_simulation_logs() trước khi start lại. + """ state_file = os.path.join(cls.RUN_STATE_DIR, simulation_id, "run_state.json") if not os.path.exists(state_file): return None - + try: with open(state_file, 'r', encoding='utf-8') as f: data = json.load(f) - + state = SimulationRunState( simulation_id=simulation_id, runner_status=RunnerStatus(data.get("runner_status", "idle")), @@ -256,7 +359,6 @@ class SimulationRunner: total_rounds=data.get("total_rounds", 0), simulated_hours=data.get("simulated_hours", 0), total_simulation_hours=data.get("total_simulation_hours", 0), - # Các vòng lặp và thời gian độc lập cho mỗi nền tảng twitter_current_round=data.get("twitter_current_round", 0), reddit_current_round=data.get("reddit_current_round", 0), twitter_simulated_hours=data.get("twitter_simulated_hours", 0), @@ -273,8 +375,8 @@ class SimulationRunner: error=data.get("error"), process_pid=data.get("process_pid"), ) - - # Tải danh sách các hành động gần đây + + # Restore 50 hành động gần nhất (để hiển thị lại sau restart) actions_data = data.get("recent_actions", []) for a in actions_data: state.recent_actions.append(AgentAction( @@ -288,76 +390,99 @@ class SimulationRunner: result=a.get("result"), success=a.get("success", True), )) - + return state except Exception as e: logger.error(f"Failed to load run state: {str(e)}") return None - + @classmethod def _save_run_state(cls, state: SimulationRunState): - """Lưu trạng thái chạy vào file""" + """ + Ghi SimulationRunState ra run_state.json VÀ cập nhật RAM cache. + + Được gọi: + 1. Sau mỗi lần đọc actions.jsonl trong monitor thread (mỗi 2 giây) + 2. Khi trạng thái thay đổi (STARTING → RUNNING → COMPLETED/FAILED) + 3. Khi process kết thúc (cleanup trong finally block) + """ sim_dir = os.path.join(cls.RUN_STATE_DIR, state.simulation_id) os.makedirs(sim_dir, exist_ok=True) state_file = os.path.join(sim_dir, "run_state.json") - + + # to_detail_dict() bao gồm recent_actions — đầy đủ hơn to_dict() data = state.to_detail_dict() - + with open(state_file, 'w', encoding='utf-8') as f: json.dump(data, f, ensure_ascii=False, indent=2) - + cls._run_states[state.simulation_id] = state - + + # -------------------------------------------------------------------------- + # PUBLIC: start_simulation — Khởi động OASIS subprocess + monitor thread + # -------------------------------------------------------------------------- + @classmethod def start_simulation( cls, simulation_id: str, - platform: str = "parallel", # twitter / reddit / parallel - max_rounds: int = None, # Số vòng mô phỏng tối đa (tùy chọn, dùng để cắt ngắn các mô phỏng quá dài) - enable_graph_memory_update: bool = False, # Có liên tục cập nhật hoạt động của Agent vào Zep graph hay không - graph_id: str = None # ID của Zep graph (Bắt buộc nếu bật tính năng cập nhật sơ đồ (graph)) + platform: str = "parallel", # "twitter" / "reddit" / "parallel" + max_rounds: int = None, # Giới hạn số vòng tối đa (None = không giới hạn) + enable_graph_memory_update: bool = False, # Ghi hành động agent vào Zep Graph + graph_id: str = None # Bắt buộc nếu enable_graph_memory_update=True ) -> SimulationRunState: """ - Bắt đầu mô phỏng - + Khởi động simulation: + 1. Đọc simulation_config.json → tính total_rounds + 2. subprocess.Popen(run_parallel_simulation.py --config ...) → OASIS chạy ngầm + 3. Thread(_monitor_simulation) → daemon thread đọc actions.jsonl mỗi 2 giây + + Sau khi hàm này return, OASIS đang chạy trong background. + Flask vẫn nhận request bình thường — không bị block. + + start_new_session=True: tạo process group mới → có thể kill toàn bộ cây process + bằng os.killpg() (thay vì chỉ kill process cha). + Args: - simulation_id: ID mô phỏng - platform: Nền tảng chạy (twitter/reddit/parallel) - max_rounds: Số vòng chạy tối đa (để cắt bớt) - enable_graph_memory_update: Có cập nhật hành vi Agent vào Zep Graph hay không - graph_id: Zep Graph ID - + simulation_id: ID của simulation đã qua prepare (status=READY) + platform: Nền tảng chạy ("twitter"/"reddit"/"parallel") + max_rounds: Cắt ngắn số vòng nếu chỉ muốn chạy thử + enable_graph_memory_update: Ghi lại hành động agent vào Zep Graph theo real-time + graph_id: Zep Graph ID (bắt buộc khi enable_graph_memory_update=True) + Returns: - SimulationRunState (Trạng thái sau khi cấu hình) + SimulationRunState với runner_status=RUNNING """ - # Kiểm tra xem có tiến trình nào đang chạy không + # Kiểm tra không cho start khi đã đang chạy existing = cls.get_run_state(simulation_id) if existing and existing.runner_status in [RunnerStatus.RUNNING, RunnerStatus.STARTING]: raise ValueError(f"Simulation is already running: {simulation_id}") - - # Tải cấu hình mô phỏng + + # Đọc simulation_config.json để tính total_rounds sim_dir = os.path.join(cls.RUN_STATE_DIR, simulation_id) config_path = os.path.join(sim_dir, "simulation_config.json") - + if not os.path.exists(config_path): raise ValueError(f"Simulation configuration not found, please call the /prepare API first") - + with open(config_path, 'r', encoding='utf-8') as f: config = json.load(f) - - # Khởi tạo trạng thái chạy + + # Tính total_rounds từ time_config time_config = config.get("time_config", {}) total_hours = time_config.get("total_simulation_hours", 72) minutes_per_round = time_config.get("minutes_per_round", 30) total_rounds = int(total_hours * 60 / minutes_per_round) - - # Nếu chỉ định maximum rounds, tiến hành việc cắt bớt + # Ví dụ: 72h × 60min / 60min/round = 72 rounds + + # Áp dụng giới hạn max_rounds nếu có (dùng để test chạy nhanh) if max_rounds is not None and max_rounds > 0: original_rounds = total_rounds total_rounds = min(total_rounds, max_rounds) if total_rounds < original_rounds: logger.info(f"Rounds truncated: {original_rounds} -> {total_rounds} (max_rounds={max_rounds})") - + + # Khởi tạo run state ban đầu với status=STARTING state = SimulationRunState( simulation_id=simulation_id, runner_status=RunnerStatus.STARTING, @@ -365,14 +490,14 @@ class SimulationRunner: total_simulation_hours=total_hours, started_at=datetime.now().isoformat(), ) - + cls._save_run_state(state) - - # Nếu tính năng cập nhật bộ nhớ graph được bật, tạo một updater + + # Khởi tạo Graph Memory Updater nếu được bật if enable_graph_memory_update: if not graph_id: raise ValueError("graph_id is required to enable graph memory updates") - + try: ZepGraphMemoryManager.create_updater(simulation_id, graph_id) cls._graph_memory_enabled[simulation_id] = True @@ -382,8 +507,8 @@ class SimulationRunner: cls._graph_memory_enabled[simulation_id] = False else: cls._graph_memory_enabled[simulation_id] = False - - # Xác định script nào sẽ chạy (các script nằm trong thư mục backend/scripts/) + + # Chọn script OASIS phù hợp với platform if platform == "twitter": script_name = "run_twitter_simulation.py" state.twitter_running = True @@ -391,71 +516,63 @@ class SimulationRunner: script_name = "run_reddit_simulation.py" state.reddit_running = True else: + # "parallel" — chạy cả hai platform đồng thời (asyncio bên trong script) script_name = "run_parallel_simulation.py" state.twitter_running = True state.reddit_running = True - + script_path = os.path.join(cls.SCRIPTS_DIR, script_name) - + if not os.path.exists(script_path): raise ValueError(f"Script no longer exists: {script_path}") - - # Tạo hàng đợi các hành động (Queue) + + # Tạo action queue (hiện chưa dùng nhưng giữ lại cho tương lai) action_queue = Queue() cls._action_queues[simulation_id] = action_queue - - # Bắt đầu chạy tiến trình mô phỏng + + # Khởi động OASIS subprocess try: - # Xây dựng lệnh chạy, sử dụng full path - # Cấu trúc log mới: - # twitter/actions.jsonl - Log cho các hành động trên Twitter - # reddit/actions.jsonl - Log cho các hành động trên Reddit - # simulation.log - Log cho tiến trình chính - cmd = [ - sys.executable, # Python Interpreter + sys.executable, # Python interpreter hiện tại script_path, - "--config", config_path, # Use full path to config + "--config", config_path, ] - - # Nếu có thiết lập giới hạn vòng tối đa, hãy truyền nó qua dòng lệnh (command line args) + + # Truyền max_rounds vào script nếu có if max_rounds is not None and max_rounds > 0: cmd.extend(["--max-rounds", str(max_rounds)]) - - # Tạo tệp log chính để tránh bộ đệm ống dẫn (pipe buffer) stdout/stderr của tiến trình đầy + + # stdout/stderr → simulation.log (tránh pipe buffer đầy gây block) main_log_path = os.path.join(sim_dir, "simulation.log") main_log_file = open(main_log_path, 'w', encoding='utf-8') - - # Đặt môi trường cho quy trình con để đảm bảo trên Windows được mã hóa thành UTF-8 - # Điều này sửa lỗi thư viện của bên thứ 3 khi họ gọi file hệ thống nếu không chỉ định rõ encode. + + # Đảm bảo UTF-8 trên mọi OS (đặc biệt Windows mặc định CP1252) env = os.environ.copy() - env['PYTHONUTF8'] = '1' # Python 3.7+ hỗ trợ điều này, giúp mọi hàm open() mặc định theo UTF-8 - env['PYTHONIOENCODING'] = 'utf-8' # Đảm bảo đầu ra có stdout/stderr dưới dạng UTF-8 - - # Đặt thư mục làm việc (CWD - Current Working Directory) thành thư mục nơi mô phỏng - # Thiết lập start_new_session=True sẽ tạo ra nhóm các tiến trình con mới, vì thế thông qua os.killpg có thể hủy toàn bộ những cái đó + env['PYTHONUTF8'] = '1' # Python 3.7+: buộc mọi open() dùng UTF-8 + env['PYTHONIOENCODING'] = 'utf-8' # stdout/stderr cũng UTF-8 + process = subprocess.Popen( cmd, - cwd=sim_dir, + cwd=sim_dir, # Working dir = thư mục simulation (script đọc file tương đối từ đây) stdout=main_log_file, - stderr=subprocess.STDOUT, # Đẩy luồng stderr cũng vào file đó + stderr=subprocess.STDOUT, # Merge stderr vào stdout → 1 file log duy nhất text=True, - encoding='utf-8', # Explicitly specify encoding + encoding='utf-8', bufsize=1, - env=env, # Đi kèm bộ setting Environment có set UTF-8 - start_new_session=True, # Bắt đầu tạo 1 luồng xử lý mới (New process group) + env=env, + start_new_session=True, # Tạo process group mới → os.killpg() kill được toàn cây ) - - # Ghi lại file để cho bước đóng (close) được thực hiện dễ dàng + cls._stdout_files[simulation_id] = main_log_file - cls._stderr_files[simulation_id] = None # Không cần lưu file stderr độc lập nữa - + cls._stderr_files[simulation_id] = None # Không cần file riêng cho stderr + state.process_pid = process.pid state.runner_status = RunnerStatus.RUNNING cls._processes[simulation_id] = process cls._save_run_state(state) - - # Khởi động tiểu trình giám sát (Monitor thread) + + # Khởi động daemon monitor thread — chạy song song với Flask + # daemon=True → thread tự chết khi main process kết thúc monitor_thread = threading.Thread( target=cls._monitor_simulation, args=(simulation_id,), @@ -463,105 +580,125 @@ class SimulationRunner: ) monitor_thread.start() cls._monitor_threads[simulation_id] = monitor_thread - + logger.info(f"Simulation started successfully: {simulation_id}, pid={process.pid}, platform={platform}") - + except Exception as e: state.runner_status = RunnerStatus.FAILED state.error = str(e) cls._save_run_state(state) raise - + return state - + + # -------------------------------------------------------------------------- + # PRIVATE: _monitor_simulation — Daemon thread theo dõi tiến độ real-time + # -------------------------------------------------------------------------- + @classmethod def _monitor_simulation(cls, simulation_id: str): - """Giám sát (Monitor) phân tích nhật ký ghi lại các hành động""" + """ + Daemon thread chạy liên tục, đọc actions.jsonl và cập nhật run_state.json. + + Cấu trúc log (mới — tách theo platform): + uploads/simulations//twitter/actions.jsonl + uploads/simulations//reddit/actions.jsonl + + Cơ chế file seek: + Mỗi lần đọc, hàm _read_action_log() trả về vị trí cuối file (f.tell()). + Lần tiếp theo, seek đến vị trí đó → chỉ đọc các dòng MỚI thêm vào. + → Tránh parse lại toàn bộ file mỗi 2 giây. + + Vòng lặp kết thúc khi process.poll() != None (process đã thoát). + Sau đó: đọc log lần cuối → xử lý exit_code → cleanup resources. + """ sim_dir = os.path.join(cls.RUN_STATE_DIR, simulation_id) - - # 新的日志结构:分平台的动作日志 + + # Cấu trúc log mới: tách log hành động theo từng nền tảng twitter_actions_log = os.path.join(sim_dir, "twitter", "actions.jsonl") reddit_actions_log = os.path.join(sim_dir, "reddit", "actions.jsonl") - + process = cls._processes.get(simulation_id) state = cls.get_run_state(simulation_id) - + if not process or not state: return - + + # Vị trí đọc cuối cùng trong mỗi file (file seek position) twitter_position = 0 reddit_position = 0 - + try: - while process.poll() is None: # 进程仍在运行 - # 读取 Twitter 动作日志 + # Vòng lặp chính: chạy cho đến khi process kết thúc + while process.poll() is None: + # Đọc log Twitter (chỉ đọc phần mới từ twitter_position) if os.path.exists(twitter_actions_log): twitter_position = cls._read_action_log( twitter_actions_log, twitter_position, state, "twitter" ) - - # 读取 Reddit 动作日志 + + # Đọc log Reddit (chỉ đọc phần mới từ reddit_position) if os.path.exists(reddit_actions_log): reddit_position = cls._read_action_log( reddit_actions_log, reddit_position, state, "reddit" ) - - # 更新状态 + + # Lưu trạng thái → Frontend có thể poll API để lấy tiến độ cls._save_run_state(state) - time.sleep(2) - - # 进程结束后,最后读取一次日志 + time.sleep(2) # Poll mỗi 2 giây + + # Process đã kết thúc — đọc log lần cuối để không bỏ sót action cuối if os.path.exists(twitter_actions_log): cls._read_action_log(twitter_actions_log, twitter_position, state, "twitter") if os.path.exists(reddit_actions_log): cls._read_action_log(reddit_actions_log, reddit_position, state, "reddit") - - # 进程结束 + + # Xử lý kết quả dựa trên exit_code exit_code = process.returncode - + if exit_code == 0: state.runner_status = RunnerStatus.COMPLETED state.completed_at = datetime.now().isoformat() - logger.info(f"模拟完成: {simulation_id}") + logger.info(f"Simulation completed: {simulation_id}") else: state.runner_status = RunnerStatus.FAILED - # 从主日志文件读取错误信息 + # Đọc 2000 ký tự cuối của simulation.log làm error message main_log_path = os.path.join(sim_dir, "simulation.log") error_info = "" try: if os.path.exists(main_log_path): with open(main_log_path, 'r', encoding='utf-8') as f: - error_info = f.read()[-2000:] # 取最后2000字符 + error_info = f.read()[-2000:] except Exception: pass - state.error = f"进程退出码: {exit_code}, 错误: {error_info}" - logger.error(f"模拟失败: {simulation_id}, error={state.error}") - + state.error = f"Process exit code: {exit_code}, error: {error_info}" + logger.error(f"Simulation failed: {simulation_id}, error={state.error}") + state.twitter_running = False state.reddit_running = False cls._save_run_state(state) - + except Exception as e: - logger.error(f"监控线程异常: {simulation_id}, error={str(e)}") + logger.error(f"Monitor thread exception: {simulation_id}, error={str(e)}") state.runner_status = RunnerStatus.FAILED state.error = str(e) cls._save_run_state(state) - + finally: - # 停止图谱记忆更新器 + # Dừng Graph Memory Updater nếu đang chạy if cls._graph_memory_enabled.get(simulation_id, False): try: ZepGraphMemoryManager.stop_updater(simulation_id) - logger.info(f"已停止图谱记忆更新: simulation_id={simulation_id}") + logger.info(f"Graph memory update stopped: simulation_id={simulation_id}") except Exception as e: - logger.error(f"停止图谱记忆更新器失败: {e}") + logger.error(f"Failed to stop graph memory updater: {e}") cls._graph_memory_enabled.pop(simulation_id, None) - - # 清理进程资源 + + # Dọn dẹp tài nguyên process cls._processes.pop(simulation_id, None) cls._action_queues.pop(simulation_id, None) - - # 关闭日志文件句柄 + + # Đóng file handle log (stdout) if simulation_id in cls._stdout_files: try: cls._stdout_files[simulation_id].close() @@ -574,48 +711,66 @@ class SimulationRunner: except Exception: pass cls._stderr_files.pop(simulation_id, None) - + + # -------------------------------------------------------------------------- + # PRIVATE: _read_action_log — Parse actions.jsonl từ vị trí đã đọc + # -------------------------------------------------------------------------- + @classmethod def _read_action_log( - cls, - log_path: str, - position: int, + cls, + log_path: str, + position: int, state: SimulationRunState, platform: str ) -> int: """ - Đọc tệp tin nhật ký (log) của hệ thống - + Đọc các dòng mới trong actions.jsonl từ vị trí `position`. + + Cơ chế file seek: + f.seek(position) → bỏ qua phần đã đọc trước + f.tell() → trả về vị trí hiện tại sau khi đọc → dùng cho lần tiếp theo + + Hai loại dòng JSON trong actions.jsonl: + + 1. Sự kiện hệ thống (có field "event_type"): + {"event_type": "round_end", "round": 15, "simulated_hours": 15} + {"event_type": "simulation_end", "total_rounds": 72, "total_actions": 3420} + → Cập nhật round_num, simulated_hours, completed flag + + 2. Hành động agent (không có "event_type", có "agent_id"): + {"round": 15, "agent_id": 3, "action_type": "CREATE_POST", ...} + → Tạo AgentAction object, thêm vào state.recent_actions + Args: - log_path: Đường dẫn tệp nhật ký - position: Vị trí đọc trước đó - state: Đối tượng trạng thái đang chạy - platform: Nền tảng (twitter/reddit) - + log_path: Đường dẫn file actions.jsonl + position: Vị trí byte đã đọc tới lần trước + state: SimulationRunState cần cập nhật + platform: "twitter" hoặc "reddit" + Returns: - Vị trí đọc mới + Vị trí byte mới (dùng cho lần gọi tiếp theo) """ - # Kiểm tra xem có bật tính năng cập nhật bộ nhớ graph hay không graph_memory_enabled = cls._graph_memory_enabled.get(state.simulation_id, False) graph_updater = None if graph_memory_enabled: graph_updater = ZepGraphMemoryManager.get_updater(state.simulation_id) - + try: with open(log_path, 'r', encoding='utf-8') as f: - f.seek(position) + f.seek(position) # Nhảy đến vị trí đã đọc lần trước for line in f: line = line.strip() if line: try: action_data = json.loads(line) - - # Xử lý các mục của loại sự kiện + + # Xử lý sự kiện hệ thống (event_type) if "event_type" in action_data: event_type = action_data.get("event_type") - - # Phát hiện sự kiện simulation_end và đánh dấu nền tảng đã hoàn thành + if event_type == "simulation_end": + # OASIS báo hiệu đã chạy hết số vòng → đánh dấu platform completed if platform == "twitter": state.twitter_completed = True state.twitter_running = False @@ -624,22 +779,20 @@ class SimulationRunner: state.reddit_completed = True state.reddit_running = False logger.info(f"Reddit simulation completed: {state.simulation_id}, total_rounds={action_data.get('total_rounds')}, total_actions={action_data.get('total_actions')}") - - # Kiểm tra xem có phải tất cả các nền tảng được bật đều đã hoàn thành hay không - # Nếu chỉ một nền tảng đang chạy, hãy chỉ kiểm tra nền tảng đó - # Nếu 2 nền tảng đang chạy thì yêu cầu phải hoàn thành cả 2 nền tảng + + # Kiểm tra nếu tất cả platform đã xong → đánh dấu COMPLETED + # Logic: platform nào bật (có actions.jsonl) thì phải xong; phải có ít nhất 1 xong all_completed = cls._check_all_platforms_completed(state) if all_completed: state.runner_status = RunnerStatus.COMPLETED state.completed_at = datetime.now().isoformat() logger.info(f"Simulation completed for all platforms: {state.simulation_id}") - - # Cập nhật thông tin vòng (round_num) (từ sự kiện round_end) + elif event_type == "round_end": + # Cập nhật số vòng và giờ đã mô phỏng (độc lập cho từng platform) round_num = action_data.get("round", 0) simulated_hours = action_data.get("simulated_hours", 0) - - # Cập nhật thời gian và vòng thứ tự độc lập cho nền tảng + if platform == "twitter": if round_num > state.twitter_current_round: state.twitter_current_round = round_num @@ -648,15 +801,15 @@ class SimulationRunner: if round_num > state.reddit_current_round: state.reddit_current_round = round_num state.reddit_simulated_hours = simulated_hours - - # Số vòng chung sẽ là số lớn nhất của hai nền tảng + + # Vòng chung = max của 2 platform (platform nhanh hơn làm mốc) if round_num > state.current_round: state.current_round = round_num - # Thời gian chung sẽ là số lớn nhất của hai nền tảng state.simulated_hours = max(state.twitter_simulated_hours, state.reddit_simulated_hours) - - continue - + + continue # Không tạo AgentAction cho sự kiện hệ thống + + # Xử lý hành động agent (có agent_id) action = AgentAction( round_num=action_data.get("round", 0), timestamp=action_data.get("timestamp", datetime.now().isoformat()), @@ -669,65 +822,83 @@ class SimulationRunner: success=action_data.get("success", True), ) state.add_action(action) - - # Cập nhật thông tin (vòng) round + + # Cập nhật current_round từ action nếu cần if action.round_num and action.round_num > state.current_round: state.current_round = action.round_num - - # Nếu cập nhật bộ nhớ graph được bật, thêm hoạt động vào Zep graph + + # Ghi hành động vào Zep Graph nếu tính năng được bật if graph_updater: graph_updater.add_activity_from_dict(action_data, platform) - + except json.JSONDecodeError: - pass - return f.tell() + pass # Dòng không phải JSON hợp lệ → bỏ qua + + return f.tell() # Trả về vị trí cuối để lần sau tiếp tục từ đây except Exception as e: logger.warning(f"Failed to read action logs: {log_path}, error={e}") - return position - + return position # Giữ nguyên position nếu đọc lỗi + @classmethod def _check_all_platforms_completed(cls, state: SimulationRunState) -> bool: """ - Kiểm tra xem tất cả các nền tảng có hoàn thành quá trình mô phỏng hay chưa? - - Kiểm tra xem nền tảng có được kích hoạt (hay enable) hay không bằng cách xem tệp tin actions.jsonl có tồn tại hay không - + Kiểm tra xem tất cả platform đã hoàn thành chưa. + + Cách xác định platform có được bật: + Kiểm tra file actions.jsonl có tồn tại không (không dùng enable_twitter/enable_reddit flag, + vì flag đó chỉ có trong SimulationState — không truyền sang đây). + + Logic: + Nếu twitter/actions.jsonl tồn tại → twitter được bật → phải twitter_completed=True + Nếu reddit/actions.jsonl tồn tại → reddit được bật → phải reddit_completed=True + Phải có ít nhất 1 platform xong (tránh trả True khi cả 2 chưa tạo file) + Returns: - True Nếu tất cả các nền tảng được bật đều đã hoàn thành + True nếu tất cả platform đã bật đều đã completed """ sim_dir = os.path.join(cls.RUN_STATE_DIR, state.simulation_id) twitter_log = os.path.join(sim_dir, "twitter", "actions.jsonl") reddit_log = os.path.join(sim_dir, "reddit", "actions.jsonl") - - # Kiểm tra xem mình có đang bật nền tảng nào không (Sử dụng cách kiểm tra tệp tin có tồn tại (exist) hay không) + twitter_enabled = os.path.exists(twitter_log) reddit_enabled = os.path.exists(reddit_log) - - # Nền tảng nào chưa xong thì trả về false + + # Platform nào được bật mà chưa xong → False if twitter_enabled and not state.twitter_completed: return False if reddit_enabled and not state.reddit_completed: return False - - # Phải có ít nhất 1 nền tảng chạy xong thì mới là true. (nếu 1 nền tảng không chạy => False. False and False = False. True and False = True) + + # Phải có ít nhất 1 platform chạy xong return twitter_enabled or reddit_enabled - + + # -------------------------------------------------------------------------- + # PRIVATE: _terminate_process — Kill process + toàn bộ process group + # -------------------------------------------------------------------------- + @classmethod def _terminate_process(cls, process: subprocess.Popen, simulation_id: str, timeout: int = 10): """ - Khả năng tương thích nền tảng, dừng một quá trình và các quá trình con (nhánh) - + Dừng process và toàn bộ process group (bao gồm child processes). + + Tại sao phải kill cả process group? + OASIS script có thể spawn thêm asyncio workers hoặc subprocess con. + Kill chỉ process cha sẽ để lại các zombie child processes. + + Unix: os.killpg(pgid, SIGTERM) → chờ 10s → os.killpg(pgid, SIGKILL) nếu không thoát + Windows: taskkill /T /PID → chờ → taskkill /F /T /PID nếu không thoát + + start_new_session=True (trong Popen) đảm bảo pgid == pid của process cha. + Args: - process: Quy trình để chấm dứt (kill) - simulation_id: Ghi log ID - timeout: Thời gian chờ cho phép tiến trình kết thúc tính bằng giây (seconds) + process: Subprocess.Popen object + simulation_id: Chỉ dùng để log + timeout: Giây chờ trước khi SIGKILL (mặc định 10s) """ if IS_WINDOWS: - # Windows: Sử dụng câu lệnh taskkill để xóa cả tiến trình theo cấu trúc branch tree - # /F = Force termination (Xoa bằng mọi giá), /T = Terminate tree (xóa nhánh tiến trình) bao gồm các sub-process - logger.info(f"Đang dừng quá trình (Windows): simulation={simulation_id}, pid={process.pid}") + logger.info(f"Terminating process (Windows): simulation={simulation_id}, pid={process.pid}") try: - # Trước hết hãy cố găng dừng mềm + # Bước 1: Dừng nhẹ nhàng (/T = kill cả tree) subprocess.run( ['taskkill', '/PID', str(process.pid), '/T'], capture_output=True, @@ -736,7 +907,7 @@ class SimulationRunner: try: process.wait(timeout=timeout) except subprocess.TimeoutExpired: - # Nếu không thể dùng dừng mềm (sau timeout), xóa cứng bằng /F + # Bước 2: Kill cưỡng bức (/F = force) logger.warning(f"Process unresponsive, force terminating: {simulation_id}") subprocess.run( ['taskkill', '/F', '/PID', str(process.pid), '/T'], @@ -752,59 +923,72 @@ class SimulationRunner: except subprocess.TimeoutExpired: process.kill() else: - # Unix: Sử dụng process group chấm dứt - # Dùng start_new_session=True, giá trị pgid sẽ bằng đúng với PID gốc của tiến trình + # Unix: dùng process group để kill toàn bộ cây pgid = os.getpgid(process.pid) - logger.info(f"Đang dừng nhóm tiến trình (Unix): simulation={simulation_id}, pgid={pgid}") - - # Gửi SIGTERM tới toàn bộ process group + logger.info(f"Terminating process group (Unix): simulation={simulation_id}, pgid={pgid}") + + # Bước 1: SIGTERM — tín hiệu nhẹ nhàng, cho phép process dọn dẹp os.killpg(pgid, signal.SIGTERM) - + try: process.wait(timeout=timeout) except subprocess.TimeoutExpired: - # Nếu xảy ra hiện tượng chưa tự hủy sau timeout, xóa cưỡng bức bằng SIGKILL + # Bước 2: SIGKILL — kill cưỡng bức, không thể bị bắt hay bỏ qua logger.warning(f"Process group unresponsive to SIGTERM, force terminating: {simulation_id}") os.killpg(pgid, signal.SIGKILL) process.wait(timeout=5) - + + # -------------------------------------------------------------------------- + # PUBLIC: stop_simulation — Dừng simulation theo yêu cầu người dùng + # -------------------------------------------------------------------------- + @classmethod def stop_simulation(cls, simulation_id: str) -> SimulationRunState: - """Dừng lại tiến trình mô phỏng""" + """ + Dừng OASIS subprocess → state = STOPPED. + + Thứ tự: + 1. Chuyển state sang STOPPING (để frontend biết đang xử lý) + 2. _terminate_process() — kill process group + 3. Chuyển state sang STOPPED + 4. Dừng Graph Memory Updater nếu đang chạy + + Khác với FAILED: STOPPED là do người dùng chủ động, FAILED là do crash. + """ state = cls.get_run_state(simulation_id) if not state: raise ValueError(f"Simulation not found: {simulation_id}") - + if state.runner_status not in [RunnerStatus.RUNNING, RunnerStatus.PAUSED]: raise ValueError(f"Simulation is not running: {simulation_id}, status={state.runner_status}") - + state.runner_status = RunnerStatus.STOPPING cls._save_run_state(state) - - # Kết thúc tiến trình con (child process) + + # Kill OASIS subprocess (và toàn bộ process group) process = cls._processes.get(simulation_id) if process and process.poll() is None: try: cls._terminate_process(process, simulation_id) except ProcessLookupError: - # Quá trình không còn ở đây nữa (Đã thoát hoặc bị đóng) + # Process đã tự thoát trước khi kịp kill — OK pass except Exception as e: logger.error(f"Failed to terminate process group: {simulation_id}, error={e}") - # Thử thêm một cách nữa để chăc chắn hủy tiến trình + # Fallback: try terminate() rồi kill() try: process.terminate() process.wait(timeout=5) except Exception: process.kill() - + state.runner_status = RunnerStatus.STOPPED state.twitter_running = False state.reddit_running = False state.completed_at = datetime.now().isoformat() cls._save_run_state(state) - - # Dừng quá trình graph memory updater + + # Dừng Graph Memory Updater if cls._graph_memory_enabled.get(simulation_id, False): try: ZepGraphMemoryManager.stop_updater(simulation_id) @@ -812,10 +996,14 @@ class SimulationRunner: except Exception as e: logger.error(f"Failed to stop graph memory updater: {e}") cls._graph_memory_enabled.pop(simulation_id, None) - + logger.info(f"Simulation stopped: {simulation_id}") return state - + + # -------------------------------------------------------------------------- + # PUBLIC: Đọc lịch sử hành động (dùng sau khi simulation kết thúc) + # -------------------------------------------------------------------------- + @classmethod def _read_actions_from_file( cls, @@ -826,48 +1014,51 @@ class SimulationRunner: round_num: Optional[int] = None ) -> List[AgentAction]: """ - Đọc các hoạt động từ một tệp duy nhất - + Đọc toàn bộ actions từ 1 file actions.jsonl với các bộ lọc tùy chọn. + + Bỏ qua: + - Dòng có "event_type" (sự kiện hệ thống: round_end, simulation_end) + - Dòng không có "agent_id" (không phải hành động của agent) + Args: - file_path: Đường dẫn tệp log của hành động đó - default_platform: Nền tảng mặc định (nếu trong nhật ký không có platform) - platform_filter: Lọc nền tảng (chỉ định platform cần đọc log) - agent_id: Lọc ID của Agent cụ thể - round_num: Lọc số vòng của Agent + file_path: Đường dẫn file actions.jsonl + default_platform: Điền tự động nếu record không có field "platform" + platform_filter: Chỉ lấy record của platform này + agent_id: Chỉ lấy record của agent này + round_num: Chỉ lấy record của vòng này """ if not os.path.exists(file_path): return [] - + actions = [] - + with open(file_path, 'r', encoding='utf-8') as f: for line in f: line = line.strip() if not line: continue - + try: data = json.loads(line) - - # Bỏ qua các bản ghi không phải hành động (chẳng hạn như là sự kiện về hệ thống: simulation_start, round_start, round_end v.v.) + + # Bỏ qua sự kiện hệ thống if "event_type" in data: continue - - # Bỏ lỡ các sự kiện không phải do agent tạo ra (Không có ID đặc trưng của Agent) + + # Bỏ qua record không có agent_id if "agent_id" not in data: continue - - # Lấy nền tảng (Platform): Ưu tiên lấy từ bản ghi nếu có `platform`, nếu không thì dùng `default_platform` + record_platform = data.get("platform") or default_platform or "" - - # Bộ lọc (Filtering) + + # Áp dụng bộ lọc if platform_filter and record_platform != platform_filter: continue if agent_id is not None and data.get("agent_id") != agent_id: continue if round_num is not None and data.get("round") != round_num: continue - + actions.append(AgentAction( round_num=data.get("round", 0), timestamp=data.get("timestamp", ""), @@ -879,12 +1070,12 @@ class SimulationRunner: result=data.get("result"), success=data.get("success", True), )) - + except json.JSONDecodeError: continue - + return actions - + @classmethod def get_all_actions( cls, @@ -894,58 +1085,55 @@ class SimulationRunner: round_num: Optional[int] = None ) -> List[AgentAction]: """ - Lấy thông tin tất cả lịch sử hoạt động của các nền tảng (không giới hạn phân trang) - - Args: - simulation_id: ID mô phỏng - platform: Bộ lọc nền tảng hoạt động (twitter/reddit) - agent_id: Lọc Agent - round_num: Lọc số vòng - - Returns: - Danh sách đầy đủ các actions (sắp xếp theo thời gian mới nhất lên trước) + Lấy toàn bộ lịch sử hành động — đọc từ file (không giới hạn phân trang). + + Thứ tự ưu tiên đọc file: + 1. twitter/actions.jsonl (cấu trúc mới — tách theo platform) + 2. reddit/actions.jsonl + 3. actions.jsonl (cấu trúc cũ — fallback tương thích ngược) + + Kết quả được sort theo timestamp giảm dần (mới nhất lên đầu). """ sim_dir = os.path.join(cls.RUN_STATE_DIR, simulation_id) actions = [] - - # Đọc tệp tin Actions của Twitter (Khai báo tự động điền twitter theo cấu trúc tệp tin log) + + # Đọc Twitter actions twitter_actions_log = os.path.join(sim_dir, "twitter", "actions.jsonl") if not platform or platform == "twitter": actions.extend(cls._read_actions_from_file( twitter_actions_log, - default_platform="twitter", # Điền dữ liệu tự động cho record `platform` + default_platform="twitter", platform_filter=platform, - agent_id=agent_id, + agent_id=agent_id, round_num=round_num )) - - # Đọc tệp tin Actions của Reddit (Tự động điền phần 'reddit' căn cứ thư mục chứa tệp tin) + + # Đọc Reddit actions reddit_actions_log = os.path.join(sim_dir, "reddit", "actions.jsonl") if not platform or platform == "reddit": actions.extend(cls._read_actions_from_file( reddit_actions_log, - default_platform="reddit", # Automatically fill the platform field + default_platform="reddit", platform_filter=platform, agent_id=agent_id, round_num=round_num )) - - # Nếu thư mục chạy các nền tảng chạy parallel này (twitter / reddit) không có ở đó. Hãy thử với các tệp định dạng cũ + + # Fallback: đọc file actions.jsonl cũ nếu không có thư mục twitter/reddit if not actions: actions_log = os.path.join(sim_dir, "actions.jsonl") actions = cls._read_actions_from_file( actions_log, - default_platform=None, # Các file json log định dạng cũ đã có sẵn record về platform nên không điền default + default_platform=None, # File cũ đã có field platform trong record platform_filter=platform, agent_id=agent_id, round_num=round_num ) - - # Sắp xếp lại log theo thời gian timestamp giảm dần (từ mới hơn lên trước) + actions.sort(key=lambda x: x.timestamp, reverse=True) - + return actions - + @classmethod def get_actions( cls, @@ -957,18 +1145,10 @@ class SimulationRunner: round_num: Optional[int] = None ) -> List[AgentAction]: """ - Lấy thông tin lịch sử diễn ra (Hỗ trợ phân trang bằng offset và limit) - - Args: - simulation_id: Simulation ID - limit: Record Count returns limit - offset: Offset - platform: Filter Platform - agent_id: Filter Agent by ID - round_num: Filter round loop - - Returns: - Actions list + Lấy lịch sử hành động có phân trang (offset + limit). + + Gọi get_all_actions() rồi cắt — không tối ưu cho file lớn nhưng đơn giản. + Nếu hiệu năng là vấn đề, cần cải thiện bằng cách chỉ đọc đến offset+limit. """ actions = cls.get_all_actions( simulation_id=simulation_id, @@ -976,10 +1156,9 @@ class SimulationRunner: agent_id=agent_id, round_num=round_num ) - - # Phân trang + return actions[offset:offset + limit] - + @classmethod def get_timeline( cls, @@ -988,29 +1167,27 @@ class SimulationRunner: end_round: Optional[int] = None ) -> List[Dict[str, Any]]: """ - Lấy thời gian (timeline) mô phỏng diễn ra (tóm tắt theo vòng được khai báo) - - Args: - simulation_id: ID Mô phỏng - start_round: Bắt đầu từ một vòng lặp nhất định (First round Number) - end_round: Kết thúc từ vòng ở đó (End round Number) - - Returns: - Cung cấp đầy đủ thông tin về các round bị gói gọn + Lấy timeline mô phỏng — tóm tắt theo từng vòng. + + Nhóm tất cả actions theo round_num, tính: + - Số action twitter/reddit trong vòng + - Danh sách agent_id đã hoạt động + - Phân phối action_type (CREATE_POST, LIKE_POST, ...) + + Dùng để hiển thị biểu đồ hoạt động theo thời gian trên frontend. """ actions = cls.get_actions(simulation_id, limit=10000) - - # Nhóm tiến trình lại vào trong Vòng (round grouping) + rounds: Dict[int, Dict[str, Any]] = {} - + for action in actions: round_num = action.round_num - + if round_num < start_round: continue if end_round is not None and round_num > end_round: continue - + if round_num not in rounds: rounds[round_num] = { "round_num": round_num, @@ -1021,19 +1198,19 @@ class SimulationRunner: "first_action_time": action.timestamp, "last_action_time": action.timestamp, } - + r = rounds[round_num] - + if action.platform == "twitter": r["twitter_actions"] += 1 else: r["reddit_actions"] += 1 - + r["active_agents"].add(action.agent_id) r["action_types"][action.action_type] = r["action_types"].get(action.action_type, 0) + 1 r["last_action_time"] = action.timestamp - - # Chuyển đổi trạng thái về Lists Arrays Data type + + # Chuyển set → list để JSON serializable, sort theo round_num tăng dần result = [] for round_num in sorted(rounds.keys()): r = rounds[round_num] @@ -1048,24 +1225,23 @@ class SimulationRunner: "first_action_time": r["first_action_time"], "last_action_time": r["last_action_time"], }) - + return result - + @classmethod def get_agent_stats(cls, simulation_id: str) -> List[Dict[str, Any]]: """ - Lấy thông kê của mọi agent - - Returns: - Danh sách thống kê Agent + Thống kê hoạt động của từng agent — sort theo tổng số hành động giảm dần. + + Dùng để hiển thị bảng xếp hạng "agent nào hoạt động nhiều nhất". """ actions = cls.get_actions(simulation_id, limit=10000) - + agent_stats: Dict[int, Dict[str, Any]] = {} - + for action in actions: agent_id = action.agent_id - + if agent_id not in agent_stats: agent_stats[agent_id] = { "agent_id": agent_id, @@ -1077,71 +1253,68 @@ class SimulationRunner: "first_action_time": action.timestamp, "last_action_time": action.timestamp, } - + stats = agent_stats[agent_id] stats["total_actions"] += 1 - + if action.platform == "twitter": stats["twitter_actions"] += 1 else: stats["reddit_actions"] += 1 - + stats["action_types"][action.action_type] = stats["action_types"].get(action.action_type, 0) + 1 stats["last_action_time"] = action.timestamp - - # Sắp xếp theo tổng số hành động giảm dần (reverse = true) + result = sorted(agent_stats.values(), key=lambda x: x["total_actions"], reverse=True) - + return result - + + # -------------------------------------------------------------------------- + # PUBLIC: cleanup — Dọn dẹp file log để cho phép chạy lại + # -------------------------------------------------------------------------- + @classmethod def cleanup_simulation_logs(cls, simulation_id: str) -> Dict[str, Any]: """ - Xóa tệp log chạy để buộc mô phỏng được khởi động lại - - Xóa sạch các tệp tin này bao gồm: - - run_state.json - - twitter/actions.jsonl - - reddit/actions.jsonl - - simulation.log - - stdout.log / stderr.log - - twitter_simulation.db(Dữ liệu nền tảng twitter) - - reddit_simulation.db(Dữ liệu nền tảng reddit) - - env_status.json(Trạng thái file Environment status) - - Chú ý: Các file liên kết đến cấu hình mô phỏng hay config thiết lập (như là simulation_config.json) hay Profile đều sẽ KHÔNG bị xóa đi. - - Args: - simulation_id: Simulation ID - + Xóa các file runtime để buộc simulation được khởi động lại từ đầu. + + CÁC FILE BỊ XÓA (runtime logs): + run_state.json, simulation.log, stdout.log, stderr.log, + twitter_simulation.db, reddit_simulation.db, env_status.json, + twitter/actions.jsonl, reddit/actions.jsonl + + CÁC FILE GIỮ LẠI (config + profiles — không xóa): + simulation_config.json, reddit_profiles.json, twitter_profiles.csv, state.json + + Khi nào cần gọi: + Sau server restart nếu run_state.json vẫn hiển thị status="running" + (process đã chết cùng server → không thể tiếp tục, phải cleanup rồi start lại) + Returns: - Kết quả của lệnh xóa sạch (clean up) + {"success": bool, "cleaned_files": [...], "errors": [...]} """ import shutil - + sim_dir = os.path.join(cls.RUN_STATE_DIR, simulation_id) - + if not os.path.exists(sim_dir): return {"success": True, "message": "Simulation directory does not exist, no need to clean."} - + cleaned_files = [] errors = [] - - # Các tệp tin cần bị loại bỏ bao gồm log, database... + files_to_delete = [ "run_state.json", "simulation.log", "stdout.log", "stderr.log", - "twitter_simulation.db", # Twitter Database - "reddit_simulation.db", # Reddit Database - "env_status.json", # Env state status file + "twitter_simulation.db", + "reddit_simulation.db", + "env_status.json", ] - - # Nhưng tệp tin có cấp quyền cần xóa (có liên quan nhật ký hoạt động actions.jsonl) + dirs_to_clean = ["twitter", "reddit"] - - # Loại bỏ các tệp không cần tới (Delete them) + for filename in files_to_delete: file_path = os.path.join(sim_dir, filename) if os.path.exists(file_path): @@ -1150,8 +1323,7 @@ class SimulationRunner: cleaned_files.append(filename) except Exception as e: errors.append(f"Failed to delete {filename}: {str(e)}") - - # Kiểm tra lại các file theo thư mục chứa action history action.jsonl files + for dir_name in dirs_to_clean: dir_path = os.path.join(sim_dir, dir_name) if os.path.exists(dir_path): @@ -1162,70 +1334,78 @@ class SimulationRunner: cleaned_files.append(f"{dir_name}/actions.jsonl") except Exception as e: errors.append(f"Failed to delete {dir_name}/actions.jsonl: {str(e)}") - - # Xóa (clear) cache nhớ run state + + # Xóa RAM cache cho simulation này if simulation_id in cls._run_states: del cls._run_states[simulation_id] - + logger.info(f"Clean up complete for simulation: {simulation_id}, Deleted files: {cleaned_files}") - + return { "success": len(errors) == 0, "cleaned_files": cleaned_files, "errors": errors if errors else None } - - # Flags Ngăn việc phải làm quá nhiều việc cho một action cleanup đã làm từ đầu hay khi gọi lần tới lệnh giống + + # Cờ chống gọi cleanup nhiều lần (atexit có thể gọi nhiều lần trong một số trường hợp) _cleanup_done = False - + @classmethod def cleanup_all_simulations(cls): """ - Dọn dẹp tất cả các tiến trình mô phỏng đang chạy - - Được gọi (call) khi đóng máy chủ, nhằm vào việc muốn các máy chủ con (child processes) bị tắt theo + Dọn dẹp tất cả simulation đang chạy — gọi khi server tắt. + + Được đăng ký bởi: + atexit.register(cls.cleanup_all_simulations) → khi Python interpreter kết thúc + signal.signal(SIGTERM, cleanup_handler) → khi nhận SIGTERM (kill server) + signal.signal(SIGINT, cleanup_handler) → khi Ctrl+C + + Thực hiện: + 1. Dừng tất cả Graph Memory Updaters + 2. Kill toàn bộ OASIS subprocess đang chạy + 3. Cập nhật state.json → status="stopped" (để frontend không hiển thị "running" sau khi restart) + 4. Cập nhật run_state.json → runner_status="stopped" + 5. Đóng tất cả file handle + 6. Clear RAM cache """ - # Nếu đã clean (dọn dẹp) thì không làm nữa if cls._cleanup_done: return cls._cleanup_done = True - - # Kiểm tra xem có gì để dọn dẹp không (tránh log rỗng khi không có tiến trình chạy) + has_processes = bool(cls._processes) has_updaters = bool(cls._graph_memory_enabled) - + if not has_processes and not has_updaters: - return # Không có gì để dọn, kết thúc - + return # Không có gì để dọn + logger.info("Cleaning up all simulation processes...") - - # Dừng tất cả cái update đồ thị nhớ (Bộ nhớ Graph) (Stop_all ghi nhận ở bên trong) + + # Dừng tất cả Graph Memory Updaters try: ZepGraphMemoryManager.stop_all() except Exception as e: logger.error(f"Failed to stop Zep Graph update daemon: {e}") cls._graph_memory_enabled.clear() - - # Tạo bản sao Dictionary dict() từ cls._processes.items để không bị hỏng List lúc lặp (iterating) + + # Tạo bản sao để tránh RuntimeError khi dict thay đổi trong khi lặp processes = list(cls._processes.items()) - + for simulation_id, process in processes: try: - if process.poll() is None: # Process (Tiến trình) Vẫn đang chạy + if process.poll() is None: # Process vẫn đang chạy logger.info(f"Terminating simulation process: {simulation_id}, pid={process.pid}") - + try: - # Áp dụng giải pháp dừng liên nền tảng (Cross-platform termination method) cls._terminate_process(process, simulation_id, timeout=5) except (ProcessLookupError, OSError): - # Trong trường hợp có thể các process này đã biến mất ở đâu đó rồi, xóa một cách bắt buộc + # Process đã biến mất → fallback terminate/kill try: process.terminate() process.wait(timeout=3) except Exception: process.kill() - - # Cập nhật run_state.json + + # Cập nhật run_state.json → status=stopped state = cls.get_run_state(simulation_id) if state: state.runner_status = RunnerStatus.STOPPED @@ -1234,8 +1414,9 @@ class SimulationRunner: state.completed_at = datetime.now().isoformat() state.error = "Server shut down, simulation terminated." cls._save_run_state(state) - - # Đồng thời cập nhật trạng thái `stopped` cho tệp (file) state.json + + # Cập nhật state.json → status="stopped" (SimulationManager's file) + # Lý do: SimulationManager không tự biết server đang tắt → phải cập nhật thủ công try: sim_dir = os.path.join(cls.RUN_STATE_DIR, simulation_id) state_file = os.path.join(sim_dir, "state.json") @@ -1252,11 +1433,11 @@ class SimulationRunner: logger.warning(f"state.json not found: {state_file}") except Exception as state_err: logger.warning(f"Failed to update state.json: {simulation_id}, error={state_err}") - + except Exception as e: logger.error(f"Failed to clean up process: {simulation_id}, error={e}") - - # Đóng tất cả tệp xử lý file handles (Log file, Errors File) + + # Đóng tất cả file handle log for simulation_id, file_handle in list(cls._stdout_files.items()): try: if file_handle: @@ -1264,7 +1445,7 @@ class SimulationRunner: except Exception: pass cls._stdout_files.clear() - + for simulation_id, file_handle in list(cls._stderr_files.items()): try: if file_handle: @@ -1272,109 +1453,105 @@ class SimulationRunner: except Exception: pass cls._stderr_files.clear() - - # Dọn dẹp trạng thái ở trong Ram Memory + + # Clear RAM cache cls._processes.clear() cls._action_queues.clear() - + logger.info("Simulation process clean up completed.") - + @classmethod def register_cleanup(cls): """ - Đăng ký một lệnh Dọn dẹp (Cleanup command) - - Trong lúc chuẩn bị khởi tạo App Flask, mình sẽ thiết lập nó sao cho gọi là máy chủ kết thúc (tắt) mọi quá trình (Simulation Process) + Đăng ký hàm dọn dẹp khi server tắt. + + Gọi 1 lần duy nhất trong app initialization (Flask app factory). + Đăng ký 3 signal handlers + atexit fallback: + SIGTERM: kill server lệnh (Linux/Mac: systemd stop, Docker stop) + SIGINT: Ctrl+C từ terminal + SIGHUP: Terminal bị đóng (Unix only) + atexit: Fallback khi signal handling không khả dụng + + Lưu ý Flask debug mode: + Werkzeug reload spawns 2 processes — chỉ đăng ký trong reloader child process + (WERKZEUG_RUN_MAIN=true). Production luôn đăng ký. """ global _cleanup_registered - + if _cleanup_registered: return - - # Flask ở trong cơ chế gỡ rối `debug`, lúc này chỉ đăng ký ứng dụng để cho thằng app.run làm (Werkzeug) - # WERKZEUG_RUN_MAIN=true Đại diện quá trình tiến trình máy chủ được nạp lại - # Nhưng nó sẽ ko apply điều này ở Production nếu app không debug + is_reloader_process = os.environ.get('WERKZEUG_RUN_MAIN') == 'true' is_debug_mode = os.environ.get('FLASK_DEBUG') == '1' or os.environ.get('WERKZEUG_RUN_MAIN') is not None - - # Trong DebugMode, chúng ta chỉ cho phép re-loader Child-process chạy. Production vẫn luôn phải chạy process này + + # Debug mode: chỉ đăng ký trong child process của Werkzeug reloader if is_debug_mode and not is_reloader_process: - _cleanup_registered = True # Check list đã lưu lại. Hủy quyền yêu cầu thêm + _cleanup_registered = True return - - # Lưu các tín hiệu để trả về Signal handling sau khi dừng Process hoàn tất + + # Lưu signal handlers gốc để gọi lại sau khi cleanup original_sigint = signal.getsignal(signal.SIGINT) original_sigterm = signal.getsignal(signal.SIGTERM) - # SIGHUP Chỉ xuất hiện trong Unix (Mac/Linux), Windows không có cái này original_sighup = None has_sighup = hasattr(signal, 'SIGHUP') if has_sighup: original_sighup = signal.getsignal(signal.SIGHUP) - + def cleanup_handler(signum=None, frame=None): - """Xử lý điều hướng tín hiệu (Signal Routing): Bắt đầu dọn tiến trình xong gởi lệnh báo (Original Processing Router)""" - # Chỉ báo nhật kí (log) nếu có tiến trình (Process) cần xử lý + """Cleanup toàn bộ simulation rồi forward signal về handler gốc.""" if cls._processes or cls._graph_memory_enabled: logger.info(f"Received signal {signum}, starting clean up...") cls.cleanup_all_simulations() - - # Gửi tín hiệu gọi hàm báo (handling functions) lúc đấy của Flask => App được tự do ngắt điện + + # Forward signal về Flask/Werkzeug handler gốc if signum == signal.SIGINT and callable(original_sigint): original_sigint(signum, frame) elif signum == signal.SIGTERM and callable(original_sigterm): original_sigterm(signum, frame) elif has_sighup and signum == signal.SIGHUP: - # SIGHUP: Được trả về khi máy chủ bị dừng (Terminal Closed) if callable(original_sighup): original_sighup(signum, frame) else: - # Mặc định hành vi: Đóng bình thường => sys.exit(0) "Tạm biệt các hành khách" sys.exit(0) else: - # Hành động ở cơ sở gốc (Root) không gọi được (SIG_DFL) => Hãy phát cảnh báo raise KeyboardInterrupt - - # Một phương án khác nếu tín hiệu đăng ký xử lý gặp khó (Fallback Option) + + # Đăng ký atexit làm fallback (nếu signal handler không chạy được) atexit.register(cls.cleanup_all_simulations) - - # Đăng ký quản lý báo tín hiệu (Chỉ riêng trong chủ / main thread có cái luồng) + + # Đăng ký signal handlers — chỉ hoạt động trong main thread try: - # SIGTERM: Tín hiệu gốc của Kill Server (Linux/Mac) signal.signal(signal.SIGTERM, cleanup_handler) - # SIGINT: Bấm lệnh Control + C / Ctrl+C signal.signal(signal.SIGINT, cleanup_handler) - # SIGHUP: Máy bị đóng (Unix OS) if has_sighup: signal.signal(signal.SIGHUP, cleanup_handler) except ValueError: - # Không ở MainThread => Sử dụng được duy nhất fallback Atexit + # Không ở main thread → chỉ dùng atexit fallback logger.warning("Failed to register signal handlers (not in main thread). Falling back to atexit.") - + _cleanup_registered = True - + @classmethod def get_running_simulations(cls) -> List[str]: - """ - Lấy danh sách tất cả các ID của các phiên mô phỏng đang hoạt động - """ + """Lấy danh sách simulation_id đang có process chạy (process.poll() is None).""" running = [] for sim_id, process in cls._processes.items(): if process.poll() is None: running.append(sim_id) return running - - # ============== Tính năng Phỏng vấn (Interview) ============== - + + # -------------------------------------------------------------------------- + # PUBLIC: Interview — Phỏng vấn agent đang chạy qua IPC + # -------------------------------------------------------------------------- + @classmethod def check_env_alive(cls, simulation_id: str) -> bool: """ - Kiểm tra xem environment còn sống không (có thể nhận lệnh Interview) + Kiểm tra xem OASIS environment có còn nhận lệnh IPC không. - Args: - simulation_id: Simulation ID - - Returns: - True Nếu environment còn sống, False nghĩa là đã đóng + Đọc env_status.json → status == "alive". + Phải kiểm tra trước khi gửi bất kỳ lệnh Interview nào. + Nếu False → raise ValueError ngay, không gửi IPC để tránh timeout. """ sim_dir = os.path.join(cls.RUN_STATE_DIR, simulation_id) if not os.path.exists(sim_dir): @@ -1386,27 +1563,27 @@ class SimulationRunner: @classmethod def get_env_status_detail(cls, simulation_id: str) -> Dict[str, Any]: """ - Lấy thông tin chi tiết về trạng thái của environment + Lấy thông tin chi tiết trạng thái IPC environment. - Args: - simulation_id: Mô phỏng ID - - Returns: - Bảng trạng thái chi tiết (Dictionary) bao gồm: status, twitter_available, reddit_available, timestamp + Returns dict: + status: "alive" hoặc "stopped" + twitter_available: True nếu Twitter environment đang active + reddit_available: True nếu Reddit environment đang active + timestamp: Thời điểm cập nhật env_status.json cuối cùng """ sim_dir = os.path.join(cls.RUN_STATE_DIR, simulation_id) status_file = os.path.join(sim_dir, "env_status.json") - + default_status = { "status": "stopped", "twitter_available": False, "reddit_available": False, "timestamp": None } - + if not os.path.exists(status_file): return default_status - + try: with open(status_file, 'r', encoding='utf-8') as f: status = json.load(f) @@ -1429,24 +1606,20 @@ class SimulationRunner: timeout: float = 60.0 ) -> Dict[str, Any]: """ - Phỏng vấn trên 1 Agent + Phỏng vấn 1 agent đang chạy qua IPC. + + Flow: + 1. check_env_alive() → nếu False → raise ValueError ngay + 2. SimulationIPCClient.send_interview() → ghi file ipc_commands/{uuid}.json + 3. Poll ipc_responses/{uuid}.json mỗi 0.5s, timeout 60s + 4. OASIS script phát hiện command → chạy ManualAction INTERVIEW → ghi response + 5. Trả về response về caller Args: - simulation_id: ID mô phỏng - agent_id: Agent ID + agent_id: ID của agent (khớp với user_id trong profile file) prompt: Câu hỏi phỏng vấn - platform: Chỉ định nền tảng (Tùy chọn/Optional) - - "twitter": Chỉ PV trên account Twitter - - "reddit": Chỉ PV trên account Reddit - - None: Phỏng vấn chéo trên cả hai nền tảng, trả về kết quả hợp lại (Nếu chạy mô phỏng nền tảng kép) - timeout: Thời gian chờ tối đa (giây) - - Returns: - Từ điền chứa kết quả PV - - Raises: - ValueError: Không có mô phỏng hoặc environment ko chạy - TimeoutError: Đang chờ phản hồi bị timeout + platform: "twitter"/"reddit"/None (None = phỏng vấn trên platform đang active) + timeout: Timeout tính bằng giây """ sim_dir = os.path.join(cls.RUN_STATE_DIR, simulation_id) if not os.path.exists(sim_dir): @@ -1482,7 +1655,7 @@ class SimulationRunner: "error": response.error, "timestamp": response.timestamp } - + @classmethod def interview_agents_batch( cls, @@ -1492,23 +1665,15 @@ class SimulationRunner: timeout: float = 120.0 ) -> Dict[str, Any]: """ - Phỏng vấn hàng loạt nhiều Agent + Phỏng vấn nhiều agent cùng lúc — gửi 1 batch command duy nhất. + + Hiệu quả hơn gọi interview_agent() N lần vì chỉ cần 1 round-trip IPC. + OASIS xử lý tất cả interviews trong batch rồi trả về kết quả gộp. Args: - simulation_id: ID mô phỏng - interviews: Danh sách nội dung phỏng vấn, mỗi phần tử (element) chứa {"agent_id": int, "prompt": str, "platform": str(tùy chọn)} - platform: Nền tảng mặc định (Nếu không chọn riêng cho từng phần tử) - - "twitter": Mặc định chỉ dùng mạng Twitter - - "reddit": Mặc định chỉ dùng mạng Reddit - - None: Phỏng vấn gộp trên cả hai nền tảng với mỗi Agent - timeout: Hết thời gian chờ (ms) (seconds) - - Returns: - Dict từ điển với các kết quả phỏng vấn hàng loạt - - Raises: - ValueError: Chưa có tiến trình chạy mô phỏng - TimeoutError: Phỏng vấn lâu quá (Timeout timeout timeout) + interviews: List[{"agent_id": int, "prompt": str, "platform": str (optional)}] + platform: Platform mặc định cho tất cả (nếu không chỉ định riêng từng phần tử) + timeout: Timeout tổng cho toàn bộ batch """ sim_dir = os.path.join(cls.RUN_STATE_DIR, simulation_id) if not os.path.exists(sim_dir): @@ -1541,7 +1706,7 @@ class SimulationRunner: "error": response.error, "timestamp": response.timestamp } - + @classmethod def interview_all_agents( cls, @@ -1551,27 +1716,18 @@ class SimulationRunner: timeout: float = 180.0 ) -> Dict[str, Any]: """ - Phỏng vấn TOÀN BỘ Agent (Phỏng vấn tổng quan - Global interview) + Phỏng vấn TẤT CẢ agent với cùng 1 câu hỏi. - Hỏi một câu hỏi với toàn bộ Agent đang có trong phiên mô phỏng hiện tại + Đọc danh sách agent từ simulation_config.json → build interviews list → + gọi interview_agents_batch(). - Args: - simulation_id: ID mô phỏng - prompt: Câu hỏi (Cho tất cả các Agent) - platform: Quyết định nền tảng (Platform decision) - - "twitter": Trỏ tới Twitter Platform - - "reddit": Trỏ tới Reddit Platform - - None: Interview kết hợp trên cả nền tảng của từng agent - timeout: Timeout - - Returns: - Kết quả của toàn thể hội đồng Agents (Lớp/Nhóm) + Dùng khi muốn biết toàn bộ quan điểm của cộng đồng về 1 chủ đề. + Timeout mặc định 180s vì số lượng agent nhiều. """ sim_dir = os.path.join(cls.RUN_STATE_DIR, simulation_id) if not os.path.exists(sim_dir): raise ValueError(f"Simulation does not exist: {simulation_id}") - # Fetch All Agents profile (Lấy thông tin agents từ phần thiết lập) config_path = os.path.join(sim_dir, "simulation_config.json") if not os.path.exists(config_path): raise ValueError(f"Simulation config not found: {simulation_id}") @@ -1583,7 +1739,6 @@ class SimulationRunner: if not agent_configs: raise ValueError(f"No Agent defined under simulation configs: {simulation_id}") - # Tập hợp danh sách phỏng vấn tất cả interviews = [] for agent_config in agent_configs: agent_id = agent_config.get("agent_id") @@ -1601,7 +1756,7 @@ class SimulationRunner: platform=platform, timeout=timeout ) - + @classmethod def close_simulation_env( cls, @@ -1609,34 +1764,32 @@ class SimulationRunner: timeout: float = 30.0 ) -> Dict[str, Any]: """ - Đóng Environment giả lập (Không phải dừng process tắt hẳn nó đi) - - Gửi lệnh ra hiệu cho Simulation ngắt bỏ tiến trình để các processes thoát ra an toàn và êm đẹp về trạng thái đang chờ nhận lệnh - - Args: - simulation_id: ID Simulation - timeout: Timeout chờ kết nối - - Returns: - Kiểu từ điển: Quá trình (Process Status execution) + Gửi lệnh đóng environment qua IPC (không kill process ngay). + + Khác với stop_simulation(): + stop_simulation() → SIGTERM ngay lập tức (brutal) + close_simulation_env() → gửi lệnh IPC để OASIS tự dọn dẹp rồi thoát (graceful) + + Dùng khi muốn kết thúc simulation sớm nhưng để OASIS lưu trạng thái trước. + Nếu timeout → trả về success=True vì environment có thể đang trong quá trình đóng. """ sim_dir = os.path.join(cls.RUN_STATE_DIR, simulation_id) if not os.path.exists(sim_dir): raise ValueError(f"Simulation does not exist: {simulation_id}") - + ipc_client = SimulationIPCClient(sim_dir) - + if not ipc_client.check_env_alive(): return { "success": True, "message": "Environment is already closed" } - + logger.info(f"Sending command to close Environment: simulation_id={simulation_id}") - + try: response = ipc_client.send_close_env(timeout=timeout) - + return { "success": response.status.value == "completed", "message": "Environment close command sent", @@ -1644,7 +1797,7 @@ class SimulationRunner: "timestamp": response.timestamp } except TimeoutError: - # Hết thời gian chờ nguyên nhân lớn nhất là vì Simulation environment đang đóng giữa chừng. + # Timeout thường xảy ra khi environment đang trong quá trình đóng → OK return { "success": True, "message": "Environment close command sent (timeout waiting for response, env might be closing)" @@ -1654,7 +1807,7 @@ class SimulationRunner: "success": False, "message": f"Failed to send close env command: {str(e)}" } - + @classmethod def _get_interview_history_from_db( cls, @@ -1663,18 +1816,24 @@ class SimulationRunner: agent_id: Optional[int] = None, limit: int = 100 ) -> List[Dict[str, Any]]: - """Lấy lịch sử phỏng vấn từ Local Database của nền tảng""" + """ + Đọc lịch sử phỏng vấn từ SQLite database của OASIS. + + OASIS lưu kết quả interview vào bảng `trace` trong database + (twitter_simulation.db hoặc reddit_simulation.db). + Truy vấn WHERE action='interview' để lọc ra các record phỏng vấn. + """ import sqlite3 - + if not os.path.exists(db_path): return [] - + results = [] - + try: conn = sqlite3.connect(db_path) cursor = conn.cursor() - + if agent_id is not None: cursor.execute(""" SELECT user_id, info, created_at @@ -1691,13 +1850,13 @@ class SimulationRunner: ORDER BY created_at DESC LIMIT ? """, (limit,)) - + for user_id, info_json, created_at in cursor.fetchall(): try: info = json.loads(info_json) if info_json else {} except json.JSONDecodeError: info = {"raw": info_json} - + results.append({ "agent_id": user_id, "response": info.get("response", info), @@ -1705,12 +1864,12 @@ class SimulationRunner: "timestamp": created_at, "platform": platform_name }) - + conn.close() - + except Exception as e: logger.error(f"Failed to load Interview history ({platform_name}): {e}") - + return results @classmethod @@ -1722,31 +1881,26 @@ class SimulationRunner: limit: int = 100 ) -> List[Dict[str, Any]]: """ - Lịch sử lấy danh sách câu trả lời của các câu hỏi với agents (Đọc từ DataBase db) - + Lấy lịch sử phỏng vấn từ database — dùng sau khi simulation đã chạy. + + Đọc từ: + twitter_simulation.db (nếu platform="twitter" hoặc None) + reddit_simulation.db (nếu platform="reddit" hoặc None) + + Kết quả sort theo timestamp giảm dần, giới hạn `limit` record. + Khi query kết hợp cả 2 platform, giới hạn tổng = limit (không phải limit×2). + Args: - simulation_id: Nhận dạng ID cho mỗi Simulation - platform: Chị định Nền tảng (reddit/twitter/None) - - "reddit": Chỉ trên reddit - - "twitter": Chỉ lấy records ghi được trên mạng xã hội twitter giả lập - - None: Kết hợp lấy logs của cả hai social network - agent_id: Cung cấp tùy chọn cho loại Agent qua ID - limit: Lượng dữ liệu load tối đa cho 1 request get query trên 1 nền tảng - - Returns: - Danh sách lưu vết record lịch sử phỏng vấn của các agents + platform: "twitter"/"reddit"/None (None = cả hai) + agent_id: Lọc theo agent cụ thể + limit: Số record tối đa trả về """ sim_dir = os.path.join(cls.RUN_STATE_DIR, simulation_id) - + results = [] - - # Xác nhận nền tảng cung cấp cho truy vấn - if platform in ("reddit", "twitter"): - platforms = [platform] - else: - # Nếu người dùng để trống, có nghĩa là gọi tất cả kết quả - platforms = ["twitter", "reddit"] - + + platforms = [platform] if platform in ("reddit", "twitter") else ["twitter", "reddit"] + for p in platforms: db_path = os.path.join(sim_dir, f"{p}_simulation.db") platform_results = cls._get_interview_history_from_db( @@ -1756,13 +1910,12 @@ class SimulationRunner: limit=limit ) results.extend(platform_results) - - # Sắp xếp chúng lại bằng thời gian gần nhất lên trước + + # Sort tổng hợp theo timestamp mới nhất results.sort(key=lambda x: x.get("timestamp", ""), reverse=True) - - # Cho trường hợp query kết hợp nhiều nền tảng, phải tiến hành gọt lấy đúng 1 giới hạn nhất định + + # Cắt về đúng limit khi query kết hợp nhiều platform if len(platforms) > 1 and len(results) > limit: results = results[:limit] - - return results + return results diff --git a/backend/app/services/zep_tools.py b/backend/app/services/zep_tools.py index 27fedc94..db1a35a5 100644 --- a/backend/app/services/zep_tools.py +++ b/backend/app/services/zep_tools.py @@ -1741,7 +1741,7 @@ Trả về định dạng JSON: {"questions": ["Câu hỏi 1", "Câu hỏi 2", . Bối cảnh mô phỏng: {simulation_requirement if simulation_requirement else "Not provided"} -Vai trò của đối tượng phỏng vấ: {', '.join(agent_roles)} +Vai trò của đối tượng phỏng vấn: {', '.join(agent_roles)} Hãy tạo từ 3-5 câu hỏi phỏng vấn.""" diff --git a/backend/app/utils/llm_client.py b/backend/app/utils/llm_client.py index 4d923cad..d1a9eb26 100644 --- a/backend/app/utils/llm_client.py +++ b/backend/app/utils/llm_client.py @@ -41,7 +41,7 @@ class LLMClient: self, messages: List[Dict[str, str]], temperature: float = 0.7, - max_tokens: int = 4096, + max_tokens: int = 16000, response_format: Optional[Dict] = None, metadata: Optional[Dict[str, Any]] = None, ) -> str: diff --git a/backend/app/utils/llm_cost.py b/backend/app/utils/llm_cost.py index 6b2e14dc..e71b3c94 100644 --- a/backend/app/utils/llm_cost.py +++ b/backend/app/utils/llm_cost.py @@ -22,6 +22,7 @@ from typing import Any, Dict, Optional MODEL_COSTS_PER_1M_TOKENS: Dict[str, Dict[str, float]] = { "Qwen/Qwen3.5-27B": {"input": 0.5, "output": 3.0}, + "Qwen/Qwen3.6-27B": {"input": 0.5, "output": 3.0}, "gemini-3-flash": {"input": 0.5, "output": 3.0}, "gemini-3.1-flash-lite": {"input": 0.25, "output": 1.5}, } diff --git a/backend/scripts/run_parallel_simulation.py b/backend/scripts/run_parallel_simulation.py index c0953f6a..ae9aa485 100644 --- a/backend/scripts/run_parallel_simulation.py +++ b/backend/scripts/run_parallel_simulation.py @@ -1,19 +1,56 @@ """ -Kịch bản mô phỏng song song hai nền tảng OASIS -Chạy đồng thời mô phỏng Twitter và Reddit, đọc cùng một tệp cấu hình +OASIS Script — Chạy simulation song song hai nền tảng Twitter + Reddit + +Vị trí trong pipeline: +───────────────────────────────────────────────────────────────────────────── + SimulationRunner.start_simulation() [Flask backend] + └─ subprocess.Popen(run_parallel_simulation.py --config ...) + │ (Script này chạy độc lập — không phải Flask process) + │ + ├── asyncio.gather(run_twitter_simulation, run_reddit_simulation) + │ └─ Hai platform chạy SONG SONG trong cùng event loop + │ + └── Sau khi xong: vào chế độ chờ lệnh IPC (Interview) +───────────────────────────────────────────────────────────────────────────── + +Luồng dữ liệu vào script: + simulation_config.json → thời gian, agent configs, initial posts + twitter_profiles.csv → hồ sơ agent Twitter (user_char = system prompt) + reddit_profiles.json → hồ sơ agent Reddit + +Luồng dữ liệu ra script: + twitter/actions.jsonl → mỗi dòng = 1 hành động của agent Twitter + reddit/actions.jsonl → mỗi dòng = 1 hành động của agent Reddit + simulation.log → stdout của script (được Flask redirect vào đây) + env_status.json → trạng thái IPC (alive/stopped) + twitter_simulation.db → SQLite database của OASIS Twitter + reddit_simulation.db → SQLite database của OASIS Reddit + +Cơ chế đọc action từ DB (quan trọng): + env.step(actions) KHÔNG trả về kết quả hữu ích — OASIS ghi action vào SQLite. + Script dùng rowid để track bản ghi đã xử lý và đọc action MỚI sau mỗi round. + Lý do dùng rowid thay vì created_at: Twitter dùng integer timestamp, + Reddit dùng datetime string — rowid là integer tự tăng, nhất quán trên cả hai. + +Hai LLM riêng biệt (tùy chọn): + Twitter → LLM_API_KEY (general) + Reddit → LLM_BOOST_API_KEY (boost, nếu có) → tăng throughput khi chạy song song + Nếu không có boost config → Reddit fallback về general LLM. Tính năng: -- Mô phỏng song song hai nền tảng (Twitter + Reddit) -- Không đóng môi trường ngay sau khi hoàn tất mô phỏng, chuyển sang chế độ chờ lệnh -- Hỗ trợ nhận lệnh Interview qua IPC -- Hỗ trợ phỏng vấn một Agent và phỏng vấn hàng loạt -- Hỗ trợ lệnh đóng môi trường từ xa +- Chạy song song hai nền tảng (Twitter + Reddit) qua asyncio.gather +- Sau simulation, KHÔNG đóng environment → vào chế độ chờ lệnh IPC +- Hỗ trợ Interview 1 agent, batch, và toàn bộ agent +- Hỗ trợ lệnh đóng environment từ xa (close_env) +- Tham số --no-wait để đóng ngay sau simulation (dùng khi test) +- Tham số --max-rounds để cắt ngắn số vòng (dùng khi test nhanh) Cách dùng: python run_parallel_simulation.py --config simulation_config.json - python run_parallel_simulation.py --config simulation_config.json --no-wait # Đóng ngay sau khi hoàn tất + python run_parallel_simulation.py --config simulation_config.json --no-wait python run_parallel_simulation.py --config simulation_config.json --twitter-only python run_parallel_simulation.py --config simulation_config.json --reddit-only + python run_parallel_simulation.py --config simulation_config.json --max-rounds 5 Cấu trúc log: sim_xxx/ @@ -21,13 +58,14 @@ Cấu trúc log: │ └── actions.jsonl # Log hành động nền tảng Twitter ├── reddit/ │ └── actions.jsonl # Log hành động nền tảng Reddit - ├── simulation.log # Log tiến trình mô phỏng chính - └── run_state.json # Trạng thái chạy (cho API truy vấn) + ├── simulation.log # Log tiến trình mô phỏng chính (stdout của script) + └── run_state.json # Trạng thái chạy (cho API truy vấn từ Flask) """ # ============================================================ -# Khắc phục vấn đề mã hóa trên Windows: đặt UTF-8 trước mọi import -# Mục tiêu là sửa lỗi thư viện OASIS bên thứ ba đọc file mà không chỉ định encoding +# Khắc phục vấn đề mã hóa trên Windows — PHẢI đặt trước mọi import +# Mục tiêu: sửa lỗi thư viện OASIS bên thứ ba đọc file không chỉ định encoding, +# gây UnicodeDecodeError khi có nội dung tiếng Việt/Unicode trong profile file. # ============================================================ import sys import os @@ -37,31 +75,28 @@ if sys.platform == 'win32': # Thiết lập này ảnh hưởng đến mọi lời gọi open() không chỉ định encoding os.environ.setdefault('PYTHONUTF8', '1') os.environ.setdefault('PYTHONIOENCODING', 'utf-8') - - # Cấu hình lại stdout/stderr sang UTF-8 (tránh lỗi hiển thị ký tự tiếng Trung trên console) + + # Cấu hình lại stdout/stderr sang UTF-8 if hasattr(sys.stdout, 'reconfigure'): sys.stdout.reconfigure(encoding='utf-8', errors='replace') if hasattr(sys.stderr, 'reconfigure'): sys.stderr.reconfigure(encoding='utf-8', errors='replace') - - # Ép đặt mã hóa mặc định (ảnh hưởng encoding mặc định của open()) - # Lưu ý: tốt nhất cần thiết lập khi Python khởi động, thiết lập lúc runtime có thể không hiệu quả - # Vì vậy cần monkey-patch thêm hàm open tích hợp + + # Monkey-patch hàm open() tích hợp để mặc định dùng UTF-8 + # Cần thiết vì PYTHONUTF8 không ảnh hưởng đến code đã chạy trước khi đặt env var. + # Lưu ý: thiết lập lúc runtime có thể không hiệu quả với một số thư viện, nhưng + # đây là phương án tốt nhất khi không kiểm soát được source code của OASIS. import builtins _original_open = builtins.open - - def _utf8_open(file, mode='r', buffering=-1, encoding=None, errors=None, + + def _utf8_open(file, mode='r', buffering=-1, encoding=None, errors=None, newline=None, closefd=True, opener=None): - """ - Wrapper cho hàm open(), mặc định dùng UTF-8 cho chế độ văn bản - Điều này giúp sửa lỗi thư viện bên thứ ba (như OASIS) đọc file không chỉ định encoding - """ - # Chỉ đặt encoding mặc định cho chế độ văn bản (không phải binary) khi chưa chỉ định encoding + """Wrapper cho hàm open(), mặc định dùng UTF-8 cho chế độ văn bản.""" if encoding is None and 'b' not in mode: encoding = 'utf-8' - return _original_open(file, mode, buffering, encoding, errors, + return _original_open(file, mode, buffering, encoding, errors, newline, closefd, opener) - + builtins.open = _utf8_open import argparse @@ -77,78 +112,85 @@ from datetime import datetime from typing import Dict, Any, List, Optional, Tuple -# Biến toàn cục dùng cho xử lý tín hiệu -_shutdown_event = None -_cleanup_done = False +# Biến toàn cục cho signal handling và shutdown coordination +_shutdown_event = None # asyncio.Event — được set khi nhận SIGTERM/SIGINT +_cleanup_done = False # Cờ chống gọi cleanup nhiều lần -# Thêm thư mục backend vào sys.path -# Script nằm cố định trong thư mục backend/scripts/ +# Thêm thư mục backend vào sys.path để import được các module nội bộ +# (action_logger, llm_cost_patch từ backend/scripts/) _scripts_dir = os.path.dirname(os.path.abspath(__file__)) _backend_dir = os.path.abspath(os.path.join(_scripts_dir, '..')) _project_root = os.path.abspath(os.path.join(_backend_dir, '..')) sys.path.insert(0, _scripts_dir) sys.path.insert(0, _backend_dir) -# Tải tệp .env ở thư mục gốc dự án (chứa các cấu hình như LLM_API_KEY) +# Tải .env ở thư mục gốc dự án (chứa LLM_API_KEY, LLM_BASE_URL, ...) +# Script chạy độc lập (subprocess) nên không thừa hưởng env từ Flask from dotenv import load_dotenv _env_file = os.path.join(_project_root, '.env') if os.path.exists(_env_file): load_dotenv(_env_file) print(f"Environment config loaded: {_env_file}") else: - # Thử tải backend/.env _backend_env = os.path.join(_backend_dir, '.env') if os.path.exists(_backend_env): load_dotenv(_backend_env) print(f"Environment config loaded: {_backend_env}") +# ============================================================================== +# Lọc cảnh báo thừa từ camel-ai về max_tokens +# ============================================================================== +# camel-ai log cảnh báo "Invalid or missing max_tokens" mỗi khi tạo LLM call. +# Đây là cảnh báo không quan trọng (chủ động không set max_tokens để model tự quyết định). +# Filter này ngăn cảnh báo đó làm ô nhiễm simulation.log. +# ============================================================================== + class MaxTokensWarningFilter(logging.Filter): - """Lọc cảnh báo max_tokens của camel-ai (chủ động không đặt max_tokens để model tự quyết định)""" - + """Lọc cảnh báo max_tokens của camel-ai.""" + def filter(self, record): - # Lọc log cảnh báo liên quan đến max_tokens if "max_tokens" in record.getMessage() and "Invalid or missing" in record.getMessage(): return False return True -# Thêm filter ngay khi module được nạp để bảo đảm có hiệu lực trước khi mã camel chạy +# Thêm filter ngay khi module được nạp — phải trước khi camel-ai chạy logging.getLogger().addFilter(MaxTokensWarningFilter()) def disable_oasis_logging(): """ - Tắt log chi tiết của thư viện OASIS - Log của OASIS quá dài dòng (ghi từng quan sát và hành động của agent), ở đây dùng action_logger riêng + Tắt log chi tiết của thư viện OASIS. + + OASIS ghi log cực kỳ dài dòng (từng quan sát + action của mỗi agent mỗi round). + MiroFish dùng action_logger riêng (ghi ra actions.jsonl) nên không cần OASIS log. + Chỉ để CRITICAL — chỉ ghi lỗi nghiêm trọng không thể bỏ qua. """ - # Tắt toàn bộ logger của OASIS oasis_loggers = [ "social.agent", - "social.twitter", + "social.twitter", "social.rec", "oasis.env", "table", ] - + for logger_name in oasis_loggers: logger = logging.getLogger(logger_name) - logger.setLevel(logging.CRITICAL) # Chỉ ghi lỗi nghiêm trọng + logger.setLevel(logging.CRITICAL) logger.handlers.clear() logger.propagate = False def init_logging_for_simulation(simulation_dir: str): """ - Khởi tạo cấu hình log cho mô phỏng - - Args: - simulation_dir: Đường dẫn thư mục mô phỏng + Khởi tạo cấu hình log cho mô phỏng. + + Dọn thư mục log/ cũ nếu còn từ lần chạy trước (cấu trúc log cũ đã deprecated). + Cấu trúc log hiện tại dùng twitter/actions.jsonl và reddit/actions.jsonl. """ - # Tắt log chi tiết của OASIS disable_oasis_logging() - - # Dọn thư mục log cũ (nếu tồn tại) + old_log_dir = os.path.join(simulation_dir, "log") if os.path.exists(old_log_dir): import shutil @@ -175,7 +217,13 @@ except ImportError as e: sys.exit(1) -# Action khả dụng trên Twitter (không gồm INTERVIEW; INTERVIEW chỉ kích hoạt thủ công qua ManualAction) +# ============================================================================== +# Action types khả dụng trên mỗi platform +# ============================================================================== +# INTERVIEW không có trong danh sách — nó chỉ được kích hoạt thủ công qua +# ManualAction (IPC command), không phải LLMAction trong vòng lặp simulation. +# ============================================================================== + TWITTER_ACTIONS = [ ActionType.CREATE_POST, ActionType.LIKE_POST, @@ -185,7 +233,6 @@ TWITTER_ACTIONS = [ ActionType.QUOTE_POST, ] -# Action khả dụng trên Reddit (không gồm INTERVIEW; INTERVIEW chỉ kích hoạt thủ công qua ManualAction) REDDIT_ACTIONS = [ ActionType.LIKE_POST, ActionType.DISLIKE_POST, @@ -203,25 +250,41 @@ REDDIT_ACTIONS = [ ] -# Hằng số liên quan đến IPC +# Tên file/folder cho cơ chế IPC IPC_COMMANDS_DIR = "ipc_commands" IPC_RESPONSES_DIR = "ipc_responses" ENV_STATUS_FILE = "env_status.json" + class CommandType: - """Hằng số loại lệnh""" + """Hằng số loại lệnh IPC""" INTERVIEW = "interview" BATCH_INTERVIEW = "batch_interview" CLOSE_ENV = "close_env" +# ============================================================================== +# CLASS: ParallelIPCHandler — Xử lý lệnh IPC trong khi environment còn sống +# ============================================================================== +# Sau khi simulation loop kết thúc, OASIS environment KHÔNG bị đóng. +# Script vào chế độ chờ — ParallelIPCHandler poll ipc_commands/ mỗi 0.5 giây. +# +# Cơ chế IPC (file-based): +# Flask ghi: ipc_commands/{uuid}.json +# {"command_id": "abc", "command_type": "interview", "args": {...}} +# Script phát hiện → xử lý → ghi: ipc_responses/{uuid}.json +# {"command_id": "abc", "status": "completed", "result": {...}} +# Flask poll ipc_responses/{uuid}.json cho đến khi có kết quả. +# Script xóa file command sau khi đã xử lý xong. +# +# Tại sao dùng file thay vì socket/queue? +# Subprocess và Flask process không chia sẻ memory. +# File là cách đơn giản nhất để giao tiếp cross-process mà không cần thêm dependency. +# ============================================================================== + class ParallelIPCHandler: - """ - Bộ xử lý lệnh IPC cho hai nền tảng - - Quản lý môi trường của cả hai nền tảng và xử lý lệnh Interview - """ - + """Bộ xử lý lệnh IPC cho hai nền tảng.""" + def __init__( self, simulation_dir: str, @@ -235,17 +298,24 @@ class ParallelIPCHandler: self.twitter_agent_graph = twitter_agent_graph self.reddit_env = reddit_env self.reddit_agent_graph = reddit_agent_graph - + self.commands_dir = os.path.join(simulation_dir, IPC_COMMANDS_DIR) self.responses_dir = os.path.join(simulation_dir, IPC_RESPONSES_DIR) self.status_file = os.path.join(simulation_dir, ENV_STATUS_FILE) - - # Đảm bảo thư mục tồn tại + os.makedirs(self.commands_dir, exist_ok=True) os.makedirs(self.responses_dir, exist_ok=True) - + def update_status(self, status: str): - """Cập nhật trạng thái môi trường""" + """ + Ghi env_status.json — Flask đọc file này để biết environment còn sống không. + + Được gọi tại 2 thời điểm: + 1. Khi vào chế độ chờ: update_status("alive") + 2. Khi thoát chế độ chờ: update_status("stopped") + + Flask's check_env_alive() đọc file này trước khi gửi bất kỳ IPC command nào. + """ with open(self.status_file, 'w', encoding='utf-8') as f: json.dump({ "status": status, @@ -253,32 +323,43 @@ class ParallelIPCHandler: "reddit_available": self.reddit_env is not None, "timestamp": datetime.now().isoformat() }, f, ensure_ascii=False, indent=2) - + def poll_command(self) -> Optional[Dict[str, Any]]: - """Poll để lấy lệnh đang chờ xử lý""" + """ + Kiểm tra xem có lệnh nào đang chờ xử lý không. + + Đọc file JSON cũ nhất trong ipc_commands/ (sort theo mtime). + Trả về None nếu không có lệnh nào, hoặc nếu file không hợp lệ. + + Flask đảm bảo chỉ ghi 1 command tại 1 thời điểm — không cần lock. + """ if not os.path.exists(self.commands_dir): return None - - # Lấy tệp lệnh (sắp xếp theo thời gian) + command_files = [] for filename in os.listdir(self.commands_dir): if filename.endswith('.json'): filepath = os.path.join(self.commands_dir, filename) command_files.append((filepath, os.path.getmtime(filepath))) - - command_files.sort(key=lambda x: x[1]) - + + command_files.sort(key=lambda x: x[1]) # Xử lý lệnh cũ nhất trước (FIFO) + for filepath, _ in command_files: try: with open(filepath, 'r', encoding='utf-8') as f: return json.load(f) except (json.JSONDecodeError, OSError): continue - + return None - + def send_response(self, command_id: str, status: str, result: Dict = None, error: str = None): - """Gửi phản hồi""" + """ + Ghi kết quả vào ipc_responses/{command_id}.json và xóa file lệnh. + + Flask poll file này (mỗi 0.5s, timeout 60s) để lấy kết quả interview. + Sau khi ghi response, xóa file command để tránh xử lý lại. + """ response = { "command_id": command_id, "status": status, @@ -286,47 +367,42 @@ class ParallelIPCHandler: "error": error, "timestamp": datetime.now().isoformat() } - + response_file = os.path.join(self.responses_dir, f"{command_id}.json") with open(response_file, 'w', encoding='utf-8') as f: json.dump(response, f, ensure_ascii=False, indent=2) - - # Xóa tệp lệnh + + # Xóa file command sau khi đã xử lý xong command_file = os.path.join(self.commands_dir, f"{command_id}.json") try: os.remove(command_file) except OSError: pass - + def _get_env_and_graph(self, platform: str): - """ - Lấy env và agent_graph của nền tảng được chỉ định - - Args: - platform: Tên nền tảng ("twitter" hoặc "reddit") - - Returns: - (env, agent_graph, platform_name) hoặc (None, None, None) - """ + """Lấy (env, agent_graph) của platform được chỉ định. Trả về (None, None, None) nếu không có.""" if platform == "twitter" and self.twitter_env: return self.twitter_env, self.twitter_agent_graph, "twitter" elif platform == "reddit" and self.reddit_env: return self.reddit_env, self.reddit_agent_graph, "reddit" else: return None, None, None - + async def _interview_single_platform(self, agent_id: int, prompt: str, platform: str) -> Dict[str, Any]: """ - Thực thi Interview trên một nền tảng - - Returns: - Dictionary chứa kết quả hoặc lỗi + Thực thi Interview trên 1 platform bằng ManualAction. + + ManualAction(INTERVIEW) là cách inject hành động thủ công vào OASIS — + thay vì để LLM quyết định action, ép agent trả lời câu hỏi cụ thể. + + OASIS ghi kết quả vào bảng `trace` trong SQLite database. + _get_interview_result() đọc bản ghi mới nhất từ đó. """ env, agent_graph, actual_platform = self._get_env_and_graph(platform) - + if not env or not agent_graph: return {"platform": platform, "error": f"{platform} platform is unavailable"} - + try: agent = agent_graph.get_agent(agent_id) interview_action = ManualAction( @@ -334,35 +410,30 @@ class ParallelIPCHandler: action_args={"prompt": prompt} ) actions = {agent: interview_action} - await env.step(actions) - + await env.step(actions) # OASIS chạy action → ghi vào DB + result = self._get_interview_result(agent_id, actual_platform) result["platform"] = actual_platform return result - + except Exception as e: return {"platform": platform, "error": str(e)} - + async def handle_interview(self, command_id: str, agent_id: int, prompt: str, platform: str = None) -> bool: """ - Xử lý lệnh phỏng vấn một Agent - - Args: - command_id: ID lệnh - agent_id: Agent ID - prompt: Câu hỏi phỏng vấn - platform: Nền tảng chỉ định (tùy chọn) - - "twitter": Chỉ phỏng vấn trên Twitter - - "reddit": Chỉ phỏng vấn trên Reddit - - None/không chỉ định: Phỏng vấn đồng thời cả hai nền tảng, trả kết quả gộp - + Xử lý lệnh phỏng vấn 1 agent. + + Nếu platform được chỉ định: phỏng vấn trên platform đó thôi. + Nếu platform=None: phỏng vấn song song cả 2 platform (asyncio.gather), + trả về kết quả gộp — hữu ích khi muốn so sánh phản ứng của agent + trên Twitter vs Reddit. + Returns: - True là thành công, False là thất bại + True nếu ít nhất 1 platform thành công, False nếu tất cả fail. """ - # Nếu có chỉ định nền tảng, chỉ phỏng vấn trên nền tảng đó if platform in ("twitter", "reddit"): - result = await self._interview_single_platform(agent_id, prompt, platform) - + result = await self._interview_single_platform(agent_id, prompt, platform) + if "error" in result: self.send_response(command_id, "failed", error=result["error"]) print(f" Interview failed: agent_id={agent_id}, platform={platform}, error={result['error']}") @@ -371,39 +442,38 @@ class ParallelIPCHandler: self.send_response(command_id, "completed", result=result) print(f" Interview completed: agent_id={agent_id}, platform={platform}") return True - - # Không chỉ định nền tảng: phỏng vấn đồng thời hai nền tảng + + # Không chỉ định platform → phỏng vấn cả 2 song song if not self.twitter_env and not self.reddit_env: self.send_response(command_id, "failed", error="No simulation environment available") return False - + results = { "agent_id": agent_id, "prompt": prompt, "platforms": {} } success_count = 0 - - # Phỏng vấn song song hai nền tảng + tasks = [] platforms_to_interview = [] - + if self.twitter_env: tasks.append(self._interview_single_platform(agent_id, prompt, "twitter")) platforms_to_interview.append("twitter") - + if self.reddit_env: tasks.append(self._interview_single_platform(agent_id, prompt, "reddit")) platforms_to_interview.append("reddit") - - # Chạy song song + + # Chạy song song — asyncio.gather không block nhau platform_results = await asyncio.gather(*tasks) - + for platform_name, platform_result in zip(platforms_to_interview, platform_results): results["platforms"][platform_name] = platform_result if "error" not in platform_result: success_count += 1 - + if success_count > 0: self.send_response(command_id, "completed", result=results) print(f" Interview completed: agent_id={agent_id}, successful platforms={success_count}/{len(platforms_to_interview)}") @@ -413,24 +483,26 @@ class ParallelIPCHandler: self.send_response(command_id, "failed", error="; ".join(errors)) print(f" Interview failed: agent_id={agent_id}, all platforms failed") return False - + async def handle_batch_interview(self, command_id: str, interviews: List[Dict], platform: str = None) -> bool: """ - Xử lý lệnh phỏng vấn hàng loạt - - Args: - command_id: ID lệnh - interviews: [{"agent_id": int, "prompt": str, "platform": str(optional)}, ...] - platform: Nền tảng mặc định (có thể bị ghi đè ở từng mục interview) - - "twitter": Chỉ phỏng vấn Twitter - - "reddit": Chỉ phỏng vấn Reddit - - None/không chỉ định: Mỗi Agent được phỏng vấn trên cả hai nền tảng + Xử lý lệnh phỏng vấn hàng loạt (batch) — hiệu quả hơn gọi N lần. + + Chiến lược tối ưu: nhóm tất cả agent cùng platform thành 1 env.step() call, + thay vì gọi env.step() riêng lẻ cho từng agent. + → Giảm số lần call asyncio, tăng throughput. + + Phân nhóm: + - platform="twitter" → twitter_interviews + - platform="reddit" → reddit_interviews + - platform=None → both_platforms_interviews → mở rộng vào cả 2 danh sách + + Mỗi nhóm được thực thi trong 1 env.step() call với dict {agent: ManualAction}. """ - # Nhóm theo nền tảng twitter_interviews = [] reddit_interviews = [] - both_platforms_interviews = [] # Cần phỏng vấn đồng thời hai nền tảng - + both_platforms_interviews = [] + for interview in interviews: item_platform = interview.get("platform", platform) if item_platform == "twitter": @@ -438,19 +510,17 @@ class ParallelIPCHandler: elif item_platform == "reddit": reddit_interviews.append(interview) else: - # Không chỉ định nền tảng: phỏng vấn cả hai nền tảng both_platforms_interviews.append(interview) - - # Tách both_platforms_interviews vào hai nền tảng + if both_platforms_interviews: if self.twitter_env: twitter_interviews.extend(both_platforms_interviews) if self.reddit_env: reddit_interviews.extend(both_platforms_interviews) - + results = {} - - # Xử lý phỏng vấn trên nền tảng Twitter + + # Xử lý Twitter batch — 1 env.step() cho tất cả agent Twitter if twitter_interviews and self.twitter_env: try: twitter_actions = {} @@ -465,10 +535,10 @@ class ParallelIPCHandler: ) except Exception as e: print(f" Warning: Cannot get Twitter Agent {agent_id}: {e}") - + if twitter_actions: await self.twitter_env.step(twitter_actions) - + for interview in twitter_interviews: agent_id = interview.get("agent_id") result = self._get_interview_result(agent_id, "twitter") @@ -476,8 +546,8 @@ class ParallelIPCHandler: results[f"twitter_{agent_id}"] = result except Exception as e: print(f" Twitter batch interview failed: {e}") - - # Xử lý phỏng vấn trên nền tảng Reddit + + # Xử lý Reddit batch — tương tự if reddit_interviews and self.reddit_env: try: reddit_actions = {} @@ -492,10 +562,10 @@ class ParallelIPCHandler: ) except Exception as e: print(f" Warning: Cannot get Reddit Agent {agent_id}: {e}") - + if reddit_actions: await self.reddit_env.step(reddit_actions) - + for interview in reddit_interviews: agent_id = interview.get("agent_id") result = self._get_interview_result(agent_id, "reddit") @@ -503,7 +573,7 @@ class ParallelIPCHandler: results[f"reddit_{agent_id}"] = result except Exception as e: print(f" Reddit batch interview failed: {e}") - + if results: self.send_response(command_id, "completed", result={ "interviews_count": len(results), @@ -514,25 +584,29 @@ class ParallelIPCHandler: else: self.send_response(command_id, "failed", error="No successful interviews") return False - + def _get_interview_result(self, agent_id: int, platform: str) -> Dict[str, Any]: - """Lấy kết quả Interview mới nhất từ cơ sở dữ liệu""" + """ + Đọc kết quả interview mới nhất từ bảng `trace` trong SQLite. + + OASIS ghi kết quả ManualAction(INTERVIEW) vào bảng trace với action='interview'. + Truy vấn ORDER BY created_at DESC LIMIT 1 → lấy bản ghi mới nhất. + """ db_path = os.path.join(self.simulation_dir, f"{platform}_simulation.db") - + result = { "agent_id": agent_id, "response": None, "timestamp": None } - + if not os.path.exists(db_path): return result - + try: conn = sqlite3.connect(db_path) cursor = conn.cursor() - - # Truy vấn bản ghi Interview mới nhất + cursor.execute(""" SELECT user_id, info, created_at FROM trace @@ -540,7 +614,7 @@ class ParallelIPCHandler: ORDER BY created_at DESC LIMIT 1 """, (ActionType.INTERVIEW.value, agent_id)) - + row = cursor.fetchone() if row: user_id, info_json, created_at = row @@ -550,31 +624,37 @@ class ParallelIPCHandler: result["timestamp"] = created_at except json.JSONDecodeError: result["response"] = info_json - + conn.close() - + except Exception as e: print(f" Failed to read interview result: {e}") - + return result - + async def process_commands(self) -> bool: """ - Xử lý tất cả lệnh đang chờ - + Xử lý 1 lệnh đang chờ (nếu có). + + Được gọi trong vòng lặp chờ mỗi 0.5 giây. + Dispatch theo command_type: + "interview" → handle_interview() + "batch_interview" → handle_batch_interview() + "close_env" → ghi response → trả về False (thoát vòng lặp) + Returns: - True để tiếp tục chạy, False để thoát + True để tiếp tục vòng lặp, False để thoát (close_env hoặc unknown) """ command = self.poll_command() if not command: return True - + command_id = command.get("command_id") command_type = command.get("command_type") args = command.get("args", {}) - + print(f"\nReceived IPC command: {command_type}, id={command_id}") - + if command_type == CommandType.INTERVIEW: await self.handle_interview( command_id, @@ -583,7 +663,7 @@ class ParallelIPCHandler: args.get("platform") ) return True - + elif command_type == CommandType.BATCH_INTERVIEW: await self.handle_batch_interview( command_id, @@ -591,27 +671,37 @@ class ParallelIPCHandler: args.get("platform") ) return True - + elif command_type == CommandType.CLOSE_ENV: print("Received close environment command") self.send_response(command_id, "completed", result={"message": "Environment will close soon"}) - return False - + return False # Thoát vòng lặp chờ + else: self.send_response(command_id, "failed", error=f"Unknown command type: {command_type}") return True def load_config(config_path: str) -> Dict[str, Any]: - """Tải tệp cấu hình""" + """Đọc simulation_config.json từ đường dẫn đã chỉ định.""" with open(config_path, 'r', encoding='utf-8') as f: return json.load(f) -# Các loại action không cốt lõi cần lọc (giá trị phân tích thấp) +# ============================================================================== +# Lọc action và ánh xạ tên +# ============================================================================== +# FILTERED_ACTIONS: các loại action không có giá trị phân tích — bỏ qua khi ghi log. +# refresh = agent "tải lại" feed (không tạo nội dung mới) +# sign_up = đăng ký tài khoản (chỉ xảy ra 1 lần lúc khởi tạo) +# ============================================================================== + FILTERED_ACTIONS = {'refresh', 'sign_up'} -# Bảng ánh xạ loại action (tên trong DB -> tên chuẩn) +# ACTION_TYPE_MAP: ánh xạ từ tên trong SQLite database của OASIS +# sang tên chuẩn viết hoa dùng trong actions.jsonl. +# OASIS lưu dạng lowercase trong DB (ví dụ: 'create_post'), +# nhưng actions.jsonl dùng UPPER_SNAKE_CASE ('CREATE_POST') để nhất quán. ACTION_TYPE_MAP = { 'create_post': 'CREATE_POST', 'like_post': 'LIKE_POST', @@ -633,25 +723,24 @@ ACTION_TYPE_MAP = { def get_agent_names_from_config(config: Dict[str, Any]) -> Dict[int, str]: """ - Lấy ánh xạ agent_id -> entity_name từ simulation_config - - Mục tiêu là hiển thị tên thực thể thật trong actions.jsonl thay vì mã như "Agent_0" - - Args: - config: Nội dung của simulation_config.json - - Returns: - Dictionary ánh xạ agent_id -> entity_name + Lấy ánh xạ agent_id → entity_name từ simulation_config.json. + + OASIS mặc định dùng tên "Agent_0", "Agent_1", ... trong DB. + Hàm này map agent_id sang tên thực thể (ví dụ: "Trần Văn An") + để actions.jsonl hiển thị tên thật thay vì "Agent_0". + + Nếu config không có entry cho agent nào, phần code gọi hàm này + sẽ fallback về tên OASIS mặc định của agent đó. """ agent_names = {} agent_configs = config.get("agent_configs", []) - + for agent_config in agent_configs: agent_id = agent_config.get("agent_id") entity_name = agent_config.get("entity_name", f"Agent_{agent_id}") if agent_id is not None: agent_names[agent_id] = entity_name - + return agent_names @@ -661,52 +750,61 @@ def fetch_new_actions_from_db( agent_names: Dict[int, str] ) -> Tuple[List[Dict[str, Any]], int]: """ - Lấy bản ghi action mới từ DB và bổ sung ngữ cảnh đầy đủ - + Đọc các action MỚI từ SQLite database kể từ last_rowid. + + Tại sao dùng rowid thay vì created_at? + Twitter lưu created_at dạng integer timestamp (Unix epoch). + Reddit lưu created_at dạng ISO 8601 string. + → So sánh/sort theo created_at không nhất quán giữa hai platform. + rowid là integer auto-increment của SQLite — đơn giản và nhất quán. + + Chiến lược: + 1. SELECT WHERE rowid > last_rowid → chỉ lấy bản ghi MỚI + 2. Cập nhật last_rowid = rowid lớn nhất trong batch này + 3. Lần tiếp theo gọi lại với last_rowid mới → không đọc lại bản ghi cũ + + Mỗi action được enrich thêm context (nội dung bài viết, tên tác giả) + bằng _enrich_action_context() để actions.jsonl có đủ thông tin. + Args: - db_path: Đường dẫn tệp cơ sở dữ liệu - last_rowid: Giá trị rowid lớn nhất đã đọc trước đó (dùng rowid thay vì created_at vì định dạng created_at khác nhau giữa nền tảng) - agent_names: Ánh xạ agent_id -> agent_name - + db_path: Đường dẫn tệp SQLite + last_rowid: rowid lớn nhất đã đọc lần trước (0 = lần đầu, đọc tất cả) + agent_names: Ánh xạ agent_id → tên thật + Returns: (actions_list, new_last_rowid) - - actions_list: Danh sách action, mỗi phần tử gồm agent_id, agent_name, action_type, action_args (có ngữ cảnh) - - new_last_rowid: Giá trị rowid lớn nhất mới """ actions = [] new_last_rowid = last_rowid - + if not os.path.exists(db_path): return actions, new_last_rowid - + try: conn = sqlite3.connect(db_path) cursor = conn.cursor() - - # Dùng rowid để theo dõi bản ghi đã xử lý (rowid là trường tự tăng tích hợp của SQLite) - # Cách này tránh vấn đề khác biệt định dạng created_at (Twitter dùng số nguyên, Reddit dùng chuỗi datetime) + cursor.execute(""" SELECT rowid, user_id, action, info FROM trace WHERE rowid > ? ORDER BY rowid ASC """, (last_rowid,)) - + for rowid, user_id, action, info_json in cursor.fetchall(): - # Cập nhật rowid lớn nhất - new_last_rowid = rowid - - # Lọc action không cốt lõi + new_last_rowid = rowid # Cập nhật cursor cho lần tiếp theo + + # Bỏ qua action không có giá trị phân tích if action in FILTERED_ACTIONS: continue - - # Parse tham số action + try: action_args = json.loads(info_json) if info_json else {} except json.JSONDecodeError: action_args = {} - - # Tinh gọn action_args, chỉ giữ trường quan trọng (giữ nguyên nội dung, không cắt) + + # Chỉ giữ các field quan trọng của action_args + # (bỏ các field nội bộ của OASIS không cần thiết cho phân tích) simplified_args = {} if 'content' in action_args: simplified_args['content'] = action_args['content'] @@ -726,24 +824,23 @@ def fetch_new_actions_from_db( simplified_args['like_id'] = action_args['like_id'] if 'dislike_id' in action_args: simplified_args['dislike_id'] = action_args['dislike_id'] - - # Chuyển tên loại action + action_type = ACTION_TYPE_MAP.get(action, action.upper()) - - # Bổ sung ngữ cảnh (nội dung bài viết, tên người dùng...) + + # Bổ sung context (nội dung bài/comment, tên tác giả) để log có nghĩa hơn _enrich_action_context(cursor, action_type, simplified_args, agent_names) - + actions.append({ 'agent_id': user_id, 'agent_name': agent_names.get(user_id, f'Agent_{user_id}'), 'action_type': action_type, 'action_args': simplified_args, }) - + conn.close() except Exception as e: print(f"Failed to read actions from database: {e}") - + return actions, new_last_rowid @@ -754,16 +851,24 @@ def _enrich_action_context( agent_names: Dict[int, str] ) -> None: """ - Bổ sung ngữ cảnh cho action (nội dung bài viết, tên người dùng...) - - Args: - cursor: DB cursor - action_type: Loại action - action_args: Tham số action (sẽ bị cập nhật) - agent_names: Ánh xạ agent_id -> agent_name + Bổ sung context vào action_args bằng cách join thêm thông tin từ DB. + + Tại sao cần enrich? + OASIS ghi action_args dạng thô: {"post_id": 42} — không có nội dung bài. + Khi đọc actions.jsonl để phân tích, cần biết agent đã like BÀI NÀO, ai viết. + Enrich thêm post_content và post_author_name để log có ý nghĩa. + + Các loại được enrich: + - LIKE/DISLIKE_POST → thêm nội dung bài + tên tác giả + - REPOST → thêm nội dung + tác giả bài gốc + - QUOTE_POST → thêm nội dung bài gốc + phần quote của agent + - FOLLOW/MUTE → thêm tên người dùng được follow/mute + - LIKE/DISLIKE_COMMENT → thêm nội dung comment + tên tác giả + - CREATE_COMMENT → thêm nội dung bài được comment vào + + Lỗi trong enrich KHÔNG ngăn luồng chính — silently pass. """ try: - # Like/dislike bài viết: bổ sung nội dung bài và tác giả if action_type in ('LIKE_POST', 'DISLIKE_POST'): post_id = action_args.get('post_id') if post_id: @@ -771,12 +876,10 @@ def _enrich_action_context( if post_info: action_args['post_content'] = post_info.get('content', '') action_args['post_author_name'] = post_info.get('author_name', '') - - # Repost: bổ sung nội dung và tác giả bài gốc + elif action_type == 'REPOST': new_post_id = action_args.get('new_post_id') if new_post_id: - # original_post_id của bài repost trỏ đến bài gốc cursor.execute(""" SELECT original_post_id FROM post WHERE post_id = ? """, (new_post_id,)) @@ -787,19 +890,17 @@ def _enrich_action_context( if original_info: action_args['original_content'] = original_info.get('content', '') action_args['original_author_name'] = original_info.get('author_name', '') - - # Quote post: bổ sung nội dung bài gốc, tác giả và phần quote + elif action_type == 'QUOTE_POST': quoted_id = action_args.get('quoted_id') new_post_id = action_args.get('new_post_id') - + if quoted_id: original_info = _get_post_info(cursor, quoted_id, agent_names) if original_info: action_args['original_content'] = original_info.get('content', '') action_args['original_author_name'] = original_info.get('author_name', '') - - # Lấy nội dung quote của bài trích dẫn (quote_content) + if new_post_id: cursor.execute(""" SELECT quote_content FROM post WHERE post_id = ? @@ -807,12 +908,10 @@ def _enrich_action_context( row = cursor.fetchone() if row and row[0]: action_args['quote_content'] = row[0] - - # Follow user: bổ sung tên người dùng được follow + elif action_type == 'FOLLOW': follow_id = action_args.get('follow_id') if follow_id: - # Lấy followee_id từ bảng follow cursor.execute(""" SELECT followee_id FROM follow WHERE follow_id = ? """, (follow_id,)) @@ -822,17 +921,14 @@ def _enrich_action_context( target_name = _get_user_name(cursor, followee_id, agent_names) if target_name: action_args['target_user_name'] = target_name - - # Mute user: bổ sung tên người dùng bị mute + elif action_type == 'MUTE': - # Lấy user_id hoặc target_id từ action_args target_id = action_args.get('user_id') or action_args.get('target_id') if target_id: target_name = _get_user_name(cursor, target_id, agent_names) if target_name: action_args['target_user_name'] = target_name - - # Like/dislike comment: bổ sung nội dung comment và tác giả + elif action_type in ('LIKE_COMMENT', 'DISLIKE_COMMENT'): comment_id = action_args.get('comment_id') if comment_id: @@ -840,8 +936,7 @@ def _enrich_action_context( if comment_info: action_args['comment_content'] = comment_info.get('content', '') action_args['comment_author_name'] = comment_info.get('author_name', '') - - # Create comment: bổ sung thông tin bài viết được bình luận + elif action_type == 'CREATE_COMMENT': post_id = action_args.get('post_id') if post_id: @@ -849,10 +944,9 @@ def _enrich_action_context( if post_info: action_args['post_content'] = post_info.get('content', '') action_args['post_author_name'] = post_info.get('author_name', '') - + except Exception as e: - # Bổ sung ngữ cảnh thất bại không ảnh hưởng luồng chính - print(f"Failed to enrich action context: {e}") + pass # Enrich thất bại không ảnh hưởng luồng chính def _get_post_info( @@ -860,17 +954,7 @@ def _get_post_info( post_id: int, agent_names: Dict[int, str] ) -> Optional[Dict[str, str]]: - """ - Lấy thông tin bài viết - - Args: - cursor: DB cursor - post_id: Post ID - agent_names: Ánh xạ agent_id -> agent_name - - Returns: - Dictionary chứa content và author_name, hoặc None - """ + """Lấy {content, author_name} của một bài viết từ DB.""" try: cursor.execute(""" SELECT p.content, p.user_id, u.agent_id @@ -883,18 +967,16 @@ def _get_post_info( content = row[0] or '' user_id = row[1] agent_id = row[2] - - # Ưu tiên dùng tên từ agent_names + author_name = '' if agent_id is not None and agent_id in agent_names: author_name = agent_names[agent_id] elif user_id: - # Lấy tên từ bảng user cursor.execute("SELECT name, user_name FROM user WHERE user_id = ?", (user_id,)) user_row = cursor.fetchone() if user_row: author_name = user_row[0] or user_row[1] or '' - + return {'content': content, 'author_name': author_name} except Exception: pass @@ -906,17 +988,7 @@ def _get_user_name( user_id: int, agent_names: Dict[int, str] ) -> Optional[str]: - """ - Lấy tên người dùng - - Args: - cursor: DB cursor - user_id: User ID - agent_names: Ánh xạ agent_id -> agent_name - - Returns: - Tên người dùng, hoặc None - """ + """Lấy tên hiển thị của một user từ DB. Ưu tiên entity_name từ agent_names.""" try: cursor.execute(""" SELECT agent_id, name, user_name FROM user WHERE user_id = ? @@ -926,8 +998,7 @@ def _get_user_name( agent_id = row[0] name = row[1] user_name = row[2] - - # Ưu tiên dùng tên từ agent_names + if agent_id is not None and agent_id in agent_names: return agent_names[agent_id] return name or user_name or '' @@ -941,17 +1012,7 @@ def _get_comment_info( comment_id: int, agent_names: Dict[int, str] ) -> Optional[Dict[str, str]]: - """ - Lấy thông tin bình luận - - Args: - cursor: DB cursor - comment_id: Comment ID - agent_names: Ánh xạ agent_id -> agent_name - - Returns: - Dictionary chứa content và author_name, hoặc None - """ + """Lấy {content, author_name} của một comment từ DB.""" try: cursor.execute(""" SELECT c.content, c.user_id, u.agent_id @@ -964,18 +1025,16 @@ def _get_comment_info( content = row[0] or '' user_id = row[1] agent_id = row[2] - - # Ưu tiên dùng tên từ agent_names + author_name = '' if agent_id is not None and agent_id in agent_names: author_name = agent_names[agent_id] elif user_id: - # Lấy tên từ bảng user cursor.execute("SELECT name, user_name FROM user WHERE user_id = ?", (user_id,)) user_row = cursor.fetchone() if user_row: author_name = user_row[0] or user_row[1] or '' - + return {'content': content, 'author_name': author_name} except Exception: pass @@ -984,17 +1043,23 @@ def _get_comment_info( def create_model(config: Dict[str, Any], use_boost: bool = False): """ - Tạo mô hình LLM - - Hỗ trợ cấu hình hai LLM để tăng tốc khi mô phỏng song song: - - Cấu hình chung: LLM_API_KEY, LLM_BASE_URL, LLM_MODEL_NAME - - Cấu hình tăng tốc (tùy chọn): LLM_BOOST_API_KEY, LLM_BOOST_BASE_URL, LLM_BOOST_MODEL_NAME - - Nếu có cấu hình LLM tăng tốc, mỗi nền tảng có thể dùng nhà cung cấp API khác nhau để tăng khả năng song song. - - Args: - config: Dictionary cấu hình mô phỏng - use_boost: Có dùng cấu hình LLM tăng tốc hay không (nếu khả dụng) + Tạo LLM model cho OASIS agent. + + Hỗ trợ 2 cấu hình LLM để tối ưu throughput khi chạy song song: + - General LLM: LLM_API_KEY + LLM_BASE_URL + LLM_MODEL_NAME + - Boost LLM: LLM_BOOST_API_KEY + LLM_BOOST_BASE_URL + LLM_BOOST_MODEL_NAME + + Twitter dùng General LLM (use_boost=False). + Reddit dùng Boost LLM nếu có (use_boost=True), fallback về General nếu không. + + Lý do tách hai LLM: khi Twitter và Reddit chạy asyncio.gather song song, + nếu cùng dùng 1 API key sẽ bị rate limit. Dùng 2 API key khác nhau (hoặc + 2 provider khác nhau) giúp tăng throughput gấp đôi. + + Validation: + - API key không được là placeholder ("your_api_key_here") + - Base URL phải bắt đầu bằng http:// hoặc https:// + - Nếu boost config không hợp lệ → in cảnh báo → fallback về general """ def _is_placeholder(value: str) -> bool: v = (value or "").strip().lower() @@ -1004,7 +1069,6 @@ def create_model(config: Dict[str, Any], use_boost: bool = False): v = (value or "").strip().lower() return v.startswith("http://") or v.startswith("https://") - # Kiểm tra có cấu hình tăng tốc hợp lệ không boost_api_key = os.environ.get("LLM_BOOST_API_KEY", "") boost_base_url = os.environ.get("LLM_BOOST_BASE_URL", "") boost_model = os.environ.get("LLM_BOOST_MODEL_NAME", "") @@ -1017,43 +1081,38 @@ def create_model(config: Dict[str, Any], use_boost: bool = False): and not _is_placeholder(boost_model) and _is_http_url(boost_base_url) ) - - # Chọn LLM theo tham số và trạng thái cấu hình + if use_boost and has_boost_config: - # Dùng cấu hình tăng tốc llm_api_key = boost_api_key llm_base_url = boost_base_url llm_model = boost_model or os.environ.get("LLM_MODEL_NAME", "") config_label = "[Boost LLM]" else: - # Dùng cấu hình chung if use_boost and not has_boost_config: print("[Boost LLM] Invalid or placeholder boost config, fallback to [General LLM].") llm_api_key = os.environ.get("LLM_API_KEY", "") llm_base_url = os.environ.get("LLM_BASE_URL", "") llm_model = os.environ.get("LLM_MODEL_NAME", "") config_label = "[General LLM]" - - # Nếu .env không có model name, dùng config làm phương án dự phòng + if not llm_model: llm_model = config.get("llm_model", "gpt-4o-mini") - - # Thiết lập biến môi trường cần thiết cho camel-ai + if llm_api_key: os.environ["OPENAI_API_KEY"] = llm_api_key - + if not os.environ.get("OPENAI_API_KEY"): raise ValueError("Missing API key config. Please set LLM_API_KEY in the project root .env file") - + if llm_base_url: if not _is_http_url(llm_base_url): raise ValueError( f"Invalid LLM base URL: {llm_base_url}. It must start with http:// or https://" ) os.environ["OPENAI_API_BASE_URL"] = llm_base_url - + print(f"{config_label} model={llm_model}, base_url={llm_base_url[:40] if llm_base_url else 'default'}...") - + return ModelFactory.create( model_platform=ModelPlatformType.OPENAI, model_type=llm_model, @@ -1067,42 +1126,68 @@ def get_active_agents_for_round( current_hour: int, round_num: int ) -> List: - """Quyết định Agent nào được kích hoạt trong round hiện tại dựa trên thời gian và cấu hình""" + """ + Quyết định agent nào được kích hoạt trong round hiện tại. + + Thuật toán 3 bước: + + Bước 1 — Tính số lượng agent cần kích hoạt: + target_count = random(base_min, base_max) × activity_multiplier + Trong đó: + peak_hours → multiplier = 1.5 (đông nhất, ví dụ 19-22h) + off_peak_hours → multiplier = 0.05 (vắng nhất, ví dụ 0-5h) + giờ khác → multiplier = 1.0 + + Bước 2 — Lọc agent ứng cử viên: + Với mỗi agent trong agent_configs: + - current_hour ∈ active_hours? → agent có thể hoạt động giờ này + - random() < activity_level? → agent "thức dậy" trong round này + → Chỉ agent thỏa mãn CẢ HAI điều kiện mới vào candidates + + Bước 3 — Chọn ngẫu nhiên: + random.sample(candidates, min(target_count, len(candidates))) + → Đảm bảo không chọn nhiều hơn số candidates thực tế + + Kết quả: List[(agent_id, agent_object)] — agent_object để truyền vào env.step() + """ time_config = config.get("time_config", {}) agent_configs = config.get("agent_configs", []) - + base_min = time_config.get("agents_per_hour_min", 5) base_max = time_config.get("agents_per_hour_max", 20) - + peak_hours = time_config.get("peak_hours", [9, 10, 11, 14, 15, 20, 21, 22]) off_peak_hours = time_config.get("off_peak_hours", [0, 1, 2, 3, 4, 5]) - + if current_hour in peak_hours: multiplier = time_config.get("peak_activity_multiplier", 1.5) elif current_hour in off_peak_hours: multiplier = time_config.get("off_peak_activity_multiplier", 0.3) else: multiplier = 1.0 - + target_count = int(random.uniform(base_min, base_max) * multiplier) - + + # Lọc ứng cử viên: đúng giờ + random activity_level candidates = [] for cfg in agent_configs: agent_id = cfg.get("agent_id", 0) active_hours = cfg.get("active_hours", list(range(8, 23))) activity_level = cfg.get("activity_level", 0.5) - + if current_hour not in active_hours: continue - + if random.random() < activity_level: candidates.append(agent_id) - + + # Chọn ngẫu nhiên từ danh sách ứng cử viên selected_ids = random.sample( - candidates, + candidates, min(target_count, len(candidates)) ) if candidates else [] - + + # Lấy agent object từ env (để truyền vào env.step() sau đó) active_agents = [] for agent_id in selected_ids: try: @@ -1110,96 +1195,125 @@ def get_active_agents_for_round( active_agents.append((agent_id, agent)) except Exception: pass - + return active_agents class PlatformSimulation: - """Container kết quả mô phỏng theo nền tảng""" + """ + Container kết quả của một platform simulation. + + Được trả về bởi run_twitter_simulation() và run_reddit_simulation(). + env và agent_graph được giữ lại sau khi vòng lặp kết thúc + để ParallelIPCHandler có thể dùng cho Interview. + """ def __init__(self): - self.env = None - self.agent_graph = None - self.total_actions = 0 + self.env = None # OASIS environment object + self.agent_graph = None # Agent graph (dùng để get_agent(id)) + self.total_actions = 0 # Tổng số action đã ghi được async def run_twitter_simulation( - config: Dict[str, Any], + config: Dict[str, Any], simulation_dir: str, action_logger: Optional[PlatformActionLogger] = None, main_logger: Optional[SimulationLogManager] = None, max_rounds: Optional[int] = None ) -> PlatformSimulation: - """Chạy mô phỏng Twitter - + """ + Khởi tạo và chạy simulation Twitter. + + Luồng chính: + 1. Tạo LLM model (General LLM) + 2. generate_twitter_agent_graph() — tải agent từ twitter_profiles.csv + 3. oasis.make() — tạo Twitter environment + SQLite DB + 4. env.reset() — khởi tạo environment + 5. Round 0: đăng initial_posts (ManualAction) + 6. Vòng lặp Round 1..N: + a. get_active_agents_for_round() — chọn agent dựa trên giờ + activity_level + b. env.step({agent: LLMAction()}) — OASIS cho agent LLM chọn action + c. fetch_new_actions_from_db() — đọc action thực tế từ DB + d. action_logger.log_action() — ghi vào twitter/actions.jsonl + 7. KHÔNG đóng env → env được giữ cho Interview + + Cơ chế đọc action từ DB (thay vì từ return value của env.step()): + env.step() không trả về action_args đủ chi tiết. + Sau mỗi round, đọc từ DB với last_rowid để lấy action MỚI. + last_rowid được cập nhật sau mỗi lần đọc → không đọc lại bản ghi cũ. + + semaphore=3 trong oasis.make(): giới hạn số LLM call đồng thời trong 1 round + → tránh bị rate limit API khi nhiều agent cùng gọi LLM. + Args: - config: Cấu hình mô phỏng - simulation_dir: Thư mục mô phỏng - action_logger: Logger hành động - main_logger: Trình quản lý log chính - max_rounds: Số round tối đa (tùy chọn, dùng để cắt ngắn mô phỏng quá dài) - + config: Nội dung simulation_config.json + simulation_dir: Thư mục chứa profile files và sẽ chứa output + action_logger: Logger ghi ra twitter/actions.jsonl + main_logger: Logger ghi ra simulation.log + max_rounds: Giới hạn số round (None = chạy đủ theo config) + Returns: - PlatformSimulation: Đối tượng kết quả chứa env và agent_graph + PlatformSimulation với env và agent_graph còn sống """ result = PlatformSimulation() - + def log_info(msg): if main_logger: main_logger.info(f"[Twitter] {msg}") print(f"[Twitter] {msg}") - + log_info("Initializing...") - - # Twitter dùng cấu hình LLM chung + + # Twitter dùng General LLM (không dùng boost) model = create_model(config, use_boost=False) - - # OASIS Twitter dùng định dạng CSV + profile_path = os.path.join(simulation_dir, "twitter_profiles.csv") if not os.path.exists(profile_path): log_info(f"Error: Profile file not found: {profile_path}") return result - + result.agent_graph = await generate_twitter_agent_graph( profile_path=profile_path, model=model, available_actions=TWITTER_ACTIONS, ) - - # Lấy ánh xạ tên thật của Agent từ config (dùng entity_name thay vì Agent_X mặc định) + + # Xây dựng agent_names map — dùng cho log và enrich context agent_names = get_agent_names_from_config(config) - # Nếu config không có Agent nào đó thì dùng tên mặc định của OASIS for agent_id, agent in result.agent_graph.get_agents(): if agent_id not in agent_names: agent_names[agent_id] = getattr(agent, 'name', f'Agent_{agent_id}') - + + # Xóa DB cũ nếu còn từ lần chạy trước (cleanup_simulation_logs() nên đã xóa rồi, + # nhưng xóa lần nữa ở đây để chắc chắn) db_path = os.path.join(simulation_dir, "twitter_simulation.db") if os.path.exists(db_path): os.remove(db_path) - + result.env = oasis.make( agent_graph=result.agent_graph, platform=oasis.DefaultPlatformType.TWITTER, database_path=db_path, - semaphore=3, # Giới hạn số request LLM đồng thời để tránh quá tải API + semaphore=3, # Tối đa 3 LLM call đồng thời trong cùng 1 round ) - + await result.env.reset() log_info("Environment started") - + if action_logger: action_logger.log_simulation_start(config) - + total_actions = 0 - last_rowid = 0 # Theo dõi row đã xử lý cuối cùng trong DB (dùng rowid để tránh khác biệt định dạng created_at) - - # Thực thi sự kiện khởi tạo + last_rowid = 0 # Theo dõi rowid đã đọc → chỉ đọc bản ghi MỚI mỗi round + + # ── Round 0: Đăng initial_posts ─────────────────────────────────────────── + # Round 0 là round đặc biệt — không phải LLMAction mà là ManualAction. + # Các bài đăng này tạo ra "tin nóng đầu tiên" để agent phản ứng từ round 1. event_config = config.get("event_config", {}) initial_posts = event_config.get("initial_posts", []) - - # Ghi log bắt đầu round 0 (giai đoạn sự kiện khởi tạo) + if action_logger: - action_logger.log_round_start(0, 0) # round 0, simulated_hour 0 - + action_logger.log_round_start(0, 0) + initial_action_count = 0 if initial_posts: initial_actions = {} @@ -1212,7 +1326,7 @@ async def run_twitter_simulation( action_type=ActionType.CREATE_POST, action_args={"content": content} ) - + if action_logger: action_logger.log_action( round_num=0, @@ -1225,63 +1339,61 @@ async def run_twitter_simulation( initial_action_count += 1 except Exception: pass - + if initial_actions: await result.env.step(initial_actions) log_info(f"Published {len(initial_actions)} initial posts") - - # Ghi log kết thúc round 0 + if action_logger: action_logger.log_round_end(0, initial_action_count) - - # Vòng lặp mô phỏng chính + + # ── Vòng lặp Round 1..N ─────────────────────────────────────────────────── time_config = config.get("time_config", {}) total_hours = time_config.get("total_simulation_hours", 72) minutes_per_round = time_config.get("minutes_per_round", 30) total_rounds = (total_hours * 60) // minutes_per_round - - # Nếu chỉ định max rounds thì cắt ngắn + if max_rounds is not None and max_rounds > 0: original_rounds = total_rounds total_rounds = min(total_rounds, max_rounds) if total_rounds < original_rounds: log_info(f"Rounds truncated: {original_rounds} -> {total_rounds} (max_rounds={max_rounds})") - + start_time = datetime.now() - + for round_num in range(total_rounds): - # Kiểm tra có nhận tín hiệu thoát không + # Kiểm tra tín hiệu thoát (SIGTERM/SIGINT) — thoát gracefully if _shutdown_event and _shutdown_event.is_set(): if main_logger: main_logger.info(f"Received shutdown signal, stop simulation at round {round_num + 1}") break - + simulated_minutes = round_num * minutes_per_round - simulated_hour = (simulated_minutes // 60) % 24 + simulated_hour = (simulated_minutes // 60) % 24 # 0-23 (giờ trong ngày) simulated_day = simulated_minutes // (60 * 24) + 1 - + active_agents = get_active_agents_for_round( result.env, config, simulated_hour, round_num ) - - # Dù có Agent hoạt động hay không, vẫn ghi log bắt đầu round + + # Log round_start dù không có agent (giúp monitor thread track round đúng) if action_logger: action_logger.log_round_start(round_num + 1, simulated_hour) - + if not active_agents: - # Không có Agent hoạt động thì vẫn ghi log kết thúc round (actions_count=0) if action_logger: action_logger.log_round_end(round_num + 1, 0) continue - + + # Cho các agent chọn action (LLMAction = OASIS tự quyết định qua LLM) actions = {agent: LLMAction() for _, agent in active_agents} await result.env.step(actions) - - # Lấy action thực tế đã chạy từ DB và ghi log + + # Đọc action thực tế từ DB (KHÔNG dùng return value của env.step()) actual_actions, last_rowid = fetch_new_actions_from_db( db_path, last_rowid, agent_names ) - + round_action_count = 0 for action_data in actual_actions: if action_logger: @@ -1294,103 +1406,102 @@ async def run_twitter_simulation( ) total_actions += 1 round_action_count += 1 - + if action_logger: action_logger.log_round_end(round_num + 1, round_action_count) - + if (round_num + 1) % 20 == 0: progress = (round_num + 1) / total_rounds * 100 log_info(f"Day {simulated_day}, {simulated_hour:02d}:00 - Round {round_num + 1}/{total_rounds} ({progress:.1f}%)") - - # Lưu ý: Không đóng environment, giữ lại để dùng cho Interview - + + # KHÔNG đóng env ở đây — giữ lại cho Interview IPC + if action_logger: action_logger.log_simulation_end(total_rounds, total_actions) - + result.total_actions = total_actions elapsed = (datetime.now() - start_time).total_seconds() log_info(f"Simulation loop completed! Elapsed: {elapsed:.1f}s, total actions: {total_actions}") - + return result async def run_reddit_simulation( - config: Dict[str, Any], + config: Dict[str, Any], simulation_dir: str, action_logger: Optional[PlatformActionLogger] = None, main_logger: Optional[SimulationLogManager] = None, max_rounds: Optional[int] = None ) -> PlatformSimulation: - """Chạy mô phỏng Reddit - - Args: - config: Cấu hình mô phỏng - simulation_dir: Thư mục mô phỏng - action_logger: Logger hành động - main_logger: Trình quản lý log chính - max_rounds: Số round tối đa (tùy chọn, dùng để cắt ngắn mô phỏng quá dài) - - Returns: - PlatformSimulation: Đối tượng kết quả chứa env và agent_graph + """ + Khởi tạo và chạy simulation Reddit. + + Tương tự run_twitter_simulation() với 2 điểm khác biệt: + + 1. Dùng Boost LLM (use_boost=True): + Reddit và Twitter chạy asyncio.gather song song. + Dùng API key khác nhau giúp tránh rate limit và tăng throughput. + + 2. Xử lý initial_posts phức tạp hơn: + OASIS Reddit cho phép 1 agent đăng nhiều bài trong initial_posts. + Nếu cùng agent được assign nhiều initial_posts, initial_actions[agent] + trở thành List[ManualAction] thay vì ManualAction đơn lẻ. + (Twitter không hỗ trợ điều này — chỉ 1 action/agent/step) """ result = PlatformSimulation() - + def log_info(msg): if main_logger: main_logger.info(f"[Reddit] {msg}") print(f"[Reddit] {msg}") - + log_info("Initializing...") - - # Reddit dùng cấu hình LLM tăng tốc (nếu có, nếu không thì fallback về cấu hình chung) + + # Reddit dùng Boost LLM (fallback về General nếu không có config boost) model = create_model(config, use_boost=True) - + profile_path = os.path.join(simulation_dir, "reddit_profiles.json") if not os.path.exists(profile_path): log_info(f"Error: Profile file not found: {profile_path}") return result - + result.agent_graph = await generate_reddit_agent_graph( profile_path=profile_path, model=model, available_actions=REDDIT_ACTIONS, ) - - # Lấy ánh xạ tên thật của Agent từ config (dùng entity_name thay vì Agent_X mặc định) + agent_names = get_agent_names_from_config(config) - # Nếu config không có Agent nào đó thì dùng tên mặc định của OASIS for agent_id, agent in result.agent_graph.get_agents(): if agent_id not in agent_names: agent_names[agent_id] = getattr(agent, 'name', f'Agent_{agent_id}') - + db_path = os.path.join(simulation_dir, "reddit_simulation.db") if os.path.exists(db_path): os.remove(db_path) - + result.env = oasis.make( agent_graph=result.agent_graph, platform=oasis.DefaultPlatformType.REDDIT, database_path=db_path, - semaphore=3, # Giới hạn số request LLM đồng thời để tránh quá tải API + semaphore=3, ) - + await result.env.reset() log_info("Environment started") - + if action_logger: action_logger.log_simulation_start(config) - + total_actions = 0 - last_rowid = 0 # Theo dõi row đã xử lý cuối cùng trong DB (dùng rowid để tránh khác biệt định dạng created_at) - - # Thực thi sự kiện khởi tạo + last_rowid = 0 + event_config = config.get("event_config", {}) initial_posts = event_config.get("initial_posts", []) - - # Ghi log bắt đầu round 0 (giai đoạn sự kiện khởi tạo) + if action_logger: - action_logger.log_round_start(0, 0) # round 0, simulated_hour 0 - + action_logger.log_round_start(0, 0) + initial_action_count = 0 if initial_posts: initial_actions = {} @@ -1399,6 +1510,7 @@ async def run_reddit_simulation( content = post.get("content", "") try: agent = result.env.agent_graph.get_agent(agent_id) + # Reddit cho phép nhiều ManualAction/agent → dùng List nếu cần if agent in initial_actions: if not isinstance(initial_actions[agent], list): initial_actions[agent] = [initial_actions[agent]] @@ -1411,7 +1523,7 @@ async def run_reddit_simulation( action_type=ActionType.CREATE_POST, action_args={"content": content} ) - + if action_logger: action_logger.log_action( round_num=0, @@ -1424,63 +1536,56 @@ async def run_reddit_simulation( initial_action_count += 1 except Exception: pass - + if initial_actions: await result.env.step(initial_actions) log_info(f"Published {len(initial_actions)} initial posts") - - # Ghi log kết thúc round 0 + if action_logger: action_logger.log_round_end(0, initial_action_count) - - # Vòng lặp mô phỏng chính + time_config = config.get("time_config", {}) total_hours = time_config.get("total_simulation_hours", 72) minutes_per_round = time_config.get("minutes_per_round", 30) total_rounds = (total_hours * 60) // minutes_per_round - - # Nếu chỉ định max rounds thì cắt ngắn + if max_rounds is not None and max_rounds > 0: original_rounds = total_rounds total_rounds = min(total_rounds, max_rounds) if total_rounds < original_rounds: log_info(f"Rounds truncated: {original_rounds} -> {total_rounds} (max_rounds={max_rounds})") - + start_time = datetime.now() - + for round_num in range(total_rounds): - # Kiểm tra có nhận tín hiệu thoát không if _shutdown_event and _shutdown_event.is_set(): if main_logger: main_logger.info(f"Received shutdown signal, stop simulation at round {round_num + 1}") break - + simulated_minutes = round_num * minutes_per_round simulated_hour = (simulated_minutes // 60) % 24 simulated_day = simulated_minutes // (60 * 24) + 1 - + active_agents = get_active_agents_for_round( result.env, config, simulated_hour, round_num ) - - # Dù có Agent hoạt động hay không, vẫn ghi log bắt đầu round + if action_logger: action_logger.log_round_start(round_num + 1, simulated_hour) - + if not active_agents: - # Không có Agent hoạt động thì vẫn ghi log kết thúc round (actions_count=0) if action_logger: action_logger.log_round_end(round_num + 1, 0) continue - + actions = {agent: LLMAction() for _, agent in active_agents} await result.env.step(actions) - - # Lấy action thực tế đã chạy từ DB và ghi log + actual_actions, last_rowid = fetch_new_actions_from_db( db_path, last_rowid, agent_names ) - + round_action_count = 0 for action_data in actual_actions: if action_logger: @@ -1493,69 +1598,53 @@ async def run_reddit_simulation( ) total_actions += 1 round_action_count += 1 - + if action_logger: action_logger.log_round_end(round_num + 1, round_action_count) - + if (round_num + 1) % 20 == 0: progress = (round_num + 1) / total_rounds * 100 log_info(f"Day {simulated_day}, {simulated_hour:02d}:00 - Round {round_num + 1}/{total_rounds} ({progress:.1f}%)") - - # Lưu ý: Không đóng environment, giữ lại để dùng cho Interview - + + # KHÔNG đóng env — giữ lại cho Interview + if action_logger: action_logger.log_simulation_end(total_rounds, total_actions) - + result.total_actions = total_actions elapsed = (datetime.now() - start_time).total_seconds() log_info(f"Simulation loop completed! Elapsed: {elapsed:.1f}s, total actions: {total_actions}") - + return result async def main(): parser = argparse.ArgumentParser(description='OASIS dual-platform parallel simulation') - parser.add_argument( - '--config', - type=str, - required=True, - help='Path to config file (simulation_config.json)' - ) - parser.add_argument( - '--twitter-only', - action='store_true', - help='Run Twitter simulation only' - ) - parser.add_argument( - '--reddit-only', - action='store_true', - help='Run Reddit simulation only' - ) - parser.add_argument( - '--max-rounds', - type=int, - default=None, - help='Maximum simulation rounds (optional, used to truncate long simulations)' - ) - parser.add_argument( - '--no-wait', - action='store_true', - default=False, - help='Close environment immediately after simulation, do not enter command wait mode' - ) - + parser.add_argument('--config', type=str, required=True, + help='Path to config file (simulation_config.json)') + parser.add_argument('--twitter-only', action='store_true', + help='Run Twitter simulation only') + parser.add_argument('--reddit-only', action='store_true', + help='Run Reddit simulation only') + parser.add_argument('--max-rounds', type=int, default=None, + help='Maximum simulation rounds (optional, used to truncate long simulations)') + parser.add_argument('--no-wait', action='store_true', default=False, + help='Close environment immediately after simulation, do not enter command wait mode') + args = parser.parse_args() - - # Tạo shutdown event khi vào main để toàn bộ chương trình có thể phản hồi tín hiệu thoát + + # Khởi tạo shutdown event — dùng để phối hợp graceful shutdown global _shutdown_event _shutdown_event = asyncio.Event() - + if not os.path.exists(args.config): print(f"Error: Config file not found: {args.config}") sys.exit(1) - + config = load_config(args.config) simulation_dir = os.path.dirname(args.config) or "." + # wait_for_commands=True: sau simulation, vào chế độ chờ lệnh IPC (Interview) + # wait_for_commands=False (--no-wait): đóng ngay — dùng khi test hoặc không cần Interview wait_for_commands = not args.no_wait install_openai_cost_patch( @@ -1565,27 +1654,25 @@ async def main(): component="scripts.run_parallel_simulation", phase="simulation_run", ) - - # Khởi tạo cấu hình log (tắt log OASIS, dọn file cũ) + init_logging_for_simulation(simulation_dir) - - # Tạo trình quản lý log + log_manager = SimulationLogManager(simulation_dir) twitter_logger = log_manager.get_twitter_logger() reddit_logger = log_manager.get_reddit_logger() - + log_manager.info("=" * 60) log_manager.info("OASIS dual-platform parallel simulation") log_manager.info(f"Config file: {args.config}") log_manager.info(f"Simulation ID: {config.get('simulation_id', 'unknown')}") log_manager.info(f"Command wait mode: {'enabled' if wait_for_commands else 'disabled'}") log_manager.info("=" * 60) - + time_config = config.get("time_config", {}) total_hours = time_config.get('total_simulation_hours', 72) minutes_per_round = time_config.get('minutes_per_round', 30) config_total_rounds = (total_hours * 60) // minutes_per_round - + log_manager.info(f"Simulation parameters:") log_manager.info(f" - Total simulation duration: {total_hours} hours") log_manager.info(f" - Minutes per round: {minutes_per_round}") @@ -1595,44 +1682,47 @@ async def main(): if args.max_rounds < config_total_rounds: log_manager.info(f" - Actual executed rounds: {args.max_rounds} (truncated)") log_manager.info(f" - Agent count: {len(config.get('agent_configs', []))}") - + log_manager.info("Log structure:") log_manager.info(f" - Main log: simulation.log") log_manager.info(f" - Twitter actions: twitter/actions.jsonl") log_manager.info(f" - Reddit actions: reddit/actions.jsonl") log_manager.info("=" * 60) - + start_time = datetime.now() - - # Lưu kết quả mô phỏng của hai nền tảng + twitter_result: Optional[PlatformSimulation] = None reddit_result: Optional[PlatformSimulation] = None - + if args.twitter_only: twitter_result = await run_twitter_simulation(config, simulation_dir, twitter_logger, log_manager, args.max_rounds) elif args.reddit_only: reddit_result = await run_reddit_simulation(config, simulation_dir, reddit_logger, log_manager, args.max_rounds) else: - # Chạy song song (mỗi nền tảng dùng logger riêng) + # Chạy song song — asyncio.gather chạy cả 2 coroutine trong cùng event loop + # Twitter dùng General LLM, Reddit dùng Boost LLM → hai API key khác nhau + # → không tranh nhau rate limit, throughput tăng đôi results = await asyncio.gather( run_twitter_simulation(config, simulation_dir, twitter_logger, log_manager, args.max_rounds), run_reddit_simulation(config, simulation_dir, reddit_logger, log_manager, args.max_rounds), ) twitter_result, reddit_result = results - + total_elapsed = (datetime.now() - start_time).total_seconds() log_manager.info("=" * 60) log_manager.info(f"Simulation loop completed! Total elapsed: {total_elapsed:.1f}s") - - # Có vào chế độ chờ lệnh hay không + + # ── Chế độ chờ lệnh IPC ─────────────────────────────────────────────────── + # Sau khi simulation xong, environment vẫn còn sống. + # Vào vòng lặp poll ipc_commands/ mỗi 0.5 giây. + # Thoát khi: nhận close_env command, SIGTERM/SIGINT, hoặc exception. if wait_for_commands: log_manager.info("") log_manager.info("=" * 60) log_manager.info("Entering command wait mode - environment stays running") log_manager.info("Supported commands: interview, batch_interview, close_env") log_manager.info("=" * 60) - - # Tạo bộ xử lý IPC + ipc_handler = ParallelIPCHandler( simulation_dir=simulation_dir, twitter_env=twitter_result.env if twitter_result else None, @@ -1640,39 +1730,39 @@ async def main(): reddit_env=reddit_result.env if reddit_result else None, reddit_agent_graph=reddit_result.agent_graph if reddit_result else None ) + # Ghi env_status.json → Flask biết có thể gửi IPC command ipc_handler.update_status("alive") - - # Vòng lặp chờ lệnh (dùng _shutdown_event toàn cục) + try: while not _shutdown_event.is_set(): should_continue = await ipc_handler.process_commands() if not should_continue: break - # Dùng wait_for thay cho sleep để có thể phản hồi shutdown_event + # asyncio.wait_for thay vì sleep — có thể interrupt ngay khi nhận signal try: await asyncio.wait_for(_shutdown_event.wait(), timeout=0.5) - break # Đã nhận tín hiệu thoát + break # Nhận signal → thoát except asyncio.TimeoutError: - pass # Timeout thì tiếp tục vòng lặp + pass # Timeout thì tiếp tục poll except KeyboardInterrupt: print("\nInterrupt signal received") except asyncio.CancelledError: print("\nTask cancelled") except Exception as e: print(f"\nCommand processing error: {e}") - + log_manager.info("\nClosing environment...") - ipc_handler.update_status("stopped") - - # Đóng environment + ipc_handler.update_status("stopped") # Flask biết environment đã đóng + + # Đóng environment sau khi thoát chế độ chờ (hoặc --no-wait) if twitter_result and twitter_result.env: await twitter_result.env.close() log_manager.info("[Twitter] Environment closed") - + if reddit_result and reddit_result.env: await reddit_result.env.close() log_manager.info("[Reddit] Environment closed") - + log_manager.info("=" * 60) log_manager.info(f"All done!") log_manager.info(f"Log files:") @@ -1684,31 +1774,38 @@ async def main(): def setup_signal_handlers(loop=None): """ - Thiết lập signal handler để thoát đúng cách khi nhận SIGTERM/SIGINT - - Kịch bản mô phỏng bền vững: không thoát ngay sau khi mô phỏng xong, tiếp tục chờ lệnh interview - Khi nhận tín hiệu dừng, cần: - 1. Thông báo vòng lặp asyncio thoát khỏi trạng thái chờ - 2. Cho chương trình cơ hội dọn tài nguyên đúng cách (đóng DB, environment...) - 3. Sau đó mới thoát + Thiết lập signal handler cho SIGTERM và SIGINT. + + Chiến lược 2 bước (thay vì sys.exit() ngay): + Lần nhận signal đầu tiên: + _shutdown_event.set() → thông báo vòng lặp asyncio thoát + Cho chương trình cơ hội dọn tài nguyên (đóng DB, env...) + Không gọi sys.exit() ngay + + Lần nhận signal thứ hai (người dùng ép thoát): + sys.exit(1) → force quit ngay lập tức + + Tại sao không gọi sys.exit() ngay lần đầu? + asyncio event loop cần được thông báo gracefully để: + 1. Hoàn thành các coroutine đang chạy + 2. Đóng SQLite connections đúng cách + 3. Ghi simulation_end event vào actions.jsonl """ def signal_handler(signum, frame): global _cleanup_done sig_name = "SIGTERM" if signum == signal.SIGTERM else "SIGINT" print(f"\nReceived {sig_name}, shutting down...") - + if not _cleanup_done: _cleanup_done = True - # Set event để thông báo vòng lặp asyncio thoát (để kịp dọn tài nguyên) + # Thông báo asyncio event loop thoát gracefully if _shutdown_event: _shutdown_event.set() - - # Không gọi sys.exit() ngay, để asyncio thoát tự nhiên và dọn tài nguyên - # Nếu nhận tín hiệu lặp lại thì mới ép thoát else: + # Nhận signal lần 2 → force exit print("Force exit...") sys.exit(1) - + signal.signal(signal.SIGTERM, signal_handler) signal.signal(signal.SIGINT, signal_handler) @@ -1722,7 +1819,8 @@ if __name__ == "__main__": except SystemExit: pass finally: - # Dọn resource tracker của multiprocessing (tránh cảnh báo khi thoát) + # Dọn resource tracker của multiprocessing — tránh warning khi thoát + # trên một số hệ thống Python/OS kết hợp nhất định try: from multiprocessing import resource_tracker resource_tracker._resource_tracker._stop() diff --git a/test_code_backend/calculate_cost.ipynb b/test_code_backend/calculate_cost.ipynb new file mode 100644 index 00000000..d9a72ee1 --- /dev/null +++ b/test_code_backend/calculate_cost.ipynb @@ -0,0 +1,107 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": 1, + "id": "1606cd6c", + "metadata": {}, + "outputs": [], + "source": [ + "import json" + ] + }, + { + "cell_type": "code", + "execution_count": 2, + "id": "69ce47c3", + "metadata": {}, + "outputs": [], + "source": [ + "input = \"/home/anman/intern/MiroFish/backend/logs/proj_ac763defa255/cost_Qwen_Qwen3.6-27B-FP8.jsonl\"\n", + "\n", + "data=[]\n", + "with open(input, 'r', encoding='utf-8') as f:\n", + " data = [json.loads(line.strip()) for line in f if line.strip()]" + ] + }, + { + "cell_type": "code", + "execution_count": 3, + "id": "6f1fe377", + "metadata": {}, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "Input Tokens: 49830604\n", + "Output Tokens: 5432209\n" + ] + } + ], + "source": [ + "input_tokens = 0\n", + "output_tokens = 0\n", + "for item in data:\n", + " input_tokens += item['input_tokens']\n", + " output_tokens += item['output_tokens']\n", + "\n", + "print(f\"Input Tokens: {input_tokens}\")\n", + "print(f\"Output Tokens: {output_tokens}\")" + ] + }, + { + "cell_type": "code", + "execution_count": 5, + "id": "95cce20c", + "metadata": {}, + "outputs": [ + { + "name": "stdout", + "output_type": "stream", + "text": [ + "Total Cost: $41.2119\n", + "Cost Breakdown: Input Cost = $24.9153, Output Cost = $16.2966\n" + ] + } + ], + "source": [ + "cost_1M_input = 0.5\n", + "cost_1M_output = 3.0\n", + "\n", + "total_cost = (input_tokens / 1_000_000) * cost_1M_input + (output_tokens / 1_000_000) * cost_1M_output\n", + "print(f\"Total Cost: ${total_cost:.4f}\")\n", + "print(f\"Cost Breakdown: Input Cost = ${(input_tokens / 1_000_000) * cost_1M_input:.4f}, Output Cost = ${(output_tokens / 1_000_000) * cost_1M_output:.4f}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "id": "273cd657", + "metadata": {}, + "outputs": [], + "source": [] + } + ], + "metadata": { + "kernelspec": { + "display_name": "mirofish_v1", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.11.15" + } + }, + "nbformat": 4, + "nbformat_minor": 5 +} diff --git a/test_code_backend/full_pipeline/README.md b/test_code_backend/full_pipeline/README.md index e9894299..0a2d121c 100644 --- a/test_code_backend/full_pipeline/README.md +++ b/test_code_backend/full_pipeline/README.md @@ -100,6 +100,8 @@ Lưu lại đường dẫn `combined_file` để dùng ở bước tiếp theo. ```bash COMBINED_FILE="/home/anman/intern/MiroFish/test_code_backend/full_pipeline/data/articles_combined.md" +COMBINED_FILE="/home/anman/intern/MiroFish/data/news/2026-02/2026-02-03_oil-prices-add-gains-after-report-says_0178c4.md" + curl -s http://localhost:5001/api/graph/ontology/generate \ -F "files=@${COMBINED_FILE};filename=articles.md" \ -F "simulation_requirement=Analyze how these oil and financial market news articles affect investor sentiment and market dynamics. Predict how different market participants (traders, analysts, retail investors) will react and what the overall price trend will be." \ @@ -248,6 +250,38 @@ curl -s http://localhost:5001/api/report/$REPORT_ID/download -o report.md echo "Saved to report.md" ``` +Nếu sim đã có report rồi mà muốn chạy lại thì dùng như sau + +```bash +SIM_ID="sim_xxxxxxxxxxxx" + +SIM_ID="sim_07ab325b3818" +# 1. Bắt đầu generate → lấy task_id và report_id +RESP=$(curl -s http://localhost:5001/api/report/generate \ + -H "Content-Type: application/json" \ + -d "{\"simulation_id\": \"$SIM_ID\", \"force_regenerate\": true}") +TASK_ID=$(echo "$RESP" | jq -r '.data.task_id') +REPORT_ID=$(echo "$RESP" | jq -r '.data.report_id') +echo "task_id=$TASK_ID report_id=$REPORT_ID" + +# 2. Poll đến khi completed (chạy lặp lại, ~5–15 phút) +while true; do + R=$(curl -s http://localhost:5001/api/report/generate/status \ + -H "Content-Type: application/json" \ + -d "{\"task_id\": \"$TASK_ID\"}") + ST=$(echo "$R" | jq -r '.data.status') + PG=$(echo "$R" | jq -r '.data.progress') + echo "$ST ${PG}%" + [[ "$ST" == "completed" ]] && break + [[ "$ST" == "failed" ]] && { echo "FAILED"; break; } + sleep 15 +done + +# 3. Download file Markdown +curl -s http://localhost:5001/api/report/$REPORT_ID/download -o report.md +echo "Saved to report.md" +``` + --- ## Output files diff --git a/test_code_backend/full_pipeline/config_articles.env b/test_code_backend/full_pipeline/config_articles.env index 30cd28cc..71cf6278 100644 --- a/test_code_backend/full_pipeline/config_articles.env +++ b/test_code_backend/full_pipeline/config_articles.env @@ -10,7 +10,7 @@ full_content = False # start_time = 2026-02-09 # end_time = 2026-02-12 start_time = 2026-04-13 -end_time = 2026-04-16 +end_time = 2026-04-14 # ─── Simulation settings ────────────────────────────────────────────────────── # What the simulation should analyse / predict (sent to the LLM) diff --git a/test_code_backend/gen_ontogogy_and_graph/build_graph.py b/test_code_backend/gen_ontogogy_and_graph/build_graph.py deleted file mode 100644 index 9d914b16..00000000 --- a/test_code_backend/gen_ontogogy_and_graph/build_graph.py +++ /dev/null @@ -1,197 +0,0 @@ -""" -Test: Knowledge Graph build on Zep Cloud -Input : - output/ontology/ontology_*.json (latest, or pass path as arg) - - news articles (same date range as ontology) -Output: output/graph/ - graph_.json — build summary (node/edge count, timing) - entities_.json — entity detail: raw nodes + filtered result - -Run: - python build_graph.py # auto-picks latest ontology - python build_graph.py output/ontology/ontology_xyz.json # specific ontology file -""" - -import sys -import json -import time -import argparse -from datetime import datetime, date -from pathlib import Path - -# ── Path setup ─────────────────────────────────────────────────────────────── -SCRIPT_DIR = Path(__file__).parent -PROJECT_ROOT = SCRIPT_DIR.parent.parent -BACKEND_DIR = PROJECT_ROOT / "backend" -ONTOLOGY_DIR = SCRIPT_DIR.parent / "output" / "ontology" -GRAPH_DIR = SCRIPT_DIR.parent / "output" / "graph" -NEWS_DIR = PROJECT_ROOT / "data" / "news" / "2026-04" - -sys.path.insert(0, str(BACKEND_DIR)) -GRAPH_DIR.mkdir(parents=True, exist_ok=True) - -# ── Config ─────────────────────────────────────────────────────────────────── -START_DATE = date(2026, 4, 20) -END_DATE = date(2026, 4, 22) -GRAPH_NAME = "MiroFish_Test_Apr2026" -CHUNK_SIZE = 500 -CHUNK_OVERLAP = 50 -BATCH_SIZE = 3 -POLL_TIMEOUT = 600 - - -# ── Helpers ────────────────────────────────────────────────────────────────── -def log(msg: str): - print(f"[{datetime.now().strftime('%H:%M:%S')}] {msg}") - - -def extract_body(md_text: str) -> str: - s = md_text.strip() - if s.startswith("---"): - end = s.find("---", 3) - if end != -1: - return s[end + 3:].strip() - return s - - -def load_combined_text(news_dir: Path, start: date, end: date) -> str: - parts = [] - for md_file in sorted(news_dir.glob("*.md")): - try: - file_date = date.fromisoformat(md_file.stem.split("_")[0]) - except ValueError: - continue - if not (start <= file_date <= end): - continue - body = extract_body(md_file.read_text(encoding="utf-8")) - if body: - parts.append(body) - return "\n\n---\n\n".join(parts) - - -def pick_ontology_file(arg_path: str | None) -> Path: - if arg_path: - p = Path(arg_path) - if not p.is_absolute(): - p = ONTOLOGY_DIR / p - if not p.exists(): - raise FileNotFoundError(f"Ontology file not found: {p}") - return p - - candidates = sorted(ONTOLOGY_DIR.glob("ontology_*.json"), key=lambda f: f.stat().st_mtime) - if not candidates: - raise FileNotFoundError( - f"No ontology_*.json found in {ONTOLOGY_DIR}. Run gen_ontology.py first." - ) - return candidates[-1] # latest - - -# ── Main ───────────────────────────────────────────────────────────────────── -def main(): - parser = argparse.ArgumentParser(description="Build Zep Knowledge Graph from ontology") - parser.add_argument("ontology_file", nargs="?", help="Path to ontology JSON (default: latest in output/)") - args = parser.parse_args() - - from app.services.graph_builder import GraphBuilderService - from app.services.text_processor import TextProcessor - from app.services.zep_entity_reader import ZepEntityReader - - # Load ontology - ontology_path = pick_ontology_file(args.ontology_file) - log(f"Using ontology: {ontology_path.name}") - with open(ontology_path, encoding="utf-8") as f: - payload = json.load(f) - ontology = payload["ontology"] - - # Load news text - log(f"Loading news text: {START_DATE} → {END_DATE}") - combined_text = load_combined_text(NEWS_DIR, START_DATE, END_DATE) - log(f" Text length: {len(combined_text):,} chars") - - svc = GraphBuilderService() - t_total = time.time() - - # Create graph - log("Creating Zep graph...") - graph_id = svc.create_graph(GRAPH_NAME) - log(f" graph_id = {graph_id}") - - # Apply ontology - log("Applying ontology schema...") - svc.set_ontology(graph_id, ontology) - - # Chunk & upload - chunks = TextProcessor.split_text(combined_text, CHUNK_SIZE, CHUNK_OVERLAP) - log(f"Uploading {len(chunks)} chunks (batch_size={BATCH_SIZE})...") - - episode_uuids = svc.add_text_batches( - graph_id, chunks, BATCH_SIZE, - progress_callback=lambda msg, _: log(f" {msg}"), - ) - log(f" {len(episode_uuids)} episodes uploaded") - - # Wait for processing - log("Waiting for Zep to process episodes...") - svc._wait_for_episodes( - episode_uuids, - progress_callback=lambda msg, _: log(f" {msg}"), - timeout=POLL_TIMEOUT, - ) - - # Fetch result - log("Fetching graph stats...") - graph_info = svc._get_graph_info(graph_id) - elapsed = round(time.time() - t_total, 2) - - log(f"Done in {elapsed}s — nodes={graph_info.node_count}, edges={graph_info.edge_count}") - log(f" Entity types: {graph_info.entity_types}") - - out_path = GRAPH_DIR / f"graph_{graph_id}.json" - with open(out_path, "w", encoding="utf-8") as f: - json.dump({ - "graph_id": graph_id, - "graph_name": GRAPH_NAME, - "built_at": datetime.now().isoformat(), - "elapsed_seconds": elapsed, - "ontology_source": ontology_path.name, - "date_range": {"start": str(START_DATE), "end": str(END_DATE)}, - "chunks_uploaded": len(chunks), - "node_count": graph_info.node_count, - "edge_count": graph_info.edge_count, - "entity_types": graph_info.entity_types, - }, f, ensure_ascii=False, indent=2) - - log(f"Saved → {out_path.name}") - - # ── Entity detail ───────────────────────────────────────────────────────── - # Raw graph data — lưu nguyên output của get_graph_data() - log("Fetching raw graph data...") - graph_data = svc.get_graph_data(graph_id) - log(f" nodes={graph_data['node_count']}, edges={graph_data['edge_count']}") - - # Filtered entities — lưu nguyên output của filter_defined_entities().to_dict() - log("Running entity filter...") - defined_types = [e["name"] for e in ontology.get("entity_types", [])] - reader = ZepEntityReader() - filtered = reader.filter_defined_entities( - graph_id=graph_id, - defined_entity_types=defined_types, - enrich_with_edges=True, - ) - log(f" total={filtered.total_count}, matched={filtered.filtered_count}") - log(f" types found: {sorted(filtered.entity_types)}") - - entities_path = GRAPH_DIR / f"entities_{graph_id}.json" - with open(entities_path, "w", encoding="utf-8") as f: - json.dump({ - "graph_id": graph_id, - "fetched_at": datetime.now().isoformat(), - "ontology_entity_types": defined_types, - "raw_graph_data": graph_data, - "filtered_entities": filtered.to_dict(), - }, f, ensure_ascii=False, indent=2) - - log(f"Saved → {entities_path.name}") - - -if __name__ == "__main__": - main() diff --git a/test_code_backend/gen_ontogogy_and_graph/gen_ontology.py b/test_code_backend/gen_ontogogy_and_graph/gen_ontology.py deleted file mode 100644 index 16b3e577..00000000 --- a/test_code_backend/gen_ontogogy_and_graph/gen_ontology.py +++ /dev/null @@ -1,109 +0,0 @@ -""" -Test: Ontology generation from news articles -Input : data/news/2026-04/ (date range configured below) -Output: test_code_backend/output/ontology/ontology__.json - -Run: - python gen_ontology.py -""" - -import sys -import json -import time -from datetime import datetime, date -from pathlib import Path - -# ── Path setup ─────────────────────────────────────────────────────────────── -SCRIPT_DIR = Path(__file__).parent -PROJECT_ROOT = SCRIPT_DIR.parent.parent -BACKEND_DIR = PROJECT_ROOT / "backend" -ONTOLOGY_DIR = SCRIPT_DIR.parent / "output" / "ontology" -NEWS_DIR = PROJECT_ROOT / "data" / "news" / "2026-04" - -sys.path.insert(0, str(BACKEND_DIR)) -ONTOLOGY_DIR.mkdir(parents=True, exist_ok=True) - -# ── Config ─────────────────────────────────────────────────────────────────── -START_DATE = date(2026, 4, 20) -END_DATE = date(2026, 4, 22) - -SIMULATION_REQUIREMENT = ( - "Simulate social media public opinion dynamics around global oil/energy markets " - "and US-Iran geopolitical tensions during April 2026. " - "Focus on how governments, corporations, media outlets, financial analysts, and " - "ordinary citizens react and interact on platforms like Twitter and Reddit." -) - - -# ── Helpers ────────────────────────────────────────────────────────────────── -def log(msg: str): - print(f"[{datetime.now().strftime('%H:%M:%S')}] {msg}") - - -def extract_body(md_text: str) -> str: - """Strip YAML frontmatter (---...---), return only article body.""" - s = md_text.strip() - if s.startswith("---"): - end = s.find("---", 3) - if end != -1: - return s[end + 3:].strip() - return s - - -def load_articles(news_dir: Path, start: date, end: date) -> list[dict]: - articles = [] - for md_file in sorted(news_dir.glob("*.md")): - try: - file_date = date.fromisoformat(md_file.stem.split("_")[0]) - except ValueError: - continue - if not (start <= file_date <= end): - continue - body = extract_body(md_file.read_text(encoding="utf-8")) - if body: - articles.append({"filename": md_file.name, "date": str(file_date), "body": body}) - return articles - - -# ── Main ───────────────────────────────────────────────────────────────────── -def main(): - from app.services.ontology_generator import OntologyGenerator - - log(f"Loading news: {START_DATE} → {END_DATE}") - articles = load_articles(NEWS_DIR, START_DATE, END_DATE) - log(f" {len(articles)} articles loaded") - - log("Calling LLM to generate ontology...") - t0 = time.time() - ontology = OntologyGenerator().generate( - document_texts=[a["body"] for a in articles], - simulation_requirement=SIMULATION_REQUIREMENT, - ) - elapsed = round(time.time() - t0, 2) - - entity_names = [e["name"] for e in ontology.get("entity_types", [])] - edge_names = [e["name"] for e in ontology.get("edge_types", [])] - log(f"Done in {elapsed}s") - log(f" Entities ({len(entity_names)}): {entity_names}") - log(f" Edges ({len(edge_names)}): {edge_names}") - - date_range = f"{START_DATE.strftime('%Y%m%d')}-{END_DATE.strftime('%Y%m%d')}" - ts = datetime.now().strftime("%H%M%S") - out_path = ONTOLOGY_DIR / f"ontology_{date_range}_{ts}.json" - - with open(out_path, "w", encoding="utf-8") as f: - json.dump({ - "meta": { - "generated_at": datetime.now().isoformat(), - "elapsed_seconds": elapsed, - "date_range": {"start": str(START_DATE), "end": str(END_DATE)}, - "article_count": len(articles), - }, - "ontology": ontology, - }, f, ensure_ascii=False, indent=2) - - log(f"Saved → {out_path.name}") - - -if __name__ == "__main__": - main()