From 8679916cedb23af1e73056a8fc95fac7cafd2e22 Mon Sep 17 00:00:00 2001 From: ththu0205 Date: Mon, 30 Mar 2026 07:07:52 +0000 Subject: [PATCH] edit prompts --- .../app/services/oasis_profile_generator.py | 24 +- backend/app/services/ontology_generator.py | 277 ++++++++++++------ .../services/simulation_config_generator.py | 248 +++++++++++----- backend/app/services/zep_tools.py | 224 ++++++++++---- backend/scripts/test_profile_format.py | 2 +- 5 files changed, 550 insertions(+), 225 deletions(-) diff --git a/backend/app/services/oasis_profile_generator.py b/backend/app/services/oasis_profile_generator.py index d5425571..b37284cf 100644 --- a/backend/app/services/oasis_profile_generator.py +++ b/backend/app/services/oasis_profile_generator.py @@ -162,7 +162,7 @@ class OasisProfileGenerator: # List mảng quốc tịch Cơ Bản COUNTRIES = [ - "China", "US", "UK", "Japan", "Germany", "France", + "Vietnam", "China", "US", "UK", "Japan", "Germany", "France", "Canada", "Australia", "Brazil", "India", "South Korea" ] @@ -676,7 +676,7 @@ class OasisProfileGenerator: def _get_system_prompt(self, is_individual: bool) -> str: """Lấy prompt cho hệ thống""" - # base_prompt = "You are an expert in generating social media user personas. Generate detailed and realistic personas for public opinion simulation to recreate existing real-world conditions to the greatest extent possible. You must return a valid JSON format; all string values must not contain unescaped line breaks. Use Chinese." + # base_prompt = "You are an expert in generating social media user personas. Generate detailed and realistic personas for public opinion simulation to recreate existing real-world conditions to the greatest extent possible. You must return a valid JSON format; all string values must not contain unescaped line breaks. Use Vietnamese." base_prompt = "Bạn là chuyên gia tạo hồ sơ người dùng mạng xã hội. Hãy tạo các nhân vật chi tiết và chân thực phục vụ cho việc mô phỏng dư luận, nhằm tái hiện tối đa các tình huống thực tế hiện có. Phải trả về định dạng JSON hợp lệ; tất cả các giá trị chuỗi không được chứa ký tự xuống dòng chưa được xử lý (unescaped). Sử dụng tiếng Việt." return base_prompt @@ -718,14 +718,14 @@ class OasisProfileGenerator: # 3. age: Age as a number (must be an integer) # 4. gender: Gender, must be in English: "male" or "female" # 5. mbti: MBTI type (e.g., INTJ, ENFP, etc.) -# 6. country: Country (use Chinese, e.g., "中国") +# 6. country: Country (use Vietnamese, e.g., "Việt Nam") # 7. profession: Occupation # 8. interested_topics: An array of interested topics # IMPORTANT: # - All field values must be strings or numbers; do not use line breaks. # - The 'persona' must be a coherent block of text description. -# - Use Chinese (except for the 'gender' field, which must be English male/female). +# - Use Vietnamese (except for the 'gender' field, which must be English male/female). # - Content must remain consistent with the entity information. # - 'age' must be a valid integer; 'gender' must be "male" or "female".""" @@ -753,14 +753,14 @@ Vui lòng tạo JSON bao gồm các trường sau: 3. age: Con số tuổi (phải là số nguyên) 4. gender: Giới tính, phải là tiếng Anh: "male" hoặc "female" 5. mbti: Loại MBTI (như INTJ, ENFP, v.v.) -6. country: Quốc gia (sử dụng tiếng Trung, ví dụ: "中国") +6. country: Quốc gia (sử dụng tiếng Việt, ví dụ: "Việt Nam") 7. profession: Nghề nghiệp 8. interested_topics: Mảng các chủ đề quan tâm QUAN TRỌNG: - Tất cả giá trị các trường phải là chuỗi hoặc số, không sử dụng ký tự xuống dòng. - 'persona' phải là một đoạn mô tả văn bản mạch lạc. -- Sử dụng tiếng Trung (ngoại trừ trường 'gender' phải dùng tiếng Anh male/female). +- Sử dụng tiếng Việt (ngoại trừ trường 'gender' phải dùng tiếng Anh male/female). - Nội dung phải nhất quán với thông tin thực thể. - 'age' phải là số nguyên hợp lệ, 'gender' phải là "male" hoặc "female". """ @@ -802,14 +802,14 @@ QUAN TRỌNG: # 3. age: Fixed at 30 (virtual age for an organizational account) # 4. gender: Fixed as "other" (representing non-individual accounts) # 5. mbti: MBTI type used to describe the account's style (e.g., ISTJ for rigorous/conservative) -# 6. country: Country (use Chinese, e.g., "中国") +# 6. country: Country (use Vietnamese, e.g., "Việt Nam") # 7. profession: Description of organizational functions # 8. interested_topics: An array of focused fields/areas of interest # IMPORTANT: # - All field values must be strings or numbers; null values are not allowed. # - 'persona' must be a coherent block of text description; do not use line breaks. -# - Use Chinese (except for the 'gender' field, which must be the English string "other"). +# - Use Vietnamese (except for the 'gender' field, which must be the English string "other"). # - 'age' must be the integer 30; 'gender' must be the string "other". # - The account's tone and discourse must strictly align with its institutional identity and positioning. # """ @@ -838,14 +838,14 @@ Vui lòng tạo JSON bao gồm các trường sau: 3. age: Cố định là 30 (tuổi ảo cho tài khoản tổ chức) 4. gender: Cố định là "other" (biểu thị tài khoản tổ chức, không phải cá nhân) 5. mbti: Loại MBTI dùng để mô tả phong cách tài khoản (ví dụ: ISTJ đại diện cho sự nghiêm túc, bảo thủ) -6. country: Quốc gia (sử dụng tiếng Trung, ví dụ: "中国") +6. country: Quốc gia (sử dụng tiếng Việt, ví dụ: "Việt Nam") 7. profession: Mô tả chức năng của tổ chức 8. interested_topics: Mảng các lĩnh vực quan tâm QUAN TRỌNG: - Tất cả giá trị các trường phải là chuỗi hoặc số, không cho phép giá trị null. - 'persona' phải là một đoạn mô tả văn bản mạch lạc, không sử dụng ký tự xuống dòng. -- Sử dụng tiếng Trung (ngoại trừ trường 'gender' phải dùng chuỗi tiếng Anh "other"). +- Sử dụng tiếng Việt (ngoại trừ trường 'gender' phải dùng chuỗi tiếng Anh "other"). - 'age' phải là số nguyên 30, 'gender' phải là chuỗi "other". - Phát ngôn và giọng điệu của tài khoản phải phù hợp tuyệt đối với định vị danh tính và đặc thù của tổ chức. """ @@ -936,6 +936,7 @@ QUAN TRỌNG: parallel_count: int = 5, realtime_output_path: Optional[str] = None, output_platform: str = "reddit", + metadata_platform: Optional[str] = None, simulation_id: Optional[str] = None, project_id: Optional[str] = None, ) -> List[OasisAgentProfile]: @@ -950,6 +951,7 @@ QUAN TRỌNG: parallel_count: Số luồng song song, mặc định 5 realtime_output_path: Đường dẫn lưu file realtime (Gen ra đứa nào auto save đứa đó luôn) output_platform: Format lưu trữ output ("reddit" hoạc "twitter") + metadata_platform: Nền tảng dùng cho metadata cost Returns: Danh sách Profile Agent @@ -966,7 +968,7 @@ QUAN TRỌNG: "phase": "generate_profiles", "simulation_id": simulation_id, "project_id": project_id, - "platform": output_platform, + "platform": metadata_platform, } total = len(entities) diff --git a/backend/app/services/ontology_generator.py b/backend/app/services/ontology_generator.py index cc44a899..fd52e3e8 100644 --- a/backend/app/services/ontology_generator.py +++ b/backend/app/services/ontology_generator.py @@ -9,32 +9,145 @@ from ..utils.llm_client import LLMClient # System prompt dùng cho việc tự động sinh Ontology -ONTOLOGY_SYSTEM_PROMPT = """Bạn là một chuyên gia thiết kế bản thể học (Ontology) cho Tri thức đồ thị (Knowledge Graph). Nhiệm vụ của bạn là phân tích nội dung văn bản được cung cấp và nhu cầu để thiết kế các loại thực thể (Entity) và loại mối quan hệ (Relationship) thiết kế phù hợp cho **Mô phỏng dư luận trên mạng xã hội**. +# ONTOLOGY_SYSTEM_PROMPT = """You are a professional Knowledge Graph Ontology Design Expert. Your task is to analyze the given text content and simulation requirements to design entity types and relationship types suitable for **social media public opinion simulation**. -**QUAN TRỌNG: Bạn BẮT BUỘC phải đầu ra một cấu trúc định dạng JSON hợp lệ, KHÔNG ĐƯỢC xuất thêm bất kỳ văn bản nào khác.** +# **IMPORTANT: You must output valid JSON format data only. Do not include any other text.** + +# ## Core Task Background + +# We are building a social media public opinion simulation system. In this system: +# - Each entity is an "account" or "subject" capable of speaking, interacting, and spreading information on social media. +# - Entities influence, forward, comment on, and respond to each other. +# - We need to simulate the reactions of all parties and the paths of information dissemination during public opinion events. + +# Therefore, **entities must be real-world subjects capable of speaking and interacting on social media**: + +# **CAN BE**: +# - Specific individuals (public figures, parties involved, opinion leaders, experts, ordinary people). +# - Companies and enterprises (including their official accounts). +# - Organizations (universities, associations, NGOs, labor unions, etc.). +# - Government departments and regulatory agencies. +# - Media outlets (newspapers, TV stations, independent media, websites). +# - Social media platforms themselves. +# - Representatives of specific groups (e.g., alumni associations, fan clubs, rights protection groups). + +# **CANNOT BE**: +# - Abstract concepts (e.g., "public opinion", "emotion", "trend"). +# - Themes/Topics (e.g., "academic integrity", "education reform"). +# - Viewpoints/Attitudes (e.g., "supporters", "opponents"). + +# ## Output Format + +# Please output in JSON format with the following structure: + +# ```json +# { +# "entity_types": [ +# { +# "name": "Entity type name (English, PascalCase)", +# "description": "Short description (English, max 100 characters)", +# "attributes": [ +# { +# "name": "Attribute name (English, snake_case)", +# "type": "text", +# "description": "Attribute description" +# } +# ], +# "examples": ["Example Entity 1", "Example Entity 2"] +# } +# ], +# "edge_types": [ +# { +# "name": "Relationship type name (English, UPPER_SNAKE_CASE)", +# "description": "Short description (English, max 100 characters)", +# "source_targets": [ +# {"source": "Source entity type", "target": "Target entity type"} +# ], +# "attributes": [] +# } +# ], +# "analysis_summary": "Brief analysis of the text content (in Vietnamese)" +# } +# ``` + +# ## Design Guidelines (Extremely Important!) + +# ### 1. Entity Type Design - Strict Compliance Required + +# **Quantity Requirement: Must be EXACTLY 10 entity types.** + +# **Hierarchy Requirements (Must include both specific types and fallback types):** + +# Your 10 entity types must include the following layers: + +# A. **Fallback Types (Required, place as the last 2 in the list)**: +# - `Person`: The fallback type for any individual natural person. Use this when a person does not fit into other specific person types. +# - `Organization`: The fallback type for any organization or institution. Use this when an organization does not fit into other specific organizational types. + +# B. **Specific Types (8 types, designed based on text content)**: +# - Design more specific types targeting the main roles appearing in the text. +# - Example: For academic events, use `Student`, `Professor`, `University` +# - VExample: For business events, use `Company`, `CEO`, `Employee` + +# **Why fallback types are needed:** +# - Various people appear in texts (e.g., "primary school teacher", "passerby", "netizen"). +# - Without a specific match, they should be categorized under `Person`. +# - Similarly, small organizations or temporary groups should fall under `Organization`. + +# **Specific Type Design Principles:** +# - Identify high-frequency or critical roles from the text. +# - Each specific type should have clear boundaries to avoid overlap. +# - The description must clearly state the difference between this type and the fallback type. + +# ### 2. Relationship Type Design + +# - Quantity: 6-10 types. +# - Relationships should reflect real-world connections in social media interactions. +# - Ensure `source_targets` cover your defined entity types. + +# ### 3. Attribute Design + +# - 1-3 key attributes per entity type. +# - **NOTE**: Do NOT use `name`, `uuid`, `group_id`, `created_at` or `summary` as attribute names (these are system reserved words). +# - Recommended: `full_name`, `title`, `role`, `position`, `location`, `description, etc. + +# ## Entity Type References + +# - Individuals (Specific): Student, Professor, Journalist, Celebrity, Executive, Official, Lawyer, Doctor. +# - Individuals (Fallback): Person. +# - Organizations (Specific): University, Company, GovernmentAgency, MediaOutlet, Hospital, School, NGO. +# - Organizations (Fallback): Organization. + +# ## Relationship Type References + +# WORKS_FOR, STUDIES_AT, AFFILIATED_WITH, REPRESENTS, REGULATES, REPORTS_ON, COMMENTS_ON, RESPONDS_TO, SUPPORTS, OPPOSES, COLLABORATES_WITH, COMPETES_WITH. +# """ + +ONTOLOGY_SYSTEM_PROMPT = """Bạn là một chuyên gia thiết kế Bản thể học (Ontology) cho Biểu đồ tri thức chuyên nghiệp. Nhiệm vụ của bạn là phân tích nội dung văn bản và yêu cầu mô phỏng được cung cấp để thiết kế các loại thực thể và loại quan hệ phù hợp cho việc **mô phỏng dư luận trên mạng xã hội**. + +**QUAN TRỌNG: Bạn phải xuất dữ liệu ở định dạng JSON hợp lệ, không xuất thêm bất kỳ nội dung nào khác.** ## Bối cảnh nhiệm vụ cốt lõi +- Chúng tôi đang xây dựng một hệ thống mô phỏng dư luận mạng xã hội. Trong hệ thống này: +- Mỗi thực thể là một "tài khoản" hoặc "chủ thể" có thể phát ngôn, tương tác và lan truyền thông tin trên mạng xã hội. +- Các thực thể sẽ ảnh hưởng, chia sẻ, bình luận và phản hồi lẫn nhau. +- Chúng tôi cần mô phỏng phản ứng của các bên và lộ trình lan truyền thông tin trong các sự kiện dư luận. -Chúng tôi đang xây dựng một **hệ thống mô phỏng tin đồn và dư luận mạng xã hội**. Trong hệ thống này: -- Mỗi thực thể là một "tài khoản" hoặc "chủ thể" có thể lên tiếng, tương tác và lan truyền thông tin trên mạng xã hội. -- Các thực thể có thể gây ảnh hưởng, chuyển tiếp (retweet), bình luận hoặc phản hồi lẫn nhau. -- Chúng tôi cần mô phỏng phản ứng của các bên và đường truyền thông tin trong các sự kiện dư luận. - -Do đó, **thực thể phải là các chủ thể có thật trong thế giới thực, có khả năng lên tiếng và tương tác trên mạng xã hội**: +Do đó, **thực thể phải là những chủ thể tồn tại thực tế, có khả năng phát ngôn và tương tác trên mạng xã hội**: **CÓ THỂ LÀ**: -- Cá nhân cụ thể (nhân vật của công chúng, các bên liên quan, KOL, chuyên gia / học giả, người bình thường) -- Công ty, doanh nghiệp (bao gồm cả tài khoản chính thức của họ) -- Tổ chức (trường đại học, hiệp hội, tổ chức phi chính phủ (NGO), công đoàn, v.v.) -- Các cơ quan chính phủ, cơ quan quản lý -- Tổ chức báo chí / truyền thông (báo đài, đài truyền hình, tự do truyền thông, trang web) -- Bản thân nền tảng mạng xã hội -- Đại diện nhóm cụ thể (như hội cựu sinh viên, fan group, nhóm bảo vệ quyền lợi, v.v.) +- Cá nhân cụ thể (người công chúng, bên liên quan, người dẫn dắt dư luận, chuyên gia, người bình thường). +- Công ty, doanh nghiệp (bao gồm cả tài khoản chính thức của họ). +- Tổ chức (trường đại học, hiệp hội, NGO, công đoàn, v.v.). +- Cơ quan chính phủ, cơ quan quản lý. +- Cơ quan truyền thông (báo chí, đài truyền hình, tự truyền thông, trang web). +- Bản thân nền tảng mạng xã hội. +- Đại diện nhóm cụ thể (như hội cựu sinh viên, nhóm người hâm mộ, nhóm bảo vệ quyền lợi, v.v.). -**KHÔNG ĐƯỢC LÀ**: -- Khái niệm trừu tượng (như "dư luận", "cảm xúc", "xu hướng") -- Chủ đề / đề tài (như "tính toàn vẹn học thuật", "cải cách giáo dục") -- Quan điểm / thái độ (như "phe ủng hộ", "bên phản đối") +**KHÔNG THỂ LÀ**: +- Khái niệm trừu tượng (như "dư luận", "cảm xúc", "xu hướng"). +- Chủ đề/Vấn đề (như "liêm chính học thuật", "cải cách giáo dục"). +- Quan điểm/Thái độ (như "bên ủng hộ", "bên phản đối"). ## Định dạng đầu ra @@ -45,38 +158,38 @@ Hãy trả về dưới định dạng JSON, bao gồm cấu trúc sau: "entity_types": [ { "name": "Tên loại thực thể (Tiếng Anh, PascalCase)", - "description": "Mô tả ngắn gọn (Tiếng Anh, tối đa 100 ký tự)", + "description": "Mô tả ngắn gọn (Tiếng Anh, không quá 100 ký tự)", "attributes": [ { "name": "Tên thuộc tính (Tiếng Anh, snake_case)", "type": "text", - "description": "Mô tả của thuộc tính" + "description": "Mô tả thuộc tính" } ], - "examples": ["Ví dụ thực thể 1", "Ví dụ thực thể 2"] + "examples": ["Thực thể ví dụ 1", "Thực thể ví dụ 2"] } ], "edge_types": [ { "name": "Tên loại quan hệ (Tiếng Anh, UPPER_SNAKE_CASE)", - "description": "Mô tả ngắn (Tiếng Anh, tối đa 100 ký tự)", + "description": "Mô tả ngắn gọn (Tiếng Anh, không quá 100 ký tự)", "source_targets": [ {"source": "Loại thực thể nguồn", "target": "Loại thực thể đích"} ], "attributes": [] } ], - "analysis_summary": "Giải thích ngắn gọn phân tích của bạn về văn bản (Tiếng Việt)" + "analysis_summary": "Phân tích ngắn gọn nội dung văn bản (bằng tiếng Việt)" } ``` -## Hướng dẫn Thiết kế (CỰC KỲ QUAN TRỌNG!) +## Hướng dẫn thiết kế (Cực kỳ quan trọng!) -### 1. Thiết kế loại Thực thể (Entity Types) - Phải tuân thủ nghiêm ngặt +### 1. Thiết kế loại thực thể - Phải tuân thủ nghiêm ngặt -**Yêu cầu số lượng: Đúng 10 loại Thực thể.** +**Yêu cầu số lượng: Phải có CHÍNH XÁC 10 loại thực thể.** -**Yêu cầu về cấu trúc phân cấp (Phải có cả Loại cụ thể và Loại bao quát/fallback):** +**Yêu cầu về cấu trúc phân cấp (Phải bao gồm cả loại cụ thể và loại dự phòng):** 10 loại thực thể của bạn phải bao gồm cấp độ sau: @@ -85,7 +198,7 @@ A. **Loại bao quát (Fallback Types) (BẮT BUỘC, phải nằm ở 2 vị tr - `Organization`: Là loại bao quát cho MỌI tổ chức. Đặc trưng cho các tổ chức nhỏ hoặc không phù hợp với các loại tổ chức cụ thể khác. B. **Loại cụ thể (8 loại, phụ thuộc vào nội dung văn bản)**: - - Thiết kế các loại cụ thể cho các vai chính được nhắc đến nhiều nhất trong văn bản. + - Thiết kế các loại cụ thể cho các vai trò chính được nhắc đến nhiều nhất trong văn bản. - Ví dụ: Nếu văn bản nói về scandal trường học, có thể có: `Student`, `Professor`, `University` - Ví dụ: Nếu văn bản là câu chuyện kinh doanh, có thể có: `Company`, `CEO`, `Employee` @@ -95,9 +208,9 @@ B. **Loại cụ thể (8 loại, phụ thuộc vào nội dung văn bản)**: - Tương tự, tổ chức nhỏ bé hoặc nhóm học tập tạm thời nên thuộc `Organization` **Nguyên tắc cho các loại Cụ thể:** -- Nhận dạng tần suất xuất hiện và sức ảnh hưởng tới cốt truyện để xây dựng loại thực thể. +- Nhận diện các vai trò xuất hiện với tần suất cao hoặc quan trọng từ văn bản. - Mỗi loại nên có một ranh giới rõ ràng, không bị chồng chéo. -- Thuộc tính mô tả (description) phải giải thích vì sao loại này tách biệt. +- Phần description phải giải thích rõ sự khác biệt giữa loại này và loại bao quát. ### 2. Thiết kế Cạnh/Quan hệ (Edge Types) @@ -113,45 +226,14 @@ B. **Loại cụ thể (8 loại, phụ thuộc vào nội dung văn bản)**: ## Loại Thực thể tham khảo -**Loại cá nhân (Cụ thể):** -- Student: Học sinh/Sinh viên -- Professor: Giáo sư/Học giả -- Journalist: Nhà báo/Phóng viên -- Celebrity: Người nổi tiếng/Idol -- Executive: Các giám đốc, CEO, cấp lãnh đạo -- Official: Các vị công chức chính phủ -- Lawyer: Luật sư -- Doctor: Y sĩ/Bác sĩ +- Nhóm cá nhân (Cụ thể): Student, Professor, Journalist, Celebrity, Executive, Official, Lawyer, Doctor. +- Nhóm cá nhân (Bao quát): Person. +- Nhóm tổ chức (Cụ thể): University, Company, GovernmentAgency, MediaOutlet, Hospital, School, NGO. +- Nhóm tổ chức (Bao quát): Organization. -**Loại cá nhân (Bao quát):** -- Person: Là loại bao quát cho MỌI cá nhân tự nhiên nào không thuộc chi tiết ở trên. +## Tham khảo loại quan hệ -**Loại tổ chức (Cụ thể):** -- University: Đại học hoặc học viện -- Company: Doanh nghiệp hay Công ty, tập đoàn -- GovernmentAgency: Cơ quan quản lý, các cơ quan ban ngành công quyền -- MediaOutlet: Truyền thông hay Tạp chí, Đài tin tức -- Hospital: Bệnh viện / Trung tâm y tế -- School: Bậc tiểu/trung học -- NGO: Các loại Tổ chức phi chính phủ hoặc từ thiện - -**Loại tổ chức (Bao quát):** -- Organization: Là loại bao quát cho MỌI cơ cấu hợp tác không thuôc chi tiết tổ chức ở trên. - -## Loại Khái niệm Liên kết (Quan Hệ) - -- WORKS_FOR: Làm việc và ăn lương bởi tổ chức -- STUDIES_AT: Đang học tại nhà trường -- AFFILIATED_WITH: Liên quan, Trực thuộc vào đơn vị -- REPRESENTS: Thể hiện tư cách hành động đại diện cho tập thể -- REGULATES: Theo dõi, quản lý, thanh tra chính sách -- REPORTS_ON: Tác nghiệp báo chí, có tin về hiện tượng -- COMMENTS_ON: Có phản hồi hoặc lên tiếng về tranh cãi -- RESPONDS_TO: Hành động đáp trả -- SUPPORTS: Theo phe ủng hộ điều luật -- OPPOSES: Phản đối chính sách -- COLLABORATES_WITH: Tham gia phối ứng xử lý sự cố. -- COMPETES_WITH: Quan hệ thù địch. +WORKS_FOR, STUDIES_AT, AFFILIATED_WITH, REPRESENTS, REGULATES, REPORTS_ON, COMMENTS_ON, RESPONDS_TO, SUPPORTS, OPPOSES, COLLABORATES_WITH, COMPETES_WITH. """ @@ -204,8 +286,8 @@ class OntologyGenerator: return result - # Định mức giới hạn độ dài ký tự tối đa của đoạn văn bản có thể gửi cho LLM (5 vạn chữ) - MAX_TEXT_LENGTH_FOR_LLM = 80000 + # Định mức giới hạn độ dài ký tự tối đa của đoạn văn bản có thể gửi cho LLM (10 vạn chữ) + MAX_TEXT_LENGTH_FOR_LLM = 100000 def _build_user_message( self, @@ -222,9 +304,36 @@ class OntologyGenerator: # Nếu vượt quá giới hạn tối đa, thực hiện cắt bớt (Việc này chỉ ảnh hưởng prompt gửi nhận diện Ontology, không ảnh hưởng thư viện Graph building ở sau) if len(combined_text) > self.MAX_TEXT_LENGTH_FOR_LLM: combined_text = combined_text[:self.MAX_TEXT_LENGTH_FOR_LLM] - combined_text += f"\n\n...(Văn bản gốc dài {original_length} chữ, đã chủ động cắt lấy {self.MAX_TEXT_LENGTH_FOR_LLM} chữ đầu tiên để phục vụ phân tích Ontology)..." + combined_text += f"\n\n...(Original text is {original_length} characters long; the first {self.MAX_TEXT_LENGTH_FOR_LLM} characters have been proactively truncated for Ontology analysis)..." + +# message = f"""## Simulation Requirements + +# {simulation_requirement} + +# ## Document Content + +# {combined_text} +# """ - message = f"""## Nhu cầu mô phỏng +# if additional_context: +# message += f""" +# ## Additional Context + +# {additional_context} +# """ + +# message += """ +# Based on the content above, please design entity types and relationship types suitable for social media public opinion simulation. + +# **Rules that MUST be followed**: +# 1. You must output EXACTLY 10 entity types. +# 2. The last 2 types must be fallback types: Person (individual fallback) and Organization (organization fallback). +# 3. The first 8 types should be specific types designed based on the text content. +# 4. All entity types must be real-world subjects capable of speaking/interacting; they cannot be abstract concepts. +# 5. Attribute names cannot use reserved words like name, uuid, or group_id; use alternatives like full_name, org_name, etc. +# """ + + message = f"""## Yêu cầu mô phỏng {simulation_requirement} @@ -241,14 +350,14 @@ class OntologyGenerator: """ message += """ -Dựa vào các thông tin trên đây, hãy thiết kế các loại mô hình Thực Thể và Quan Hệ phù hợp để phục vụ việc mô phỏng dư luận trên mạng xã hội. +Dựa trên các nội dung trên, hãy thiết kế các loại thực thể và loại quan hệ phù hợp cho việc mô phỏng dư luận xã hội. -**Các quy tắc BẮT BUỘC tuân thủ**: -1. Số lượng chính xác: Xuất phải CHUẨN XÁC 10 loại Thực thể -2. 2 vị trí cuối cùng bắt buộc là từ Khoá phụ (Fallback): Person (Cho cá nhân) và Organization (Cho Tổ chức) -3. 8 vị trí đầu tiên phải phân tích và suy luận dựa vào cấu trúc của chính văn bản truyền vào -4. Tất cả các thực thể được liệt kê phải đóng vai trò là Chủ thể (nhân vật có thể lên tiếng ngoài đời thực), KHÔNG ĐƯỢC dùng làm khái niệm trừu tượng. -5. Tên biến thuộc tính KHÔNG ĐƯỢC là name, uuid, group_id hay các biến số bảo lưu của hệ thống khác. Vui lòng chuyển thành full_name, org_name, v.v. +**Các quy tắc BẮT BUỘC phải tuân thủ**: +1. Phải xuất chính xác 10 loại thực thể. +2. 2 loại cuối cùng phải là loại dự phòng: Person (Cá nhân dự phòng) và Organization (Tổ chức dự phòng). +3. 8 loại đầu tiên là các loại cụ thể được thiết kế dựa trên nội dung văn bản. +4. Tất cả các loại thực thể phải là những chủ thể có thể phát ngôn trong thực tế, không được là các khái niệm trừu tượng. +5. Tên thuộc tính không được sử dụng các từ khóa hệ thống như name, uuid, group_id; hãy thay thế bằng full_name, org_name, v.v. """ return message @@ -355,15 +464,15 @@ Dựa vào các thông tin trên đây, hãy thiết kế các loại mô hình """ code_lines = [ '"""', - 'Các loại đối tượng (Thực thể) tuỳ chỉnh', - 'Được khởi tạo tự động bởi công cụ MiroFish, ứng dụng vào việc chạy giả lập diễn biến dư luận', + 'Custom Entity Type Definitions', + 'Automatically generated by MiroFish for social media public opinion simulation', '"""', '', 'from pydantic import Field', 'from zep_cloud.external_clients.ontology import EntityModel, EntityText, EdgeModel', '', '', - '# ============== Định nghĩa Tên Lớp Các thực thể (Entity) ==============', + '# ============== Entity Type Definitions ==============', '', ] @@ -390,7 +499,7 @@ Dựa vào các thông tin trên đây, hãy thiết kế các loại mô hình code_lines.append('') code_lines.append('') - code_lines.append('# ============== Định nghĩa Các Nhóm Quan Hệ/Hành Vi (Edge) ==============') + code_lines.append('# ============== Relationship Type Definitions ==============') code_lines.append('') # Khởi tạo các đoạn mã tạo lập Relationship (Edges) @@ -419,7 +528,7 @@ Dựa vào các thông tin trên đây, hãy thiết kế các loại mô hình code_lines.append('') # Tự động kết xuất ra dictionary mapping từ Tên Loại - sang class Object - code_lines.append('# ============== Các tuỳ chỉnh Map Cấu Hình ==============') + code_lines.append('# ============== Type Configuration ==============') code_lines.append('') code_lines.append('ENTITY_TYPES = {') for entity in ontology.get("entity_types", []): diff --git a/backend/app/services/simulation_config_generator.py b/backend/app/services/simulation_config_generator.py index 3b794969..7dc9d92d 100644 --- a/backend/app/services/simulation_config_generator.py +++ b/backend/app/services/simulation_config_generator.py @@ -546,27 +546,72 @@ class SimulationConfigGenerator: # Cắt lấy số lượng Tối đa số lượng (Chiếm 80% từ số lượng lượng Agent thực thể) max_agents_allowed = max(1, int(num_entities * 0.9)) + +# prompt = f"""Based on the following simulation requirements, generate a time simulation configuration. + +# {context_truncated} + +# ## Task +# Please generate a time configuration JSON. + +# ### General Principles (For reference only; adjust flexibly based on specific events and participant groups): +# - The user group consists of Vietnamese people; must comply with Hanoi Time (CST) daily routines. +# - 0:00–5:00 AM: Almost no activity (Activity Coefficient: 0.05). +# - 6:00–8:00 AM: Gradual increase in activity (Activity Coefficient: 0.4). +# - 9:00 AM–6:00 PM (Work hours): Moderate activity (Activity Coefficient: 0.7). +# - 7:00 PM–10:00 PM: Peak period (Activity Coefficient: 1.5). +# - After 11:00 PM: Activity declines (Activity Coefficient: 0.5). +# - General Pattern: Low activity in the early morning, gradual increase in the morning, moderate during work hours, and peak in the evening. +# - **Important:**: The example values below are for reference only. You need to adjust specific periods based on the nature of the event and characteristics of the participant group. +# - e.g., The peak for students might be 9:00 PM–11:00 PM; Media groups remain active all day; Official organizations only during work hours. +# - e.g., Breaking news may lead to discussions late at night; off_peak_hours can be shortened accordingly. + +# ### Return JSON Format (Do not use Markdown) + +# Example: +# {{ +# "total_simulation_hours": 72, +# "minutes_per_round": 60, +# "agents_per_hour_min": 5, +# "agents_per_hour_max": 50, +# "peak_hours": [19, 20, 21, 22], +# "off_peak_hours": [0, 1, 2, 3, 4, 5], +# "morning_hours": [6, 7, 8], +# "work_hours": [9, 10, 11, 12, 13, 14, 15, 16, 17, 18], +# "reasoning": "Time configuration explanation for this specific event." +# }} + +# Field Descriptions: +# - total_simulation_hours (int): Total simulation duration, 24–168 hours. Short for breaking news, long for sustained topics. +# - minutes_per_round (int): Duration per round, 30–120 minutes, suggested 60 minutes. +# - agents_per_hour_min (int): Minimum activated agents per hour (Range: 1-{max_agents_allowed}). +# - agents_per_hour_max (int): Maximum activated agents per hour (Range: 1-{max_agents_allowed}). +# - peak_hours (int array): Peak hours, adjusted based on the participant group. +# - off_peak_hours (int array): Off-peak hours, usually late night/early morning. +# - morning_hours (int array): Morning hours. +# - work_hours (int array): Working hours. +# - reasoning (string): Brief explanation of why this configuration was chosen.""" - prompt = f"""Dựa vào yêu cầu Mô phỏng này, hãy tự động gen cho 1 file Thông số thời gian + prompt = f"""Dựa trên các yêu cầu mô phỏng dưới đây, hãy tạo cấu hình mô phỏng thời gian. {context_truncated} -## Task công việc -Vui lòng xuất ra kết quả Thời gian dưới định dạng format JSON +## Nhiệm vụ +Hãy tạo JSON cấu hình thời gian. -### Logic cơ bản có thể cần để tham khảo (Hãy dựa trên nhu cầu của user và hoàn cảnh để suy ra): -- Vị trí của người tham gia là người dùng mạng Trung Quốc, cần sinh hoạt bằng thói quen sinh học giờ chuẩn Bắc Kinh (China Time: GMT+8). -- Không xuất hiện hay có dấu hiệu online của người dùng từ 0-5 giờ sáng (Hệ số Active 0.05). -- Tăng nhẹ số lượng truy cập lại mức trung thành khoảng giữa 6-8 giờ sáng (Hệ số Active 0.4). -- Số lượng active hoạt động ở mức bình ổn khoảng từ 9-18 giờ sáng (Hệ số Active 0.7). -- Khung giờ sôi động nhất sẽ tập trung quanh 19-22 giờ tối (Hệ số Active 1.5). -- Tỷ lệ giảm lại sau 23 giờ (Hệ số Active 0.5). -- Cơ chế bình thường: Đêm không online, sáng bắt đầu đăng bài, giờ hành chính bình bình và cao trào trong buổi tối thức đêm -- **Chỉ Dẫn Rất Quan Trọng**: Những thông tin từ list được lấy để tham chiếu. Còn thông số thật sự còn phải tùy theo Đặc điểm, Tình Huống đối tượng ở Mạng và Thời Điểm Sự Kiện để gen ra - - Ví dụ: Số lượng sinh viên thức đêm ở từ 21-23 giờ thường là lớn; Media báo đài thì hay đăng tin liên tục cả ngày theo ca; Tài khoản văn phòng của các Cơ Quan chức năng chỉ trả lời giờ làm việc Hành Chính... - - Hoặc ví dụ: Các Biến Cố hoặc Drama xảy ra trong đêm khuya thì sẽ dẫn đến Lượng truy cập ban đêm có dấu hiệu đi lên, trong khi đó Off_peak_hours vì lẽ đó mà sẽ có khi co lại cho ngắn... +### Nguyên tắc cơ bản (Chỉ mang tính chất tham khảo, cần điều chỉnh linh hoạt theo sự kiện cụ thể và nhóm đối tượng tham gia): +- Nhóm người dùng là người Việt Nam, cần phù hợp với thói quen sinh hoạt theo giờ Hà Nội. +- 0-5 giờ sáng: Hầu như không có hoạt động (Hệ số hoạt động: 0.05). +- 6-8 giờ sáng: Hoạt động tăng dần (Hệ số hoạt động: 0.4). +- 9-18 giờ (Giờ làm việc): Hoạt động trung bình (Hệ số hoạt động: 0.7). +- 19-22 giờ tối: Giai đoạn cao điểm (Hệ số hoạt động: 1.5). +- Sau 23 giờ: Mức độ hoạt động giảm xuống (Hệ số hoạt động: 0.5). +- Quy luật chung: Thấp vào rạng sáng, tăng dần vào buổi sáng, trung bình trong giờ làm việc và cao điểm vào buổi tối. +- **Quan trọng**: Các giá trị ví dụ dưới đây chỉ mang tính chất tham khảo, bạn cần điều chỉnh các khung giờ cụ thể dựa trên tính chất sự kiện và đặc điểm của nhóm đối tượng tham gia. + - Ví dụ: Cao điểm của nhóm sinh viên có thể là 21-23 giờ; Nhóm truyền thông hoạt động cả ngày; Các cơ quan chính thống chỉ hoạt động trong giờ hành chính. + - Ví dụ: Tin tức nóng hổi (hotspot) có thể dẫn đến thảo luận vào đêm muộn; off_peak_hours có thể được rút ngắn tương ứng. -### Định dạng Format của JSON Return (Lưu ý Tuyệt đối Không Return Markdown code block, chỉ Return Format Json Thuần Túy) +### Định dạng JSON trả về (Không sử dụng markdown) Ví dụ Format như sau: {{ @@ -578,21 +623,23 @@ Ví dụ Format như sau: "off_peak_hours": [0, 1, 2, 3, 4, 5], "morning_hours": [6, 7, 8], "work_hours": [9, 10, 11, 12, 13, 14, 15, 16, 17, 18], - "reasoning": "Một đoạn văn lời nói cho biết Bạn đã dựa theo yêu cầu như thế nào để gen các Thông Số trên" + "reasoning": "Giải thích cấu hình thời gian cho sự kiện này." }} -Các Khóa của Json có nghĩa là: -- total_simulation_hours (int): Mô tả tổng giới hạn thời gian (Đơn vị giờ), Có giá trị trong khung 24-168h, Tùy vào biến cố Drama nóng để chọn. Các chủ đề chạy Drama ít hơn thì nên cấp ngắn -- minutes_per_round (int): Số Time trên mỗi Khung Đợt Thời Gian Thực Của Simulation để mô phỏng cho 1 phiên trong game, lấy giá trị 30-120 phút. Đề xuất: 60 (1 giờ) -- agents_per_hour_min (int): Số Agent online tối thiểu trong một tiếng mô phỏng (Phạm vi {1}-{max_agents_allowed}) -- agents_per_hour_max (int): Số lượng lên mạng tối đa (Phạm vi {1}-{max_agents_allowed}) -- peak_hours (mảng int list): Thời điểm đỉnh sóng Cao Điểm, cân nhắc theo Đối tượng để quyết định -- off_peak_hours (mảng int list): Đỉnh sóng Đáy, ít ai quan tâm -- morning_hours (mảng int list): Khoảng thời điểm đầu buổi sáng -- work_hours (mảng int list): Khung hành chính công việc -- reasoning (string): Sự giải thích từ LLM""" +Mô tả các trường: +- total_simulation_hours (int): Tổng thời gian mô phỏng, từ 24-168 giờ. Ngắn cho sự kiện đột xuất, dài cho các chủ đề kéo dài. +- minutes_per_round (int): Thời gian mỗi hiệp, 30-120 phút, khuyến nghị 60 phút. +- agents_per_hour_min (int): Số lượng Agent kích hoạt tối thiểu mỗi giờ (Phạm vi: 1-{max_agents_allowed}). +- agents_per_hour_max (int): Số lượng Agent kích hoạt tối đa mỗi giờ (Phạm vi: 1-{max_agents_allowed}). +- peak_hours (int array): Khung giờ cao điểm, điều chỉnh theo nhóm tham gia. +- off_peak_hours (int array): Khung giờ thấp điểm, thường là đêm khuya rạng sáng. +- morning_hours (mảng int): Khung giờ buổi sáng. +- work_hours (mảng int): Khung giờ làm việc. +- reasoning (string): Giải thích ngắn gọn lý do tại sao cấu hình như vậy.""" + + # system_prompt = "You are a social media simulation expert. Return in pure JSON format; time configurations must comply with Vietnamese daily routines." - system_prompt = "Bạn là 1 Tool chuyên mô phỏng môi trường làm việc trên mxh bằng thuật toán LLM để cung cấp ra Cấu hình Time. Hãy xuất JSON." + system_prompt = "Bạn là chuyên gia mô phỏng mạng xã hội. Trả về định dạng JSON thuần túy; cấu hình thời gian cần phù hợp với thói quen sinh hoạt của người Việt Nam." try: return self._call_llm_with_retry(prompt, system_prompt) @@ -604,14 +651,14 @@ Các Khóa của Json có nghĩa là: """Tạo sẵn file chuẩn nếu bị đơ để trả ra theo múi giờ chuẩn sinh hoạt China""" return { "total_simulation_hours": 72, - "minutes_per_round": 60, # 1 Hour / Vòng -> Rút gắn Time + "minutes_per_round": 60, # 1 Hour / Vòng -> Rút ngắn Time "agents_per_hour_min": max(1, num_entities // 15), "agents_per_hour_max": max(5, num_entities // 5), "peak_hours": [19, 20, 21, 22], "off_peak_hours": [0, 1, 2, 3, 4, 5], "morning_hours": [6, 7, 8], "work_hours": [9, 10, 11, 12, 13, 14, 15, 16, 17, 18], - "reasoning": "Mặc định sử dụng Thời gian làm việc của người dùng Trung Quốc (1 Giờ/vòng)" + "reasoning": "Defaults to Vietnamese users' daily routines and working hours (1 hour/round)" } def _parse_time_config(self, result: Dict[str, Any], num_entities: int) -> TimeSimulationConfig: @@ -678,37 +725,68 @@ Các Khóa của Json có nghĩa là: # Có chặn để lấy chuỗi theo cấu hình chiều dài giới hạn context_truncated = context[:self.EVENT_CONFIG_CONTEXT_LENGTH] - - prompt = f"""Gen cấu hình Event dưới các tham chiếu từ Yêu cầu (Requirements): -Simulation Requirements: {simulation_requirement} +# prompt = f"""Based on the following simulation requirements, generate an event configuration. + +# Simulation Requirements: {simulation_requirement} + +# {context_truncated} + +# ## Available Entity Types and Examples +# {type_info} + +# ## Task +# Please generate an event configuration JSON: +# - Extract key hot topic keywords. +# - Describe the direction of public opinion development. +# - Design initial post content; **each post must specify a poster_type (publisher type)**. + +# **IMPORTANT**: The poster_type must be selected from the "Available Entity Types" above so that initial posts can be assigned to the appropriate Agent for publishing. +# For example: Official statements should be posted by Official/University types, news by MediaOutlet, and student perspectives by Student. + +# Return in JSON format (no markdown): +# {{ +# "hot_topics": ["Keyword1", "Keyword2", ...], +# "narrative_direction": "", +# "initial_posts": [ +# {{"content": "Post Content...", "poster_type": "Entity type (must be selected from available types)"}}, +# ... +# ], +# "reasoning": "" +# }}""" + +# system_prompt = "You are a public opinion analysis expert. Return in pure JSON format. Ensure that poster_type exactly matches the available entity types." + + prompt = f"""Dựa trên các yêu cầu mô phỏng sau đây, hãy tạo cấu hình sự kiện. + +Yêu cầu mô phỏng: {simulation_requirement} {context_truncated} -## Các Entity Type có cung cấp & VD minh họa: +## Các loại thực thể khả dụng và ví dụ {type_info} -## Task công việc -Vui lòng xuất ra kết quả Thời gian dưới định dạng format JSON: -- Chỉ định List các Hot Keyword để kéo trend -- Miêu tả định hướng thảo luận cho trend hiện tại -- Đăng tải Post đầu tiên (Initial_Post) lên với nguyên tắc: **Phải đi kèm với tham số người up Post (poster_type)** +## Nhiệm vụ +Vui lòng tạo JSON cấu hình sự kiện: +- Trích xuất các từ khóa chủ đề nóng (hot topics). +- Mô tả hướng phát triển của dư luận. +- Thiết kế nội dung các bài đăng khởi tạo, **mỗi bài đăng phải chỉ định poster_type (loại người đăng)**. -**RẤT QUAN TRỌNG**: Người Poster Type (poster_type) Phải trùng khớp/được lấy từ danh mục từ mục "Các Entity Type" đã cho để gán. Tránh báo lỗi cho Agent - Ví dụ: official announcements should be posted by Official/University type, news by MediaOutlet, and student opinions by Student. +**QUAN TRỌNG**: poster_type phải được chọn từ "Các loại thực thể khả dụng" ở trên để các bài đăng khởi tạo có thể được phân bổ cho đúng Agent phù hợp. + Ví dụ: Các tuyên bố chính thức nên được đăng bởi loại Official/University, tin tức bởi MediaOutlet, và quan điểm sinh viên bởi Student. -Format trả ra (Tuyệt đối Không Markdown, chỉ lấy format chuỗi chuẩn): +Trả về định dạng JSON (không sử dụng markdown): {{ - "hot_topics": ["Keyword1", "Keyword2", ...], - "narrative_direction": "<Đoạn Text dài định hướng Dư luận (Narrative)>", + "hot_topics": ["Từ khóa 1", "Từ khóa 2", ...], + "narrative_direction": "", "initial_posts": [ - {{"content": "Post Content...", "poster_type": "Người sẽ Post ra nội dung (Hạn chế tùy tiện vì nó lấy từ mảng danh sách Loại Entity cho trước)"}}, + {{"content": "Nội dung bài đăng...", "poster_type": "Loại thực thể (phải chọn từ các loại khả dụng)"}}, ... ], - "reasoning": "" + "reasoning": "" }}""" - system_prompt = "Bạn là 1 Chuyên gia về Data Dư luận, Yêu cầu làm việc trên chuỗi JSON nghiêm ngặt. Tránh Lỗi." + system_prompt = "Bạn là chuyên gia phân tích dư luận. Trả về định dạng JSON thuần túy. Lưu ý rằng poster_type phải khớp chính xác với các loại thực thể khả dụng." try: return self._call_llm_with_retry(prompt, system_prompt) @@ -835,43 +913,81 @@ Format trả ra (Tuyệt đối Không Markdown, chỉ lấy format chuỗi chu "summary": e.summary[:summary_len] if e.summary else "" }) - prompt = f"""Tạo profile Social Media Activity Configs cho từng Thực thể sau. +# prompt = f"""Based on the following information, generate social media activity configurations for each entity. -Nhu cầu: {simulation_requirement} +# Simulation Requirements: {simulation_requirement} -## List các thực thể +# ## Entity List +# ```json +# {json.dumps(entity_list, ensure_ascii=False, indent=2)} +# ``` + +# ## Task +# Generate activity configurations for each entity, noting: +# - **Time aligns with Vietnamese daily routines**: Almost no activity between 0-5 AM; most active during 7-10 PM (19:00-22:00). +# - **Official Institutions (University/GovernmentAgency)**: Low activity (0.1-0.3), active during work hours (9:00-17:00), slow response (60-240 mins), high influence (2.5-3.0). +# - **Media (MediaOutlet)**: Medium activity (0.4-0.6), active all day (8:00-23:00), fast response (5-30 mins), high influence (2.0-2.5). +# - **Individuals (Student/Person/Alumni)**: High activity (0.6-0.9), active mainly in the evening (18:00-23:00), fast response (1-15 mins), low influence (0.8-1.2). +# - **Public Figures/Experts**: Medium activity (0.4-0.6), medium-high influence (1.5-2.0). + +# Return in JSON format (no markdown): +# {{ +# "agent_configs": [ +# {{ +# "agent_id": , +# "activity_level": <0.0-1.0>, +# "posts_per_hour": , +# "comments_per_hour": , +# "active_hours": [], +# "response_delay_min": , +# "response_delay_max": , +# "sentiment_bias": <-1.0 to 1.0>, +# "stance": "", +# "influence_weight": +# }}, +# ... +# ] +# }}""" + +# system_prompt = "You are a social media behavior analysis expert. Return pure JSON. Configurations must comply with Vietnamese daily routines." + + prompt = f"""Dựa trên các thông tin sau đây, hãy tạo cấu hình hoạt động trên mạng xã hội cho từng thực thể. + +Yêu cầu mô phỏng: {simulation_requirement} + +## Danh sách thực thể ```json {json.dumps(entity_list, ensure_ascii=False, indent=2)} ``` -## Task Công Việc -Trả ra cho Từng Entity các bộ Activity Profile tham chiều theo các quy tắc ngầm sau: -- **Tập quán Sinh hoạt Trung Quốc**: 0-5h sáng gần như sẽ hiếm ai onl, 19-22h tối lượng tương tác rất sôi nổi -- **Đại diện Cơ quan (University/GovernmentAgency)**: Tần suất (0.1-0.3), làm việc trong giờ hành chính (9-17h), delay hơi trễ (60-240 phút), Trọng lượng lời nói cao (2.5-3.0) -- **Truyền Thông Báo Đài (MediaOutlet)**: Tần suất TB (0.4-0.6), Hầu như online nguyên ngày (8-23h), Trễ ít (5-30 phút), Trọng lượng cũng Cao (2.0-2.5) -- **Người Dùng Bình thường (Student/Person/Alumni)**: Tần suất cao (0.6-0.9), Onl chủ yếu để cãi nhau buổi tối (18-23h), Tương tác lẹ như hack (1-15 min), Uy tín lời nói khá lèo tèo (0.8-1.2) -- **Học giả/Chuyên gia/Kols**: Tần suất TB (0.4-0.6), Uy tín tương đối (1.5-2.0) +## Nhiệm vụ +Tạo cấu hình hoạt động cho từng thực thể, lưu ý: +- **Thời gian phù hợp với thói quen của người Việt Nam**: Gần như không hoạt động từ 0-5 giờ sáng, hoạt động mạnh nhất từ 19-22 giờ tối. +- **Cơ quan chính thống (University/GovernmentAgency)**: Hoạt động thấp (0.1-0.3), hoạt động trong giờ làm việc (9:00-17:00), phản hồi chậm (60-240 phút), tầm ảnh hưởng cao (2.5-3.0). +- **Truyền thông (MediaOutlet)**: Hoạt động trung bình (0.4-0.6), hoạt động cả ngày (8:00-23:00), phản hồi nhanh (5-30 phút), tầm ảnh hưởng cao (2.0-2.5). +- **Cá nhân (Student/Person/Alumni)**: Hoạt động cao (0.6-0.9), hoạt động chủ yếu vào buổi tối (18:00-23:00), phản hồi nhanh (1-15 phút), tầm ảnh hưởng thấp (0.8-1.2). +- **Người công chúng/Chuyên gia**: Hoạt động trung bình (0.4-0.6), tầm ảnh hưởng trung bình cao (1.5-2.0). -Trả đúng 1 object JSON ko format MD: +Trả về định dạng JSON (không sử dụng markdown): {{ "agent_configs": [ {{ - "agent_id": , + "agent_id": , "activity_level": <0.0-1.0>, - "posts_per_hour": , - "comments_per_hour": , - "active_hours": [], - "response_delay_min": , - "response_delay_max": , - "sentiment_bias": <-1.0 đến 1.0 (Tiêu cực sang Tích Cực)>, + "posts_per_hour": , + "comments_per_hour": , + "active_hours": [], + "response_delay_min": <Độ trễ phản hồi tối thiểu tính bằng phút>, + "response_delay_max": <Độ trễ phản hồi tối đa tính bằng phút>, + "sentiment_bias": <-1.0 đến 1.0>, "stance": "", - "influence_weight": + "influence_weight": }}, ... ] }}""" - system_prompt = "Hệ thống Analysis Chuyên gia. Luôn trả về Format Object Array bằng JSON. Và tuân thủ sinh học." + system_prompt = "Bạn là chuyên gia phân tích hành vi mạng xã hội. Trả về JSON thuần túy. Cấu hình phải phù hợp với thói quen sinh hoạt của người Việt Nam." try: result = self._call_llm_with_retry(prompt, system_prompt) diff --git a/backend/app/services/zep_tools.py b/backend/app/services/zep_tools.py index d5230d2f..4de8331b 100644 --- a/backend/app/services/zep_tools.py +++ b/backend/app/services/zep_tools.py @@ -1109,23 +1109,41 @@ class ZepToolsService: Giúp phân rã một câu hỏi lớn / phức tạp thành nhiều câu hỏi nhỏ lẻ có thể query độc lập trên cơ sở dữ liệu. """ - system_prompt = """You are a professional problem analysis expert. Your task is to break down a complex query into multiple sub-queries that can be independently observed in the simulated world. +# system_prompt = """You are a professional problem analysis expert. Your task is to decompose a complex problem into multiple sub-questions that can be independently observed within a simulated world. -Requirements: -1. Each sub-query should be specific enough to find concrete Agent behaviors or events. -2. Sub-queries should cover different dimensions of the original query (Who, What, Why, How, When, Where). -3. Sub-queries must relate to the simulation context. -4. Return exactly in JSON format: {"sub_queries": ["sub_query 1", "sub_query 2", ...]}""" +# Requirements: +# 1. Each sub-question should be specific enough to identify relevant Agent behaviors or events within the simulation. +# 2. Sub-questions should cover different dimensions of the original problem (e.g., Who, What, Why, How, When, Where). +# 3. Sub-questions must be relevant to the simulation scenario. +# 4. Return in JSON format: {"sub_queries": ["Sub-question 1", "Sub-question 2", ...]}""" - user_prompt = f"""Simulation background: +# user_prompt = f"""Simulation Requirement Background: +# {simulation_requirement} + +# {f"Report context: {report_context[:500]}" if report_context else ""} + +# Please decompose the following question into {max_queries} sub-questions: +# {query} + +# Return the list of sub-questions in JSON format.""" + + system_prompt = """Bạn là một chuyên gia phân tích vấn đề chuyên nghiệp. Nhiệm vụ của bạn là chia nhỏ một vấn đề phức tạp thành nhiều câu hỏi phụ có thể quan sát độc lập trong thế giới mô phỏng. + +Yêu cầu: +1. Mỗi câu hỏi phụ phải đủ cụ thể để có thể tìm thấy các hành vi của Agent hoặc các sự kiện liên quan trong thế giới mô phỏng. +2. Các câu hỏi phụ nên bao quát các khía cạnh khác nhau của vấn đề gốc (Ví dụ: Ai, Cái gì, Tại sao, Như thế nào, Khi nào, Ở đâu). +3. Các câu hỏi phụ phải liên quan đến kịch bản mô phỏng. +4. Trả về định dạng JSON: {"sub_queries": ["Câu hỏi phụ 1", "Câu hỏi phụ 2", ...]}""" + + user_prompt = f"""Bối cảnh yêu cầu mô phỏng: {simulation_requirement} -{f"Report context: {report_context[:500]}" if report_context else ""} +{f"Ngữ cảnh báo cáo: {report_context[:500]}" if report_context else ""} -Please break down the following query into {max_queries} sub-queries: +Hãy chia nhỏ vấn đề sau đây thành {max_queries} các câu hỏi phụ: {query} -Return the JSON format.""" +Trả về danh sách các câu hỏi phụ dưới định dạng JSON.""" try: response = self.llm.chat_json( @@ -1358,16 +1376,28 @@ Return the JSON format.""" combined_prompt = "\n".join([f"{i+1}. {q}" for i, q in enumerate(result.interview_questions)]) # Thêm các prefix tối ưu hoá, ràng buộc format câu trả lời của Agent + # INTERVIEW_PROMPT_PREFIX = ( + # "You are currently being interviewed. Please combine your persona, " + # "all past memories, and actions to answer the following questions directly in plain text.\n" + # "Response Requirements:\n" + # "1. Answer directly using natural language; do not call any tools.\n" + # "2. Do not return JSON format or tool-call formats.\n" + # "3. Do not use Markdown headers (e.g., #, ##, ###).\n" + # "4. Answer questions one by one according to their numbers. Each answer must start with 'Question X:' (where X is the question number).\n" + # "5. Use a blank line to separate the answers for each question.\n" + # "6. Answers must have substantive content; provide at least 2-3 sentences for each question.\n\n" + # ) + INTERVIEW_PROMPT_PREFIX = ( - "You are being interviewed. Please combine your profile, all your past memories and actions, " - "and directly answer the following questions in pure text.\n" - "Reply requirements:\n" - "1. Answer directly in natural language, do not call any tools.\n" - "2. Do not return JSON format or tool call formats.\n" - "3. Do not use Markdown headers (like #, ##, ###).\n" - "4. Answer questions one by one according to their numbers, start each answer with 'Question X:' (X is the number).\n" - "5. Separate each answer with a blank line.\n" - "6. Answers must have substance, at least 2-3 sentences per question.\n\n" + "Bạn đang tham gia một buổi phỏng vấn. Hãy kết hợp nhân vật (persona), " + "tất cả ký ức và hành động trong quá khứ của bạn để trả lời trực tiếp các câu hỏi dưới đây bằng văn bản thuần túy.\n" + "Yêu cầu phản hồi:\n" + "1. Trả lời trực tiếp bằng ngôn ngữ tự nhiên, không gọi bất kỳ công cụ nào.\n" + "2. Không trả về định dạng JSON hoặc định dạng gọi công cụ.\n" + "3. Không sử dụng tiêu đề Markdown (như #, ##, ###).\n" + "4. Trả lời từng câu một theo số thứ tự, mỗi câu trả lời bắt đầu bằng 'Câu hỏi X:' (X là số thứ tự câu hỏi).\n" + "5. Sử dụng một dòng trống để phân tách giữa các câu trả lời.\n" + "6. Câu trả lời phải có nội dung thực tế, mỗi câu hỏi cần trả lời ít nhất 2-3 câu văn.\n\n" ) optimized_prompt = f"{INTERVIEW_PROMPT_PREFIX}{combined_prompt}" @@ -1585,31 +1615,56 @@ Return the JSON format.""" "interested_topics": profile.get("interested_topics", []) } agent_summaries.append(summary) - - system_prompt = """You are a professional interview planning expert. Your task is to select the most suitable target agents for an interview based on requirements. -Selection criteria: -1. Agent identity/profession is related to the interview topic. -2. Agent might hold unique or valuable opinions. -3. Select diverse perspectives (e.g., supporters, opponents, neutrals, professionals, etc.). -4. Prioritize characters directly related to the event. + system_prompt = """Bạn là một chuyên gia lập kế hoạch phỏng vấn chuyên nghiệp. Nhiệm vụ của bạn là dựa trên yêu cầu phỏng vấn để chọn ra những đối tượng phù hợp nhất từ danh sách các Agent mô phỏng. -Return in JSON format: +Tiêu chí lựa chọn: +1. Danh tính/Nghề nghiệp của Agent phải liên quan đến chủ đề phỏng vấn. +2. Agent có khả năng nắm giữ những quan điểm độc đáo hoặc có giá trị. +3. Lựa chọn các góc nhìn đa dạng (Ví dụ: bên ủng hộ, bên phản đối, bên trung lập, chuyên gia, v.v.). +4. Ưu tiên các nhân vật có liên quan trực tiếp đến sự kiện. + +Trả về định dạng JSON: { - "selected_indices": [array of selected agent indices], - "reasoning": "explanation for the selection" + "selected_indices": [Danh sách chỉ mục của các Agent được chọn], + "reasoning": "Giải thích lý do lựa chọn" }""" - user_prompt = f"""Interview requirements: + user_prompt = f"""Yêu cầu phỏng vấn: {interview_requirement} -Simulation background: +Bối cảnh mô phỏng: {simulation_requirement if simulation_requirement else "Not provided"} -Available Agents (total {len(agent_summaries)}): +Danh sách các Agent có thể chọn (Tổng cộng {len(agent_summaries)}): {json.dumps(agent_summaries, ensure_ascii=False, indent=2)} -Please select up to {max_agents} most suitable agents for the interview and explain your reasoning.""" +Hãy chọn tối đa {max_agents} Agent phù hợp nhất để phỏng vấn và giải thích lý do lựa chọn của bạn.""" + +# system_prompt = """You are a professional interview planning expert. Your task is to select the most suitable subjects for an interview from a list of simulated Agents based on the interview requirements. + +# Selection Criteria: +# 1. The Agent's identity/profession must be relevant to the interview topic. +# 2. The Agent is likely to hold a unique or valuable perspective. +# 3. Select diverse perspectives (e.g., supporters, opponents, neutral parties, professionals, etc.). +# 4. Prioritize characters directly related to the event. + +# Return in JSON format: +# { +# "selected_indices": [List of indices of selected Agents], +# "reasoning": "Explanation for the selection" +# }""" + +# user_prompt = f"""Interview Requirements: +# {interview_requirement} + +# Simulation Background: +# {simulation_requirement if simulation_requirement else "Not provided"} + +# List of selectable Agents (Total {len(agent_summaries)}): +# {json.dumps(agent_summaries, ensure_ascii=False, indent=2)} + +# Please select a maximum of {max_agents} most suitable Agents for the interview and explain the reasons for your selection.""" try: response = self.llm.chat_json( @@ -1649,27 +1704,47 @@ Please select up to {max_agents} most suitable agents for the interview and expl """Sử dụng LLM để sinh ra các câu chất vấn hợp với tính chất sự việc""" agent_roles = [a.get("profession", "Unknown") for a in selected_agents] + +# system_prompt = """You are a professional journalist/interviewer. Based on the interview requirements, generate 3-5 in-depth interview questions. + +# Question Requirements: +# 1. Open-ended questions that encourage detailed answers. +# 2. Questions that may elicit different answers from different roles. +# 3. Cover multiple dimensions such as facts, opinions, and feelings. +# 4. Use natural language, sounding like a real-life interview. +# 5. Keep each question under 50 words, concise and clear. +# 6. Ask the questions directly; do not include background explanations or prefixes. + +# Return in JSON format: {"questions": ["Question 1", "Question 2", ...]}""" + +# user_prompt = f"""Interview Requirements: {interview_requirement} + +# Simulation Background: {simulation_requirement if simulation_requirement else "Not provided"} + +# Interviewee Roles: {', '.join(agent_roles)} + +# Please generate 3-5 interview questions.""" - system_prompt = """You are a professional journalist/interviewer. Generate 3-5 deep interview questions based on requirements. + system_prompt = """Bạn là một nhà báo/người phỏng vấn chuyên nghiệp. Dựa trên yêu cầu phỏng vấn, hãy tạo từ 3-5 câu hỏi phỏng vấn chuyên sâu. -Question requirements: -1. Open-ended questions, encourage detailed answers. -2. Formulated so different roles might have different answers. -3. Cover multiple dimensions like facts, opinions, feelings, etc. -4. Natural language, sounds like a real interview. -5. Keep each question within 50 words, concise and clear. -6. Ask directly, do not include background explanations or prefixes. +Yêu cầu đối với câu hỏi: +1. Câu hỏi mở, khuyến khích câu trả lời chi tiết. +2. Các câu hỏi có thể nhận được những câu trả lời khác nhau tùy theo từng vai trò. +3. Bao quát nhiều khía cạnh như sự thật, quan điểm và cảm xúc. +4. Ngôn ngữ tự nhiên, giống như một buổi phỏng vấn thực tế. +5. Mỗi câu hỏi khống chế dưới 50 chữ, ngắn gọn và súc tích. +6. Đặt câu hỏi trực tiếp, không bao gồm giải thích bối cảnh hoặc tiền tố. -Return in JSON format: {"questions": ["question 1", "question 2", ...]}""" +Trả về định dạng JSON: {"questions": ["Câu hỏi 1", "Câu hỏi 2", ...]}""" - user_prompt = f"""Interview requirements: {interview_requirement} + user_prompt = f"""Yêu cầu phỏng vấn: {interview_requirement} -Simulation background: {simulation_requirement if simulation_requirement else "Not provided"} +Bối cảnh mô phỏng: {simulation_requirement if simulation_requirement else "Not provided"} -Interviewee roles: {', '.join(agent_roles)} - -Please generate 3-5 interview questions.""" +Vai trò của đối tượng phỏng vấ: {', '.join(agent_roles)} +Hãy tạo từ 3-5 câu hỏi phỏng vấn.""" + try: response = self.llm.chat_json( messages=[ @@ -1703,29 +1778,52 @@ Please generate 3-5 interview questions.""" interview_texts = [] for interview in interviews: interview_texts.append(f"【{interview.agent_name}({interview.agent_role})】\n{interview.response[:500]}") + +# system_prompt = """You are a professional news editor. Please generate an interview summary based on the responses from multiple interviewees. + +# Summary Requirements: +# 1. Synthesize the main viewpoints from all parties. +# 2. Identify areas of consensus and divergence among the viewpoints. +# 3. Highlight valuable quotes. +# 4. Maintain objectivity and neutrality, without favoring any side. +# 5. Limit the summary to 1000 words. + +# Formatting Constraints (MUST be followed): +# - Use plain text paragraphs, separated by blank lines. +# - Do not use Markdown headers (e.g., #, ##, ###). +# - Do not use horizontal rules (e.g., ---, ***). +# - Use Vietnamese quotation marks 「」 when quoting the interviewees' original words. +# - You may use **bold** to highlight keywords, but do not use any other Markdown syntax.""" + +# user_prompt = f"""Interview Topic: {interview_requirement} + +# Interview Content: +# {"\n\n".join(interview_texts)} + +# Please generate the interview summary.""" - system_prompt = """You are a professional news editor. Please generate an interview summary based on the answers from multiple interviewees. + system_prompt = """Bạn là một biên tập viên tin tức chuyên nghiệp. Hãy tạo một bản tóm tắt phỏng vấn dựa trên câu trả lời từ nhiều đối tượng được phỏng vấn. -Summary requirements: -1. Extract main viewpoints from all parties. -2. Point out consensus and disagreements among opinions. -3. Highlight valuable quotes. -4. Objective and neutral, do not favor any side. -5. Keep it within 1000 words. +Yêu cầu tóm tắt: +1. Đúc kết các quan điểm chính của các bên. +2. Chỉ ra những điểm đồng thuận và khác biệt trong các quan điểm. +3. Làm nổi bật các câu trích dẫn có giá trị. +4. Đảm bảo tính khách quan và trung lập, không thiên vị bất kỳ bên nào. +5. Giới hạn trong khoảng 1000 chữ. -Formatting constraints (Must obey): -- Use plain text paragraphs, separate different sections with blank lines. -- Do not use Markdown headers (like #, ##, ###). -- Do not use dividers (like ---, ***). -- Use normal quotes when citing interviewee actions/words. -- You can use **bold** to mark keywords, but no other Markdown syntax.""" +Ràng buộc định dạng (BẮT BUỘC tuân thủ): +- Sử dụng các đoạn văn bản thuần túy, phân tách các phần bằng dòng trống. +- Không sử dụng tiêu đề Markdown (như #, ##, ###). +- Không sử dụng đường kẻ phân cách (như ---, ***). +- Sử dụng dấu ngoặc kép kiểu Việt Nam 「」 khi trích dẫn nguyên văn lời của người được phỏng vấn. +- Có thể sử dụng dấu **in đậm** cho các từ khóa, nhưng không sử dụng bất kỳ cú pháp Markdown nào khác.""" - user_prompt = f"""Interview topic: {interview_requirement} + user_prompt = f"""Chủ đề phỏng vấn: {interview_requirement} -Interview content: +Nội dung phỏng vấn: {"\n\n".join(interview_texts)} -Please generate the interview summary.""" +Hãy tạo bản tóm tắt phỏng vấn.""" try: summary = self.llm.chat( diff --git a/backend/scripts/test_profile_format.py b/backend/scripts/test_profile_format.py index a2029b21..6e96f451 100644 --- a/backend/scripts/test_profile_format.py +++ b/backend/scripts/test_profile_format.py @@ -38,7 +38,7 @@ def test_profile_formats(): age=25, gender="male", mbti="INTJ", - country="China", + country="Vietnam", profession="Student", interested_topics=["Technology", "Education"], source_entity_uuid="test-uuid-123",