From 8032288d737fa1f98989bad32b7b1256fba53040 Mon Sep 17 00:00:00 2001 From: Armor <2654988228@qq.com> Date: Mon, 5 Jan 2026 21:14:28 +0800 Subject: [PATCH] update --- config/config.json | 1 + src/{TG-tools => TG_tools}/__init__.py | 0 .../history_message_upload.py | 104 ++++++--- src/log_tool/__init__.py | 0 src/log_tool/analysis_via_history_message.py | 119 ++++++++++ src/log_tool/log_analysis_via_LLM.py | 220 ++++++++++++++++++ 6 files changed, 412 insertions(+), 32 deletions(-) rename src/{TG-tools => TG_tools}/__init__.py (100%) rename src/{TG-tools => TG_tools}/history_message_upload.py (63%) create mode 100644 src/log_tool/__init__.py create mode 100644 src/log_tool/analysis_via_history_message.py create mode 100644 src/log_tool/log_analysis_via_LLM.py diff --git a/config/config.json b/config/config.json index 48b9291..053ff27 100644 --- a/config/config.json +++ b/config/config.json @@ -4,6 +4,7 @@ "host": "", "port": "", "file_db_path": "", + "database": "", "db_host": "", "db_port": "", diff --git a/src/TG-tools/__init__.py b/src/TG_tools/__init__.py similarity index 100% rename from src/TG-tools/__init__.py rename to src/TG_tools/__init__.py diff --git a/src/TG-tools/history_message_upload.py b/src/TG_tools/history_message_upload.py similarity index 63% rename from src/TG-tools/history_message_upload.py rename to src/TG_tools/history_message_upload.py index 8767a65..6f069d7 100644 --- a/src/TG-tools/history_message_upload.py +++ b/src/TG_tools/history_message_upload.py @@ -127,47 +127,87 @@ class MessageUploader: } for file_param in self.file_param_list: if file_param in message: - file_location = self.transfer_file_to_db(message[file_param]) + # 尝试复制文件到文件库,transfer_file_to_db 在失败时返回 None + file_path = message[file_param] + file_location = self.transfer_file_to_db(file_path) target['file_location'] = file_location - target['file_size_raw'] = os.path.getsize(os.path.join(self.load_path, message[file_param])) - target['file_size'] = size_of_file(target['file_size_raw']) - if file_param == 'photo': - target['file_type'] = file_param - elif 'sticker_emoji' in message: - target['file_type'] = 'sticker' - elif 'media_type' in message: - target['file_type'] = message['media_type'] + if file_location: + source_path = os.path.join(self.load_path, file_path) + try: + file_size_raw = os.path.getsize(source_path) + target['file_size_raw'] = file_size_raw + target['file_size'] = size_of_file(file_size_raw) + except Exception: + # 无法获取大小,置为 None 并记录 + self.logger.warning(f'Could not get size for source file: {source_path}') + target['file_size_raw'] = None + target['file_size'] = None + if file_param == 'photo': + target['file_type'] = file_param + elif 'sticker_emoji' in message: + target['file_type'] = 'sticker' + elif 'media_type' in message: + target['file_type'] = message['media_type'] + else: + target['file_type'] = 'file' else: - target['file_type'] = 'file' + # 复制失败或文件不存在,保持 file_* 字段为 None + self.logger.warning(f'File not copied, skipped file field for message id {message.get("id")}') target_list.append(target) self.logger.info(f'Data process complete, time: {time.time() - start_time}') return target_list def transfer_file_to_db(self, file_path): - source_path = os.path.join(self.load_path, file_path) - destination_path = os.path.join(self.file_db_path, file_path) - destination_folder, file_name = os.path.split(destination_path) - new_file_name = file_name - # 路径不存在则创建路径 - if not os.path.exists(destination_folder): - os.mkdir(destination_folder) + # 防御性检查:file_path 不是字符串或是占位文本时直接返回 None + try: + if not isinstance(file_path, str): + self.logger.warning(f'Invalid file path type: {type(file_path)}') + return None + low = file_path.lower() + if 'exceeds maximum size' in low or 'file exceeds' in low: + self.logger.warning(f'File path appears to be placeholder (too large): {file_path}') + return None - # 已存在相同文件名的情况 - if os.path.exists(destination_path): - # 如果文件大小不相等,或者大小相同但不是sticker文件,直接进行复制操作。 - if (os.path.getsize(source_path) != os.path.getsize(destination_path) or - ('sticker' not in source_path) and - os.path.getsize(source_path) == os.path.getsize(destination_path)): - i = 0 - while os.path.exists(destination_path): - new_file_name = f"{i}_{file_name}" - destination_path = os.path.join(destination_folder, new_file_name) - i += 1 - # 如果是sticker文件,则直接进行映射(sticker文件出错问题应该不大) + source_path = os.path.join(self.load_path, file_path) + if not os.path.exists(source_path): + self.logger.warning(f'Source file does not exist: {source_path}') + return None - shutil.copy2(source_path, destination_path) - return file_path.replace(file_name, new_file_name) + destination_path = os.path.join(self.file_db_path, file_path) + destination_folder, file_name = os.path.split(destination_path) + new_file_name = file_name + + # 确保目标目录存在 + os.makedirs(destination_folder, exist_ok=True) + + # 已存在相同文件名的情况 + if os.path.exists(destination_path): + try: + src_size = os.path.getsize(source_path) + dst_size = os.path.getsize(destination_path) + except Exception: + src_size = dst_size = None + # 如果文件大小不相等,或大小相同但非 sticker,改名复制 + if (src_size is not None and dst_size is not None and src_size != dst_size) or ( + src_size is not None and dst_size is not None and src_size == dst_size and 'sticker' not in source_path): + i = 0 + base_name = file_name + while os.path.exists(destination_path): + new_file_name = f"{i}_{base_name}" + destination_path = os.path.join(destination_folder, new_file_name) + i += 1 + + try: + shutil.copy2(source_path, destination_path) + except Exception as e: + self.logger.warning(f'Failed to copy file {source_path} to {destination_path}: {e}') + return None + + return file_path.replace(file_name, new_file_name) + except Exception as e: + self.logger.exception(f'Unexpected error in transfer_file_to_db for {file_path}: {e}') + return None def logger_test(self): self.logger.info('this is a test') @@ -203,7 +243,7 @@ def upload_didi_saved_message_to_db(): if __name__ == "__main__": tg = MessageUploader( local_logger=LoggerClass().get_log(logging.DEBUG), - local_load_path='E:\\SNSFiles\\ChatExport_2024-07-10', + local_load_path='E:\\SNSFiles\\ChatExport_2026-01-01', local_file_db_path=project_config['file_db_path'], size_limit=0.25 ) diff --git a/src/log_tool/__init__.py b/src/log_tool/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/src/log_tool/analysis_via_history_message.py b/src/log_tool/analysis_via_history_message.py new file mode 100644 index 0000000..4481cdd --- /dev/null +++ b/src/log_tool/analysis_via_history_message.py @@ -0,0 +1,119 @@ +from typing import Optional +import pandas as pd +from langchain_openai import ChatOpenAI + +from src.PostgresSQL.db_handler import PgHandler +from src.config.load_config import project_config + +LLM = ChatOpenAI( + model="qwen-plus-latest", + api_key=project_config.get('bailian_API_key', ''), + base_url=project_config.get('bailian_API_URL', ''), +) + + +def load_history_by_year_month(year: int, month: Optional[int] = None, + table_name: str = '王适意_personal_chat', + schema: str = 'public', + db_config: Optional[dict] = None) -> pd.DataFrame: + """从 Postgres 中读取历史消息并按年/月过滤。 + + 参数: + - year: 年份(四位数) + - month: 可选的月份(1-12),若提供则只返回该月的数据 + - table_name: 数据库表名,默认 '王适意_personal_chat' + - schema: schema,默认 'public' + - db_config: 可选的数据库连接参数字典(含 host, port, user, password, database) + + 返回:pandas.DataFrame,包含原始列并新增一列'dt'(datetime 类型)用于时间筛选与排序。 + """ + # 使用传入的 db_config 或 project_config + cfg = db_config or project_config + host = cfg.get('host') or cfg.get('db_host') + port = int(cfg.get('port') or cfg.get('db_port') or 5432) + user = cfg.get('user') or cfg.get('db_user') + password = cfg.get('password') or cfg.get('db_password') + database = cfg.get('database') or cfg.get('db_name') + + pg = PgHandler(host=host, port=port, user=user, password=password, database=database) + try: + if not pg.table_exists(table_name, schema=schema): + raise ValueError(f"table {schema}.{table_name} not exists") + df = pg.read_table(table_name, schema=schema) + finally: + pg.close() + + if df.empty: + return df + + # 尝试解析时间列:优先 message_datetime,再尝试 message_unix + dt = None + if 'message_datetime' in df.columns: + try: + dt = pd.to_datetime(df['message_datetime'], errors='coerce') + except Exception: + dt = None + if (dt is None or dt.isna().all()) and 'message_unix' in df.columns: + # message_unix 可能为字符串或数值 + try: + dt = pd.to_datetime(df['message_unix'].astype(float), unit='s', errors='coerce') + except Exception: + dt = pd.to_datetime(df['message_unix'], errors='coerce') + + if dt is None: + # 无法解析时间,返回原表 + return df + + df = df.copy() + df['dt'] = dt + # 过滤年份和可选月份 + df = df[df['dt'].dt.year == int(year)] + if month is not None: + df = df[df['dt'].dt.month == int(month)] + + df = df.sort_values('dt').reset_index(drop=True) + return df + +def do_analysis_on_history_messages(df: pd.DataFrame): + data = df.to_dict(orient='records') + to_model_data = [ + { + 'message_datetime': item.get('message_datetime'), + 'message_from': item.get('message_from'), + 'message_text': item.get('message_text'), + 'file_type': item.get('file_type'), + } + for item in data + ] + to_model_data = to_model_data + prompt = f""" +你是一个聊天记录分析专家,擅长根据用户的聊天记录,分析用户的行为,然后撰写报告。 +你的报告应当包含以下内容: +1. 以天为单位,分析用户在这一天内的主要活动和重大事件。每一天的总结应当独立成段。 +2. 按照月时间维度,输出总结。 + +你的总结应当包括但不限于以下内容: +- 用户在这个月内的主要活动和重大事件。 +- 用户的情绪变化和心理状态。 +- 用户的兴趣爱好和关注点。 +- 用户的社交活动和人际关系。 +- 用户的生活习惯和日常规律。 +- 任何其他你认为重要的信息。 + +注意:不要对于进行过度的推断,只基于用户的聊天记录内容进行分析。 + +注意:用户是Z Armor,与他对话的是王适意,与用户是恋人关系。司,王适意,Skuld都是指王适意;装,Armor,周义人都是指Z Armor。一定要搞清楚双方的身份关系,message_from字段指的是,这个消息是谁发的。 + +以下是用户的日记内容: +{to_model_data} + """ + llm_resp = LLM.invoke(prompt) + print(llm_resp.content) + return llm_resp.content + + +if __name__ == "__main__": + df_month = load_history_by_year_month(2025, month=1) + summary = do_analysis_on_history_messages(df_month) + print(f"Loaded {len(df_month)} messages for May 2023") + diff --git a/src/log_tool/log_analysis_via_LLM.py b/src/log_tool/log_analysis_via_LLM.py new file mode 100644 index 0000000..c854797 --- /dev/null +++ b/src/log_tool/log_analysis_via_LLM.py @@ -0,0 +1,220 @@ +import os +import json + +from langchain_openai import ChatOpenAI +from pathlib import Path + +from src.config.load_config import project_config + +PROJECT_PATH = Path(__file__).resolve().parent.parent.parent.absolute() +DATA_PATH = os.path.join(PROJECT_PATH, 'data') + +LLM = ChatOpenAI( + model="qwen-plus-latest", + api_key=project_config.get('bailian_API_key', ''), + base_url=project_config.get('bailian_API_URL', ''), +) # type: ignore + + +def load_year_logs(year: int = 2025, base_data_path: str | None = None) -> dict: + """读取 data/Log/ 下的 .md 文件,返回结构化字典。 + + 返回示例结构: + { + 'year': 2025, + 'year_summary': '...', + 'months': { + '01': { + 'monthly_summary': '...', + 'entries': { + '2025-01-01': '...' + }, + 'other_files': { 'notes.md': '...' } + }, + ... + }, + 'other_files': [ { 'path': 'Log/2025/some.md', 'content': '...' } ] + } + + 规则说明: + - 识别形如 YYYY-MM-DD.md 的为按日条目,日期作为键。 + - 识别月份名称(January..December)或数字为月度总结(例如 December.md 或 12.md)。 + - 当文件位于月份子目录(例如 .../2025/December/*.md)时,也会归到对应月份。 + - 未能确定归属的文件会被放入 other_files。 + """ + import re + from pathlib import Path + + if base_data_path is None: + base_data_path = DATA_PATH + + year_path = Path(base_data_path) / 'Log' / str(year) + if not year_path.exists(): + return {} + + month_map = { + 'january': '01', 'february': '02', 'march': '03', 'april': '04', + 'may': '05', 'june': '06', 'july': '07', 'august': '08', + 'september': '09', 'october': '10', 'november': '11', 'december': '12' + } + + def month_name_to_num(name: str) -> str | None: + if not name: + return None + s = name.lower().strip() + s = s.replace('.md', '') + # 直接数字 + if s.isdigit(): + n = int(s) + if 1 <= n <= 12: + return f"{n:02d}" + # 完整英文月份 + if s in month_map: + return month_map[s] + # 缩写或开头匹配 + for k in month_map: + if k.startswith(s): + return month_map[k] + return None + + def ensure_month(res: dict, m: str): + if m not in res['months']: + res['months'][m] = {'monthly_summary': None, 'entries': {}, 'other_files': {}} + + def parse_date_from_stem(stem: str, default_year: int | None = None) -> str | None: + """尝试从文件名 stem 中解析日期,支持: + - YYYY-MM-DD + - YY-MM-DD(自动扩展为 2000+YY) + 如果匹配成功返回 'YYYY-MM-DD',否则返回 None。 + """ + import re + + # 允许后缀,比如 '25-08-11 Mon',只取开头部分进行解析 + m = re.match(r'^(?P\d{2,4})-(?P\d{1,2})-(?P\d{1,2})', stem) + if not m: + return None + y = m.group('y') + mo = int(m.group('m')) + d = int(m.group('d')) + if len(y) == 2: + yy = int(y) + yyyy = 2000 + yy + else: + yyyy = int(y) + try: + return f"{yyyy:04d}-{mo:02d}-{d:02d}" + except Exception: + return None + + res: dict = {'year': year, 'year_summary': None, 'months': {}, 'other_files': []} + + # 遍历年份目录下的文件和子目录 + for item in sorted(year_path.iterdir()): + if item.is_file() and item.suffix.lower() == '.md': + stem = item.stem + try: + text = item.read_text(encoding='utf-8') + except Exception: + text = item.read_text(encoding='utf-8', errors='ignore') + + # YYYY-MM-DD.md or YY-MM-DD... -> 日志条目 + parsed = parse_date_from_stem(stem) + if parsed: + month = parsed[5:7] + ensure_month(res, month) + res['months'][month]['entries'][parsed] = text + continue + + # Month name like December.md + mn = month_name_to_num(stem) + if mn: + ensure_month(res, mn) + # 将视为当月的月度记录 + res['months'][mn]['monthly_summary'] = text + continue + + # 整体年度总结 + if stem.lower() == 'summary' or 'summary' in stem.lower(): + # 若带有月份名称则上面已经捕获,这里为年度/通用总结 + res['year_summary'] = text + continue + + # 其他无法分类的文件 + res['other_files'].append({'path': str(item.relative_to(base_data_path)), 'content': text}) + + elif item.is_dir(): + # 目录通常是月份目录 + mnum = month_name_to_num(item.name) + for f in sorted(item.glob('*.md')): + stem = f.stem + try: + text = f.read_text(encoding='utf-8') + except Exception: + text = f.read_text(encoding='utf-8', errors='ignore') + + # summary 文件 + if 'summary' in stem.lower(): + if mnum: + ensure_month(res, mnum) + res['months'][mnum]['monthly_summary'] = text + else: + res['other_files'].append({'path': str(f.relative_to(base_data_path)), 'content': text}) + continue + + # 完整日期格式 + # 支持 YYYY-MM-DD 或 YY-MM-DD 开头(允许后缀,例如 '25-08-11 Mon') + parsed = parse_date_from_stem(stem, default_year=year) + if parsed: + date = parsed + month = date[5:7] + ensure_month(res, month) + res['months'][month]['entries'][date] = text + continue + + # 仅为日号,如 1.md, 31.md + if re.match(r'^\d{1,2}$', stem): + if mnum: + day = int(stem) + date = f"{year}-{mnum}-{day:02d}" + ensure_month(res, mnum) + res['months'][mnum]['entries'][date] = text + else: + res['other_files'].append({'path': str(f.relative_to(base_data_path)), 'content': text}) + continue + + # 其他文件,放入该月份的 other_files + if mnum: + ensure_month(res, mnum) + res['months'][mnum]['other_files'][stem] = text + else: + res['other_files'].append({'path': str(f.relative_to(base_data_path)), 'content': text}) + + return res + + +if __name__ == "__main__": + year_logs = load_year_logs(2025) + + prompt = f""" +你是一个日记分析专家,擅长根据用户的日记,分析用户的行为,按照月和年为时间维度,输出总结。 + +你的总结应当包括但不限于以下内容: +1. 用户在这个时间段内的主要活动和重大事件。 +2. 用户的情绪变化和心理状态。 +3. 用户的兴趣爱好和关注点。 +4. 用户的社交活动和人际关系。 +5. 用户的生活习惯和日常规律。 +6. 任何其他你认为重要的信息。 + +注意:不要对于进行过度的推断,只基于用户的日记内容进行分析。 + +你的输出应当包含分月份的总结和整体的年度总结。 + +以下是用户的日记内容: +{json.dumps(year_logs, ensure_ascii=False)} + """ + + llm_resp = LLM.invoke(prompt) + print(llm_resp.content) + + print('done') \ No newline at end of file