update
This commit is contained in:
Binary file not shown.
@@ -27,8 +27,8 @@ from openpyxl.worksheet.worksheet import Worksheet
|
|||||||
from openpyxl.cell.cell import Cell, MergedCell
|
from openpyxl.cell.cell import Cell, MergedCell
|
||||||
|
|
||||||
# Paths and constants
|
# Paths and constants
|
||||||
PROJECT_ROOT = Path(__file__).resolve().parents[2]
|
PROJECT_ROOT = Path(__file__).resolve().parents[4]
|
||||||
LOG_BOOK_PATH = PROJECT_ROOT / "local_content_dir" / "log_book.xlsx"
|
LOG_BOOK_PATH = PROJECT_ROOT / "data" / "KSP_data" / "log_book.xlsx"
|
||||||
|
|
||||||
OVERVIEW_SHEET_NAME = "Overview"
|
OVERVIEW_SHEET_NAME = "Overview"
|
||||||
HEADER_ROW_INDEX = 1
|
HEADER_ROW_INDEX = 1
|
||||||
@@ -695,6 +695,6 @@ def main(sheet: str, workbook: Optional[str] = None) -> None:
|
|||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
# Configure these values here and call main directly.
|
# Configure these values here and call main directly.
|
||||||
# Edit these lines to change which sheet / workbook are processed.
|
# Edit these lines to change which sheet / workbook are processed.
|
||||||
sheet_name = "XH-01"
|
sheet_name = "ST-01"
|
||||||
workbook_path = str(LOG_BOOK_PATH)
|
workbook_path = str(LOG_BOOK_PATH)
|
||||||
main(sheet_name, workbook_path)
|
main(sheet_name, workbook_path)
|
||||||
|
|||||||
@@ -13,7 +13,7 @@ operating_history_raw = """
|
|||||||
Uncrewed mission to Earth-Sun Lagrange Point 2 to test long-duration operations
|
Uncrewed mission to Earth-Sun Lagrange Point 2 to test long-duration operations
|
||||||
2054-11-28 Test Mission Complete, all systems nominal. Returning to LEO.
|
2054-11-28 Test Mission Complete, all systems nominal. Returning to LEO.
|
||||||
|
|
||||||
2055-12-01 Start maintenance
|
2054-12-01 Start maintenance
|
||||||
2055-02-28 Maintenance complete, all systems operational, ready for commissioning.
|
2055-02-28 Maintenance complete, all systems operational, ready for commissioning.
|
||||||
|
|
||||||
2055-03-20 Picking up crew for Inner Solar System Exploration Mission 1
|
2055-03-20 Picking up crew for Inner Solar System Exploration Mission 1
|
||||||
@@ -37,7 +37,7 @@ operating_history_raw = """
|
|||||||
2056-04-12 Earth Arrival
|
2056-04-12 Earth Arrival
|
||||||
|
|
||||||
2056-04-15 Start maintenance
|
2056-04-15 Start maintenance
|
||||||
2056-06-01 Maintenance complete.
|
2056-05-31 Maintenance complete.
|
||||||
|
|
||||||
2056-06-02 Picking up crew for Mars One Construction Mission 2
|
2056-06-02 Picking up crew for Mars One Construction Mission 2
|
||||||
2056-06-04 Picking up Mars One construction cargo
|
2056-06-04 Picking up Mars One construction cargo
|
||||||
@@ -205,7 +205,7 @@ operating_history = [
|
|||||||
{
|
{
|
||||||
'log_name': "Maintenance",
|
'log_name': "Maintenance",
|
||||||
'start_time': "2056-04-15",
|
'start_time': "2056-04-15",
|
||||||
'end_time': "2056-06-01",
|
'end_time': "2056-05-31",
|
||||||
'mission_detail': "",
|
'mission_detail': "",
|
||||||
'sub_log': []
|
'sub_log': []
|
||||||
},
|
},
|
||||||
@@ -312,4 +312,34 @@ operating_history = [
|
|||||||
'sub_log': [
|
'sub_log': [
|
||||||
"2059-05-31 Parked into Iapetus orbit, orbital science completed",
|
"2059-05-31 Parked into Iapetus orbit, orbital science completed",
|
||||||
"2059-06-15 Start Iapetus Surface operation",
|
"2059-06-15 Start Iapetus Surface operation",
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
'log_name': "Saturn Exploration Mission 1 return to Earth",
|
||||||
|
'start_time': "2059-07-01",
|
||||||
|
'end_time': "2059-11-24",
|
||||||
|
'mission_detail': "Saturn-Earth transfer, using standard trajectory with 162km/s of departure burn",
|
||||||
|
'sub_log': [
|
||||||
|
"2059-07-01 Wrapping up Saturn Exploration Mission. Saturn Departure for Earth",
|
||||||
|
"2059-11-24 Back from Saturn, Earth Arrival",
|
||||||
|
]
|
||||||
|
},
|
||||||
|
{
|
||||||
|
'log_name': "Upgrades at Star Port",
|
||||||
|
'start_time': "2059-11-25",
|
||||||
|
'end_time': "2060-02-10",
|
||||||
|
'mission_detail': "Upgrades complete, ST-01 now fully meets Stellaria specifications.",
|
||||||
|
'sub_log': []
|
||||||
|
},
|
||||||
|
{
|
||||||
|
'log_name': "Outer Solar System Exploration Mission 1 outbound to Neptune",
|
||||||
|
'start_time': "2060-02-11",
|
||||||
|
'end_time': None,
|
||||||
|
'mission_detail': "Earth-Neptune transfer, using 160km/s departure burn",
|
||||||
|
'sub_log': [
|
||||||
|
"2060-02-11 Out of Star Port, config for Outer Solar System Exploration",
|
||||||
|
"2060-02-12 Pick up Outer Solar System Exploration Crew 1 and supply",
|
||||||
|
"2060-02-12 Earth Departure for Neptune",
|
||||||
|
]
|
||||||
|
},
|
||||||
]
|
]
|
||||||
@@ -1,6 +1,7 @@
|
|||||||
from typing import Optional
|
from typing import Optional
|
||||||
import pandas as pd
|
import pandas as pd
|
||||||
from langchain_openai import ChatOpenAI
|
from langchain_openai import ChatOpenAI
|
||||||
|
import warnings
|
||||||
|
|
||||||
from src.PostgresSQL.db_handler import PgHandler
|
from src.PostgresSQL.db_handler import PgHandler
|
||||||
from src.config.load_config import project_config
|
from src.config.load_config import project_config
|
||||||
@@ -15,12 +16,14 @@ LLM = ChatOpenAI(
|
|||||||
def load_history_by_year_month(year: int, month: Optional[int] = None,
|
def load_history_by_year_month(year: int, month: Optional[int] = None,
|
||||||
table_name: str = '王适意_personal_chat',
|
table_name: str = '王适意_personal_chat',
|
||||||
schema: str = 'public',
|
schema: str = 'public',
|
||||||
db_config: Optional[dict] = None) -> pd.DataFrame:
|
db_config: Optional[dict] = None,
|
||||||
"""从 Postgres 中读取历史消息并按年/月过滤。
|
week: Optional[int] = None) -> pd.DataFrame:
|
||||||
|
"""从 Postgres 中读取历史消息并按年/月或年/周过滤。
|
||||||
|
|
||||||
参数:
|
参数:
|
||||||
- year: 年份(四位数)
|
- year: 年份(四位数)
|
||||||
- month: 可选的月份(1-12),若提供则只返回该月的数据
|
- month: 可选的月份(1-12),若提供则只返回该月的数据
|
||||||
|
- week: 可选的每年的第几周(ISO week)。如果提供则优先按周过滤,会 bypass month,但会发出警告。
|
||||||
- table_name: 数据库表名,默认 '王适意_personal_chat'
|
- table_name: 数据库表名,默认 '王适意_personal_chat'
|
||||||
- schema: schema,默认 'public'
|
- schema: schema,默认 'public'
|
||||||
- db_config: 可选的数据库连接参数字典(含 host, port, user, password, database)
|
- db_config: 可选的数据库连接参数字典(含 host, port, user, password, database)
|
||||||
@@ -33,7 +36,7 @@ def load_history_by_year_month(year: int, month: Optional[int] = None,
|
|||||||
port = int(cfg.get('port') or cfg.get('db_port') or 5432)
|
port = int(cfg.get('port') or cfg.get('db_port') or 5432)
|
||||||
user = cfg.get('user') or cfg.get('db_user')
|
user = cfg.get('user') or cfg.get('db_user')
|
||||||
password = cfg.get('password') or cfg.get('db_password')
|
password = cfg.get('password') or cfg.get('db_password')
|
||||||
database = cfg.get('database') or cfg.get('db_name')
|
database = cfg.get('tg_message_db_name')
|
||||||
|
|
||||||
pg = PgHandler(host=host, port=port, user=user, password=password, database=database)
|
pg = PgHandler(host=host, port=port, user=user, password=password, database=database)
|
||||||
try:
|
try:
|
||||||
@@ -66,21 +69,39 @@ def load_history_by_year_month(year: int, month: Optional[int] = None,
|
|||||||
|
|
||||||
df = df.copy()
|
df = df.copy()
|
||||||
df['dt'] = dt
|
df['dt'] = dt
|
||||||
# 过滤年份和可选月份
|
|
||||||
df = df[df['dt'].dt.year == int(year)]
|
# 如果同时提供 month 和 week,警告并以 week 为准(week 优先)
|
||||||
if month is not None:
|
if week is not None and month is not None:
|
||||||
df = df[df['dt'].dt.month == int(month)]
|
warnings.warn('Both "month" and "week" provided; "week" will take precedence and "month" will be ignored.', UserWarning)
|
||||||
|
|
||||||
|
# 过滤年份和可选月份/周
|
||||||
|
# 当提供 week 时,按 ISO week 过滤,并且考虑 ISO week 的 year(isocalendar year)以确保正确匹配
|
||||||
|
if week is not None:
|
||||||
|
# Pandas >= 1.1 提供 dt.isocalendar()
|
||||||
|
try:
|
||||||
|
iso = df['dt'].dt.isocalendar()
|
||||||
|
# iso is a DataFrame-like with 'year' and 'week' columns
|
||||||
|
df = df[(iso['year'] == int(year)) & (iso['week'] == int(week))]
|
||||||
|
except Exception:
|
||||||
|
# 备选:使用 datetime.isocalendar() per-row(较慢)
|
||||||
|
df = df[df['dt'].apply(lambda x: x.isocalendar()[0] == int(year) and x.isocalendar()[1] == int(week))]
|
||||||
|
else:
|
||||||
|
# 默认按 calendar 年过滤
|
||||||
|
df = df[df['dt'].dt.year == int(year)]
|
||||||
|
if month is not None:
|
||||||
|
df = df[df['dt'].dt.month == int(month)]
|
||||||
|
|
||||||
df = df.sort_values('dt').reset_index(drop=True)
|
df = df.sort_values('dt').reset_index(drop=True)
|
||||||
return df
|
return df
|
||||||
|
|
||||||
|
|
||||||
def do_analysis_on_history_messages(df: pd.DataFrame):
|
def do_analysis_on_history_messages(df: pd.DataFrame):
|
||||||
data = df.to_dict(orient='records')
|
data = df.to_dict(orient='records')
|
||||||
to_model_data = [
|
to_model_data = [
|
||||||
{
|
{
|
||||||
'message_datetime': item.get('message_datetime'),
|
'msg_time': item.get('message_datetime'),
|
||||||
'message_from': item.get('message_from'),
|
'msg_from': item.get('message_from'),
|
||||||
'message_text': item.get('message_text'),
|
'msg_text': f"转发来自{item['forward_from']}的消息: {item.get('message_text')}" if item.get('forward_from') else item.get('message_text'),
|
||||||
'file_type': item.get('file_type'),
|
'file_type': item.get('file_type'),
|
||||||
}
|
}
|
||||||
for item in data
|
for item in data
|
||||||
@@ -89,8 +110,8 @@ def do_analysis_on_history_messages(df: pd.DataFrame):
|
|||||||
prompt = f"""
|
prompt = f"""
|
||||||
你是一个聊天记录分析专家,擅长根据用户的聊天记录,分析用户的行为,然后撰写报告。
|
你是一个聊天记录分析专家,擅长根据用户的聊天记录,分析用户的行为,然后撰写报告。
|
||||||
你的报告应当包含以下内容:
|
你的报告应当包含以下内容:
|
||||||
1. 以天为单位,分析用户在这一天内的主要活动和重大事件。每一天的总结应当独立成段。
|
1. 以天为单位,分析用户在这一天内的主要活动和重大事件。每一天的总结应当独立成段。每一天的重大事件应当按照时间顺序排列。
|
||||||
2. 按照月时间维度,输出总结。
|
2. 输出总结报告。
|
||||||
|
|
||||||
你的总结应当包括但不限于以下内容:
|
你的总结应当包括但不限于以下内容:
|
||||||
- 用户在这个月内的主要活动和重大事件。
|
- 用户在这个月内的主要活动和重大事件。
|
||||||
@@ -113,7 +134,7 @@ def do_analysis_on_history_messages(df: pd.DataFrame):
|
|||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
df_month = load_history_by_year_month(2025, month=1)
|
# df = load_history_by_year_month(2025, month=1)
|
||||||
summary = do_analysis_on_history_messages(df_month)
|
df = load_history_by_year_month(2025, week=2)
|
||||||
print(f"Loaded {len(df_month)} messages for May 2023")
|
summary = do_analysis_on_history_messages(df)
|
||||||
|
print(f"done")
|
||||||
|
|||||||
Reference in New Issue
Block a user