Files
Hermes-Skills/strategy-management/scripts/fetch_policy.py
T

128 lines
4.2 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""
fetch_policy.py - 读取 policy-news-monitor cron 最新 session,提取政策标题摘要
用法:
python3 fetch_policy.py --type cn # 国内政策 → /tmp/policy_cn.txt
python3 fetch_policy.py --type intl # 国际政策 → /tmp/policy_intl.txt
python3 fetch_policy.py --type both # 两者都抓
集成方式:
cron 先跑 policy-news-monitor 生成 session 文件,
本脚本读取最新 session 提取标题,供股票分析脚本注入。
"""
import argparse
import re
import sys
import os
from pathlib import Path
from datetime import datetime, date
REF_DIR = Path(__file__).parent.parent / "references"
_ref_dir = REF_DIR # 运行时覆盖
def latest_session(pattern):
"""返回匹配 pattern 的最新 session 文件路径"""
sessions = sorted(
_ref_dir.glob(f"session-*{pattern}*.md"),
key=lambda p: p.stat().st_mtime,
reverse=True
)
return sessions[0] if sessions else None
def extract_headlines(session_path, max_lines=8):
"""从 session 文件提取前 N 条政策新闻标题"""
if not session_path or not session_path.exists():
return None
text = session_path.read_text(encoding="utf-8")
lines = []
# 匹配形如 "· 央行宣布降准" 或 "1. 政策标题" 的行
for line in text.splitlines():
line = line.strip()
if not line:
continue
# 跳过代码块、标题、URL
if line.startswith("#") or line.startswith("```") or line.startswith("http"):
continue
# 提取列表项和政策标题
m = re.match(r"^[\d\.\、\-\•\◉]+\s*[\[【]?\s*([^\]\n]{8,60})", line)
if m:
title = m.group(1).strip()
if "免责" not in title and len(title) > 6:
lines.append(f" · {title}")
elif len(line) > 8 and len(line) < 80 and "" in line:
# "央行:降准" 格式
title = line.strip()
if "免责" not in title:
lines.append(f" · {title}")
if len(lines) >= max_lines:
break
return "\n".join(lines) if lines else None
def fetch_cn():
"""国内政策: 读最新 domestic session"""
session = latest_session("domestic")
if not session:
return None, "无 domestic session"
headlines = extract_headlines(session)
if not headlines:
return None, f"提取失败: {session.name}"
return headlines, session.name
def fetch_intl():
"""国际政策: 读最新 international session"""
session = latest_session("international")
if not session:
return None, "无 international session"
headlines = extract_headlines(session)
if not headlines:
return None, f"提取失败: {session.name}"
return headlines, session.name
def main():
ap = argparse.ArgumentParser()
ap.add_argument('--type', default='both', choices=['cn', 'intl', 'both'])
ap.add_argument('--ref', default=None, help='references 目录')
args = ap.parse_args()
# 动态设置 ref dir(允许 cron 指定不同路径)
ref_path = Path(args.ref) if args.ref else REF_DIR
if ref_path.exists():
_ref_dir = ref_path
else:
_ref_dir = REF_DIR
date_str = date.today().isoformat()
done = []
if args.type in ('cn', 'both'):
headlines, info = fetch_cn()
if headlines:
Path("/tmp/policy_cn.txt").write_text(f"# 国内政策摘要 {date_str}(来源: {info}\n{headlines}\n")
print(f"✅ 国内政策 → /tmp/policy_cn.txt{info}")
done.append("cn")
else:
print(f"⚠️ 国内政策: {info}")
if args.type in ('intl', 'both'):
headlines, info = fetch_intl()
if headlines:
Path("/tmp/policy_intl.txt").write_text(f"# 国际政策摘要 {date_str}(来源: {info}\n{headlines}\n")
print(f"✅ 国际政策 → /tmp/policy_intl.txt{info}")
done.append("intl")
else:
print(f"⚠️ 国际政策: {info}")
if not done:
print("⚠️ 警告: 两个都没抓到,exit 1")
sys.exit(1)
if __name__ == '__main__':
main()