Initial commit: Hermes Agent skills collection
- Trading skills (OKX, dividend, lottery, quantitative) - Creative skills (ASCII art, diagrams, video) - Development skills (GitHub, debugging, TDD) - Research skills (arXiv, blog monitoring) - Productivity skills (email, documents, notes) - MCP integration skills - Custom user skills
This commit is contained in:
@@ -0,0 +1,365 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
用户痛点自动抓取与分析工具
|
||||
从 Reddit、Hacker News 等平台采集用户评论,用 LLM 提取痛点。
|
||||
|
||||
用法:
|
||||
python3 pain_point_scraper.py --topic "AI写作工具" --sources reddit,hn --limit 20
|
||||
python3 pain_point_scraper.py --topic "OKX交易" --sources reddit --limit 30 --output report.md
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
import time
|
||||
from datetime import datetime
|
||||
from typing import Optional
|
||||
|
||||
import requests
|
||||
|
||||
# ─────────────────────────────────────────────
|
||||
# 配置
|
||||
# ─────────────────────────────────────────────
|
||||
|
||||
# LLM API (使用本地 Ollama 或 LM Studio)
|
||||
LLM_API_URL = "http://192.168.199.100:11434/v1/chat/completions"
|
||||
LLM_MODEL = "gemma-4-26b-a4b-it-4bit"
|
||||
LLM_API_KEY = "" # 本地服务通常不需要
|
||||
|
||||
# 如果本地不可用,fallback到远程
|
||||
LLM_FALLBACK_URL = "https://token-plan-cn.xiaomimimo.com/v1/chat/completions"
|
||||
LLM_FALLBACK_MODEL = "mimo-v2.5-pro"
|
||||
LLM_FALLBACK_KEY = "tp-cd64i50fo6ifihhiupspq9dypmzeud3u77tia5mlz0rb572u"
|
||||
|
||||
HEADERS = {
|
||||
"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36"
|
||||
}
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────
|
||||
# Reddit 采集
|
||||
# ─────────────────────────────────────────────
|
||||
|
||||
def scrape_reddit(topic: str, limit: int = 20) -> list[dict]:
|
||||
"""从 Reddit 搜索相关讨论"""
|
||||
results = []
|
||||
|
||||
# 使用 Reddit JSON API (无需认证)
|
||||
url = "https://www.reddit.com/search.json"
|
||||
params = {
|
||||
"q": topic,
|
||||
"sort": "relevance",
|
||||
"t": "month", # 最近一个月
|
||||
"limit": min(limit, 25),
|
||||
"type": "link"
|
||||
}
|
||||
|
||||
try:
|
||||
resp = requests.get(url, params=params, headers=HEADERS, timeout=15)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
|
||||
for post in data.get("data", {}).get("children", []):
|
||||
d = post.get("data", {})
|
||||
results.append({
|
||||
"source": "Reddit",
|
||||
"title": d.get("title", ""),
|
||||
"text": d.get("selftext", "")[:500],
|
||||
"score": d.get("score", 0),
|
||||
"comments": d.get("num_comments", 0),
|
||||
"url": f"https://reddit.com{d.get('permalink', '')}",
|
||||
"subreddit": d.get("subreddit", ""),
|
||||
"created": datetime.fromtimestamp(d.get("created_utc", 0)).isoformat()
|
||||
})
|
||||
|
||||
print(f" ✅ Reddit: 获取 {len(results)} 条", file=sys.stderr)
|
||||
except Exception as e:
|
||||
print(f" ❌ Reddit 采集失败: {e}", file=sys.stderr)
|
||||
|
||||
return results
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────
|
||||
# Hacker News 采集
|
||||
# ─────────────────────────────────────────────
|
||||
|
||||
def scrape_hn(topic: str, limit: int = 20) -> list[dict]:
|
||||
"""从 Hacker News 搜索相关讨论"""
|
||||
results = []
|
||||
|
||||
url = "https://hn.algolia.com/api/v1/search"
|
||||
params = {
|
||||
"query": topic,
|
||||
"tags": "story",
|
||||
"hitsPerPage": min(limit, 30),
|
||||
"numericFilters": "created_at_i>" + str(int(time.time()) - 30*86400)
|
||||
}
|
||||
|
||||
try:
|
||||
resp = requests.get(url, params=params, timeout=15)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
|
||||
for hit in data.get("hits", []):
|
||||
results.append({
|
||||
"source": "Hacker News",
|
||||
"title": hit.get("title", ""),
|
||||
"text": (hit.get("story_text") or "")[:500],
|
||||
"score": hit.get("points", 0),
|
||||
"comments": hit.get("num_comments", 0),
|
||||
"url": hit.get("url", f"https://news.ycombinator.com/item?id={hit.get('objectID', '')}"),
|
||||
"author": hit.get("author", ""),
|
||||
"created": hit.get("created_at", "")
|
||||
})
|
||||
|
||||
print(f" ✅ HN: 获取 {len(results)} 条", file=sys.stderr)
|
||||
except Exception as e:
|
||||
print(f" ❌ HN 采集失败: {e}", file=sys.stderr)
|
||||
|
||||
return results
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────
|
||||
# Product Hunt 采集 (通过搜索)
|
||||
# ─────────────────────────────────────────────
|
||||
|
||||
def scrape_producthunt(topic: str, limit: int = 10) -> list[dict]:
|
||||
"""从 Product Hunt 搜索相关产品评论"""
|
||||
results = []
|
||||
|
||||
# 使用 Google 搜索 Product Hunt 上的内容
|
||||
url = "https://www.google.com/search"
|
||||
params = {
|
||||
"q": f"site:producthunt.com {topic} review complaints",
|
||||
"num": min(limit, 10)
|
||||
}
|
||||
|
||||
try:
|
||||
resp = requests.get(url, params=params, headers=HEADERS, timeout=15)
|
||||
# Google 可能会block,所以这里简单处理
|
||||
if resp.status_code == 200:
|
||||
# 简单提取标题
|
||||
import re
|
||||
titles = re.findall(r'<h3[^>]*>(.*?)</h3>', resp.text)
|
||||
for t in titles[:limit]:
|
||||
clean = re.sub(r'<[^>]+>', '', t)
|
||||
if clean and len(clean) > 10:
|
||||
results.append({
|
||||
"source": "Product Hunt",
|
||||
"title": clean,
|
||||
"text": "",
|
||||
"score": 0,
|
||||
"comments": 0,
|
||||
"url": "",
|
||||
"created": ""
|
||||
})
|
||||
|
||||
print(f" ✅ Product Hunt: 获取 {len(results)} 条", file=sys.stderr)
|
||||
except Exception as e:
|
||||
print(f" ❌ Product Hunt 采集失败: {e}", file=sys.stderr)
|
||||
|
||||
return results
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────
|
||||
# LLM 分析
|
||||
# ─────────────────────────────────────────────
|
||||
|
||||
def call_llm(prompt: str, api_url: str = None, model: str = None, api_key: str = None) -> str:
|
||||
"""调用 LLM API 进行分析"""
|
||||
api_url = api_url or LLM_API_URL
|
||||
model = model or LLM_MODEL
|
||||
api_key = api_key or LLM_API_KEY
|
||||
|
||||
headers = {"Content-Type": "application/json"}
|
||||
if api_key:
|
||||
headers["Authorization"] = f"Bearer {api_key}"
|
||||
|
||||
payload = {
|
||||
"model": model,
|
||||
"messages": [
|
||||
{"role": "system", "content": "你是一个专业的用户研究分析师。你擅长从用户评论中提取痛点、需求和机会。请用中文回答。"},
|
||||
{"role": "user", "content": prompt}
|
||||
],
|
||||
"temperature": 0.3,
|
||||
"max_tokens": 4000
|
||||
}
|
||||
|
||||
try:
|
||||
resp = requests.post(api_url, json=payload, headers=headers, timeout=120)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
return data["choices"][0]["message"]["content"]
|
||||
except Exception as e:
|
||||
print(f" ⚠️ 主API失败 ({e}),尝试fallback...", file=sys.stderr)
|
||||
# Fallback
|
||||
try:
|
||||
fallback_headers = {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": f"Bearer {LLM_FALLBACK_KEY}"
|
||||
}
|
||||
payload["model"] = LLM_FALLBACK_MODEL
|
||||
resp = requests.post(LLM_FALLBACK_URL, json=payload, headers=fallback_headers, timeout=120)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
return data["choices"][0]["message"]["content"]
|
||||
except Exception as e2:
|
||||
return f"❌ LLM 分析失败: {e2}"
|
||||
|
||||
|
||||
def analyze_pain_points(items: list[dict], topic: str) -> str:
|
||||
"""用 LLM 分析采集到的数据,提取痛点"""
|
||||
|
||||
# 构建分析prompt
|
||||
reviews_text = ""
|
||||
for i, item in enumerate(items[:30], 1): # 最多30条,避免token超限
|
||||
reviews_text += f"\n--- [{i}] {item['source']} | 👍{item['score']} 💬{item['comments']} ---\n"
|
||||
reviews_text += f"标题: {item['title']}\n"
|
||||
if item.get('text'):
|
||||
reviews_text += f"内容: {item['text'][:300]}\n"
|
||||
|
||||
prompt = f"""请分析以下关于"{topic}"的用户评论/讨论,提取用户痛点、需求和机会。
|
||||
|
||||
## 原始数据
|
||||
{reviews_text}
|
||||
|
||||
## 分析要求
|
||||
|
||||
请按以下格式输出:
|
||||
|
||||
### 📊 概览
|
||||
- 总评论数:N
|
||||
- 主要痛点数:N
|
||||
- 整体情绪:正面/负面/混合
|
||||
|
||||
### 🔥 TOP 痛点(按严重程度排序)
|
||||
|
||||
对每个痛点:
|
||||
1. **痛点标题**(一句话概括)
|
||||
2. **用户原话**(引用1-2条最能说明问题的原文)
|
||||
3. **痛点描述**(为什么这是问题)
|
||||
4. **影响人群**(谁会遇到这个问题)
|
||||
5. **现有解决方案**(有没有竞品在解决)
|
||||
6. **机会评估**(⭐1-5星,这个痛点值不值得做)
|
||||
|
||||
### 💡 机会洞察
|
||||
- 基于以上痛点,有哪些产品/功能机会
|
||||
- 有哪些未被满足的需求
|
||||
|
||||
### 📈 趋势观察
|
||||
- 近期讨论的热点变化
|
||||
- 用户关注点的转移
|
||||
|
||||
请用中文回答,保持分析的客观性和实用性。"""
|
||||
|
||||
return call_llm(prompt)
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────
|
||||
# 报告生成
|
||||
# ─────────────────────────────────────────────
|
||||
|
||||
def generate_report(topic: str, items: list[dict], analysis: str) -> str:
|
||||
"""生成完整的Markdown报告"""
|
||||
|
||||
sources = {}
|
||||
for item in items:
|
||||
src = item["source"]
|
||||
sources[src] = sources.get(src, 0) + 1
|
||||
|
||||
source_summary = "、".join([f"{k}({v}条)" for k, v in sources.items()])
|
||||
|
||||
report = f"""# 🔍 痛点分析报告:{topic}
|
||||
> 数据来源:{source_summary} | 采集时间:{datetime.now().strftime('%Y-%m-%d %H:%M')} | 分析评论数:{len(items)}
|
||||
|
||||
---
|
||||
|
||||
{analysis}
|
||||
|
||||
---
|
||||
|
||||
## 📋 原始数据摘要
|
||||
|
||||
| # | 来源 | 标题 | 👍 | 💬 |
|
||||
|---|------|------|-----|-----|
|
||||
"""
|
||||
|
||||
for i, item in enumerate(items[:20], 1):
|
||||
title = item["title"][:50] + ("..." if len(item["title"]) > 50 else "")
|
||||
report += f"| {i} | {item['source']} | {title} | {item['score']} | {item['comments']} |\n"
|
||||
|
||||
report += f"\n*共 {len(items)} 条数据,显示前 20 条*\n"
|
||||
|
||||
return report
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────
|
||||
# 主流程
|
||||
# ─────────────────────────────────────────────
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description="用户痛点自动抓取与分析")
|
||||
parser.add_argument("--topic", "-t", required=True, help="分析主题/关键词")
|
||||
parser.add_argument("--sources", "-s", default="reddit,hn", help="数据源 (逗号分隔: reddit,hn,producthunt)")
|
||||
parser.add_argument("--limit", "-l", type=int, default=20, help="每个源采集数量")
|
||||
parser.add_argument("--output", "-o", help="输出文件路径 (默认stdout)")
|
||||
parser.add_argument("--raw", action="store_true", help="只输出原始数据,不做LLM分析")
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
print(f"\n🔍 痛点分析:{args.topic}", file=sys.stderr)
|
||||
print(f"📡 数据源:{args.sources}", file=sys.stderr)
|
||||
print(f"📊 采集数量:每个源 {args.limit} 条\n", file=sys.stderr)
|
||||
|
||||
# 1. 采集数据
|
||||
all_items = []
|
||||
source_map = {
|
||||
"reddit": scrape_reddit,
|
||||
"hn": scrape_hn,
|
||||
"producthunt": scrape_producthunt
|
||||
}
|
||||
|
||||
for source in args.sources.split(","):
|
||||
source = source.strip().lower()
|
||||
if source in source_map:
|
||||
items = source_map[source](args.topic, args.limit)
|
||||
all_items.extend(items)
|
||||
else:
|
||||
print(f" ⚠️ 未知数据源: {source}", file=sys.stderr)
|
||||
|
||||
if not all_items:
|
||||
print("\n❌ 没有采集到任何数据", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
print(f"\n📊 总计采集: {len(all_items)} 条\n", file=sys.stderr)
|
||||
|
||||
# 2. 如果只输出原始数据
|
||||
if args.raw:
|
||||
output = json.dumps(all_items, ensure_ascii=False, indent=2)
|
||||
if args.output:
|
||||
with open(args.output, "w") as f:
|
||||
f.write(output)
|
||||
print(f"✅ 原始数据已保存到: {args.output}", file=sys.stderr)
|
||||
else:
|
||||
print(output)
|
||||
return
|
||||
|
||||
# 3. LLM 分析
|
||||
print("🧠 正在分析痛点...", file=sys.stderr)
|
||||
analysis = analyze_pain_points(all_items, args.topic)
|
||||
|
||||
# 4. 生成报告
|
||||
report = generate_report(args.topic, all_items, analysis)
|
||||
|
||||
if args.output:
|
||||
with open(args.output, "w") as f:
|
||||
f.write(report)
|
||||
print(f"\n✅ 报告已保存到: {args.output}", file=sys.stderr)
|
||||
else:
|
||||
print(report)
|
||||
|
||||
print(f"\n✅ 分析完成!", file=sys.stderr)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user