4f63ec56dc
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
200 lines
9.5 KiB
Python
200 lines
9.5 KiB
Python
#!/usr/bin/env python3
|
|
"""비하이브 '[🐝 공모청약]' 영상 → 종목별 청약 의견 요약 (자산웹 공모주 탭 표시용).
|
|
|
|
흐름: 채널 영상 목록(youtube_briefing_digest 재사용) → 제목 '공모청약' 필터 → 처리 안 한 영상만
|
|
자막 → LLM 이 '지금 진행·예정 공모주 목록' 중 영상이 다룬 종목별로 요약 → state 에 영상 단위 저장.
|
|
behive_web 은 state 파일만 읽는다(렌더 경로 LLM·네트워크 0).
|
|
|
|
⚠️ 영상 하나가 여러 종목을 다룬다(예: 멜콘 영상 = 1주차 3종목) → 제목 해시태그로 매칭하지 않고
|
|
후보 목록을 LLM 에 주고 ipoCode 로 돌려받는다. 목록에 없는 코드는 버린다.
|
|
⚠️ 자막은 업로드 직후 없을 수 있다 → 실패는 기록만 하고 다음 실행에서 재시도, MAX_ATTEMPTS 후 포기.
|
|
트리거: ipo_alert.py(stock.ipo-alert 08:30·10:00)가 끝에 run() 을 부른다. 새 트리거 없음.
|
|
|
|
CLI: python3 ipo_youtube_digest.py [run|show] [--dry-run]
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import re
|
|
import subprocess
|
|
import sys
|
|
from datetime import date, datetime, timedelta
|
|
from pathlib import Path
|
|
|
|
sys.path.insert(0, '/Users/snowoyh/.openclaw/workspace/scripts')
|
|
import youtube_briefing_digest as ytd # noqa: E402
|
|
from ipo_calendar_sync import KST, NAVER_IPO_API_URL, _naver_api_rows, fetch # noqa: E402
|
|
|
|
STATE_FILE = Path('/Users/snowoyh/.openclaw/agents/stock/workspace/state/ipo_youtube_summaries.json')
|
|
VIDEOS_URL = ytd.CHANNELS['behive']['videos_url']
|
|
TITLE_FILTER = '공모청약'
|
|
# launchd PATH 의 /opt/homebrew/bin/openclaw 는 node22 라 안 뜬다 — node24 래퍼를 절대경로로.
|
|
OPENCLAW_BIN = str(Path.home() / '.local/bin/openclaw')
|
|
LLM_MODEL = 'openai/gpt-5.6-sol'
|
|
LLM_TIMEOUT_SEC = 180
|
|
MAX_ATTEMPTS = 6 # 자막 미생성·LLM 실패 재시도 상한 (하루 2회 실행 → 약 3일)
|
|
KEEP_DAYS = 120 # 상장 끝난 지 오래된 영상 기록 정리
|
|
VERDICTS = {'positive': '청약 매력 높음', 'neutral': '중립·수요예측 확인', 'negative': '청약 부담'}
|
|
|
|
|
|
def load_state() -> dict:
|
|
try:
|
|
return json.loads(STATE_FILE.read_text())
|
|
except Exception:
|
|
return {'videos': {}}
|
|
|
|
|
|
def save_state(state: dict) -> None:
|
|
tmp = STATE_FILE.with_suffix('.json.tmp')
|
|
tmp.write_text(json.dumps(state, ensure_ascii=False, indent=2))
|
|
tmp.replace(STATE_FILE)
|
|
|
|
|
|
def _candidates(today: date) -> list[dict]:
|
|
"""요약 대상 후보 = 아직 상장 안 했거나 상장 2주 이내인 공모주(청약 전·중·후 모두)."""
|
|
rows = _naver_api_rows(json.loads(fetch(NAVER_IPO_API_URL, encoding='utf-8')))
|
|
floor = (today - timedelta(days=14)).isoformat()
|
|
out = []
|
|
for r in rows:
|
|
code = (r.get('ipoCode') or r.get('itemCode') or '').strip()
|
|
name = (r.get('compName') or '').strip()
|
|
if not code or not name:
|
|
continue
|
|
if (r.get('lcalDate') or '9999') < floor:
|
|
continue
|
|
out.append({'ipo_code': code, 'name': name, 'po_start': r.get('poStartDate') or '',
|
|
'po_end': r.get('poEndDate') or '', 'lcal_date': r.get('lcalDate') or '',
|
|
'hope': f"{r.get('hopePubStart') or '?'}~{r.get('hopePubEnd') or '?'}",
|
|
'fix': r.get('fixPubPrice') or ''})
|
|
return out
|
|
|
|
|
|
def _call_llm(prompt: str) -> str:
|
|
cmd = [OPENCLAW_BIN, 'capability', 'model', 'run', '--prompt', prompt, '--model', LLM_MODEL, '--json']
|
|
p = subprocess.run(cmd, capture_output=True, text=True, timeout=LLM_TIMEOUT_SEC)
|
|
if p.returncode != 0:
|
|
raise RuntimeError(f'LLM rc={p.returncode}: {p.stderr[-300:]}')
|
|
data = json.loads(p.stdout[p.stdout.index('{'):])
|
|
text = ((data.get('outputs') or [{}])[0].get('text') or '').strip()
|
|
if not data.get('ok') or not text:
|
|
raise RuntimeError(f'LLM 응답 비정상: {str(data)[:300]}')
|
|
return text
|
|
|
|
|
|
def _prompt(title: str, transcript: str, cands: list[dict]) -> str:
|
|
cand_lines = '\n'.join(
|
|
f'- {c["ipo_code"]} {c["name"]} (청약 {c["po_start"]}~{c["po_end"]}, 희망공모가 {c["hope"]}원'
|
|
+ (f', 확정공모가 {c["fix"]}원' if c['fix'] else '') + ')' for c in cands)
|
|
return f"""아래는 유튜브 채널 '비하이브 투자자문'의 공모주 청약 영상 자막(자동생성, 오타 많음)이다.
|
|
영상이 **구체적으로 분석한** 종목만 골라 종목별로 요약하라. 일정만 언급하고 지나간 종목은 제외.
|
|
종목은 반드시 아래 후보 목록의 코드로 지정한다(자막의 종목명 오타는 후보명으로 맞춰 판단). 후보에 없으면 제외.
|
|
|
|
[후보]
|
|
{cand_lines}
|
|
|
|
[영상 제목] {title}
|
|
[자막]
|
|
{transcript}
|
|
|
|
출력은 JSON 배열 하나만(설명·코드펜스 금지):
|
|
[{{"ipo_code":"A123456","verdict":"positive|neutral|negative","verdict_text":"영상 결론 한 구절(15자 이내)","points":["핵심 3~4개, 각 40자 이내"]}}]
|
|
- verdict: 영상 화자의 청약 의견. 긍정=positive, 유보·수요예측 보고 판단=neutral, 부정·부담=negative
|
|
- 자막 숫자는 받아쓰기 오류가 잦다(예: 74,000→7,400). 공모가는 후보 정보가 정답이니 그 값을 쓰고, 다른 수치도 앞뒤가 안 맞으면 빼라
|
|
- points: 사업 한 줄 / 실적 추이(수치) / 유통가능물량(%) / 공모가·밸류 부담 등 화자가 짚은 장단점. 영상에 없는 내용 지어내지 말 것
|
|
- 해당 종목이 없으면 []"""
|
|
|
|
|
|
def _parse(text: str, cands: list[dict]) -> list[dict]:
|
|
m = re.search(r'\[.*\]', text, re.S)
|
|
arr = json.loads(m.group(0)) if m else []
|
|
names = {c['ipo_code']: c['name'] for c in cands}
|
|
out = []
|
|
for it in arr:
|
|
code = str(it.get('ipo_code') or '').strip()
|
|
if code not in names:
|
|
continue
|
|
verdict = it.get('verdict') if it.get('verdict') in VERDICTS else 'neutral'
|
|
points = [str(p).strip() for p in (it.get('points') or []) if str(p).strip()][:4]
|
|
out.append({'ipo_code': code, 'name': names[code], 'verdict': verdict,
|
|
'verdict_text': str(it.get('verdict_text') or VERDICTS[verdict]).strip()[:30],
|
|
'points': points})
|
|
return out
|
|
|
|
|
|
def run(dry_run: bool = False) -> dict:
|
|
"""새 공모청약 영상만 요약. 반환: {new, summarized, failed}."""
|
|
now = datetime.now(KST)
|
|
state = load_state()
|
|
videos = state.setdefault('videos', {})
|
|
listing = [v for v in ytd.fetch_channel_videos(VIDEOS_URL) if TITLE_FILTER in v.get('title', '')]
|
|
todo = [v for v in listing
|
|
if not videos.get(v['video_id'], {}).get('done')
|
|
and videos.get(v['video_id'], {}).get('attempts', 0) < MAX_ATTEMPTS]
|
|
result = {'new': len(todo), 'summarized': 0, 'failed': 0}
|
|
if not todo:
|
|
return result
|
|
cands = _candidates(now.date())
|
|
for v in todo:
|
|
vid = v['video_id']
|
|
rec = videos.setdefault(vid, {'title': v['title'], 'url': ytd.WATCH_URL.format(vid=vid), 'attempts': 0})
|
|
try:
|
|
transcript, status = ytd.fetch_transcript(vid)
|
|
if status != 'ok':
|
|
raise RuntimeError(f'자막 {status}')
|
|
stocks = _parse(_call_llm(_prompt(v['title'], transcript, cands)), cands)
|
|
except Exception as e:
|
|
rec['attempts'] = rec.get('attempts', 0) + 1
|
|
rec['last_error'] = f'{type(e).__name__}: {e}'[:300]
|
|
result['failed'] += 1
|
|
print(f'ipo yt {vid} 실패({rec["attempts"]}): {rec["last_error"]}', file=sys.stderr)
|
|
continue
|
|
published = ytd.get_publish_kst(vid)
|
|
rec.update({'done': True, 'stocks': stocks, 'summarized_at': now.isoformat(timespec='seconds'),
|
|
'published_kst': published.isoformat() if published else None})
|
|
rec.pop('last_error', None)
|
|
result['summarized'] += 1
|
|
if dry_run:
|
|
print(json.dumps({vid: rec}, ensure_ascii=False, indent=2))
|
|
# 오래된 기록 정리 — 목록에서 빠졌고 요약 시각이 KEEP_DAYS 넘은 것
|
|
cutoff = (now - timedelta(days=KEEP_DAYS)).isoformat()
|
|
for vid in [k for k, r in videos.items() if (r.get('summarized_at') or now.isoformat()) < cutoff]:
|
|
videos.pop(vid)
|
|
if not dry_run:
|
|
save_state(state)
|
|
return result
|
|
|
|
|
|
def summaries_by_code() -> dict[str, list[dict]]:
|
|
"""behive_web 용: ipoCode → [{video_id,title,url,published_kst,verdict,verdict_text,points,tagged}].
|
|
제목 해시태그에 그 종목이 있는 전용 영상을 앞에, 그다음 최신순."""
|
|
out: dict[str, list[dict]] = {}
|
|
for vid, rec in load_state().get('videos', {}).items():
|
|
for s in rec.get('stocks') or []:
|
|
out.setdefault(s['ipo_code'], []).append({
|
|
'video_id': vid, 'title': rec.get('title', ''), 'url': rec.get('url', ''),
|
|
'published_kst': rec.get('published_kst') or rec.get('summarized_at') or '',
|
|
'verdict': s['verdict'], 'verdict_text': s['verdict_text'], 'points': s['points'],
|
|
'tagged': f'#{s["name"]}' in rec.get('title', ''),
|
|
})
|
|
for lst in out.values():
|
|
lst.sort(key=lambda x: x['published_kst'], reverse=True)
|
|
lst.sort(key=lambda x: not x['tagged'])
|
|
return out
|
|
|
|
|
|
def main() -> int:
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument('cmd', nargs='?', default='run', choices=('run', 'show'))
|
|
ap.add_argument('--dry-run', action='store_true')
|
|
args = ap.parse_args()
|
|
if args.cmd == 'show':
|
|
print(json.dumps(summaries_by_code(), ensure_ascii=False, indent=2))
|
|
return 0
|
|
print(json.dumps(run(args.dry_run), ensure_ascii=False))
|
|
return 0
|
|
|
|
|
|
if __name__ == '__main__':
|
|
raise SystemExit(main())
|