carbon-ai-chatbot / monitor_methodologies.py
ruffy1601's picture
Claude Sonnet 4.6
feat: μžλ™ λͺ¨λ‹ˆν„°λ§ νŒŒμ΄ν”„λΌμΈ ꡬ좕 (APScheduler + SQLite + monitoring_reports)
18f8ce8
Raw History Blame Contribute Delete
15.8 kB
"""
방법둠 μžλ™ λͺ¨λ‹ˆν„°λ§ νŒŒμ΄ν”„λΌμΈ
================================
주기적으둜 μ‹€ν–‰ν•˜μ—¬ μ‹ κ·œ/κ°œμ • 방법둠을 κ°μ§€ν•˜κ³ 
knowledge_baseλ₯Ό μžλ™μœΌλ‘œ κ°±μ‹ ν•©λ‹ˆλ‹€.
μ‹€ν–‰:
python monitor_methodologies.py # λ³€κ²½ 감지 + μžλ™ μ—…λ°μ΄νŠΈ
python monitor_methodologies.py --check-only # λ³€κ²½ κ°μ§€λ§Œ (μ—…λ°μ΄νŠΈ μ—†μŒ)
python monitor_methodologies.py --scrape # 원격 λ ˆμ§€μŠ€νŠΈλ¦¬ μŠ€ν¬λž˜ν•‘ 포함
python monitor_methodologies.py --dry-run # 미리보기
흐름:
β‘  둜컬 방법둠 디렉토리 μŠ€μΊ” (CDM/ORS/GS/Verra)
β‘‘ methodology_state.json κ³Ό 비ꡐ β†’ μ‹ κ·œ/κ°œμ • 감지
β‘’ λ³€κ²½λœ 파일 β†’ knowledge_base/ 볡사
β‘£ λ³€κ²½ 있으면 β†’ update_all.py μ‹€ν–‰ (ChromaDB μž¬κ΅¬μΆ•)
β‘€ state 파일 κ°±μ‹ 
"""
from __future__ import annotations
import argparse
import hashlib
import json
import shutil
import subprocess
import sys
import time
from dataclasses import asdict, dataclass
from datetime import datetime
from pathlib import Path
from typing import Optional
# ── 경둜 μ„€μ • ──────────────────────────────────────────────────────────────
ROOT = Path(__file__).parent
KNOWLEDGE_BASE = ROOT / "react-agent" / "knowledge_base"
STATE_FILE = ROOT / "methodology_state.json"
# λ ˆμ§€μŠ€νŠΈλ¦¬λ³„ 둜컬 디렉토리 및 prefix μ„€μ •
REGISTRIES = {
"CDM_PA": {
"dir": ROOT / "cdm_methodologies" / "CDM_PA_Large_Scale_Methodologies",
"prefix": "cdm_pa_",
"label": "CDM λŒ€κ·œλͺ¨",
},
"CDM_SSC": {
"dir": ROOT / "cdm_methodologies" / "CDM_SSC_Small_Scale_Methodologies",
"prefix": "cdm_ssc_",
"label": "CDM μ†Œκ·œλͺ¨",
},
"CDM_AR": {
"dir": ROOT / "cdm_methodologies" / "CDM_AR_SSCAR_Afforestation_Reforestation_Methodologies",
"prefix": "cdm_ar_",
"label": "CDM 쑰림/재쑰림",
},
"ORS": {
"dir": ROOT / "ors_methodologies",
"prefix": "ors_",
"label": "ORS (κ΅­λ‚΄ μƒμ‡„λ°°μΆœμ›)",
},
"GS": {
"dir": ROOT / "gs_methodologies",
"prefix": "gs_",
"label": "Gold Standard",
},
"VERRA": {
"dir": ROOT / "verra_methodologies",
"prefix": "verra_",
"label": "Verra VCS",
},
}
# knowledge_base에 볡사할 파일 ν™•μž₯자
ALLOWED_EXT = {".pdf", ".hwp", ".docx", ".txt"}
# 슀크래퍼 슀크립트 (--scrape μ˜΅μ…˜ μ‚¬μš© μ‹œ)
SCRAPERS = {
"ORS": ROOT / "scrape_ors.py",
"GS": ROOT / "scrape_gs.py",
"VERRA": ROOT / "scrape_verra.py",
}
# ── 데이터 ꡬ쑰 ────────────────────────────────────────────────────────────
@dataclass
class FileRecord:
registry: str
filename: str # 원본 파일λͺ…
kb_filename: str # knowledge_base λ‚΄ 파일λͺ…
file_hash: str # MD5 ν•΄μ‹œ
file_size: int
last_seen: str # ISO λ‚ μ§œ
in_kb: bool = False # knowledge_base에 λ³΅μ‚¬λλŠ”μ§€
def compute_hash(path: Path) -> str:
"""파일 MD5 ν•΄μ‹œ 계산"""
h = hashlib.md5()
with open(path, "rb") as f:
for chunk in iter(lambda: f.read(65536), b""):
h.update(chunk)
return h.hexdigest()
def load_state() -> dict[str, FileRecord]:
"""μƒνƒœ 파일 λ‘œλ“œ"""
if not STATE_FILE.exists():
return {}
try:
with open(STATE_FILE, "r", encoding="utf-8") as f:
raw = json.load(f)
return {k: FileRecord(**v) for k, v in raw.items()}
except Exception:
return {}
def save_state(state: dict[str, FileRecord]) -> None:
"""μƒνƒœ 파일 μ €μž₯"""
with open(STATE_FILE, "w", encoding="utf-8") as f:
json.dump({k: asdict(v) for k, v in state.items()}, f,
ensure_ascii=False, indent=2)
def make_state_key(registry: str, filename: str) -> str:
return f"{registry}:{filename}"
# ── μŠ€μΊ” ──────────────────────────────────────────────────────────────────
def scan_registry(registry_id: str, config: dict) -> dict[str, FileRecord]:
"""λ ˆμ§€μŠ€νŠΈλ¦¬ 디렉토리 μŠ€μΊ” β†’ ν˜„μž¬ 파일 λͺ©λ‘ λ°˜ν™˜"""
dir_path: Path = config["dir"]
prefix: str = config["prefix"]
records = {}
if not dir_path.exists():
return records
for file_path in sorted(dir_path.iterdir()):
if file_path.suffix.lower() not in ALLOWED_EXT:
continue
if not file_path.is_file():
continue
key = make_state_key(registry_id, file_path.name)
file_hash = compute_hash(file_path)
kb_filename = prefix + file_path.name
records[key] = FileRecord(
registry=registry_id,
filename=file_path.name,
kb_filename=kb_filename,
file_hash=file_hash,
file_size=file_path.stat().st_size,
last_seen=datetime.now().isoformat(),
in_kb=(KNOWLEDGE_BASE / kb_filename).exists(),
)
return records
def scan_all() -> dict[str, FileRecord]:
"""λͺ¨λ“  λ ˆμ§€μŠ€νŠΈλ¦¬ μŠ€μΊ”"""
all_records = {}
for registry_id, config in REGISTRIES.items():
records = scan_registry(registry_id, config)
all_records.update(records)
return all_records
# ── λ³€κ²½ 감지 ─────────────────────────────────────────────────────────────
@dataclass
class ChangeReport:
new_files: list[FileRecord] # μ‹ κ·œ 방법둠
updated_files: list[FileRecord] # κ°œμ •λœ 방법둠
removed_keys: list[str] # μ‚­μ œλœ 방법둠
not_in_kb: list[FileRecord] # λ‘œμ»¬μ—” μžˆμ§€λ§Œ KB에 μ—†λŠ” 파일
def detect_changes(
current: dict[str, FileRecord],
previous: dict[str, FileRecord],
) -> ChangeReport:
"""ν˜„μž¬ μƒνƒœ vs 이전 μƒνƒœ 비ꡐ"""
new_files = []
updated_files = []
not_in_kb = []
for key, rec in current.items():
if key not in previous:
new_files.append(rec)
elif previous[key].file_hash != rec.file_hash:
updated_files.append(rec)
elif not rec.in_kb:
not_in_kb.append(rec)
removed_keys = [k for k in previous if k not in current]
return ChangeReport(
new_files=new_files,
updated_files=updated_files,
removed_keys=removed_keys,
not_in_kb=not_in_kb,
)
# ── Knowledge Base 동기화 ─────────────────────────────────────────────────
def sync_to_knowledge_base(
records: list[FileRecord],
dry_run: bool = False,
) -> list[FileRecord]:
"""νŒŒμΌλ“€μ„ knowledge_base/에 볡사"""
synced = []
KNOWLEDGE_BASE.mkdir(parents=True, exist_ok=True)
for rec in records:
# 원본 파일 경둜 볡원
config = REGISTRIES[rec.registry]
src = config["dir"] / rec.filename
dst = KNOWLEDGE_BASE / rec.kb_filename
if not src.exists():
print(f" [κ²½κ³ ] 원본 파일 μ—†μŒ: {src}")
continue
if dry_run:
print(f" [DRY-RUN] 볡사 μ˜ˆμ •: {rec.kb_filename}")
synced.append(rec)
continue
shutil.copy2(src, dst)
rec.in_kb = True
synced.append(rec)
print(f" 볡사: {rec.kb_filename} ({rec.file_size / 1024:.0f}KB)")
return synced
def remove_from_knowledge_base(
keys: list[str],
state: dict[str, FileRecord],
dry_run: bool = False,
) -> None:
"""μ‚­μ œλœ 방법둠을 knowledge_baseμ—μ„œλ„ 제거"""
for key in keys:
if key not in state:
continue
rec = state[key]
dst = KNOWLEDGE_BASE / rec.kb_filename
if dst.exists():
if dry_run:
print(f" [DRY-RUN] μ‚­μ œ μ˜ˆμ •: {rec.kb_filename}")
else:
dst.unlink()
print(f" μ‚­μ œ: {rec.kb_filename}")
# ── 슀크래퍼 μ‹€ν–‰ ────────────────────────────────────────────────────────
def run_scrapers(targets: Optional[list[str]] = None) -> None:
"""원격 λ ˆμ§€μŠ€νŠΈλ¦¬ μŠ€ν¬λž˜ν•‘ (μƒˆ 파일 λ‹€μš΄λ‘œλ“œ)"""
targets = targets or list(SCRAPERS.keys())
for name in targets:
scraper = SCRAPERS.get(name)
if not scraper or not scraper.exists():
print(f" [κ²½κ³ ] 슀크래퍼 μ—†μŒ: {name}")
continue
print(f"\n μŠ€ν¬λž˜ν•‘: {name}...")
result = subprocess.run(
[sys.executable, str(scraper)],
cwd=str(ROOT),
)
if result.returncode != 0:
print(f" [κ²½κ³ ] {name} 슀크래퍼 μ‹€νŒ¨ (μ’…λ£Œμ½”λ“œ {result.returncode})")
# ── update_all.py μ‹€ν–‰ ────────────────────────────────────────────────────
def run_update_pipeline(dry_run: bool = False) -> bool:
"""update_all.py μ‹€ν–‰ (ChromaDB + GraphDB + 배포)"""
update_script = ROOT / "update_all.py"
if not update_script.exists():
print(" [였λ₯˜] update_all.py μ—†μŒ")
return False
cmd = [sys.executable, str(update_script), "--yes", "--skip-graph", "--skip-community"]
if dry_run:
cmd.append("--dry-run")
print(f"\n update_all.py μ‹€ν–‰ 쀑...")
result = subprocess.run(cmd, cwd=str(ROOT))
return result.returncode == 0
# ── 리포트 좜λ ₯ ──────────────────────────────────────────────────────────
def print_report(report: ChangeReport, current: dict[str, FileRecord]) -> None:
"""λ³€κ²½ 감지 κ²°κ³Ό 좜λ ₯"""
total = len(report.new_files) + len(report.updated_files) + len(report.not_in_kb)
print("\n" + "=" * 60)
print(" 방법둠 λͺ¨λ‹ˆν„°λ§ κ²°κ³Ό")
print("=" * 60)
# λ ˆμ§€μŠ€νŠΈλ¦¬λ³„ 톡계
registry_counts = {}
for rec in current.values():
registry_counts[rec.registry] = registry_counts.get(rec.registry, 0) + 1
print("\n λ ˆμ§€μŠ€νŠΈλ¦¬λ³„ 보유 방법둠:")
for rid, config in REGISTRIES.items():
count = registry_counts.get(rid, 0)
label = config["label"]
print(f" {label:25s}: {count:4d}개")
print(f"\n 총 보유 방법둠: {len(current)}개")
# 변경사항
if report.new_files:
print(f"\n [μ‹ κ·œ] {len(report.new_files)}개 발견:")
for rec in report.new_files[:10]:
label = REGISTRIES[rec.registry]["label"]
print(f" [{label}] {rec.filename[:60]}")
if len(report.new_files) > 10:
print(f" ... μ™Έ {len(report.new_files)-10}개")
if report.updated_files:
print(f"\n [κ°œμ •] {len(report.updated_files)}개 감지:")
for rec in report.updated_files[:10]:
label = REGISTRIES[rec.registry]["label"]
print(f" [{label}] {rec.filename[:60]}")
if report.not_in_kb:
print(f"\n [미동기] KB에 μ—†λŠ” 파일 {len(report.not_in_kb)}개")
if report.removed_keys:
print(f"\n [μ‚­μ œ] {len(report.removed_keys)}개 방법둠 제거됨")
if total == 0:
print("\n 변경사항 μ—†μŒ - Knowledge Base μ΅œμ‹  μƒνƒœ")
else:
print(f"\n β†’ Knowledge Base κ°±μ‹  ν•„μš”: {total}개 파일")
print("=" * 60)
# ── 메인 ─────────────────────────────────────────────────────────────────
def main():
parser = argparse.ArgumentParser(
description="방법둠 μžλ™ λͺ¨λ‹ˆν„°λ§ νŒŒμ΄ν”„λΌμΈ"
)
parser.add_argument("--check-only", action="store_true",
help="λ³€κ²½ κ°μ§€λ§Œ (KB κ°±μ‹  μ—†μŒ)")
parser.add_argument("--scrape", action="store_true",
help="원격 λ ˆμ§€μŠ€νŠΈλ¦¬ μŠ€ν¬λž˜ν•‘ ν›„ 감지")
parser.add_argument("--scrape-only", nargs="+", metavar="REGISTRY",
help=f"νŠΉμ • λ ˆμ§€μŠ€νŠΈλ¦¬λ§Œ μŠ€ν¬λž˜ν•‘ ({', '.join(SCRAPERS.keys())})")
parser.add_argument("--dry-run", action="store_true",
help="미리보기 (μ‹€μ œ 파일 λ³€κ²½ μ—†μŒ)")
parser.add_argument("--skip-update", action="store_true",
help="update_all.py μ‹€ν–‰ κ±΄λ„ˆλœ€")
args = parser.parse_args()
print("=" * 60)
print(" 방법둠 μžλ™ λͺ¨λ‹ˆν„°λ§ νŒŒμ΄ν”„λΌμΈ")
print(f" μ‹€ν–‰ μ‹œκ°: {datetime.now().strftime('%Y-%m-%d %H:%M:%S')}")
print("=" * 60)
# Step 0: μŠ€ν¬λž˜ν•‘ (μ˜΅μ…˜)
if args.scrape or args.scrape_only:
print("\n[Step 0] 원격 λ ˆμ§€μŠ€νŠΈλ¦¬ μŠ€ν¬λž˜ν•‘")
targets = args.scrape_only or None
run_scrapers(targets)
# Step 1: ν˜„μž¬ μƒνƒœ μŠ€μΊ”
print("\n[Step 1] 둜컬 방법둠 디렉토리 μŠ€μΊ” 쀑...")
t0 = time.perf_counter()
current_state = scan_all()
print(f" μ™„λ£Œ: {len(current_state)}개 파일 ({time.perf_counter()-t0:.1f}초)")
# Step 2: 이전 μƒνƒœ λ‘œλ“œ & 비ꡐ
print("\n[Step 2] 변경사항 감지 쀑...")
previous_state = load_state()
report = detect_changes(current_state, previous_state)
# 리포트 좜λ ₯
print_report(report, current_state)
# --check-only 이면 μ—¬κΈ°μ„œ μ’…λ£Œ
if args.check_only:
save_state(current_state)
return
# Step 3: Knowledge Base 동기화
files_to_sync = report.new_files + report.updated_files + report.not_in_kb
if files_to_sync:
print(f"\n[Step 3] Knowledge Base 동기화 ({len(files_to_sync)}개)")
synced = sync_to_knowledge_base(files_to_sync, dry_run=args.dry_run)
# μ‚­μ œλœ 방법둠 KBμ—μ„œλ„ 제거
if report.removed_keys:
remove_from_knowledge_base(
report.removed_keys, previous_state, dry_run=args.dry_run
)
# μƒνƒœ 파일 κ°±μ‹ 
if not args.dry_run:
# ν˜„μž¬ μƒνƒœλ‘œ μ—…λ°μ΄νŠΈ (removedλŠ” μ œμ™Έ)
new_state = {k: v for k, v in current_state.items()}
save_state(new_state)
print(f"\n μƒνƒœ 파일 μ €μž₯: {STATE_FILE}")
# Step 4: update_all.py μ‹€ν–‰
if not args.skip_update and synced:
print(f"\n[Step 4] Knowledge Base μž¬κ΅¬μΆ• (update_all.py)")
if args.dry_run:
print(" [DRY-RUN] update_all.py μ‹€ν–‰ κ±΄λ„ˆλœ€")
else:
success = run_update_pipeline(dry_run=False)
if success:
print(" Knowledge Base μž¬κ΅¬μΆ• μ™„λ£Œ")
print("\n λ‹€μŒ 단계:")
print(" git push origin main # HuggingFace Spaces 배포")
else:
print(" [κ²½κ³ ] update_all.py μ‹€νŒ¨ - μˆ˜λ™ 확인 ν•„μš”")
else:
# λ³€κ²½ 없어도 μƒνƒœλŠ” μ €μž₯ (last_seen κ°±μ‹ )
if not args.dry_run:
save_state(current_state)
print("\n Knowledge Base κ°±μ‹  λΆˆν•„μš”")
print(f"\nμ™„λ£Œ: {datetime.now().strftime('%Y-%m-%d %H:%M:%S')}\n")
if __name__ == "__main__":
main()