"""
Thu thập toàn bộ lịch sử chat Messenger của một Fanpage về tệp JSONL (mỗi dòng một hội thoại).

Dùng để đọc lại cách chủ trang đã tư vấn, rút ra câu trả lời tiêu chuẩn cho kho tư vấn.
Tệp ra chứa tin nhắn thật của khách: để trong private/, KHÔNG commit, KHÔNG đưa lên web.

Chạy (trên máy chủ, thư mục chatbot):
    moitruong/bin/python -m kenh.thu_thap_fanpage <id kênh trong chat_kenh> <tệp ra>
"""
from __future__ import annotations

import json
import os
import sys
import time

try:
    from congcu import csdl
except ImportError:
    sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
    from congcu import csdl  # type: ignore

import httpx

from kenh.messenger import GRAPH


def _lay_kenh(kenh_id: int) -> dict:
    conn = csdl.ket_noi()
    with conn:
        with conn.cursor() as ct:
            ct.execute('SELECT page_id, page_token FROM chat_kenh WHERE id = %s', (kenh_id,))
            return ct.fetchone() or {}


def _goi(url: str, token: str, params: dict | None = None) -> dict:
    for lan in range(4):
        r = httpx.get(url, params=params, headers={'Authorization': f'Bearer {token}'}, timeout=30)
        d = r.json()
        if 'error' not in d:
            return d
        if lan == 3:
            raise RuntimeError(d['error'].get('message'))
        time.sleep(2 * (lan + 1))
    return {}


def thu_thap(kenh_id: int, tep_ra: str) -> None:
    k = _lay_kenh(kenh_id)
    page_id, token = str(k.get('page_id') or ''), str(k.get('page_token') or '')
    if not page_id or not token:
        raise SystemExit(f'Không thấy kênh {kenh_id} hoặc thiếu token')

    url = f'{GRAPH}/{page_id}/conversations'
    params: dict | None = {'platform': 'messenger', 'limit': 50,
                           'fields': 'id,updated_time,participants'}
    so_hoi_thoai = so_tin = 0
    with open(tep_ra, 'w', encoding='utf-8') as f:
        while url:
            trang = _goi(url, token, params)
            params = None  # link "next" đã chứa đủ tham số
            for hc in trang.get('data', []):
                tin = []
                u2 = f"{GRAPH}/{hc['id']}/messages"
                p2: dict | None = {'limit': 100, 'fields': 'message,from,created_time'}
                while u2:
                    t2 = _goi(u2, token, p2)
                    p2 = None
                    for m in t2.get('data', []):
                        tu = m.get('from') or {}
                        tin.append({
                            'luc': m.get('created_time'),
                            'ai': 'trang' if str(tu.get('id')) == page_id else 'khach',
                            'noi_dung': m.get('message') or '',
                        })
                    u2 = t2.get('paging', {}).get('next')
                tin.sort(key=lambda x: x['luc'] or '')
                f.write(json.dumps({'id': hc['id'], 'cap_nhat': hc.get('updated_time'), 'tin': tin},
                                   ensure_ascii=False) + '\n')
                so_hoi_thoai += 1
                so_tin += len(tin)
            print(f'... {so_hoi_thoai} hội thoại, {so_tin} tin', flush=True)
            url = trang.get('paging', {}).get('next')
    print(f'XONG: {so_hoi_thoai} hội thoại, {so_tin} tin -> {tep_ra}')


if __name__ == '__main__':
    if len(sys.argv) != 3:
        raise SystemExit(__doc__)
    thu_thap(int(sys.argv[1]), sys.argv[2])
