"""Recompute a bounded audit of a public research dataset; never runs remote code.

Source: GEO Citation Lab contributors, GEO Citation Lab,
https://github.com/yaojingang/geo-citation-lab
See its LICENSE-CONTENT and THIRD_PARTY_NOTICES for separate material rights.
Output: counts and means only. Original rows are read in memory, not republished.
"""
import collections
import csv
import hashlib
import io
import json
from pathlib import Path
from urllib.request import Request, urlopen

URL = (
    'https://raw.githubusercontent.com/yaojingang/geo-citation-lab/'
    '25dd7e66324e4acd005ef146d38f57d87b6977df/'
    '01-geo-experiment-data-report/02-data/features_all_platforms_72.csv'
)
EXPECTED_SHA256 = 'd67ca08531cbc72347d02fda186e8aa5650866930b13c08acfc26888959dca67'

def main():
    with urlopen(Request(URL, headers={'User-Agent': 'SEARCH-KOREA-Research/1.0'}), timeout=30) as response:
        raw = response.read()
    digest = hashlib.sha256(raw).hexdigest()
    if digest != EXPECTED_SHA256:
        raise ValueError('Source bytes differ from the reviewed snapshot; inspect before reuse.')
    reader = csv.DictReader(io.StringIO(raw.decode('utf-8-sig')))
    rows = list(reader)
    result = {
        'source_url': URL, 'sha256': digest, 'bytes': len(raw),
        'rows': len(rows), 'columns': reader.fieldnames,
        'audit_scope': 'Counts, fetch status and stored score means only; not a reproduction of original collection, factual verification, or causal effects.',
        'by_platform': [],
    }
    for platform in ['chatgpt', 'google', 'perplexity']:
        group = [row for row in rows if row['platform'] == platform]
        fetched = [row for row in group if row['fetch_ok'] == 'True']
        scores = [float(row['influence_score']) for row in fetched if row['influence_score']]
        result['by_platform'].append({
            'platform': platform, 'all_rows': len(group), 'fetch_ok': len(fetched),
            'fetch_fail': len(group) - len(fetched),
            'fetch_ok_percent': round(100 * len(fetched) / len(group), 2),
            'mean_influence_score': sum(scores) / len(scores),
        })
    for column in ['platform', 'fetch_ok']:
        counts = collections.Counter(row[column] for row in rows)
        result[column] = {'unique': len(counts), 'top_values': counts.most_common()}
    target = Path(__file__).with_name('dataset_audit.json')
    target.write_text(json.dumps(result, ensure_ascii=False, indent=2), encoding='utf-8')
    print(json.dumps({'rows': len(rows), 'fields': len(reader.fieldnames), 'by_platform': result['by_platform']}, ensure_ascii=True))

if __name__ == '__main__':
    main()
