"""Reproduce the public-library calculations offline using Python 3 standard library. Usage: python reproduce.py [--check] Input is operator-transcribed published figures, not respondent microdata. No service accounts, network, customer data, API calls or publication. """ from pathlib import Path from decimal import Decimal, ROUND_HALF_UP import csv,hashlib,json,sys ROOT=Path(__file__).resolve().parent def read(name):return json.loads((ROOT/name).read_text(encoding='utf-8')) def rounded(value):return float(value.quantize(Decimal('0.01'),rounding=ROUND_HALF_UP)) def calculate(): data=read('raw-metrics.json'); by={r['code']:r for r in data} assert len(by)==len(data),'duplicate metric code' def n(c):return Decimal(str(by[c]['value'])) specs=[ ('D01','연간 발표 방문량 차이',['M03','M04'],n('M04')-n('M03'),'명','2830000 - 3000000','발표된 반올림·추정 수치의 단순 차이; 운영일·범위 미보정'), ('D02','연간 발표 방문량 변화율',['M03','M04'],(n('M04')/n('M03')-1)*100,'%','(2830000 / 3000000 - 1) * 100','인과 효과·운영 효율·고유 방문자 변화가 아님'), ('D03','팝업 운영처당 북키트 대여량',['M06','M07'],n('M07')/n('M06'),'개/곳','4962 / 108','배포 총량의 산술평균; 모든 기관이 이 수량을 받았다는 뜻이 아님'), ('D04','발표 독서자 비율 차이',['M08','M09'],n('M09')-n('M08'),'%p','85.4 - 83.5','서로 다른 해의 집계 차이; 동일인 개선·전국 독서율 효과가 아님'), ('D05','발표 미반납률 차이',['M10','M11'],n('M11')-n('M10'),'%p','0.51 - 0.71','책 수 분모가 없어 분실 권수나 비용 절감으로 환산하지 않음'), ] rows=[{'code':c,'measure':label,'inputCodes':codes,'value':rounded(v),'unit':unit,'formula':formula,'interpretation':note,'status':'calculated-from-published-values'} for c,label,codes,v,unit,formula,note in specs] rows.append({'code':'D06','measure':'독서자 비율 2024→2025 정확 변화','inputCodes':['M09','M14'],'value':None,'unit':'%p','formula':'not computed','interpretation':'해당 연도 원문은 10명 중 8명의 반올림 표현. 동일 정밀도·분모·조사설계가 확인되지 않아 정확한 전년 대비 효과를 만들지 않음.','status':'not-comparable-for-exact-change'}) return rows def main(): rows=calculate(); raw=read('raw-metrics.json'); comparison=read('comparison-conditions.json') assert len(comparison)==24 assert {r['service'] for r in comparison}=={'general','smart','relay','seoul-smart'} assert len({(r['service'],r['criterion']) for r in comparison})==24 assert rows[0]['value']==-170000 and rows[1]['value']==-5.67 assert rows[2]['value']==45.94 and rows[3]['value']==1.9 and rows[4]['value']==-0.2 assert rows[5]['value'] is None hashes={f:hashlib.sha256((ROOT/f).read_bytes()).hexdigest() for f in ['raw-metrics.json','comparison-conditions.json','sources.json']} result={'scope':'offline public-document analysis; no live service or AI experiment','rawMetricRows':len(raw),'comparisonCells':len(comparison),'derivedRows':len(rows),'numericDerivedRows':5,'withheldDerivedRows':1,'inputSha256':hashes,'calculations':rows} target=ROOT/'reproduced-results.json' if '--check' in sys.argv: assert read(target.name)==result,'stored output does not reproduce' print('PASS: stored calculations, input hashes and non-comparable guard reproduce') else: target.write_text(json.dumps(result,ensure_ascii=False,indent=2)+'\n',encoding='utf-8',newline='\n') with (ROOT/'analysis-results.csv').open('w',encoding='utf-8-sig',newline='') as f: w=csv.DictWriter(f,fieldnames=list(rows[0]));w.writeheader();w.writerows({**r,'inputCodes':','.join(r['inputCodes'])} for r in rows) print(json.dumps(result,ensure_ascii=False,indent=2)) if __name__=='__main__':main()