"""Reproduce article evidence from UCI Online Retail; never modifies the source.

Usage: python3 analyze_retail.py /path/to/online-retail.zip [output-directory]
Outputs are derived teaching data, not an operational customer targeting model.
"""
from pathlib import Path
import hashlib
import io
import json
import sys
import zipfile
import pandas as pd

OUT = Path(sys.argv[2]).resolve() if len(sys.argv) > 2 else Path.cwd() / "rfm-results"
OUT.mkdir(parents=True, exist_ok=True)
source = Path(sys.argv[1])
archive = source.read_bytes()
with zipfile.ZipFile(io.BytesIO(archive)) as z:
    name = next(n for n in z.namelist() if n.endswith('.xlsx'))
    xlsx = z.read(name)
df = pd.read_excel(io.BytesIO(xlsx), dtype={'InvoiceNo': str, 'StockCode': str})
counts = {'source_rows': len(df), 'exact_duplicate_rows_preserved': int(df.duplicated().sum())}
working = df.copy()
rules = [
    ('missing_customer_id', lambda d: d.CustomerID.isna()),
    ('cancelled_invoice', lambda d: d.InvoiceNo.str.upper().str.startswith('C')),
    ('nonpositive_quantity', lambda d: d.Quantity.le(0)),
    ('nonpositive_unit_price', lambda d: d.UnitPrice.le(0)),
]
for label, mask_fn in rules:
    mask = mask_fn(working)
    counts[label] = int(mask.sum())
    working = working.loc[~mask].copy()
working['CustomerID'] = working.CustomerID.astype(int).astype(str)
working['AmountGBP'] = working.Quantity * working.UnitPrice
counts['eligible_rows'] = len(working)
assert sum(counts[k] for k, _ in rules) + len(working) == len(df)
assert working[['InvoiceDate', 'InvoiceNo']].notna().all().all()
reference = pd.Timestamp('2011-12-10')
rfm = working.groupby('CustomerID').agg(
    last_purchase=('InvoiceDate', 'max'),
    frequency=('InvoiceNo', 'nunique'),
    monetary_gbp=('AmountGBP', 'sum'),
    source_lines=('InvoiceNo', 'size'),
)
rfm['recency_days'] = (reference - rfm.last_purchase.dt.normalize()).dt.days
rfm = rfm[['recency_days', 'frequency', 'monetary_gbp', 'last_purchase', 'source_lines']]
assert rfm.recency_days.min() >= 1
assert abs(rfm.monetary_gbp.sum() - working.AmountGBP.sum()) < 1e-6
ranked = rfm.sort_values(['monetary_gbp', 'frequency'], ascending=False)
top_n = __import__('math').ceil(len(rfm) * .2)
ids = ['12347', '12348', '12349', '12350', '12352', '12353']
example = rfm.loc[ids]
raw_sample = df.loc[df.CustomerID.isin([int(x) for x in ids])].copy()
raw_sample.to_csv(OUT / 'six-customers-source.csv', index=False, float_format='%.6f')
rfm.to_csv(OUT / 'rfm-all-customers.csv', float_format='%.6f', date_format='%Y-%m-%d %H:%M:%S')
example.to_csv(OUT / 'rfm-six-customers.csv', float_format='%.6f', date_format='%Y-%m-%d %H:%M:%S')
summary = {
    'dataset_url': 'https://archive.ics.uci.edu/dataset/352/online+retail',
    'download_url': 'https://archive.ics.uci.edu/static/public/352/online+retail.zip',
    'attribution': 'Chen, D. (2015). Online Retail. UCI Machine Learning Repository. DOI: 10.24432/C5BW33. CC BY 4.0.',
    'analysis_date': '2026-10-03',
    'archive_sha256': hashlib.sha256(archive).hexdigest(),
    'xlsx_sha256': hashlib.sha256(xlsx).hexdigest(),
    'source_period': [str(df.InvoiceDate.min()), str(df.InvoiceDate.max())],
    'reference_date': str(reference.date()),
    'currency': 'GBP',
    'monetary_definition': 'Sum of positive quantity times positive unit price on non-cancelled invoices with CustomerID; not net revenue or profit.',
    'duplicate_policy': 'Preserve exact duplicates; no unique line ID establishes accidental duplication.',
    'counts': counts,
    'customers': len(rfm),
    'invoices': int(working.InvoiceNo.nunique()),
    'positive_purchase_amount_gbp': round(float(rfm.monetary_gbp.sum()), 6),
    'top_20_percent_customer_count': top_n,
    'top_20_percent_positive_purchase_share': float(ranked.head(top_n).monetary_gbp.sum() / ranked.monetary_gbp.sum()),
    'practice_recent_repeat_customers_R_le_30_F_ge_5': int(((rfm.recency_days <= 30) & (rfm.frequency >= 5)).sum()),
    'practice_lapsed_repeat_customers_R_gt_90_F_ge_5': int(((rfm.recency_days > 90) & (rfm.frequency >= 5)).sum()),
    'practice_thresholds': 'Illustrative analyst choices, not validated segment boundaries; recent and lapsed groups are disjoint but do not cover all customers.',
    'example_customers': json.loads(example.reset_index().to_json(orient='records', date_format='iso')),
    'sample_selection': 'Six fixed IDs for hand calculation; deliberately selected, not statistically representative.',
    'sample_source_rows': len(raw_sample),
}
(OUT / 'analysis-summary.json').write_text(json.dumps(summary, ensure_ascii=False, indent=2) + '\n')
print(json.dumps(summary, ensure_ascii=False, indent=2))
