#!/usr/bin/env python3
"""Summarize manually reviewed benchmark CSV rows. Python 3.9+, standard library only.
No ASR, WER calculation, automatic meaning judgment, or network access.
Usage: python3 summarize-dictation-benchmark.py results.csv > summary.json
"""
import argparse
import csv
import json
import math
import statistics
import sys
from collections import Counter, defaultdict
from pathlib import Path

GROUP_FIELDS = ['speaker_id', 'test_date', 'tool', 'app_version', 'os', 'hardware',
                'mic', 'target_app', 'language_selection', 'condition',
                'dictionary_changes', 'config_id', 'voice_model', 'cleanup_model',
                'context_setting', 'input_mode', 'reference_revision', 'group']
RATINGS = ['meaning_preserved', 'terms_preserved', 'numbers_preserved']
TIMES = ['latency_ms', 'correction_seconds']
STATUSES = ['not_run', 'completed', 'failed', 'not_applicable']
REQUIRED = GROUP_FIELDS + RATINGS + TIMES + ['sample_id', 'trial', 'reference', 'output', 'status', 'notes']


def summarize(path):
    groups = defaultdict(list)
    counts = Counter()
    seen = set()
    with Path(path).open(encoding='utf-8-sig', newline='') as source:
        reader = csv.DictReader(source)
        headers = reader.fieldnames or []
        if len(headers) != len(set(headers)) or not set(REQUIRED).issubset(headers):
            raise ValueError('Missing required columns or duplicate CSV headers.')
        for line, row in enumerate(reader, 2):
            if None in row or any(value is None for value in row.values()):
                raise ValueError(f'Row {line}: wrong column count.')
            status = row['status'].strip()
            if status not in STATUSES:
                raise ValueError(f'Row {line}: invalid status.')
            counts[status] += 1
            for name in RATINGS:
                value = row[name].strip()
                if value not in ['', 'yes', 'no', 'not_applicable']:
                    raise ValueError(f'Row {line}: invalid {name}.')
                row[name] = value
            for name in TIMES:
                value = row[name].strip()
                if value:
                    try:
                        value = float(value)
                    except ValueError:
                        raise ValueError(f'Row {line}: invalid {name}.') from None
                    if not math.isfinite(value) or value < 0:
                        raise ValueError(f'Row {line}: {name} must be finite and non-negative.')
                row[name] = value if value != '' else None
            if status in ['not_run', 'not_applicable']:
                if row['output'].strip() or any(row[n] for n in RATINGS) or any(row[n] is not None for n in TIMES):
                    raise ValueError(f'Row {line}: unrun/inapplicable row contains results.')
                continue
            for name in GROUP_FIELDS + ['sample_id', 'trial', 'reference']:
                if name != 'dictionary_changes' and not row[name].strip():
                    raise ValueError(f'Row {line}: fill {name}; use unknown if unavailable.')
            if not row['trial'].strip().isascii() or not row['trial'].strip().isdigit() or int(row['trial']) < 1:
                raise ValueError(f'Row {line}: trial must be a positive integer.')
            if status == 'failed' and not row['notes'].strip():
                raise ValueError(f'Row {line}: failed attempt needs notes.')
            key = tuple(row[name].strip() for name in GROUP_FIELDS)
            attempt = key + (row['sample_id'].strip(), int(row['trial']))
            if attempt in seen:
                raise ValueError(f'Row {line}: duplicate attempt; assign a new trial number.')
            seen.add(attempt)
            groups[key].append(row)
    output = []
    for key, rows in sorted(groups.items()):
        completed = [r for r in rows if r['status'].strip() == 'completed']
        entry = {'configuration': dict(zip(GROUP_FIELDS, key)), 'attempted': len(rows),
                 'completed': len(completed), 'failed': len(rows) - len(completed),
                 'sample_ids': sorted({r['sample_id'].strip() for r in rows}),
                 'ratings': {}, 'timings': {}}
        for name in RATINGS:
            tally = Counter(r[name] for r in completed)
            reviewed = tally['yes'] + tally['no']
            entry['ratings'][name] = {'yes': tally['yes'], 'no': tally['no'],
                'not_applicable': tally['not_applicable'], 'missing': tally[''],
                'reviewed': reviewed, 'yes_fraction': tally['yes'] / reviewed if reviewed else None}
        for name in TIMES:
            values = [r[name] for r in completed if r[name] is not None]
            entry['timings'][name] = {'observations': len(values), 'missing': len(completed) - len(values),
                                     'median': statistics.median(values) if values else None}
        output.append(entry)
    return {'schema_version': 1, 'scope': 'Manual ratings and completed-attempt timings; not ASR accuracy or WER.',
            'row_counts': {s: counts[s] for s in STATUSES}, 'groups': output}


def main():
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument('csv_path')
    args = parser.parse_args()
    try:
        result = summarize(args.csv_path)
    except (ValueError, OSError, csv.Error) as error:
        parser.exit(2, f'Error: {error}\n')
    json.dump(result, sys.stdout, ensure_ascii=False, indent=2, allow_nan=False)
    sys.stdout.write('\n')


if __name__ == '__main__':
    main()
