#!/usr/bin/env python3
"""Local stable-ID facility review with exact checks and a fresh evidence export.

No network, accounts or hosted upload service. CSV decimals are converted exactly;
source bytes and the normalized model are preserved. Existing reports are never
replaced. Optional OR-Tools is used only when explicitly selected by the caller.
"""
import argparse
import csv
from datetime import datetime, timezone
from fractions import Fraction
from functools import reduce
import hashlib
import html
import io
import json
import math
from pathlib import Path
import re
import sys

from metric_facility_reference import MODEL,MAX_POINTS,calculate,parse,primal,dual,rational

SCHEMA='facility-review-v1'
FILES=('study.json','locations.csv','distances.csv','plan.csv')
MAX_FILE_BYTES=2*1024*1024
ID=re.compile(r'[A-Za-z0-9][A-Za-z0-9_.:-]{0,63}\Z')
NUMBER=re.compile(r'-?(?:[0-9]+(?:\.[0-9]+)?|[0-9]+/[0-9]+)\Z')


def sha(data):return hashlib.sha256(data).hexdigest()


def unique_object(pairs):
    out={}
    for k,v in pairs:
        if k in out:raise ValueError('Duplicate JSON key: '+k)
        out[k]=v
    return out


def reject_constant(value):raise ValueError('Non-finite JSON number: '+value)


def decoded_json(raw):return json.loads(raw.decode('utf-8-sig'),object_pairs_hook=unique_object,parse_constant=reject_constant)


def csv_rows(raw,columns):
    reader=csv.DictReader(io.StringIO(raw.decode('utf-8-sig'),newline=''),strict=True)
    if reader.fieldnames is None or len(reader.fieldnames)!=len(columns) or set(reader.fieldnames)!=set(columns):
        raise ValueError('CSV header must contain exactly: '+', '.join(columns))
    rows=[]
    for row in reader:
        if None in row or any(v is None for v in row.values()):raise ValueError('CSV row has wrong field count')
        rows.append(row)
    return rows


def text(value,label,limit,empty=False):
    if not isinstance(value,str) or (not value and not empty) or len(value)>limit or any(ord(c)<32 for c in value):
        raise ValueError('Invalid '+label)
    return value


def number(value):
    if not isinstance(value,str) or len(value)>160 or not NUMBER.fullmatch(value):
        raise ValueError('Use bounded decimal or rational strings, without rounding or exponents')
    try:q=Fraction(value)
    except (ValueError,ZeroDivisionError):raise ValueError('Invalid rational cell') from None
    rational(str(q))
    return q


def load_inputs(directory):
    unexpected=sorted(p.name for p in directory.iterdir() if p.name not in FILES+('dual.json',))
    if unexpected:raise ValueError('Unexpected entries in input directory: '+', '.join(unexpected))
    raw={}
    for name in FILES+('dual.json',):
        path=directory/name
        if name=='dual.json' and not path.exists():continue
        with path.open('rb') as f:data=f.read(MAX_FILE_BYTES+1)
        if len(data)>MAX_FILE_BYTES:raise ValueError(name+' exceeds two MiB')
        raw[name]=data
    return raw


def normalize(raw):
    study=decoded_json(raw['study.json'])
    fields={'schema','title','units','distance_provenance','k','model','additional_constraints'}
    if not isinstance(study,dict) or set(study)!=fields or study['schema']!=SCHEMA:raise ValueError('Invalid study schema/fields')
    for name,limit in [('title',160),('units',64),('distance_provenance',2000)]:text(study[name],name,limit)
    if not isinstance(study['model'],dict) or not isinstance(study['additional_constraints'],list):raise ValueError('Model and constraint declarations are required')
    for value in study['additional_constraints']:text(value,'additional constraint',500)
    rows=csv_rows(raw['locations.csv'],('location_id','label','is_client','is_candidate','demand_weight','capacity','mandatory'))
    if not 1<=len(rows)<=MAX_POINTS:raise ValueError('Require 1..48 locations')
    locations={};unsupported=[]
    if study['model']!=MODEL:unsupported.append('The declared model differs from the supported unweighted, uncapacitated strict-metric model.')
    if study['additional_constraints']:unsupported.append('Additional constraints are not implemented: '+'; '.join(study['additional_constraints']))
    for row in rows:
        key=row['location_id']
        if not ID.fullmatch(key):raise ValueError('Invalid location ID: '+key)
        if key in locations:raise ValueError('Duplicate location ID: '+key)
        text(row['label'],'location label',160,True)
        if any(row[k] not in ('true','false') for k in ('is_client','is_candidate','mandatory')):raise ValueError('Flags must be exactly true or false')
        weight=number(row['demand_weight'])
        if weight<0:raise ValueError('Demand weight must be nonnegative')
        expected=1 if row['is_client']=='true' else 0
        if weight!=expected:unsupported.append('Demand weight at '+key+' is unsupported for its declared client role.')
        if row['capacity']!='':
            if number(row['capacity'])<0:raise ValueError('Capacity must be nonnegative')
            unsupported.append('Finite capacity at '+key+' is not implemented.')
        if row['mandatory']=='true':unsupported.append('Mandatory-site constraint at '+key+' is not implemented.')
        locations[key]=dict(row)
    ids=sorted(locations);indices={k:i for i,k in enumerate(ids)};n=len(ids)
    clients=[indices[k] for k in ids if locations[k]['is_client']=='true']
    facilities=[indices[k] for k in ids if locations[k]['is_candidate']=='true']
    if type(study['k']) is not int or not 1<=study['k']<=len(facilities):raise ValueError('k must be an integer from 1 through candidate count')
    cells=csv_rows(raw['distances.csv'],('from_id','to_id','distance'))
    if len(cells)!=n*n:raise ValueError('Require exactly one distance for every ordered location pair')
    seen=set();d=[[None]*n for _ in ids]
    for row in cells:
        a,b=row['from_id'],row['to_id']
        if a not in indices or b not in indices:raise ValueError('Distance references an unknown ID')
        if (a,b) in seen:raise ValueError('Duplicate distance pair: '+a+', '+b)
        seen.add((a,b));q=number(row['distance'])
        if q<0:raise ValueError('Distance must be nonnegative')
        d[indices[a]][indices[b]]=str(q)
    plan=csv_rows(raw['plan.csv'],('selected_site_id',));selected=[]
    for row in plan:
        key=row['selected_site_id']
        if key not in indices:raise ValueError('Plan references unknown ID: '+key)
        if indices[key] in selected:raise ValueError('Duplicate selected ID: '+key)
        selected.append(indices[key])
    record=dict(model=dict(study['model']),distances=d,clients=clients,facilities=facilities,k=study['k'])
    return dict(study=study,ids=ids,locations=locations,record=record,selected=sorted(selected),unsupported=unsupported)


def convert_dual(raw,ids,record):
    cert=decoded_json(raw)
    if not isinstance(cert,dict) or set(cert)!={'alpha','beta','lambda'}:raise ValueError('Dual needs alpha, beta and lambda')
    clients=[ids[i] for i in record['clients']];facilities=[ids[i] for i in record['facilities']]
    if not isinstance(cert['alpha'],dict) or set(cert['alpha'])!=set(clients):raise ValueError('Dual alpha IDs must equal client IDs')
    if not isinstance(cert['beta'],dict) or set(cert['beta'])!=set(clients):raise ValueError('Dual beta rows must equal client IDs')
    for row in cert['beta'].values():
        if not isinstance(row,dict) or set(row)!=set(facilities):raise ValueError('Dual beta columns must equal candidate IDs')
    return dict(alpha=[str(number(cert['alpha'][c])) for c in clients],beta=[[str(number(cert['beta'][c][f])) for f in facilities] for c in clients],**{'lambda':str(number(cert['lambda']))})


def named_primal(value,ids):
    return dict(valid=value['valid'],selected_ids=[ids[i] for i in value['selected']],exact_cost=value['exact_cost'],
                assignments=[dict(client_id=ids[c],site_id=ids[f]) for c,f in zip(value['clients'],value['assignments'])])


def solver_proposal(record):
    import ortools
    from ortools.sat.python import cp_model
    data=parse(record);d,clients,facilities,k=data
    scale=reduce(math.lcm,(d[c][f].denominator for c in clients for f in facilities),1)
    costs=[[int(d[c][f]*scale) for f in facilities] for c in clients]
    if scale>1_000_000 or sum(sum(r) for r in costs)>2**60:raise ValueError('Solver adapter scale/range limit; exact input retained without rounding')
    model=cp_model.CpModel();ys=[model.new_bool_var('site_'+str(i)) for i in facilities]
    model.add(sum(ys)>=1);model.add(sum(ys)<=k);terms=[]
    for c,row in zip(clients,costs):
        xs=[model.new_bool_var(f'assign_{c}_{f}') for f in facilities];model.add(sum(xs)==1)
        for x,y,cost in zip(xs,ys,row):model.add(x<=y);terms.append(x*cost)
    model.minimize(sum(terms))
    if model.validate():raise ValueError(model.validate())
    solver=cp_model.CpSolver();solver.parameters.max_time_in_seconds=2.0
    solver.parameters.num_search_workers=1;solver.parameters.random_seed=125
    status=solver.solve(model);selected=[i for i,v in zip(facilities,ys) if solver.value(v)] if status in (cp_model.OPTIMAL,cp_model.FEASIBLE) else None
    return dict(status=solver.status_name(status),selected=selected,checked_primal=primal(data,selected) if selected else None,
                runtime_version=ortools.__version__,seconds_limit=2,workers=1,seed=125,exact_integer_scale=scale,
                reported_objective=float(solver.objective_value) if selected else None,reported_best_bound=float(solver.best_objective_bound),
                reported_wall_seconds=float(solver.wall_time),model_sha256=sha(str(model.proto).encode()),
                scope='Solver status and numerical best bound are reported only. Returned sites are checked independently. No certified gap is derived from the solver bound.')


def review(raw,solver='none',subset_budget=1000,lookup_budget=5_000_000):
    if solver not in ('none','ortools'):raise ValueError('Unsupported solver selection')
    if not isinstance(raw,dict) or any(name not in FILES+('dual.json',) or not isinstance(data,bytes) or len(data)>MAX_FILE_BYTES for name,data in raw.items()):
        raise ValueError('Require known input filenames and byte strings within two MiB each')
    # Validate execution options even when an input is invalid or unsupported.
    from metric_facility_reference import integer,MAX_SUBSETS,MAX_LOOKUPS
    integer(subset_budget,0,MAX_SUBSETS,'subset budget');integer(lookup_budget,0,MAX_LOOKUPS,'lookup budget')
    result=dict(schema=SCHEMA,generated_at_utc=datetime.now(timezone.utc).isoformat(),status='input_invalid',
                input_files=[dict(name=name,bytes=len(data),sha256=sha(data)) for name,data in sorted(raw.items())],
                errors=[],model_support=None,metric_scope=None,submitted_plan=None,solver=None,comparison=None,dual=None,
                execution=dict(solver_requested=solver,subset_budget=subset_budget,lookup_budget=lookup_budget),
                interpretation='Exact frozen input review only. Distance provenance is supplied, not verified. No source approximation algorithm, customer benefit or physical planning validity is established.')
    try:
        normalized=normalize(raw)
    except (ValueError,KeyError,UnicodeDecodeError,csv.Error) as e:
        result['errors'].append(str(e));return result,None
    result['study']=normalized['study'];result['location_ids']=normalized['ids'];result['normalized_input_sha256']=sha(json.dumps(normalized,sort_keys=True,separators=(',',':')).encode())
    ids=normalized['ids'];record=normalized['record'];selected=normalized['selected']
    result['model_support']=dict(supported=not normalized['unsupported'],issues=normalized['unsupported'])
    if normalized['unsupported']:
        result['status']='unsupported_model';return result,normalized
    try:data=parse(record)
    except ValueError as e:
        reason=str(e)
        match=re.search(r'\(([0-9,]+)\)',reason)
        if match:
            mapped=', '.join(ids[int(x)] for x in match.group(1).split(','))
            reason=reason[:match.start()]+'locations '+mapped
        result['status']='metric_ineligible';result['metric_scope']=dict(eligible=False,reason=reason,interpretation='This input is outside the strict-metric reference. It does not establish that a directed or other planning model is invalid.')
        return result,normalized
    result['metric_scope']=dict(eligible=True,scope='Exact symmetry, diagonal/strict positivity and every triangle on the supplied matrix')
    try:submitted=primal(data,selected);result['submitted_plan']=named_primal(submitted,ids)
    except ValueError as e:submitted=None;result['submitted_plan']=dict(valid=False,reason=str(e),selected_ids=[ids[i] for i in selected])
    certificate=None
    if 'dual.json' in raw:
        try:
            certificate=convert_dual(raw['dual.json'],ids,record);checked=dual(data,certificate)
            result['dual']=dict(status='accepted',check=checked)
        except (ValueError,UnicodeDecodeError) as e:
            certificate=None;result['dual']=dict(status='rejected',reason=str(e))
    seed=submitted['selected'] if submitted else None
    if solver=='ortools':
        try:
            proposal=solver_proposal(record);result['solver']=proposal
            if proposal['checked_primal']:
                checked=proposal['checked_primal'];proposal['named_primal']=named_primal(checked,ids)
                if submitted is None or Fraction(checked['exact_cost'])<Fraction(submitted['exact_cost']):seed=checked['selected']
        except ImportError as e:result['solver']=dict(status='unavailable',reason=str(e))
        except ValueError as e:result['solver']=dict(status='refused',reason=str(e))
    else:result['solver']=dict(status='not_requested')
    evidence=calculate(record,selected=seed,certificate=certificate,subset_budget=subset_budget,lookup_budget=lookup_budget)
    result['comparison']=dict(reference=evidence,best_checked_plan=named_primal(evidence['primal'],ids),
                              improvement_over_submitted=str(Fraction(submitted['exact_cost'])-Fraction(evidence['exact_upper_bound'])) if submitted else None)
    result['status']='review_complete' if submitted else 'submitted_plan_infeasible'
    return result,normalized


def report_html(result):
    esc=lambda x:html.escape(str(x),quote=True)
    study=result.get('study',{});comparison=result.get('comparison');submitted=result.get('submitted_plan');evidence=comparison['reference'] if comparison else None
    status=dict(review_complete='Review complete',submitted_plan_infeasible='Submitted plan fails the supported constraints',input_invalid='Input needs correction',unsupported_model='Model needs an additional checker',metric_ineligible='Outside the strict-metric reference')[result['status']]
    def card(label,value):return '<div class="card"><span>'+esc(label)+'</span><strong>'+esc(value)+'</strong></div>'
    cards=card('Submitted cost',submitted.get('exact_cost','Not accepted') if submitted else 'Not assessed')
    cards+=card('Best checked cost',evidence['exact_upper_bound'] if evidence else 'Not assessed')
    cards+=card('Certified lower bound',evidence['exact_lower_bound'] if evidence else 'Not assessed')
    cards+=card('Best plan proven optimal?', 'Yes, for this model' if evidence and evidence['independently_known_optimum'] is not None else 'Unknown')
    body='<h2>Review findings</h2><ul>'
    issues=result['errors']+(result['model_support']['issues'] if result['model_support'] else [])
    if result['metric_scope'] and not result['metric_scope']['eligible']:issues.extend([result['metric_scope']['reason'],result['metric_scope']['interpretation']])
    if submitted and not submitted['valid']:issues.append(submitted['reason'])
    if result['dual'] and result['dual']['status']=='rejected':issues.append('Supplied lower-bound certificate rejected: '+result['dual']['reason'])
    if not issues:issues=['Input model and metric checks passed. Costs and bounds below apply to the frozen, declared model.']
    body+=''.join('<li>'+esc(x)+'</li>' for x in issues)+'</ul>'
    if evidence:
        reason={'all_k_subsets_completed':'Every k-site combination checked','primal_matches_lower_bound':'Checked cost matches a valid lower bound','budget_exhausted':'Search limit reached; exact optimum remains unknown'}[evidence['reason']]
        body+='<p>Quality evidence: <b>'+esc(reason)+'</b>. Absolute certified gap: '+esc(evidence['absolute_gap'])+'. Enumeration inspected '+esc(evidence['examined_subsets'])+' of '+esc(evidence['total_k_subsets'])+' possible k-site sets.</p>'
        if comparison['improvement_over_submitted'] is not None:body+='<p>Best checked plan reduces the submitted cost by <b>'+esc(comparison['improvement_over_submitted'])+' '+esc(study['units'])+'</b>.</p>'
    for label,plan in [('Submitted plan',submitted),('Best checked plan',comparison['best_checked_plan'] if comparison else None)]:
        if not plan or not plan['valid']:continue
        body+='<h2>'+label+'</h2><p>Selected sites: '+', '.join(esc(x) for x in plan['selected_ids'])+'</p><div class="table-wrap"><table><thead><tr><th>Client ID</th><th>Assigned site ID</th></tr></thead><tbody>'
        body+=''.join('<tr><td>'+esc(a['client_id'])+'</td><td>'+esc(a['site_id'])+'</td></tr>' for a in plan['assignments'])+'</tbody></table></div>'
    if result['solver']:
        solver_label={'not_requested':'Not run','unavailable':'Unavailable in this Python environment','refused':'Input exceeds the solver adapter limits','OPTIMAL':'Optimal, as reported by the solver','FEASIBLE':'Feasible plan reported; solver optimality not established','UNKNOWN':'No completed solver conclusion','INFEASIBLE':'Infeasible, as reported by the solver','MODEL_INVALID':'Solver model rejected'}.get(result['solver']['status'],result['solver']['status'])
        body+='<h2>Conventional solver</h2><p>'+esc(solver_label)+'. Solver status and numerical bounds are reports; the certified bounds above come from independent checks.</p>'
    body+='<h2>Input provenance</h2><p>'+esc(study.get('distance_provenance','No accepted study declaration.'))+'</p><p>This provenance description is supplied, not independently verified.</p><ul>'
    body+=''.join('<li><a href="inputs/'+esc(x['name'])+'" download>'+esc(x['name'])+'</a> — '+esc(x['bytes'])+' bytes; SHA-256 <code>'+esc(x['sha256'])+'</code></li>' for x in result['input_files'])+'</ul>'
    body+='<details><summary>Scope, limits and machine-readable evidence</summary><p>'+esc(result['interpretation'])+'</p>'
    if evidence:body+='<p>'+esc(evidence['resource_scope'])+'</p>'
    body+='<p><a href="report.json" download>Download review JSON</a> · <a href="manifest.json" download>Download file manifest</a></p></details>'
    css='''body{font:16px/1.55 system-ui,sans-serif;color:#173b39;background:#f2f5f0;margin:0}main{max-width:1080px;margin:40px auto;padding:32px;background:white;border:1px solid #d6e1db;border-radius:12px}h1{font-size:34px;line-height:1.15;margin:12px 0 20px}h2{font-size:22px;margin-top:32px}.eyebrow{letter-spacing:.13em;text-transform:uppercase;font-size:12px}.status{padding:12px 18px;background:#e8f3ed;border-left:4px solid #357b68}.cards{display:grid;grid-template-columns:repeat(4,1fr);gap:12px;margin:24px 0}.card{border:1px solid #ccdcd2;border-radius:8px;padding:16px}.card span{font-size:13px;display:block}.card strong{display:block;font-size:23px;margin-top:8px;overflow-wrap:anywhere}a{color:#286653}code{overflow-wrap:anywhere;font-size:12px}table{border-collapse:collapse;width:100%}th,td{padding:10px 14px;border:1px solid #d9e2dd;text-align:left}th{background:#e8f3ed}.table-wrap{overflow-x:auto}details{margin-top:32px;padding:16px;border:1px solid #d9e2dd}summary{cursor:pointer;font-weight:600}@media(max-width:700px){main{margin:12px;padding:20px}.cards{grid-template-columns:repeat(2,1fr)}h1{font-size:28px}}@media print{body{background:white}main{border:0;margin:0}details{display:block}a{color:inherit}}'''
    displayed_time=datetime.fromisoformat(result['generated_at_utc']).strftime('%d %B %Y, %H:%M UTC')
    return '<!doctype html><html lang="en"><meta charset="utf-8"><meta name="viewport" content="width=device-width,initial-scale=1"><title>'+esc(study.get('title','Facility review'))+'</title><style>'+css+'</style><main><div class="eyebrow">Optimization Assurance · Local review</div><h1>'+esc(study.get('title','Facility review'))+'</h1><div class="status">'+esc(status)+'</div><p>Units: '+esc(study.get('units','Unspecified'))+' · Generated '+esc(displayed_time)+'</p><div class="cards">'+cards+'</div>'+body+'<footer><p>Review sign-off: not supplied. This report records computational checks and unresolved items.</p></footer></main></html>\n'


def export_review(raw,output,solver='none',subset_budget=1000,lookup_budget=5_000_000):
    if output.exists():raise FileExistsError('Choose a fresh output directory; existing evidence is preserved')
    result,normalized=review(raw,solver,subset_budget,lookup_budget)
    output.mkdir(parents=True,exist_ok=False);(output/'inputs').mkdir()
    for name,data in raw.items():
        if name not in FILES+('dual.json',):raise ValueError('Unexpected input name')
        with (output/'inputs'/name).open('xb') as f:f.write(data)
    def write(name,value):
        with (output/name).open('x') as f:json.dump(value,f,indent=2,allow_nan=False);f.write('\n')
    write('report.json',result)
    if normalized:write('normalized.json',normalized)
    with (output/'report.html').open('x') as f:f.write(report_html(result))
    files=[dict(path=str(p.relative_to(output)),bytes=p.stat().st_size,sha256=sha(p.read_bytes())) for p in sorted(output.rglob('*')) if p.is_file()]
    tool_paths=[Path(__file__),Path(__file__).with_name('metric_facility_reference.py')]
    write('manifest.json',dict(schema=SCHEMA,files=files,tools=[dict(name=p.name,sha256=sha(p.read_bytes())) for p in tool_paths],python=sys.version,execution=result['execution'],scope='Checksums bind local bytes; no independent authenticity, source theorem or customer validation.'))
    return result


def main():
    parser=argparse.ArgumentParser(description=__doc__);parser.add_argument('input_directory',type=Path);parser.add_argument('--output',required=True,type=Path)
    parser.add_argument('--solver',choices=('none','ortools'),default='none');parser.add_argument('--subset-budget',type=int,default=1000);parser.add_argument('--lookup-budget',type=int,default=5_000_000);args=parser.parse_args()
    result=export_review(load_inputs(args.input_directory),args.output,args.solver,args.subset_budget,args.lookup_budget)
    print(json.dumps(dict(status=result['status'],report=str((args.output/'report.html').resolve()),submitted=result['submitted_plan'],comparison=result['comparison'])))

if __name__=='__main__':main()
