"""Saved-only metrics and strictly selected local exports for Lab22.

No Torch/Transformers imports, tokenizer/model loads, installs or network calls.
"""
import csv,io,re
from microscope_io import (SCHEMA,Store,relative,separate,bounded,read_json,digest,encode,inventory,size,
    hashes,check_hashes,finalized,integrity_files,npz,events,PAYLOAD_CAP)
from microscope_imports import CAPTURE_KEYS,ARMS19,HELDOUT,lab17,lab19


def csv_bytes(rows):
    stream=io.StringIO();writer=csv.DictWriter(stream,fieldnames=list(rows[0]));writer.writeheader();writer.writerows(rows)
    return stream.getvalue().encode()

def log_softmax(value):
    import numpy as np
    value=value.astype(np.float64);shift=value-float(np.max(value))
    return shift-float(np.log(np.exp(shift).sum()))

def scalar(logits,baseline,original=None,replacement=None):
    import numpy as np
    z=logits.astype(np.float64);b=baseline.astype(np.float64)
    lp=log_softmax(z);bp=log_softmax(b)
    result={'eligible':True,'y':float(z[1175]-z[3076]),'delta_y':float((z[1175]-z[3076])-(b[1175]-b[3076])),
        'kl_baseline_to_arm_full_vocab':float(np.sum(np.exp(bp)*(bp-lp))),
        'activation_norm':None,'perturbation_norm':None,'relative_perturbation_norm':None,'norm_status':'missing: no boundary capture in this arm'}
    if original is not None:
        norm=float(np.linalg.norm(original.astype(np.float64)));change=(replacement if replacement is not None else original).astype(np.float64)-original.astype(np.float64)
        perturb=float(np.linalg.norm(change))
        result.update(activation_norm=norm,perturbation_norm=perturb,relative_perturbation_norm=perturb/norm if norm else None,
                      norm_status='defined' if norm else 'undefined_zero_norm')
    return result

def geometry(vector,previous=None):
    import numpy as np
    vector=vector.astype(np.float64);norm=float(np.linalg.norm(vector))
    result={'norm':norm,'difference_norm':None,'cosine':None,'cosine_status':'not_applicable: no comparison boundary'}
    if previous is not None:
        previous=previous.astype(np.float64);pn=float(np.linalg.norm(previous))
        result.update(difference_norm=float(np.linalg.norm(vector-previous)),cosine=float(vector.dot(previous)/(norm*pn)) if norm and pn else None,
                      cosine_status='defined' if norm and pn else 'undefined_zero_norm')
    return result

def fresh_arrays(root,current,fixture=False):
    import numpy as np
    from microscope import SCHEDULE,measurement_address
    if current.get('provenance')!=('fixture' if fixture else 'measured'):raise ValueError('Raw provenance does not match analysis mode')
    if fixture and current['demonstrate']=='completed':raise ValueError('Fixture cannot satisfy measured core completion')
    record=events(relative(root,'attempts.jsonl'));n=current['attempted']
    if type(n) is not int or not 0<=n<=14:raise ValueError('Invalid fresh attempt count')
    starts=[e for e in record if e.get('outcome')=='started'];finishes=[e for e in record if e.get('outcome')!='started']
    complete=current['demonstrate']=='completed'
    if len({e.get('attempt') for e in starts})!=len(starts) or len({e.get('attempt') for e in finishes})!=len(finishes):raise ValueError('Fresh duplicate ledger events')
    if len(starts)>n or (complete and len(starts)!=n):raise ValueError('Fresh start/count mismatch')
    by_number={e['attempt']:e for e in finishes};started={e['attempt']:e for e in starts}
    for number,e in enumerate(starts,1):
        if e.get('attempt')!=number or any(e.get(k)!=v for k,v in SCHEDULE[number-1].items()):raise ValueError('Fresh fixed chronological schedule')
        finish=by_number.get(number)
        if finish and (any(finish.get(k)!=e[k] for k in ('attempt','prompt','arm')) or finish.get('timestamp')<e.get('timestamp')):raise ValueError('Fresh finish provenance')
    for number,e in by_number.items():
        if type(number) is not int or not 1<=number<=n:raise ValueError('Fresh orphan finish')
        if number not in started and (complete or number!=n or e.get('outcome')!='partial' or e.get('start_event_missing') is not True):
            raise ValueError('Unexplained missing start event')
    # For a partial run, an atomic array snapshot may precede its metadata or
    # completion event. Only the fixed schedule authorizes possible keys/shapes;
    # unfinished measurements remain quarantined and ineligible, never zeros.
    allowed_logits={f'attempt_{number:03d}_logits':((50304,),'float32') for number in range(1,n+1) if number!=13}
    allowed_captures={'positive_addition':((128,),'float32'),'negative_addition':((128,),'float32')}
    for number,item in enumerate(SCHEDULE[:n],1):
        keys=CAPTURE_KEYS if item['arm']=='observe' else ('original','replacement') if item['arm'] in ('zero','positive','negative') else ()
        for key in keys:allowed_captures[f'attempt_{number:03d}_{key}']=((128,),'float32')
    metadata_path=relative(root,'array-metadata.json')
    metadata=read_json(metadata_path) if metadata_path.exists() else {}
    logits,count=npz(relative(root,'logits.npz'),allowed_logits,optional=True) if relative(root,'logits.npz').exists() else ({},0)
    captures,ccount=npz(relative(root,'captures.npz'),allowed_captures,optional=True) if relative(root,'captures.npz').exists() else ({},0)
    if not set(metadata).issubset(set(allowed_logits)|set(allowed_captures)):raise ValueError('Fresh unexpected metadata key')
    rows={r['id']:r for r in read_json(relative(root,'inputs.json'))['rows']}
    for key,record in metadata.items():
        addition=key in ('positive_addition','negative_addition')
        if addition:
            arm=key.split('_')[0];row=rows['H1+'];number=0;k=key;parents=['imports/lab19/discovery-vectors.npz','imports/lab19/direction.json']
        else:
            number=int(key.split('_')[1]);k='_'.join(key.split('_')[2:]);item=SCHEDULE[number-1];arm=item['arm'];row=rows[item['prompt']];parents=None
        expected=measurement_address(current['run_id'],row,arm,number,k,[128] if addition else [1,1,50304] if k=='logits' else [1,row['length'],128],
                                     'derived' if addition else 'fixture' if fixture else 'measured',parents)
        if any(record.get(k)!=v for k,v in expected.items()):raise ValueError('Fresh ambiguous/fixture measurement address: '+key)
    eligible=[]
    for number,e in by_number.items():
        if e.get('outcome')!='completed_logits' or e.get('cleanup')!='verified' or number not in started:continue
        key=f'attempt_{number:03d}_logits'
        required=[key]+[k for k in allowed_captures if k.startswith(f'attempt_{number:03d}_')]
        if all(k in metadata and k in (logits if k.endswith('_logits') else captures) for k in required):eligible.append(number)
        elif complete:raise ValueError('Completed run missing committed array/metadata')
    if complete:
        expected_logits={f'attempt_{number:03d}_logits' for number in range(1,15) if number!=13}
        if set(logits)!=expected_logits or set(captures)!=set(allowed_captures) or set(metadata)!=set(logits)|set(captures):raise ValueError('Completed raw array inventory')
    # Numeric bytes include unfinished snapshots even though they do not enter
    # denominators. Preserve those raw files; analysis does not repair them.
    logits={k:v for k,v in logits.items() if int(k.split('_')[1]) in eligible}
    captures={k:v for k,v in captures.items() if k in ('positive_addition','negative_addition') or int(k.split('_')[1]) in eligible}
    completed=[e for e in finishes if e.get('outcome')=='completed_logits']
    if current['demonstrate']=='completed':
        if n!=14 or current['finished']!=14 or len(completed)!=13 or len(finishes)!=14:raise ValueError('Core completion requires exact14/13 logits')
        for number,e in by_number.items():
            if e['outcome']!=('expected_exception' if number==13 else 'completed_logits') or e.get('cleanup')!='verified':raise ValueError('Core schedule cleanup/outcome')
        if by_number[13].get('exception')!='CleanupCheck: deliberate_cleanup_check':raise ValueError('Named expected exception required')
        checks=read_json(relative(root,'checks.json'))
        expected_checks=[p+' '+arm for p in ('H1+','H1-') for arm in ('observe','zero','restore')]
        expected_checks += [p+' replay '+arm for p in ('H1+','H1-') for arm in ('baseline','positive','negative')]
        expected_checks += [p+f' parallel_addition_{i}' for p in ('H1+','H1-') for i in range(6)]
        expected_checks += [p+' '+arm for p in ('H1+','H1-') for arm in ('final_normalization','normalized_head','raw_head')]
        expected_checks += ['expected_exception_cleanup','H1+ recovery','demonstrate terminal']
        if len(checks)!=len(expected_checks) or {c.get('name') for c in checks}!=set(expected_checks) or not all(c.get('passed') is True and c.get('eligible') is True for c in checks):
            raise ValueError('Core numerical checks incomplete')
        if read_json(relative(root,'diagnostic-module-calls.json'))!={'final_layer_norm':4,'output_head':4}:raise ValueError('Exact diagnostic module-call budget')
    return logits,captures,finishes,count+ccount

def analyze(args,store):
    from microscope import state,frozen_map,SCHEDULE
    raw=safe_run=args.run;final_hash=finalized(raw);current=state(raw)
    if current.get('active_command') is not None or current['demonstrate'] not in ('completed','failed','partial','not_run'):
        raise ValueError('Unfinalized/crashed run is not analyzable')
    if frozen_map(raw)!=read_json(relative(raw,'frozen.sha256.json')):raise ValueError('Frozen original import/source changed')
    a=lab17(relative(raw,'imports/lab17'));b=lab19(relative(raw,'imports/lab19'))
    imported=read_json(relative(raw,'imports.json'))
    for name,result in [('lab17',a),('lab19',b)]:
        if imported[name].get('provenance')!='measured' or result['file_hashes']!=imported[name]['file_hashes']:raise ValueError('Imported provenance/hash mismatch')
    logits,captures,attempts,amount=fresh_arrays(raw,current)
    if amount+b['payload_bytes']+sum(v.nbytes for v in a['vectors'].values())>PAYLOAD_CAP:raise ValueError('Raw numeric payload ceiling')
    geom=[]
    for row in a['prompts']:
        p=row['id']
        for key in [f'r{i}' for i in range(7)]+['h_f']:
            previous=a['vectors'][p+f'_r{int(key[1:])-1}'] if key.startswith('r') and key!='r0' else None
            geom.append({'prompt':p,'boundary':key,'normalization':'final normalized (separate)' if key=='h_f' else 'raw',
                'origin_run_id':a['origin_run_id'],'provenance':'derived','parents':['imports/lab17/captures.json:'+p+'/'+key],
                **geometry(a['vectors'][p+'_'+key],previous)})
    store.json('lab17-geometry.json',geom);store.put('lab17-geometry.csv',csv_bytes(geom))
    effect=[]
    for offset,p in enumerate(HELDOUT):
        baseline=b['logits'][f'attempt_{23+offset*9:03d}']
        for arm_index,arm in enumerate(ARMS19):
            n=23+offset*9+arm_index;key=f'attempt_{n:03d}'
            original=b['vectors'].get(key+'_original');replacement=b['vectors'].get(key+'_replacement')
            if arm=='baseline':original=b['vectors'][f'attempt_{n+1:03d}_original']
            effect.append({'prompt':p,'arm':arm,'attempt':n,'comparison_group':'primary_four_reviews' if offset<4 else 'offtask_controls',
                'origin_run_id':b['origin_run_id'],'provenance':'derived','parents':['imports/lab19/logits.npz:'+key],
                **scalar(b['logits'][key],baseline,original,replacement)})
    means=[]
    for arm in ARMS19:
        selected=[x['delta_y'] for x in effect if x['arm']==arm and x['comparison_group']=='primary_four_reviews']
        means.append({'arm':arm,'review_count':4,'mean_delta_y':sum(selected)/4,'eligible':True,'provenance':'derived',
                     'endpoint':'original four-review mean; all arms retained'})
    store.json('lab19-effects.json',{'rows':effect,'primary_mean':means,'prior_exposure':b['prior_exposure']});store.put('lab19-effects.csv',csv_bytes(effect))
    fresh=[]
    for n,item in enumerate(SCHEDULE,1):
        key=f'attempt_{n:03d}_logits';baseline=logits.get('attempt_001_logits' if item['prompt']=='H1+' else 'attempt_007_logits')
        values=scalar(logits[key],baseline,captures.get(f'attempt_{n:03d}_original'),captures.get(f'attempt_{n:03d}_replacement')) if key in logits and baseline is not None else {
            'eligible':False,'y':None,'delta_y':None,'kl_baseline_to_arm_full_vocab':None,'activation_norm':None,'perturbation_norm':None,
            'relative_perturbation_norm':None,'norm_status':'missing: expected_exception' if n==13 else 'missing: no successful saved logits/baseline'}
        fresh.append({'attempt':n,**item,'provenance':'derived' if values['eligible'] else 'missing','parents':[key] if values['eligible'] else [],**values})
    checks=read_json(relative(raw,'checks.json'))
    descriptor=read_json(relative(raw,'files.sha256.json'));receipt=descriptor.get('processing_receipt',{})
    durations=dict(current['command_durations'])
    if receipt.get('command') and isinstance(receipt.get('actual_parent_observation'),(int,float)):
        durations[receipt['command']]=receipt['actual_parent_observation']
    summary={'demonstration_state':current['demonstrate'],'attempted':current['attempted'],'finished':current['finished'],
        'successful_logits':len(logits),'eligible_signed_comparisons':sum(x['eligible'] for x in fresh if x['arm'] in ('positive','negative')),
        'attempts':attempts,'checks':checks,'processing_seconds':current['processing_seconds'],'command_durations':durations,'processing_receipt':receipt,'processing_total_kind':'conservative upper bound including terminal close',
        'raw_storage_bytes':size(raw),'numeric_payload_bytes':amount+b['payload_bytes']+sum(v.nbytes for v in a['vectors'].values()),
        'diagnostic_module_calls':current.get('diagnostic_module_calls',{}),'reason':current.get('reason'),
        'scientific_outcome':current['scientific_outcome'],'routing':'not_applicable: dense specimen'}
    kind=current.get('verification_kind','canonical_same_environment')
    summary['verification_kind']=kind
    if kind=='supplemental_portability':
        LABEL=read_json(relative(raw,'portability.json'))['label']
        summary.update(canonical_same_environment=False,label=LABEL,portability=read_json(relative(raw,'portability.json')))
    store.json('integration-summary.json',summary);store.json('fresh-effects.json',fresh);store.put('fresh-effects.csv',csv_bytes(fresh))
    complete=current['demonstrate']=='completed'
    conclusion=('The fixed replay passed the required instrumentation checks. Any signed effects describe only this frozen direction, dose, boundary and two previously inspected prompts.' if complete else
                'The fresh integration is not complete. Missing measurements remain unavailable, and this record does not satisfy the core measured integration criterion.')
    report=f'''This report separates imported scientific evidence from new integration evidence. The imported Lab17 record contains the eight core prompts and all twenty final-position captures per prompt. The geometry table reports all seven raw residual boundaries, with final normalization presented separately. Historical observer, restoration, addition and reconstruction checks remain identified as original checks. The original full logit arrays were not saved in that Lab, so importing those checks does not recreate their numerical execution. The optional probe is excluded from the integration summaries.

The imported corrected Lab19 record retains its complete eighty-two-attempt schedule, including the intentional cleanup exception, discovery vectors, frozen direction and dose, tokenization, predictions and all recorded controls. The primary endpoint remains the four-review mean for every original arm. The two off-task prompts remain separate. Negative, zero and larger-dose results are retained alongside favorable results. Derived tables are calculated from the preserved float32 measurements in float64. Original hashes are distinguished from hashes first recorded at import; neither kind authenticates the honesty of the original experiment.

The corrected Lab19 predictions were explicitly post-exposure validation expectations. Its review prompts and results had already been inspected. The present plan discloses that access, and its predictions concern integration behavior and numerical replay. Reusing those prompts is not a fresh held-out test or another independent observation of generalization. The imported direction is recomputed for verification from the six saved discovery vectors without fitting a replacement direction, reversing its orientation, changing its dose, or selecting a favorable layer.

The fresh demonstration state is {current['demonstrate']}. Its saved ledger records {current['attempted']} attempted top-level forwards and {len(logits)} successful logit arrays. The fixed plan permits fourteen attempts, with one named expected exception and thirteen successful outputs. The integration summary lists every available check and the reasons for unavailable comparisons. Parent-observed per-command durations and storage totals are retained. The cumulative allowance conservatively charges the bounded terminal close, including final metadata and integrity hashing; the final receipt states its measurement boundary. Diagnostic final-normalization and output-head calls are counted separately from top-level Transformer forwards. A forced termination cannot establish hook cleanup and is recorded as cleanup unverified.

Fresh signed effects use the leading-space good-minus-bad logit difference, within-prompt changes, actual boundary perturbation norms and full-vocabulary stable-log-softmax divergence. Undefined values are null with reasons rather than zero. All six ordinary arms per prompt and both closing cleanup attempts appear in the summary. Repeating this offline analysis regenerates those comparisons from files without loading a model, changing the raw run, or claiming another experiment.

{conclusion} This small dense checkpoint, fixed computational address and hand-authored synthetic input set do not establish a universal sentiment mechanism, subjective experience, a population uncertainty interval or unseen task performance. Source and runtime compatibility checks delimit the replay, while missing historical NumPy metadata and changes to host metadata formatting are explicitly disclosed. A new wording family or scientific hypothesis requires its own prospectively bounded plan. This implementation-generated mechanical report requires learner review and explanation before submission or sharing.
'''
    if kind=='supplemental_portability':
        report=LABEL+' Historical NumPy is unrecorded; current NumPy is 2.5.3.\n\n'+report.replace('The fixed replay passed the required instrumentation checks.', 'The supplemental two-prompt replay passed its required instrumentation checks; canonical same-environment acceptance remains unavailable.')
    if not 400<=len(report.split())<=600:raise ValueError('Mechanical report word bound')
    store.put('report.md',report.encode())
    manifest=read_json(relative(store.root,'manifest.json'));manifest.update(raw_run_id=current['run_id'],raw_final_manifest_sha256=final_hash,
        source_sha256=digest(bounded(__file__)),source_kind='saved-only NumPy analysis',provenance='derived',
        verification_kind=kind,canonical_same_environment=kind=='canonical_same_environment',
        raw_provenance='measured historical imports; fresh state '+current['demonstrate'],model_loaded=False,forwards=0)
    store.json('manifest.json',manifest,replace=True)
    if finalized(raw)!=final_hash:raise ValueError('Analysis changed raw evidence')


def export(args,store):
    from microscope import state,SOURCE_NAMES
    raw=args.run;analysis=args.analysis;final_hash=finalized(raw);current=state(raw)
    analysis_hash=finalized(analysis);manifest=read_json(relative(analysis,'manifest.json'))
    if manifest.get('state')!='completed' or manifest.get('raw_run_id')!=current['run_id'] or manifest.get('raw_final_manifest_sha256')!=final_hash:
        raise ValueError('Selected analysis is not bound to this finalized raw run')
    verification_kind=current.get('verification_kind','canonical_same_environment')
    if manifest.get('verification_kind','canonical_same_environment')!=verification_kind:raise ValueError('Analysis verification kind differs')
    entries=read_json(args.allowlist,100_000)
    if not isinstance(entries,list) or not entries:raise ValueError('Explicit reviewed export entries required')
    permitted_raw={'environment.json','artifact-manifest.json','source-manifest.json','inputs.json','predictions.json','plan.json',
                   'checks.json','status.json','array-metadata.json','captures.npz','logits.npz','metrics.csv','diagnostic-module-calls.json'}
    permitted_raw|={'source/'+n for n in SOURCE_NAMES}
    permitted_analysis={'manifest.json','lab17-geometry.json','lab17-geometry.csv','lab19-effects.json','lab19-effects.csv',
                        'integration-summary.json','fresh-effects.json','fresh-effects.csv','report.md'}
    selected=[];seen=set()
    for entry in entries:
        if not isinstance(entry,dict) or set(entry)!={'kind','path'} or entry['kind'] not in ('raw','analysis'):raise ValueError('Unknown export configuration')
        kind,name=entry['kind'],entry['path'];root=raw if kind=='raw' else analysis
        source=relative(root,name)
        if name not in (permitted_raw if kind=='raw' else permitted_analysis):raise ValueError('Unapproved export source')
        destination=kind+'/'+name
        if destination in seen:raise ValueError('Duplicate export entry')
        seen.add(destination);data=bounded(source)
        if source.suffix!='.npz':
            text=data.decode('utf-8')
            if re.search(r'/(?:Users|home)/|(?i:AKIA[0-9A-Z]{16}|-----BEGIN .*PRIVATE KEY-----|sk-[A-Za-z0-9_-]{20,}|gh[pousr]_[A-Za-z0-9]{20,})',text):
                raise ValueError('Export contains private path/credential pattern')
        selected.append((kind,name,destination,data))
    if ('analysis','manifest.json','analysis/manifest.json') not in [(k,n,d) for k,n,d,_ in selected]:raise ValueError('Selected analysis manifest must be exported')
    if sum(len(d) for _,_,_,d in selected)+100_000>store.cap:raise ValueError('Export storage preflight')
    copied=[]
    raw_hashes=integrity_files(raw);analysis_hashes=integrity_files(analysis)
    for kind,name,destination,data in selected:
        if digest(data)!=(raw_hashes if kind=='raw' else analysis_hashes)[name]:raise ValueError('Export source hash changed')
        store.put(destination,data);copied.append({'source_kind':kind,'relative_source_path':name,'destination_path':destination,'sha256':digest(data)})
    if finalized(raw)!=final_hash or finalized(analysis)!=analysis_hash:raise ValueError('Export source mutated')
    m=read_json(relative(store.root,'manifest.json'));m.update(raw_run_id=current['run_id'],raw_final_manifest_sha256=final_hash,
        selected_analysis_run_id=manifest['run_id'],selected_analysis_final_manifest_sha256=analysis_hash,
        verification_kind=verification_kind,canonical_same_environment=verification_kind=='canonical_same_environment',
        label=read_json(relative(raw,'portability.json'))['label'] if verification_kind=='supplemental_portability' else 'Canonical same-environment replay',
        source_sha256=digest(bounded(__file__)),copied=copied,local_only=True,uploaded=False)
    store.json('manifest.json',m,replace=True)
