# /// script
# requires-python = ">=3.11"
# dependencies = ["marimo==0.24.0", "numpy>=2.1", "torch>=2.10; sys_platform != 'emscripten'", "transformers==4.57.6; sys_platform != 'emscripten'"]
# ///
import marimo
__generated_with = "0.24.0"
app=marimo.App(width="medium",app_title="Inside a pretrained language model")

@app.cell
def _():
    import marimo as mo
    import json,sys,html,math
    import numpy as np
    from pathlib import Path
    return mo,json,sys,html,math,np,Path

@app.cell
def _():
    recorded={'model': 'EleutherAI/pythia-70m-deduped', 'revision': 'e93a9faa9c77e5d09219f6c868bfc7a1bd65593c', 'torch': '2.14.0', 'transformers': '4.57.6', 'device': 'CPU', 'inference_seconds': 3.6340336670109536, 'collected_at_utc': '2026-09-07T01:31:08.751532+00:00', 'precision': 'float32', 'method': 'Clean post-block residual vectors patched into corrupted prompts at changed-token and final-token positions. No weight changes. Controls are interventions, not guaranteed null effects. Candidate probabilities use the full vocabulary denominator.', 'limits': 'Six selected capital-city prompt pairs, with no-op, zero and one seeded norm-matched random intervention control in a small pretrained research model. Not a representative accuracy evaluation, a Goodfire replication or a frontier-model result.', 'records': [{'clean': 'The capital of France is', 'corrupt': 'The capital of England is', 'clean_answer': ' Paris', 'corrupt_answer': ' London', 'clean_tokens': ['The', 'Ġcapital', 'Ġof', 'ĠFrance', 'Ġis'], 'corrupt_tokens': ['The', 'Ġcapital', 'Ġof', 'ĠEngland', 'Ġis'], 'positions': [3, 4], 'rows': [{'condition': 'Clean', 'layer': -1, 'position': -1, 'logit_difference': 1.55712890625, 'p_clean_answer': 0.0019448678940534592, 'p_corrupt_answer': 0.0004098619392607361}, {'condition': 'Corrupted', 'layer': -1, 'position': -1, 'logit_difference': -2.048828125, 'p_clean_answer': 0.000294992933049798, 'p_corrupt_answer': 0.0022887922823429108}, {'condition': 'Patched', 'layer': 0, 'position': 3, 'logit_difference': 1.423828125, 'p_clean_answer': 0.002181690651923418, 'p_corrupt_answer': 0.0005253303097561002}, {'condition': 'No-op control', 'layer': 0, 'position': 3, 'logit_difference': -2.048828125, 'p_clean_answer': 0.000294992933049798, 'p_corrupt_answer': 0.0022887922823429108}, {'condition': 'Zero control', 'layer': 0, 'position': 3, 'logit_difference': -0.5413818359375, 'p_clean_answer': 0.0003822331200353801, 'p_corrupt_answer': 0.0006568216485902667}, {'condition': 'Random norm-matched control', 'layer': 0, 'position': 3, 'logit_difference': -0.289794921875, 'p_clean_answer': 0.00020540517289191484, 'p_corrupt_answer': 0.0002744528464972973}, {'condition': 'Patched', 'layer': 0, 'position': 4, 'logit_difference': -2.025634765625, 'p_clean_answer': 0.0002625109045766294, 'p_corrupt_answer': 0.001990074524655938}, {'condition': 'No-op control', 'layer': 0, 'position': 4, 'logit_difference': -2.048828125, 'p_clean_answer': 0.000294992933049798, 'p_corrupt_answer': 0.0022887922823429108}, {'condition': 'Zero control', 'layer': 0, 'position': 4, 'logit_difference': -0.6724853515625, 'p_clean_answer': 0.0008020123350434005, 'p_corrupt_answer': 0.0015712225576862693}, {'condition': 'Random norm-matched control', 'layer': 0, 'position': 4, 'logit_difference': -1.01953125, 'p_clean_answer': 0.000542323337867856, 'p_corrupt_answer': 0.0015032633673399687}, {'condition': 'Patched', 'layer': 1, 'position': 3, 'logit_difference': 1.3741455078125, 'p_clean_answer': 0.002201675670221448, 'p_corrupt_answer': 0.0005571466172114015}, {'condition': 'No-op control', 'layer': 1, 'position': 3, 'logit_difference': -2.048828125, 'p_clean_answer': 0.000294992933049798, 'p_corrupt_answer': 0.0022887922823429108}, {'condition': 'Zero control', 'layer': 1, 'position': 3, 'logit_difference': -0.8779296875, 'p_clean_answer': 0.00014560851559508592, 'p_corrupt_answer': 0.00035032149753533304}, {'condition': 'Random norm-matched control', 'layer': 1, 'position': 3, 'logit_difference': -0.7838134765625, 'p_clean_answer': 0.00025201603421010077, 'p_corrupt_answer': 0.000551866483874619}, {'condition': 'Patched', 'layer': 1, 'position': 4, 'logit_difference': -1.9620361328125, 'p_clean_answer': 0.0002597339334897697, 'p_corrupt_answer': 0.001847694511525333}, {'condition': 'No-op control', 'layer': 1, 'position': 4, 'logit_difference': -2.048828125, 'p_clean_answer': 0.000294992933049798, 'p_corrupt_answer': 0.0022887922823429108}, {'condition': 'Zero control', 'layer': 1, 'position': 4, 'logit_difference': -1.0125732421875, 'p_clean_answer': 0.00027147491346113384, 'p_corrupt_answer': 0.0007472822908312082}, {'condition': 'Random norm-matched control', 'layer': 1, 'position': 4, 'logit_difference': 1.680419921875, 'p_clean_answer': 0.0005524078151211143, 'p_corrupt_answer': 0.00010291121725458652}, {'condition': 'Patched', 'layer': 2, 'position': 3, 'logit_difference': 1.2349853515625, 'p_clean_answer': 0.0023293474223464727, 'p_corrupt_answer': 0.0006774651119485497}, {'condition': 'No-op control', 'layer': 2, 'position': 3, 'logit_difference': -2.048828125, 'p_clean_answer': 0.000294992933049798, 'p_corrupt_answer': 0.0022887922823429108}, {'condition': 'Zero control', 'layer': 2, 'position': 3, 'logit_difference': -0.679443359375, 'p_clean_answer': 0.00027209415566176176, 'p_corrupt_answer': 0.0005367817357182503}, {'condition': 'Random norm-matched control', 'layer': 2, 'position': 3, 'logit_difference': -0.760986328125, 'p_clean_answer': 0.00014679215382784605, 'p_corrupt_answer': 0.00031419191509485245}, {'condition': 'Patched', 'layer': 2, 'position': 4, 'logit_difference': -1.751953125, 'p_clean_answer': 0.000279109226539731, 'p_corrupt_answer': 0.0016093028243631124}, {'condition': 'No-op control', 'layer': 2, 'position': 4, 'logit_difference': -2.048828125, 'p_clean_answer': 0.000294992933049798, 'p_corrupt_answer': 0.0022887922823429108}, {'condition': 'Zero control', 'layer': 2, 'position': 4, 'logit_difference': -0.4215087890625, 'p_clean_answer': 0.001912839012220502, 'p_corrupt_answer': 0.002915663179010153}, {'condition': 'Random norm-matched control', 'layer': 2, 'position': 4, 'logit_difference': -0.4456787109375, 'p_clean_answer': 2.0159914129180834e-06, 'p_corrupt_answer': 3.1480708457820583e-06}, {'condition': 'Patched', 'layer': 3, 'position': 3, 'logit_difference': 1.1746826171875, 'p_clean_answer': 0.0022832814138382673, 'p_corrupt_answer': 0.0007053444278426468}, {'condition': 'No-op control', 'layer': 3, 'position': 3, 'logit_difference': -2.048828125, 'p_clean_answer': 0.000294992933049798, 'p_corrupt_answer': 0.0022887922823429108}, {'condition': 'Zero control', 'layer': 3, 'position': 3, 'logit_difference': -1.0277099609375, 'p_clean_answer': 0.0003015504917129874, 'p_corrupt_answer': 0.0008427306893281639}, {'condition': 'Random norm-matched control', 'layer': 3, 'position': 3, 'logit_difference': -1.00439453125, 'p_clean_answer': 0.0002978213015012443, 'p_corrupt_answer': 0.0008131277281790972}, {'condition': 'Patched', 'layer': 3, 'position': 4, 'logit_difference': -1.6658935546875, 'p_clean_answer': 0.00030106809572316706, 'p_corrupt_answer': 0.0015927701024338603}, {'condition': 'No-op control', 'layer': 3, 'position': 4, 'logit_difference': -2.048828125, 'p_clean_answer': 0.000294992933049798, 'p_corrupt_answer': 0.0022887922823429108}, {'condition': 'Zero control', 'layer': 3, 'position': 4, 'logit_difference': -0.0343017578125, 'p_clean_answer': 8.643014734843746e-05, 'p_corrupt_answer': 8.944627916207537e-05}, {'condition': 'Random norm-matched control', 'layer': 3, 'position': 4, 'logit_difference': 0.8345947265625, 'p_clean_answer': 0.000573255936615169, 'p_corrupt_answer': 0.00024882194702513516}, {'condition': 'Patched', 'layer': 4, 'position': 3, 'logit_difference': -1.8699951171875, 'p_clean_answer': 0.0003143977955915034, 'p_corrupt_answer': 0.0020398960914462805}, {'condition': 'No-op control', 'layer': 4, 'position': 3, 'logit_difference': -2.048828125, 'p_clean_answer': 0.000294992933049798, 'p_corrupt_answer': 0.0022887922823429108}, {'condition': 'Zero control', 'layer': 4, 'position': 3, 'logit_difference': -1.9287109375, 'p_clean_answer': 0.0003367163008078933, 'p_corrupt_answer': 0.002316822065040469}, {'condition': 'Random norm-matched control', 'layer': 4, 'position': 3, 'logit_difference': -1.923583984375, 'p_clean_answer': 0.00032731282408349216, 'p_corrupt_answer': 0.0022406030911952257}, {'condition': 'Patched', 'layer': 4, 'position': 4, 'logit_difference': 1.3609619140625, 'p_clean_answer': 0.001813808805309236, 'p_corrupt_answer': 0.00046508596278727055}, {'condition': 'No-op control', 'layer': 4, 'position': 4, 'logit_difference': -2.048828125, 'p_clean_answer': 0.000294992933049798, 'p_corrupt_answer': 0.0022887922823429108}, {'condition': 'Zero control', 'layer': 4, 'position': 4, 'logit_difference': -1.01708984375, 'p_clean_answer': 4.087056004209444e-05, 'p_corrupt_answer': 0.00011301265476504341}, {'condition': 'Random norm-matched control', 'layer': 4, 'position': 4, 'logit_difference': -0.6378173828125, 'p_clean_answer': 2.122823389072437e-05, 'p_corrupt_answer': 4.0171162254409865e-05}, {'condition': 'Patched', 'layer': 5, 'position': 3, 'logit_difference': -2.048828125, 'p_clean_answer': 0.000294992933049798, 'p_corrupt_answer': 0.0022887922823429108}, {'condition': 'No-op control', 'layer': 5, 'position': 3, 'logit_difference': -2.048828125, 'p_clean_answer': 0.000294992933049798, 'p_corrupt_answer': 0.0022887922823429108}, {'condition': 'Zero control', 'layer': 5, 'position': 3, 'logit_difference': -2.048828125, 'p_clean_answer': 0.000294992933049798, 'p_corrupt_answer': 0.0022887922823429108}, {'condition': 'Random norm-matched control', 'layer': 5, 'position': 3, 'logit_difference': -2.048828125, 'p_clean_answer': 0.000294992933049798, 'p_corrupt_answer': 0.0022887922823429108}, {'condition': 'Patched', 'layer': 5, 'position': 4, 'logit_difference': 1.55712890625, 'p_clean_answer': 0.0019448678940534592, 'p_corrupt_answer': 0.0004098619392607361}, {'condition': 'No-op control', 'layer': 5, 'position': 4, 'logit_difference': -2.048828125, 'p_clean_answer': 0.000294992933049798, 'p_corrupt_answer': 0.0022887922823429108}, {'condition': 'Zero control', 'layer': 5, 'position': 4, 'logit_difference': -0.29443359375, 'p_clean_answer': 5.812497693113983e-05, 'p_corrupt_answer': 7.802498294040561e-05}, {'condition': 'Random norm-matched control', 'layer': 5, 'position': 4, 'logit_difference': 14.4002685546875, 'p_clean_answer': 1.686214883001025e-11, 'p_corrupt_answer': 9.396276277415475e-18}], 'residual_norms': [7.000734329223633, 9.803852081298828, 10.563419342041016, 11.329485893249512, 16.035472869873047, 93.23194885253906]}, {'clean': 'The capital of Germany is', 'corrupt': 'The capital of Spain is', 'clean_answer': ' Berlin', 'corrupt_answer': ' Madrid', 'clean_tokens': ['The', 'Ġcapital', 'Ġof', 'ĠGermany', 'Ġis'], 'corrupt_tokens': ['The', 'Ġcapital', 'Ġof', 'ĠSpain', 'Ġis'], 'positions': [3, 4], 'rows': [{'condition': 'Clean', 'layer': -1, 'position': -1, 'logit_difference': 3.068603515625, 'p_clean_answer': 0.001607217127457261, 'p_corrupt_answer': 7.471314165741205e-05}, {'condition': 'Corrupted', 'layer': -1, 'position': -1, 'logit_difference': -2.5987548828125, 'p_clean_answer': 9.333924390375614e-05, 'p_corrupt_answer': 0.0012551313266158104}, {'condition': 'Patched', 'layer': 0, 'position': 3, 'logit_difference': 3.0379638671875, 'p_clean_answer': 0.0014795990427955985, 'p_corrupt_answer': 7.092070882208645e-05}, {'condition': 'No-op control', 'layer': 0, 'position': 3, 'logit_difference': -2.5987548828125, 'p_clean_answer': 9.333924390375614e-05, 'p_corrupt_answer': 0.0012551313266158104}, {'condition': 'Zero control', 'layer': 0, 'position': 3, 'logit_difference': 0.7447509765625, 'p_clean_answer': 0.00010520592331886292, 'p_corrupt_answer': 4.995729977963492e-05}, {'condition': 'Random norm-matched control', 'layer': 0, 'position': 3, 'logit_difference': 1.0115966796875, 'p_clean_answer': 5.9207144659012556e-05, 'p_corrupt_answer': 2.1529962396016344e-05}, {'condition': 'Patched', 'layer': 0, 'position': 4, 'logit_difference': -2.58154296875, 'p_clean_answer': 0.00010416343138786033, 'p_corrupt_answer': 0.0013767818454653025}, {'condition': 'No-op control', 'layer': 0, 'position': 4, 'logit_difference': -2.5987548828125, 'p_clean_answer': 9.333924390375614e-05, 'p_corrupt_answer': 0.0012551313266158104}, {'condition': 'Zero control', 'layer': 0, 'position': 4, 'logit_difference': -1.5528564453125, 'p_clean_answer': 0.00025087196263484657, 'p_corrupt_answer': 0.0011853568721562624}, {'condition': 'Random norm-matched control', 'layer': 0, 'position': 4, 'logit_difference': -2.76513671875, 'p_clean_answer': 0.0003131017729174346, 'p_corrupt_answer': 0.004972435068339109}, {'condition': 'Patched', 'layer': 1, 'position': 3, 'logit_difference': 2.9693603515625, 'p_clean_answer': 0.0013163266703486443, 'p_corrupt_answer': 6.757512164767832e-05}, {'condition': 'No-op control', 'layer': 1, 'position': 3, 'logit_difference': -2.5987548828125, 'p_clean_answer': 9.333924390375614e-05, 'p_corrupt_answer': 0.0012551313266158104}, {'condition': 'Zero control', 'layer': 1, 'position': 3, 'logit_difference': 0.5950927734375, 'p_clean_answer': 3.713394471560605e-05, 'p_corrupt_answer': 2.0479794329730794e-05}, {'condition': 'Random norm-matched control', 'layer': 1, 'position': 3, 'logit_difference': 0.602294921875, 'p_clean_answer': 7.514503522543237e-05, 'p_corrupt_answer': 4.114593320991844e-05}, {'condition': 'Patched', 'layer': 1, 'position': 4, 'logit_difference': -2.5191650390625, 'p_clean_answer': 0.00012140027683926746, 'p_corrupt_answer': 0.0015075757401064038}, {'condition': 'No-op control', 'layer': 1, 'position': 4, 'logit_difference': -2.5987548828125, 'p_clean_answer': 9.333924390375614e-05, 'p_corrupt_answer': 0.0012551313266158104}, {'condition': 'Zero control', 'layer': 1, 'position': 4, 'logit_difference': -0.78173828125, 'p_clean_answer': 5.7456709328107536e-05, 'p_corrupt_answer': 0.0001255582901649177}, {'condition': 'Random norm-matched control', 'layer': 1, 'position': 4, 'logit_difference': -1.922119140625, 'p_clean_answer': 3.0606686777900904e-05, 'p_corrupt_answer': 0.00020920981478411704}, {'condition': 'Patched', 'layer': 2, 'position': 3, 'logit_difference': 2.7733154296875, 'p_clean_answer': 0.0010440130718052387, 'p_corrupt_answer': 6.520341412397102e-05}, {'condition': 'No-op control', 'layer': 2, 'position': 3, 'logit_difference': -2.5987548828125, 'p_clean_answer': 9.333924390375614e-05, 'p_corrupt_answer': 0.0012551313266158104}, {'condition': 'Zero control', 'layer': 2, 'position': 3, 'logit_difference': 0.6639404296875, 'p_clean_answer': 7.7886987128295e-05, 'p_corrupt_answer': 4.009767872048542e-05}, {'condition': 'Random norm-matched control', 'layer': 2, 'position': 3, 'logit_difference': -0.1409912109375, 'p_clean_answer': 2.412022695352789e-05, 'p_corrupt_answer': 2.777237750706263e-05}, {'condition': 'Patched', 'layer': 2, 'position': 4, 'logit_difference': -2.3896484375, 'p_clean_answer': 0.00015468271158169955, 'p_corrupt_answer': 0.0016875354340299964}, {'condition': 'No-op control', 'layer': 2, 'position': 4, 'logit_difference': -2.5987548828125, 'p_clean_answer': 9.333924390375614e-05, 'p_corrupt_answer': 0.0012551313266158104}, {'condition': 'Zero control', 'layer': 2, 'position': 4, 'logit_difference': -2.9462890625, 'p_clean_answer': 0.0007418054738081992, 'p_corrupt_answer': 0.014120404608547688}, {'condition': 'Random norm-matched control', 'layer': 2, 'position': 4, 'logit_difference': 2.842529296875, 'p_clean_answer': 9.238103666575626e-05, 'p_corrupt_answer': 5.383789357438218e-06}, {'condition': 'Patched', 'layer': 3, 'position': 3, 'logit_difference': 2.75830078125, 'p_clean_answer': 0.0010704115265980363, 'p_corrupt_answer': 6.786346057197079e-05}, {'condition': 'No-op control', 'layer': 3, 'position': 3, 'logit_difference': -2.5987548828125, 'p_clean_answer': 9.333924390375614e-05, 'p_corrupt_answer': 0.0012551313266158104}, {'condition': 'Zero control', 'layer': 3, 'position': 3, 'logit_difference': 0.6549072265625, 'p_clean_answer': 8.715256990399212e-05, 'p_corrupt_answer': 4.5274911826709285e-05}, {'condition': 'Random norm-matched control', 'layer': 3, 'position': 3, 'logit_difference': -0.4622802734375, 'p_clean_answer': 7.985283446032554e-06, 'p_corrupt_answer': 1.2678156963374931e-05}, {'condition': 'Patched', 'layer': 3, 'position': 4, 'logit_difference': -2.3626708984375, 'p_clean_answer': 0.00015361467376351357, 'p_corrupt_answer': 0.0016312767984345555}, {'condition': 'No-op control', 'layer': 3, 'position': 4, 'logit_difference': -2.5987548828125, 'p_clean_answer': 9.333924390375614e-05, 'p_corrupt_answer': 0.0012551313266158104}, {'condition': 'Zero control', 'layer': 3, 'position': 4, 'logit_difference': 0.63916015625, 'p_clean_answer': 5.1978200644953176e-05, 'p_corrupt_answer': 2.7430738555267453e-05}, {'condition': 'Random norm-matched control', 'layer': 3, 'position': 4, 'logit_difference': -2.4149169921875, 'p_clean_answer': 1.9674960640259087e-05, 'p_corrupt_answer': 0.0002201400202466175}, {'condition': 'Patched', 'layer': 4, 'position': 3, 'logit_difference': -2.4090576171875, 'p_clean_answer': 0.00010042615758720785, 'p_corrupt_answer': 0.0011170876678079367}, {'condition': 'No-op control', 'layer': 4, 'position': 3, 'logit_difference': -2.5987548828125, 'p_clean_answer': 9.333924390375614e-05, 'p_corrupt_answer': 0.0012551313266158104}, {'condition': 'Zero control', 'layer': 4, 'position': 3, 'logit_difference': -2.3438720703125, 'p_clean_answer': 0.00010624151036608964, 'p_corrupt_answer': 0.0011071971384808421}, {'condition': 'Random norm-matched control', 'layer': 4, 'position': 3, 'logit_difference': -2.3389892578125, 'p_clean_answer': 0.00010488642146810889, 'p_corrupt_answer': 0.0010877508902922273}, {'condition': 'Patched', 'layer': 4, 'position': 4, 'logit_difference': 2.8773193359375, 'p_clean_answer': 0.0015002540312707424, 'p_corrupt_answer': 8.444246486760676e-05}, {'condition': 'No-op control', 'layer': 4, 'position': 4, 'logit_difference': -2.5987548828125, 'p_clean_answer': 9.333924390375614e-05, 'p_corrupt_answer': 0.0012551313266158104}, {'condition': 'Zero control', 'layer': 4, 'position': 4, 'logit_difference': 0.6298828125, 'p_clean_answer': 2.1702875528717414e-05, 'p_corrupt_answer': 1.1560127859411296e-05}, {'condition': 'Random norm-matched control', 'layer': 4, 'position': 4, 'logit_difference': -1.8475341796875, 'p_clean_answer': 5.886911367269931e-06, 'p_corrupt_answer': 3.7347490433603525e-05}, {'condition': 'Patched', 'layer': 5, 'position': 3, 'logit_difference': -2.5987548828125, 'p_clean_answer': 9.333924390375614e-05, 'p_corrupt_answer': 0.0012551313266158104}, {'condition': 'No-op control', 'layer': 5, 'position': 3, 'logit_difference': -2.5987548828125, 'p_clean_answer': 9.333924390375614e-05, 'p_corrupt_answer': 0.0012551313266158104}, {'condition': 'Zero control', 'layer': 5, 'position': 3, 'logit_difference': -2.5987548828125, 'p_clean_answer': 9.333924390375614e-05, 'p_corrupt_answer': 0.0012551313266158104}, {'condition': 'Random norm-matched control', 'layer': 5, 'position': 3, 'logit_difference': -2.5987548828125, 'p_clean_answer': 9.333924390375614e-05, 'p_corrupt_answer': 0.0012551313266158104}, {'condition': 'Patched', 'layer': 5, 'position': 4, 'logit_difference': 3.068603515625, 'p_clean_answer': 0.001607217127457261, 'p_corrupt_answer': 7.471314165741205e-05}, {'condition': 'No-op control', 'layer': 5, 'position': 4, 'logit_difference': -2.5987548828125, 'p_clean_answer': 9.333924390375614e-05, 'p_corrupt_answer': 0.0012551313266158104}, {'condition': 'Zero control', 'layer': 5, 'position': 4, 'logit_difference': 1.03857421875, 'p_clean_answer': 5.7739562180358917e-05, 'p_corrupt_answer': 2.0437437342479825e-05}, {'condition': 'Random norm-matched control', 'layer': 5, 'position': 4, 'logit_difference': -4.07440185546875, 'p_clean_answer': 2.09249480733763e-17, 'p_corrupt_answer': 1.230707002745858e-15}], 'residual_norms': [7.110810279846191, 9.787389755249023, 10.446525573730469, 11.36587905883789, 15.912463188171387, 94.16485595703125]}, {'clean': 'The capital of Japan is', 'corrupt': 'The capital of China is', 'clean_answer': ' Tokyo', 'corrupt_answer': ' Beijing', 'clean_tokens': ['The', 'Ġcapital', 'Ġof', 'ĠJapan', 'Ġis'], 'corrupt_tokens': ['The', 'Ġcapital', 'Ġof', 'ĠChina', 'Ġis'], 'positions': [3, 4], 'rows': [{'condition': 'Clean', 'layer': -1, 'position': -1, 'logit_difference': 2.912353515625, 'p_clean_answer': 0.004140540491789579, 'p_corrupt_answer': 0.00022502873616758734}, {'condition': 'Corrupted', 'layer': -1, 'position': -1, 'logit_difference': -1.34375, 'p_clean_answer': 0.00020214311371091753, 'p_corrupt_answer': 0.0007748937350697815}, {'condition': 'Patched', 'layer': 0, 'position': 3, 'logit_difference': 2.80615234375, 'p_clean_answer': 0.003330013481900096, 'p_corrupt_answer': 0.00020125630544498563}, {'condition': 'No-op control', 'layer': 0, 'position': 3, 'logit_difference': -1.34375, 'p_clean_answer': 0.00020214311371091753, 'p_corrupt_answer': 0.0007748937350697815}, {'condition': 'Zero control', 'layer': 0, 'position': 3, 'logit_difference': 0.9674072265625, 'p_clean_answer': 7.84801522968337e-05, 'p_corrupt_answer': 2.9827731850673445e-05}, {'condition': 'Random norm-matched control', 'layer': 0, 'position': 3, 'logit_difference': 0.894287109375, 'p_clean_answer': 7.142584945540875e-05, 'p_corrupt_answer': 2.9205957616795786e-05}, {'condition': 'Patched', 'layer': 0, 'position': 4, 'logit_difference': -1.2821044921875, 'p_clean_answer': 0.000229804907576181, 'p_corrupt_answer': 0.0008282666094601154}, {'condition': 'No-op control', 'layer': 0, 'position': 4, 'logit_difference': -1.34375, 'p_clean_answer': 0.00020214311371091753, 'p_corrupt_answer': 0.0007748937350697815}, {'condition': 'Zero control', 'layer': 0, 'position': 4, 'logit_difference': 0.213134765625, 'p_clean_answer': 0.00010500505595700815, 'p_corrupt_answer': 8.484904537908733e-05}, {'condition': 'Random norm-matched control', 'layer': 0, 'position': 4, 'logit_difference': -1.3072509765625, 'p_clean_answer': 9.503376350039616e-05, 'p_corrupt_answer': 0.0003512447001412511}, {'condition': 'Patched', 'layer': 1, 'position': 3, 'logit_difference': 2.79345703125, 'p_clean_answer': 0.0033180085010826588, 'p_corrupt_answer': 0.0002030928008025512}, {'condition': 'No-op control', 'layer': 1, 'position': 3, 'logit_difference': -1.34375, 'p_clean_answer': 0.00020214311371091753, 'p_corrupt_answer': 0.0007748937350697815}, {'condition': 'Zero control', 'layer': 1, 'position': 3, 'logit_difference': 0.787841796875, 'p_clean_answer': 9.197239705827087e-05, 'p_corrupt_answer': 4.183137571089901e-05}, {'condition': 'Random norm-matched control', 'layer': 1, 'position': 3, 'logit_difference': 0.947998046875, 'p_clean_answer': 3.354357977514155e-05, 'p_corrupt_answer': 1.2998675629205536e-05}, {'condition': 'Patched', 'layer': 1, 'position': 4, 'logit_difference': -1.24609375, 'p_clean_answer': 0.00023990280169527978, 'p_corrupt_answer': 0.0008340785861946642}, {'condition': 'No-op control', 'layer': 1, 'position': 4, 'logit_difference': -1.34375, 'p_clean_answer': 0.00020214311371091753, 'p_corrupt_answer': 0.0007748937350697815}, {'condition': 'Zero control', 'layer': 1, 'position': 4, 'logit_difference': 0.51416015625, 'p_clean_answer': 2.9260985684231855e-05, 'p_corrupt_answer': 1.7498146917205304e-05}, {'condition': 'Random norm-matched control', 'layer': 1, 'position': 4, 'logit_difference': -4.5281982421875, 'p_clean_answer': 3.490705466902e-07, 'p_corrupt_answer': 3.23209933412727e-05}, {'condition': 'Patched', 'layer': 2, 'position': 3, 'logit_difference': 2.75732421875, 'p_clean_answer': 0.002236707368865609, 'p_corrupt_answer': 0.0001419444743078202}, {'condition': 'No-op control', 'layer': 2, 'position': 3, 'logit_difference': -1.34375, 'p_clean_answer': 0.00020214311371091753, 'p_corrupt_answer': 0.0007748937350697815}, {'condition': 'Zero control', 'layer': 2, 'position': 3, 'logit_difference': 0.585693359375, 'p_clean_answer': 6.781737465644255e-05, 'p_corrupt_answer': 3.775526784011163e-05}, {'condition': 'Random norm-matched control', 'layer': 2, 'position': 3, 'logit_difference': 0.908447265625, 'p_clean_answer': 6.399814446922392e-05, 'p_corrupt_answer': 2.5800834919209592e-05}, {'condition': 'Patched', 'layer': 2, 'position': 4, 'logit_difference': -1.1846923828125, 'p_clean_answer': 0.0003886999620590359, 'p_corrupt_answer': 0.0012709248112514615}, {'condition': 'No-op control', 'layer': 2, 'position': 4, 'logit_difference': -1.34375, 'p_clean_answer': 0.00020214311371091753, 'p_corrupt_answer': 0.0007748937350697815}, {'condition': 'Zero control', 'layer': 2, 'position': 4, 'logit_difference': -0.406982421875, 'p_clean_answer': 0.0013223610585555434, 'p_corrupt_answer': 0.0019865534268319607}, {'condition': 'Random norm-matched control', 'layer': 2, 'position': 4, 'logit_difference': 1.3817138671875, 'p_clean_answer': 7.933635970402975e-06, 'p_corrupt_answer': 1.9925148535548942e-06}, {'condition': 'Patched', 'layer': 3, 'position': 3, 'logit_difference': 2.6856689453125, 'p_clean_answer': 0.002437145449221134, 'p_corrupt_answer': 0.00016615379718132317}, {'condition': 'No-op control', 'layer': 3, 'position': 3, 'logit_difference': -1.34375, 'p_clean_answer': 0.00020214311371091753, 'p_corrupt_answer': 0.0007748937350697815}, {'condition': 'Zero control', 'layer': 3, 'position': 3, 'logit_difference': 0.5565185546875, 'p_clean_answer': 9.430326463188976e-05, 'p_corrupt_answer': 5.405474075814709e-05}, {'condition': 'Random norm-matched control', 'layer': 3, 'position': 3, 'logit_difference': 0.895263671875, 'p_clean_answer': 0.0001781375758582726, 'p_corrupt_answer': 7.276918040588498e-05}, {'condition': 'Patched', 'layer': 3, 'position': 4, 'logit_difference': -1.1141357421875, 'p_clean_answer': 0.00035266217309981585, 'p_corrupt_answer': 0.0010745382169261575}, {'condition': 'No-op control', 'layer': 3, 'position': 4, 'logit_difference': -1.34375, 'p_clean_answer': 0.00020214311371091753, 'p_corrupt_answer': 0.0007748937350697815}, {'condition': 'Zero control', 'layer': 3, 'position': 4, 'logit_difference': 0.84033203125, 'p_clean_answer': 9.894199138216209e-06, 'p_corrupt_answer': 4.2700121412053704e-06}, {'condition': 'Random norm-matched control', 'layer': 3, 'position': 4, 'logit_difference': -0.87255859375, 'p_clean_answer': 3.9202728657983243e-05, 'p_corrupt_answer': 9.381313429912552e-05}, {'condition': 'Patched', 'layer': 4, 'position': 3, 'logit_difference': -1.252685546875, 'p_clean_answer': 0.00022109344718046486, 'p_corrupt_answer': 0.0007737671840004623}, {'condition': 'No-op control', 'layer': 4, 'position': 3, 'logit_difference': -1.34375, 'p_clean_answer': 0.00020214311371091753, 'p_corrupt_answer': 0.0007748937350697815}, {'condition': 'Zero control', 'layer': 4, 'position': 3, 'logit_difference': -1.31884765625, 'p_clean_answer': 0.00021481336443684995, 'p_corrupt_answer': 0.0008032108889892697}, {'condition': 'Random norm-matched control', 'layer': 4, 'position': 3, 'logit_difference': -1.75, 'p_clean_answer': 6.665904220426455e-05, 'p_corrupt_answer': 0.0003835963143501431}, {'condition': 'Patched', 'layer': 4, 'position': 4, 'logit_difference': 2.8165283203125, 'p_clean_answer': 0.00378618948161602, 'p_corrupt_answer': 0.00022646423894912004}, {'condition': 'No-op control', 'layer': 4, 'position': 4, 'logit_difference': -1.34375, 'p_clean_answer': 0.00020214311371091753, 'p_corrupt_answer': 0.0007748937350697815}, {'condition': 'Zero control', 'layer': 4, 'position': 4, 'logit_difference': 1.42626953125, 'p_clean_answer': 3.3595075365155935e-05, 'p_corrupt_answer': 8.06964817456901e-06}, {'condition': 'Random norm-matched control', 'layer': 4, 'position': 4, 'logit_difference': 2.61376953125, 'p_clean_answer': 2.0397645130287856e-05, 'p_corrupt_answer': 1.494287971581798e-06}, {'condition': 'Patched', 'layer': 5, 'position': 3, 'logit_difference': -1.34375, 'p_clean_answer': 0.00020214311371091753, 'p_corrupt_answer': 0.0007748937350697815}, {'condition': 'No-op control', 'layer': 5, 'position': 3, 'logit_difference': -1.34375, 'p_clean_answer': 0.00020214311371091753, 'p_corrupt_answer': 0.0007748937350697815}, {'condition': 'Zero control', 'layer': 5, 'position': 3, 'logit_difference': -1.34375, 'p_clean_answer': 0.00020214311371091753, 'p_corrupt_answer': 0.0007748937350697815}, {'condition': 'Random norm-matched control', 'layer': 5, 'position': 3, 'logit_difference': -1.34375, 'p_clean_answer': 0.00020214311371091753, 'p_corrupt_answer': 0.0007748937350697815}, {'condition': 'Patched', 'layer': 5, 'position': 4, 'logit_difference': 2.912353515625, 'p_clean_answer': 0.004140540491789579, 'p_corrupt_answer': 0.00022502873616758734}, {'condition': 'No-op control', 'layer': 5, 'position': 4, 'logit_difference': -1.34375, 'p_clean_answer': 0.00020214311371091753, 'p_corrupt_answer': 0.0007748937350697815}, {'condition': 'Zero control', 'layer': 5, 'position': 4, 'logit_difference': -0.04510498046875, 'p_clean_answer': 2.5115472453762777e-05, 'p_corrupt_answer': 2.6274243282387033e-05}, {'condition': 'Random norm-matched control', 'layer': 5, 'position': 4, 'logit_difference': -16.3575439453125, 'p_clean_answer': 1.546693080470438e-17, 'p_corrupt_answer': 1.9651476279580748e-10}], 'residual_norms': [6.926535606384277, 9.85919189453125, 10.356697082519531, 11.179608345031738, 16.143373489379883, 92.03011322021484]}, {'clean': 'The capital of Italy is', 'corrupt': 'The capital of Greece is', 'clean_answer': ' Rome', 'corrupt_answer': ' Athens', 'clean_tokens': ['The', 'Ġcapital', 'Ġof', 'ĠItaly', 'Ġis'], 'corrupt_tokens': ['The', 'Ġcapital', 'Ġof', 'ĠGreece', 'Ġis'], 'positions': [3, 4], 'rows': [{'condition': 'Clean', 'layer': -1, 'position': -1, 'logit_difference': 2.646728515625, 'p_clean_answer': 0.0008934771176427603, 'p_corrupt_answer': 6.333208875730634e-05}, {'condition': 'Corrupted', 'layer': -1, 'position': -1, 'logit_difference': -0.56689453125, 'p_clean_answer': 0.0002926239976659417, 'p_corrupt_answer': 0.0005158329731784761}, {'condition': 'Patched', 'layer': 0, 'position': 3, 'logit_difference': 2.6839599609375, 'p_clean_answer': 0.0007971825543791056, 'p_corrupt_answer': 5.4441337852040306e-05}, {'condition': 'No-op control', 'layer': 0, 'position': 3, 'logit_difference': -0.56689453125, 'p_clean_answer': 0.0002926239976659417, 'p_corrupt_answer': 0.0005158329731784761}, {'condition': 'Zero control', 'layer': 0, 'position': 3, 'logit_difference': 1.5247802734375, 'p_clean_answer': 0.00011289031681371853, 'p_corrupt_answer': 2.4572707843617536e-05}, {'condition': 'Random norm-matched control', 'layer': 0, 'position': 3, 'logit_difference': 1.103515625, 'p_clean_answer': 6.137391756055877e-05, 'p_corrupt_answer': 2.0357907487777993e-05}, {'condition': 'Patched', 'layer': 0, 'position': 4, 'logit_difference': -0.634521484375, 'p_clean_answer': 0.0003284046542830765, 'p_corrupt_answer': 0.0006194104207679629}, {'condition': 'No-op control', 'layer': 0, 'position': 4, 'logit_difference': -0.56689453125, 'p_clean_answer': 0.0002926239976659417, 'p_corrupt_answer': 0.0005158329731784761}, {'condition': 'Zero control', 'layer': 0, 'position': 4, 'logit_difference': -0.3017578125, 'p_clean_answer': 0.0017463015392422676, 'p_corrupt_answer': 0.0023614075034856796}, {'condition': 'Random norm-matched control', 'layer': 0, 'position': 4, 'logit_difference': -4.5218505859375, 'p_clean_answer': 3.3553880030012806e-07, 'p_corrupt_answer': 3.087148434133269e-05}, {'condition': 'Patched', 'layer': 1, 'position': 3, 'logit_difference': 2.6199951171875, 'p_clean_answer': 0.0007483805529773235, 'p_corrupt_answer': 5.448451338452287e-05}, {'condition': 'No-op control', 'layer': 1, 'position': 3, 'logit_difference': -0.56689453125, 'p_clean_answer': 0.0002926239976659417, 'p_corrupt_answer': 0.0005158329731784761}, {'condition': 'Zero control', 'layer': 1, 'position': 3, 'logit_difference': 1.555419921875, 'p_clean_answer': 8.325278031406924e-05, 'p_corrupt_answer': 1.7574720914126374e-05}, {'condition': 'Random norm-matched control', 'layer': 1, 'position': 3, 'logit_difference': -0.076171875, 'p_clean_answer': 2.1386196749517694e-05, 'p_corrupt_answer': 2.307887189090252e-05}, {'condition': 'Patched', 'layer': 1, 'position': 4, 'logit_difference': -0.58056640625, 'p_clean_answer': 0.0003615020541474223, 'p_corrupt_answer': 0.0006460223812609911}, {'condition': 'No-op control', 'layer': 1, 'position': 4, 'logit_difference': -0.56689453125, 'p_clean_answer': 0.0002926239976659417, 'p_corrupt_answer': 0.0005158329731784761}, {'condition': 'Zero control', 'layer': 1, 'position': 4, 'logit_difference': 0.3721923828125, 'p_clean_answer': 3.796036617131904e-05, 'p_corrupt_answer': 2.616310666780919e-05}, {'condition': 'Random norm-matched control', 'layer': 1, 'position': 4, 'logit_difference': 0.5045166015625, 'p_clean_answer': 0.00011763974180212244, 'p_corrupt_answer': 7.103056850610301e-05}, {'condition': 'Patched', 'layer': 2, 'position': 3, 'logit_difference': 2.546630859375, 'p_clean_answer': 0.0006700000376440585, 'p_corrupt_answer': 5.249126843409613e-05}, {'condition': 'No-op control', 'layer': 2, 'position': 3, 'logit_difference': -0.56689453125, 'p_clean_answer': 0.0002926239976659417, 'p_corrupt_answer': 0.0005158329731784761}, {'condition': 'Zero control', 'layer': 2, 'position': 3, 'logit_difference': 1.061767578125, 'p_clean_answer': 5.919741670368239e-05, 'p_corrupt_answer': 2.0473069525905885e-05}, {'condition': 'Random norm-matched control', 'layer': 2, 'position': 3, 'logit_difference': 0.6181640625, 'p_clean_answer': 1.8874567103921436e-05, 'p_corrupt_answer': 1.0172126167162787e-05}, {'condition': 'Patched', 'layer': 2, 'position': 4, 'logit_difference': -0.5078125, 'p_clean_answer': 0.0004022792272735387, 'p_corrupt_answer': 0.0006684482214041054}, {'condition': 'No-op control', 'layer': 2, 'position': 4, 'logit_difference': -0.56689453125, 'p_clean_answer': 0.0002926239976659417, 'p_corrupt_answer': 0.0005158329731784761}, {'condition': 'Zero control', 'layer': 2, 'position': 4, 'logit_difference': -0.3221435546875, 'p_clean_answer': 0.001148705487139523, 'p_corrupt_answer': 0.0015853088116273284}, {'condition': 'Random norm-matched control', 'layer': 2, 'position': 4, 'logit_difference': -0.8779296875, 'p_clean_answer': 0.00021564339112956077, 'p_corrupt_answer': 0.0005188193754293025}, {'condition': 'Patched', 'layer': 3, 'position': 3, 'logit_difference': 2.509765625, 'p_clean_answer': 0.0006377698155120015, 'p_corrupt_answer': 5.184258407098241e-05}, {'condition': 'No-op control', 'layer': 3, 'position': 3, 'logit_difference': -0.56689453125, 'p_clean_answer': 0.0002926239976659417, 'p_corrupt_answer': 0.0005158329731784761}, {'condition': 'Zero control', 'layer': 3, 'position': 3, 'logit_difference': 1.18310546875, 'p_clean_answer': 8.506407175445929e-05, 'p_corrupt_answer': 2.6057337890961207e-05}, {'condition': 'Random norm-matched control', 'layer': 3, 'position': 3, 'logit_difference': 0.120849609375, 'p_clean_answer': 1.0392931471869815e-05, 'p_corrupt_answer': 9.209875315718818e-06}, {'condition': 'Patched', 'layer': 3, 'position': 4, 'logit_difference': -0.4931640625, 'p_clean_answer': 0.00042447124724276364, 'p_corrupt_answer': 0.0006950670504011214}, {'condition': 'No-op control', 'layer': 3, 'position': 4, 'logit_difference': -0.56689453125, 'p_clean_answer': 0.0002926239976659417, 'p_corrupt_answer': 0.0005158329731784761}, {'condition': 'Zero control', 'layer': 3, 'position': 4, 'logit_difference': 1.39208984375, 'p_clean_answer': 3.484515764284879e-05, 'p_corrupt_answer': 8.660949788463768e-06}, {'condition': 'Random norm-matched control', 'layer': 3, 'position': 4, 'logit_difference': 1.23583984375, 'p_clean_answer': 4.303562946006423e-07, 'p_corrupt_answer': 1.2505749680258305e-07}, {'condition': 'Patched', 'layer': 4, 'position': 3, 'logit_difference': -0.4681396484375, 'p_clean_answer': 0.000313681666739285, 'p_corrupt_answer': 0.0005009560263715684}, {'condition': 'No-op control', 'layer': 4, 'position': 3, 'logit_difference': -0.56689453125, 'p_clean_answer': 0.0002926239976659417, 'p_corrupt_answer': 0.0005158329731784761}, {'condition': 'Zero control', 'layer': 4, 'position': 3, 'logit_difference': -0.4801025390625, 'p_clean_answer': 0.00029381661443039775, 'p_corrupt_answer': 0.00047487823758274317}, {'condition': 'Random norm-matched control', 'layer': 4, 'position': 3, 'logit_difference': -0.8463134765625, 'p_clean_answer': 0.00015750282909721136, 'p_corrupt_answer': 0.00036714502493850887}, {'condition': 'Patched', 'layer': 4, 'position': 4, 'logit_difference': 2.5517578125, 'p_clean_answer': 0.0008376878686249256, 'p_corrupt_answer': 6.529319216497242e-05}, {'condition': 'No-op control', 'layer': 4, 'position': 4, 'logit_difference': -0.56689453125, 'p_clean_answer': 0.0002926239976659417, 'p_corrupt_answer': 0.0005158329731784761}, {'condition': 'Zero control', 'layer': 4, 'position': 4, 'logit_difference': 2.6104736328125, 'p_clean_answer': 4.381401959108189e-05, 'p_corrupt_answer': 3.2203183764067944e-06}, {'condition': 'Random norm-matched control', 'layer': 4, 'position': 4, 'logit_difference': 2.60595703125, 'p_clean_answer': 4.3775358790298924e-05, 'p_corrupt_answer': 3.2320417631126475e-06}, {'condition': 'Patched', 'layer': 5, 'position': 3, 'logit_difference': -0.56689453125, 'p_clean_answer': 0.0002926239976659417, 'p_corrupt_answer': 0.0005158329731784761}, {'condition': 'No-op control', 'layer': 5, 'position': 3, 'logit_difference': -0.56689453125, 'p_clean_answer': 0.0002926239976659417, 'p_corrupt_answer': 0.0005158329731784761}, {'condition': 'Zero control', 'layer': 5, 'position': 3, 'logit_difference': -0.56689453125, 'p_clean_answer': 0.0002926239976659417, 'p_corrupt_answer': 0.0005158329731784761}, {'condition': 'Random norm-matched control', 'layer': 5, 'position': 3, 'logit_difference': -0.56689453125, 'p_clean_answer': 0.0002926239976659417, 'p_corrupt_answer': 0.0005158329731784761}, {'condition': 'Patched', 'layer': 5, 'position': 4, 'logit_difference': 2.646728515625, 'p_clean_answer': 0.0008934771176427603, 'p_corrupt_answer': 6.333208875730634e-05}, {'condition': 'No-op control', 'layer': 5, 'position': 4, 'logit_difference': -0.56689453125, 'p_clean_answer': 0.0002926239976659417, 'p_corrupt_answer': 0.0005158329731784761}, {'condition': 'Zero control', 'layer': 5, 'position': 4, 'logit_difference': 0.4127197265625, 'p_clean_answer': 2.0489895177888684e-05, 'p_corrupt_answer': 1.356119173578918e-05}, {'condition': 'Random norm-matched control', 'layer': 5, 'position': 4, 'logit_difference': -9.8231201171875, 'p_clean_answer': 9.144730573166536e-23, 'p_corrupt_answer': 1.6877098386817432e-18}], 'residual_norms': [7.046963691711426, 9.859155654907227, 10.483169555664062, 11.39472484588623, 15.967036247253418, 91.78528594970703]}, {'clean': 'The capital of Canada is', 'corrupt': 'The capital of Russia is', 'clean_answer': ' Ottawa', 'corrupt_answer': ' Moscow', 'clean_tokens': ['The', 'Ġcapital', 'Ġof', 'ĠCanada', 'Ġis'], 'corrupt_tokens': ['The', 'Ġcapital', 'Ġof', 'ĠRussia', 'Ġis'], 'positions': [3, 4], 'rows': [{'condition': 'Clean', 'layer': -1, 'position': -1, 'logit_difference': 1.9915771484375, 'p_clean_answer': 0.000249287870246917, 'p_corrupt_answer': 3.4022810723399743e-05}, {'condition': 'Corrupted', 'layer': -1, 'position': -1, 'logit_difference': -5.2939453125, 'p_clean_answer': 7.749201358819846e-06, 'p_corrupt_answer': 0.001543079037219286}, {'condition': 'Patched', 'layer': 0, 'position': 3, 'logit_difference': 1.727783203125, 'p_clean_answer': 0.000198131674551405, 'p_corrupt_answer': 3.5203607694711536e-05}, {'condition': 'No-op control', 'layer': 0, 'position': 3, 'logit_difference': -5.2939453125, 'p_clean_answer': 7.749201358819846e-06, 'p_corrupt_answer': 0.001543079037219286}, {'condition': 'Zero control', 'layer': 0, 'position': 3, 'logit_difference': -2.55615234375, 'p_clean_answer': 7.68435620557284e-06, 'p_corrupt_answer': 9.902170131681487e-05}, {'condition': 'Random norm-matched control', 'layer': 0, 'position': 3, 'logit_difference': -1.98291015625, 'p_clean_answer': 1.0834867680387106e-05, 'p_corrupt_answer': 7.870286935940385e-05}, {'condition': 'Patched', 'layer': 0, 'position': 4, 'logit_difference': -5.1505126953125, 'p_clean_answer': 9.333541129308287e-06, 'p_corrupt_answer': 0.0016102218069136143}, {'condition': 'No-op control', 'layer': 0, 'position': 4, 'logit_difference': -5.2939453125, 'p_clean_answer': 7.749201358819846e-06, 'p_corrupt_answer': 0.001543079037219286}, {'condition': 'Zero control', 'layer': 0, 'position': 4, 'logit_difference': -4.3345947265625, 'p_clean_answer': 4.410222027217969e-05, 'p_corrupt_answer': 0.003364735981449485}, {'condition': 'Random norm-matched control', 'layer': 0, 'position': 4, 'logit_difference': -4.1114501953125, 'p_clean_answer': 2.190113491451484e-06, 'p_corrupt_answer': 0.00013367393694352359}, {'condition': 'Patched', 'layer': 1, 'position': 3, 'logit_difference': 1.60693359375, 'p_clean_answer': 0.00018559007730800658, 'p_corrupt_answer': 3.721108805621043e-05}, {'condition': 'No-op control', 'layer': 1, 'position': 3, 'logit_difference': -5.2939453125, 'p_clean_answer': 7.749201358819846e-06, 'p_corrupt_answer': 0.001543079037219286}, {'condition': 'Zero control', 'layer': 1, 'position': 3, 'logit_difference': -2.63134765625, 'p_clean_answer': 4.751538654090837e-06, 'p_corrupt_answer': 6.601065251743421e-05}, {'condition': 'Random norm-matched control', 'layer': 1, 'position': 3, 'logit_difference': -1.353759765625, 'p_clean_answer': 1.3753314306086395e-05, 'p_corrupt_answer': 5.325222809915431e-05}, {'condition': 'Patched', 'layer': 1, 'position': 4, 'logit_difference': -5.082275390625, 'p_clean_answer': 9.889869033941068e-06, 'p_corrupt_answer': 0.0015936564886942506}, {'condition': 'No-op control', 'layer': 1, 'position': 4, 'logit_difference': -5.2939453125, 'p_clean_answer': 7.749201358819846e-06, 'p_corrupt_answer': 0.001543079037219286}, {'condition': 'Zero control', 'layer': 1, 'position': 4, 'logit_difference': -3.106689453125, 'p_clean_answer': 3.0528742627211614e-06, 'p_corrupt_answer': 6.82224053889513e-05}, {'condition': 'Random norm-matched control', 'layer': 1, 'position': 4, 'logit_difference': -2.8201904296875, 'p_clean_answer': 1.8729202793110744e-06, 'p_corrupt_answer': 3.142768764519133e-05}, {'condition': 'Patched', 'layer': 2, 'position': 3, 'logit_difference': 1.2847900390625, 'p_clean_answer': 0.0001490953000029549, 'p_corrupt_answer': 4.125596387893893e-05}, {'condition': 'No-op control', 'layer': 2, 'position': 3, 'logit_difference': -5.2939453125, 'p_clean_answer': 7.749201358819846e-06, 'p_corrupt_answer': 0.001543079037219286}, {'condition': 'Zero control', 'layer': 2, 'position': 3, 'logit_difference': -2.1053466796875, 'p_clean_answer': 1.1799415915447753e-05, 'p_corrupt_answer': 9.68725944403559e-05}, {'condition': 'Random norm-matched control', 'layer': 2, 'position': 3, 'logit_difference': -2.1002197265625, 'p_clean_answer': 1.2542876902443822e-05, 'p_corrupt_answer': 0.00010244976874673739}, {'condition': 'Patched', 'layer': 2, 'position': 4, 'logit_difference': -4.8040771484375, 'p_clean_answer': 1.393308411934413e-05, 'p_corrupt_answer': 0.001699931686744094}, {'condition': 'No-op control', 'layer': 2, 'position': 4, 'logit_difference': -5.2939453125, 'p_clean_answer': 7.749201358819846e-06, 'p_corrupt_answer': 0.001543079037219286}, {'condition': 'Zero control', 'layer': 2, 'position': 4, 'logit_difference': -3.924072265625, 'p_clean_answer': 4.863757567363791e-05, 'p_corrupt_answer': 0.0024613584391772747}, {'condition': 'Random norm-matched control', 'layer': 2, 'position': 4, 'logit_difference': -4.6396484375, 'p_clean_answer': 3.1710940220364137e-06, 'p_corrupt_answer': 0.00032823346555233}, {'condition': 'Patched', 'layer': 3, 'position': 3, 'logit_difference': 1.1851806640625, 'p_clean_answer': 0.0001394124119542539, 'p_corrupt_answer': 4.261710637365468e-05}, {'condition': 'No-op control', 'layer': 3, 'position': 3, 'logit_difference': -5.2939453125, 'p_clean_answer': 7.749201358819846e-06, 'p_corrupt_answer': 0.001543079037219286}, {'condition': 'Zero control', 'layer': 3, 'position': 3, 'logit_difference': -2.3052978515625, 'p_clean_answer': 1.0947535884042736e-05, 'p_corrupt_answer': 0.00010977275087498128}, {'condition': 'Random norm-matched control', 'layer': 3, 'position': 3, 'logit_difference': -2.5252685546875, 'p_clean_answer': 7.5007337727583945e-06, 'p_corrupt_answer': 9.371604392072186e-05}, {'condition': 'Patched', 'layer': 3, 'position': 4, 'logit_difference': -4.6798095703125, 'p_clean_answer': 1.5385352526209317e-05, 'p_corrupt_answer': 0.0016577648930251598}, {'condition': 'No-op control', 'layer': 3, 'position': 4, 'logit_difference': -5.2939453125, 'p_clean_answer': 7.749201358819846e-06, 'p_corrupt_answer': 0.001543079037219286}, {'condition': 'Zero control', 'layer': 3, 'position': 4, 'logit_difference': -1.06689453125, 'p_clean_answer': 3.9897677197586745e-06, 'p_corrupt_answer': 1.1595619980653282e-05}, {'condition': 'Random norm-matched control', 'layer': 3, 'position': 4, 'logit_difference': -0.5289306640625, 'p_clean_answer': 1.5792038539075293e-05, 'p_corrupt_answer': 2.6800931664183736e-05}, {'condition': 'Patched', 'layer': 4, 'position': 3, 'logit_difference': -5.163818359375, 'p_clean_answer': 8.411583621636964e-06, 'p_corrupt_answer': 0.0014706035144627094}, {'condition': 'No-op control', 'layer': 4, 'position': 3, 'logit_difference': -5.2939453125, 'p_clean_answer': 7.749201358819846e-06, 'p_corrupt_answer': 0.001543079037219286}, {'condition': 'Zero control', 'layer': 4, 'position': 3, 'logit_difference': -5.1756591796875, 'p_clean_answer': 8.221449206757825e-06, 'p_corrupt_answer': 0.0014544829027727246}, {'condition': 'Random norm-matched control', 'layer': 4, 'position': 3, 'logit_difference': -5.2236328125, 'p_clean_answer': 3.665065833047265e-06, 'p_corrupt_answer': 0.0006802627467550337}, {'condition': 'Patched', 'layer': 4, 'position': 4, 'logit_difference': 1.839599609375, 'p_clean_answer': 0.00022672058548778296, 'p_corrupt_answer': 3.6021599953528494e-05}, {'condition': 'No-op control', 'layer': 4, 'position': 4, 'logit_difference': -5.2939453125, 'p_clean_answer': 7.749201358819846e-06, 'p_corrupt_answer': 0.001543079037219286}, {'condition': 'Zero control', 'layer': 4, 'position': 4, 'logit_difference': -1.8505859375, 'p_clean_answer': 2.8020317586197052e-06, 'p_corrupt_answer': 1.7830861906986684e-05}, {'condition': 'Random norm-matched control', 'layer': 4, 'position': 4, 'logit_difference': -3.47607421875, 'p_clean_answer': 4.356832903340546e-07, 'p_corrupt_answer': 1.4086747796682175e-05}, {'condition': 'Patched', 'layer': 5, 'position': 3, 'logit_difference': -5.2939453125, 'p_clean_answer': 7.749201358819846e-06, 'p_corrupt_answer': 0.001543079037219286}, {'condition': 'No-op control', 'layer': 5, 'position': 3, 'logit_difference': -5.2939453125, 'p_clean_answer': 7.749201358819846e-06, 'p_corrupt_answer': 0.001543079037219286}, {'condition': 'Zero control', 'layer': 5, 'position': 3, 'logit_difference': -5.2939453125, 'p_clean_answer': 7.749201358819846e-06, 'p_corrupt_answer': 0.001543079037219286}, {'condition': 'Random norm-matched control', 'layer': 5, 'position': 3, 'logit_difference': -5.2939453125, 'p_clean_answer': 7.749201358819846e-06, 'p_corrupt_answer': 0.001543079037219286}, {'condition': 'Patched', 'layer': 5, 'position': 4, 'logit_difference': 1.9915771484375, 'p_clean_answer': 0.000249287870246917, 'p_corrupt_answer': 3.4022810723399743e-05}, {'condition': 'No-op control', 'layer': 5, 'position': 4, 'logit_difference': -5.2939453125, 'p_clean_answer': 7.749201358819846e-06, 'p_corrupt_answer': 0.001543079037219286}, {'condition': 'Zero control', 'layer': 5, 'position': 4, 'logit_difference': -0.98699951171875, 'p_clean_answer': 1.2674412573687732e-05, 'p_corrupt_answer': 3.400762579985894e-05}, {'condition': 'Random norm-matched control', 'layer': 5, 'position': 4, 'logit_difference': -9.03704833984375, 'p_clean_answer': 7.568397582009324e-19, 'p_corrupt_answer': 6.3642049190847615e-15}], 'residual_norms': [7.054368019104004, 9.872750282287598, 10.385741233825684, 11.390432357788086, 16.195310592651367, 91.97669982910156]}, {'clean': 'The capital of Portugal is', 'corrupt': 'The capital of France is', 'clean_answer': ' Lisbon', 'corrupt_answer': ' Paris', 'clean_tokens': ['The', 'Ġcapital', 'Ġof', 'ĠPortugal', 'Ġis'], 'corrupt_tokens': ['The', 'Ġcapital', 'Ġof', 'ĠFrance', 'Ġis'], 'positions': [3, 4], 'rows': [{'condition': 'Clean', 'layer': -1, 'position': -1, 'logit_difference': -0.0050048828125, 'p_clean_answer': 0.0002310975978616625, 'p_corrupt_answer': 0.00023225712357088923}, {'condition': 'Corrupted', 'layer': -1, 'position': -1, 'logit_difference': -4.726318359375, 'p_clean_answer': 1.722963497741148e-05, 'p_corrupt_answer': 0.0019448678940534592}, {'condition': 'Patched', 'layer': 0, 'position': 3, 'logit_difference': -0.1097412109375, 'p_clean_answer': 0.000205121366889216, 'p_corrupt_answer': 0.00022891323897056282}, {'condition': 'No-op control', 'layer': 0, 'position': 3, 'logit_difference': -4.726318359375, 'p_clean_answer': 1.722963497741148e-05, 'p_corrupt_answer': 0.0019448678940534592}, {'condition': 'Zero control', 'layer': 0, 'position': 3, 'logit_difference': -4.471923828125, 'p_clean_answer': 3.965176802012138e-06, 'p_corrupt_answer': 0.00034705185680650175}, {'condition': 'Random norm-matched control', 'layer': 0, 'position': 3, 'logit_difference': -4.05078125, 'p_clean_answer': 2.254277205793187e-06, 'p_corrupt_answer': 0.00012949090159963816}, {'condition': 'Patched', 'layer': 0, 'position': 4, 'logit_difference': -4.6053466796875, 'p_clean_answer': 1.9812881873804145e-05, 'p_corrupt_answer': 0.00198163790628314}, {'condition': 'No-op control', 'layer': 0, 'position': 4, 'logit_difference': -4.726318359375, 'p_clean_answer': 1.722963497741148e-05, 'p_corrupt_answer': 0.0019448678940534592}, {'condition': 'Zero control', 'layer': 0, 'position': 4, 'logit_difference': -6.358154296875, 'p_clean_answer': 1.5533585610683076e-05, 'p_corrupt_answer': 0.00896567665040493}, {'condition': 'Random norm-matched control', 'layer': 0, 'position': 4, 'logit_difference': -4.5860595703125, 'p_clean_answer': 3.0882301871315576e-06, 'p_corrupt_answer': 0.0003029772487934679}, {'condition': 'Patched', 'layer': 1, 'position': 3, 'logit_difference': -0.2073974609375, 'p_clean_answer': 0.0001777676079655066, 'p_corrupt_answer': 0.0002187379723181948}, {'condition': 'No-op control', 'layer': 1, 'position': 3, 'logit_difference': -4.726318359375, 'p_clean_answer': 1.722963497741148e-05, 'p_corrupt_answer': 0.0019448678940534592}, {'condition': 'Zero control', 'layer': 1, 'position': 3, 'logit_difference': -4.3712158203125, 'p_clean_answer': 1.618573605810525e-06, 'p_corrupt_answer': 0.0001280935830436647}, {'condition': 'Random norm-matched control', 'layer': 1, 'position': 3, 'logit_difference': -3.488037109375, 'p_clean_answer': 1.1298652680125087e-06, 'p_corrupt_answer': 3.6971057852497324e-05}, {'condition': 'Patched', 'layer': 1, 'position': 4, 'logit_difference': -4.541748046875, 'p_clean_answer': 2.278800093336031e-05, 'p_corrupt_answer': 0.0021387613378465176}, {'condition': 'No-op control', 'layer': 1, 'position': 4, 'logit_difference': -4.726318359375, 'p_clean_answer': 1.722963497741148e-05, 'p_corrupt_answer': 0.0019448678940534592}, {'condition': 'Zero control', 'layer': 1, 'position': 4, 'logit_difference': -6.08984375, 'p_clean_answer': 1.8905900560639566e-06, 'p_corrupt_answer': 0.0008344165980815887}, {'condition': 'Random norm-matched control', 'layer': 1, 'position': 4, 'logit_difference': -4.4884033203125, 'p_clean_answer': 6.413524261006387e-06, 'p_corrupt_answer': 0.0005706706433556974}, {'condition': 'Patched', 'layer': 2, 'position': 3, 'logit_difference': -0.380859375, 'p_clean_answer': 0.00017902175022754818, 'p_corrupt_answer': 0.00026200583670288324}, {'condition': 'No-op control', 'layer': 2, 'position': 3, 'logit_difference': -4.726318359375, 'p_clean_answer': 1.722963497741148e-05, 'p_corrupt_answer': 0.0019448678940534592}, {'condition': 'Zero control', 'layer': 2, 'position': 3, 'logit_difference': -4.1737060546875, 'p_clean_answer': 3.926917088392656e-06, 'p_corrupt_answer': 0.00025507580721750855}, {'condition': 'Random norm-matched control', 'layer': 2, 'position': 3, 'logit_difference': -4.936279296875, 'p_clean_answer': 1.8976224964717403e-06, 'p_corrupt_answer': 0.00026424616225995123}, {'condition': 'Patched', 'layer': 2, 'position': 4, 'logit_difference': -4.4349365234375, 'p_clean_answer': 2.077850695059169e-05, 'p_corrupt_answer': 0.0017525998409837484}, {'condition': 'No-op control', 'layer': 2, 'position': 4, 'logit_difference': -4.726318359375, 'p_clean_answer': 1.722963497741148e-05, 'p_corrupt_answer': 0.0019448678940534592}, {'condition': 'Zero control', 'layer': 2, 'position': 4, 'logit_difference': -4.041015625, 'p_clean_answer': 0.00020550392218865454, 'p_corrupt_answer': 0.01168990321457386}, {'condition': 'Random norm-matched control', 'layer': 2, 'position': 4, 'logit_difference': -3.605712890625, 'p_clean_answer': 1.7486372883013246e-07, 'p_corrupt_answer': 6.436368948925519e-06}, {'condition': 'Patched', 'layer': 3, 'position': 3, 'logit_difference': -0.481201171875, 'p_clean_answer': 0.00020049703016411513, 'p_corrupt_answer': 0.0003244075633119792}, {'condition': 'No-op control', 'layer': 3, 'position': 3, 'logit_difference': -4.726318359375, 'p_clean_answer': 1.722963497741148e-05, 'p_corrupt_answer': 0.0019448678940534592}, {'condition': 'Zero control', 'layer': 3, 'position': 3, 'logit_difference': -4.09765625, 'p_clean_answer': 4.207162419334054e-06, 'p_corrupt_answer': 0.0002532671205699444}, {'condition': 'Random norm-matched control', 'layer': 3, 'position': 3, 'logit_difference': -4.1517333984375, 'p_clean_answer': 5.153463462193031e-06, 'p_corrupt_answer': 0.0003274719638284296}, {'condition': 'Patched', 'layer': 3, 'position': 4, 'logit_difference': -4.2991943359375, 'p_clean_answer': 1.964739931281656e-05, 'p_corrupt_answer': 0.0014468432636931539}, {'condition': 'No-op control', 'layer': 3, 'position': 4, 'logit_difference': -4.726318359375, 'p_clean_answer': 1.722963497741148e-05, 'p_corrupt_answer': 0.0019448678940534592}, {'condition': 'Zero control', 'layer': 3, 'position': 4, 'logit_difference': -3.3099365234375, 'p_clean_answer': 3.3058024655474583e-06, 'p_corrupt_answer': 9.052407403942198e-05}, {'condition': 'Random norm-matched control', 'layer': 3, 'position': 4, 'logit_difference': -3.127197265625, 'p_clean_answer': 7.2276470746146515e-06, 'p_corrupt_answer': 0.00016486232925672084}, {'condition': 'Patched', 'layer': 4, 'position': 3, 'logit_difference': -4.6561279296875, 'p_clean_answer': 1.7825590475695208e-05, 'p_corrupt_answer': 0.0018757483921945095}, {'condition': 'No-op control', 'layer': 4, 'position': 3, 'logit_difference': -4.726318359375, 'p_clean_answer': 1.722963497741148e-05, 'p_corrupt_answer': 0.0019448678940534592}, {'condition': 'Zero control', 'layer': 4, 'position': 3, 'logit_difference': -4.9031982421875, 'p_clean_answer': 1.5437568436027505e-05, 'p_corrupt_answer': 0.002079748548567295}, {'condition': 'Random norm-matched control', 'layer': 4, 'position': 3, 'logit_difference': -5.0360107421875, 'p_clean_answer': 1.3213696547609288e-05, 'p_corrupt_answer': 0.002032993594184518}, {'condition': 'Patched', 'layer': 4, 'position': 4, 'logit_difference': -0.0762939453125, 'p_clean_answer': 0.00022222462575882673, 'p_corrupt_answer': 0.0002398425422143191}, {'condition': 'No-op control', 'layer': 4, 'position': 4, 'logit_difference': -4.726318359375, 'p_clean_answer': 1.722963497741148e-05, 'p_corrupt_answer': 0.0019448678940534592}, {'condition': 'Zero control', 'layer': 4, 'position': 4, 'logit_difference': -2.8394775390625, 'p_clean_answer': 2.4132782527885865e-06, 'p_corrupt_answer': 4.128352884436026e-05}, {'condition': 'Random norm-matched control', 'layer': 4, 'position': 4, 'logit_difference': -4.4090576171875, 'p_clean_answer': 4.190845857010572e-07, 'p_corrupt_answer': 3.4445387427695096e-05}, {'condition': 'Patched', 'layer': 5, 'position': 3, 'logit_difference': -4.726318359375, 'p_clean_answer': 1.722963497741148e-05, 'p_corrupt_answer': 0.0019448678940534592}, {'condition': 'No-op control', 'layer': 5, 'position': 3, 'logit_difference': -4.726318359375, 'p_clean_answer': 1.722963497741148e-05, 'p_corrupt_answer': 0.0019448678940534592}, {'condition': 'Zero control', 'layer': 5, 'position': 3, 'logit_difference': -4.726318359375, 'p_clean_answer': 1.722963497741148e-05, 'p_corrupt_answer': 0.0019448678940534592}, {'condition': 'Random norm-matched control', 'layer': 5, 'position': 3, 'logit_difference': -4.726318359375, 'p_clean_answer': 1.722963497741148e-05, 'p_corrupt_answer': 0.0019448678940534592}, {'condition': 'Patched', 'layer': 5, 'position': 4, 'logit_difference': -0.0050048828125, 'p_clean_answer': 0.0002310975978616625, 'p_corrupt_answer': 0.00023225712357088923}, {'condition': 'No-op control', 'layer': 5, 'position': 4, 'logit_difference': -4.726318359375, 'p_clean_answer': 1.722963497741148e-05, 'p_corrupt_answer': 0.0019448678940534592}, {'condition': 'Zero control', 'layer': 5, 'position': 4, 'logit_difference': -1.751220703125, 'p_clean_answer': 1.0088283488585148e-05, 'p_corrupt_answer': 5.812497693113983e-05}, {'condition': 'Random norm-matched control', 'layer': 5, 'position': 4, 'logit_difference': -10.49072265625, 'p_clean_answer': 2.72739842459549e-24, 'p_corrupt_answer': 9.813219715401338e-20}], 'residual_norms': [7.041056156158447, 9.729270935058594, 10.59091854095459, 11.194535255432129, 15.990078926086426, 91.74433898925781]}]}
    config={'id':'u12-llm','controls':[]}
    return recorded,config

@app.cell
def _(mo):
    mo.md("""
    # Inside a pretrained language model

    **Year 12 · 55 minutes · Recorded Pythia measurements, with an optional live native rerun**

    ## Can restoring an internal state recover a language model's answer preference?

    Explore real activation-patching measurements collected from EleutherAI's Pythia-70M-deduped. Six fixed capital-city prompt pairs keep the comparison inspectable. Candidate probabilities use the entire vocabulary, not just the two answer choices.

    **Browser mode:** Python analyses recorded model runs; moving these controls does not rerun the language model. **Native mode:** download this notebook to repeat the measured experiment on a CPU or your own H100. Neither mode is a reproduction of Goodfire's models or a representative factual-accuracy test.
    """)
    return

@app.cell
def _(mo):
    project_file = mo.ui.file(filetypes=['.json'], multiple=False, max_size=2000000, label='Open a saved Brightlab notebook project')
    mo.vstack([mo.md('**Keep your work:** download a project before leaving. Reopen it here to restore settings, comparisons and writing. Files are read inside this notebook; use fictional classroom data.'),project_file])
    return (project_file,)

@app.cell
def _(project_file, json, config, mo):
    restored = {}
    if project_file.contents():
        try:
            _candidate = json.loads(project_file.contents().decode('utf-8'))
            def _tree(value,depth=0):
                if depth>16:raise ValueError('Project nesting is too deep.')
                if isinstance(value,dict):
                    if len(value)>100:raise ValueError('Too many object fields.')
                    for child in value.values():_tree(child,depth+1)
                elif isinstance(value,list):
                    if len(value)>20000:raise ValueError('Too many rows.')
                    for child in value:_tree(child,depth+1)
                elif isinstance(value,float) and not (-1e100<value<1e100):raise ValueError('Project contains a non-finite or excessive number.')
                elif isinstance(value,str) and len(value)>200000:raise ValueError('Project text is too long.')
            _tree(_candidate)
            if not isinstance(_candidate,dict) or _candidate.get('version')!=1:raise ValueError('Choose a supported version 1 notebook project.')
            if _candidate.get('format') != 'brightlab-notebook-project' or _candidate.get('lesson') != config.get('id',config.get('lesson_id')):
                raise ValueError('Choose a project for this notebook.')
            if not isinstance(_candidate.get('settings'),dict) or not isinstance(_candidate.get('prediction'),str) or not isinstance(_candidate.get('saved_runs',[]),list):
                raise ValueError('The project is missing its settings, prediction or comparison list.')
            for _spec in config['controls']:
                _value=_candidate['settings'].get(_spec['key'],_spec['default'])
                if _spec['kind']=='slider' and (type(_value) not in [int,float] or not _spec['min']<=_value<=_spec['max']):raise ValueError('A saved slider is outside its allowed range.')
                if _spec['kind']=='choice' and (type(_value) is not int or not 0<=_value<len(_spec['options'])):raise ValueError('A saved choice is invalid.')
                if _spec['kind']=='text' and (not isinstance(_value,str) or len(_value)>1000):raise ValueError('A saved text control is invalid.')
            if not isinstance(_candidate.get('conclusion',''),str) or len(_candidate.get('saved_runs',[]))>6:raise ValueError('Invalid writing or comparison count.')
            if len(_candidate.get('prediction',''))>10000 or not isinstance(_candidate.get('conclusion',''),str) or len(_candidate.get('conclusion',''))>20000:raise ValueError('Invalid writing in the saved project.')
            _runs=_candidate.get('saved_runs',[])
            if len(_runs)>6:raise ValueError('A project may contain up to six comparisons.')
            for _run in _runs:
                if not isinstance(_run,dict) or not isinstance(_run.get('settings'),dict):raise ValueError('Invalid saved comparison.')
                _rows=_run.get('measurements',_run.get('result',{}).get('rows') if isinstance(_run.get('result'),dict) else None)
                if not isinstance(_rows,list) or not all(isinstance(row,dict) for row in _rows):raise ValueError('A saved comparison needs a measurement table.')
            restored = _candidate
            mo.output.replace(mo.md('Project read. Save the restored prediction to continue your investigation.'))
        except (ValueError,TypeError,UnicodeError,KeyError,AttributeError) as _error:
            mo.output.replace(mo.callout(mo.md('Could not open this project: '+str(_error)),kind='warn'))
    return (restored,)

@app.cell
def _(mo, restored):
    reset_controls = mo.ui.button(value=0,on_click=lambda count:count+1,label='Reset controls to starting settings')
    return (reset_controls,)

@app.cell
def _(mo, restored):
    prediction=mo.ui.text_area(value=restored.get('prediction',''),label='My prediction — which patch will recover the clean preference?',placeholder='Compare the final token after the last block with an earlier patch.',full_width=True).form(submit_button_label='Save prediction & open the lab',validate=lambda v: None if v and len(v.strip())>=3 else 'Write a testable prediction first.')
    prediction
    return (prediction,)

@app.cell
def _(mo):
    get_attempt,set_attempt=mo.state(None)
    return get_attempt,set_attempt

@app.cell
def _(mo,sys,prediction):
    mo.stop(prediction.value is None)
    native_run=mo.ui.dropdown(options=['CPU','H100 CUDA'],value='CPU',label='Device for a fresh model run').form(submit_button_label='Repeat the model experiment',clear_on_submit=False)
    mo.vstack([mo.md('### Optional: repeat the original measurements\nThe first native run downloads the pinned model weights (roughly 300 MB) from Hugging Face. CPU is sufficient. H100 uses your own runtime and may incur your provider’s charges. Nothing provisions a GPU automatically. The browser copy cannot run PyTorch.'),mo.md('Download the Python notebook and run `uv run marimo edit --sandbox u12-llm.py`, then submit this section. The [H100 setup guide](https://brightlab-ai-creators.ian347727.chatgpt.site/marimo-labs/h100) explains runtime access, device verification and shutdown.') if sys.platform=='emscripten' else native_run])
    return (native_run,)

@app.cell
def _():
    """Pinned Pythia activation patching on fixed, teacher-reviewable prompts.
    CPU works; H100 is optional. No arbitrary text generation or remote model code.
    """
    import time
    from datetime import datetime, timezone
    MODEL_ID='EleutherAI/pythia-70m-deduped'
    PAIRS=[('The capital of France is','The capital of England is',' Paris',' London'),('The capital of Germany is','The capital of Spain is',' Berlin',' Madrid'),('The capital of Japan is','The capital of China is',' Tokyo',' Beijing'),('The capital of Italy is','The capital of Greece is',' Rome',' Athens'),('The capital of Canada is','The capital of Russia is',' Ottawa',' Moscow'),('The capital of Portugal is','The capital of France is',' Lisbon',' Paris')]

    def collect(revision,device='cpu',cache_dir=None):
        import os
        os.environ.setdefault("HF_HUB_DISABLE_XET","1")
        import torch, transformers
        from transformers import AutoTokenizer,AutoModelForCausalLM
        if device=='cuda':
            if not torch.cuda.is_available():raise RuntimeError('CUDA is unavailable; select CPU or attach your GPU runtime.')
            if 'H100' not in torch.cuda.get_device_name(0):raise RuntimeError('The selected GPU is not an H100. Select CPU or attach an H100.')
        torch.manual_seed(51);torch.set_num_threads(2)
        tokenizer=AutoTokenizer.from_pretrained(MODEL_ID,revision=revision,cache_dir=cache_dir,trust_remote_code=False)
        model=AutoModelForCausalLM.from_pretrained(MODEL_ID,revision=revision,cache_dir=cache_dir,trust_remote_code=False,use_safetensors=True,attn_implementation='eager').to(device).eval()
        def hidden(output):return output[0] if isinstance(output,tuple) else output
        def replace(output,new):return (new,)+output[1:] if isinstance(output,tuple) else new
        records=[];start=time.perf_counter()
        with torch.inference_mode():
            for clean,corrupt,a,b in PAIRS:
                x=tokenizer(clean,return_tensors='pt').to(device);y=tokenizer(corrupt,return_tensors='pt').to(device)
                assert x['input_ids'].shape==y['input_ids'].shape,'Unequal token positions require an explicit alignment policy.'
                aid=tokenizer.encode(a,add_special_tokens=False);bid=tokenizer.encode(b,add_special_tokens=False)
                assert len(aid)==len(bid)==1,'This experiment requires single-token answer candidates.'
                aid,bid=aid[0],bid[0];saved={};handles=[]
                for layer,block in enumerate(model.gpt_neox.layers):
                    def capture(module,args,out,index=layer):saved[index]=hidden(out).detach().clone()
                    handles.append(block.register_forward_hook(capture))
                original=model(**x).logits[0,-1]
                for h in handles:h.remove()
                bad_saved={};bad_handles=[]
                for layer,block in enumerate(model.gpt_neox.layers):
                    def capture_bad(module,args,out,index=layer):bad_saved[index]=hidden(out).detach().clone()
                    bad_handles.append(block.register_forward_hook(capture_bad))
                broken=model(**y).logits[0,-1]
                for h in bad_handles:h.remove()
                metric=lambda logits:dict(logit_difference=float((logits[aid]-logits[bid]).item()),p_clean_answer=float(logits.softmax(-1)[aid].item()),p_corrupt_answer=float(logits.softmax(-1)[bid].item()))
                rows=[dict(condition='Clean',layer=-1,position=-1,**metric(original)),dict(condition='Corrupted',layer=-1,position=-1,**metric(broken))]
                positions=sorted(set((x['input_ids'][0]!=y['input_ids'][0]).nonzero().flatten().tolist()+[x['input_ids'].shape[1]-1]))
                for layer,block in enumerate(model.gpt_neox.layers):
                    for position in positions:
                        random_vector=torch.randn_like(saved[layer][:,position,:]);random_vector=random_vector/random_vector.norm()*saved[layer][:,position,:].norm()
                        interventions=[('Patched',saved[layer][:,position,:]),('No-op control',bad_saved[layer][:,position,:]),('Zero control',torch.zeros_like(random_vector)),('Random norm-matched control',random_vector)]
                        for condition,vector in interventions:
                            def patch(module,args,out):
                                new=hidden(out).clone();new[:,position,:]=vector;return replace(out,new)
                            handle=block.register_forward_hook(patch)
                            try:patched=model(**y).logits[0,-1]
                            finally:handle.remove()
                            rows.append(dict(condition=condition,layer=layer,position=position,**metric(patched)))
                records.append(dict(clean=clean,corrupt=corrupt,clean_answer=a,corrupt_answer=b,clean_tokens=tokenizer.convert_ids_to_tokens(x['input_ids'][0].tolist()),corrupt_tokens=tokenizer.convert_ids_to_tokens(y['input_ids'][0].tolist()),positions=positions,rows=rows,residual_norms=[float(saved[i][0,-1].norm().item()) for i in range(len(saved))]))
        if device=='cuda':torch.cuda.synchronize()
        return dict(model=MODEL_ID,revision=revision,torch=torch.__version__,transformers=transformers.__version__,device=torch.cuda.get_device_name(0) if device=='cuda' else 'CPU',inference_seconds=time.perf_counter()-start,collected_at_utc=datetime.now(timezone.utc).isoformat(),precision='float32',method='Clean post-block residual vectors patched into corrupted prompts at changed-token and final-token positions. No weight changes. Controls are interventions, not guaranteed null effects. Candidate probabilities use the full vocabulary denominator.',limits='Six selected capital-city prompt pairs, with no-op, zero and one seeded norm-matched random intervention control in a small pretrained research model. Not a representative accuracy evaluation, a Goodfire replication or a frontier-model result.',records=records)


    return (collect,)

@app.cell
def _(mo,sys,native_run,recorded,collect,set_attempt,Path):
    if sys.platform!='emscripten' and native_run.value is not None:
        try:
            with mo.status.spinner(title='Loading pinned model and collecting matched interventions…'):
                _data=collect(recorded['revision'],'cuda' if native_run.value=='H100 CUDA' else 'cpu',str(Path.home()/'.cache'/'brightlab-models'))
            set_attempt({'data':_data,'error':None})
        except Exception as _error:
            set_attempt({'data':None,'error':str(_error)})
    return

@app.cell
def _(recorded,get_attempt):
    attempt=get_attempt()
    dataset=attempt['data'] if attempt and attempt['data'] else recorded
    origin='Fresh native model run' if attempt and attempt['data'] else 'Recorded '+recorded['device']+' measurements: '+recorded['collected_at_utc']
    return dataset,origin,attempt

@app.cell
def _(mo,dataset,origin,attempt,prediction,restored,reset_controls):
    mo.stop(prediction.value is None,mo.md('Save a prediction to inspect the measurements.'))
    _saved=restored.get('settings',{}) if not reset_controls.value else {}
    _pair=_saved.get('pair',0)
    if type(_pair) is not int or not 0<=_pair<len(dataset['records']):_pair=0
    control=mo.ui.dictionary({'pair':mo.ui.dropdown(options={r['clean']+' / '+r['corrupt']:i for i,r in enumerate(dataset['records'])},value=dataset['records'][_pair]['clean']+' / '+dataset['records'][_pair]['corrupt'],label='Matched prompt pair',full_width=True),'position':mo.ui.dropdown(options={'Changed country token':0,'Final token':1},value='Changed country token' if _saved.get('position')==0 else 'Final token',label='Patched token position'),'metric':mo.ui.dropdown(options=['logit_difference','p_clean_answer','p_corrupt_answer'],value=_saved.get('metric') if _saved.get('metric') in ['logit_difference','p_clean_answer','p_corrupt_answer'] else 'logit_difference',label='Measurement'),'condition':mo.ui.dropdown(options=['Patched','No-op control','Zero control','Random norm-matched control'],value=_saved.get('condition') if _saved.get('condition') in ['Patched','No-op control','Zero control','Random norm-matched control'] else 'Patched',label='Intervention or control')})
    mo.vstack([mo.callout(mo.md('**Current evidence:** '+origin),kind='info'),mo.callout(mo.md('The fresh run failed; recorded evidence remains selected. '+attempt['error']),kind='warn') if attempt and attempt['error'] else mo.md(''),reset_controls,control.vstack()])
    return (control,)

@app.cell
def _(dataset,control):
    record=dataset['records'][control.value['pair']]
    position=record['positions'][control.value['position']]
    rows=[r for r in record['rows'] if r['condition']==control.value['condition'] and r['position']==position]
    reference=record['rows'][:2]
    den=reference[0]['logit_difference']-reference[1]['logit_difference']
    analysis_rows=[{**r,'normalised_recovery':(r['logit_difference']-reference[1]['logit_difference'])/den if abs(den)>1e-6 else None} for r in rows]
    chart={'kind':'line','rows':analysis_rows,'x':'layer','y':control.value['metric']}
    return record,position,analysis_rows,reference,chart

@app.cell
def _(html,np,math):
    def draw(out):
        """Accessible SVG views, with exact values duplicated in tables."""
        esc=lambda s:html.escape(str(s),quote=True)
        marks=[];kind=out['kind'];rows=out.get('chart_rows',out['rows']);x=out['x'];y=out['y']
        if kind=='circle':
            marks.append('<circle cx="290" cy="180" r="130" fill="none" stroke="#778d85" stroke-width="2"/>')
            for i in range(12):
                a=2*math.pi*i/12;label=out.get('labels',[str(i*30)+'°' for i in range(12)])[i]
                marks.append(f'<text x="{290+152*math.cos(a):.1f}" y="{185-152*math.sin(a):.1f}" text-anchor="middle">{esc(label)}</text>')
            points=' '.join(f'{290+130*r[x]:.2f},{180-130*r[y]:.2f}' for r in rows)
            marks.append(f'<polyline points="{points}" fill="none" stroke="#9f3e20" stroke-width="4"/>')
            for i,r in enumerate(rows):marks.append(f'<circle cx="{290+130*r[x]:.2f}" cy="{180-130*r[y]:.2f}" r="4" fill="#9f3e20"><title>Step {i}: ({r[x]:.3f}, {r[y]:.3f})</title></circle>')
            marks.append('<text x="485" y="160">Orange: trajectory</text><text x="485" y="185">Grey: unit circle</text>')
        elif kind=='line':
            values=[float(r[y]) for r in rows];lo=min(0,min(values));hi=max(.01,max(values));px=lambda v:70+(v-float(rows[0][x]))/max(1e-9,float(rows[-1][x])-float(rows[0][x]))*570;py=lambda v:300-(v-lo)/max(1e-9,hi-lo)*245
            points=' '.join(f'{px(float(r[x])):.2f},{py(float(r[y])):.2f}' for r in rows)
            marks.append(f'<polyline points="{points}" fill="none" stroke="#9f3e20" stroke-width="3"/>')
            for r in rows:marks.append(f'<circle cx="{px(float(r[x])):.2f}" cy="{py(float(r[y])):.2f}" r="4" fill="#9f3e20"><title>{esc(x)} {r[x]}: {r[y]:.4f}</title></circle>')
            for frac in [0,.5,1]:
                val=lo+frac*(hi-lo);marks.append(f'<text x="60" y="{py(val)+5}" text-anchor="end">{val:.3g}</text>')
            marks.append(f'<path d="M70 40 V300 H645" fill="none" stroke="#59756b"/><text x="70" y="325">{rows[0][x]}</text><text x="640" y="325" text-anchor="end">{rows[-1][x]}</text><text x="340" y="350" text-anchor="middle">{esc(x)}</text><text x="70" y="25">{esc(y)}</text>')
        else:
            vals=[float(r[y]) for r in rows];lo=min(0,min(vals));hi=max(.001,max(vals));scale=lambda v:200+440*(v-lo)/max(1e-9,hi-lo);zero=scale(0);height=min(46,270/max(1,len(rows)))
            for i,r in enumerate(rows):
                yy=50+i*height;end=scale(float(r[y]));marks.append(f'<text x="187" y="{yy+14}" text-anchor="end">{esc(r[x])}</text><rect x="{min(zero,end):.2f}" y="{yy}" width="{max(.5,abs(end-zero)):.2f}" height="{height*.65}" rx="2" fill="#26735c"/><text x="650" y="{yy+14}">{float(r[y]):.3g}</text>')
            marks.append(f'<text x="200" y="25">{esc(y)}</text>')
        graphic=f'<svg viewBox="0 0 720 365" role="img" aria-label="{esc(y)} chart; exact values in the evidence table" style="width:100%;min-width:560px;background:#f7f8f2;border-radius:12px;font:14px system-ui;fill:#173d36"><title>{esc(y)} chart</title>{"".join(marks)}</svg>'
        if out.get('heat') is not None:
            heat=np.asarray(out['heat']);maxv=max(1e-9,float(np.max(np.abs(heat))));cells=[]
            for i,row in enumerate(heat):
                cells.append(f'<text x="80" y="{44+i*24}" text-anchor="end">{esc(out["heat_labels"][i])}</text>')
                for j,val in enumerate(row):
                    opacity=.1+.9*abs(float(val))/maxv;color='#26735c' if val>=0 else '#a94322';cells.append(f'<rect x="{90+j*32}" y="{27+i*24}" width="29" height="21" fill="{color}" opacity="{opacity:.3f}"><title>Row {i}, column {j}: {val:.5f}</title></rect>')
            for j in range(heat.shape[1]):cells.append(f'<text x="{96+j*32}" y="18">{j}</text>')
            graphic+=f'<svg viewBox="0 0 720 {65+24*len(heat)}" role="img" aria-label="Value heat map. Green positive, rust negative. Exact values in the heat map table." style="width:100%;min-width:560px;font:12px system-ui;fill:#173d36"><title>Heat map with indexed columns</title>{"".join(cells)}</svg>'
        return '<div tabindex="0" role="group" aria-label="Scrollable experiment charts" style="max-width:100%;overflow-x:auto">'+graphic+'</div>'

    return (draw,)

@app.cell
def _(mo,record,position,analysis_rows,reference,chart,draw):
    mo.vstack([mo.md('**Clean prompt:** '+record['clean']+'\n\n**Corrupted prompt:** '+record['corrupt']+'\n\n**Candidate comparison:** `'+record['clean_answer']+'` minus `'+record['corrupt_answer']+'`. Patched position: '+str(position)+'.'),mo.Html(draw(chart)),mo.ui.table(analysis_rows,selection=None),mo.accordion({'Clean and corrupted baselines':mo.ui.table(reference,selection=None),'Token alignment':mo.ui.table([{'position':i,'clean':a,'corrupted':b} for i,(a,b) in enumerate(zip(record['clean_tokens'],record['corrupt_tokens']))],selection=None)})])
    return

@app.cell
def _(mo,restored):
    get_runs,set_runs=mo.state(restored.get('saved_runs',[])[:6])
    return get_runs,set_runs

@app.cell
def _(mo,control,analysis_rows,set_runs):
    _snapshot={'settings':dict(control.value),'measurements':analysis_rows}
    keep=mo.ui.button(label='Save this run for comparison',on_click=lambda _:set_runs(lambda old:(old+[_snapshot])[-6:]))
    clear=mo.ui.button(label='Clear saved comparisons',on_click=lambda _:set_runs([]))
    mo.hstack([keep,clear])
    return

@app.cell
def _(mo,get_runs):
    mo.vstack([mo.md(f'Saved comparisons: {len(get_runs())} / 6. A seventh replaces the oldest.'),mo.accordion({f'Run {i+1}':mo.ui.table(r['measurements'],selection=None) for i,r in enumerate(get_runs())})])
    return

@app.cell
def _(mo,restored):
    reflection=mo.ui.text_area(value=restored.get('conclusion',''),label='My conclusion — intervention, evidence and scope',placeholder='Which positive control worked? What does this experiment leave unresolved?',full_width=True,debounce=False)
    mo.vstack([mo.md('## Challenge the explanation\nCompare changed-token and final-token patches across all six blocks. The post-final-block final-token patch should recover the clean score: it replaces the state read by the output head. Treat that as a positive control, not discovery of a complete circuit. Repeat across all six prompt pairs.\n\n**Make an intervention report:** name the checkpoint, prompt pair, token position, layer, metric and normalised recovery. Recovery can exceed 0–1; a near-zero baseline gap makes it unstable.'),reflection])
    return (reflection,)

@app.cell
def _(mo,json,html,dataset,origin,control,analysis_rows,reference,prediction,reflection,get_runs):
    journal={'format':'brightlab-notebook-project','version':1,'saved_runs':get_runs(),'lesson':'u12-llm','evidence_origin':origin,'prediction':prediction.value,'settings':control.value,'measurements':analysis_rows,'baselines':reference,'conclusion':reflection.value,'provenance':{k:v for k,v in dataset.items() if k!='records'}}
    _e=html.escape
    def _readable(value):
        if isinstance(value,dict):return '<dl>'+''.join('<dt><strong>'+_e(str(k).replace('_',' '))+'</strong></dt><dd>'+_readable(v)+'</dd>' for k,v in value.items())+'</dl>'
        if isinstance(value,list):
            if value and all(isinstance(row,dict) for row in value):
                _keys=list(dict.fromkeys(k for row in value for k in row))
                return '<table><thead><tr>'+''.join('<th>'+_e(str(k).replace('_',' '))+'</th>' for k in _keys)+'</tr></thead><tbody>'+''.join('<tr>'+''.join('<td>'+_readable(row.get(k,''))+'</td>' for k in _keys)+'</tr>' for row in value)+'</tbody></table>'
            return '<ol>'+''.join('<li>'+_readable(v)+'</li>' for v in value)+'</ol>'
        return _e(str(value)) if value is not None else 'Not recorded'
    _report='<html lang="en-AU"><meta charset="utf-8"><title>My Pythia investigation</title><style>body{font:18px/1.6 system-ui;max-width:950px;margin:auto;padding:25px}td,th{border:1px solid #999;padding:8px}table{border-collapse:collapse}</style><h1>My Pythia investigation</h1><p>'+_e(origin)+'</p><h2>Prediction</h2><p>'+_e(prediction.value or '')+'</p><h2>Settings</h2><p>'+_e(str(control.value))+'</p><table>'+''.join('<tr>'+''.join('<td>'+_e(str(v))+'</td>' for v in row.values())+'</tr>' for row in analysis_rows)+'</table><h2>Conclusion</h2><p>'+_e(reflection.value)+'</p><h2>Scope</h2><p>'+_e(dataset['limits'])+'</p></html>'
    _report=_report.replace('</html>','<h2>Baselines</h2>'+_readable(reference)+'<h2>Saved comparisons</h2>'+_readable(get_runs())+'</html>')
    mo.hstack([mo.download(data=json.dumps(journal,indent=2,allow_nan=False).encode(),filename='u12-llm-project.json',label='Download resumable notebook project'),mo.download(data=_report.encode(),filename='u12-llm-report.html',label='Download readable report / print')])
    return

@app.cell
def _(mo,dataset):
    mo.accordion({'Teacher lesson plan':mo.md('**Preparation:** browser runtime access; no personal text, accounts or GPU needed for the recorded-data lesson. Review the six fixed prompt pairs.\n\n**0–8 min:** distinguish a recorded observation from a new intervention run; predict the final-block positive control. **8–25:** compare positions and layers. **25–40:** repeat across prompt pairs and inspect token alignment. **40–55:** write a claim with checkpoint and scope.\n\n**Assessment (0–2 each):** matched comparison; correctly defined logit difference; positive-control interpretation; limited claim. Support with the clean/corrupted baseline table. Extend by repeating the native experiment, then adding one teacher-reviewed pair with equal token lengths and single-token answer candidates. Preserve the original data and reserve new pairs for evaluation.\n\n**Misconceptions:** high probability is not truth; patch recovery does not identify a unique circuit; a small research model does not represent all LLMs. H100 acceleration is unnecessary for this small workload.'),'Glossary and method':mo.md('**Activation:** a measured intermediate tensor. **Residual patch:** replacing an internal vector. **Logit:** an unnormalised token score. **Recovery:** (patched − corrupted)/(clean − corrupted), using the chosen candidate logit difference. **Recorded trace:** values from a previous identified model run, not a live model response.'),'Provenance':mo.json({k:v for k,v in dataset.items() if k!='records'}),'Read the research':mo.md('[Pythia model card](https://huggingface.co/EleutherAI/pythia-70m-deduped) · [Goodfire: Mixing Mechanisms](https://www.goodfire.com/research/mixing-mechanisms) · [Goodfire: Replicating Circuit Tracing](https://www.goodfire.com/research/replicating-circuit-tracing-for-a-simple-mechanism) · [Community circuit research guide](https://www.neuronpedia.org/graph/info)\n\nThese papers motivate controlled interventions; this notebook uses a different model and task. It is independently authored and is not endorsed by Goodfire.')})
    return

if __name__=='__main__':
    app.run()
