lab02-deid / app.py
ruthlesslearner's picture
Create app.py
7323418 verified
Raw History Blame Contribute Delete
1.34 kB
import gradio as gr
import pandas as pd
# The only data in this Space: the de-identified sample. Raw identifiers were
# stripped in the lab and never uploaded - that is the whole point.
deid = pd.read_csv("sample_deid.csv")
# Recompute k-anonymity at startup (same groupby as the lab).
qi = [c for c in ["AGE_BAND", "GENDER", "RACE", "ZIP3"] if c in deid.columns]
deid["_k"] = deid.groupby(qi)[qi[0]].transform("size")
def show_record(row_index):
i = max(0, min(int(float(row_index)), len(deid) - 1))
record = deid.drop(columns=["_k"]).loc[[i]].reset_index(drop=True)
k = int(deid.loc[i, "_k"])
verdict = (f"k = {k} -> SAFE (hides in a crowd of {k})" if k >= 5
else f"k = {k} -> AT RISK: uniquely / near-uniquely identifiable")
return record, verdict
with gr.Blocks() as demo:
gr.Markdown("# Patient De-Identification Viewer\n"
"Every record below is de-identified. Pick one and check its re-identification risk.")
idx = gr.Slider(0, max(len(deid) - 1, 0), value=0, step=1, label="Record (row number)")
out = gr.Dataframe(label="De-identified record")
kmeter = gr.Textbox(label="Re-identification risk (k-anonymity)")
idx.change(fn=show_record, inputs=idx, outputs=[out, kmeter])
demo.load(fn=show_record, inputs=idx, outputs=[out, kmeter])
demo.launch()