import gradio as gr import pandas as pd # The only data in this Space: the de-identified sample. Raw identifiers were # stripped in the lab and never uploaded - that is the whole point. deid = pd.read_csv("sample_deid.csv") # Recompute k-anonymity at startup (same groupby as the lab). qi = [c for c in ["AGE_BAND", "GENDER", "RACE", "ZIP3"] if c in deid.columns] deid["_k"] = deid.groupby(qi)[qi[0]].transform("size") def show_record(row_index): i = max(0, min(int(float(row_index)), len(deid) - 1)) record = deid.drop(columns=["_k"]).loc[[i]].reset_index(drop=True) k = int(deid.loc[i, "_k"]) verdict = (f"k = {k} -> SAFE (hides in a crowd of {k})" if k >= 5 else f"k = {k} -> AT RISK: uniquely / near-uniquely identifiable") return record, verdict with gr.Blocks() as demo: gr.Markdown("# Patient De-Identification Viewer\n" "Every record below is de-identified. Pick one and check its re-identification risk.") idx = gr.Slider(0, max(len(deid) - 1, 0), value=0, step=1, label="Record (row number)") out = gr.Dataframe(label="De-identified record") kmeter = gr.Textbox(label="Re-identification risk (k-anonymity)") idx.change(fn=show_record, inputs=idx, outputs=[out, kmeter]) demo.load(fn=show_record, inputs=idx, outputs=[out, kmeter]) demo.launch()