import html import json from pathlib import Path import gradio as gr ROOT = Path(__file__).parent DATA = ROOT / "data" MODEL_REPO = "facebook/meta-encoder" MODEL_URL = f"https://huggingface.co/{MODEL_REPO}" GALLERY = json.load(open(DATA / "gallery.json")) gr.set_static_paths([str(DATA)]) def url(rel): return f"/gradio_api/file={DATA / rel}" def esc(s): return html.escape(str(s)).replace("\n", "
") # ---------------------------------------------------------------- static HTML def header(): return f"""
MetaEncoder

Multimodal System-1 Encoder
Powered by Natural Language

Describe tasks and candidate options using free-form prompts and rich media. One unified model for multimodal decision making and retrieval.

""" def section_head(line1, line2): return f"

{line1}{line2}

" def state_cell(r, state_text): if r["media"]: return f"" if state_text: return f"
{esc(state_text)}
" return "" def prompt_cell(r, query): parts = [f"

{esc(r['task'])}

", f"

{esc(r['source'])}

"] if r["instruction"]: parts.append(f"

{esc(r['instruction'])}

") if query: parts.append(f"

{esc(query)}

") return "".join(parts) def example_row(r, outcome, outcome_label): return row(state_cell(r, r["state"]), prompt_cell(r, r["query"]), outcome, outcome_label) def row(state, prompt, outcome, outcome_label): state_col = f"State{state}" if state else "" return (f"
{state_col}
" f"
Instruction{prompt}
" f"
{outcome_label}{outcome}
" f"
") def table(groups): return "".join(f"

{name}" f"{len(rows)} examples

{''.join(rows)}
" for name, rows in groups if rows) MODALITY_ORDER = {"video": 0, "image": 1} GROUPS = ("Video", "Image", "Text") def modality(r): kinds = [r["media"]["type"]] if r["media"] else [] kinds += [x["type"] for x in r.get("results", [])] return min((MODALITY_ORDER.get(k, 2) for k in kinds), default=2) def by_modality(rows, render): return [(name, [render(r) for r in rows if modality(r) == k]) for k, name in enumerate(GROUPS)] def pct(p): if p >= 0.995: return ">99%" if p < 0.005: return "<1%" return f"{p:.0%}" def option_note(n): return f"All {n} options" if n <= 3 else f"Top 3 of {n:,} options" def option_label(o): detail = f": {esc(o['detail'])}" if o.get("detail") else "" return esc(o["label"]) + detail def decision_row(r): opts = "".join( f"
  • " f"{option_label(o)}{pct(o['prob'])}
  • " for o in r["outcome"]) outcome = (f"" f"

    {option_note(r['n_options'])}

    ") return example_row(r, outcome, "Outcome") def decision_list(): return table(by_modality(GALLERY["decisions"], decision_row)) def retrieval_row(r): if r["results"][0]["type"] == "text": shown = r["results"][:3] res = "
      " + "".join(f"
    1. {esc(x['text'])}
    2. " for x in shown) + "
    " else: shown = r["results"] res = ("
    " + "".join(f"" for x in shown) + "
    ") outcome = f"{res}

    Top {len(shown)} of {r['corpus_size']:,} candidates

    " return example_row(r, outcome, "Top results") def retrieval_list(): return table(by_modality(GALLERY["retrieval"], retrieval_row)) def footer(): return (f"") # ---------------------------------------------------------------- layout THEME = gr.themes.Base( primary_hue="neutral", neutral_hue="neutral", radius_size=gr.themes.sizes.radius_none, font=["-apple-system", "BlinkMacSystemFont", "SF Pro Text", "Helvetica Neue", "Inter", "sans-serif"], ).set(body_background_fill="#ffffff", block_background_fill="transparent", block_border_width="0px", block_shadow="none") FORCE_LIGHT = "() => { document.body.classList.remove('dark'); " \ "document.documentElement.classList.remove('dark'); }" with gr.Blocks(css=open(ROOT / "style.css").read(), theme=THEME, js=FORCE_LIGHT, title="MetaEncoder") as demo: gr.HTML(header()) with gr.Column(elem_classes="page"): with gr.Tabs(elem_classes="plain-tabs main-tabs"): with gr.Tab("Decision making"): gr.HTML(section_head( "Define the task state, instructions, and criteria in text with auxiliary media, " "alongside a set of multimodal candidates.", "MetaEncoder performs instruction-following decision-making.") + decision_list()) with gr.Tab("Retrieval"): gr.HTML(section_head( "Define the query, context, and instructions in text with auxiliary media, " "alongside a pool of multimodal candidates.", "MetaEncoder performs instruction-following retrieval.") + retrieval_list()) gr.HTML(footer()) demo.launch()