{text}{meta}""" + assistant_name + " — Shared conversation
" + page_title + "
" + source + """" Security contract for server-backed conversation shares. The browser is untrusted. Share requests carry structured conversation data and an allowlisted representation id; callers never choose response MIME types or submit rendered HTML for the server to host. """ from __future__ import annotations import hashlib import hmac import html import json import math import re import secrets from typing import Any from urllib.parse import urlsplit, urlunsplit SHARE_SCHEMA_VERSION = "2.1" SHARE_ACCEPTED_SCHEMA_VERSIONS = frozenset({"2.0", "2.1"}) SHARE_FORMATS: dict[str, tuple[str, str]] = { "html": ("text/html; charset=utf-8", ".html"), "json": ("application/json; charset=utf-8", ".json"), "txt": ("text/plain; charset=utf-8", ".txt"), "yaml": ("application/yaml", ".yaml"), "toml": ("application/toml", ".toml"), } def share_artifact_filename(fmt: str) -> str: """ Return the stable human-facing filename for a Global Share artifact. The public read capability is intentionally excluded from the filename. """ fmt = validate_share_format(fmt) _mime, ext = SHARE_FORMATS[fmt] return f"ai-conversation-global-share-{fmt}{ext}" MAX_SHARE_RECORDS = 1000 MAX_SHARE_TEXT_CHARS = 200_000 MAX_SHARE_METADATA_CHARS = 2048 MAX_SHARE_RESOURCES_PER_MESSAGE = 512 MAX_SHARE_RESOURCES_TOTAL = 4096 MAX_SAFE_INTEGER = 9_007_199_254_740_991 _SHARE_ID_RE = re.compile( r"^(?:[0-9a-f]{32}|[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12})$" ) class ShareValidationError(ValueError): """Raised when an untrusted share snapshot violates the public contract.""" def _bounded_string( value: Any, *, limit: int, field: str, nullable: bool = True ) -> str | None: if value is None and nullable: return None if not isinstance(value, str): raise ShareValidationError( f"{field} must be a string" + (" or null" if nullable else "") ) if len(value) > limit: raise ShareValidationError(f"{field} is too long") return value def _bounded_int(value: Any, *, field: str, nullable: bool = True) -> int | None: if value is None and nullable: return None if isinstance(value, bool) or not isinstance(value, int): raise ShareValidationError( f"{field} must be an integer" + (" or null" if nullable else "") ) return value def _safe_scalar(value: Any, *, field: str) -> str | int | float | bool | None: if value is None or isinstance(value, (str, bool, int)): if isinstance(value, str) and len(value) > MAX_SHARE_METADATA_CHARS: raise ShareValidationError(f"{field} is too long") return value if isinstance(value, float) and math.isfinite(value): return value raise ShareValidationError(f"{field} must be a finite primitive value") def sanitize_share_page_url(value: Any) -> str: """Return an HTTP(S) source URL without credentials, query, or fragment.""" if not isinstance(value, str) or not value: return "" if len(value) > 8192: # ruff: ignore[magic-value-comparison] return "" try: parts = urlsplit(value) except ValueError: return "" if parts.scheme.lower() not in {"http", "https"} or not parts.hostname: return "" host = parts.hostname if ":" in host and not host.startswith("["): host = f"[{host}]" try: port = parts.port except ValueError: return "" if port is not None: host = f"{host}:{port}" return urlunsplit((parts.scheme.lower(), host, parts.path or "/", "", "")) _RESOURCE_KINDS = frozenset( { "text", "image", "vector_image", "audio", "video", "data", "file", "replay", "page", "pdf", "archive", } ) _RESOURCE_DELIVERIES = frozenset({"context", "raw", "not_sent"}) _RESOURCE_INTENTS = frozenset({"", "auto", "raw", "extract", "context"}) def _resource_count( value: Any, *, field: str, maximum: int = MAX_SHARE_RESOURCES_PER_MESSAGE ) -> int: if ( isinstance(value, bool) or not isinstance(value, int) or value < 0 or value > maximum ): raise ShareValidationError( f"{field} must be an integer between 0 and {maximum}" ) return value def _resource_bool(value: Any, *, field: str) -> bool: if not isinstance(value, bool): raise ShareValidationError(f"{field} must be a boolean") return value def _safe_relative_path(value: Any, *, field: str) -> str: text = _bounded_string(value, limit=1024, field=field) or "" if not text: return "" text = text.replace("\\", "/") if text.startswith("/") or re.match(r"^[A-Za-z]:", text): return "" parts = [part for part in text.split("/") if part not in {"", "."}] if any( part == ".." # lint or any(ord(ch) < 32 for ch in part) # ruff: ignore[magic-value-comparison] for part in parts ): return "" return "/".join(parts) def _safe_resource_name(value: Any, *, field: str, limit: int = 512) -> str: text = _bounded_string(value, limit=limit, field=field, nullable=False) or "" return ( "".join( ch # lint for ch in text # lint if ord(ch) >= 32 # ruff: ignore[magic-value-comparison] ) .replace("/", "_") .replace("\\", "_")[:limit] ) def _canonical_resource_item( raw: Any, record_index: int, item_index: int, *, allow_source_urls: bool ) -> dict[str, Any]: prefix = f"records[{record_index}].resources.items[{item_index}]" if not isinstance(raw, dict): raise ShareValidationError(f"{prefix} must be an object") kind = _bounded_string( raw.get("kind"), limit=32, field=f"{prefix}.kind", nullable=False ) if kind not in _RESOURCE_KINDS: raise ShareValidationError(f"{prefix}.kind is not allowed") delivery = _bounded_string( raw.get("delivery"), limit=32, field=f"{prefix}.delivery", nullable=False ) if delivery not in _RESOURCE_DELIVERIES: raise ShareValidationError(f"{prefix}.delivery is not allowed") intent = ( _bounded_string(raw.get("intent"), limit=32, field=f"{prefix}.intent") or "" ) if intent not in _RESOURCE_INTENTS: raise ShareValidationError(f"{prefix}.intent is not allowed") size = _bounded_int(raw.get("size"), field=f"{prefix}.size", nullable=False) line_count = _bounded_int( raw.get("lineCount"), field=f"{prefix}.lineCount", nullable=False ) if size is None or size < 0 or size > MAX_SAFE_INTEGER: raise ShareValidationError(f"{prefix}.size is out of range") if ( line_count is None # lint or line_count < 0 # lint or line_count > 1_000_000 # ruff: ignore[magic-value-comparison] ): raise ShareValidationError(f"{prefix}.lineCount is out of range") badge = ( _bounded_string( raw.get("badge"), limit=12, field=f"{prefix}.badge", nullable=False ) or "FILE" ) badge = re.sub(r"[^A-Za-z0-9+._-]", "", badge).upper()[:12] or "FILE" modality = ( _bounded_string(raw.get("modality"), limit=32, field=f"{prefix}.modality") or kind ) modality = re.sub(r"[^A-Za-z0-9_-]", "", modality)[:32] source_kind = ( _bounded_string(raw.get("sourceKind"), limit=24, field=f"{prefix}.sourceKind") or "" ) source_kind = re.sub(r"[^A-Za-z0-9_-]", "", source_kind)[:24] archive_name = ( _bounded_string( raw.get("archiveName"), limit=512, field=f"{prefix}.archiveName" ) or "" ) if archive_name: archive_name = _safe_resource_name(archive_name, field=f"{prefix}.archiveName") item = { "name": _safe_resource_name(raw.get("name"), field=f"{prefix}.name"), "badge": badge, "kind": kind, "size": size, "lineCount": line_count, "included": _resource_bool(raw.get("included"), field=f"{prefix}.included"), "localOnly": _resource_bool(raw.get("localOnly"), field=f"{prefix}.localOnly"), "delivery": delivery, "modality": modality, "intent": intent, "replay": _resource_bool(raw.get("replay"), field=f"{prefix}.replay"), "boundedExcerpt": _resource_bool( raw.get("boundedExcerpt"), field=f"{prefix}.boundedExcerpt" ), "status": ( _bounded_string(raw.get("status"), limit=96, field=f"{prefix}.status") or "" ), "type": ( _bounded_string(raw.get("type"), limit=120, field=f"{prefix}.type") or "" ), "relativePath": _safe_relative_path( raw.get("relativePath"), field=f"{prefix}.relativePath" ), "sourceKind": source_kind, "archiveName": archive_name, } if kind == "page": role = ( _bounded_string( raw.get("contextRole"), limit=16, field=f"{prefix}.contextRole" ) or "pinned" ) item["contextRole"] = "current" if role == "current" else "pinned" item["sourceUrl"] = ( sanitize_share_page_url(raw.get("sourceUrl")) if allow_source_urls else "" ) return item def _resource_aggregates(items: list[dict[str, Any]]) -> dict[str, int]: return { "includedCount": sum(1 for item in items if item["included"]), "localOnlyCount": sum(1 for item in items if item["localOnly"]), "contextCount": sum(1 for item in items if item["delivery"] == "context"), "rawCount": sum(1 for item in items if item["delivery"] == "raw"), "notSentCount": sum(1 for item in items if item["delivery"] == "not_sent"), "pageCount": sum(1 for item in items if item["kind"] == "page"), "replayCount": sum( 1 for item in items if item["kind"] == "replay" or item["replay"] ), "totalBytes": min(MAX_SAFE_INTEGER, sum(item["size"] for item in items)), } def _canonical_resources( raw: Any, record_index: int, *, allow_source_urls: bool ) -> dict[str, Any]: prefix = f"records[{record_index}].resources" if not isinstance(raw, dict): raise ShareValidationError(f"{prefix} must be an object") raw_items = raw.get("items") if not isinstance(raw_items, list): raise ShareValidationError(f"{prefix}.items must be an array") if len(raw_items) > MAX_SHARE_RESOURCES_PER_MESSAGE: raise ShareValidationError(f"{prefix}.items contains too many resources") items = [ _canonical_resource_item( item, record_index, i, allow_source_urls=allow_source_urls ) for i, item in enumerate(raw_items) ] derived = _resource_aggregates(items) total = _resource_count( raw.get("totalCount", len(items)), field=f"{prefix}.totalCount" ) if total < len(items): raise ShareValidationError(f"{prefix}.totalCount cannot be smaller than items") def merged(name: str) -> int: value = _resource_count(raw.get(name, derived[name]), field=f"{prefix}.{name}") return max(derived[name], min(total, value)) total_bytes = _bounded_int( raw.get("totalBytes", derived["totalBytes"]), field=f"{prefix}.totalBytes", nullable=False, ) if total_bytes is None or total_bytes < 0 or total_bytes > MAX_SAFE_INTEGER: raise ShareValidationError(f"{prefix}.totalBytes is out of range") omitted = _resource_count( raw.get("omittedCount", max(0, total - len(items))), field=f"{prefix}.omittedCount", ) omitted = max(omitted, total - len(items)) complete = raw.get("complete") if not isinstance(complete, bool): raise ShareValidationError(f"{prefix}.complete must be a boolean") version = _resource_count( raw.get("version", 2), field=f"{prefix}.version", maximum=2 ) if version != 2: # ruff: ignore[magic-value-comparison] raise ShareValidationError(f"{prefix}.version must be 2") return { "version": 2, "totalCount": total, "includedCount": merged("includedCount"), "localOnlyCount": merged("localOnlyCount"), "contextCount": merged("contextCount"), "rawCount": merged("rawCount"), "notSentCount": merged("notSentCount"), "pageCount": merged("pageCount"), "replayCount": merged("replayCount"), "totalBytes": max(derived["totalBytes"], total_bytes), "itemCount": len(items), "omittedCount": omitted, "complete": complete and omitted == 0 and total == len(items), "items": items, } def _canonical_record( raw: Any, index: int, safe_page: str | None, session_id: str | None ) -> dict[str, Any]: if not isinstance(raw, dict): raise ShareValidationError(f"records[{index}] must be an object") role = raw.get("role") if role not in {"user", "assistant", "error"}: raise ShareValidationError(f"records[{index}].role is not allowed") text = _bounded_string( raw.get("text"), limit=MAX_SHARE_TEXT_CHARS, field=f"records[{index}].text", nullable=False, ) turn_index = _bounded_int( raw.get("turn_index"), field=f"records[{index}].turn_index", nullable=False ) message_index = _bounded_int( raw.get("message_index"), field=f"records[{index}].message_index", nullable=False, ) ts = _bounded_int(raw.get("ts"), field=f"records[{index}].ts") ts_iso = _bounded_string( raw.get("ts_iso"), limit=128, field=f"records[{index}].ts_iso" ) if raw.get("resources") is not None and role != "user": raise ShareValidationError( f"records[{index}].resources is allowed only for user messages" ) return { "turn_index": turn_index, "message_index": message_index, "role": role, "text": text, "ts": ts, "ts_iso": ts_iso, "model_id": _bounded_string( raw.get("model_id"), limit=MAX_SHARE_METADATA_CHARS, field=f"records[{index}].model_id", ), "model_provider": _bounded_string( raw.get("model_provider"), limit=MAX_SHARE_METADATA_CHARS, field=f"records[{index}].model_provider", ), "model_name": _bounded_string( raw.get("model_name"), limit=MAX_SHARE_METADATA_CHARS, field=f"records[{index}].model_name", ), "feedback_rating_value": _safe_scalar( raw.get("feedback_rating_value"), field=f"records[{index}].feedback_rating_value", ), "feedback_rating_label": _bounded_string( raw.get("feedback_rating_label"), limit=MAX_SHARE_METADATA_CHARS, field=f"records[{index}].feedback_rating_label", ), "feedback_message": _bounded_string( raw.get("feedback_message"), limit=MAX_SHARE_TEXT_CHARS, field=f"records[{index}].feedback_message", ), "resources": ( _canonical_resources( raw.get("resources"), index, allow_source_urls=bool(safe_page) ) if raw.get("resources") is not None and role == "user" else None ), # Never trust duplicated per-record identity/page claims from the client; # bind them to the canonical session values reconstructed above. "session_id": session_id, "page_url": safe_page, } def _build_turns(records: list[dict[str, Any]]) -> list[dict[str, Any]]: turns: list[dict[str, Any]] = [] current: dict[str, Any] | None = None for row in records: if row["role"] == "user": current = { "turn_index": row["turn_index"], "user": { "text": row["text"], "ts": row["ts"], "ts_iso": row["ts_iso"], "resources": row.get("resources"), }, "assistant": None, } turns.append(current) elif ( row["role"] == "assistant" and current is not None and current["assistant"] is None ): current["assistant"] = { "text": row["text"], "ts": row["ts"], "ts_iso": row["ts_iso"], "model_id": row["model_id"], "model_provider": row["model_provider"], "model_name": row["model_name"], "feedback_rating_value": row["feedback_rating_value"], "feedback_rating_label": row["feedback_rating_label"], "feedback_message": row["feedback_message"], } return turns def canonicalize_share_snapshot(raw: Any) -> dict[str, Any]: """Validate and reconstruct the allowlisted canonical schema-v2.1 Share snapshot.""" if not isinstance(raw, dict): raise ShareValidationError("snapshot must be an object") if raw.get("schema_version") not in SHARE_ACCEPTED_SCHEMA_VERSIONS: raise ShareValidationError( "snapshot.schema_version must be one of: '2.0', '2.1'" ) raw_session = raw.get("session") if not isinstance(raw_session, dict): raise ShareValidationError("snapshot.session must be an object") session_id = _bounded_string(raw_session.get("id"), limit=256, field="session.id") safe_page = sanitize_share_page_url(raw_session.get("page_url")) or None session = { "id": session_id, "page_url": safe_page, "page_title": _bounded_string( raw_session.get("page_title"), limit=2048, field="session.page_title" ), "assistant_name": ( _bounded_string( raw_session.get("assistant_name"), limit=256, field="session.assistant_name", ) or "AI Assistant" ), "exported_at": _bounded_int( raw_session.get("exported_at"), field="session.exported_at" ), "exported_at_iso": _bounded_string( raw_session.get("exported_at_iso"), limit=128, field="session.exported_at_iso", ), } raw_records = raw.get("records") if not isinstance(raw_records, list) or not raw_records: raise ShareValidationError("snapshot.records must be a non-empty array") if len(raw_records) > MAX_SHARE_RECORDS: raise ShareValidationError("snapshot.records contains too many messages") records = [ _canonical_record(row, i, safe_page, session_id) for i, row in enumerate(raw_records) ] if ( sum((row.get("resources") or {}).get("itemCount", 0) for row in records) > MAX_SHARE_RESOURCES_TOTAL ): raise ShareValidationError( "snapshot.records contains too many resource metadata rows" ) # Never accept caller-supplied turns/unknown root data as trusted. Turns are # a derived view of validated records and are rebuilt server-side. return { "schema_version": SHARE_SCHEMA_VERSION, "session": session, "turns": _build_turns(records), "records": records, } def validate_share_format(value: Any) -> str: if not isinstance(value, str) or value not in SHARE_FORMATS: raise ShareValidationError("format must be one of: html, json, txt, yaml, toml") return value def _resource_summary(manifest: dict[str, Any] | None) -> str: if not manifest or not manifest.get("totalCount"): return "" return ( f"Resources used for this question: {manifest['totalCount']}" f" · context {manifest['contextCount']}" f" · raw {manifest['rawCount']}" f" · not-sent {manifest['notSentCount']}" ) def _resource_text_lines(manifest: dict[str, Any] | None) -> list[str]: if not manifest or not manifest.get("totalCount"): return [] lines = [f"[{_resource_summary(manifest)}]"] for item in manifest.get("items") or []: line = f"- [{item.get('badge') or 'FILE'}] {item.get('name') or 'file'}" if item.get("status"): line += f" — {item['status']}" lines.append(line) if manifest.get("omittedCount"): count = manifest["omittedCount"] lines.append( f"- … {count} resource metadata row{'s' if count != 1 else ''} omitted from restored state" ) return lines def _resource_html(manifest: dict[str, Any] | None) -> str: if not manifest or not manifest.get("totalCount"): return "" rows: list[str] = [] for item in manifest.get("items") or []: status = ( f'{html.escape(str(item.get("status") or ""))}' if item.get("status") else "" ) rows.append( '
Source: {escaped_url}
' messages: list[str] = [] for row in snapshot["records"]: role = row["role"] label = ( "You" if role == "user" else ("Error" if role == "error" else assistant_name) ) text = html.escape(str(row.get("text") or "")) cls = ( "user" if role == "user" else ("error" if role == "error" else "assistant") ) meta_parts: list[str] = [] if row.get("ts_iso"): meta_parts.append(html.escape(str(row["ts_iso"]))) if row.get("model_name"): meta_parts.append(html.escape(str(row["model_name"]))) if row.get("model_provider"): meta_parts.append(html.escape(str(row["model_provider"]))) if ( row.get("feedback_rating_label") or row.get("feedback_rating_value") is not None ): rating = str( row.get("feedback_rating_label") or row.get("feedback_rating_value") ) if row.get("feedback_rating_value") is not None and row.get( "feedback_rating_label" ): rating += f" ({row['feedback_rating_value']})" meta_parts.append("Rating: " + html.escape(rating)) if row.get("feedback_message"): meta_parts.append("Feedback: " + html.escape(str(row["feedback_message"]))) meta = f'' if meta_parts else "" resources = _resource_html(row.get("resources")) if role == "user" else "" messages.append( f'{text}{meta}" + page_title + "
" + source + "Loading shared conversation…