[ "any-to-any", "audio-classification", "audio-to-audio", "automatic-speech-recognition", "depth-estimation", "feature-extraction", "fill-mask", "graph-ml", "image-classification", "image-feature-extraction", "image-segmentation", "image-text-to-image", "image-text-to-text", "image-text-to-video", "image-to-3d", "image-to-image", "image-to-text", "image-to-video", "mask-generation", "multiple-choice", "object-detection", "question-answering", "reinforcement-learning", "robotics", "sentence-similarity", "summarization", "table-question-answering", "table-to-text", "tabular-classification", "tabular-regression", "tabular-to-text", "text-classification", "text-generation", "text-ranking", "text-retrieval", "text-to-3d", "text-to-audio", "text-to-image", "text-to-speech", "text-to-video", "time-series-forecasting", "token-classification", "translation", "unconditional-image-generation", "video-classification", "video-text-to-text", "visual-document-retrieval", "visual-question-answering", "voice-activity-detection", "zero-shot-classification", "zero-shot-image-classification", "zero-shot-object-detection" ]