Update scores
#2
by dinintagoto - opened
- config/model_performance.jsonl +0 -0
- src/config.py +6 -5
config/model_performance.jsonl
CHANGED
|
The diff for this file is too large to render.
See raw diff
|
|
|
src/config.py
CHANGED
|
@@ -5,7 +5,7 @@ TITLE = """<h1 align="center" id="space-title">GoTo-AI Leaderboard</h1>"""
|
|
| 5 |
|
| 6 |
# Introduction text providing an overview of the leaderboard
|
| 7 |
INTRODUCTION_TEXT = """
|
| 8 |
-
GoTo-AI
|
| 9 |
This leaderboard evaluates general language capabilities of GoTo-AI and other open source models using SEA-HELM and IndoMMLU, focusing on Indonesian, Javanese, Sundanese, Balinese, and Batak.
|
| 10 |
"""
|
| 11 |
|
|
@@ -48,14 +48,15 @@ IndoMMLU covers various subjects and educational levels, including STEM, social
|
|
| 48 |
|
| 49 |
# Explanation of score calculation methodology
|
| 50 |
INFO_SCORE_CALCULATION = """
|
| 51 |
-
- The
|
| 52 |
-
-
|
| 53 |
-
-
|
|
|
|
| 54 |
"""
|
| 55 |
|
| 56 |
# Placeholder information about GoTo and GoTo AI
|
| 57 |
INFO_GOTO_AI = """
|
| 58 |
-
GoTo-AI
|
| 59 |
|
| 60 |
We are supported by research centers and global tech experts such as AI Singapore to train the model to gain general language understanding.
|
| 61 |
|
|
|
|
| 5 |
|
| 6 |
# Introduction text providing an overview of the leaderboard
|
| 7 |
INTRODUCTION_TEXT = """
|
| 8 |
+
GoTo-AI is a collection of large language models which has been pretrained and instruct-tuned for Indonesian language and its various local languages.
|
| 9 |
This leaderboard evaluates general language capabilities of GoTo-AI and other open source models using SEA-HELM and IndoMMLU, focusing on Indonesian, Javanese, Sundanese, Balinese, and Batak.
|
| 10 |
"""
|
| 11 |
|
|
|
|
| 48 |
|
| 49 |
# Explanation of score calculation methodology
|
| 50 |
INFO_SCORE_CALCULATION = """
|
| 51 |
+
- The evaluation runs 8 separate times to capture variations in non-deterministic outputs.
|
| 52 |
+
- 30 bootstrap samples are randomly drawn with replacement from the pooled evaluation data and the scores are averaged to a final score.
|
| 53 |
+
- The overall score for a language is computed as the average of all competency scores. Each competency score is computed as the average of its tasks.
|
| 54 |
+
- Normalization is applied for multi-choice task by substracting the random baseline score and scaling it to the range of 0-100.
|
| 55 |
"""
|
| 56 |
|
| 57 |
# Placeholder information about GoTo and GoTo AI
|
| 58 |
INFO_GOTO_AI = """
|
| 59 |
+
GoTo-AI is a local open source Large Language Model (LLM) ecosystem in Indonesian language, co-initiated by Indonesian tech and telecommunication companies: GoTo Group and Indosat Ooredoo Hutchison. GoTo-AI ecosystem aims to empower Indonesians who want to develop AI-based services and applications using Bahasa Indonesia and its various local languages.
|
| 60 |
|
| 61 |
We are supported by research centers and global tech experts such as AI Singapore to train the model to gain general language understanding.
|
| 62 |
|