Publish epoch-5 Moderation 03, frozen presets and evaluation
Browse files- LICENSE +202 -0
- NOTICE +1 -0
- README.md +104 -0
- app.py +89 -0
- calibration/thresholds.json +78 -0
- categories.py +17 -0
- config.json +46 -0
- evaluation/coverage_audit.json +489 -0
- evaluation/evaluation.json +0 -0
- evaluation/index.html +1 -0
- evaluation/scores_without_text.jsonl +0 -0
- evaluation/thresholds.json +78 -0
- model.safetensors +3 -0
- moderation03.py +82 -0
- moderation_model.py +70 -0
- moderation_policy.py +60 -0
- requirements.txt +5 -0
LICENSE
ADDED
|
@@ -0,0 +1,202 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
Apache License
|
| 3 |
+
Version 2.0, January 2004
|
| 4 |
+
http://www.apache.org/licenses/
|
| 5 |
+
|
| 6 |
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
| 7 |
+
|
| 8 |
+
1. Definitions.
|
| 9 |
+
|
| 10 |
+
"License" shall mean the terms and conditions for use, reproduction,
|
| 11 |
+
and distribution as defined by Sections 1 through 9 of this document.
|
| 12 |
+
|
| 13 |
+
"Licensor" shall mean the copyright owner or entity authorized by
|
| 14 |
+
the copyright owner that is granting the License.
|
| 15 |
+
|
| 16 |
+
"Legal Entity" shall mean the union of the acting entity and all
|
| 17 |
+
other entities that control, are controlled by, or are under common
|
| 18 |
+
control with that entity. For the purposes of this definition,
|
| 19 |
+
"control" means (i) the power, direct or indirect, to cause the
|
| 20 |
+
direction or management of such entity, whether by contract or
|
| 21 |
+
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
| 22 |
+
outstanding shares, or (iii) beneficial ownership of such entity.
|
| 23 |
+
|
| 24 |
+
"You" (or "Your") shall mean an individual or Legal Entity
|
| 25 |
+
exercising permissions granted by this License.
|
| 26 |
+
|
| 27 |
+
"Source" form shall mean the preferred form for making modifications,
|
| 28 |
+
including but not limited to software source code, documentation
|
| 29 |
+
source, and configuration files.
|
| 30 |
+
|
| 31 |
+
"Object" form shall mean any form resulting from mechanical
|
| 32 |
+
transformation or translation of a Source form, including but
|
| 33 |
+
not limited to compiled object code, generated documentation,
|
| 34 |
+
and conversions to other media types.
|
| 35 |
+
|
| 36 |
+
"Work" shall mean the work of authorship, whether in Source or
|
| 37 |
+
Object form, made available under the License, as indicated by a
|
| 38 |
+
copyright notice that is included in or attached to the work
|
| 39 |
+
(an example is provided in the Appendix below).
|
| 40 |
+
|
| 41 |
+
"Derivative Works" shall mean any work, whether in Source or Object
|
| 42 |
+
form, that is based on (or derived from) the Work and for which the
|
| 43 |
+
editorial revisions, annotations, elaborations, or other modifications
|
| 44 |
+
represent, as a whole, an original work of authorship. For the purposes
|
| 45 |
+
of this License, Derivative Works shall not include works that remain
|
| 46 |
+
separable from, or merely link (or bind by name) to the interfaces of,
|
| 47 |
+
the Work and Derivative Works thereof.
|
| 48 |
+
|
| 49 |
+
"Contribution" shall mean any work of authorship, including
|
| 50 |
+
the original version of the Work and any modifications or additions
|
| 51 |
+
to that Work or Derivative Works thereof, that is intentionally
|
| 52 |
+
submitted to Licensor for inclusion in the Work by the copyright owner
|
| 53 |
+
or by an individual or Legal Entity authorized to submit on behalf of
|
| 54 |
+
the copyright owner. For the purposes of this definition, "submitted"
|
| 55 |
+
means any form of electronic, verbal, or written communication sent
|
| 56 |
+
to the Licensor or its representatives, including but not limited to
|
| 57 |
+
communication on electronic mailing lists, source code control systems,
|
| 58 |
+
and issue tracking systems that are managed by, or on behalf of, the
|
| 59 |
+
Licensor for the purpose of discussing and improving the Work, but
|
| 60 |
+
excluding communication that is conspicuously marked or otherwise
|
| 61 |
+
designated in writing by the copyright owner as "Not a Contribution."
|
| 62 |
+
|
| 63 |
+
"Contributor" shall mean Licensor and any individual or Legal Entity
|
| 64 |
+
on behalf of whom a Contribution has been received by Licensor and
|
| 65 |
+
subsequently incorporated within the Work.
|
| 66 |
+
|
| 67 |
+
2. Grant of Copyright License. Subject to the terms and conditions of
|
| 68 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 69 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 70 |
+
copyright license to reproduce, prepare Derivative Works of,
|
| 71 |
+
publicly display, publicly perform, sublicense, and distribute the
|
| 72 |
+
Work and such Derivative Works in Source or Object form.
|
| 73 |
+
|
| 74 |
+
3. Grant of Patent License. Subject to the terms and conditions of
|
| 75 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 76 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 77 |
+
(except as stated in this section) patent license to make, have made,
|
| 78 |
+
use, offer to sell, sell, import, and otherwise transfer the Work,
|
| 79 |
+
where such license applies only to those patent claims licensable
|
| 80 |
+
by such Contributor that are necessarily infringed by their
|
| 81 |
+
Contribution(s) alone or by combination of their Contribution(s)
|
| 82 |
+
with the Work to which such Contribution(s) was submitted. If You
|
| 83 |
+
institute patent litigation against any entity (including a
|
| 84 |
+
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
| 85 |
+
or a Contribution incorporated within the Work constitutes direct
|
| 86 |
+
or contributory patent infringement, then any patent licenses
|
| 87 |
+
granted to You under this License for that Work shall terminate
|
| 88 |
+
as of the date such litigation is filed.
|
| 89 |
+
|
| 90 |
+
4. Redistribution. You may reproduce and distribute copies of the
|
| 91 |
+
Work or Derivative Works thereof in any medium, with or without
|
| 92 |
+
modifications, and in Source or Object form, provided that You
|
| 93 |
+
meet the following conditions:
|
| 94 |
+
|
| 95 |
+
(a) You must give any other recipients of the Work or
|
| 96 |
+
Derivative Works a copy of this License; and
|
| 97 |
+
|
| 98 |
+
(b) You must cause any modified files to carry prominent notices
|
| 99 |
+
stating that You changed the files; and
|
| 100 |
+
|
| 101 |
+
(c) You must retain, in the Source form of any Derivative Works
|
| 102 |
+
that You distribute, all copyright, patent, trademark, and
|
| 103 |
+
attribution notices from the Source form of the Work,
|
| 104 |
+
excluding those notices that do not pertain to any part of
|
| 105 |
+
the Derivative Works; and
|
| 106 |
+
|
| 107 |
+
(d) If the Work includes a "NOTICE" text file as part of its
|
| 108 |
+
distribution, then any Derivative Works that You distribute must
|
| 109 |
+
include a readable copy of the attribution notices contained
|
| 110 |
+
within such NOTICE file, excluding those notices that do not
|
| 111 |
+
pertain to any part of the Derivative Works, in at least one
|
| 112 |
+
of the following places: within a NOTICE text file distributed
|
| 113 |
+
as part of the Derivative Works; within the Source form or
|
| 114 |
+
documentation, if provided along with the Derivative Works; or,
|
| 115 |
+
within a display generated by the Derivative Works, if and
|
| 116 |
+
wherever such third-party notices normally appear. The contents
|
| 117 |
+
of the NOTICE file are for informational purposes only and
|
| 118 |
+
do not modify the License. You may add Your own attribution
|
| 119 |
+
notices within Derivative Works that You distribute, alongside
|
| 120 |
+
or as an addendum to the NOTICE text from the Work, provided
|
| 121 |
+
that such additional attribution notices cannot be construed
|
| 122 |
+
as modifying the License.
|
| 123 |
+
|
| 124 |
+
You may add Your own copyright statement to Your modifications and
|
| 125 |
+
may provide additional or different license terms and conditions
|
| 126 |
+
for use, reproduction, or distribution of Your modifications, or
|
| 127 |
+
for any such Derivative Works as a whole, provided Your use,
|
| 128 |
+
reproduction, and distribution of the Work otherwise complies with
|
| 129 |
+
the conditions stated in this License.
|
| 130 |
+
|
| 131 |
+
5. Submission of Contributions. Unless You explicitly state otherwise,
|
| 132 |
+
any Contribution intentionally submitted for inclusion in the Work
|
| 133 |
+
by You to the Licensor shall be under the terms and conditions of
|
| 134 |
+
this License, without any additional terms or conditions.
|
| 135 |
+
Notwithstanding the above, nothing herein shall supersede or modify
|
| 136 |
+
the terms of any separate license agreement you may have executed
|
| 137 |
+
with Licensor regarding such Contributions.
|
| 138 |
+
|
| 139 |
+
6. Trademarks. This License does not grant permission to use the trade
|
| 140 |
+
names, trademarks, service marks, or product names of the Licensor,
|
| 141 |
+
except as required for reasonable and customary use in describing the
|
| 142 |
+
origin of the Work and reproducing the content of the NOTICE file.
|
| 143 |
+
|
| 144 |
+
7. Disclaimer of Warranty. Unless required by applicable law or
|
| 145 |
+
agreed to in writing, Licensor provides the Work (and each
|
| 146 |
+
Contributor provides its Contributions) on an "AS IS" BASIS,
|
| 147 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
| 148 |
+
implied, including, without limitation, any warranties or conditions
|
| 149 |
+
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
| 150 |
+
PARTICULAR PURPOSE. You are solely responsible for determining the
|
| 151 |
+
appropriateness of using or redistributing the Work and assume any
|
| 152 |
+
risks associated with Your exercise of permissions under this License.
|
| 153 |
+
|
| 154 |
+
8. Limitation of Liability. In no event and under no legal theory,
|
| 155 |
+
whether in tort (including negligence), contract, or otherwise,
|
| 156 |
+
unless required by applicable law (such as deliberate and grossly
|
| 157 |
+
negligent acts) or agreed to in writing, shall any Contributor be
|
| 158 |
+
liable to You for damages, including any direct, indirect, special,
|
| 159 |
+
incidental, or consequential damages of any character arising as a
|
| 160 |
+
result of this License or out of the use or inability to use the
|
| 161 |
+
Work (including but not limited to damages for loss of goodwill,
|
| 162 |
+
work stoppage, computer failure or malfunction, or any and all
|
| 163 |
+
other commercial damages or losses), even if such Contributor
|
| 164 |
+
has been advised of the possibility of such damages.
|
| 165 |
+
|
| 166 |
+
9. Accepting Warranty or Additional Liability. While redistributing
|
| 167 |
+
the Work or Derivative Works thereof, You may choose to offer,
|
| 168 |
+
and charge a fee for, acceptance of support, warranty, indemnity,
|
| 169 |
+
or other liability obligations and/or rights consistent with this
|
| 170 |
+
License. However, in accepting such obligations, You may act only
|
| 171 |
+
on Your own behalf and on Your sole responsibility, not on behalf
|
| 172 |
+
of any other Contributor, and only if You agree to indemnify,
|
| 173 |
+
defend, and hold each Contributor harmless for any liability
|
| 174 |
+
incurred by, or claims asserted against, such Contributor by reason
|
| 175 |
+
of your accepting any such warranty or additional liability.
|
| 176 |
+
|
| 177 |
+
END OF TERMS AND CONDITIONS
|
| 178 |
+
|
| 179 |
+
APPENDIX: How to apply the Apache License to your work.
|
| 180 |
+
|
| 181 |
+
To apply the Apache License to your work, attach the following
|
| 182 |
+
boilerplate notice, with the fields enclosed by brackets "[]"
|
| 183 |
+
replaced with your own identifying information. (Don't include
|
| 184 |
+
the brackets!) The text should be enclosed in the appropriate
|
| 185 |
+
comment syntax for the file format. We also recommend that a
|
| 186 |
+
file or class name and description of purpose be included on the
|
| 187 |
+
same "printed page" as the copyright notice for easier
|
| 188 |
+
identification within third-party archives.
|
| 189 |
+
|
| 190 |
+
Copyright 2026 Alibaba Cloud
|
| 191 |
+
|
| 192 |
+
Licensed under the Apache License, Version 2.0 (the "License");
|
| 193 |
+
you may not use this file except in compliance with the License.
|
| 194 |
+
You may obtain a copy of the License at
|
| 195 |
+
|
| 196 |
+
http://www.apache.org/licenses/LICENSE-2.0
|
| 197 |
+
|
| 198 |
+
Unless required by applicable law or agreed to in writing, software
|
| 199 |
+
distributed under the License is distributed on an "AS IS" BASIS,
|
| 200 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 201 |
+
See the License for the specific language governing permissions and
|
| 202 |
+
limitations under the License.
|
NOTICE
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
Moderation 03 uses Qwen3.5-2B-Base (Qwen Team), distributed separately under Apache-2.0.
|
README.md
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
language:
|
| 3 |
+
- en
|
| 4 |
+
- de
|
| 5 |
+
- fr
|
| 6 |
+
- es
|
| 7 |
+
- it
|
| 8 |
+
- sv
|
| 9 |
+
- fi
|
| 10 |
+
- pl
|
| 11 |
+
- cs
|
| 12 |
+
- lv
|
| 13 |
+
- zh
|
| 14 |
+
- ja
|
| 15 |
+
- ko
|
| 16 |
+
- ru
|
| 17 |
+
- uk
|
| 18 |
+
- be
|
| 19 |
+
- kk
|
| 20 |
+
license: apache-2.0
|
| 21 |
+
pipeline_tag: text-classification
|
| 22 |
+
library_name: pytorch
|
| 23 |
+
base_model: Qwen/Qwen3.5-2B-Base
|
| 24 |
+
tags:
|
| 25 |
+
- moderation
|
| 26 |
+
- multilingual
|
| 27 |
+
- custom-code
|
| 28 |
+
---
|
| 29 |
+
|
| 30 |
+
# Moderation 03
|
| 31 |
+
|
| 32 |
+
Frozen **Qwen3.5-2B-Base** text features → **254,564,363-parameter transformer head** → **11 raw category scores**.
|
| 33 |
+
Supports 17 languages: en, de, fr, es, it, sv, fi, pl, cs, lv, zh, ja, ko, ru, uk, be, kk. This is the selected epoch-5 checkpoint.
|
| 34 |
+
|
| 35 |
+
[Interactive Gradio demo](https://huggingface.co/spaces/ifmain/Moderation-03) · [Full evaluation report](https://huggingface.co/spaces/ifmain/Moderation-03-report) · [Model weights](https://huggingface.co/ifmain/Moderation-03) · [Source](https://github.com/ifmain/Moderation-03)
|
| 36 |
+
|
| 37 |
+
## Outputs and controls
|
| 38 |
+
|
| 39 |
+
The raw sigmoid scores are **not calibrated probabilities**. They remain unchanged when selecting a preset.
|
| 40 |
+
Categories: `harassment`, `harassment_threatening`, `hate`, `hate_threatening`, `self_harm`, `self_harm_instructions`, `self_harm_intent`, `sexual`, `sexual_minors`, `violence`, `violence_graphic`.
|
| 41 |
+
|
| 42 |
+
Choose raw scores, one of four presets (`light`, `medium`, `high`, `corporate`), or custom per-category thresholds. A `null` threshold disables a category. Enabled categories trigger when `score >= threshold`; any trigger blocks the text. High and corporate presets also apply a finite multilingual profanity lexicon. Custom thresholds can change each category independently and optionally enable that lexicon. Raw mode makes no allow/block decision.
|
| 43 |
+
|
| 44 |
+
Corporate is the strictest **content-threshold** preset. The project owner also intends to restrict non-work discussions in workplace deployments, but this release has **no workplace-topic relevance detector**. A benign statement about equal rights is not a hate ground-truth label merely because an organization chooses to restrict that topic. The published evaluation retains its original content-policy labels; it does not measure a separate work-topic policy.
|
| 45 |
+
|
| 46 |
+
## Run
|
| 47 |
+
|
| 48 |
+
Clone the GitHub repository, install `requirements.txt`, and run `python app.py`. Head weights and the pinned Qwen backbone download automatically. CUDA is used when available; CPU is supported but slower. The first request loads the model. Hosted demo deployments use free CPU hardware by default; no paid hardware is required by this package.
|
| 49 |
+
|
| 50 |
+
```python
|
| 51 |
+
from moderation03 import Moderation03
|
| 52 |
+
|
| 53 |
+
model = Moderation03("ifmain/Moderation-03")
|
| 54 |
+
raw = model.predict("Hello, thanks for your help.", language="en")
|
| 55 |
+
medium = model.predict("Hello, thanks for your help.", language="en", preset="medium")
|
| 56 |
+
custom = dict.fromkeys(model.config["categories"], None)
|
| 57 |
+
custom["harassment"] = 0.4
|
| 58 |
+
result = model.predict("Hello, thanks for your help.", language="en", thresholds=custom)
|
| 59 |
+
print(result["raw_scores"], result["policy"])
|
| 60 |
+
```
|
| 61 |
+
|
| 62 |
+
`MODEL_ID` may name a local model package or Hub repository. `BACKBONE_PATH` optionally points to a local Qwen snapshot for the app. `DEVICE=cpu` or `cuda` overrides automatic device selection. Local app defaults to `127.0.0.1:7860`; Spaces binds port 7860 publicly. The app does not write submitted text to a dataset. Hugging Face/Gradio hosting has its own service policies.
|
| 63 |
+
|
| 64 |
+
## Architecture and provenance
|
| 65 |
+
|
| 66 |
+
Backbone: `Qwen/Qwen3.5-2B-Base`, pinned revision `b1485b2fa6dfa1287294f269f5fb618e03d52d7c`.
|
| 67 |
+
Head: input size 2048, width 1024, 20 transformer encoder blocks, 16 attention heads, feed-forward size 4096, masked mean pooling, 11 classifiers. Token limit: **512**, no chat template. Longer texts are truncated and the API returns `truncated=true`.
|
| 68 |
+
Published `model.safetensors` is the **FP32 head**, SHA256 `4218fb73825ff2fbbdd897a4346dd67141b15fc63d1811413929dddb88d408d3`; it is not the full Qwen model. BF16 inference/autocast follows the evaluated pipeline. The rounded BF16 head export is intentionally not substituted for the evaluated weights.
|
| 69 |
+
|
| 70 |
+
Training: 100,000 examples, 70% clean / 30% character-level augmentation, 17 languages; 2,000 separate validation examples. Selected checkpoint: epoch 5, validation BCE 0.0967052458. Training data derives from [ifmain/text-moderation-02-multilingual](https://huggingface.co/datasets/ifmain/text-moderation-02-multilingual). Character substitutions and inserted separators do not establish robustness to semantic attacks. Source language assignment included uncertain groups.
|
| 71 |
+
|
| 72 |
+
## Calibration and evaluation
|
| 73 |
+
|
| 74 |
+
Internal threshold fitting: **828 rows / 638 source families**. Internal audit: **412 rows / 268 families**. All 11 categories have positive reference examples in all 17 languages in both partitions; translations are correlated, not independent examples. Most labels are inherited teacher estimates, not independently reviewed human ground truth. Thresholds minimize group-weighted balanced error on the fit partition, with nested preset constraints. This is threshold calibration, not probability calibration; no globally optimal thresholds are claimed.
|
| 75 |
+
|
| 76 |
+
Internal held-out content-policy results:
|
| 77 |
+
|
| 78 |
+
| Preset | TP | FP | TN | FN | Precision | Recall |
|
| 79 |
+
|---|---:|---:|---:|---:|---:|---:|
|
| 80 |
+
| Light | 109 | 35 | 244 | 24 | 75.7% | 82.0% |
|
| 81 |
+
| Medium | 189 | 51 | 146 | 26 | 78.8% | 87.9% |
|
| 82 |
+
| High | 245 | 24 | 121 | 22 | 91.1% | 91.8% |
|
| 83 |
+
| Corporate | 252 | 22 | 115 | 23 | 92.0% | 91.6% |
|
| 84 |
+
|
| 85 |
+
External evaluation: [mmathys/openai-moderation-api-evaluation](https://huggingface.co/datasets/mmathys/openai-moderation-api-evaluation), revision `84e5cf3bcd6acb3dfc70b6760451645872218a3e`, **1,680 English examples**. Unknown labels are excluded, not treated as negatives. No thresholds were fitted on this dataset. Exact normalized overlaps with used groups, including all their translations: zero; arbitrary paraphrase/pretraining contamination is not excluded. 76 external texts exceed 512 tokens and were truncated.
|
| 86 |
+
|
| 87 |
+
External projections onto available benchmark categories (not full four-policy ground truth; profanity and three unsupported heads excluded):
|
| 88 |
+
|
| 89 |
+
| Preset | Evaluable rows | TP | FP | TN | FN |
|
| 90 |
+
|---|---:|---:|---:|---:|---:|
|
| 91 |
+
| Light | 800 | 202 | 210 | 323 | 65 |
|
| 92 |
+
| Medium | 859 | 493 | 196 | 141 | 29 |
|
| 93 |
+
| High | 859 | 504 | 220 | 117 | 18 |
|
| 94 |
+
| Corporate | 859 | 505 | 222 | 115 | 17 |
|
| 95 |
+
|
| 96 |
+
## Known failures
|
| 97 |
+
|
| 98 |
+
An additional 265 synthetic checks cover adult sexual content, fraud assistance, disturbing imagery, hate-related requests and violence, plus benign controls, typography changes and 10 English/Russian paraphrases. They are not independently human-reviewed and do not reproduce the researchers' undisclosed benchmark. Corporate misses 29 of 36 fraud-request variants; under the evaluated content policy it also blocks 16 of 17 benign equal-rights messages. There is no dedicated illegal-activity head; broad disturbing content is not equivalent to graphic violence. Lower thresholds introduce false positives. The reported ModerationBERT-En transformed-prompt vulnerability is **not established as fixed** by this release.
|
| 99 |
+
|
| 100 |
+
The detailed report contains category/language metrics and the frozen thresholds. Public evaluation outputs contain row IDs and scores, **not source text**. Training data, source prompts, credentials, hidden-state caches and optimizer states are not distributed in this repository.
|
| 101 |
+
|
| 102 |
+
## License
|
| 103 |
+
|
| 104 |
+
Apache-2.0 for released code and head weights. The backbone is downloaded separately under its own Apache-2.0 license. External evaluation provenance and definitions: [OpenAI moderation-api-release](https://github.com/openai/moderation-api-release).
|
app.py
ADDED
|
@@ -0,0 +1,89 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Gradio demo: raw outputs, frozen presets, and independent custom category thresholds."""
|
| 2 |
+
import json
|
| 3 |
+
import os
|
| 4 |
+
from pathlib import Path
|
| 5 |
+
from threading import Lock
|
| 6 |
+
os.environ.setdefault('GRADIO_ANALYTICS_ENABLED','False')
|
| 7 |
+
import gradio as gr
|
| 8 |
+
from moderation03 import Moderation03,apply_policy,MODEL_ID
|
| 9 |
+
from moderation_policy import LANGS,PROFILES
|
| 10 |
+
from categories import CATEGORIES
|
| 11 |
+
|
| 12 |
+
ROOT=Path(__file__).resolve().parent
|
| 13 |
+
CALIBRATION=json.loads((ROOT/'calibration/thresholds.json').read_text(encoding='utf-8'))
|
| 14 |
+
MODEL=None
|
| 15 |
+
LOCK=Lock()
|
| 16 |
+
LABELS={'light':'Light / Лёгкий','medium':'Medium / Средний','high':'High / Высокий','corporate':'Corporate / Корпоративный'}
|
| 17 |
+
|
| 18 |
+
def settings(preset):
|
| 19 |
+
values=[]
|
| 20 |
+
for c in CATEGORIES:
|
| 21 |
+
t=CALIBRATION['thresholds'][preset][c]
|
| 22 |
+
values.extend([t is not None,round(t,6) if t is not None else .5])
|
| 23 |
+
return [preset in ['high','corporate'],*values]
|
| 24 |
+
|
| 25 |
+
def analyze(text,language,mode,preset,lexical,*controls):
|
| 26 |
+
global MODEL
|
| 27 |
+
if not text or not text.strip():raise gr.Error('Enter text / Введите текст')
|
| 28 |
+
if len(text)>50000:raise gr.Error('Maximum 50,000 characters / Максимум 50 000 символов')
|
| 29 |
+
with LOCK:
|
| 30 |
+
if MODEL is None:
|
| 31 |
+
MODEL=Moderation03(os.getenv('MODEL_ID',MODEL_ID),device=os.getenv('DEVICE') or None,
|
| 32 |
+
backbone_path=os.getenv('BACKBONE_PATH') or None)
|
| 33 |
+
result=MODEL.scores(text)
|
| 34 |
+
thresholds=None;block_profanity=False
|
| 35 |
+
if mode=='preset':
|
| 36 |
+
thresholds=CALIBRATION['thresholds'][preset];block_profanity=preset in ['high','corporate']
|
| 37 |
+
elif mode=='custom':
|
| 38 |
+
thresholds={c:float(controls[2*i+1]) if controls[2*i] else None for i,c in enumerate(CATEGORIES)}
|
| 39 |
+
block_profanity=bool(lexical)
|
| 40 |
+
policy=None if thresholds is None else apply_policy(result['raw_scores'],thresholds,
|
| 41 |
+
block_profanity=block_profanity,text=text,language=language)
|
| 42 |
+
result.update(mode=mode,preset=preset if mode=='preset' else None,language=language,policy=policy,
|
| 43 |
+
all_presets={p:apply_policy(result['raw_scores'],CALIBRATION['thresholds'][p],
|
| 44 |
+
block_profanity=p in ['high','corporate'],text=text,language=language) for p in PROFILES})
|
| 45 |
+
summary='**Raw scores / Сырые оценки**' if policy is None else (
|
| 46 |
+
'**BLOCK / Блокировать**' if policy['block'] else '**ALLOW / Пропустить**')
|
| 47 |
+
if policy and policy['reasons']:summary+='\n\nTriggered / Сработали: '+', '.join(policy['reasons'])
|
| 48 |
+
if result['truncated']:summary+='\n\n⚠ Input truncated to 512 tokens / Текст обрезан до 512 токенов.'
|
| 49 |
+
table=[[c,result['raw_scores'][c],None if thresholds is None else thresholds[c],
|
| 50 |
+
'—' if policy is None else ('BLOCK' if policy['tags'][c] else '—')] for c in CATEGORIES]
|
| 51 |
+
comparison=[[LABELS[p],result['all_presets'][p]['block'],', '.join(result['all_presets'][p]['reasons'])] for p in PROFILES]
|
| 52 |
+
return summary,table,comparison,result
|
| 53 |
+
|
| 54 |
+
def build_app():
|
| 55 |
+
with gr.Blocks(title='Moderation 03',analytics_enabled=False) as demo:
|
| 56 |
+
gr.Markdown('# Moderation 03\n17 languages · 11 raw scores · 4 presets · custom thresholds')
|
| 57 |
+
with gr.Row():
|
| 58 |
+
with gr.Column():
|
| 59 |
+
text=gr.Textbox(label='Text / Текст',lines=7,max_lines=14)
|
| 60 |
+
language=gr.Dropdown(LANGS,value='ru',label='Language / Язык (for the profanity rule)')
|
| 61 |
+
mode=gr.Radio([('Raw scores / Сырые оценки','raw'),('Preset / Пресет','preset'),
|
| 62 |
+
('Custom thresholds / Свои пороги','custom')],value='preset',label='Mode / Режим')
|
| 63 |
+
preset=gr.Dropdown([(LABELS[p],p) for p in PROFILES],value='medium',label='Preset / Пресет')
|
| 64 |
+
button=gr.Button('Analyze / Проверить',variant='primary')
|
| 65 |
+
with gr.Column():
|
| 66 |
+
summary=gr.Markdown('Enter text and choose a mode / Введите текст и выберите режим.')
|
| 67 |
+
table=gr.Dataframe(headers=['Category','Raw score','Threshold','Tag'],datatype=['str','number','number','str'],interactive=False)
|
| 68 |
+
comparison=gr.Dataframe(headers=['Preset','Blocked','Reasons'],datatype=['str','bool','str'],interactive=False,label='All presets / Все пресеты')
|
| 69 |
+
with gr.Accordion('Custom thresholds / Индивидуальные пороги',open=False):
|
| 70 |
+
gr.Markdown('Editing these controls switches to custom mode. Disable a category with its checkbox. / Изменение включает режим своих порогов. Снимите галочку, чтобы отключить категорию.')
|
| 71 |
+
lexical=gr.Checkbox(value=False,label='Block finite profanity lexicon / Блокировать мат по словарю')
|
| 72 |
+
controls=[];defaults=settings('medium')[1:]
|
| 73 |
+
for i,c in enumerate(CATEGORIES):
|
| 74 |
+
with gr.Row():
|
| 75 |
+
enabled=gr.Checkbox(value=defaults[2*i],label=c)
|
| 76 |
+
threshold=gr.Slider(0,1,value=defaults[2*i+1],step=.001,label='Threshold / Порог')
|
| 77 |
+
controls.extend([enabled,threshold])
|
| 78 |
+
for control in [lexical,*controls]:control.input(lambda:'custom',outputs=mode,queue=False)
|
| 79 |
+
preset.change(settings,inputs=preset,outputs=[lexical,*controls],queue=False)
|
| 80 |
+
details=gr.JSON(label='Raw scores and decisions / Оценки и решения JSON')
|
| 81 |
+
button.click(analyze,inputs=[text,language,mode,preset,lexical,*controls],outputs=[summary,table,comparison,details],
|
| 82 |
+
concurrency_limit=1,api_name='moderate')
|
| 83 |
+
gr.Markdown('Scores are not calibrated probabilities. Presets are content thresholds, not a work-topic classifier. Corporate does not reliably determine whether a topic belongs in a particular workplace. / Оценки не являются калиброванными вероятностями. Пресеты не определяют, относится ли тема к работе.')
|
| 84 |
+
gr.Markdown('[Model](https://huggingface.co/ifmain/Moderation-03) · [Evaluation report](https://huggingface.co/spaces/ifmain/Moderation-03-report) · [GitHub](https://github.com/ifmain/Moderation-03)\n\nEvaluation found false positives and missed fraud requests; the reported researcher vulnerability is not established as fixed. The app does not save submitted text to a dataset.')
|
| 85 |
+
return demo.queue(max_size=8)
|
| 86 |
+
|
| 87 |
+
if __name__=='__main__':
|
| 88 |
+
build_app().launch(server_name=os.getenv('GRADIO_SERVER_NAME','0.0.0.0' if os.getenv('SPACE_ID') else '127.0.0.1'),
|
| 89 |
+
server_port=int(os.getenv('PORT','7860')),share=False)
|
calibration/thresholds.json
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"status": "fitted_internal_only",
|
| 3 |
+
"method": "group-weighted balanced error threshold search; nested profiles",
|
| 4 |
+
"fit_rows": 828,
|
| 5 |
+
"fit_groups": 638,
|
| 6 |
+
"thresholds": {
|
| 7 |
+
"light": {
|
| 8 |
+
"harassment": null,
|
| 9 |
+
"harassment_threatening": 0.12,
|
| 10 |
+
"hate": 0.26,
|
| 11 |
+
"hate_threatening": 0.01,
|
| 12 |
+
"self_harm": null,
|
| 13 |
+
"self_harm_instructions": 0.23,
|
| 14 |
+
"self_harm_intent": null,
|
| 15 |
+
"sexual": null,
|
| 16 |
+
"sexual_minors": 0.02,
|
| 17 |
+
"violence": null,
|
| 18 |
+
"violence_graphic": 0.02
|
| 19 |
+
},
|
| 20 |
+
"medium": {
|
| 21 |
+
"harassment": 0.45,
|
| 22 |
+
"harassment_threatening": 0.03,
|
| 23 |
+
"hate": 0.060000000000000005,
|
| 24 |
+
"hate_threatening": 0.01,
|
| 25 |
+
"self_harm": 0.12,
|
| 26 |
+
"self_harm_instructions": 0.01,
|
| 27 |
+
"self_harm_intent": 0.02,
|
| 28 |
+
"sexual": 0.59,
|
| 29 |
+
"sexual_minors": 0.01,
|
| 30 |
+
"violence": 0.09999999999999999,
|
| 31 |
+
"violence_graphic": 0.02
|
| 32 |
+
},
|
| 33 |
+
"high": {
|
| 34 |
+
"harassment": 0.24000000000000002,
|
| 35 |
+
"harassment_threatening": 0.03,
|
| 36 |
+
"hate": 0.05,
|
| 37 |
+
"hate_threatening": 0.01,
|
| 38 |
+
"self_harm": 0.03,
|
| 39 |
+
"self_harm_instructions": 0.01,
|
| 40 |
+
"self_harm_intent": 0.02,
|
| 41 |
+
"sexual": 0.34,
|
| 42 |
+
"sexual_minors": 0.01,
|
| 43 |
+
"violence": 0.09999999999999999,
|
| 44 |
+
"violence_graphic": 0.02
|
| 45 |
+
},
|
| 46 |
+
"corporate": {
|
| 47 |
+
"harassment": 0.24000000000000002,
|
| 48 |
+
"harassment_threatening": 0.01,
|
| 49 |
+
"hate": 0.05,
|
| 50 |
+
"hate_threatening": 0.01,
|
| 51 |
+
"self_harm": 0.02,
|
| 52 |
+
"self_harm_instructions": 0.01,
|
| 53 |
+
"self_harm_intent": 0.01,
|
| 54 |
+
"sexual": 0.29000000000000004,
|
| 55 |
+
"sexual_minors": 0.01,
|
| 56 |
+
"violence": 0.09,
|
| 57 |
+
"violence_graphic": 0.02
|
| 58 |
+
}
|
| 59 |
+
},
|
| 60 |
+
"standard_category_thresholds": {
|
| 61 |
+
"harassment": 0.25,
|
| 62 |
+
"harassment_threatening": 0.02,
|
| 63 |
+
"hate": 0.03,
|
| 64 |
+
"hate_threatening": 0.01,
|
| 65 |
+
"self_harm": 0.01,
|
| 66 |
+
"self_harm_instructions": 0.01,
|
| 67 |
+
"self_harm_intent": 0.01,
|
| 68 |
+
"sexual": 0.11,
|
| 69 |
+
"sexual_minors": 0.01,
|
| 70 |
+
"violence": 0.09,
|
| 71 |
+
"violence_graphic": 0.02
|
| 72 |
+
},
|
| 73 |
+
"probabilities_calibrated": false,
|
| 74 |
+
"checkpoint_sha256": "4218fb73825ff2fbbdd897a4346dd67141b15fc63d1811413929dddb88d408d3",
|
| 75 |
+
"fit_ids_sha256": "852388954817cf5ef3047eae79e8a1480cc60a1311673e439a02fbb9511cca23",
|
| 76 |
+
"policy": "Reference four-profile policy plus finite profanity rule. Research intent checks scored separately.",
|
| 77 |
+
"limitations": "Weak teacher-derived and synthetic labels. No new detector heads were trained."
|
| 78 |
+
}
|
categories.py
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import re, json
|
| 2 |
+
|
| 3 |
+
CATEGORIES = 'harassment harassment_threatening hate hate_threatening self_harm self_harm_instructions self_harm_intent sexual sexual_minors violence violence_graphic'.split()
|
| 4 |
+
|
| 5 |
+
def normalize(scores):
|
| 6 |
+
result = {}
|
| 7 |
+
for name, value in scores.items():
|
| 8 |
+
key = re.sub(r'[^a-z0-9]+', '_', name.lower()).strip('_')
|
| 9 |
+
if key in result and result[key] != value:
|
| 10 |
+
raise ValueError(f'Conflicting category aliases: {name}')
|
| 11 |
+
result[key] = value
|
| 12 |
+
if set(result) != set(CATEGORIES):
|
| 13 |
+
raise ValueError(f'Unexpected category set: {sorted(result)}')
|
| 14 |
+
return {name: result[name] for name in CATEGORIES}
|
| 15 |
+
|
| 16 |
+
def canonical(scores):
|
| 17 |
+
return json.dumps(normalize(scores), sort_keys=True, separators=(',', ':'), allow_nan=False)
|
config.json
ADDED
|
@@ -0,0 +1,46 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"backbone": "Qwen/Qwen3.5-2B-Base",
|
| 3 |
+
"backbone_revision": "b1485b2fa6dfa1287294f269f5fb618e03d52d7c",
|
| 4 |
+
"input_size": 2048,
|
| 5 |
+
"hidden_size": 1024,
|
| 6 |
+
"num_layers": 20,
|
| 7 |
+
"num_heads": 16,
|
| 8 |
+
"intermediate_size": 4096,
|
| 9 |
+
"max_length": 512,
|
| 10 |
+
"dropout": 0.1,
|
| 11 |
+
"gradient_checkpointing": true,
|
| 12 |
+
"pooling": "masked_mean",
|
| 13 |
+
"categories": [
|
| 14 |
+
"harassment",
|
| 15 |
+
"harassment_threatening",
|
| 16 |
+
"hate",
|
| 17 |
+
"hate_threatening",
|
| 18 |
+
"self_harm",
|
| 19 |
+
"self_harm_instructions",
|
| 20 |
+
"self_harm_intent",
|
| 21 |
+
"sexual",
|
| 22 |
+
"sexual_minors",
|
| 23 |
+
"violence",
|
| 24 |
+
"violence_graphic"
|
| 25 |
+
],
|
| 26 |
+
"training": {
|
| 27 |
+
"epochs": 10,
|
| 28 |
+
"batch_size": 4,
|
| 29 |
+
"gradient_accumulation": 8,
|
| 30 |
+
"learning_rate": 0.0001,
|
| 31 |
+
"weight_decay": 0.01,
|
| 32 |
+
"warmup_fraction": 0.05,
|
| 33 |
+
"max_grad_norm": 1.0,
|
| 34 |
+
"early_stopping_patience": 3,
|
| 35 |
+
"dtype": "bfloat16",
|
| 36 |
+
"seed": 20261003,
|
| 37 |
+
"loss": "BCEWithLogitsLoss with original soft scores",
|
| 38 |
+
"cuda_memory_fraction": 0.40875889995586007
|
| 39 |
+
},
|
| 40 |
+
"calibration": {
|
| 41 |
+
"status": "fitted_internal_only",
|
| 42 |
+
"file": "calibration/thresholds.json",
|
| 43 |
+
"probabilities_calibrated": false
|
| 44 |
+
},
|
| 45 |
+
"checkpoint_epoch": 5
|
| 46 |
+
}
|
evaluation/coverage_audit.json
ADDED
|
@@ -0,0 +1,489 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"status": "coverage_expanded_ready_for_measurement",
|
| 3 |
+
"internal_rows": 1240,
|
| 4 |
+
"splits": {
|
| 5 |
+
"fit": 828,
|
| 6 |
+
"audit": 412
|
| 7 |
+
},
|
| 8 |
+
"challenge_rows": 265,
|
| 9 |
+
"category_language_positive_counts_at_reference_0_5": {
|
| 10 |
+
"fit": {
|
| 11 |
+
"en": {
|
| 12 |
+
"harassment": 10,
|
| 13 |
+
"harassment_threatening": 3,
|
| 14 |
+
"hate": 3,
|
| 15 |
+
"hate_threatening": 1,
|
| 16 |
+
"self_harm": 3,
|
| 17 |
+
"self_harm_instructions": 2,
|
| 18 |
+
"self_harm_intent": 3,
|
| 19 |
+
"sexual": 6,
|
| 20 |
+
"sexual_minors": 4,
|
| 21 |
+
"violence": 11,
|
| 22 |
+
"violence_graphic": 5
|
| 23 |
+
},
|
| 24 |
+
"de": {
|
| 25 |
+
"harassment": 17,
|
| 26 |
+
"harassment_threatening": 5,
|
| 27 |
+
"hate": 4,
|
| 28 |
+
"hate_threatening": 3,
|
| 29 |
+
"self_harm": 2,
|
| 30 |
+
"self_harm_instructions": 1,
|
| 31 |
+
"self_harm_intent": 2,
|
| 32 |
+
"sexual": 7,
|
| 33 |
+
"sexual_minors": 5,
|
| 34 |
+
"violence": 9,
|
| 35 |
+
"violence_graphic": 3
|
| 36 |
+
},
|
| 37 |
+
"fr": {
|
| 38 |
+
"harassment": 12,
|
| 39 |
+
"harassment_threatening": 4,
|
| 40 |
+
"hate": 3,
|
| 41 |
+
"hate_threatening": 2,
|
| 42 |
+
"self_harm": 4,
|
| 43 |
+
"self_harm_instructions": 3,
|
| 44 |
+
"self_harm_intent": 3,
|
| 45 |
+
"sexual": 10,
|
| 46 |
+
"sexual_minors": 5,
|
| 47 |
+
"violence": 12,
|
| 48 |
+
"violence_graphic": 5
|
| 49 |
+
},
|
| 50 |
+
"es": {
|
| 51 |
+
"harassment": 17,
|
| 52 |
+
"harassment_threatening": 3,
|
| 53 |
+
"hate": 4,
|
| 54 |
+
"hate_threatening": 2,
|
| 55 |
+
"self_harm": 5,
|
| 56 |
+
"self_harm_instructions": 3,
|
| 57 |
+
"self_harm_intent": 5,
|
| 58 |
+
"sexual": 7,
|
| 59 |
+
"sexual_minors": 5,
|
| 60 |
+
"violence": 8,
|
| 61 |
+
"violence_graphic": 3
|
| 62 |
+
},
|
| 63 |
+
"it": {
|
| 64 |
+
"harassment": 12,
|
| 65 |
+
"harassment_threatening": 4,
|
| 66 |
+
"hate": 4,
|
| 67 |
+
"hate_threatening": 2,
|
| 68 |
+
"self_harm": 4,
|
| 69 |
+
"self_harm_instructions": 2,
|
| 70 |
+
"self_harm_intent": 4,
|
| 71 |
+
"sexual": 10,
|
| 72 |
+
"sexual_minors": 5,
|
| 73 |
+
"violence": 10,
|
| 74 |
+
"violence_graphic": 4
|
| 75 |
+
},
|
| 76 |
+
"sv": {
|
| 77 |
+
"harassment": 14,
|
| 78 |
+
"harassment_threatening": 4,
|
| 79 |
+
"hate": 3,
|
| 80 |
+
"hate_threatening": 2,
|
| 81 |
+
"self_harm": 4,
|
| 82 |
+
"self_harm_instructions": 3,
|
| 83 |
+
"self_harm_intent": 4,
|
| 84 |
+
"sexual": 7,
|
| 85 |
+
"sexual_minors": 4,
|
| 86 |
+
"violence": 7,
|
| 87 |
+
"violence_graphic": 3
|
| 88 |
+
},
|
| 89 |
+
"fi": {
|
| 90 |
+
"harassment": 11,
|
| 91 |
+
"harassment_threatening": 4,
|
| 92 |
+
"hate": 4,
|
| 93 |
+
"hate_threatening": 2,
|
| 94 |
+
"self_harm": 3,
|
| 95 |
+
"self_harm_instructions": 2,
|
| 96 |
+
"self_harm_intent": 3,
|
| 97 |
+
"sexual": 9,
|
| 98 |
+
"sexual_minors": 5,
|
| 99 |
+
"violence": 10,
|
| 100 |
+
"violence_graphic": 4
|
| 101 |
+
},
|
| 102 |
+
"pl": {
|
| 103 |
+
"harassment": 14,
|
| 104 |
+
"harassment_threatening": 4,
|
| 105 |
+
"hate": 3,
|
| 106 |
+
"hate_threatening": 1,
|
| 107 |
+
"self_harm": 4,
|
| 108 |
+
"self_harm_instructions": 2,
|
| 109 |
+
"self_harm_intent": 4,
|
| 110 |
+
"sexual": 8,
|
| 111 |
+
"sexual_minors": 5,
|
| 112 |
+
"violence": 10,
|
| 113 |
+
"violence_graphic": 4
|
| 114 |
+
},
|
| 115 |
+
"cs": {
|
| 116 |
+
"harassment": 16,
|
| 117 |
+
"harassment_threatening": 5,
|
| 118 |
+
"hate": 7,
|
| 119 |
+
"hate_threatening": 2,
|
| 120 |
+
"self_harm": 5,
|
| 121 |
+
"self_harm_instructions": 2,
|
| 122 |
+
"self_harm_intent": 4,
|
| 123 |
+
"sexual": 7,
|
| 124 |
+
"sexual_minors": 4,
|
| 125 |
+
"violence": 7,
|
| 126 |
+
"violence_graphic": 3
|
| 127 |
+
},
|
| 128 |
+
"lv": {
|
| 129 |
+
"harassment": 17,
|
| 130 |
+
"harassment_threatening": 4,
|
| 131 |
+
"hate": 4,
|
| 132 |
+
"hate_threatening": 2,
|
| 133 |
+
"self_harm": 3,
|
| 134 |
+
"self_harm_instructions": 1,
|
| 135 |
+
"self_harm_intent": 3,
|
| 136 |
+
"sexual": 9,
|
| 137 |
+
"sexual_minors": 4,
|
| 138 |
+
"violence": 8,
|
| 139 |
+
"violence_graphic": 4
|
| 140 |
+
},
|
| 141 |
+
"zh": {
|
| 142 |
+
"harassment": 15,
|
| 143 |
+
"harassment_threatening": 4,
|
| 144 |
+
"hate": 3,
|
| 145 |
+
"hate_threatening": 2,
|
| 146 |
+
"self_harm": 5,
|
| 147 |
+
"self_harm_instructions": 3,
|
| 148 |
+
"self_harm_intent": 4,
|
| 149 |
+
"sexual": 7,
|
| 150 |
+
"sexual_minors": 4,
|
| 151 |
+
"violence": 9,
|
| 152 |
+
"violence_graphic": 3
|
| 153 |
+
},
|
| 154 |
+
"ja": {
|
| 155 |
+
"harassment": 13,
|
| 156 |
+
"harassment_threatening": 3,
|
| 157 |
+
"hate": 5,
|
| 158 |
+
"hate_threatening": 2,
|
| 159 |
+
"self_harm": 4,
|
| 160 |
+
"self_harm_instructions": 2,
|
| 161 |
+
"self_harm_intent": 4,
|
| 162 |
+
"sexual": 6,
|
| 163 |
+
"sexual_minors": 4,
|
| 164 |
+
"violence": 7,
|
| 165 |
+
"violence_graphic": 3
|
| 166 |
+
},
|
| 167 |
+
"ko": {
|
| 168 |
+
"harassment": 12,
|
| 169 |
+
"harassment_threatening": 2,
|
| 170 |
+
"hate": 3,
|
| 171 |
+
"hate_threatening": 1,
|
| 172 |
+
"self_harm": 4,
|
| 173 |
+
"self_harm_instructions": 2,
|
| 174 |
+
"self_harm_intent": 4,
|
| 175 |
+
"sexual": 9,
|
| 176 |
+
"sexual_minors": 5,
|
| 177 |
+
"violence": 8,
|
| 178 |
+
"violence_graphic": 4
|
| 179 |
+
},
|
| 180 |
+
"ru": {
|
| 181 |
+
"harassment": 17,
|
| 182 |
+
"harassment_threatening": 3,
|
| 183 |
+
"hate": 4,
|
| 184 |
+
"hate_threatening": 1,
|
| 185 |
+
"self_harm": 3,
|
| 186 |
+
"self_harm_instructions": 2,
|
| 187 |
+
"self_harm_intent": 3,
|
| 188 |
+
"sexual": 8,
|
| 189 |
+
"sexual_minors": 5,
|
| 190 |
+
"violence": 5,
|
| 191 |
+
"violence_graphic": 2
|
| 192 |
+
},
|
| 193 |
+
"uk": {
|
| 194 |
+
"harassment": 14,
|
| 195 |
+
"harassment_threatening": 3,
|
| 196 |
+
"hate": 3,
|
| 197 |
+
"hate_threatening": 1,
|
| 198 |
+
"self_harm": 4,
|
| 199 |
+
"self_harm_instructions": 1,
|
| 200 |
+
"self_harm_intent": 4,
|
| 201 |
+
"sexual": 6,
|
| 202 |
+
"sexual_minors": 5,
|
| 203 |
+
"violence": 7,
|
| 204 |
+
"violence_graphic": 3
|
| 205 |
+
},
|
| 206 |
+
"be": {
|
| 207 |
+
"harassment": 14,
|
| 208 |
+
"harassment_threatening": 3,
|
| 209 |
+
"hate": 5,
|
| 210 |
+
"hate_threatening": 2,
|
| 211 |
+
"self_harm": 4,
|
| 212 |
+
"self_harm_instructions": 2,
|
| 213 |
+
"self_harm_intent": 3,
|
| 214 |
+
"sexual": 12,
|
| 215 |
+
"sexual_minors": 5,
|
| 216 |
+
"violence": 9,
|
| 217 |
+
"violence_graphic": 3
|
| 218 |
+
},
|
| 219 |
+
"kk": {
|
| 220 |
+
"harassment": 12,
|
| 221 |
+
"harassment_threatening": 3,
|
| 222 |
+
"hate": 2,
|
| 223 |
+
"hate_threatening": 1,
|
| 224 |
+
"self_harm": 3,
|
| 225 |
+
"self_harm_instructions": 1,
|
| 226 |
+
"self_harm_intent": 3,
|
| 227 |
+
"sexual": 10,
|
| 228 |
+
"sexual_minors": 5,
|
| 229 |
+
"violence": 10,
|
| 230 |
+
"violence_graphic": 5
|
| 231 |
+
}
|
| 232 |
+
},
|
| 233 |
+
"audit": {
|
| 234 |
+
"en": {
|
| 235 |
+
"harassment": 5,
|
| 236 |
+
"harassment_threatening": 4,
|
| 237 |
+
"hate": 4,
|
| 238 |
+
"hate_threatening": 3,
|
| 239 |
+
"self_harm": 3,
|
| 240 |
+
"self_harm_instructions": 1,
|
| 241 |
+
"self_harm_intent": 3,
|
| 242 |
+
"sexual": 3,
|
| 243 |
+
"sexual_minors": 2,
|
| 244 |
+
"violence": 5,
|
| 245 |
+
"violence_graphic": 1
|
| 246 |
+
},
|
| 247 |
+
"de": {
|
| 248 |
+
"harassment": 6,
|
| 249 |
+
"harassment_threatening": 4,
|
| 250 |
+
"hate": 3,
|
| 251 |
+
"hate_threatening": 2,
|
| 252 |
+
"self_harm": 4,
|
| 253 |
+
"self_harm_instructions": 2,
|
| 254 |
+
"self_harm_intent": 4,
|
| 255 |
+
"sexual": 3,
|
| 256 |
+
"sexual_minors": 1,
|
| 257 |
+
"violence": 5,
|
| 258 |
+
"violence_graphic": 3
|
| 259 |
+
},
|
| 260 |
+
"fr": {
|
| 261 |
+
"harassment": 9,
|
| 262 |
+
"harassment_threatening": 3,
|
| 263 |
+
"hate": 4,
|
| 264 |
+
"hate_threatening": 2,
|
| 265 |
+
"self_harm": 2,
|
| 266 |
+
"self_harm_instructions": 1,
|
| 267 |
+
"self_harm_intent": 2,
|
| 268 |
+
"sexual": 3,
|
| 269 |
+
"sexual_minors": 2,
|
| 270 |
+
"violence": 3,
|
| 271 |
+
"violence_graphic": 1
|
| 272 |
+
},
|
| 273 |
+
"es": {
|
| 274 |
+
"harassment": 8,
|
| 275 |
+
"harassment_threatening": 4,
|
| 276 |
+
"hate": 3,
|
| 277 |
+
"hate_threatening": 2,
|
| 278 |
+
"self_harm": 2,
|
| 279 |
+
"self_harm_instructions": 1,
|
| 280 |
+
"self_harm_intent": 2,
|
| 281 |
+
"sexual": 4,
|
| 282 |
+
"sexual_minors": 2,
|
| 283 |
+
"violence": 5,
|
| 284 |
+
"violence_graphic": 2
|
| 285 |
+
},
|
| 286 |
+
"it": {
|
| 287 |
+
"harassment": 7,
|
| 288 |
+
"harassment_threatening": 3,
|
| 289 |
+
"hate": 3,
|
| 290 |
+
"hate_threatening": 2,
|
| 291 |
+
"self_harm": 3,
|
| 292 |
+
"self_harm_instructions": 2,
|
| 293 |
+
"self_harm_intent": 3,
|
| 294 |
+
"sexual": 5,
|
| 295 |
+
"sexual_minors": 2,
|
| 296 |
+
"violence": 4,
|
| 297 |
+
"violence_graphic": 1
|
| 298 |
+
},
|
| 299 |
+
"sv": {
|
| 300 |
+
"harassment": 9,
|
| 301 |
+
"harassment_threatening": 3,
|
| 302 |
+
"hate": 4,
|
| 303 |
+
"hate_threatening": 2,
|
| 304 |
+
"self_harm": 2,
|
| 305 |
+
"self_harm_instructions": 1,
|
| 306 |
+
"self_harm_intent": 2,
|
| 307 |
+
"sexual": 4,
|
| 308 |
+
"sexual_minors": 2,
|
| 309 |
+
"violence": 4,
|
| 310 |
+
"violence_graphic": 2
|
| 311 |
+
},
|
| 312 |
+
"fi": {
|
| 313 |
+
"harassment": 7,
|
| 314 |
+
"harassment_threatening": 3,
|
| 315 |
+
"hate": 3,
|
| 316 |
+
"hate_threatening": 2,
|
| 317 |
+
"self_harm": 4,
|
| 318 |
+
"self_harm_instructions": 2,
|
| 319 |
+
"self_harm_intent": 4,
|
| 320 |
+
"sexual": 4,
|
| 321 |
+
"sexual_minors": 2,
|
| 322 |
+
"violence": 5,
|
| 323 |
+
"violence_graphic": 1
|
| 324 |
+
},
|
| 325 |
+
"pl": {
|
| 326 |
+
"harassment": 7,
|
| 327 |
+
"harassment_threatening": 3,
|
| 328 |
+
"hate": 4,
|
| 329 |
+
"hate_threatening": 2,
|
| 330 |
+
"self_harm": 2,
|
| 331 |
+
"self_harm_instructions": 1,
|
| 332 |
+
"self_harm_intent": 2,
|
| 333 |
+
"sexual": 3,
|
| 334 |
+
"sexual_minors": 1,
|
| 335 |
+
"violence": 4,
|
| 336 |
+
"violence_graphic": 1
|
| 337 |
+
},
|
| 338 |
+
"cs": {
|
| 339 |
+
"harassment": 6,
|
| 340 |
+
"harassment_threatening": 3,
|
| 341 |
+
"hate": 3,
|
| 342 |
+
"hate_threatening": 2,
|
| 343 |
+
"self_harm": 3,
|
| 344 |
+
"self_harm_instructions": 2,
|
| 345 |
+
"self_harm_intent": 2,
|
| 346 |
+
"sexual": 3,
|
| 347 |
+
"sexual_minors": 1,
|
| 348 |
+
"violence": 4,
|
| 349 |
+
"violence_graphic": 2
|
| 350 |
+
},
|
| 351 |
+
"lv": {
|
| 352 |
+
"harassment": 7,
|
| 353 |
+
"harassment_threatening": 3,
|
| 354 |
+
"hate": 4,
|
| 355 |
+
"hate_threatening": 2,
|
| 356 |
+
"self_harm": 2,
|
| 357 |
+
"self_harm_instructions": 2,
|
| 358 |
+
"self_harm_intent": 2,
|
| 359 |
+
"sexual": 5,
|
| 360 |
+
"sexual_minors": 2,
|
| 361 |
+
"violence": 4,
|
| 362 |
+
"violence_graphic": 1
|
| 363 |
+
},
|
| 364 |
+
"zh": {
|
| 365 |
+
"harassment": 6,
|
| 366 |
+
"harassment_threatening": 3,
|
| 367 |
+
"hate": 4,
|
| 368 |
+
"hate_threatening": 2,
|
| 369 |
+
"self_harm": 2,
|
| 370 |
+
"self_harm_instructions": 1,
|
| 371 |
+
"self_harm_intent": 2,
|
| 372 |
+
"sexual": 4,
|
| 373 |
+
"sexual_minors": 2,
|
| 374 |
+
"violence": 6,
|
| 375 |
+
"violence_graphic": 2
|
| 376 |
+
},
|
| 377 |
+
"ja": {
|
| 378 |
+
"harassment": 6,
|
| 379 |
+
"harassment_threatening": 5,
|
| 380 |
+
"hate": 4,
|
| 381 |
+
"hate_threatening": 2,
|
| 382 |
+
"self_harm": 3,
|
| 383 |
+
"self_harm_instructions": 2,
|
| 384 |
+
"self_harm_intent": 3,
|
| 385 |
+
"sexual": 3,
|
| 386 |
+
"sexual_minors": 1,
|
| 387 |
+
"violence": 6,
|
| 388 |
+
"violence_graphic": 2
|
| 389 |
+
},
|
| 390 |
+
"ko": {
|
| 391 |
+
"harassment": 7,
|
| 392 |
+
"harassment_threatening": 4,
|
| 393 |
+
"hate": 3,
|
| 394 |
+
"hate_threatening": 2,
|
| 395 |
+
"self_harm": 2,
|
| 396 |
+
"self_harm_instructions": 1,
|
| 397 |
+
"self_harm_intent": 3,
|
| 398 |
+
"sexual": 5,
|
| 399 |
+
"sexual_minors": 1,
|
| 400 |
+
"violence": 4,
|
| 401 |
+
"violence_graphic": 1
|
| 402 |
+
},
|
| 403 |
+
"ru": {
|
| 404 |
+
"harassment": 6,
|
| 405 |
+
"harassment_threatening": 3,
|
| 406 |
+
"hate": 3,
|
| 407 |
+
"hate_threatening": 3,
|
| 408 |
+
"self_harm": 3,
|
| 409 |
+
"self_harm_instructions": 1,
|
| 410 |
+
"self_harm_intent": 3,
|
| 411 |
+
"sexual": 2,
|
| 412 |
+
"sexual_minors": 1,
|
| 413 |
+
"violence": 6,
|
| 414 |
+
"violence_graphic": 2
|
| 415 |
+
},
|
| 416 |
+
"uk": {
|
| 417 |
+
"harassment": 7,
|
| 418 |
+
"harassment_threatening": 3,
|
| 419 |
+
"hate": 3,
|
| 420 |
+
"hate_threatening": 2,
|
| 421 |
+
"self_harm": 3,
|
| 422 |
+
"self_harm_instructions": 2,
|
| 423 |
+
"self_harm_intent": 2,
|
| 424 |
+
"sexual": 3,
|
| 425 |
+
"sexual_minors": 1,
|
| 426 |
+
"violence": 4,
|
| 427 |
+
"violence_graphic": 2
|
| 428 |
+
},
|
| 429 |
+
"be": {
|
| 430 |
+
"harassment": 6,
|
| 431 |
+
"harassment_threatening": 4,
|
| 432 |
+
"hate": 3,
|
| 433 |
+
"hate_threatening": 2,
|
| 434 |
+
"self_harm": 2,
|
| 435 |
+
"self_harm_instructions": 1,
|
| 436 |
+
"self_harm_intent": 2,
|
| 437 |
+
"sexual": 2,
|
| 438 |
+
"sexual_minors": 1,
|
| 439 |
+
"violence": 7,
|
| 440 |
+
"violence_graphic": 2
|
| 441 |
+
},
|
| 442 |
+
"kk": {
|
| 443 |
+
"harassment": 9,
|
| 444 |
+
"harassment_threatening": 4,
|
| 445 |
+
"hate": 5,
|
| 446 |
+
"hate_threatening": 3,
|
| 447 |
+
"self_harm": 3,
|
| 448 |
+
"self_harm_instructions": 2,
|
| 449 |
+
"self_harm_intent": 3,
|
| 450 |
+
"sexual": 3,
|
| 451 |
+
"sexual_minors": 1,
|
| 452 |
+
"violence": 5,
|
| 453 |
+
"violence_graphic": 1
|
| 454 |
+
}
|
| 455 |
+
}
|
| 456 |
+
},
|
| 457 |
+
"all_11_categories_positive_in_all_17_languages_both_splits": true,
|
| 458 |
+
"external_rows": 1680,
|
| 459 |
+
"external_overlap_excluded_from_primary_test": 0,
|
| 460 |
+
"external_revision": "84e5cf3bcd6acb3dfc70b6760451645872218a3e",
|
| 461 |
+
"external_sha256": "ef3f7de86ee394fe337b0a566ba4b9c7db058f3caa8d6a4a0006138b663ab9a8",
|
| 462 |
+
"research_mapping": {
|
| 463 |
+
"pornography": [
|
| 464 |
+
"sexual"
|
| 465 |
+
],
|
| 466 |
+
"illegal_activity": [],
|
| 467 |
+
"disturbing_content": [
|
| 468 |
+
"violence_graphic",
|
| 469 |
+
"self_harm"
|
| 470 |
+
],
|
| 471 |
+
"hate": [
|
| 472 |
+
"hate",
|
| 473 |
+
"hate_threatening"
|
| 474 |
+
],
|
| 475 |
+
"violence": [
|
| 476 |
+
"violence",
|
| 477 |
+
"violence_graphic"
|
| 478 |
+
]
|
| 479 |
+
},
|
| 480 |
+
"limitations": [
|
| 481 |
+
"Coverage is not proof of detection. Illegal activity has no dedicated output.",
|
| 482 |
+
"Broad disturbing content is not equivalent to graphic violence.",
|
| 483 |
+
"Internal source labels are inherited teacher estimates; authored checks are not human-reviewed.",
|
| 484 |
+
"Translations share semantic groups; reported row counts are not independent sample counts.",
|
| 485 |
+
"Typography and 10 authored EN/RU semantic paraphrases do not reproduce the undisclosed researcher benchmark.",
|
| 486 |
+
"External data has unknown labels and no gold four-profile labels.",
|
| 487 |
+
"Challenge policy: fraud assistance, hateful content, praise of violence and gross imagery blocked at all levels; adult erotica allowed only at light."
|
| 488 |
+
]
|
| 489 |
+
}
|
evaluation/evaluation.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evaluation/index.html
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
<!doctype html><html lang="ru"><meta charset="utf-8"><meta name="viewport" content="width=device-width,initial-scale=1"><title>Moderation 03 — калибровка и тест</title><style>body{font:16px system-ui;background:#101723;color:#eef3fc;max-width:1150px;margin:36px auto;padding:0 24px}h1,h2{color:#8cc9ff}h2{margin-top:36px}table{border-collapse:collapse;width:100%;font-variant-numeric:tabular-nums}th,td{text-align:right;padding:10px;border-bottom:1px solid #354457}th:first-child{text-align:left}details{margin:14px 0}summary{cursor:pointer;color:#8cc9ff}.note{padding:20px;background:#302719}.scroll{overflow:auto}a{color:#8cc9ff}p{line-height:1.6}</style><h1>Moderation 03 · эпоха 5</h1><p class="note">Этот отчёт проверяет исходную политику контента. Владелец также предполагает запрет нерабочих тем в корпоративных чатах; отдельный детектор релевантности рабочей теме здесь не реализован и не оценивался. Запрет темы организацией не меняет её сырой тег на hate.</p><p>Пороги подобраны только на 828 внутренних примерах. Отложенная проверка: 412 примеров. Внешний тест: 1 680. Веса модели фиксированы.</p><p class="note">Это калибровка порогов, не подтверждение исправления атаки исследователей. Внутренние метки преимущественно унаследованы от исходного датасета; новые проверки синтетические. Переводы связаны: число строк не равно числу независимых ситуаций. Незаконная деятельность не имеет отдельного выхода модели.</p><h2>Внутренний отложенный тест · после калибровки</h2><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>Лёгкий</th><td>412</td><td>109</td><td>35</td><td>244</td><td>24</td><td>75.7%</td><td>82.0%</td><td>12.5%</td><td>78.7%</td></tr><tr><th>Средний</th><td>412</td><td>189</td><td>51</td><td>146</td><td>26</td><td>78.8%</td><td>87.9%</td><td>25.9%</td><td>83.1%</td></tr><tr><th>Высокий</th><td>412</td><td>245</td><td>24</td><td>121</td><td>22</td><td>91.1%</td><td>91.8%</td><td>16.6%</td><td>91.4%</td></tr><tr><th>Корпоративный</th><td>412</td><td>252</td><td>22</td><td>115</td><td>23</td><td>92.0%</td><td>91.6%</td><td>16.1%</td><td>91.8%</td></tr></table></div><details><summary>До калибровки</summary><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>Лёгкий</th><td>412</td><td>20</td><td>0</td><td>279</td><td>113</td><td>100.0%</td><td>15.0%</td><td>0.0%</td><td>26.1%</td></tr><tr><th>Средний</th><td>412</td><td>118</td><td>19</td><td>178</td><td>97</td><td>86.1%</td><td>54.9%</td><td>9.6%</td><td>67.0%</td></tr><tr><th>Высокий</th><td>412</td><td>208</td><td>7</td><td>138</td><td>59</td><td>96.7%</td><td>77.9%</td><td>4.8%</td><td>86.3%</td></tr><tr><th>Корпоративный</th><td>412</td><td>246</td><td>19</td><td>118</td><td>29</td><td>92.8%</td><td>89.5%</td><td>13.9%</td><td>91.1%</td></tr></table></div></details><h2>Теги по уровням · внутренний тест</h2><details><summary>Лёгкий</summary><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>harassment_threatening</th><td>412</td><td>17</td><td>22</td><td>351</td><td>22</td><td>43.6%</td><td>43.6%</td><td>5.9%</td><td>43.6%</td></tr><tr><th>hate</th><td>412</td><td>39</td><td>7</td><td>350</td><td>16</td><td>84.8%</td><td>70.9%</td><td>2.0%</td><td>77.2%</td></tr><tr><th>hate_threatening</th><td>412</td><td>26</td><td>10</td><td>365</td><td>11</td><td>72.2%</td><td>70.3%</td><td>2.7%</td><td>71.2%</td></tr><tr><th>self_harm_instructions</th><td>412</td><td>4</td><td>1</td><td>388</td><td>19</td><td>80.0%</td><td>17.4%</td><td>0.3%</td><td>28.6%</td></tr><tr><th>sexual_minors</th><td>412</td><td>23</td><td>18</td><td>369</td><td>2</td><td>56.1%</td><td>92.0%</td><td>4.7%</td><td>69.7%</td></tr><tr><th>violence_graphic</th><td>412</td><td>22</td><td>46</td><td>340</td><td>4</td><td>32.4%</td><td>84.6%</td><td>11.9%</td><td>46.8%</td></tr></table></div></details><details><summary>Средний</summary><div class="scroll"><table><tr><th>Уровен�� / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>harassment</th><td>412</td><td>82</td><td>37</td><td>265</td><td>28</td><td>68.9%</td><td>74.5%</td><td>12.3%</td><td>71.6%</td></tr><tr><th>harassment_threatening</th><td>412</td><td>53</td><td>59</td><td>294</td><td>6</td><td>47.3%</td><td>89.8%</td><td>16.7%</td><td>62.0%</td></tr><tr><th>hate</th><td>412</td><td>55</td><td>40</td><td>315</td><td>2</td><td>57.9%</td><td>96.5%</td><td>11.3%</td><td>72.4%</td></tr><tr><th>hate_threatening</th><td>412</td><td>26</td><td>10</td><td>365</td><td>11</td><td>72.2%</td><td>70.3%</td><td>2.7%</td><td>71.2%</td></tr><tr><th>self_harm</th><td>412</td><td>18</td><td>5</td><td>370</td><td>19</td><td>78.3%</td><td>48.6%</td><td>1.3%</td><td>60.0%</td></tr><tr><th>self_harm_instructions</th><td>412</td><td>18</td><td>24</td><td>363</td><td>7</td><td>42.9%</td><td>72.0%</td><td>6.2%</td><td>53.7%</td></tr><tr><th>self_harm_intent</th><td>412</td><td>27</td><td>12</td><td>364</td><td>9</td><td>69.2%</td><td>75.0%</td><td>3.2%</td><td>72.0%</td></tr><tr><th>sexual</th><td>412</td><td>21</td><td>6</td><td>378</td><td>7</td><td>77.8%</td><td>75.0%</td><td>1.6%</td><td>76.4%</td></tr><tr><th>sexual_minors</th><td>412</td><td>23</td><td>26</td><td>361</td><td>2</td><td>46.9%</td><td>92.0%</td><td>6.7%</td><td>62.2%</td></tr><tr><th>violence</th><td>412</td><td>63</td><td>66</td><td>279</td><td>4</td><td>48.8%</td><td>94.0%</td><td>19.1%</td><td>64.3%</td></tr><tr><th>violence_graphic</th><td>412</td><td>23</td><td>45</td><td>339</td><td>5</td><td>33.8%</td><td>82.1%</td><td>11.7%</td><td>47.9%</td></tr></table></div></details><details><summary>Высокий</summary><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>harassment</th><td>412</td><td>130</td><td>43</td><td>215</td><td>24</td><td>75.1%</td><td>84.4%</td><td>16.7%</td><td>79.5%</td></tr><tr><th>harassment_threatening</th><td>412</td><td>56</td><td>56</td><td>293</td><td>7</td><td>50.0%</td><td>88.9%</td><td>16.0%</td><td>64.0%</td></tr><tr><th>hate</th><td>412</td><td>62</td><td>35</td><td>311</td><td>4</td><td>63.9%</td><td>93.9%</td><td>10.1%</td><td>76.1%</td></tr><tr><th>hate_threatening</th><td>412</td><td>26</td><td>10</td><td>365</td><td>11</td><td>72.2%</td><td>70.3%</td><td>2.7%</td><td>71.2%</td></tr><tr><th>self_harm</th><td>412</td><td>31</td><td>15</td><td>354</td><td>12</td><td>67.4%</td><td>72.1%</td><td>4.1%</td><td>69.7%</td></tr><tr><th>self_harm_instructions</th><td>412</td><td>18</td><td>24</td><td>363</td><td>7</td><td>42.9%</td><td>72.0%</td><td>6.2%</td><td>53.7%</td></tr><tr><th>self_harm_intent</th><td>412</td><td>29</td><td>10</td><td>363</td><td>10</td><td>74.4%</td><td>74.4%</td><td>2.7%</td><td>74.4%</td></tr><tr><th>sexual</th><td>412</td><td>28</td><td>11</td><td>342</td><td>31</td><td>71.8%</td><td>47.5%</td><td>3.1%</td><td>57.1%</td></tr><tr><th>sexual_minors</th><td>412</td><td>23</td><td>26</td><td>361</td><td>2</td><td>46.9%</td><td>92.0%</td><td>6.7%</td><td>62.2%</td></tr><tr><th>violence</th><td>412</td><td>68</td><td>61</td><td>274</td><td>9</td><td>52.7%</td><td>88.3%</td><td>18.2%</td><td>66.0%</td></tr><tr><th>violence_graphic</th><td>412</td><td>24</td><td>44</td><td>339</td><td>5</td><td>35.3%</td><td>82.8%</td><td>11.5%</td><td>49.5%</td></tr></table></div></details><details><summary>Корпоративный</summary><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>harassment</th><td>412</td><td>136</td><td>37</td><td>208</td><td>31</td><td>78.6%</td><td>81.4%</td><td>15.1%</td><td>80.0%</td></tr><tr><th>harassment_threatening</th><td>412</td><td>62</td><td>97</td><td>250</td><td>3</td><td>39.0%</td><td>95.4%</td><td>28.0%</td><td>55.4%</td></tr><tr><th>hate</th><td>412</td><td>70</td><td>27</td><td>310</td><td>5</td><td>72.2%</td><td>93.3%</td><td>8.0%</td><td>81.4%</td></tr><tr><th>hate_threatening</th><td>412</td><td>26</td><td>10</td><td>365</td><td>11</td><td>72.2%</td><td>70.3%</td><td>2.7%</td><td>71.2%</td></tr><tr><th>self_harm</th><td>412</td><td>35</td><td>19</td><td>348</td><td>10</td><td>64.8%</td><td>77.8%</td><td>5.2%</td><td>70.7%</td></tr><tr><th>self_harm_instructions</th><td>412</td><td>18</td><td>24</td><td>363</td><td>7</td><td>42.9%</td><td>72.0%</td><td>6.2%</td><td>53.7%</td></tr><tr><th>self_harm_intent</th><td>412</td><td>37</td><td>16</td><td>352</td><td>7</td><td>69.8%</td><td>84.1%</td><td>4.3%</td><td>76.3%</td></tr><tr><th>sexual</th><td>412</td><td>32</td><td>10</td><td>336</td><td>34</td><td>76.2%</td><td>48.5%</td><td>2.9%</td><td>59.3%</td></tr><tr><th>sexual_minors</th><td>412</td><td>24</td><td>25</td><td>361</td><td>2</td><td>49.0%</td><td>92.3%</td><td>6.5%</td><td>64.0%</td></tr><tr><th>violence</th><td>412</td><td>90</td><td>46</td><td>264</td><td>12</td><td>66.2%</td><td>88.2%</td><td>14.8%</td><td>75.6%</td></tr><tr><th>violence_graphic</th><td>412</td><td>24</td><td>44</td><td>339</td><td>5</td><td>35.3%</td><td>82.8%</td><td>11.5%</td><td>49.5%</td></tr></table></div></details><h2>Пять направлений из письма · дополнительные проверки</h2><p>Положительные запросы, безопасные контрпримеры, изменения написания; 10 переформулировок EN/RU. Это небольшой диагностический набор, не внешний бенчмарк. Эротика взрослых разрешена только на лёгком уровне; остальные четыре направления блокируются на всех уровнях.</p><h3>disturbing_content</h3><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>Лёгкий</th><td>53</td><td>23</td><td>0</td><td>17</td><td>13</td><td>100.0%</td><td>63.9%</td><td>0.0%</td><td>78.0%</td></tr><tr><th>Средний</th><td>53</td><td>24</td><td>0</td><td>17</td><td>12</td><td>100.0%</td><td>66.7%</td><td>0.0%</td><td>80.0%</td></tr><tr><th>Высокий</th><td>53</td><td>27</td><td>1</td><td>16</td><td>9</td><td>96.4%</td><td>75.0%</td><td>5.9%</td><td>84.4%</td></tr><tr><th>Корпоративный</th><td>53</td><td>27</td><td>1</td><td>16</td><td>9</td><td>96.4%</td><td>75.0%</td><td>5.9%</td><td>84.4%</td></tr></table></div><h3>hate</h3><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>Лёгкий</th><td>53</td><td>22</td><td>4</td><td>13</td><td>14</td><td>84.6%</td><td>61.1%</td><td>23.5%</td><td>71.0%</td></tr><tr><th>Средний</th><td>53</td><td>36</td><td>16</td><td>1</td><td>0</td><td>69.2%</td><td>100.0%</td><td>94.1%</td><td>81.8%</td></tr><tr><th>Высокий</th><td>53</td><td>36</td><td>16</td><td>1</td><td>0</td><td>69.2%</td><td>100.0%</td><td>94.1%</td><td>81.8%</td></tr><tr><th>Корпоративный</th><td>53</td><td>36</td><td>16</td><td>1</td><td>0</td><td>69.2%</td><td>100.0%</td><td>94.1%</td><td>81.8%</td></tr></table></div><h3>illegal_activity</h3><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>Лёгкий</th><td>53</td><td>0</td><td>0</td><td>17</td><td>36</td><td>—</td><td>0.0%</td><td>0.0%</td><td>0.0%</td></tr><tr><th>Средний</th><td>53</td><td>0</td><td>0</td><td>17</td><td>36</td><td>—</td><td>0.0%</td><td>0.0%</td><td>0.0%</td></tr><tr><th>Высокий</th><td>53</td><td>2</td><td>0</td><td>17</td><td>34</td><td>100.0%</td><td>5.6%</td><td>0.0%</td><td>10.5%</td></tr><tr><th>Корпоративный</th><td>53</td><td>7</td><td>0</td><td>17</td><td>29</td><td>100.0%</td><td>19.4%</td><td>0.0%</td><td>32.6%</td></tr></table></div><h3>pornography</h3><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>Лёгкий</th><td>53</td><td>0</td><td>32</td><td>21</td><td>0</td><td>0.0%</td><td>—</td><td>60.4%</td><td>0.0%</td></tr><tr><th>Средний</th><td>53</td><td>35</td><td>6</td><td>11</td><td>1</td><td>85.4%</td><td>97.2%</td><td>35.3%</td><td>90.9%</td></tr><tr><th>Высокий</th><td>53</td><td>36</td><td>7</td><td>10</td><td>0</td><td>83.7%</td><td>100.0%</td><td>41.2%</td><td>91.1%</td></tr><tr><th>Корпоративный</th><td>53</td><td>36</td><td>7</td><td>10</td><td>0</td><td>83.7%</td><td>100.0%</td><td>41.2%</td><td>91.1%</td></tr></table></div><h3>violence</h3><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>Лёгкий</th><td>53</td><td>23</td><td>7</td><td>10</td><td>13</td><td>76.7%</td><td>63.9%</td><td>41.2%</td><td>69.7%</td></tr><tr><th>Средний</th><td>53</td><td>32</td><td>11</td><td>6</td><td>4</td><td>74.4%</td><td>88.9%</td><td>64.7%</td><td>81.0%</td></tr><tr><th>Высокий</th><td>53</td><td>32</td><td>12</td><td>5</td><td>4</td><td>72.7%</td><td>88.9%</td><td>70.6%</td><td>80.0%</td></tr><tr><th>Корпоративный</th><td>53</td><td>32</td><td>17</td><td>0</td><td>4</td><td>65.3%</td><td>88.9%</td><td>100.0%</td><td>75.3%</td></tr></table></div><h2>Внешний тест · восемь категорий</h2><p>Неизвестные метки исключены. Ста��дартные пороги категорий также подобраны только на внутренней части.</p><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>sexual</th><td>984</td><td>228</td><td>77</td><td>670</td><td>9</td><td>74.8%</td><td>96.2%</td><td>10.3%</td><td>84.1%</td></tr><tr><th>hate</th><td>771</td><td>158</td><td>196</td><td>413</td><td>4</td><td>44.6%</td><td>97.5%</td><td>32.2%</td><td>61.2%</td></tr><tr><th>violence</th><td>1450</td><td>86</td><td>189</td><td>1167</td><td>8</td><td>31.3%</td><td>91.5%</td><td>13.9%</td><td>46.6%</td></tr><tr><th>harassment</th><td>1444</td><td>67</td><td>443</td><td>925</td><td>9</td><td>13.1%</td><td>88.2%</td><td>32.4%</td><td>22.9%</td></tr><tr><th>self_harm</th><td>1447</td><td>43</td><td>106</td><td>1290</td><td>8</td><td>28.9%</td><td>84.3%</td><td>7.6%</td><td>43.0%</td></tr><tr><th>sexual_minors</th><td>994</td><td>85</td><td>212</td><td>697</td><td>0</td><td>28.6%</td><td>100.0%</td><td>23.3%</td><td>44.5%</td></tr><tr><th>hate_threatening</th><td>761</td><td>36</td><td>69</td><td>651</td><td>5</td><td>34.3%</td><td>87.8%</td><td>9.6%</td><td>49.3%</td></tr><tr><th>violence_graphic</th><td>1447</td><td>21</td><td>147</td><td>1276</td><td>3</td><td>12.5%</td><td>87.5%</td><td>10.3%</td><td>21.9%</td></tr></table></div><p>Лимит модели: 512 токенов. Обрезано 76 из 1680 внешних текстов. Они остаются в основном тесте, отражая реальный режим работы.</p><h2>Внешний тест · проекция четырёх уровней</h2><p>Внешний набор не размечен по нашим четырём политикам. Здесь сравнивается объединение доступных бинарных категорий. Правило мата и три отсутствующие категории исключены; это не полная точность уровня модерации.</p><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>Лёгкий</th><td>800</td><td>202</td><td>210</td><td>323</td><td>65</td><td>49.0%</td><td>75.7%</td><td>39.4%</td><td>59.5%</td></tr><tr><th>Средний</th><td>859</td><td>493</td><td>196</td><td>141</td><td>29</td><td>71.6%</td><td>94.4%</td><td>58.2%</td><td>81.4%</td></tr><tr><th>Высокий</th><td>859</td><td>504</td><td>220</td><td>117</td><td>18</td><td>69.6%</td><td>96.6%</td><td>65.3%</td><td>80.9%</td></tr><tr><th>Корпоративный</th><td>859</td><td>505</td><td>222</td><td>115</td><td>17</td><td>69.5%</td><td>96.7%</td><td>65.9%</td><td>80.9%</td></tr></table></div><h2>Внутренний тест по языкам</h2><details><summary>be</summary><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>Лёгкий</th><td>24</td><td>6</td><td>2</td><td>15</td><td>1</td><td>75.0%</td><td>85.7%</td><td>11.8%</td><td>80.0%</td></tr><tr><th>Средний</th><td>24</td><td>9</td><td>5</td><td>7</td><td>3</td><td>64.3%</td><td>75.0%</td><td>41.7%</td><td>69.2%</td></tr><tr><th>Высокий</th><td>24</td><td>14</td><td>3</td><td>6</td><td>1</td><td>82.4%</td><td>93.3%</td><td>33.3%</td><td>87.5%</td></tr><tr><th>Корпоративный</th><td>24</td><td>15</td><td>2</td><td>6</td><td>1</td><td>88.2%</td><td>93.8%</td><td>25.0%</td><td>90.9%</td></tr></table></div></details><details><summary>cs</summary><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>Лёгкий</th><td>23</td><td>7</td><td>0</td><td>16</td><td>0</td><td>100.0%</td><td>100.0%</td><td>0.0%</td><td>100.0%</td></tr><tr><th>Средний</th><td>23</td><td>11</td><td>3</td><td>8</td><td>1</td><td>78.6%</td><td>91.7%</td><td>27.3%</td><td>84.6%</td></tr><tr><th>Высокий</th><td>23</td><td>15</td><td>2</td><td>6</td><td>0</td><td>88.2%</td><td>100.0%</td><td>25.0%</td><td>93.8%</td></tr><tr><th>Корпоративный</th><td>23</td><td>15</td><td>2</td><td>6</td><td>0</td><td>88.2%</td><td>100.0%</td><td>25.0%</td><td>93.8%</td></tr></table></div></details><details><summary>de</summary><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>Лёгкий</th><td>24</td><td>7</td><td>1</td><td>14</td><td>2</td><td>87.5%</td><td>77.8%</td><td>6.7%</td><td>82.4%</td></tr><tr><th>Средний</th><td>24</td><td>13</td><td>1</td><td>9</td><td>1</td><td>92.9%</td><td>92.9%</td><td>10.0%</td><td>92.9%</td></tr><tr><th>Высокий</th><td>24</td><td>15</td><td>0</td><td>8</td><td>1</td><td>100.0%</td><td>93.8%</td><td>0.0%</td><td>96.8%</td></tr><tr><th>Корпоративный</th><td>24</td><td>15</td><td>0</td><td>8</td><td>1</td><td>100.0%</td><td>93.8%</td><td>0.0%</td><td>96.8%</td></tr></table></div></details><details><summary>en</summary><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>Лёгкий</th><td>25</td><td>8</td><td>2</td><td>15</td><td>0</td><td>80.0%</td><td>100.0%</td><td>11.8%</td><td>88.9%</td></tr><tr><th>Средний</th><td>25</td><td>12</td><td>0</td><td>13</td><td>0</td><td>100.0%</td><td>100.0%</td><td>0.0%</td><td>100.0%</td></tr><tr><th>Высокий</th><td>25</td><td>14</td><td>0</td><td>11</td><td>0</td><td>100.0%</td><td>100.0%</td><td>0.0%</td><td>100.0%</td></tr><tr><th>Корпоративный</th><td>25</td><td>14</td><td>0</td><td>10</td><td>1</td><td>100.0%</td><td>93.3%</td><td>0.0%</td><td>96.6%</td></tr></table></div></details><details><summary>es</summary><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>Лёгкий</th><td>25</td><td>6</td><td>2</td><td>15</td><td>2</td><td>75.0%</td><td>75.0%</td><td>11.8%</td><td>75.0%</td></tr><tr><th>Средний</th><td>25</td><td>13</td><td>2</td><td>9</td><td>1</td><td>86.7%</td><td>92.9%</td><td>18.2%</td><td>89.7%</td></tr><tr><th>Высокий</th><td>25</td><td>16</td><td>1</td><td>6</td><td>2</td><td>94.1%</td><td>88.9%</td><td>14.3%</td><td>91.4%</td></tr><tr><th>Корпоративный</th><td>25</td><td>17</td><td>0</td><td>6</td><td>2</td><td>100.0%</td><td>89.5%</td><td>0.0%</td><td>94.4%</td></tr></table></div></details><details><summary>fi</summary><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>Лёгкий</th><td>24</td><td>5</td><td>3</td><td>13</td><td>3</td><td>62.5%</td><td>62.5%</td><td>18.8%</td><td>62.5%</td></tr><tr><th>Средний</th><td>24</td><td>11</td><td>4</td><td>8</td><td>1</td><td>73.3%</td><td>91.7%</td><td>33.3%</td><td>81.5%</td></tr><tr><th>Высокий</th><td>24</td><td>13</td><td>3</td><td>6</td><td>2</td><td>81.2%</td><td>86.7%</td><td>33.3%</td><td>83.9%</td></tr><tr><th>Корпоративный</th><td>24</td><td>13</td><td>3</td><td>6</td><td>2</td><td>81.2%</td><td>86.7%</td><td>33.3%</td><td>83.9%</td></tr></table></div></details><details><summary>fr</summary><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>Лёгкий</th><td>23</td><td>7</td><td>3</td><td>12</td><td>1</td><td>70.0%</td><td>87.5%</td><td>20.0%</td><td>77.8%</td></tr><tr><th>Средний</th><td>23</td><td>12</td><td>4</td><td>7</td><td>0</td><td>75.0%</td><td>100.0%</td><td>36.4%</td><td>85.7%</td></tr><tr><th>Высокий</th><td>23</td><td>16</td><td>0</td><td>7</td><td>0</td><td>100.0%</td><td>100.0%</td><td>0.0%</td><td>100.0%</td></tr><tr><th>Корпоративный</th><td>23</td><td>16</td><td>1</td><td>6</td><td>0</td><td>94.1%</td><td>100.0%</td><td>14.3%</td><td>97.0%</td></tr></table></div></details><details><summary>it</summary><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>Лёгкий</th><td>25</td><td>7</td><td>4</td><td>13</td><td>1</td><td>63.6%</td><td>87.5%</td><td>23.5%</td><td>73.7%</td></tr><tr><th>Средний</th><td>25</td><td>13</td><td>4</td><td>8</td><td>0</td><td>76.5%</td><td>100.0%</td><td>33.3%</td><td>86.7%</td></tr><tr><th>Высокий</th><td>25</td><td>16</td><td>2</td><td>7</td><td>0</td><td>88.9%</td><td>100.0%</td><td>22.2%</td><td>94.1%</td></tr><tr><th>Корпоративный</th><td>25</td><td>16</td><td>2</td><td>7</td><td>0</td><td>88.9%</td><td>100.0%</td><td>22.2%</td><td>94.1%</td></tr></table></div></details><details><summary>ja</summary><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>Лёгкий</th><td>24</td><td>6</td><td>1</td><td>14</td><td>3</td><td>85.7%</td><td>66.7%</td><td>6.7%</td><td>75.0%</td></tr><tr><th>Средний</th><td>24</td><td>8</td><td>4</td><td>8</td><td>4</td><td>66.7%</td><td>66.7%</td><td>33.3%</td><td>66.7%</td></tr><tr><th>Высокий</th><td>24</td><td>12</td><td>3</td><td>7</td><td>2</td><td>80.0%</td><td>85.7%</td><td>30.0%</td><td>82.8%</td></tr><tr><th>Корпоративный</th><td>24</td><td>13</td><td>2</td><td>7</td><td>2</td><td>86.7%</td><td>86.7%</td><td>22.2%</td><td>86.7%</td></tr></table></div></details><details><summary>kk</summary><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>Лёгкий</th><td>25</td><td>6</td><td>3</td><td>13</td><td>3</td><td>66.7%</td><td>66.7%</td><td>18.8%</td><td>66.7%</td></tr><tr><th>Средний</th><td>25</td><td>12</td><td>5</td><td>7</td><td>1</td><td>70.6%</td><td>92.3%</td><td>41.7%</td><td>80.0%</td></tr><tr><th>Высокий</th><td>25</td><td>17</td><td>2</td><td>6</td><td>0</td><td>89.5%</td><td>100.0%</td><td>25.0%</td><td>94.4%</td></tr><tr><th>Корпоративный</th><td>25</td><td>17</td><td>3</td><td>5</td><td>0</td><td>85.0%</td><td>100.0%</td><td>37.5%</td><td>91.9%</td></tr></table></div></details><details><summary>ko</summary><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>Лёгкий</th><td>24</td><td>6</td><td>2</td><td>15</td><td>1</td><td>75.0%</td><td>85.7%</td><td>11.8%</td><td>80.0%</td></tr><tr><th>Средний</th><td>24</td><td>10</td><td>3</td><td>10</td><td>1</td><td>76.9%</td><td>90.9%</td><td>23.1%</td><td>83.3%</td></tr><tr><th>Высокий</th><td>24</td><td>14</td><td>2</td><td>7</td><td>1</td><td>87.5%</td><td>93.3%</td><td>22.2%</td><td>90.3%</td></tr><tr><th>Корпоративный</th><td>24</td><td>15</td><td>1</td><td>7</td><td>1</td><td>93.8%</td><td>93.8%</td><td>12.5%</td><td>93.8%</td></tr></table></div></details><details><summary>lv</summary><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>Лёгкий</th><td>24</td><td>5</td><td>1</td><td>15</td><td>3</td><td>83.3%</td><td>62.5%</td><td>6.2%</td><td>71.4%</td></tr><tr><th>Средний</th><td>24</td><td>9</td><td>1</td><td>11</td><td>3</td><td>90.0%</td><td>75.0%</td><td>8.3%</td><td>81.8%</td></tr><tr><th>Высокий</th><td>24</td><td>13</td><td>0</td><td>8</td><td>3</td><td>100.0%</td><td>81.2%</td><td>0.0%</td><td>89.7%</td></tr><tr><th>Корпоративный</th><td>24</td><td>13</td><td>0</td><td>8</td><td>3</td><td>100.0%</td><td>81.2%</td><td>0.0%</td><td>89.7%</td></tr></table></div></details><details><summary>pl</summary><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>Лёгкий</th><td>23</td><td>5</td><td>4</td><td>12</td><td>2</td><td>55.6%</td><td>71.4%</td><td>25.0%</td><td>62.5%</td></tr><tr><th>Средний</th><td>23</td><td>11</td><td>3</td><td>7</td><td>2</td><td>78.6%</td><td>84.6%</td><td>30.0%</td><td>81.5%</td></tr><tr><th>Высокий</th><td>23</td><td>12</td><td>2</td><td>6</td><td>3</td><td>85.7%</td><td>80.0%</td><td>25.0%</td><td>82.8%</td></tr><tr><th>Корпоративный</th><td>23</td><td>12</td><td>2</td><td>5</td><td>4</td><td>85.7%</td><td>75.0%</td><td>28.6%</td><td>80.0%</td></tr></table></div></details><details><summary>ru</summary><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>Лёгкий</th><td>24</td><td>7</td><td>3</td><td>14</td><td>0</td><td>70.0%</td><td>100.0%</td><td>17.6%</td><td>82.4%</td></tr><tr><th>Средний</th><td>24</td><td>11</td><td>3</td><td>8</td><td>2</td><td>78.6%</td><td>84.6%</td><td>27.3%</td><td>81.5%</td></tr><tr><th>Высокий</th><td>24</td><td>14</td><td>0</td><td>8</td><td>2</td><td>100.0%</td><td>87.5%</td><td>0.0%</td><td>93.3%</td></tr><tr><th>Корпоративный</th><td>24</td><td>14</td><td>0</td><td>8</td><td>2</td><td>100.0%</td><td>87.5%</td><td>0.0%</td><td>93.3%</td></tr></table></div></details><details><summary>sv</summary><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>Лёгкий</th><td>25</td><td>7</td><td>0</td><td>17</td><td>1</td><td>100.0%</td><td>87.5%</td><td>0.0%</td><td>93.3%</td></tr><tr><th>Средний</th><td>25</td><td>12</td><td>2</td><td>9</td><td>2</td><td>85.7%</td><td>85.7%</td><td>18.2%</td><td>85.7%</td></tr><tr><th>Высокий</th><td>25</td><td>16</td><td>0</td><td>8</td><td>1</td><td>100.0%</td><td>94.1%</td><td>0.0%</td><td>97.0%</td></tr><tr><th>Корпоративный</th><td>25</td><td>17</td><td>1</td><td>7</td><td>0</td><td>94.4%</td><td>100.0%</td><td>12.5%</td><td>97.1%</td></tr></table></div></details><details><summary>uk</summary><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>Лёгкий</th><td>25</td><td>7</td><td>2</td><td>16</td><td>0</td><td>77.8%</td><td>100.0%</td><td>11.1%</td><td>87.5%</td></tr><tr><th>Средний</th><td>25</td><td>12</td><td>3</td><td>9</td><td>1</td><td>80.0%</td><td>92.3%</td><td>25.0%</td><td>85.7%</td></tr><tr><th>Высокий</th><td>25</td><td>14</td><td>2</td><td>7</td><td>2</td><td>87.5%</td><td>87.5%</td><td>22.2%</td><td>87.5%</td></tr><tr><th>Корпоративный</th><td>25</td><td>15</td><td>2</td><td>6</td><td>2</td><td>88.2%</td><td>88.2%</td><td>25.0%</td><td>88.2%</td></tr></table></div></details><details><summary>zh</summary><div class="scroll"><table><tr><th>Уровень / категория</th><th>n</th><th>TP</th><th>FP</th><th>TN</th><th>FN</th><th>precision</th><th>recall</th><th>false_positive_rate</th><th>f1</th></tr><tr><th>Лёгкий</th><td>25</td><td>7</td><td>2</td><td>15</td><td>1</td><td>77.8%</td><td>87.5%</td><td>11.8%</td><td>82.4%</td></tr><tr><th>Средний</th><td>25</td><td>10</td><td>4</td><td>8</td><td>3</td><td>71.4%</td><td>76.9%</td><td>33.3%</td><td>74.1%</td></tr><tr><th>Высокий</th><td>25</td><td>14</td><td>2</td><td>7</td><td>2</td><td>87.5%</td><td>87.5%</td><td>22.2%</td><td>87.5%</td></tr><tr><th>Корпоративный</th><td>25</td><td>15</td><td>1</td><td>7</td><td>2</td><td>93.8%</td><td>88.2%</td><td>12.5%</td><td>90.9%</td></tr></table></div></details><h2>Карта покрытия 11 × 17</h2><p>Положительные примеры по исходной оценке ≥ 0,5. В ячейках: калибровка / отложенная проверка. Переводы связаны общими группами.</p><div class="scroll"><table><tr><th>Категория</th><th>en</th><th>de</th><th>fr</th><th>es</th><th>it</th><th>sv</th><th>fi</th><th>pl</th><th>cs</th><th>lv</th><th>zh</th><th>ja</th><th>ko</th><th>ru</th><th>uk</th><th>be</th><th>kk</th></tr><tr><th>harassment</th><td>10 / 5</td><td>17 / 6</td><td>12 / 9</td><td>17 / 8</td><td>12 / 7</td><td>14 / 9</td><td>11 / 7</td><td>14 / 7</td><td>16 / 6</td><td>17 / 7</td><td>15 / 6</td><td>13 / 6</td><td>12 / 7</td><td>17 / 6</td><td>14 / 7</td><td>14 / 6</td><td>12 / 9</td></tr><tr><th>harassment_threatening</th><td>3 / 4</td><td>5 / 4</td><td>4 / 3</td><td>3 / 4</td><td>4 / 3</td><td>4 / 3</td><td>4 / 3</td><td>4 / 3</td><td>5 / 3</td><td>4 / 3</td><td>4 / 3</td><td>3 / 5</td><td>2 / 4</td><td>3 / 3</td><td>3 / 3</td><td>3 / 4</td><td>3 / 4</td></tr><tr><th>hate</th><td>3 / 4</td><td>4 / 3</td><td>3 / 4</td><td>4 / 3</td><td>4 / 3</td><td>3 / 4</td><td>4 / 3</td><td>3 / 4</td><td>7 / 3</td><td>4 / 4</td><td>3 / 4</td><td>5 / 4</td><td>3 / 3</td><td>4 / 3</td><td>3 / 3</td><td>5 / 3</td><td>2 / 5</td></tr><tr><th>hate_threatening</th><td>1 / 3</td><td>3 / 2</td><td>2 / 2</td><td>2 / 2</td><td>2 / 2</td><td>2 / 2</td><td>2 / 2</td><td>1 / 2</td><td>2 / 2</td><td>2 / 2</td><td>2 / 2</td><td>2 / 2</td><td>1 / 2</td><td>1 / 3</td><td>1 / 2</td><td>2 / 2</td><td>1 / 3</td></tr><tr><th>self_harm</th><td>3 / 3</td><td>2 / 4</td><td>4 / 2</td><td>5 / 2</td><td>4 / 3</td><td>4 / 2</td><td>3 / 4</td><td>4 / 2</td><td>5 / 3</td><td>3 / 2</td><td>5 / 2</td><td>4 / 3</td><td>4 / 2</td><td>3 / 3</td><td>4 / 3</td><td>4 / 2</td><td>3 / 3</td></tr><tr><th>self_harm_instructions</th><td>2 / 1</td><td>1 / 2</td><td>3 / 1</td><td>3 / 1</td><td>2 / 2</td><td>3 / 1</td><td>2 / 2</td><td>2 / 1</td><td>2 / 2</td><td>1 / 2</td><td>3 / 1</td><td>2 / 2</td><td>2 / 1</td><td>2 / 1</td><td>1 / 2</td><td>2 / 1</td><td>1 / 2</td></tr><tr><th>self_harm_intent</th><td>3 / 3</td><td>2 / 4</td><td>3 / 2</td><td>5 / 2</td><td>4 / 3</td><td>4 / 2</td><td>3 / 4</td><td>4 / 2</td><td>4 / 2</td><td>3 / 2</td><td>4 / 2</td><td>4 / 3</td><td>4 / 3</td><td>3 / 3</td><td>4 / 2</td><td>3 / 2</td><td>3 / 3</td></tr><tr><th>sexual</th><td>6 / 3</td><td>7 / 3</td><td>10 / 3</td><td>7 / 4</td><td>10 / 5</td><td>7 / 4</td><td>9 / 4</td><td>8 / 3</td><td>7 / 3</td><td>9 / 5</td><td>7 / 4</td><td>6 / 3</td><td>9 / 5</td><td>8 / 2</td><td>6 / 3</td><td>12 / 2</td><td>10 / 3</td></tr><tr><th>sexual_minors</th><td>4 / 2</td><td>5 / 1</td><td>5 / 2</td><td>5 / 2</td><td>5 / 2</td><td>4 / 2</td><td>5 / 2</td><td>5 / 1</td><td>4 / 1</td><td>4 / 2</td><td>4 / 2</td><td>4 / 1</td><td>5 / 1</td><td>5 / 1</td><td>5 / 1</td><td>5 / 1</td><td>5 / 1</td></tr><tr><th>violence</th><td>11 / 5</td><td>9 / 5</td><td>12 / 3</td><td>8 / 5</td><td>10 / 4</td><td>7 / 4</td><td>10 / 5</td><td>10 / 4</td><td>7 / 4</td><td>8 / 4</td><td>9 / 6</td><td>7 / 6</td><td>8 / 4</td><td>5 / 6</td><td>7 / 4</td><td>9 / 7</td><td>10 / 5</td></tr><tr><th>violence_graphic</th><td>5 / 1</td><td>3 / 3</td><td>5 / 1</td><td>3 / 2</td><td>4 / 1</td><td>3 / 2</td><td>4 / 1</td><td>4 / 1</td><td>3 / 2</td><td>4 / 1</td><td>3 / 2</td><td>3 / 2</td><td>4 / 1</td><td>2 / 2</td><td>3 / 2</td><td>3 / 2</td><td>5 / 1</td></tr></table></div><p>TP — верно заблокировано; FP — лишняя блокировка; TN — верно пропущено; FN — пропуск запрещённого. Precision — доля верных блокировок; recall — доля обнаруженного запрещённого.</p><p><a href="evaluation.json">Полные метрики JSON</a> · <a href="thresholds.json">Зафиксированные пороги</a> · <a href="scores_without_text.jsonl">Оценки по строкам без исходных текстов</a></p></html>
|
evaluation/scores_without_text.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
evaluation/thresholds.json
ADDED
|
@@ -0,0 +1,78 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"status": "fitted_internal_only",
|
| 3 |
+
"method": "group-weighted balanced error threshold search; nested profiles",
|
| 4 |
+
"fit_rows": 828,
|
| 5 |
+
"fit_groups": 638,
|
| 6 |
+
"thresholds": {
|
| 7 |
+
"light": {
|
| 8 |
+
"harassment": null,
|
| 9 |
+
"harassment_threatening": 0.12,
|
| 10 |
+
"hate": 0.26,
|
| 11 |
+
"hate_threatening": 0.01,
|
| 12 |
+
"self_harm": null,
|
| 13 |
+
"self_harm_instructions": 0.23,
|
| 14 |
+
"self_harm_intent": null,
|
| 15 |
+
"sexual": null,
|
| 16 |
+
"sexual_minors": 0.02,
|
| 17 |
+
"violence": null,
|
| 18 |
+
"violence_graphic": 0.02
|
| 19 |
+
},
|
| 20 |
+
"medium": {
|
| 21 |
+
"harassment": 0.45,
|
| 22 |
+
"harassment_threatening": 0.03,
|
| 23 |
+
"hate": 0.060000000000000005,
|
| 24 |
+
"hate_threatening": 0.01,
|
| 25 |
+
"self_harm": 0.12,
|
| 26 |
+
"self_harm_instructions": 0.01,
|
| 27 |
+
"self_harm_intent": 0.02,
|
| 28 |
+
"sexual": 0.59,
|
| 29 |
+
"sexual_minors": 0.01,
|
| 30 |
+
"violence": 0.09999999999999999,
|
| 31 |
+
"violence_graphic": 0.02
|
| 32 |
+
},
|
| 33 |
+
"high": {
|
| 34 |
+
"harassment": 0.24000000000000002,
|
| 35 |
+
"harassment_threatening": 0.03,
|
| 36 |
+
"hate": 0.05,
|
| 37 |
+
"hate_threatening": 0.01,
|
| 38 |
+
"self_harm": 0.03,
|
| 39 |
+
"self_harm_instructions": 0.01,
|
| 40 |
+
"self_harm_intent": 0.02,
|
| 41 |
+
"sexual": 0.34,
|
| 42 |
+
"sexual_minors": 0.01,
|
| 43 |
+
"violence": 0.09999999999999999,
|
| 44 |
+
"violence_graphic": 0.02
|
| 45 |
+
},
|
| 46 |
+
"corporate": {
|
| 47 |
+
"harassment": 0.24000000000000002,
|
| 48 |
+
"harassment_threatening": 0.01,
|
| 49 |
+
"hate": 0.05,
|
| 50 |
+
"hate_threatening": 0.01,
|
| 51 |
+
"self_harm": 0.02,
|
| 52 |
+
"self_harm_instructions": 0.01,
|
| 53 |
+
"self_harm_intent": 0.01,
|
| 54 |
+
"sexual": 0.29000000000000004,
|
| 55 |
+
"sexual_minors": 0.01,
|
| 56 |
+
"violence": 0.09,
|
| 57 |
+
"violence_graphic": 0.02
|
| 58 |
+
}
|
| 59 |
+
},
|
| 60 |
+
"standard_category_thresholds": {
|
| 61 |
+
"harassment": 0.25,
|
| 62 |
+
"harassment_threatening": 0.02,
|
| 63 |
+
"hate": 0.03,
|
| 64 |
+
"hate_threatening": 0.01,
|
| 65 |
+
"self_harm": 0.01,
|
| 66 |
+
"self_harm_instructions": 0.01,
|
| 67 |
+
"self_harm_intent": 0.01,
|
| 68 |
+
"sexual": 0.11,
|
| 69 |
+
"sexual_minors": 0.01,
|
| 70 |
+
"violence": 0.09,
|
| 71 |
+
"violence_graphic": 0.02
|
| 72 |
+
},
|
| 73 |
+
"probabilities_calibrated": false,
|
| 74 |
+
"checkpoint_sha256": "4218fb73825ff2fbbdd897a4346dd67141b15fc63d1811413929dddb88d408d3",
|
| 75 |
+
"fit_ids_sha256": "852388954817cf5ef3047eae79e8a1480cc60a1311673e439a02fbb9511cca23",
|
| 76 |
+
"policy": "Reference four-profile policy plus finite profanity rule. Research intent checks scored separately.",
|
| 77 |
+
"limitations": "Weak teacher-derived and synthetic labels. No new detector heads were trained."
|
| 78 |
+
}
|
model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4218fb73825ff2fbbdd897a4346dd67141b15fc63d1811413929dddb88d408d3
|
| 3 |
+
size 1018281676
|
moderation03.py
ADDED
|
@@ -0,0 +1,82 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Portable inference for the exact FP32 epoch-5 head used in the published evaluation."""
|
| 2 |
+
import gc
|
| 3 |
+
import json
|
| 4 |
+
import os
|
| 5 |
+
from pathlib import Path
|
| 6 |
+
import torch
|
| 7 |
+
from huggingface_hub import snapshot_download
|
| 8 |
+
from safetensors.torch import load_file
|
| 9 |
+
from transformers import AutoTokenizer, Qwen3_5ForConditionalGeneration
|
| 10 |
+
from moderation_model import ModerationHead
|
| 11 |
+
from moderation_policy import LANGS, PROFILES, profanity
|
| 12 |
+
from categories import CATEGORIES
|
| 13 |
+
|
| 14 |
+
MODEL_ID='ifmain/Moderation-03'
|
| 15 |
+
|
| 16 |
+
def apply_policy(scores, thresholds, *, block_profanity=False, text='', language='en'):
|
| 17 |
+
if set(scores)!=set(CATEGORIES) or set(thresholds)!=set(CATEGORIES):
|
| 18 |
+
raise ValueError('Exactly 11 named categories are required')
|
| 19 |
+
if language not in LANGS:raise ValueError('Unsupported language')
|
| 20 |
+
for value in thresholds.values():
|
| 21 |
+
if value is not None and not 0<=value<=1:raise ValueError('Thresholds must be in [0,1] or null (disabled)')
|
| 22 |
+
tags={c:thresholds[c] is not None and scores[c]>=thresholds[c] for c in CATEGORIES}
|
| 23 |
+
reasons=[c for c,flagged in tags.items() if flagged]
|
| 24 |
+
lexical=block_profanity and profanity(text,language)
|
| 25 |
+
if lexical:reasons.append('profanity_rule')
|
| 26 |
+
return {'block':bool(reasons),'tags':tags,'reasons':reasons,'thresholds':thresholds,
|
| 27 |
+
'profanity_rule_enabled':block_profanity,'profanity_rule_triggered':bool(lexical)}
|
| 28 |
+
|
| 29 |
+
class Moderation03:
|
| 30 |
+
def __init__(self, model_id=MODEL_ID, device=None, backbone_path=None):
|
| 31 |
+
self.device=device or ('cuda' if torch.cuda.is_available() else 'cpu')
|
| 32 |
+
torch.set_num_threads(int(os.getenv('TORCH_NUM_THREADS','4')))
|
| 33 |
+
torch.backends.mha.set_fastpath_enabled(False)
|
| 34 |
+
path=Path(model_id)
|
| 35 |
+
self.root=path if path.is_dir() else Path(snapshot_download(model_id,allow_patterns=[
|
| 36 |
+
'config.json','model.safetensors','calibration/thresholds.json']))
|
| 37 |
+
self.config=json.loads((self.root/'config.json').read_text(encoding='utf-8'))
|
| 38 |
+
self.calibration=json.loads((self.root/'calibration/thresholds.json').read_text(encoding='utf-8'))
|
| 39 |
+
backbone=backbone_path or self.config['backbone']
|
| 40 |
+
local=Path(backbone).is_dir()
|
| 41 |
+
kwargs={'local_files_only':True} if local else {'revision':self.config['backbone_revision']}
|
| 42 |
+
self.tokenizer=AutoTokenizer.from_pretrained(backbone,**kwargs)
|
| 43 |
+
self.tokenizer.padding_side='right'
|
| 44 |
+
if self.tokenizer.pad_token_id is None:self.tokenizer.pad_token=self.tokenizer.eos_token
|
| 45 |
+
full=Qwen3_5ForConditionalGeneration.from_pretrained(backbone,dtype=torch.bfloat16,
|
| 46 |
+
attn_implementation='sdpa',**kwargs)
|
| 47 |
+
self.backbone=full.model.language_model
|
| 48 |
+
del full;gc.collect()
|
| 49 |
+
self.backbone.requires_grad_(False).eval().to(self.device)
|
| 50 |
+
self.head=ModerationHead(self.config)
|
| 51 |
+
self.head.load_state_dict(load_file(str(self.root/'model.safetensors'),device='cpu'))
|
| 52 |
+
self.head.requires_grad_(False).eval().to(self.device)
|
| 53 |
+
|
| 54 |
+
@torch.inference_mode()
|
| 55 |
+
def scores(self,text):
|
| 56 |
+
if not isinstance(text,str) or not text.strip():raise ValueError('Enter non-empty text')
|
| 57 |
+
if len(text)>50000:raise ValueError('Maximum input length is 50,000 characters')
|
| 58 |
+
ids=self.tokenizer(text,add_special_tokens=False,truncation=False)['input_ids']
|
| 59 |
+
total=len(ids);ids=ids[:self.config['max_length']] or [self.tokenizer.eos_token_id]
|
| 60 |
+
encoded={'input_ids':torch.tensor([ids],device=self.device),
|
| 61 |
+
'attention_mask':torch.ones((1,len(ids)),dtype=torch.long,device=self.device)}
|
| 62 |
+
hidden=self.backbone(**encoded,use_cache=False,return_dict=True).last_hidden_state
|
| 63 |
+
with torch.autocast(self.device,dtype=torch.bfloat16):
|
| 64 |
+
logits=self.head(hidden.float(),encoded['attention_mask'])
|
| 65 |
+
if not bool(torch.isfinite(logits).all()):raise ValueError('Non-finite scores')
|
| 66 |
+
values=logits.float().sigmoid()[0].cpu().tolist()
|
| 67 |
+
return {'raw_scores':dict(zip(CATEGORIES,values)), 'input_tokens':total,
|
| 68 |
+
'processed_tokens':len(ids),'truncated':total>self.config['max_length'],
|
| 69 |
+
'checkpoint_epoch':5,'score_type':'uncalibrated sigmoid score'}
|
| 70 |
+
|
| 71 |
+
def predict(self,text,language='en',preset=None,thresholds=None,block_profanity=False):
|
| 72 |
+
if language not in LANGS:raise ValueError('Unsupported language')
|
| 73 |
+
if preset is not None and preset not in PROFILES:raise ValueError('Unknown preset')
|
| 74 |
+
if preset is not None and thresholds is not None:raise ValueError('Choose preset or custom thresholds')
|
| 75 |
+
result=self.scores(text)
|
| 76 |
+
if preset is not None:
|
| 77 |
+
thresholds=self.calibration['thresholds'][preset]
|
| 78 |
+
block_profanity=preset in ['high','corporate']
|
| 79 |
+
result['preset']=preset
|
| 80 |
+
result['policy']=None if thresholds is None else apply_policy(result['raw_scores'],thresholds,
|
| 81 |
+
block_profanity=block_profanity,text=text,language=language)
|
| 82 |
+
return result
|
moderation_model.py
ADDED
|
@@ -0,0 +1,70 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Trainable transformer head over the full sequence of frozen Qwen hidden states."""
|
| 2 |
+
import json
|
| 3 |
+
from pathlib import Path
|
| 4 |
+
import torch
|
| 5 |
+
from torch import nn
|
| 6 |
+
from torch.utils.checkpoint import checkpoint
|
| 7 |
+
|
| 8 |
+
DEFAULT_CONFIG = Path(__file__).resolve().parent / 'models/moderation_head/config.json'
|
| 9 |
+
|
| 10 |
+
|
| 11 |
+
def load_config(path=DEFAULT_CONFIG):
|
| 12 |
+
return json.loads(Path(path).read_text(encoding='utf-8'))
|
| 13 |
+
|
| 14 |
+
|
| 15 |
+
class ModerationHead(nn.Module):
|
| 16 |
+
def __init__(self, config):
|
| 17 |
+
super().__init__()
|
| 18 |
+
self.config = config
|
| 19 |
+
self.input_norm = nn.LayerNorm(config['input_size'])
|
| 20 |
+
self.projection = nn.Linear(config['input_size'], config['hidden_size'])
|
| 21 |
+
self.positions = nn.Embedding(config['max_length'], config['hidden_size'])
|
| 22 |
+
# Construct independently: every layer receives its own random initialization.
|
| 23 |
+
self.layers = nn.ModuleList([
|
| 24 |
+
nn.TransformerEncoderLayer(
|
| 25 |
+
d_model=config['hidden_size'], nhead=config['num_heads'],
|
| 26 |
+
dim_feedforward=config['intermediate_size'], dropout=config['dropout'],
|
| 27 |
+
activation='gelu', batch_first=True, norm_first=True)
|
| 28 |
+
for _ in range(config['num_layers'])
|
| 29 |
+
])
|
| 30 |
+
self.final_norm = nn.LayerNorm(config['hidden_size'])
|
| 31 |
+
self.classifier = nn.Linear(config['hidden_size'], len(config['categories']))
|
| 32 |
+
|
| 33 |
+
def forward(self, hidden_states, attention_mask):
|
| 34 |
+
if hidden_states.ndim != 3 or hidden_states.shape[-1] != self.config['input_size']:
|
| 35 |
+
raise ValueError('Expected [batch, sequence, Qwen_hidden_size]')
|
| 36 |
+
if hidden_states.shape[:2] != attention_mask.shape:
|
| 37 |
+
raise ValueError('Mask shape mismatch')
|
| 38 |
+
if hidden_states.shape[1] > self.config['max_length']:
|
| 39 |
+
raise ValueError('Sequence exceeds configured maximum')
|
| 40 |
+
mask = attention_mask.bool()
|
| 41 |
+
if not bool(mask.any(dim=1).all()):
|
| 42 |
+
raise ValueError('Every example requires at least one non-padding token')
|
| 43 |
+
x = self.projection(self.input_norm(hidden_states))
|
| 44 |
+
positions = (mask.long().cumsum(dim=1) - 1).clamp_min(0)
|
| 45 |
+
x = x + self.positions(positions)
|
| 46 |
+
padding = ~mask
|
| 47 |
+
for layer in self.layers:
|
| 48 |
+
if self.training and self.config.get('gradient_checkpointing', False):
|
| 49 |
+
x = checkpoint(layer, x, src_key_padding_mask=padding, use_reentrant=False)
|
| 50 |
+
else:
|
| 51 |
+
x = layer(x, src_key_padding_mask=padding)
|
| 52 |
+
x = self.final_norm(x)
|
| 53 |
+
pooled = (x * mask.unsqueeze(-1)).sum(dim=1) / mask.sum(dim=1, keepdim=True)
|
| 54 |
+
return self.classifier(pooled)
|
| 55 |
+
|
| 56 |
+
@torch.inference_mode()
|
| 57 |
+
def predict_scores(self, hidden_states, attention_mask):
|
| 58 |
+
if self.training:
|
| 59 |
+
raise RuntimeError('Call eval() before inference')
|
| 60 |
+
probabilities = self(hidden_states, attention_mask).float().sigmoid()
|
| 61 |
+
return [dict(zip(self.config['categories'], row)) for row in probabilities.cpu().tolist()]
|
| 62 |
+
|
| 63 |
+
|
| 64 |
+
def apply_profile(scores, thresholds):
|
| 65 |
+
"""Future calibrated thresholds must be provided explicitly for all categories."""
|
| 66 |
+
if thresholds is None or set(scores) != set(thresholds):
|
| 67 |
+
raise ValueError('This profile has not been calibrated for all 11 categories')
|
| 68 |
+
if any(not 0 <= v <= 1 for v in thresholds.values()):
|
| 69 |
+
raise ValueError('Thresholds must lie in [0, 1]')
|
| 70 |
+
return {name: score >= thresholds[name] for name, score in scores.items()}
|
moderation_policy.py
ADDED
|
@@ -0,0 +1,60 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Four nested policies over 11 raw outputs plus a separate explicit profanity rule."""
|
| 2 |
+
import re
|
| 3 |
+
import unicodedata
|
| 4 |
+
from categories import CATEGORIES
|
| 5 |
+
|
| 6 |
+
LANGS='en de fr es it sv fi pl cs lv zh ja ko ru uk be kk'.split()
|
| 7 |
+
PROFILES=['light','medium','high','corporate']
|
| 8 |
+
# These are policy definitions, not measured model probability thresholds.
|
| 9 |
+
REFERENCE={
|
| 10 |
+
'light':[None,.60,.80,.50,None,.60,None,None,.50,None,.60],
|
| 11 |
+
'medium':[.55,.45,.60,.40,.85,.45,.85,.80,.40,.85,.45],
|
| 12 |
+
'high':[.30,.30,.40,.25,.65,.30,.65,.50,.25,.60,.30],
|
| 13 |
+
'corporate':[.15,.20,.20,.15,.45,.20,.45,.20,.15,.30,.20],
|
| 14 |
+
}
|
| 15 |
+
TERMS={
|
| 16 |
+
'en':['fuck','fucking','shit','bullshit','motherfucker'],
|
| 17 |
+
'de':['scheiße','scheisse','verdammt','arschloch'],
|
| 18 |
+
'fr':['putain','merde','bordel'], 'es':['mierda','joder','coño'],
|
| 19 |
+
'it':['cazzo','merda','vaffanculo'], 'sv':['jävla','jävlar','helvete'],
|
| 20 |
+
'fi':['vittu','perkele','saatana'], 'pl':['kurwa','kurwy','pierdolić'],
|
| 21 |
+
'cs':['kurva','kurvy','do prdele'], 'lv':['bļeģ','bļadj','pizģec'],
|
| 22 |
+
'zh':['他妈的','操你妈','傻逼'], 'ja':['くそ','クソ','ちくしょう'],
|
| 23 |
+
'ko':['씨발','시발','좆'], 'ru':['блядь','блять','хуй','хуя','пиздец','ебать','ебаный'],
|
| 24 |
+
'uk':['блядь','блять','хуй','хуя','пиздець','їбаний'],
|
| 25 |
+
'be':['блядзь','блядь','хуй','піздзец'], 'kk':['қотақ','блядь','пиздец'],
|
| 26 |
+
}
|
| 27 |
+
HOMO=str.maketrans({'а':'a','е':'e','ё':'e','о':'o','р':'p','с':'c','х':'x','у':'y','і':'i'})
|
| 28 |
+
|
| 29 |
+
|
| 30 |
+
def norm(text):
|
| 31 |
+
return unicodedata.normalize('NFKC',text).casefold().translate(HOMO)
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
def pattern(word, lang):
|
| 35 |
+
# Inserted separators and common digit substitutions are accepted inside known words.
|
| 36 |
+
variants={'a':'a4','e':'e3','i':'i1','o':'o0','s':'s5','t':'t7','b':'b8','g':'g9','z':'z2'}
|
| 37 |
+
letters=[]
|
| 38 |
+
for ch in norm(word):
|
| 39 |
+
if ch.isspace():continue
|
| 40 |
+
letters.append('['+re.escape(variants.get(ch,ch))+']')
|
| 41 |
+
core=r'[\s_.\-]*'.join(letters)
|
| 42 |
+
return re.compile(core if lang in {'zh','ja','ko'} else r'(?<!\w)'+core+r'(?!\w)')
|
| 43 |
+
|
| 44 |
+
PATTERNS={lang:[pattern(word,lang) for word in terms] for lang,terms in TERMS.items()}
|
| 45 |
+
|
| 46 |
+
|
| 47 |
+
def profanity(text,language):
|
| 48 |
+
normalized=norm(text)
|
| 49 |
+
langs=[language] if language in PATTERNS else LANGS
|
| 50 |
+
return any(p.search(normalized) for lang in langs for p in PATTERNS[lang])
|
| 51 |
+
|
| 52 |
+
|
| 53 |
+
def decisions(scores, profane, thresholds=None):
|
| 54 |
+
thresholds=thresholds or {p:dict(zip(CATEGORIES,REFERENCE[p])) for p in PROFILES}
|
| 55 |
+
result={}
|
| 56 |
+
for profile in PROFILES:
|
| 57 |
+
reasons=[c for c in CATEGORIES if thresholds[profile][c] is not None and scores[c]>=thresholds[profile][c]]
|
| 58 |
+
if profane and profile in {'high','corporate'}:reasons.append('profanity_rule')
|
| 59 |
+
result[profile]={'block':bool(reasons),'reasons':reasons}
|
| 60 |
+
return result
|
requirements.txt
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
torch==2.13.0
|
| 2 |
+
transformers==5.14.1
|
| 3 |
+
safetensors==0.8.0
|
| 4 |
+
huggingface-hub==1.26.0
|
| 5 |
+
gradio==6.13.0
|