champollion 0.3.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +133 -0
- package/README.md +387 -0
- package/bin/cli.js +278 -0
- package/index.js +135 -0
- package/lib/api-key.js +127 -0
- package/lib/autofix.js +432 -0
- package/lib/bridge/method_bridge.py +430 -0
- package/lib/card-source-resolution.mjs +284 -0
- package/lib/cards/cache.js +169 -0
- package/lib/cards/env.js +82 -0
- package/lib/cards/fetch-card-child.js +38 -0
- package/lib/cards/reader.js +435 -0
- package/lib/cards/refresh.js +111 -0
- package/lib/cards/remote.js +387 -0
- package/lib/cldf-export.mjs +540 -0
- package/lib/cldf-terms.mjs +62 -0
- package/lib/command-help.js +790 -0
- package/lib/commands/audit.js +49 -0
- package/lib/commands/card.js +454 -0
- package/lib/commands/doctor.js +559 -0
- package/lib/commands/fonts.js +489 -0
- package/lib/commands/help.js +91 -0
- package/lib/commands/init.js +1259 -0
- package/lib/commands/integrity.js +148 -0
- package/lib/commands/leaderboard.js +478 -0
- package/lib/commands/lint.js +30 -0
- package/lib/commands/models.js +177 -0
- package/lib/commands/plugin.js +103 -0
- package/lib/commands/provenance.js +45 -0
- package/lib/commands/recommend.js +75 -0
- package/lib/commands/register-corpus.js +678 -0
- package/lib/commands/repair-script.js +42 -0
- package/lib/commands/seal-corpus.js +355 -0
- package/lib/commands/seo.js +72 -0
- package/lib/commands/serve.js +147 -0
- package/lib/commands/status.js +265 -0
- package/lib/commands/submit.js +332 -0
- package/lib/commands/sync.js +89 -0
- package/lib/commands/tm.js +573 -0
- package/lib/commands/verify.js +39 -0
- package/lib/commands/watch.js +20 -0
- package/lib/commands/wrap.js +138 -0
- package/lib/commands/xliff.js +327 -0
- package/lib/commercial-eligibility.js +235 -0
- package/lib/concurrent.js +87 -0
- package/lib/config.js +523 -0
- package/lib/contamination-lane.js +76 -0
- package/lib/content-sync.js +731 -0
- package/lib/content.js +733 -0
- package/lib/corpus-registration.mjs +608 -0
- package/lib/cost-report.js +346 -0
- package/lib/diff.js +155 -0
- package/lib/docusaurus-sync.js +1256 -0
- package/lib/flatten.js +55 -0
- package/lib/format.js +954 -0
- package/lib/hash.js +159 -0
- package/lib/icu.js +473 -0
- package/lib/integrity.js +689 -0
- package/lib/license-gate.mjs +478 -0
- package/lib/license-identify.mjs +229 -0
- package/lib/lint.js +629 -0
- package/lib/method-manifest.js +60 -0
- package/lib/methods/anthropic.js +140 -0
- package/lib/methods/apertium.js +163 -0
- package/lib/methods/api.js +316 -0
- package/lib/methods/base.js +184 -0
- package/lib/methods/content-separator.js +45 -0
- package/lib/methods/deepl.js +426 -0
- package/lib/methods/direct-llm.js +586 -0
- package/lib/methods/external.js +332 -0
- package/lib/methods/fetch-with-retry.js +124 -0
- package/lib/methods/gemini.js +147 -0
- package/lib/methods/google-translate.js +402 -0
- package/lib/methods/http-utils.js +122 -0
- package/lib/methods/libretranslate.js +314 -0
- package/lib/methods/llm-coached.js +670 -0
- package/lib/methods/llm.js +592 -0
- package/lib/methods/local.js +76 -0
- package/lib/methods/microsoft-translator.js +331 -0
- package/lib/methods/openai.js +131 -0
- package/lib/methods/openrouter-client.js +327 -0
- package/lib/methods/openrouter-pricing.js +156 -0
- package/lib/methods/provider-env.js +115 -0
- package/lib/methods/provider-pricing.js +310 -0
- package/lib/methods/tilde.js +150 -0
- package/lib/methods/translated.js +229 -0
- package/lib/methods/translation-error.js +80 -0
- package/lib/models.js +258 -0
- package/lib/no-translate.js +233 -0
- package/lib/output.js +238 -0
- package/lib/pairs.js +547 -0
- package/lib/plugins.js +447 -0
- package/lib/provenance.js +323 -0
- package/lib/recommend.js +648 -0
- package/lib/registers.js +1185 -0
- package/lib/repair-script.js +266 -0
- package/lib/scripts.js +994 -0
- package/lib/seal.mjs +464 -0
- package/lib/sealed-qualifier.mjs +211 -0
- package/lib/security.js +59 -0
- package/lib/segment.js +369 -0
- package/lib/seo.js +275 -0
- package/lib/serve.js +854 -0
- package/lib/string-classify.js +85 -0
- package/lib/submit.mjs +344 -0
- package/lib/sync.js +969 -0
- package/lib/tags/bcp47.js +202 -0
- package/lib/tags/resolve.js +314 -0
- package/lib/terminology.js +111 -0
- package/lib/tm-seed.js +294 -0
- package/lib/tm.js +515 -0
- package/lib/translate-pair.js +197 -0
- package/lib/translate.js +203 -0
- package/lib/types.js +230 -0
- package/lib/validate.js +510 -0
- package/lib/verify.js +451 -0
- package/lib/watch.js +145 -0
- package/lib/xliff.js +184 -0
- package/package.json +93 -0
- package/shared/ATTRIBUTION.md +145 -0
- package/shared/CORPORA-CARDS.md +288 -0
- package/shared/DATA-SOVEREIGNTY.md +500 -0
- package/shared/LANGUAGE-CARD-FIELDS.md +532 -0
- package/shared/card-lint-baseline.json +3189 -0
- package/shared/cards-fallback.json +1 -0
- package/shared/catalogue/card-config.json +6091 -0
- package/shared/catalogue/external-results.json +3888 -0
- package/shared/catalogue/gender-guidance.json +1038 -0
- package/shared/catalogue/method-coverage.json +1751 -0
- package/shared/catalogue/metric-coverage.json +170 -0
- package/shared/catalogue/metric-reliability.json +1 -0
- package/shared/catalogue/register-presets.json +3180 -0
- package/shared/catalogue/vitality-scales.json +55 -0
- package/shared/cldr-index.json +1115 -0
- package/shared/code-bridge.json +253 -0
- package/shared/corpora-cards-v1-reference.md +281 -0
- package/shared/curated-dictionary-flags.json +35 -0
- package/shared/curated-endonyms.json +35 -0
- package/shared/curated-fsts.json +51 -0
- package/shared/curated-orthography-conventions.json +26 -0
- package/shared/curated-sil-resources.json +374 -0
- package/shared/curated-tools.json +41 -0
- package/shared/docent/corpus.json +11333 -0
- package/shared/docent/faq.en.json +564 -0
- package/shared/docent/register-blocks.json +60 -0
- package/shared/docent/system-prompt.md +144 -0
- package/shared/domain-taxonomy.json +35 -0
- package/shared/explainers/glossary.json +2975 -0
- package/shared/explainers/tc-features.json +20112 -0
- package/shared/explainers/term-watchlist.json +147 -0
- package/shared/human-services.json +59 -0
- package/shared/license-corrections.json +261 -0
- package/shared/license-evidence.json +13452 -0
- package/shared/licenses.json +6781 -0
- package/shared/method-registry.json +236 -0
- package/shared/metric-registry.json +620 -0
- package/shared/model-aliases.json +7 -0
- package/shared/schemas/champollion-plugin.schema.json +206 -0
- package/shared/schemas/corpora-card.schema.json +957 -0
- package/shared/schemas/domain-taxonomy.schema.json +64 -0
- package/shared/schemas/external-results.schema.json +314 -0
- package/shared/schemas/human-services.schema.json +90 -0
- package/shared/schemas/language-card.schema.json +1308 -0
- package/shared/schemas/licenses.schema.json +155 -0
- package/shared/schemas/method-card.schema.json +412 -0
- package/shared/schemas/method-registry.schema.json +85 -0
- package/shared/schemas/metric-registry.schema.json +96 -0
- package/shared/schemas/metric-reliability.schema.json +178 -0
- package/shared/schemas/model-aliases.schema.json +27 -0
- package/shared/schemas/source-snapshot.schema.json +96 -0
|
@@ -0,0 +1,564 @@
|
|
|
1
|
+
{
|
|
2
|
+
"_note": "Docent FAQ SSOT (en). Authored+verified 2026-07-20 by the docent-content-authoring swarm, grounded in cli/website/docs. Served free (no model call) by docent-chat on a strong match. Translated per-locale via the champollion sync pipeline.",
|
|
3
|
+
"faq": [
|
|
4
|
+
{
|
|
5
|
+
"id": "what-is-champollion",
|
|
6
|
+
"question": "What is Champollion?",
|
|
7
|
+
"answer": "Champollion is open-source infrastructure for trustworthy machine translation across the world's languages, with a focus on low-resource and Indigenous ones. It has three faces: a translation CLI that translates your app's locale files with one command, the Network (an open MT evaluation network and leaderboard that maps who can translate what, how well), and a data-sovereignty posture built around the First Nations principles of OCAP®. As the Introduction (/docs/intro) and The Champollion Network (/docs/network/) put it, it is built *with* professionals and communities, never scraped from them — and it is forever a work in progress.",
|
|
8
|
+
"sources": [
|
|
9
|
+
"/docs/intro",
|
|
10
|
+
"/docs/network/",
|
|
11
|
+
"/docs/network/how-it-works"
|
|
12
|
+
],
|
|
13
|
+
"keywords": [
|
|
14
|
+
"what is champollion",
|
|
15
|
+
"about",
|
|
16
|
+
"overview",
|
|
17
|
+
"mission",
|
|
18
|
+
"what does champollion do"
|
|
19
|
+
]
|
|
20
|
+
},
|
|
21
|
+
{
|
|
22
|
+
"id": "free-and-open-source",
|
|
23
|
+
"question": "Is Champollion free? Is it open source? Does it cost anything?",
|
|
24
|
+
"answer": "Yes on both counts. Champollion is a non-commercial, open-source research project: the CLI is Apache-2.0 and the evaluation harness is AGPL-3.0, per How the Work Is Funded (/docs/network/sovereignty/economic-model). There is no paid API, no metering, and no revenue share — the only costs you'd incur are the API tokens you spend on your chosen model provider when you translate or benchmark. Champollion itself takes nothing.",
|
|
25
|
+
"sources": [
|
|
26
|
+
"/docs/network/sovereignty/economic-model",
|
|
27
|
+
"/docs/intro"
|
|
28
|
+
],
|
|
29
|
+
"keywords": [
|
|
30
|
+
"free",
|
|
31
|
+
"open source",
|
|
32
|
+
"cost",
|
|
33
|
+
"price",
|
|
34
|
+
"license",
|
|
35
|
+
"pricing",
|
|
36
|
+
"how much"
|
|
37
|
+
]
|
|
38
|
+
},
|
|
39
|
+
{
|
|
40
|
+
"id": "the-cli",
|
|
41
|
+
"question": "What is the champollion CLI and what does it do?",
|
|
42
|
+
"answer": "The CLI is the deployment end of Champollion: run `npx champollion sync` and it auto-detects your locale files, format, and target languages, translates what's missing, skips what's done, validates every result through a quality gate, and writes clean output. Per How It Works (/docs/how-it-works), it tracks every source string with SHA-256 hashes so re-runs only re-translate what changed, caches results in Translation Memory to save cost, and lets you pick a different translation method per language pair. It's a full i18n pipeline — sync, watch, lint, verify, audit, XLIFF export, and more (/docs/intro).",
|
|
43
|
+
"sources": [
|
|
44
|
+
"/docs/intro",
|
|
45
|
+
"/docs/how-it-works"
|
|
46
|
+
],
|
|
47
|
+
"keywords": [
|
|
48
|
+
"cli",
|
|
49
|
+
"command line",
|
|
50
|
+
"sync",
|
|
51
|
+
"tool",
|
|
52
|
+
"npx champollion",
|
|
53
|
+
"locale files",
|
|
54
|
+
"i18n"
|
|
55
|
+
]
|
|
56
|
+
},
|
|
57
|
+
{
|
|
58
|
+
"id": "install-and-run-cli",
|
|
59
|
+
"question": "How do I install and run the CLI?",
|
|
60
|
+
"answer": "You need Node.js 20.11+ and a translation API key. The fastest path is `npx champollion sync` with no install, or `npm install --save-dev champollion` — see Installation (/docs/getting-started/installation). Set a provider key (OpenRouter is recommended; Gemini has a free tier), then run `npx champollion sync` and it will translate your locale files in about 60 seconds, as walked through in the Quick Start (/docs/getting-started/quick-start).",
|
|
61
|
+
"sources": [
|
|
62
|
+
"/docs/getting-started/installation",
|
|
63
|
+
"/docs/getting-started/quick-start"
|
|
64
|
+
],
|
|
65
|
+
"keywords": [
|
|
66
|
+
"install",
|
|
67
|
+
"setup",
|
|
68
|
+
"getting started",
|
|
69
|
+
"run",
|
|
70
|
+
"api key",
|
|
71
|
+
"node",
|
|
72
|
+
"npm",
|
|
73
|
+
"how to install"
|
|
74
|
+
]
|
|
75
|
+
},
|
|
76
|
+
{
|
|
77
|
+
"id": "cli-methods",
|
|
78
|
+
"question": "What translation methods does the CLI support?",
|
|
79
|
+
"answer": "Methods are configured per language pair, so you can mix them in one project. How It Works (/docs/how-it-works) describes the core four — `llm` (any OpenRouter model), `llm-coached` (the same prompt plus grammar rules, a dictionary, and style notes), `google-translate`, and `api` (an HTTP endpoint you host) — plus `plugin` methods installed locally. Installation (/docs/getting-started/installation) lists the supported backends (OpenRouter, Gemini, OpenAI, Anthropic, DeepL, Google Translate, Microsoft Translator, LibreTranslate, Tilde, and Translated/Lara). You might use Google Translate for French and a coached LLM for a low-resource language, all under the same `sync` command.",
|
|
80
|
+
"sources": [
|
|
81
|
+
"/docs/how-it-works",
|
|
82
|
+
"/docs/getting-started/installation",
|
|
83
|
+
"/docs/guides/agent-guide"
|
|
84
|
+
],
|
|
85
|
+
"keywords": [
|
|
86
|
+
"methods",
|
|
87
|
+
"llm",
|
|
88
|
+
"coached",
|
|
89
|
+
"google translate",
|
|
90
|
+
"deepl",
|
|
91
|
+
"api",
|
|
92
|
+
"plugin",
|
|
93
|
+
"providers",
|
|
94
|
+
"per pair"
|
|
95
|
+
]
|
|
96
|
+
},
|
|
97
|
+
{
|
|
98
|
+
"id": "what-is-the-network",
|
|
99
|
+
"question": "What is the Network and the leaderboard?",
|
|
100
|
+
"answer": "The Champollion Network is open infrastructure to create and trust translation test sets and to map who can translate what, how well, across as many language pairs as possible. As The Champollion Network (/docs/network/) explains, you build a translation method, the eval harness benchmarks it with reproducible, fingerprinted scoring, and results appear on the public Method Leaderboard (/leaderboard). It runs on two kinds of benchmark: public benchmarks on open data, and sovereign benchmarks — secret community-owned test sets that Champollion never sees. Every method is welcome, human and machine.",
|
|
101
|
+
"sources": [
|
|
102
|
+
"/docs/network/",
|
|
103
|
+
"/docs/network/how-it-works",
|
|
104
|
+
"/leaderboard"
|
|
105
|
+
],
|
|
106
|
+
"keywords": [
|
|
107
|
+
"network",
|
|
108
|
+
"arena",
|
|
109
|
+
"leaderboard",
|
|
110
|
+
"benchmark",
|
|
111
|
+
"evaluation",
|
|
112
|
+
"harness",
|
|
113
|
+
"scoreboard"
|
|
114
|
+
]
|
|
115
|
+
},
|
|
116
|
+
{
|
|
117
|
+
"id": "leaderboard-status",
|
|
118
|
+
"question": "Is the leaderboard populated yet? What can I actually trust on it?",
|
|
119
|
+
"answer": "Honestly, not much is on it yet: the board is open for submissions and currently seeding — no runs have published, so there is nothing to rank (/docs/network/). Read the trust badge on any future row: 'self-reported' is the default and means exactly that, and no score on the site has yet cleared community speaker validation. Honest Limitations (/docs/network/honest-limitations) states these edges plainly — deep morphological validation currently covers one language pair, and community speaker-validation has not happened yet.",
|
|
120
|
+
"sources": [
|
|
121
|
+
"/docs/network/",
|
|
122
|
+
"/docs/network/honest-limitations",
|
|
123
|
+
"/docs/network/leaderboard/rules"
|
|
124
|
+
],
|
|
125
|
+
"keywords": [
|
|
126
|
+
"leaderboard empty",
|
|
127
|
+
"status",
|
|
128
|
+
"populated",
|
|
129
|
+
"results",
|
|
130
|
+
"trust",
|
|
131
|
+
"verified",
|
|
132
|
+
"honest limitations"
|
|
133
|
+
]
|
|
134
|
+
},
|
|
135
|
+
{
|
|
136
|
+
"id": "submit-a-method",
|
|
137
|
+
"question": "How do I submit a benchmark run or a method to the leaderboard?",
|
|
138
|
+
"answer": "Install the harness with `pip install mt-eval`, run it against a registered corpus (for example `mt-eval run --corpus eval-amh-fra-globalvoices-test-v1 --model gemini-pro`), review the run card it produces, then publish with `mt-eval publish` — the full walkthrough is Submit a Method (/docs/network/getting-started/submit-a-method). Your run first appears as self-benchmarked; the server then re-scores your outputs against the sha-pinned corpus and, when it reproduces, promotes it to Champollion Verified. A hosted upload API and web UI are planned but not yet live, so `mt-eval publish` or a pull request to the harness repo are the working paths today.",
|
|
139
|
+
"sources": [
|
|
140
|
+
"/docs/network/getting-started/submit-a-method",
|
|
141
|
+
"/docs/network/leaderboard/rules"
|
|
142
|
+
],
|
|
143
|
+
"keywords": [
|
|
144
|
+
"submit",
|
|
145
|
+
"publish",
|
|
146
|
+
"run card",
|
|
147
|
+
"mt-eval",
|
|
148
|
+
"leaderboard submission",
|
|
149
|
+
"benchmark run"
|
|
150
|
+
]
|
|
151
|
+
},
|
|
152
|
+
{
|
|
153
|
+
"id": "scoring-and-metrics",
|
|
154
|
+
"question": "How is a method scored, and what do the quality tiers mean?",
|
|
155
|
+
"answer": "The leaderboard scores up to five metrics — chrF++, exact match, FST acceptance (morphological validity), equivalent match, and semantic score — combined into a composite whose exact weights live in the Scoring Specification (/docs/network/specifications/scoring), the single source of truth. MT Evaluation (/docs/network/leaderboard/rules) and the Network Agent Guide (/docs/network/getting-started/agent-guide) lay out the quality tiers: Baseline (0.00–0.30), Emerging (0.30–0.50), Functional (0.50–0.70), Deployable (0.70–0.85), and Fluent (0.85–1.00). These are automated proxies, not quality guarantees — only community speaker review confirms real usability.",
|
|
156
|
+
"sources": [
|
|
157
|
+
"/docs/network/leaderboard/rules",
|
|
158
|
+
"/docs/network/getting-started/agent-guide",
|
|
159
|
+
"/docs/network/specifications/scoring"
|
|
160
|
+
],
|
|
161
|
+
"keywords": [
|
|
162
|
+
"scoring",
|
|
163
|
+
"metrics",
|
|
164
|
+
"chrf",
|
|
165
|
+
"fst",
|
|
166
|
+
"composite",
|
|
167
|
+
"quality tiers",
|
|
168
|
+
"deployable",
|
|
169
|
+
"how scored"
|
|
170
|
+
]
|
|
171
|
+
},
|
|
172
|
+
{
|
|
173
|
+
"id": "the-one-rule",
|
|
174
|
+
"question": "What is 'the one rule' about training data?",
|
|
175
|
+
"answer": "Do not train on the evaluation data. As stated in The Champollion Network (/docs/network/) and MT Evaluation (/docs/network/leaderboard/rules): methods exposed to the benchmark dataset — as training data, few-shot examples, dictionary entries, or prompt material — will be disqualified. Fine-tune on whatever you want; just not on the test set. Every dataset carries a `do_not_train` flag for exactly this reason.",
|
|
176
|
+
"sources": [
|
|
177
|
+
"/docs/network/",
|
|
178
|
+
"/docs/network/leaderboard/rules",
|
|
179
|
+
"/docs/network/leaderboard/datasets"
|
|
180
|
+
],
|
|
181
|
+
"keywords": [
|
|
182
|
+
"do not train",
|
|
183
|
+
"one rule",
|
|
184
|
+
"training data",
|
|
185
|
+
"contamination",
|
|
186
|
+
"disqualified",
|
|
187
|
+
"eval data"
|
|
188
|
+
]
|
|
189
|
+
},
|
|
190
|
+
{
|
|
191
|
+
"id": "non-programmer-help",
|
|
192
|
+
"question": "Can I really help if I'm not a programmer?",
|
|
193
|
+
"answer": "Absolutely — and if you speak an Indigenous or low-resource language, you may be the most important person in this ecosystem. For Language Communities (/docs/network/community/for-language-communities) explains that reference translations, translation review, and coaching data (your knowledge of how the language works) need no code at all, and this work is paid, not volunteered. You could even start today by adding sentences to Tatoeba, which can become benchmark data at the next corpus build. No programming is required to make a real contribution.",
|
|
194
|
+
"sources": [
|
|
195
|
+
"/docs/network/community/for-language-communities",
|
|
196
|
+
"/docs/network/perspectives/how-speakers-get-paid",
|
|
197
|
+
"/get-involved"
|
|
198
|
+
],
|
|
199
|
+
"keywords": [
|
|
200
|
+
"not a programmer",
|
|
201
|
+
"no coding",
|
|
202
|
+
"non technical",
|
|
203
|
+
"contribute",
|
|
204
|
+
"help",
|
|
205
|
+
"speaker",
|
|
206
|
+
"volunteer"
|
|
207
|
+
]
|
|
208
|
+
},
|
|
209
|
+
{
|
|
210
|
+
"id": "language-community-involved",
|
|
211
|
+
"question": "I'm a language community or speaker — how do I get involved, and can I own my test set?",
|
|
212
|
+
"answer": "Start with Your Language (/my-language) and For Language Communities (/docs/network/community/for-language-communities). The strongest position you can hold is owning the benchmark itself: register your corpus as a metadata card (never the content) and choose an exposure lane — open, research-only, or fully sovereign, where the test set never leaves your infrastructure and we never see it. You decide who may develop methods for your language, and a high score never grants anyone permission to deploy. Reach out via info@champollion.dev, and note the ownership-transfer model is a committed design, not yet a running program.",
|
|
213
|
+
"sources": [
|
|
214
|
+
"/my-language",
|
|
215
|
+
"/docs/network/community/for-language-communities",
|
|
216
|
+
"/docs/network/sovereignty/registering-corpora"
|
|
217
|
+
],
|
|
218
|
+
"keywords": [
|
|
219
|
+
"community",
|
|
220
|
+
"speaker",
|
|
221
|
+
"indigenous",
|
|
222
|
+
"my language",
|
|
223
|
+
"own test set",
|
|
224
|
+
"sovereign",
|
|
225
|
+
"get involved"
|
|
226
|
+
]
|
|
227
|
+
},
|
|
228
|
+
{
|
|
229
|
+
"id": "researcher-involved",
|
|
230
|
+
"question": "I'm an ML researcher — how do I get involved?",
|
|
231
|
+
"answer": "Build a method and benchmark it. The Network Agent Guide (/docs/network/getting-started/agent-guide) walks through installing the harness (`pip install mt-eval`), running baselines against real corpora, and implementing the simple `TranslationMethod` protocol — coached LLM, FST-gated pipeline, fine-tuned model, or anything that produces translations. You get standardized, reproducible, fingerprinted scoring against a shared corpus catalogue, and every run adds a point to a shared map of what works. Get Involved (/get-involved) points to the developer paths, including running the public benchmark queue.",
|
|
232
|
+
"sources": [
|
|
233
|
+
"/docs/network/getting-started/agent-guide",
|
|
234
|
+
"/get-involved",
|
|
235
|
+
"/docs/network/how-it-works"
|
|
236
|
+
],
|
|
237
|
+
"keywords": [
|
|
238
|
+
"researcher",
|
|
239
|
+
"ml engineer",
|
|
240
|
+
"build a method",
|
|
241
|
+
"harness",
|
|
242
|
+
"research",
|
|
243
|
+
"developer",
|
|
244
|
+
"reproducible"
|
|
245
|
+
]
|
|
246
|
+
},
|
|
247
|
+
{
|
|
248
|
+
"id": "translation-vendor-involved",
|
|
249
|
+
"question": "I'm a professional translator or a translation vendor — is there a place for me here?",
|
|
250
|
+
"answer": "Yes — human translation is a first-class method on the Network, listed and benchmarked alongside the machines, not treated as an afterthought (/docs/network/). To be listed, use Submit to the Index (/docs/network/getting-started/submit-to-the-index) and propose a 'human translation service' entry; a maintainer reviews every submission by hand, and your contact details are handled out-of-band, never posted in a public issue. Separately, the CLI's XLIFF workflow (/docs/guides/professional-translators) lets developers export machine drafts for professional review in memoQ, Trados, Phrase, or OmegaT and import your corrections back.",
|
|
251
|
+
"sources": [
|
|
252
|
+
"/docs/network/getting-started/submit-to-the-index",
|
|
253
|
+
"/docs/network/",
|
|
254
|
+
"/docs/guides/professional-translators"
|
|
255
|
+
],
|
|
256
|
+
"keywords": [
|
|
257
|
+
"translator",
|
|
258
|
+
"vendor",
|
|
259
|
+
"agency",
|
|
260
|
+
"human translation",
|
|
261
|
+
"professional",
|
|
262
|
+
"xliff",
|
|
263
|
+
"service listing"
|
|
264
|
+
]
|
|
265
|
+
},
|
|
266
|
+
{
|
|
267
|
+
"id": "ocap-aspirant-meaning",
|
|
268
|
+
"question": "What does 'OCAP®-aspirant' mean?",
|
|
269
|
+
"answer": "OCAP® — Ownership, Control, Access, and Possession — is a set of First Nations data-governance principles, and it is a registered trademark of the First Nations Information Governance Centre (FNIGC). Champollion's posture is OCAP®-aspirant: as Contact, Objections & Takedowns (/docs/network/community/contact-objections-takedown) puts it, the design is built so communities *can* exercise ownership, control, access, and possession of their data and anything derived from it — but whether it *achieves* that is for communities to decide, which is exactly why the objections channel exists. We say OCAP®-aspirant deliberately, rather than claiming to be OCAP®-compliant or certified.",
|
|
270
|
+
"sources": [
|
|
271
|
+
"/docs/network/community/contact-objections-takedown",
|
|
272
|
+
"/docs/network/sovereignty/data-sovereignty"
|
|
273
|
+
],
|
|
274
|
+
"keywords": [
|
|
275
|
+
"ocap",
|
|
276
|
+
"ocap aspirant",
|
|
277
|
+
"fnigc",
|
|
278
|
+
"sovereignty principles",
|
|
279
|
+
"ownership control access possession",
|
|
280
|
+
"indigenous"
|
|
281
|
+
]
|
|
282
|
+
},
|
|
283
|
+
{
|
|
284
|
+
"id": "data-sovereignty",
|
|
285
|
+
"question": "What is data sovereignty here — do you store or scrape my language data?",
|
|
286
|
+
"answer": "No. Data Stewardship (/docs/network/sovereignty/data-sovereignty) states the position: language data is biodata, so the people who provide a corpus hold the keys to it and to anything measured against it. Champollion never holds the data — corpora are registered as hash-pinned metadata cards and fetched from the steward's own hosting at evaluation time; take your archive offline and evaluation simply stops. Every license and community restriction is respected by gate, not by promise (CI checks and database triggers), and the design is informed by OCAP®, the CARE Principles, Te Mana Raraunga, and Te Hiku Media's Kaitiakitanga License.",
|
|
287
|
+
"sources": [
|
|
288
|
+
"/docs/network/sovereignty/data-sovereignty",
|
|
289
|
+
"/docs/network/sovereignty/registering-corpora"
|
|
290
|
+
],
|
|
291
|
+
"keywords": [
|
|
292
|
+
"data sovereignty",
|
|
293
|
+
"store data",
|
|
294
|
+
"scrape",
|
|
295
|
+
"biodata",
|
|
296
|
+
"fetch from source",
|
|
297
|
+
"care",
|
|
298
|
+
"kaitiakitanga",
|
|
299
|
+
"privacy"
|
|
300
|
+
]
|
|
301
|
+
},
|
|
302
|
+
{
|
|
303
|
+
"id": "licensing-lanes",
|
|
304
|
+
"question": "What are the licensing lanes, and how do they work?",
|
|
305
|
+
"answer": "When you register a corpus you pick an exposure lane, described in Registering Corpora & Exposure Lanes (/docs/network/sovereignty/registering-corpora): public (open license, may rank on the public board), non-commercial research-only (benchmarkable but carved out of every commercial, prize, and API path), or private (your own scored runs, references never published). Enforcement is use-based and mechanical — the commercial lane is strict, the research lane is lenient, and quarantine always wins so an improper subset can never rank in any lane. The Evaluation Datasets page (/docs/network/leaderboard/datasets) shows the per-family license table; some restricted-license sets additionally refuse remote model-API evaluation until the rights-holder's consent is recorded.",
|
|
306
|
+
"sources": [
|
|
307
|
+
"/docs/network/sovereignty/registering-corpora",
|
|
308
|
+
"/docs/network/leaderboard/datasets"
|
|
309
|
+
],
|
|
310
|
+
"keywords": [
|
|
311
|
+
"license",
|
|
312
|
+
"lanes",
|
|
313
|
+
"non-commercial",
|
|
314
|
+
"research-only",
|
|
315
|
+
"private",
|
|
316
|
+
"public",
|
|
317
|
+
"quarantine",
|
|
318
|
+
"commercial"
|
|
319
|
+
]
|
|
320
|
+
},
|
|
321
|
+
{
|
|
322
|
+
"id": "secret-test-set",
|
|
323
|
+
"question": "Can my community keep its test set secret and still have methods scored against it?",
|
|
324
|
+
"answer": "Yes — that's the sovereign lane, and it is treated as core architecture, not an exception (/docs/network/sovereignty/data-sovereignty). In the sovereign/private lane the test set never leaves community infrastructure and Champollion never sees it; methods are scored on your side and only the score travels (/docs/network/sovereignty/registering-corpora). For Language Communities (/docs/network/community/for-language-communities) points to the Run a Sovereign Contest runbook for hosting a community-controlled evaluation on your own terms — your test set, your rules, your decision about what, if anything, gets published.",
|
|
325
|
+
"sources": [
|
|
326
|
+
"/docs/network/sovereignty/registering-corpora",
|
|
327
|
+
"/docs/network/community/for-language-communities",
|
|
328
|
+
"/docs/network/sovereignty/data-sovereignty"
|
|
329
|
+
],
|
|
330
|
+
"keywords": [
|
|
331
|
+
"secret",
|
|
332
|
+
"sealed",
|
|
333
|
+
"sovereign",
|
|
334
|
+
"held-out",
|
|
335
|
+
"private test set",
|
|
336
|
+
"contest",
|
|
337
|
+
"keys",
|
|
338
|
+
"custody"
|
|
339
|
+
]
|
|
340
|
+
},
|
|
341
|
+
{
|
|
342
|
+
"id": "speakers-paid",
|
|
343
|
+
"question": "Do speakers get paid, and how much?",
|
|
344
|
+
"answer": "Yes — paying speakers is treated as non-negotiable, and payment is unconditional (you're paid whether or not your ratings are used). How Speakers Get Paid (/docs/network/perspectives/how-speakers-get-paid) publishes the rates: bilingual speaker work at $50–65 CAD per hour, with corpus curation budgeted at $2,500–6,000 and a full metric-validation round at $1,475–1,920. Payment does not buy your data — you are paid for the work and remain the steward of what you build. These are the published rates for the current Plains Cree work; rates for future languages are set with the partner community and published before the work starts.",
|
|
345
|
+
"sources": [
|
|
346
|
+
"/docs/network/perspectives/how-speakers-get-paid",
|
|
347
|
+
"/docs/network/sovereignty/economic-model"
|
|
348
|
+
],
|
|
349
|
+
"keywords": [
|
|
350
|
+
"paid",
|
|
351
|
+
"payment",
|
|
352
|
+
"speakers",
|
|
353
|
+
"rates",
|
|
354
|
+
"compensation",
|
|
355
|
+
"corpus work",
|
|
356
|
+
"validation",
|
|
357
|
+
"how much"
|
|
358
|
+
]
|
|
359
|
+
},
|
|
360
|
+
{
|
|
361
|
+
"id": "ownership-transfer",
|
|
362
|
+
"question": "What happens when a method wins — who owns it?",
|
|
363
|
+
"answer": "For Indigenous-language corpora, the default Community Transfer Template (/docs/network/sovereignty/ownership-transfer) has the winning method — source code, weights, configuration, and coaching data — transfer to the community's governance organization, which owns it outright to inspect, modify, deploy, shelve, or license with no ongoing claim from Champollion. The developer keeps attribution and the right to publish and reuse their techniques. Deployment is entirely the community's decision, and Champollion takes no share of anything a community earns. Note this is a documented template, not yet a track record — no prize has opened and no transfer has happened.",
|
|
364
|
+
"sources": [
|
|
365
|
+
"/docs/network/sovereignty/ownership-transfer",
|
|
366
|
+
"/docs/network/getting-started/agent-guide"
|
|
367
|
+
],
|
|
368
|
+
"keywords": [
|
|
369
|
+
"ownership",
|
|
370
|
+
"transfer",
|
|
371
|
+
"who owns",
|
|
372
|
+
"winning method",
|
|
373
|
+
"prize",
|
|
374
|
+
"community owns",
|
|
375
|
+
"deployment"
|
|
376
|
+
]
|
|
377
|
+
},
|
|
378
|
+
{
|
|
379
|
+
"id": "explore-and-correct-languages",
|
|
380
|
+
"question": "How do I explore the languages Champollion knows, and fix something that's wrong?",
|
|
381
|
+
"answer": "Browse the Language Atlas at /languages — the one language-browse surface, with search, filters, and a detail card for each language showing classification, vitality, scripts, and resources. Champollion is an index, not an authority: every value on a card cites its source, and when sources disagree the card shows all of them attributed. If something is wrong, outdated, or missing, use Submit to the Index (/docs/network/getting-started/submit-to-the-index) — every card carries a 'Suggest a correction' link — and a maintainer applies a cited fix at the data source. Corrections from speakers outrank any database we ingest.",
|
|
382
|
+
"sources": [
|
|
383
|
+
"/languages",
|
|
384
|
+
"/docs/network/getting-started/submit-to-the-index",
|
|
385
|
+
"/my-language"
|
|
386
|
+
],
|
|
387
|
+
"keywords": [
|
|
388
|
+
"languages",
|
|
389
|
+
"atlas",
|
|
390
|
+
"browse",
|
|
391
|
+
"language card",
|
|
392
|
+
"correction",
|
|
393
|
+
"fix",
|
|
394
|
+
"catalogue",
|
|
395
|
+
"explore"
|
|
396
|
+
]
|
|
397
|
+
},
|
|
398
|
+
{
|
|
399
|
+
"id": "ai-agent-usage",
|
|
400
|
+
"question": "How should an AI agent use Champollion?",
|
|
401
|
+
"answer": "Two things are built for agents. First, the machine-readable /llms.txt index maps the whole site, the license lanes, and the public data feeds (queue, mesh, corpus registry). Second, the MCP server `@champollion/mcp-server` (on npm) is the agent-facing door — translate with Translation-Memory caching and a deterministic quality gate, browse the benchmark queue, estimate cost, run benchmarks, read the leaderboard, and query per-language metric-trust evidence, all as tools. The Agent Guides (/docs/guides/agent-guide for the CLI, /docs/network/getting-started/agent-guide for benchmarking) give step-by-step recipes. The best pattern is to set up your own agent with a workspace and your own API key — tokens aren't free.",
|
|
402
|
+
"sources": [
|
|
403
|
+
"/llms.txt",
|
|
404
|
+
"/docs/guides/agent-guide",
|
|
405
|
+
"/docs/network/getting-started/agent-guide"
|
|
406
|
+
],
|
|
407
|
+
"keywords": [
|
|
408
|
+
"ai agent",
|
|
409
|
+
"mcp",
|
|
410
|
+
"llms.txt",
|
|
411
|
+
"mcp server",
|
|
412
|
+
"automation",
|
|
413
|
+
"tools",
|
|
414
|
+
"claude",
|
|
415
|
+
"agent guide"
|
|
416
|
+
]
|
|
417
|
+
},
|
|
418
|
+
{
|
|
419
|
+
"id": "contribute-compute",
|
|
420
|
+
"question": "How do I contribute compute by running the benchmark queue?",
|
|
421
|
+
"answer": "The leaderboard has empty squares — (language pair, model, condition) combinations nobody has measured — kept in a public queue at /queue.json. Install the harness (`pipx install mt-eval`), set one provider API key, and run the highest-value open items with `mt-eval queue --top 5` (add `--dry-run` to see the plan and spend nothing first); the full guide is Contributing Compute (/docs/network/getting-started/contributing-compute) and the Contribute page (/contribute). No account is needed — results publish as 'anonymous' unless you sign in to put your name on the board — and every run is a real, citable contribution to low-resource MT evaluation.",
|
|
422
|
+
"sources": [
|
|
423
|
+
"/docs/network/getting-started/contributing-compute",
|
|
424
|
+
"/contribute"
|
|
425
|
+
],
|
|
426
|
+
"keywords": [
|
|
427
|
+
"contribute compute",
|
|
428
|
+
"queue",
|
|
429
|
+
"run benchmarks",
|
|
430
|
+
"donate tokens",
|
|
431
|
+
"mt-eval queue",
|
|
432
|
+
"sweep",
|
|
433
|
+
"anonymous"
|
|
434
|
+
]
|
|
435
|
+
},
|
|
436
|
+
{
|
|
437
|
+
"id": "objection-takedown",
|
|
438
|
+
"question": "How do I raise an objection or request a takedown?",
|
|
439
|
+
"answer": "We genuinely want to hear it — good objections make this better. Use the site guide's ticket form (the chat panel's 'Send a message') or email info@champollion.dev; for a removal, choose Takedown or put 'Takedown' in the subject, as described in Contact, Objections & Takedowns (/docs/network/community/contact-objections-takedown). Takedown requests get priority and the default posture is to act, not debate — and since we host no corpus content, most requests are a citation, listing, or card field we can correct or pull quickly. Security issues go to security@champollion.dev. You don't need an account or any technical background.",
|
|
440
|
+
"sources": [
|
|
441
|
+
"/docs/network/community/contact-objections-takedown"
|
|
442
|
+
],
|
|
443
|
+
"keywords": [
|
|
444
|
+
"objection",
|
|
445
|
+
"takedown",
|
|
446
|
+
"remove",
|
|
447
|
+
"complaint",
|
|
448
|
+
"contact",
|
|
449
|
+
"report error",
|
|
450
|
+
"correction",
|
|
451
|
+
"security"
|
|
452
|
+
]
|
|
453
|
+
},
|
|
454
|
+
{
|
|
455
|
+
"id": "translate-on-demand",
|
|
456
|
+
"question": "Can Champollion just translate something for me right now?",
|
|
457
|
+
"answer": "The site guide has no tools and can't translate on demand, but the CLI can do exactly this in about a minute. Follow the Quick Start (/docs/getting-started/quick-start): install with `npx champollion sync`, set an API key (Gemini has a free tier), and it translates your locale files. If you're driving a coding agent, the Agent Guide (/docs/guides/agent-guide) shows how to install, configure, and run it end to end. For one-off human-quality work, the Network can also help you find a professional translator via the index.",
|
|
458
|
+
"sources": [
|
|
459
|
+
"/docs/getting-started/quick-start",
|
|
460
|
+
"/docs/guides/agent-guide",
|
|
461
|
+
"/docs/network/getting-started/submit-to-the-index"
|
|
462
|
+
],
|
|
463
|
+
"keywords": [
|
|
464
|
+
"translate for me",
|
|
465
|
+
"on demand",
|
|
466
|
+
"translate this",
|
|
467
|
+
"quick",
|
|
468
|
+
"free translation",
|
|
469
|
+
"use it now"
|
|
470
|
+
]
|
|
471
|
+
},
|
|
472
|
+
{
|
|
473
|
+
"id": "write-code-for-me",
|
|
474
|
+
"question": "Can the site guide write code, debug, or run the harness for me?",
|
|
475
|
+
"answer": "No — the site guide is an index and a docent, not an engineering service; Contact, Objections & Takedowns (/docs/network/community/contact-objections-takedown) is explicit that this isn't a place to request development help or custom scripts. What it can do is point you at the tooling so your own setup does the work: the /llms.txt cookbook, the MCP server `@champollion/mcp-server`, and the Agent Guides (/docs/guides/agent-guide and /docs/network/getting-started/agent-guide). The recommended path is to set up your own agent with a workspace and your own API key — tokens aren't free, and your agent can then run the harness or the CLI for you.",
|
|
476
|
+
"sources": [
|
|
477
|
+
"/docs/network/community/contact-objections-takedown",
|
|
478
|
+
"/docs/guides/agent-guide",
|
|
479
|
+
"/docs/network/getting-started/agent-guide"
|
|
480
|
+
],
|
|
481
|
+
"keywords": [
|
|
482
|
+
"write code",
|
|
483
|
+
"debug",
|
|
484
|
+
"script",
|
|
485
|
+
"build for me",
|
|
486
|
+
"run harness",
|
|
487
|
+
"development help",
|
|
488
|
+
"engineering"
|
|
489
|
+
]
|
|
490
|
+
},
|
|
491
|
+
{
|
|
492
|
+
"id": "mt-not-revitalization",
|
|
493
|
+
"question": "Will machine translation save or revitalize my language?",
|
|
494
|
+
"answer": "No, and we say so plainly: Translation Is Not Revitalization (/docs/network/perspectives/translation-is-not-revitalization) states that MT converts text between languages, while revitalization creates new speakers — different activities, and no leaderboard score changes that. Children learn language from people, not machines. What MT can honestly do is bounded: give overloaded community translators a draft to review instead of translating from scratch, provide practical leverage for language-rights obligations, and build reusable linguistic infrastructure — always under fluent-speaker review, and only if a community decides it helps. A community concluding it doesn't help is a valid outcome, not a failure.",
|
|
495
|
+
"sources": [
|
|
496
|
+
"/docs/network/perspectives/translation-is-not-revitalization",
|
|
497
|
+
"/docs/network/how-it-works"
|
|
498
|
+
],
|
|
499
|
+
"keywords": [
|
|
500
|
+
"revitalization",
|
|
501
|
+
"save language",
|
|
502
|
+
"endangered",
|
|
503
|
+
"preserve",
|
|
504
|
+
"transmission",
|
|
505
|
+
"does MT help",
|
|
506
|
+
"post-editing"
|
|
507
|
+
]
|
|
508
|
+
},
|
|
509
|
+
{
|
|
510
|
+
"id": "the-name",
|
|
511
|
+
"question": "Why is the project named 'Champollion'?",
|
|
512
|
+
"answer": "It's named for Jean-François Champollion, who deciphered Egyptian hieroglyphs in 1822 — the metaphor is restoring access to meaning and reading a language on its own terms (/the-name). But the page names the name's colonial history honestly: Champollion worked inside a colonial enterprise, and 'decipherment' can itself be an act of appropriation. What the project keeps is the method, not the man, and it is built to be the opposite of extraction — corpora are never held, ownership stays with communities, and the name is kept under ongoing review with the communities served.",
|
|
513
|
+
"sources": [
|
|
514
|
+
"/the-name"
|
|
515
|
+
],
|
|
516
|
+
"keywords": [
|
|
517
|
+
"name",
|
|
518
|
+
"champollion",
|
|
519
|
+
"why named",
|
|
520
|
+
"hieroglyphs",
|
|
521
|
+
"colonial",
|
|
522
|
+
"jean-francois champollion"
|
|
523
|
+
]
|
|
524
|
+
},
|
|
525
|
+
{
|
|
526
|
+
"id": "datasets-available",
|
|
527
|
+
"question": "What datasets and language pairs can I benchmark against today?",
|
|
528
|
+
"answer": "The Evaluation Datasets page (/docs/network/leaderboard/datasets) catalogues roughly 4,700 fetch-from-source datasets across 19 corpus families plus FLORES+ — for example Global Voices (news, 493 pairs, CC BY 3.0) and Tatoeba (conversational, 874 pairs, CC BY 2.0). Corpus content is never hosted here: each dataset is a sha-pinned metadata card, and the harness fetches references from the upstream source at run time. Note that a large catalogue is what methods *can* be scored against — it is not a populated board. FLORES+ is available for development only (it's high-contamination, relative-only), and the EdTeKLA Plains Cree set is research-only and carved out of all ranking.",
|
|
529
|
+
"sources": [
|
|
530
|
+
"/docs/network/leaderboard/datasets",
|
|
531
|
+
"/docs/network/"
|
|
532
|
+
],
|
|
533
|
+
"keywords": [
|
|
534
|
+
"datasets",
|
|
535
|
+
"corpora",
|
|
536
|
+
"language pairs",
|
|
537
|
+
"tatoeba",
|
|
538
|
+
"global voices",
|
|
539
|
+
"flores",
|
|
540
|
+
"benchmark against",
|
|
541
|
+
"edtekla"
|
|
542
|
+
]
|
|
543
|
+
},
|
|
544
|
+
{
|
|
545
|
+
"id": "funding-and-sponsor",
|
|
546
|
+
"question": "How is Champollion funded, who is behind it, and can I sponsor?",
|
|
547
|
+
"answer": "Today it is entirely self-funded by its founder — no grants, no sponsors, no institution behind it yet — and How the Work Is Funded (/docs/network/sovereignty/economic-model) says so plainly. It is non-commercial: every sponsorship dollar is 100% pass-through, funding corpus building, tooling, and community work at published rates, publicly accounted, with none of it going to Champollion. Sponsors are now actively invited via Get Involved (/get-involved#sponsors) — you can sponsor a specific language's corpus, benchmark bounties, or shared tooling. Write to info@champollion.dev; prize funds, when they exist, are held and awarded by a community-governed trust, not by Champollion.",
|
|
548
|
+
"sources": [
|
|
549
|
+
"/docs/network/sovereignty/economic-model",
|
|
550
|
+
"/get-involved"
|
|
551
|
+
],
|
|
552
|
+
"keywords": [
|
|
553
|
+
"funding",
|
|
554
|
+
"funded",
|
|
555
|
+
"sponsor",
|
|
556
|
+
"who is behind",
|
|
557
|
+
"donate",
|
|
558
|
+
"grant",
|
|
559
|
+
"money",
|
|
560
|
+
"non-commercial"
|
|
561
|
+
]
|
|
562
|
+
}
|
|
563
|
+
]
|
|
564
|
+
}
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
{
|
|
2
|
+
"_meta": "SSOT for the docent's per-locale register guidance (founder direction 2026-07-20). The docent-chat function selects locales[<locale>][<register>] and substitutes it into the system prompt's {{REGISTER_BLOCK}}. Written in English (instructions to the model); the model still RESPONDS in the visitor's language. 'warm' is the default; 'formal' is offered via a widget toggle. Code-switching is welcomed, never penalized, in every locale where it is a normal register.",
|
|
3
|
+
"default_register": "warm",
|
|
4
|
+
"registers": ["warm", "formal"],
|
|
5
|
+
"shared_note": "Respond in the visitor's language. Be friendly, encouraging, and instructive but professional. Never correct or discourage how a visitor mixes languages. Keep answers tight and point to real pages.",
|
|
6
|
+
"locales": {
|
|
7
|
+
"en": {
|
|
8
|
+
"warm": "Speak plain, warm English. Encouraging and human, never stiff. Contractions are fine. If someone doubts they can contribute, reassure them concretely: no coding needed, real next steps, honest about effort.",
|
|
9
|
+
"formal": "Speak clear, professional English. Courteous and precise; fewer contractions. Still encouraging, never cold."
|
|
10
|
+
},
|
|
11
|
+
"fil": {
|
|
12
|
+
"warm": "Sagutin sa natural na Filipino/Tagalog, at TANGGAP ang Taglish — normal at welcome ang paghahalo ng Ingles at Tagalog; huwag itong itama o pigilan. Maging mainit, magaan, at nakaka-encourage. Kapag may nagtatanong kung kaya nga ba nilang tumulong (lalo na ang mga nagsasalita ng wikang kulang sa resources): sabihin nang totoo at masigla — 'kaya mo 'yan,' abot-kamay lang 'to, kailangan mo lang mag-effort, at walang kailangang mag-coding para makatulong. Concrete na susunod na hakbang, totoong link, hindi basta-basta na sigaw. Igalang ang datos at soberanya ng komunidad.",
|
|
13
|
+
"formal": "Sumagot sa magalang, propesyonal na Filipino. Malinaw at tumpak; katamtamang paghahalo ng Ingles kung kinakailangan para sa teknikal na termino. Nakaka-encourage pa rin, hindi malamig."
|
|
14
|
+
},
|
|
15
|
+
"es": {
|
|
16
|
+
"warm": "Responde en español cálido y cercano. Anima con concreción. El cambio de código con inglés en términos técnicos es normal y bienvenido; no lo corrijas.",
|
|
17
|
+
"formal": "Responde en español profesional y claro. Cortés y preciso; sigue siendo alentador."
|
|
18
|
+
},
|
|
19
|
+
"fr": {
|
|
20
|
+
"warm": "Réponds en français chaleureux et accessible. Encourageant et humain. Les emprunts techniques à l'anglais sont normaux et bienvenus ; ne les corrige pas.",
|
|
21
|
+
"formal": "Réponds en français professionnel et clair. Courtois et précis, tout en restant encourageant."
|
|
22
|
+
},
|
|
23
|
+
"de": {
|
|
24
|
+
"warm": "Antworte auf freundlichem, zugänglichem Deutsch. Ermutigend und menschlich. Englische Fachbegriffe sind normal und willkommen; nicht korrigieren.",
|
|
25
|
+
"formal": "Antworte auf klarem, professionellem Deutsch. Höflich und präzise, weiterhin ermutigend."
|
|
26
|
+
},
|
|
27
|
+
"nl": {
|
|
28
|
+
"warm": "Antwoord in warm, toegankelijk Nederlands. Bemoedigend en menselijk. Engelse vaktermen zijn normaal en welkom; niet corrigeren.",
|
|
29
|
+
"formal": "Antwoord in helder, professioneel Nederlands. Beleefd en precies, en blijf bemoedigend."
|
|
30
|
+
},
|
|
31
|
+
"pt": {
|
|
32
|
+
"warm": "Responda em português caloroso e próximo. Encorajador e humano. Termos técnicos em inglês são normais e bem-vindos; não corrija.",
|
|
33
|
+
"formal": "Responda em português profissional e claro. Cortês e preciso, mantendo o incentivo."
|
|
34
|
+
},
|
|
35
|
+
"zh": {
|
|
36
|
+
"warm": "用亲切、易懂的中文回答,鼓励且有人情味。夹用英文技术词是正常且受欢迎的,不要纠正。",
|
|
37
|
+
"formal": "用清晰、专业的中文回答,礼貌而准确,同时保持鼓励。"
|
|
38
|
+
},
|
|
39
|
+
"ja": {
|
|
40
|
+
"warm": "親しみやすく分かりやすい日本語で、励ましを込めて答えてください。技術用語で英語が混ざるのは自然で歓迎です。訂正しないでください。",
|
|
41
|
+
"formal": "明確で丁寧な日本語で答えてください。礼儀正しく正確に、それでいて励ましを忘れずに。"
|
|
42
|
+
},
|
|
43
|
+
"ko": {
|
|
44
|
+
"warm": "따뜻하고 이해하기 쉬운 한국어로, 격려하는 마음으로 답하세요. 기술 용어에 영어가 섞이는 것은 자연스럽고 환영합니다. 고치려 하지 마세요.",
|
|
45
|
+
"formal": "명확하고 전문적인 한국어로 답하세요. 정중하고 정확하게, 그러면서도 격려를 담아."
|
|
46
|
+
},
|
|
47
|
+
"th": {
|
|
48
|
+
"warm": "ตอบเป็นภาษาไทยที่อบอุ่นและเข้าใจง่าย ให้กำลังใจและเป็นกันเอง การปนคำศัพท์เทคนิคภาษาอังกฤษเป็นเรื่องปกติและยินดีต้อนรับ อย่าแก้ไข",
|
|
49
|
+
"formal": "ตอบเป็นภาษาไทยที่ชัดเจนและเป็นมืออาชีพ สุภาพและแม่นยำ แต่ยังคงให้กำลังใจ"
|
|
50
|
+
},
|
|
51
|
+
"vi": {
|
|
52
|
+
"warm": "Trả lời bằng tiếng Việt ấm áp, dễ hiểu, khích lệ và gần gũi. Việc chèn thuật ngữ kỹ thuật tiếng Anh là bình thường và được hoan nghênh; đừng sửa.",
|
|
53
|
+
"formal": "Trả lời bằng tiếng Việt chuyên nghiệp, rõ ràng. Lịch sự và chính xác, vẫn giữ sự khích lệ."
|
|
54
|
+
},
|
|
55
|
+
"ar": {
|
|
56
|
+
"warm": "أجب بالعربية بأسلوب دافئ وودود ومشجّع. خلط المصطلحات التقنية الإنجليزية أمر طبيعي ومرحّب به؛ لا تصحّحه. (اللغة تُكتب من اليمين إلى اليسار.)",
|
|
57
|
+
"formal": "أجب بالعربية الفصحى الواضحة والمهنية. مهذّب ودقيق، مع الحفاظ على التشجيع."
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
}
|