aontu 0.60.0 → 0.62.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -3
- package/dist/agentsmd.d.ts +1 -0
- package/dist/agentsmd.js +7 -1
- package/dist/agentsmd.js.map +1 -1
- package/dist/allow.d.ts +23 -0
- package/dist/allow.js +230 -0
- package/dist/allow.js.map +1 -0
- package/dist/aontu.d.ts +4 -2
- package/dist/aontu.js +4 -2
- package/dist/aontu.js.map +1 -1
- package/dist/cli.d.ts +10 -1
- package/dist/cli.js +1000 -47
- package/dist/cli.js.map +1 -1
- package/dist/format.d.ts +1 -0
- package/dist/format.js +154 -12
- package/dist/format.js.map +1 -1
- package/dist/helpdoc.d.ts +16 -0
- package/dist/helpdoc.js +59 -0
- package/dist/helpdoc.js.map +1 -0
- package/dist/hints.js +15 -2
- package/dist/hints.js.map +1 -1
- package/dist/lang.js +93 -40
- package/dist/lang.js.map +1 -1
- package/dist/lower.d.ts +3 -0
- package/dist/lower.js +3 -0
- package/dist/lower.js.map +1 -1
- package/dist/lsp.d.ts +1 -1
- package/dist/lsp.js +1 -1
- package/dist/lsp.js.map +1 -1
- package/dist/relation.d.ts +2 -0
- package/dist/relation.js +7 -1
- package/dist/relation.js.map +1 -1
- package/dist/render.js +26 -21
- package/dist/render.js.map +1 -1
- package/dist/sigdecl.js +1 -1
- package/dist/sigdecl.js.map +1 -1
- package/dist/std.js +112 -69
- package/dist/std.js.map +1 -1
- package/dist/template.d.ts +2 -1
- package/dist/template.js +51 -15
- package/dist/template.js.map +1 -1
- package/dist/tsconfig.tsbuildinfo +1 -1
- package/dist/val/AggFuncVal.js +2 -2
- package/dist/val/AggFuncVal.js.map +1 -1
- package/dist/val/CmpFuncVal.d.ts +20 -0
- package/dist/val/CmpFuncVal.js +245 -0
- package/dist/val/CmpFuncVal.js.map +1 -0
- package/dist/val/EachFuncVal.d.ts +1 -2
- package/dist/val/EachFuncVal.js +15 -29
- package/dist/val/EachFuncVal.js.map +1 -1
- package/dist/val/FormFuncVal.js.map +1 -1
- package/dist/val/LowerFuncVal.js +17 -1
- package/dist/val/LowerFuncVal.js.map +1 -1
- package/dist/val/NamerFuncVal.d.ts +12 -0
- package/dist/val/NamerFuncVal.js +176 -0
- package/dist/val/NamerFuncVal.js.map +1 -0
- package/dist/val/NilVal.js +29 -3
- package/dist/val/NilVal.js.map +1 -1
- package/dist/val/NomFuncVal.d.ts +12 -0
- package/dist/val/NomFuncVal.js +187 -0
- package/dist/val/NomFuncVal.js.map +1 -0
- package/dist/val/PackFuncVal.js +3 -3
- package/dist/val/PackFuncVal.js.map +1 -1
- package/dist/val/RefVal.js +1 -1
- package/dist/val/TranslateFuncVal.d.ts +12 -0
- package/dist/val/TranslateFuncVal.js +101 -0
- package/dist/val/TranslateFuncVal.js.map +1 -0
- package/dist/val/UpperFuncVal.js +17 -1
- package/dist/val/UpperFuncVal.js.map +1 -1
- package/dist/val/caserange.d.ts +3 -0
- package/dist/val/caserange.js +110 -0
- package/dist/val/caserange.js.map +1 -0
- package/dist/vet.d.ts +12 -0
- package/dist/vet.js +209 -1
- package/dist/vet.js.map +1 -1
- package/grammar/aontu.abnf +1 -1
- package/grammar/aontu.gbnf +1 -1
- package/grammar/aontu.lark +1 -1
- package/grammar/aontu.tmLanguage.json +1 -1
- package/package.json +1 -1
- package/skill/SKILL.md +8 -0
- package/skill/init/check.sh +28 -0
- package/skill/init/data.aon +12 -0
- package/skill/init/model.aon +19 -0
- package/skill/tasks.md +151 -0
- package/src/agentsmd.ts +13 -2
- package/src/allow.ts +316 -0
- package/src/aontu.ts +10 -1
- package/src/cli.ts +1131 -53
- package/src/format.ts +191 -14
- package/src/helpdoc.ts +77 -0
- package/src/hints.ts +16 -2
- package/src/lang.ts +98 -42
- package/src/lower.ts +3 -3
- package/src/lsp.ts +1 -1
- package/src/relation.ts +21 -1
- package/src/render.ts +26 -21
- package/src/sigdecl.ts +1 -1
- package/src/std.ts +112 -69
- package/src/template.ts +55 -15
- package/src/val/AggFuncVal.ts +2 -2
- package/src/val/CmpFuncVal.ts +411 -0
- package/src/val/EachFuncVal.ts +49 -50
- package/src/val/LowerFuncVal.ts +18 -1
- package/src/val/NilVal.ts +29 -3
- package/src/val/NomFuncVal.ts +287 -0
- package/src/val/PackFuncVal.ts +3 -3
- package/src/val/RefVal.ts +1 -1
- package/src/val/TranslateFuncVal.ts +182 -0
- package/src/val/UpperFuncVal.ts +18 -1
- package/src/val/caserange.ts +115 -0
- package/src/vet.ts +287 -1
- package/src/val/FormFuncVal.ts +0 -119
|
@@ -0,0 +1,287 @@
|
|
|
1
|
+
/* Copyright (c) 2026 Richard Rodger, MIT License */
|
|
2
|
+
|
|
3
|
+
// `nom` -- NAME TRANSFORMATION, THE GENERAL CASE (SPIKE,
|
|
4
|
+
// docs/design/JOSTRACA.0.md).
|
|
5
|
+
//
|
|
6
|
+
// Generated code is mostly names, and no two targets spell them the
|
|
7
|
+
// same way: one model field is `user_id` in SQL, `userId` in
|
|
8
|
+
// TypeScript, `UserID` in Go, `USER_ID` in an environment variable and
|
|
9
|
+
// `user-id` in a URL. Before this, aontu had whole-string `upper` and
|
|
10
|
+
// `lower` and nothing else, so the spike's own worked example wrote
|
|
11
|
+
// `Planet` out by hand rather than deriving it -- `upper("planet")` is
|
|
12
|
+
// `PLANET`.
|
|
13
|
+
//
|
|
14
|
+
// nom(s) every spelling, as a map
|
|
15
|
+
// nom(s, style) one spelling
|
|
16
|
+
// nom(s, acronyms) every spelling, with an acronym set
|
|
17
|
+
// nom(s, style, acronyms) one spelling, with an acronym set
|
|
18
|
+
//
|
|
19
|
+
// THE SOURCE FORMAT IS NOT DECLARED, and that is what makes this the
|
|
20
|
+
// general case rather than a family of pairwise converters. A name is
|
|
21
|
+
// SPLIT INTO WORDS first -- `user_id`, `userId`, `UserID`, `USER_ID`,
|
|
22
|
+
// `user-id`, `user.id` and `user/id` all split to `[user, id]` -- and
|
|
23
|
+
// then rendered in the target style. N formats in and M out is one
|
|
24
|
+
// splitter and M renderers, not N*M.
|
|
25
|
+
//
|
|
26
|
+
// THE SPLITTER IS THE RENDERER'S OWN (`splitWords`, ts/src/lower.ts):
|
|
27
|
+
// the same one `aontu:profile`'s `%case` uses to lower a declaration,
|
|
28
|
+
// so a name derived here and a name the renderer derives cannot
|
|
29
|
+
// disagree. It breaks at `_`, `-` and space, at a lower-to-upper
|
|
30
|
+
// boundary, at a letter-to-digit boundary either way, and before the
|
|
31
|
+
// last capital of a capital run a lower case letter follows
|
|
32
|
+
// (`HTTPServer` is HTTP, Server).
|
|
33
|
+
//
|
|
34
|
+
// `.` AND `/` ARE NORMALISED HERE, not in the splitter. A dotted path
|
|
35
|
+
// and a slash path are name formats an author converts FROM, but the
|
|
36
|
+
// splitter's behaviour is pinned in cross-port parity by the
|
|
37
|
+
// renderer's rows, so the extra separators are this function's
|
|
38
|
+
// vocabulary and are folded to `_` before it is asked. The same
|
|
39
|
+
// boundary applies to the four styles below that `caseName` does not
|
|
40
|
+
// serve: `title`, `text`, `dot` and `path` are nom's, and
|
|
41
|
+
// `aontu:profile`'s `%case` set is unchanged.
|
|
42
|
+
//
|
|
43
|
+
// THE ACRONYM SET is the reason a style alone is not enough. `ledgerId`
|
|
44
|
+
// is `LedgerID` in Go and `ledgerId` in TypeScript, which is a fact
|
|
45
|
+
// about the TARGET and not about the name -- so it is an argument, and
|
|
46
|
+
// a document generating for two targets passes a different one to each.
|
|
47
|
+
//
|
|
48
|
+
// SPIKE SCOPE. TypeScript only, so -- as with the component primitives
|
|
49
|
+
// -- deliberately absent from test/spec/signature.tsv, BUILTIN_FUNCS
|
|
50
|
+
// and grammar/, all pinned in cross-port parity. Arity and argument
|
|
51
|
+
// shape are refused here rather than by the parse-time table or the
|
|
52
|
+
// signature gate, and every refusal is `invalid-arg`, because a new
|
|
53
|
+
// code needs a row in test/spec/errcodes.tsv and an entry in BOTH
|
|
54
|
+
// ports' tables.
|
|
55
|
+
|
|
56
|
+
import type {
|
|
57
|
+
Val,
|
|
58
|
+
ValSpec,
|
|
59
|
+
} from '../type'
|
|
60
|
+
|
|
61
|
+
import {
|
|
62
|
+
AontuContext,
|
|
63
|
+
} from '../ctx'
|
|
64
|
+
|
|
65
|
+
import { makeNilErr } from '../err'
|
|
66
|
+
|
|
67
|
+
import {
|
|
68
|
+
splitWords,
|
|
69
|
+
caseName,
|
|
70
|
+
lowerASCII,
|
|
71
|
+
capitalise,
|
|
72
|
+
} from '../lower'
|
|
73
|
+
|
|
74
|
+
import { MapVal } from './MapVal'
|
|
75
|
+
import { StringVal } from './StringVal'
|
|
76
|
+
import { FuncBaseVal } from './FuncBaseVal'
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
// The styles, and the map's keys. The first five are `caseName`'s --
|
|
80
|
+
// `aontu:profile`'s `%case` vocabulary, shared with the renderer --
|
|
81
|
+
// and the last four are nom's own (see the note above). `as-is` is
|
|
82
|
+
// not among them: it is the profile's way of saying "do nothing",
|
|
83
|
+
// which is not a spelling anyone asks a namer for.
|
|
84
|
+
const NOM_STYLES = [
|
|
85
|
+
'camel', // userId
|
|
86
|
+
'dot', // user.id
|
|
87
|
+
'kebab', // user-id
|
|
88
|
+
'pascal', // UserId
|
|
89
|
+
'path', // user/id
|
|
90
|
+
'snake', // user_id
|
|
91
|
+
'text', // User id
|
|
92
|
+
'title', // User Id
|
|
93
|
+
'upper', // USER_ID
|
|
94
|
+
]
|
|
95
|
+
|
|
96
|
+
// nom's style name -> the `%case` style `caseName` serves. `upper` and
|
|
97
|
+
// `text` are nom's spellings: `screaming` is what `aontu:profile` calls
|
|
98
|
+
// SCREAMING_SNAKE and that name is pinned cross-port, so the mapping
|
|
99
|
+
// lives here rather than in the shared vocabulary.
|
|
100
|
+
const CASENAME_STYLES: Record<string, string> = {
|
|
101
|
+
camel: 'camel',
|
|
102
|
+
kebab: 'kebab',
|
|
103
|
+
pascal: 'pascal',
|
|
104
|
+
snake: 'snake',
|
|
105
|
+
upper: 'screaming',
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
// One name in one style, or undefined when the style is not one, or
|
|
110
|
+
// when the name holds no words at all.
|
|
111
|
+
function styleName(name: string, style: string, acronyms: string[]):
|
|
112
|
+
string | undefined {
|
|
113
|
+
// `.` and `/` are nom's separators, folded before the shared
|
|
114
|
+
// splitter is asked (see the note above). Everything else that is
|
|
115
|
+
// not a separator is word content: a `$` or a `@` rides into the
|
|
116
|
+
// word it sits in, which is why `nom` renames a spelled path's
|
|
117
|
+
// text and does not tidy it.
|
|
118
|
+
const src = name.replace(/[./]/g, '_')
|
|
119
|
+
|
|
120
|
+
// A NAME WITH NO WORDS IS NOT A NAME, and it is refused in every
|
|
121
|
+
// style. `caseName` answers its INPUT for one (it is lowering a
|
|
122
|
+
// declaration, where the name has already been vetted), so
|
|
123
|
+
// `nom("_", pascal)` came back as `"_"` while `nom("_")`
|
|
124
|
+
// refused -- the same argument, accepted by one spelling of the
|
|
125
|
+
// call and refused by the other.
|
|
126
|
+
const words = splitWords(src)
|
|
127
|
+
if (0 === words.length) {
|
|
128
|
+
return undefined
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
const cased = CASENAME_STYLES[style]
|
|
132
|
+
if (undefined !== cased) {
|
|
133
|
+
return caseName(src, cased, acronyms)
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
if ('dot' === style) {
|
|
137
|
+
return words.map(lowerASCII).join('.')
|
|
138
|
+
}
|
|
139
|
+
if ('path' === style) {
|
|
140
|
+
return words.map(lowerASCII).join('/')
|
|
141
|
+
}
|
|
142
|
+
if ('title' === style) {
|
|
143
|
+
return words.map((w) => capitalise(w, acronyms)).join(' ')
|
|
144
|
+
}
|
|
145
|
+
if ('text' === style) {
|
|
146
|
+
// Sentence case: the first word capitalised, the rest lower --
|
|
147
|
+
// EXCEPT an acronym, which stays one, because `ledger id` loses
|
|
148
|
+
// what `ID` was. Membership in the set decides that, not how the
|
|
149
|
+
// input happened to spell the word: asking whether `capitalise`
|
|
150
|
+
// changed it made `nom("ledgerId", text, [ID])` answer
|
|
151
|
+
// `Ledger id` while `nom("ledgerID", text, [ID])` answered
|
|
152
|
+
// `Ledger ID` -- the same name, two answers, decided by its
|
|
153
|
+
// source spelling, which is the one thing a namer must not do.
|
|
154
|
+
const isAcronym = (w: string) =>
|
|
155
|
+
acronyms.some((a) => lowerASCII(a) === lowerASCII(w))
|
|
156
|
+
return [capitalise(words[0], acronyms)]
|
|
157
|
+
.concat(words.slice(1).map((w) =>
|
|
158
|
+
isAcronym(w) ? capitalise(w, acronyms) : lowerASCII(w)))
|
|
159
|
+
.join(' ')
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
return undefined
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
// The text a value carries, when it is a concrete string. A path is a
|
|
167
|
+
// string too (ScalarKindVal's Path sits under String), so a spelled
|
|
168
|
+
// address may be renamed like any other text.
|
|
169
|
+
function textOf(v: Val | undefined): string | undefined {
|
|
170
|
+
const s: any = v
|
|
171
|
+
return (true === s?.isScalar && 'string' === typeof s.peg) ? s.peg : undefined
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
// An acronym set: a list of concrete strings, or undefined when the
|
|
176
|
+
// value is not one.
|
|
177
|
+
function acronymsOf(v: Val | undefined): string[] | undefined {
|
|
178
|
+
const l: any = v
|
|
179
|
+
if (true !== l?.isList) {
|
|
180
|
+
return undefined
|
|
181
|
+
}
|
|
182
|
+
const out: string[] = []
|
|
183
|
+
for (const el of l.peg as Val[]) {
|
|
184
|
+
const t = textOf(el)
|
|
185
|
+
if (undefined === t || '' === t) {
|
|
186
|
+
return undefined
|
|
187
|
+
}
|
|
188
|
+
out.push(t)
|
|
189
|
+
}
|
|
190
|
+
return out
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
class NomFuncVal extends FuncBaseVal {
|
|
195
|
+
isNamerFunc = true
|
|
196
|
+
|
|
197
|
+
constructor(spec: ValSpec, ctx?: AontuContext) {
|
|
198
|
+
super(spec, ctx)
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
make(_ctx: AontuContext, spec: ValSpec): Val {
|
|
202
|
+
return new NomFuncVal(spec)
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
funcname() {
|
|
206
|
+
return 'nom'
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
resolve(ctx: AontuContext, args: Val[]): Val {
|
|
211
|
+
if (args.length < 1 || 3 < args.length) {
|
|
212
|
+
return makeNilErr(ctx, 'invalid-arg', this, undefined, 'arity')
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
const name = textOf(args[0])
|
|
216
|
+
if (undefined === name || '' === name) {
|
|
217
|
+
return makeNilErr(ctx, 'invalid-arg', this, args[0], 'name')
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
// THE SECOND ARGUMENT SAYS WHICH OF THE FOUR CALLS THIS IS, by
|
|
221
|
+
// its shape rather than by its position: a STRING is the style, a
|
|
222
|
+
// LIST is the acronym set. The same rule the component primitives
|
|
223
|
+
// read their spec by, and it is what keeps the acronym set
|
|
224
|
+
// reachable from the map form without a placeholder argument.
|
|
225
|
+
let style: string | undefined
|
|
226
|
+
let acronyms: string[] = []
|
|
227
|
+
|
|
228
|
+
if (2 <= args.length) {
|
|
229
|
+
const second: any = args[1]
|
|
230
|
+
if (true === second?.isList) {
|
|
231
|
+
if (3 === args.length) {
|
|
232
|
+
return makeNilErr(ctx, 'invalid-arg', this, args[2], 'arity')
|
|
233
|
+
}
|
|
234
|
+
const acr = acronymsOf(second)
|
|
235
|
+
if (undefined === acr) {
|
|
236
|
+
return makeNilErr(ctx, 'invalid-arg', this, second, 'acronyms')
|
|
237
|
+
}
|
|
238
|
+
acronyms = acr
|
|
239
|
+
}
|
|
240
|
+
else {
|
|
241
|
+
style = textOf(second)
|
|
242
|
+
if (undefined === style) {
|
|
243
|
+
return makeNilErr(ctx, 'invalid-arg', this, second, 'style')
|
|
244
|
+
}
|
|
245
|
+
if (3 === args.length) {
|
|
246
|
+
const acr = acronymsOf(args[2])
|
|
247
|
+
if (undefined === acr) {
|
|
248
|
+
return makeNilErr(ctx, 'invalid-arg', this, args[2], 'acronyms')
|
|
249
|
+
}
|
|
250
|
+
acronyms = acr
|
|
251
|
+
}
|
|
252
|
+
}
|
|
253
|
+
}
|
|
254
|
+
|
|
255
|
+
// One style: the string.
|
|
256
|
+
if (undefined !== style) {
|
|
257
|
+
const out = styleName(name, style, acronyms)
|
|
258
|
+
if (undefined === out) {
|
|
259
|
+
return makeNilErr(ctx, 'invalid-arg', this, args[1], 'style')
|
|
260
|
+
}
|
|
261
|
+
return this.place(new StringVal({ peg: out }, ctx))
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
// Every style: the map. Closed, because the nine keys ARE the
|
|
265
|
+
// vocabulary and a tenth is a typo -- `nom($.n).pascel` is
|
|
266
|
+
// refused where every other mistake in an aontu document is.
|
|
267
|
+
const peg: Record<string, Val> = {}
|
|
268
|
+
for (const s of NOM_STYLES) {
|
|
269
|
+
const out = styleName(name, s, acronyms)
|
|
270
|
+
if (undefined === out) {
|
|
271
|
+
return makeNilErr(ctx, 'invalid-arg', this, args[0], 'name')
|
|
272
|
+
}
|
|
273
|
+
peg[s] = new StringVal({ peg: out }, ctx)
|
|
274
|
+
}
|
|
275
|
+
const map = new MapVal({ peg }, ctx)
|
|
276
|
+
map.closed = true
|
|
277
|
+
|
|
278
|
+
return this.place(map)
|
|
279
|
+
}
|
|
280
|
+
|
|
281
|
+
} /* node:coverage ignore next 6 */
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
export {
|
|
285
|
+
NOM_STYLES,
|
|
286
|
+
NomFuncVal,
|
|
287
|
+
}
|
package/src/val/PackFuncVal.ts
CHANGED
|
@@ -48,9 +48,9 @@ import { bagMembers } from './members'
|
|
|
48
48
|
|
|
49
49
|
|
|
50
50
|
// The keys a data bag names, in the order the result must carry them,
|
|
51
|
-
// or a code naming what is wrong with it.
|
|
52
|
-
//
|
|
53
|
-
//
|
|
51
|
+
// or a code naming what is wrong with it. `each` asks the same
|
|
52
|
+
// question of the same argument and answers it with the values rather
|
|
53
|
+
// than the keys (members.ts).
|
|
54
54
|
function dataKeys(data: Val | undefined, ctx: AontuContext): string[] | string {
|
|
55
55
|
// The candidates are the bag's MEMBERS -- what generation would
|
|
56
56
|
// emit (./members.ts, BUGS.md §79) -- so a hidden key, or a hidden
|
package/src/val/RefVal.ts
CHANGED
|
@@ -657,7 +657,7 @@ class RefVal extends FeatureVal {
|
|
|
657
657
|
// hidden IN ITS OWN RIGHT, and the verb's enumeration
|
|
658
658
|
// (ts/src/val/members.ts, BUGS.md §79) needs to see that
|
|
659
659
|
// mark to leave the member out, as generation does. A marked
|
|
660
|
-
// target lifts even there, or `
|
|
660
|
+
// target lifts even there, or `form($.schema.entities, _)`
|
|
661
661
|
// under `schema: hide({...})` would see every entity as
|
|
662
662
|
// hidden, since hide() marks to the leaves.
|
|
663
663
|
const lifted = true !== (ctx as any).argsnap
|
|
@@ -0,0 +1,182 @@
|
|
|
1
|
+
/* Copyright (c) 2026 Richard Rodger, MIT License */
|
|
2
|
+
|
|
3
|
+
// `translate` -- PER-CHARACTER SUBSTITUTION AND DELETION (SPIKE,
|
|
4
|
+
// docs/design/JOSTRACA.0.md), after the `tr` command.
|
|
5
|
+
//
|
|
6
|
+
// translate(s, from, to) each character of `from` becomes the one
|
|
7
|
+
// at the same position in `to`
|
|
8
|
+
// translate(s, from) each character of `from` is deleted
|
|
9
|
+
//
|
|
10
|
+
// WHY IT IS NOT `rep`. `rep(s, pattern, sub)` matches a REGION and
|
|
11
|
+
// substitutes text for it, so a per-character map is N calls over N
|
|
12
|
+
// passes, each seeing the previous one's output -- and that composition
|
|
13
|
+
// is wrong, not merely slow: `rep(rep(x, "a", "b"), "b", "a")` maps
|
|
14
|
+
// every original `a` to `a` again. `translate` reads the source once
|
|
15
|
+
// and consults a table, so `translate(x, "ab", "ba")` SWAPS them, which
|
|
16
|
+
// is the operation `tr` exists for and the one a generator wants when
|
|
17
|
+
// it rewrites a delimiter set or strips a character class.
|
|
18
|
+
//
|
|
19
|
+
// RANGES, as `tr` has them: `a-z` is every code point from `a` to `z`
|
|
20
|
+
// inclusive, in both sets. A `-` first or last in a set is itself,
|
|
21
|
+
// which is the only way to mean a literal one -- the sets take no
|
|
22
|
+
// escape, because an aontu string literal has already processed its
|
|
23
|
+
// own and a second escape layer over the first is a trap rather than a
|
|
24
|
+
// feature. A descending range (`z-a`) is refused rather than read as
|
|
25
|
+
// empty: it is always a mistake, and answering nothing for it hides it.
|
|
26
|
+
//
|
|
27
|
+
// A SHORT `to` PADS WITH ITS LAST CHARACTER, which is `tr`'s rule:
|
|
28
|
+
// `translate(s, "abc", "x")` maps all three to `x`. An EMPTY or absent
|
|
29
|
+
// `to` deletes instead, which is `tr -d`. Those are the same rule read
|
|
30
|
+
// two ways -- there is no last character to pad with -- so the second
|
|
31
|
+
// argument alone is deletion and needs no flag.
|
|
32
|
+
//
|
|
33
|
+
// A CHARACTER NAMED TWICE IN `from` TAKES ITS LAST MAPPING, because the
|
|
34
|
+
// table is built left to right and a later entry overwrites an earlier.
|
|
35
|
+
// `tr` does the same.
|
|
36
|
+
//
|
|
37
|
+
// CODE POINTS, NOT UTF-16 UNITS: one entry is one code point, so an
|
|
38
|
+
// astral character maps as a unit and cannot be half-matched.
|
|
39
|
+
//
|
|
40
|
+
// NOT DONE, and deliberately: `tr`'s `-s` (squeeze repeats), `-c`
|
|
41
|
+
// (complement) and its character classes (`[:alpha:]`). Each is a
|
|
42
|
+
// second vocabulary on top of the sets, and none is needed by anything
|
|
43
|
+
// the spike found. `re()` already spells a class for `rep`.
|
|
44
|
+
//
|
|
45
|
+
// SPIKE SCOPE. TypeScript only, so -- as with `nom` and the component
|
|
46
|
+
// primitives -- deliberately absent from test/spec/signature.tsv,
|
|
47
|
+
// BUILTIN_FUNCS and grammar/, all pinned in cross-port parity. Arity
|
|
48
|
+
// and argument shape are refused here, and every refusal is
|
|
49
|
+
// `invalid-arg`.
|
|
50
|
+
|
|
51
|
+
import type {
|
|
52
|
+
Val,
|
|
53
|
+
ValSpec,
|
|
54
|
+
} from '../type'
|
|
55
|
+
|
|
56
|
+
import {
|
|
57
|
+
AontuContext,
|
|
58
|
+
} from '../ctx'
|
|
59
|
+
|
|
60
|
+
import { makeNilErr } from '../err'
|
|
61
|
+
|
|
62
|
+
import { StringVal } from './StringVal'
|
|
63
|
+
import { FuncBaseVal } from './FuncBaseVal'
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
// A set with its ranges expanded, or undefined when a range descends.
|
|
67
|
+
// `-` first or last is a literal.
|
|
68
|
+
function expandSet(set: string): string[] | undefined {
|
|
69
|
+
const cps = Array.from(set)
|
|
70
|
+
const out: string[] = []
|
|
71
|
+
|
|
72
|
+
for (let i = 0; i < cps.length; i++) {
|
|
73
|
+
const c = cps[i]
|
|
74
|
+
const next = cps[i + 1]
|
|
75
|
+
const after = cps[i + 2]
|
|
76
|
+
|
|
77
|
+
if ('-' === next && undefined !== after && 0 < i + 2) {
|
|
78
|
+
const lo = c.codePointAt(0) as number
|
|
79
|
+
const hi = after.codePointAt(0) as number
|
|
80
|
+
if (hi < lo) {
|
|
81
|
+
return undefined
|
|
82
|
+
}
|
|
83
|
+
for (let cp = lo; cp <= hi; cp++) {
|
|
84
|
+
out.push(String.fromCodePoint(cp))
|
|
85
|
+
}
|
|
86
|
+
i += 2
|
|
87
|
+
continue
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
out.push(c)
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
return out
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
// The text a value carries, when it is a concrete string.
|
|
98
|
+
function textOf(v: Val | undefined): string | undefined {
|
|
99
|
+
const s: any = v
|
|
100
|
+
return (true === s?.isScalar && 'string' === typeof s.peg) ? s.peg : undefined
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
class TranslateFuncVal extends FuncBaseVal {
|
|
105
|
+
isTranslateFunc = true
|
|
106
|
+
|
|
107
|
+
constructor(spec: ValSpec, ctx?: AontuContext) {
|
|
108
|
+
super(spec, ctx)
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
make(_ctx: AontuContext, spec: ValSpec): Val {
|
|
112
|
+
return new TranslateFuncVal(spec)
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
funcname() {
|
|
116
|
+
return 'translate'
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
resolve(ctx: AontuContext, args: Val[]): Val {
|
|
121
|
+
if (args.length < 2 || 3 < args.length) {
|
|
122
|
+
return makeNilErr(ctx, 'invalid-arg', this, undefined, 'arity')
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
const src = textOf(args[0])
|
|
126
|
+
if (undefined === src) {
|
|
127
|
+
return makeNilErr(ctx, 'invalid-arg', this, args[0], 'src')
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
const fromText = textOf(args[1])
|
|
131
|
+
if (undefined === fromText) {
|
|
132
|
+
return makeNilErr(ctx, 'invalid-arg', this, args[1], 'from')
|
|
133
|
+
}
|
|
134
|
+
const from = expandSet(fromText)
|
|
135
|
+
if (undefined === from) {
|
|
136
|
+
return makeNilErr(ctx, 'invalid-arg', this, args[1], 'from')
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
let to: string[] = []
|
|
140
|
+
if (3 === args.length) {
|
|
141
|
+
const toText = textOf(args[2])
|
|
142
|
+
if (undefined === toText) {
|
|
143
|
+
return makeNilErr(ctx, 'invalid-arg', this, args[2], 'to')
|
|
144
|
+
}
|
|
145
|
+
const expanded = expandSet(toText)
|
|
146
|
+
if (undefined === expanded) {
|
|
147
|
+
return makeNilErr(ctx, 'invalid-arg', this, args[2], 'to')
|
|
148
|
+
}
|
|
149
|
+
to = expanded
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
// The table, built left to right so a character named twice takes
|
|
153
|
+
// its last mapping. An empty `to` maps to nothing, which is the
|
|
154
|
+
// deletion: there is no last character to pad with.
|
|
155
|
+
const pad = 0 < to.length ? to[to.length - 1] : undefined
|
|
156
|
+
const table = new Map<string, string | undefined>()
|
|
157
|
+
for (let i = 0; i < from.length; i++) {
|
|
158
|
+
table.set(from[i], i < to.length ? to[i] : pad)
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
let out = ''
|
|
162
|
+
for (const c of Array.from(src)) {
|
|
163
|
+
if (table.has(c)) {
|
|
164
|
+
const sub = table.get(c)
|
|
165
|
+
if (undefined !== sub) {
|
|
166
|
+
out += sub
|
|
167
|
+
}
|
|
168
|
+
continue
|
|
169
|
+
}
|
|
170
|
+
out += c
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
return this.place(new StringVal({ peg: out }, ctx))
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
} /* node:coverage ignore next 6 */
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
export {
|
|
180
|
+
expandSet,
|
|
181
|
+
TranslateFuncVal,
|
|
182
|
+
}
|
package/src/val/UpperFuncVal.ts
CHANGED
|
@@ -16,6 +16,7 @@ import { makeNilErr } from '../err'
|
|
|
16
16
|
import { ScalarKindVal } from '../val/ScalarKindVal'
|
|
17
17
|
import { makeScalarLike } from '../val/valutil'
|
|
18
18
|
import { Decimal } from '../val/Decimal'
|
|
19
|
+
import { caseRange, rangeArg } from './caserange'
|
|
19
20
|
|
|
20
21
|
|
|
21
22
|
|
|
@@ -49,7 +50,23 @@ class UpperFuncVal extends FuncBaseVal {
|
|
|
49
50
|
// internal error.
|
|
50
51
|
const arg = args?.[0]
|
|
51
52
|
const oldpeg = arg?.peg
|
|
52
|
-
|
|
53
|
+
|
|
54
|
+
// THE RANGE (ts/src/val/caserange.ts): `start` names the first
|
|
55
|
+
// character of the run when it is zero or positive and the last
|
|
56
|
+
// when it is negative; `len` of -1, and the absent argument, are
|
|
57
|
+
// the source's length. Refused on a NUMBER, where a run of
|
|
58
|
+
// characters means nothing -- the numeric arm below is a ceiling,
|
|
59
|
+
// not a case mapping.
|
|
60
|
+
const start = rangeArg(args?.[1])
|
|
61
|
+
const len = rangeArg(args?.[2])
|
|
62
|
+
const ranged = undefined !== start || undefined !== len
|
|
63
|
+
if (ranged &&
|
|
64
|
+
(Number.isNaN(start as number) || Number.isNaN(len as number) ||
|
|
65
|
+
'string' !== typeof oldpeg)) {
|
|
66
|
+
return this.place(makeNilErr(ctx, 'invalid-arg', this, arg, 'range'))
|
|
67
|
+
}
|
|
68
|
+
const peg = 'string' === typeof oldpeg ?
|
|
69
|
+
caseRange(oldpeg, start ?? 0, len ?? -1, true) :
|
|
53
70
|
'number' === typeof oldpeg ? Math.ceil(oldpeg) :
|
|
54
71
|
// The exact leaves take an EXACT ceiling and keep their kind: a
|
|
55
72
|
// biginteger is already integral so it is its own ceiling, and a
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
/* Copyright (c) 2026 Richard Rodger, MIT License */
|
|
2
|
+
|
|
3
|
+
// THE RANGE `upper` AND `lower` TAKE.
|
|
4
|
+
//
|
|
5
|
+
// upper(s) the whole string
|
|
6
|
+
// upper(s, start) from `start` to the end
|
|
7
|
+
// upper(s, start, len) `len` characters
|
|
8
|
+
//
|
|
9
|
+
// `start` IS A BOUNDARY, NOT A CHARACTER. Zero or positive, it is
|
|
10
|
+
// where the run BEGINS and the run reaches forward; negative, it counts
|
|
11
|
+
// from the end and is where the run STOPS -- the character it lands on
|
|
12
|
+
// is the first one NOT modified. One index, two directions, and no
|
|
13
|
+
// second argument to say which:
|
|
14
|
+
//
|
|
15
|
+
// upper("foo", 0, 1) "Foo" begins at 0, one character
|
|
16
|
+
// lower("FOOBAR", -3, -1) "fooBAR" stops before the last three
|
|
17
|
+
//
|
|
18
|
+
// `len` of -1 means THE LENGTH OF THE SOURCE, which is how "and the
|
|
19
|
+
// rest of it" is spelled in either direction, and is also the default:
|
|
20
|
+
//
|
|
21
|
+
// lower("FOO", 1, -1) "Foo" from 1 to the end
|
|
22
|
+
// lower("FOOBAR", -1) "foobaR" everything but the last
|
|
23
|
+
//
|
|
24
|
+
// The exclusive reading is why `-0` is not a spelling: it is `0`, and
|
|
25
|
+
// so begins a forward run. The whole string is `upper(s)`.
|
|
26
|
+
//
|
|
27
|
+
// BOTH ENDS CLAMP rather than refuse. A run reaching past either end
|
|
28
|
+
// modifies as much of the string as exists, and one that lands wholly
|
|
29
|
+
// outside it modifies nothing — an out-of-range index is not a
|
|
30
|
+
// different KIND of answer, it is the same answer over a shorter run.
|
|
31
|
+
//
|
|
32
|
+
// CODE POINTS, NOT UTF-16 UNITS, so an index means the same thing in
|
|
33
|
+
// both ports: one index is one Go rune. Twin: caseSpan/caseRange in
|
|
34
|
+
// go/func.go.
|
|
35
|
+
//
|
|
36
|
+
// THE CASE MAPPING IS THE WHOLE-STRING ONE, applied to the slice, so
|
|
37
|
+
// `upper(s)` and `upper(s, 0, -1)` are the same bytes. Two consequences
|
|
38
|
+
// follow from full Unicode case mapping and are properties of the
|
|
39
|
+
// mapping rather than of this range:
|
|
40
|
+
//
|
|
41
|
+
// - THE RESULT MAY BE LONGER than the source, because full mapping is
|
|
42
|
+
// not one-in-one-out: `upper("straße", 3, 3)` is "straSSE".
|
|
43
|
+
// - FINAL SIGMA IS DECIDED WITHIN THE SLICE, since a slice taken out
|
|
44
|
+
// of its word has no following letter to see. A medial sigma cased
|
|
45
|
+
// alone lowercases as a final one.
|
|
46
|
+
//
|
|
47
|
+
// Both are stated here rather than worked around: silently widening
|
|
48
|
+
// the slice to keep context would make the run something other than
|
|
49
|
+
// what the author asked for.
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
// The half-open code-point span [lo, hi) a (start, len) selects over a
|
|
53
|
+
// string of `n` code points.
|
|
54
|
+
export function caseSpan(n: number, start: number, len: number): [number, number] {
|
|
55
|
+
// -1 is the source's length, and so is any other negative: there is
|
|
56
|
+
// no meaningful run of "minus two" characters, and refusing one would
|
|
57
|
+
// be a second rule for no gain.
|
|
58
|
+
const span = len < 0 ? n : len
|
|
59
|
+
|
|
60
|
+
if (0 <= start) {
|
|
61
|
+
const lo = Math.min(start, n)
|
|
62
|
+
return [lo, Math.min(lo + span, n)]
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
// A negative start is where the run STOPS: the character it lands on
|
|
66
|
+
// is the first one not modified, so the run reaches back from it. No
|
|
67
|
+
// upper clamp: start is negative here, so n+start is below n by
|
|
68
|
+
// construction.
|
|
69
|
+
const hi = Math.max(0, n + start)
|
|
70
|
+
return [Math.max(0, hi - span), hi]
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
// `text` with the selected run cased. Outside the run the source is
|
|
75
|
+
// returned verbatim, code point for code point.
|
|
76
|
+
export function caseRange(
|
|
77
|
+
text: string, start: number, len: number, up: boolean
|
|
78
|
+
): string {
|
|
79
|
+
const cps = Array.from(text)
|
|
80
|
+
const [lo, hi] = caseSpan(cps.length, start, len)
|
|
81
|
+
if (hi <= lo) {
|
|
82
|
+
return text
|
|
83
|
+
}
|
|
84
|
+
const mid = cps.slice(lo, hi).join('')
|
|
85
|
+
return cps.slice(0, lo).join('') +
|
|
86
|
+
(up ? mid.toUpperCase() : mid.toLowerCase()) +
|
|
87
|
+
cps.slice(hi).join('')
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
// The integer a range argument carries, or undefined when it is not
|
|
92
|
+
// one. `undefined` for an absent argument is the caller's default, not
|
|
93
|
+
// a refusal — these are optional.
|
|
94
|
+
export function rangeArg(v: any): number | undefined {
|
|
95
|
+
if (null == v) {
|
|
96
|
+
return undefined
|
|
97
|
+
}
|
|
98
|
+
const p = (true === v?.isScalar) ? v.peg : undefined
|
|
99
|
+
if ('number' === typeof p && Number.isInteger(p)) {
|
|
100
|
+
return p
|
|
101
|
+
}
|
|
102
|
+
// A biginteger index is an index: a position in a string is small by
|
|
103
|
+
// construction, or it is not an index at all.
|
|
104
|
+
if ('bigint' === typeof p) {
|
|
105
|
+
const n = Number(p)
|
|
106
|
+
if (Number.isSafeInteger(n)) {
|
|
107
|
+
return n
|
|
108
|
+
}
|
|
109
|
+
}
|
|
110
|
+
// ONE REFUSAL for every way an argument can fail to be an index: not
|
|
111
|
+
// a scalar at all (a preference reaches here, the signature gate
|
|
112
|
+
// having nothing to check), a scalar of another kind, or a biginteger
|
|
113
|
+
// too large to be a position in a string.
|
|
114
|
+
return NaN
|
|
115
|
+
}
|