@vertesia/common 1.5.0-dev.20260714.072725Z → 1.5.0-dev.20260722.120446Z
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/apikey.d.ts +1 -0
- package/lib/apikey.d.ts.map +1 -1
- package/lib/apikey.js.map +1 -1
- package/lib/apps.d.ts +500 -34
- package/lib/apps.d.ts.map +1 -1
- package/lib/apps.js +63 -70
- package/lib/apps.js.map +1 -1
- package/lib/audit-trail.d.ts +61 -1
- package/lib/audit-trail.d.ts.map +1 -1
- package/lib/audit-trail.js +15 -0
- package/lib/audit-trail.js.map +1 -1
- package/lib/data-platform.d.ts +121 -5
- package/lib/data-platform.d.ts.map +1 -1
- package/lib/index.d.ts +7 -0
- package/lib/index.d.ts.map +1 -1
- package/lib/index.js +7 -0
- package/lib/index.js.map +1 -1
- package/lib/interaction.d.ts +19 -1
- package/lib/interaction.d.ts.map +1 -1
- package/lib/interaction.js.map +1 -1
- package/lib/json-schema.d.ts +1 -1
- package/lib/json-schema.d.ts.map +1 -1
- package/lib/platform-event.d.ts +79 -3
- package/lib/platform-event.d.ts.map +1 -1
- package/lib/platform-event.js.map +1 -1
- package/lib/project.d.ts +119 -20
- package/lib/project.d.ts.map +1 -1
- package/lib/project.js +65 -0
- package/lib/project.js.map +1 -1
- package/lib/query.d.ts +6 -0
- package/lib/query.d.ts.map +1 -1
- package/lib/refs.d.ts +1 -0
- package/lib/refs.d.ts.map +1 -1
- package/lib/schema-for-extraction.d.ts +21 -0
- package/lib/schema-for-extraction.d.ts.map +1 -0
- package/lib/schema-for-extraction.js +205 -0
- package/lib/schema-for-extraction.js.map +1 -0
- package/lib/store/agent-run.d.ts +2 -0
- package/lib/store/agent-run.d.ts.map +1 -1
- package/lib/store/conversation-state.d.ts +39 -0
- package/lib/store/conversation-state.d.ts.map +1 -1
- package/lib/store/conversation-state.js +3 -0
- package/lib/store/conversation-state.js.map +1 -1
- package/lib/store/doc-analyzer.d.ts +10 -67
- package/lib/store/doc-analyzer.d.ts.map +1 -1
- package/lib/store/dsl-workflow.d.ts +1 -0
- package/lib/store/dsl-workflow.d.ts.map +1 -1
- package/lib/store/dsl-workflow.js.map +1 -1
- package/lib/store/grounded-extraction.d.ts +146 -0
- package/lib/store/grounded-extraction.d.ts.map +1 -0
- package/lib/store/grounded-extraction.js +8 -0
- package/lib/store/grounded-extraction.js.map +1 -0
- package/lib/store/index.d.ts +1 -0
- package/lib/store/index.d.ts.map +1 -1
- package/lib/store/index.js +1 -0
- package/lib/store/index.js.map +1 -1
- package/lib/store/store.d.ts +307 -1
- package/lib/store/store.d.ts.map +1 -1
- package/lib/store/store.js +438 -0
- package/lib/store/store.js.map +1 -1
- package/lib/store/workflow.d.ts +3 -0
- package/lib/store/workflow.d.ts.map +1 -1
- package/lib/store/workflow.js.map +1 -1
- package/lib/user.d.ts +14 -0
- package/lib/user.d.ts.map +1 -1
- package/lib/user.js +31 -0
- package/lib/user.js.map +1 -1
- package/lib/vertesia-common.js +2 -2
- package/lib/vertesia-common.js.map +1 -1
- package/lib/view-configuration-validation.d.ts +15 -0
- package/lib/view-configuration-validation.d.ts.map +1 -0
- package/lib/view-configuration-validation.js +63 -0
- package/lib/view-configuration-validation.js.map +1 -0
- package/lib/view-query-validation.d.ts +13 -0
- package/lib/view-query-validation.d.ts.map +1 -0
- package/lib/view-query-validation.js +266 -0
- package/lib/view-query-validation.js.map +1 -0
- package/lib/view-validation-helpers.d.ts +15 -0
- package/lib/view-validation-helpers.d.ts.map +1 -0
- package/lib/view-validation-helpers.js +25 -0
- package/lib/view-validation-helpers.js.map +1 -0
- package/lib/views-schema.d.ts +1992 -0
- package/lib/views-schema.d.ts.map +1 -0
- package/lib/views-schema.js +674 -0
- package/lib/views-schema.js.map +1 -0
- package/lib/views-validation.d.ts +21 -0
- package/lib/views-validation.d.ts.map +1 -0
- package/lib/views-validation.js +164 -0
- package/lib/views-validation.js.map +1 -0
- package/lib/views.d.ts +381 -0
- package/lib/views.d.ts.map +1 -0
- package/lib/views.js +41 -0
- package/lib/views.js.map +1 -0
- package/package.json +5 -4
- package/src/apikey.ts +1 -0
- package/src/apps.test.ts +9 -1
- package/src/apps.ts +583 -90
- package/src/audit-trail.ts +83 -0
- package/src/data-platform.ts +129 -5
- package/src/index.ts +12 -0
- package/src/interaction.ts +20 -1
- package/src/json-schema.ts +0 -1
- package/src/platform-event.ts +92 -2
- package/src/project.test.ts +44 -0
- package/src/project.ts +205 -22
- package/src/query.ts +6 -0
- package/src/refs.ts +1 -0
- package/src/roles.test.ts +32 -0
- package/src/schema-for-extraction.test.ts +191 -0
- package/src/schema-for-extraction.ts +231 -0
- package/src/store/agent-run.ts +2 -0
- package/src/store/conversation-state.ts +47 -0
- package/src/store/doc-analyzer.ts +10 -76
- package/src/store/dsl-workflow.ts +1 -0
- package/src/store/grounded-extraction.ts +154 -0
- package/src/store/index.ts +1 -0
- package/src/store/store.ts +778 -1
- package/src/store/workflow.ts +3 -0
- package/src/user.ts +46 -0
- package/src/view-configuration-validation.ts +74 -0
- package/src/view-query-validation.test.ts +21 -0
- package/src/view-query-validation.ts +319 -0
- package/src/view-validation-helpers.ts +28 -0
- package/src/views-schema.test.ts +364 -0
- package/src/views-schema.ts +689 -0
- package/src/views-validation.ts +234 -0
- package/src/views.ts +484 -0
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Extraction schema filtering.
|
|
3
|
+
*
|
|
4
|
+
* Content-type object_schema properties may set `"x-extract": false` to mark fields
|
|
5
|
+
* that must not be filled by extraction models (match scores, ERP ids filled later,
|
|
6
|
+
* internal bookkeeping). The full schema remains the object model for UI/storage;
|
|
7
|
+
* only the filtered copy is used as result_schema for constrained decoding.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
export const X_EXTRACT = 'x-extract';
|
|
11
|
+
|
|
12
|
+
function unescapeJsonPointerSegment(segment: string): string {
|
|
13
|
+
return segment.replace(/~1/g, '/').replace(/~0/g, '~');
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
function resolveLocalRef(ref: string, root: unknown): unknown {
|
|
17
|
+
if (!ref.startsWith('#/')) {
|
|
18
|
+
return undefined;
|
|
19
|
+
}
|
|
20
|
+
let current = root;
|
|
21
|
+
for (const segment of ref.slice(2).split('/').map(unescapeJsonPointerSegment)) {
|
|
22
|
+
if (!current || typeof current !== 'object' || Array.isArray(current)) {
|
|
23
|
+
return undefined;
|
|
24
|
+
}
|
|
25
|
+
current = (current as Record<string, unknown>)[segment];
|
|
26
|
+
}
|
|
27
|
+
return current;
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
function isExtractableSchemaNodeInternal(schema: unknown, root: unknown, seenRefs: Set<string>): boolean {
|
|
31
|
+
if (!schema || typeof schema !== 'object' || Array.isArray(schema)) {
|
|
32
|
+
return true;
|
|
33
|
+
}
|
|
34
|
+
const schemaObj = schema as Record<string, unknown>;
|
|
35
|
+
if (schemaObj[X_EXTRACT] === false) {
|
|
36
|
+
return false;
|
|
37
|
+
}
|
|
38
|
+
if (typeof schemaObj.$ref === 'string' && !seenRefs.has(schemaObj.$ref)) {
|
|
39
|
+
seenRefs.add(schemaObj.$ref);
|
|
40
|
+
const resolved = resolveLocalRef(schemaObj.$ref, root);
|
|
41
|
+
if (resolved !== undefined) {
|
|
42
|
+
return isExtractableSchemaNodeInternal(resolved, root, seenRefs);
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
return true;
|
|
46
|
+
}
|
|
47
|
+
|
|
48
|
+
export function isExtractableSchemaNode(schema: unknown, root: unknown = schema): boolean {
|
|
49
|
+
return isExtractableSchemaNodeInternal(schema, root, new Set());
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
function filterSchemaArray(value: unknown, root: unknown): unknown[] | undefined {
|
|
53
|
+
if (!Array.isArray(value)) {
|
|
54
|
+
return undefined;
|
|
55
|
+
}
|
|
56
|
+
const filtered = value
|
|
57
|
+
.map((item) => schemaForExtractionNode(item, root))
|
|
58
|
+
.filter((item): item is unknown => item !== undefined);
|
|
59
|
+
return filtered.length > 0 ? filtered : undefined;
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
function filterSchemaMap(value: unknown, root: unknown): Record<string, unknown> | undefined {
|
|
63
|
+
if (!value || typeof value !== 'object' || Array.isArray(value)) {
|
|
64
|
+
return undefined;
|
|
65
|
+
}
|
|
66
|
+
const filtered: Record<string, unknown> = {};
|
|
67
|
+
for (const [key, child] of Object.entries(value)) {
|
|
68
|
+
const childFiltered = schemaForExtractionNode(child, root);
|
|
69
|
+
if (childFiltered !== undefined) {
|
|
70
|
+
filtered[key] = childFiltered;
|
|
71
|
+
}
|
|
72
|
+
}
|
|
73
|
+
return Object.keys(filtered).length > 0 ? filtered : undefined;
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
function schemaForExtractionNode(schema: unknown, root: unknown): unknown {
|
|
77
|
+
if (!schema || typeof schema !== 'object') {
|
|
78
|
+
return schema;
|
|
79
|
+
}
|
|
80
|
+
if (Array.isArray(schema)) {
|
|
81
|
+
return schema.map((item) => schemaForExtractionNode(item, root));
|
|
82
|
+
}
|
|
83
|
+
if (!isExtractableSchemaNode(schema, root)) {
|
|
84
|
+
return undefined;
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
const node = { ...(schema as Record<string, unknown>) };
|
|
88
|
+
|
|
89
|
+
for (const defsKey of ['$defs', 'definitions']) {
|
|
90
|
+
const defs = filterSchemaMap(node[defsKey], root);
|
|
91
|
+
if (defs) {
|
|
92
|
+
node[defsKey] = defs;
|
|
93
|
+
} else {
|
|
94
|
+
delete node[defsKey];
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
if (node.items && typeof node.items === 'object') {
|
|
99
|
+
const items = schemaForExtractionNode(node.items, root);
|
|
100
|
+
if (items !== undefined) {
|
|
101
|
+
node.items = items;
|
|
102
|
+
} else {
|
|
103
|
+
delete node.items;
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
for (const unionKey of ['anyOf', 'oneOf', 'allOf']) {
|
|
107
|
+
if (Array.isArray(node[unionKey])) {
|
|
108
|
+
const filtered = filterSchemaArray(node[unionKey], root);
|
|
109
|
+
if (!filtered) {
|
|
110
|
+
return undefined;
|
|
111
|
+
}
|
|
112
|
+
node[unionKey] = filtered;
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
if (node.additionalProperties && typeof node.additionalProperties === 'object') {
|
|
116
|
+
const additionalProperties = schemaForExtractionNode(node.additionalProperties, root);
|
|
117
|
+
node.additionalProperties = additionalProperties ?? false;
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
if (node.properties && typeof node.properties === 'object' && !Array.isArray(node.properties)) {
|
|
121
|
+
const props: Record<string, unknown> = {};
|
|
122
|
+
for (const [key, child] of Object.entries(node.properties as Record<string, unknown>)) {
|
|
123
|
+
const childFiltered = schemaForExtractionNode(child, root);
|
|
124
|
+
if (childFiltered !== undefined) {
|
|
125
|
+
props[key] = childFiltered;
|
|
126
|
+
}
|
|
127
|
+
}
|
|
128
|
+
node.properties = props;
|
|
129
|
+
|
|
130
|
+
if (Array.isArray(node.required)) {
|
|
131
|
+
const kept = node.required.filter((r) => typeof r === 'string' && r in props);
|
|
132
|
+
if (kept.length > 0) {
|
|
133
|
+
node.required = kept;
|
|
134
|
+
} else {
|
|
135
|
+
delete node.required;
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
// Drop the flag from the extraction copy (not needed by providers)
|
|
141
|
+
delete node[X_EXTRACT];
|
|
142
|
+
|
|
143
|
+
return node;
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
/**
|
|
147
|
+
* Deep-clones a JSON schema and removes properties (recursively) marked
|
|
148
|
+
* `x-extract: false`, cleaning `required` arrays to match.
|
|
149
|
+
*/
|
|
150
|
+
export function schemaForExtraction<T>(schema: T): T {
|
|
151
|
+
return (schemaForExtractionNode(schema, schema) ?? {}) as T;
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
function resolveSchemaNode(schema: unknown, root: unknown): unknown {
|
|
155
|
+
if (!schema || typeof schema !== 'object' || Array.isArray(schema)) {
|
|
156
|
+
return schema;
|
|
157
|
+
}
|
|
158
|
+
const schemaObj = schema as Record<string, unknown>;
|
|
159
|
+
if (typeof schemaObj.$ref !== 'string') {
|
|
160
|
+
return schema;
|
|
161
|
+
}
|
|
162
|
+
return resolveLocalRef(schemaObj.$ref, root) ?? schema;
|
|
163
|
+
}
|
|
164
|
+
|
|
165
|
+
function mergePreservingNonExtractableNode(
|
|
166
|
+
existing: unknown,
|
|
167
|
+
extracted: unknown,
|
|
168
|
+
schema: unknown,
|
|
169
|
+
root: unknown,
|
|
170
|
+
): unknown {
|
|
171
|
+
if (!schema || typeof schema !== 'object' || Array.isArray(schema)) {
|
|
172
|
+
return extracted;
|
|
173
|
+
}
|
|
174
|
+
const resolved = resolveSchemaNode(schema, root);
|
|
175
|
+
if (!resolved || typeof resolved !== 'object' || Array.isArray(resolved)) {
|
|
176
|
+
return extracted;
|
|
177
|
+
}
|
|
178
|
+
const schemaObj = resolved as Record<string, unknown>;
|
|
179
|
+
|
|
180
|
+
// Arrays are new extraction output. Without a declared stable item key, index
|
|
181
|
+
// merging can attach preserved ERP/match fields to the wrong row.
|
|
182
|
+
if (Array.isArray(extracted)) {
|
|
183
|
+
const itemSchema = schemaObj.items;
|
|
184
|
+
if (!itemSchema || typeof itemSchema !== 'object') {
|
|
185
|
+
return extracted;
|
|
186
|
+
}
|
|
187
|
+
return extracted.map((item) => mergePreservingNonExtractableNode(undefined, item, itemSchema, root));
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
if (!extracted || typeof extracted !== 'object' || Array.isArray(extracted)) {
|
|
191
|
+
return extracted;
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
const props = schemaObj.properties;
|
|
195
|
+
if (!props || typeof props !== 'object' || Array.isArray(props)) {
|
|
196
|
+
return extracted;
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
const existingObj =
|
|
200
|
+
existing && typeof existing === 'object' && !Array.isArray(existing)
|
|
201
|
+
? (existing as Record<string, unknown>)
|
|
202
|
+
: {};
|
|
203
|
+
const extractedObj = extracted as Record<string, unknown>;
|
|
204
|
+
const propSchemas = props as Record<string, unknown>;
|
|
205
|
+
const result: Record<string, unknown> = { ...extractedObj };
|
|
206
|
+
|
|
207
|
+
for (const [key, propSchema] of Object.entries(propSchemas)) {
|
|
208
|
+
if (!isExtractableSchemaNode(propSchema, root)) {
|
|
209
|
+
// Model should not have filled this; restore prior value if any.
|
|
210
|
+
if (key in existingObj) {
|
|
211
|
+
result[key] = existingObj[key];
|
|
212
|
+
} else {
|
|
213
|
+
delete result[key];
|
|
214
|
+
}
|
|
215
|
+
continue;
|
|
216
|
+
}
|
|
217
|
+
if (key in extractedObj) {
|
|
218
|
+
result[key] = mergePreservingNonExtractableNode(existingObj[key], extractedObj[key], propSchema, root);
|
|
219
|
+
}
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
return result;
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
/**
|
|
226
|
+
* Merge model output over existing properties while preserving values at paths
|
|
227
|
+
* marked `x-extract: false` on the full (unfiltered) schema.
|
|
228
|
+
*/
|
|
229
|
+
export function mergePreservingNonExtractable(existing: unknown, extracted: unknown, schema: unknown): unknown {
|
|
230
|
+
return mergePreservingNonExtractableNode(existing, extracted, schema, schema);
|
|
231
|
+
}
|
package/src/store/agent-run.ts
CHANGED
|
@@ -453,6 +453,8 @@ export interface UpdateAgentRunStatusPayload {
|
|
|
453
453
|
title?: string;
|
|
454
454
|
topic?: string;
|
|
455
455
|
lessons_learned?: string[];
|
|
456
|
+
/** Shallow-merged into the run's existing properties. */
|
|
457
|
+
properties?: Record<string, unknown>;
|
|
456
458
|
/** ES-only: conversation content text (not stored in MongoDB) */
|
|
457
459
|
content?: string;
|
|
458
460
|
/**
|
|
@@ -14,6 +14,34 @@ export interface ToolReference {
|
|
|
14
14
|
stored_at: string;
|
|
15
15
|
}
|
|
16
16
|
|
|
17
|
+
/** Reference to text content externalized to agent artifact storage. */
|
|
18
|
+
export interface TextArtifactReference {
|
|
19
|
+
storage_id: string;
|
|
20
|
+
artifact_path: string;
|
|
21
|
+
display_ref: string;
|
|
22
|
+
sha256: string;
|
|
23
|
+
size_bytes: number;
|
|
24
|
+
content_type: string;
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* Sidecar metadata for generated tool input fields that were stored outside
|
|
29
|
+
* model-visible tool_input. Keyed by tool_use.id on ConversationState.
|
|
30
|
+
*/
|
|
31
|
+
export interface ExternalizedToolInputRef {
|
|
32
|
+
tool_name: string;
|
|
33
|
+
input_path: ['content'];
|
|
34
|
+
ref: TextArtifactReference;
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
export interface ExternalizedToolInputRefs {
|
|
38
|
+
[toolUseId: string]: ExternalizedToolInputRef[];
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
export function toolInputRefsArtifactPath(storageId: string): string {
|
|
42
|
+
return `agents/${storageId}/tool-input-refs.json`;
|
|
43
|
+
}
|
|
44
|
+
|
|
17
45
|
/**
|
|
18
46
|
* Conversation state passed between workflow activities.
|
|
19
47
|
* Contains all context needed to continue a multi-turn agent conversation.
|
|
@@ -51,6 +79,15 @@ export interface ConversationState {
|
|
|
51
79
|
/** Compact, redacted latest user intent for reviewer-style system interactions. */
|
|
52
80
|
latest_user_message?: string;
|
|
53
81
|
|
|
82
|
+
/**
|
|
83
|
+
* Transport sidecar for large generated tool input fields.
|
|
84
|
+
*
|
|
85
|
+
* These refs are intentionally kept out of tool_use.tool_input so they are
|
|
86
|
+
* not shown to the model. Tool execution hydrates them from artifact storage
|
|
87
|
+
* immediately before activity validation.
|
|
88
|
+
*/
|
|
89
|
+
tool_input_refs?: ExternalizedToolInputRefs;
|
|
90
|
+
|
|
54
91
|
/**
|
|
55
92
|
* The output of the this conversation step
|
|
56
93
|
*/
|
|
@@ -204,6 +241,16 @@ export interface ConversationState {
|
|
|
204
241
|
* to consolidate all artifacts under the parent agent run.
|
|
205
242
|
*/
|
|
206
243
|
launch_id?: string;
|
|
244
|
+
|
|
245
|
+
/**
|
|
246
|
+
* The exact app version this run is pinned to, derived from the `@version` on the
|
|
247
|
+
* started interaction ref / the `x-vertesia-app-version` header at start. Persisted on the state
|
|
248
|
+
* so it survives resume, and applied to the activity client (`withAppVersion`) so every app-owned
|
|
249
|
+
* ref the run resolves — interactions, types, processes, tools — targets this version instead of
|
|
250
|
+
* the current/promoted one. Undefined → current/promoted. Resolution-time only; never a stored
|
|
251
|
+
* capability-ref version.
|
|
252
|
+
*/
|
|
253
|
+
app_version?: string;
|
|
207
254
|
}
|
|
208
255
|
|
|
209
256
|
/**
|
|
@@ -1,104 +1,42 @@
|
|
|
1
|
-
import type { JSONObject } from '../json.js';
|
|
2
1
|
import type { WorkflowExecutionPayload, WorkflowRunStatus } from './workflow.js';
|
|
3
2
|
|
|
4
|
-
export interface
|
|
3
|
+
export interface DocumentPrepOptions {
|
|
5
4
|
features?: string[];
|
|
6
5
|
debug?: boolean;
|
|
6
|
+
output_format?: DocProcessorOutputFormat;
|
|
7
7
|
[key: string]: unknown;
|
|
8
8
|
}
|
|
9
9
|
|
|
10
|
-
export interface
|
|
11
|
-
vars:
|
|
12
|
-
}
|
|
13
|
-
|
|
14
|
-
export interface TransformTablesWorkflowPayload extends Omit<WorkflowExecutionPayload, 'vars'> {
|
|
15
|
-
vars: AdaptTablesParams;
|
|
16
|
-
environment?: string;
|
|
10
|
+
export interface DocumentPrepWorkflowPayload extends Omit<WorkflowExecutionPayload, 'vars'> {
|
|
11
|
+
vars: DocumentPrepOptions;
|
|
17
12
|
}
|
|
18
13
|
|
|
19
|
-
export
|
|
20
|
-
result_path: string;
|
|
21
|
-
status: string;
|
|
22
|
-
table_count: number;
|
|
23
|
-
item_count: number;
|
|
24
|
-
}
|
|
25
|
-
|
|
26
|
-
/**
|
|
27
|
-
* Represents a image in a document that has been analyzed
|
|
28
|
-
*/
|
|
29
|
-
export interface DocImage {
|
|
30
|
-
id?: string;
|
|
31
|
-
page_number?: number;
|
|
32
|
-
description?: string;
|
|
33
|
-
is_meaningful?: boolean;
|
|
34
|
-
width?: number;
|
|
35
|
-
height?: number;
|
|
36
|
-
}
|
|
37
|
-
|
|
38
|
-
/**
|
|
39
|
-
* The export type formats for tables.
|
|
40
|
-
*/
|
|
41
|
-
export type ExportTableFormats = 'json' | 'csv' | 'xml';
|
|
42
|
-
|
|
43
|
-
/**
|
|
44
|
-
* Represents a table in a document that has been analyzed
|
|
45
|
-
*/
|
|
46
|
-
export interface DocTable {
|
|
47
|
-
page_number?: number;
|
|
48
|
-
table_number?: number;
|
|
49
|
-
title?: string;
|
|
50
|
-
format: 'application/csv' | 'application/json';
|
|
51
|
-
}
|
|
52
|
-
|
|
53
|
-
/**
|
|
54
|
-
* Represents a table in a document that has been analyzed in CSV format
|
|
55
|
-
*/
|
|
56
|
-
export interface DocTableCsv extends DocTable {
|
|
57
|
-
format: 'application/csv';
|
|
58
|
-
title?: string;
|
|
59
|
-
data: string;
|
|
60
|
-
}
|
|
61
|
-
|
|
62
|
-
/**
|
|
63
|
-
* Represents a table in a document that has been analyzed in JSON format
|
|
64
|
-
*/
|
|
65
|
-
export interface DocTableJson extends DocTable {
|
|
66
|
-
format: 'application/json';
|
|
67
|
-
title?: string;
|
|
68
|
-
data: JSONObject[];
|
|
69
|
-
}
|
|
70
|
-
|
|
71
|
-
export type DocTableResponse = DocTableCsv | DocTableJson;
|
|
14
|
+
export type DocumentProcessingPhase = 'markdown' | 'grounded_extraction';
|
|
72
15
|
|
|
73
16
|
/**
|
|
74
17
|
* Output format for document processing workflows
|
|
75
18
|
*/
|
|
76
|
-
export type DocProcessorOutputFormat = '
|
|
19
|
+
export type DocProcessorOutputFormat = 'markdown';
|
|
77
20
|
|
|
78
21
|
/**
|
|
79
22
|
* Represents a document analysis run status
|
|
80
23
|
*/
|
|
81
24
|
export interface DocAnalyzeRunStatusResponse extends WorkflowRunStatus {
|
|
25
|
+
phase?: DocumentProcessingPhase;
|
|
82
26
|
progress?: DocAnalyzerProgress;
|
|
83
|
-
/** The output format being used for processing
|
|
27
|
+
/** The output format being used for processing. */
|
|
84
28
|
output_format?: DocProcessorOutputFormat;
|
|
85
29
|
}
|
|
86
30
|
|
|
87
|
-
export interface DocAnalyzerResultResponse {
|
|
88
|
-
document?: string;
|
|
89
|
-
tables?: DocTableResponse[];
|
|
90
|
-
images?: DocImage[];
|
|
91
|
-
annotated?: string | null;
|
|
92
|
-
}
|
|
93
|
-
|
|
94
31
|
export interface DocAnalyzerProgress {
|
|
32
|
+
phase?: DocumentProcessingPhase;
|
|
95
33
|
pages: DocAnalyzerProgressStatus;
|
|
96
34
|
images: DocAnalyzerProgressStatus;
|
|
97
35
|
tables: DocAnalyzerProgressStatus;
|
|
98
36
|
visuals: DocAnalyzerProgressStatus;
|
|
99
37
|
started_at?: number;
|
|
100
38
|
percent: number;
|
|
101
|
-
/** The output format being used for processing
|
|
39
|
+
/** The output format being used for processing. */
|
|
102
40
|
output_format?: DocProcessorOutputFormat;
|
|
103
41
|
}
|
|
104
42
|
|
|
@@ -176,7 +114,3 @@ export interface AdaptedTable {
|
|
|
176
114
|
}
|
|
177
115
|
|
|
178
116
|
export type AdaptedTableResponse = Record<string, AdaptedTable>;
|
|
179
|
-
|
|
180
|
-
export interface AnnotatedPdfResponse {
|
|
181
|
-
url: string | null;
|
|
182
|
-
}
|
|
@@ -48,6 +48,7 @@ export interface DSLWorkflowExecutionPayload extends WorkflowExecutionPayload<Re
|
|
|
48
48
|
*/
|
|
49
49
|
export interface DSLActivityOptions {
|
|
50
50
|
startToCloseTimeout?: DurationValue;
|
|
51
|
+
heartbeatTimeout?: DurationValue;
|
|
51
52
|
scheduleToStartTimeout?: DurationValue;
|
|
52
53
|
scheduleToCloseTimeout?: DurationValue;
|
|
53
54
|
retry?: DSLRetryPolicy;
|
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
import type { InteractionExecutionConfiguration } from '../interaction.js';
|
|
2
|
+
import type { WorkflowRunStatus } from './workflow.js';
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Document-level trust verdict for a grounded extraction. `good_to_go` means the
|
|
6
|
+
* extracted content (after any review corrections) can be used without a human
|
|
7
|
+
* check; `needs_review` means a human should verify it. This reflects content
|
|
8
|
+
* correctness, not how many citation boxes rendered.
|
|
9
|
+
*/
|
|
10
|
+
export type GroundedExtractionVerdict = 'good_to_go' | 'needs_review';
|
|
11
|
+
|
|
12
|
+
/**
|
|
13
|
+
* Canonical workflow id for object-scoped grounded extraction. Every entry point
|
|
14
|
+
* must use this id so Temporal prevents overlapping extraction runs per object.
|
|
15
|
+
*/
|
|
16
|
+
export function getGroundedExtractionWorkflowId(accountId: string, objectId: string): string {
|
|
17
|
+
return `${accountId.slice(0, 6)}:workflow_execution_request:${objectId}:grounded`;
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
/**
|
|
21
|
+
* Request body to start a grounded extraction on a content object. All fields are
|
|
22
|
+
* optional: with none set, the object's own content-type schema drives the
|
|
23
|
+
* extraction with default models and settings.
|
|
24
|
+
*/
|
|
25
|
+
export interface GroundedExtractionRequest {
|
|
26
|
+
/** JSON schema describing the data to extract. Takes precedence over type_ref. */
|
|
27
|
+
schema?: Record<string, unknown>;
|
|
28
|
+
/** Content type id or catalog ref whose object_schema drives the extraction. */
|
|
29
|
+
type_ref?: string;
|
|
30
|
+
/** Interaction to use. Defaults to sys:ExtractInformationGrounded. */
|
|
31
|
+
interaction_name?: string;
|
|
32
|
+
/** Maximum number of pages to process. */
|
|
33
|
+
max_pages?: number;
|
|
34
|
+
/** Run OCR on every page even when a text layer exists. */
|
|
35
|
+
force_ocr?: boolean;
|
|
36
|
+
/** Re-run OCR on pages that need it instead of restoring the stored OCR result. */
|
|
37
|
+
refresh_ocr?: boolean;
|
|
38
|
+
/** Attach clean page images for layout/semantic context; direct-vision pages also receive checkerboards. */
|
|
39
|
+
use_vision?: boolean;
|
|
40
|
+
/**
|
|
41
|
+
* A1 locate-grid cell size in PDF points for vision pages (drives both the drawn
|
|
42
|
+
* grid and cell→box resolution). Smaller = finer grid / more cells. Default: 14.
|
|
43
|
+
*/
|
|
44
|
+
grid_cell_pt?: number;
|
|
45
|
+
/**
|
|
46
|
+
* How to read pages that have no digital text layer (scans / image-only pages).
|
|
47
|
+
* 'vision' (default): read them off the page image with the extraction model,
|
|
48
|
+
* skipping OCR entirely. 'ocr': legacy path — OCR those pages and block-ground on
|
|
49
|
+
* the (lossy) OCR text. Set to 'ocr' to revert to the pre-vision behavior.
|
|
50
|
+
*/
|
|
51
|
+
raster_mode?: 'vision' | 'ocr';
|
|
52
|
+
/** Maximum pages per extraction call; larger documents are split into sequential windows. */
|
|
53
|
+
window_pages?: number;
|
|
54
|
+
/**
|
|
55
|
+
* Extract with an autonomous agent (views the whole document at once) instead of
|
|
56
|
+
* the deterministic windowed pipeline. Sidesteps window-boundary splits on long
|
|
57
|
+
* documents. The workflow stages the artifacts into an agent space, runs a
|
|
58
|
+
* conversation agent that writes the extraction, then folds it back.
|
|
59
|
+
*/
|
|
60
|
+
agentic_extraction?: boolean;
|
|
61
|
+
/** Agent interaction for agentic_extraction. Defaults to sys:GeneralAgent. */
|
|
62
|
+
extract_agent?: string;
|
|
63
|
+
/** Update the object's properties with the extracted data. Default: true. */
|
|
64
|
+
update_properties?: boolean;
|
|
65
|
+
/** LLM execution configuration (model, environment, ...) for the main pass. */
|
|
66
|
+
config?: InteractionExecutionConfiguration;
|
|
67
|
+
/** Execution configuration used instead of `config` on hard content (scans, handwriting). */
|
|
68
|
+
hard_config?: InteractionExecutionConfiguration;
|
|
69
|
+
/** Hardness score (0..1) at or above which `hard_config` is used. Default: 0.5. */
|
|
70
|
+
hardness_threshold?: number;
|
|
71
|
+
/** Execution configuration for the post-extraction review pass. No review runs when absent. */
|
|
72
|
+
review_config?: InteractionExecutionConfiguration;
|
|
73
|
+
/** Hardness score (0..1) at or above which the review runs. Defaults to hardness_threshold. */
|
|
74
|
+
review_threshold?: number;
|
|
75
|
+
/** Review triggers when any page's citation coverage falls below this floor. Default: 0.2. */
|
|
76
|
+
coverage_review_threshold?: number;
|
|
77
|
+
/** Run the model review even when every citation was digitally verified. Requires review_config. */
|
|
78
|
+
force_review?: boolean;
|
|
79
|
+
/**
|
|
80
|
+
* Free-text operator guidance folded into the extraction prompt to steer a
|
|
81
|
+
* (re-)extraction, e.g. "part numbers are in the third column; some line items
|
|
82
|
+
* wrap onto the next row".
|
|
83
|
+
*/
|
|
84
|
+
operator_instructions?: string;
|
|
85
|
+
/** Interactive assistant only: the operator's opening message for the assistant conversation. */
|
|
86
|
+
user_prompt?: string;
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
/**
|
|
90
|
+
* Response from starting the interactive grounded extraction assistant. The agent
|
|
91
|
+
* run + conversation workflow are launched server-side (recordRun -> stage the
|
|
92
|
+
* document into the agent space -> launch the interactive conversation); the
|
|
93
|
+
* client renders the conversation with `agent_run_id`.
|
|
94
|
+
*/
|
|
95
|
+
export interface GroundedAssistantResponse {
|
|
96
|
+
/** The AgentRun id to stream/render the conversation. */
|
|
97
|
+
agent_run_id: string;
|
|
98
|
+
/** The conversation workflow id backing the run. */
|
|
99
|
+
workflow_id: string;
|
|
100
|
+
/** The object the assistant is scoped to. */
|
|
101
|
+
object_id: string;
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
/**
|
|
105
|
+
* Status of a grounded extraction workflow. Carries the doc-level verdict once the
|
|
106
|
+
* run has completed and written its result.
|
|
107
|
+
*/
|
|
108
|
+
export interface GroundedExtractionRunStatusResponse extends WorkflowRunStatus {
|
|
109
|
+
/** The trust verdict, present once the run has completed. */
|
|
110
|
+
verdict?: GroundedExtractionVerdict;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/**
|
|
114
|
+
* How each extracted value was verified. Two kinds, both trustworthy:
|
|
115
|
+
* digitally verified (matched the document's text — digital layer or OCR) and
|
|
116
|
+
* AI verified (the reviewer confirmed it against the page image — the primary
|
|
117
|
+
* signal for scanned or handwritten content, which has no text layer to match).
|
|
118
|
+
*/
|
|
119
|
+
export interface GroundedVerificationBreakdown {
|
|
120
|
+
/** Total number of cited values. */
|
|
121
|
+
total: number;
|
|
122
|
+
/** Values matched verbatim against the document's text (digital layer or OCR). */
|
|
123
|
+
digitally_verified: number;
|
|
124
|
+
/** Values the reviewer model confirmed against the page image. */
|
|
125
|
+
ai_verified: number;
|
|
126
|
+
/** Values read from the image but neither text-matched nor reviewer-confirmed. */
|
|
127
|
+
unverified: number;
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
/**
|
|
131
|
+
* Completed grounded extraction result: the extracted data with its trust verdict
|
|
132
|
+
* and verification breakdown, plus a download URL for the full citations artifact.
|
|
133
|
+
*/
|
|
134
|
+
export interface GroundedExtractionResultResponse {
|
|
135
|
+
object_id: string;
|
|
136
|
+
/** The extracted data, shaped by the requested schema. */
|
|
137
|
+
data: Record<string, unknown>;
|
|
138
|
+
/** Document-level trust verdict. */
|
|
139
|
+
verdict?: GroundedExtractionVerdict;
|
|
140
|
+
/** One-sentence rationale for the verdict. */
|
|
141
|
+
verdict_reason?: string;
|
|
142
|
+
/** Mean citation confidence in [0,1]. */
|
|
143
|
+
confidence?: number;
|
|
144
|
+
/** Per-value verification breakdown. */
|
|
145
|
+
verification: GroundedVerificationBreakdown;
|
|
146
|
+
/** Review outcome, when a review pass ran. */
|
|
147
|
+
review?: {
|
|
148
|
+
assessment: 'complete' | 'issues_found';
|
|
149
|
+
summary?: string;
|
|
150
|
+
corrections_applied?: number;
|
|
151
|
+
};
|
|
152
|
+
/** Signed download URL for the full grounded-extraction.json (data + citations + boxes). */
|
|
153
|
+
result_url?: string | null;
|
|
154
|
+
}
|
package/src/store/index.ts
CHANGED
|
@@ -6,6 +6,7 @@ export * from './common.js';
|
|
|
6
6
|
export * from './conversation-state.js';
|
|
7
7
|
export * from './doc-analyzer.js';
|
|
8
8
|
export * from './dsl-workflow.js';
|
|
9
|
+
export * from './grounded-extraction.js';
|
|
9
10
|
export * from './hive-memory.js';
|
|
10
11
|
export * from './object-types.js';
|
|
11
12
|
export * from './process.js';
|