@vertesia/common 1.5.0-dev.20260714.072725Z → 1.5.0-dev.20260722.120446Z

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (127) hide show
  1. package/lib/apikey.d.ts +1 -0
  2. package/lib/apikey.d.ts.map +1 -1
  3. package/lib/apikey.js.map +1 -1
  4. package/lib/apps.d.ts +500 -34
  5. package/lib/apps.d.ts.map +1 -1
  6. package/lib/apps.js +63 -70
  7. package/lib/apps.js.map +1 -1
  8. package/lib/audit-trail.d.ts +61 -1
  9. package/lib/audit-trail.d.ts.map +1 -1
  10. package/lib/audit-trail.js +15 -0
  11. package/lib/audit-trail.js.map +1 -1
  12. package/lib/data-platform.d.ts +121 -5
  13. package/lib/data-platform.d.ts.map +1 -1
  14. package/lib/index.d.ts +7 -0
  15. package/lib/index.d.ts.map +1 -1
  16. package/lib/index.js +7 -0
  17. package/lib/index.js.map +1 -1
  18. package/lib/interaction.d.ts +19 -1
  19. package/lib/interaction.d.ts.map +1 -1
  20. package/lib/interaction.js.map +1 -1
  21. package/lib/json-schema.d.ts +1 -1
  22. package/lib/json-schema.d.ts.map +1 -1
  23. package/lib/platform-event.d.ts +79 -3
  24. package/lib/platform-event.d.ts.map +1 -1
  25. package/lib/platform-event.js.map +1 -1
  26. package/lib/project.d.ts +119 -20
  27. package/lib/project.d.ts.map +1 -1
  28. package/lib/project.js +65 -0
  29. package/lib/project.js.map +1 -1
  30. package/lib/query.d.ts +6 -0
  31. package/lib/query.d.ts.map +1 -1
  32. package/lib/refs.d.ts +1 -0
  33. package/lib/refs.d.ts.map +1 -1
  34. package/lib/schema-for-extraction.d.ts +21 -0
  35. package/lib/schema-for-extraction.d.ts.map +1 -0
  36. package/lib/schema-for-extraction.js +205 -0
  37. package/lib/schema-for-extraction.js.map +1 -0
  38. package/lib/store/agent-run.d.ts +2 -0
  39. package/lib/store/agent-run.d.ts.map +1 -1
  40. package/lib/store/conversation-state.d.ts +39 -0
  41. package/lib/store/conversation-state.d.ts.map +1 -1
  42. package/lib/store/conversation-state.js +3 -0
  43. package/lib/store/conversation-state.js.map +1 -1
  44. package/lib/store/doc-analyzer.d.ts +10 -67
  45. package/lib/store/doc-analyzer.d.ts.map +1 -1
  46. package/lib/store/dsl-workflow.d.ts +1 -0
  47. package/lib/store/dsl-workflow.d.ts.map +1 -1
  48. package/lib/store/dsl-workflow.js.map +1 -1
  49. package/lib/store/grounded-extraction.d.ts +146 -0
  50. package/lib/store/grounded-extraction.d.ts.map +1 -0
  51. package/lib/store/grounded-extraction.js +8 -0
  52. package/lib/store/grounded-extraction.js.map +1 -0
  53. package/lib/store/index.d.ts +1 -0
  54. package/lib/store/index.d.ts.map +1 -1
  55. package/lib/store/index.js +1 -0
  56. package/lib/store/index.js.map +1 -1
  57. package/lib/store/store.d.ts +307 -1
  58. package/lib/store/store.d.ts.map +1 -1
  59. package/lib/store/store.js +438 -0
  60. package/lib/store/store.js.map +1 -1
  61. package/lib/store/workflow.d.ts +3 -0
  62. package/lib/store/workflow.d.ts.map +1 -1
  63. package/lib/store/workflow.js.map +1 -1
  64. package/lib/user.d.ts +14 -0
  65. package/lib/user.d.ts.map +1 -1
  66. package/lib/user.js +31 -0
  67. package/lib/user.js.map +1 -1
  68. package/lib/vertesia-common.js +2 -2
  69. package/lib/vertesia-common.js.map +1 -1
  70. package/lib/view-configuration-validation.d.ts +15 -0
  71. package/lib/view-configuration-validation.d.ts.map +1 -0
  72. package/lib/view-configuration-validation.js +63 -0
  73. package/lib/view-configuration-validation.js.map +1 -0
  74. package/lib/view-query-validation.d.ts +13 -0
  75. package/lib/view-query-validation.d.ts.map +1 -0
  76. package/lib/view-query-validation.js +266 -0
  77. package/lib/view-query-validation.js.map +1 -0
  78. package/lib/view-validation-helpers.d.ts +15 -0
  79. package/lib/view-validation-helpers.d.ts.map +1 -0
  80. package/lib/view-validation-helpers.js +25 -0
  81. package/lib/view-validation-helpers.js.map +1 -0
  82. package/lib/views-schema.d.ts +1992 -0
  83. package/lib/views-schema.d.ts.map +1 -0
  84. package/lib/views-schema.js +674 -0
  85. package/lib/views-schema.js.map +1 -0
  86. package/lib/views-validation.d.ts +21 -0
  87. package/lib/views-validation.d.ts.map +1 -0
  88. package/lib/views-validation.js +164 -0
  89. package/lib/views-validation.js.map +1 -0
  90. package/lib/views.d.ts +381 -0
  91. package/lib/views.d.ts.map +1 -0
  92. package/lib/views.js +41 -0
  93. package/lib/views.js.map +1 -0
  94. package/package.json +5 -4
  95. package/src/apikey.ts +1 -0
  96. package/src/apps.test.ts +9 -1
  97. package/src/apps.ts +583 -90
  98. package/src/audit-trail.ts +83 -0
  99. package/src/data-platform.ts +129 -5
  100. package/src/index.ts +12 -0
  101. package/src/interaction.ts +20 -1
  102. package/src/json-schema.ts +0 -1
  103. package/src/platform-event.ts +92 -2
  104. package/src/project.test.ts +44 -0
  105. package/src/project.ts +205 -22
  106. package/src/query.ts +6 -0
  107. package/src/refs.ts +1 -0
  108. package/src/roles.test.ts +32 -0
  109. package/src/schema-for-extraction.test.ts +191 -0
  110. package/src/schema-for-extraction.ts +231 -0
  111. package/src/store/agent-run.ts +2 -0
  112. package/src/store/conversation-state.ts +47 -0
  113. package/src/store/doc-analyzer.ts +10 -76
  114. package/src/store/dsl-workflow.ts +1 -0
  115. package/src/store/grounded-extraction.ts +154 -0
  116. package/src/store/index.ts +1 -0
  117. package/src/store/store.ts +778 -1
  118. package/src/store/workflow.ts +3 -0
  119. package/src/user.ts +46 -0
  120. package/src/view-configuration-validation.ts +74 -0
  121. package/src/view-query-validation.test.ts +21 -0
  122. package/src/view-query-validation.ts +319 -0
  123. package/src/view-validation-helpers.ts +28 -0
  124. package/src/views-schema.test.ts +364 -0
  125. package/src/views-schema.ts +689 -0
  126. package/src/views-validation.ts +234 -0
  127. package/src/views.ts +484 -0
@@ -0,0 +1,231 @@
1
+ /**
2
+ * Extraction schema filtering.
3
+ *
4
+ * Content-type object_schema properties may set `"x-extract": false` to mark fields
5
+ * that must not be filled by extraction models (match scores, ERP ids filled later,
6
+ * internal bookkeeping). The full schema remains the object model for UI/storage;
7
+ * only the filtered copy is used as result_schema for constrained decoding.
8
+ */
9
+
10
+ export const X_EXTRACT = 'x-extract';
11
+
12
+ function unescapeJsonPointerSegment(segment: string): string {
13
+ return segment.replace(/~1/g, '/').replace(/~0/g, '~');
14
+ }
15
+
16
+ function resolveLocalRef(ref: string, root: unknown): unknown {
17
+ if (!ref.startsWith('#/')) {
18
+ return undefined;
19
+ }
20
+ let current = root;
21
+ for (const segment of ref.slice(2).split('/').map(unescapeJsonPointerSegment)) {
22
+ if (!current || typeof current !== 'object' || Array.isArray(current)) {
23
+ return undefined;
24
+ }
25
+ current = (current as Record<string, unknown>)[segment];
26
+ }
27
+ return current;
28
+ }
29
+
30
+ function isExtractableSchemaNodeInternal(schema: unknown, root: unknown, seenRefs: Set<string>): boolean {
31
+ if (!schema || typeof schema !== 'object' || Array.isArray(schema)) {
32
+ return true;
33
+ }
34
+ const schemaObj = schema as Record<string, unknown>;
35
+ if (schemaObj[X_EXTRACT] === false) {
36
+ return false;
37
+ }
38
+ if (typeof schemaObj.$ref === 'string' && !seenRefs.has(schemaObj.$ref)) {
39
+ seenRefs.add(schemaObj.$ref);
40
+ const resolved = resolveLocalRef(schemaObj.$ref, root);
41
+ if (resolved !== undefined) {
42
+ return isExtractableSchemaNodeInternal(resolved, root, seenRefs);
43
+ }
44
+ }
45
+ return true;
46
+ }
47
+
48
+ export function isExtractableSchemaNode(schema: unknown, root: unknown = schema): boolean {
49
+ return isExtractableSchemaNodeInternal(schema, root, new Set());
50
+ }
51
+
52
+ function filterSchemaArray(value: unknown, root: unknown): unknown[] | undefined {
53
+ if (!Array.isArray(value)) {
54
+ return undefined;
55
+ }
56
+ const filtered = value
57
+ .map((item) => schemaForExtractionNode(item, root))
58
+ .filter((item): item is unknown => item !== undefined);
59
+ return filtered.length > 0 ? filtered : undefined;
60
+ }
61
+
62
+ function filterSchemaMap(value: unknown, root: unknown): Record<string, unknown> | undefined {
63
+ if (!value || typeof value !== 'object' || Array.isArray(value)) {
64
+ return undefined;
65
+ }
66
+ const filtered: Record<string, unknown> = {};
67
+ for (const [key, child] of Object.entries(value)) {
68
+ const childFiltered = schemaForExtractionNode(child, root);
69
+ if (childFiltered !== undefined) {
70
+ filtered[key] = childFiltered;
71
+ }
72
+ }
73
+ return Object.keys(filtered).length > 0 ? filtered : undefined;
74
+ }
75
+
76
+ function schemaForExtractionNode(schema: unknown, root: unknown): unknown {
77
+ if (!schema || typeof schema !== 'object') {
78
+ return schema;
79
+ }
80
+ if (Array.isArray(schema)) {
81
+ return schema.map((item) => schemaForExtractionNode(item, root));
82
+ }
83
+ if (!isExtractableSchemaNode(schema, root)) {
84
+ return undefined;
85
+ }
86
+
87
+ const node = { ...(schema as Record<string, unknown>) };
88
+
89
+ for (const defsKey of ['$defs', 'definitions']) {
90
+ const defs = filterSchemaMap(node[defsKey], root);
91
+ if (defs) {
92
+ node[defsKey] = defs;
93
+ } else {
94
+ delete node[defsKey];
95
+ }
96
+ }
97
+
98
+ if (node.items && typeof node.items === 'object') {
99
+ const items = schemaForExtractionNode(node.items, root);
100
+ if (items !== undefined) {
101
+ node.items = items;
102
+ } else {
103
+ delete node.items;
104
+ }
105
+ }
106
+ for (const unionKey of ['anyOf', 'oneOf', 'allOf']) {
107
+ if (Array.isArray(node[unionKey])) {
108
+ const filtered = filterSchemaArray(node[unionKey], root);
109
+ if (!filtered) {
110
+ return undefined;
111
+ }
112
+ node[unionKey] = filtered;
113
+ }
114
+ }
115
+ if (node.additionalProperties && typeof node.additionalProperties === 'object') {
116
+ const additionalProperties = schemaForExtractionNode(node.additionalProperties, root);
117
+ node.additionalProperties = additionalProperties ?? false;
118
+ }
119
+
120
+ if (node.properties && typeof node.properties === 'object' && !Array.isArray(node.properties)) {
121
+ const props: Record<string, unknown> = {};
122
+ for (const [key, child] of Object.entries(node.properties as Record<string, unknown>)) {
123
+ const childFiltered = schemaForExtractionNode(child, root);
124
+ if (childFiltered !== undefined) {
125
+ props[key] = childFiltered;
126
+ }
127
+ }
128
+ node.properties = props;
129
+
130
+ if (Array.isArray(node.required)) {
131
+ const kept = node.required.filter((r) => typeof r === 'string' && r in props);
132
+ if (kept.length > 0) {
133
+ node.required = kept;
134
+ } else {
135
+ delete node.required;
136
+ }
137
+ }
138
+ }
139
+
140
+ // Drop the flag from the extraction copy (not needed by providers)
141
+ delete node[X_EXTRACT];
142
+
143
+ return node;
144
+ }
145
+
146
+ /**
147
+ * Deep-clones a JSON schema and removes properties (recursively) marked
148
+ * `x-extract: false`, cleaning `required` arrays to match.
149
+ */
150
+ export function schemaForExtraction<T>(schema: T): T {
151
+ return (schemaForExtractionNode(schema, schema) ?? {}) as T;
152
+ }
153
+
154
+ function resolveSchemaNode(schema: unknown, root: unknown): unknown {
155
+ if (!schema || typeof schema !== 'object' || Array.isArray(schema)) {
156
+ return schema;
157
+ }
158
+ const schemaObj = schema as Record<string, unknown>;
159
+ if (typeof schemaObj.$ref !== 'string') {
160
+ return schema;
161
+ }
162
+ return resolveLocalRef(schemaObj.$ref, root) ?? schema;
163
+ }
164
+
165
+ function mergePreservingNonExtractableNode(
166
+ existing: unknown,
167
+ extracted: unknown,
168
+ schema: unknown,
169
+ root: unknown,
170
+ ): unknown {
171
+ if (!schema || typeof schema !== 'object' || Array.isArray(schema)) {
172
+ return extracted;
173
+ }
174
+ const resolved = resolveSchemaNode(schema, root);
175
+ if (!resolved || typeof resolved !== 'object' || Array.isArray(resolved)) {
176
+ return extracted;
177
+ }
178
+ const schemaObj = resolved as Record<string, unknown>;
179
+
180
+ // Arrays are new extraction output. Without a declared stable item key, index
181
+ // merging can attach preserved ERP/match fields to the wrong row.
182
+ if (Array.isArray(extracted)) {
183
+ const itemSchema = schemaObj.items;
184
+ if (!itemSchema || typeof itemSchema !== 'object') {
185
+ return extracted;
186
+ }
187
+ return extracted.map((item) => mergePreservingNonExtractableNode(undefined, item, itemSchema, root));
188
+ }
189
+
190
+ if (!extracted || typeof extracted !== 'object' || Array.isArray(extracted)) {
191
+ return extracted;
192
+ }
193
+
194
+ const props = schemaObj.properties;
195
+ if (!props || typeof props !== 'object' || Array.isArray(props)) {
196
+ return extracted;
197
+ }
198
+
199
+ const existingObj =
200
+ existing && typeof existing === 'object' && !Array.isArray(existing)
201
+ ? (existing as Record<string, unknown>)
202
+ : {};
203
+ const extractedObj = extracted as Record<string, unknown>;
204
+ const propSchemas = props as Record<string, unknown>;
205
+ const result: Record<string, unknown> = { ...extractedObj };
206
+
207
+ for (const [key, propSchema] of Object.entries(propSchemas)) {
208
+ if (!isExtractableSchemaNode(propSchema, root)) {
209
+ // Model should not have filled this; restore prior value if any.
210
+ if (key in existingObj) {
211
+ result[key] = existingObj[key];
212
+ } else {
213
+ delete result[key];
214
+ }
215
+ continue;
216
+ }
217
+ if (key in extractedObj) {
218
+ result[key] = mergePreservingNonExtractableNode(existingObj[key], extractedObj[key], propSchema, root);
219
+ }
220
+ }
221
+
222
+ return result;
223
+ }
224
+
225
+ /**
226
+ * Merge model output over existing properties while preserving values at paths
227
+ * marked `x-extract: false` on the full (unfiltered) schema.
228
+ */
229
+ export function mergePreservingNonExtractable(existing: unknown, extracted: unknown, schema: unknown): unknown {
230
+ return mergePreservingNonExtractableNode(existing, extracted, schema, schema);
231
+ }
@@ -453,6 +453,8 @@ export interface UpdateAgentRunStatusPayload {
453
453
  title?: string;
454
454
  topic?: string;
455
455
  lessons_learned?: string[];
456
+ /** Shallow-merged into the run's existing properties. */
457
+ properties?: Record<string, unknown>;
456
458
  /** ES-only: conversation content text (not stored in MongoDB) */
457
459
  content?: string;
458
460
  /**
@@ -14,6 +14,34 @@ export interface ToolReference {
14
14
  stored_at: string;
15
15
  }
16
16
 
17
+ /** Reference to text content externalized to agent artifact storage. */
18
+ export interface TextArtifactReference {
19
+ storage_id: string;
20
+ artifact_path: string;
21
+ display_ref: string;
22
+ sha256: string;
23
+ size_bytes: number;
24
+ content_type: string;
25
+ }
26
+
27
+ /**
28
+ * Sidecar metadata for generated tool input fields that were stored outside
29
+ * model-visible tool_input. Keyed by tool_use.id on ConversationState.
30
+ */
31
+ export interface ExternalizedToolInputRef {
32
+ tool_name: string;
33
+ input_path: ['content'];
34
+ ref: TextArtifactReference;
35
+ }
36
+
37
+ export interface ExternalizedToolInputRefs {
38
+ [toolUseId: string]: ExternalizedToolInputRef[];
39
+ }
40
+
41
+ export function toolInputRefsArtifactPath(storageId: string): string {
42
+ return `agents/${storageId}/tool-input-refs.json`;
43
+ }
44
+
17
45
  /**
18
46
  * Conversation state passed between workflow activities.
19
47
  * Contains all context needed to continue a multi-turn agent conversation.
@@ -51,6 +79,15 @@ export interface ConversationState {
51
79
  /** Compact, redacted latest user intent for reviewer-style system interactions. */
52
80
  latest_user_message?: string;
53
81
 
82
+ /**
83
+ * Transport sidecar for large generated tool input fields.
84
+ *
85
+ * These refs are intentionally kept out of tool_use.tool_input so they are
86
+ * not shown to the model. Tool execution hydrates them from artifact storage
87
+ * immediately before activity validation.
88
+ */
89
+ tool_input_refs?: ExternalizedToolInputRefs;
90
+
54
91
  /**
55
92
  * The output of the this conversation step
56
93
  */
@@ -204,6 +241,16 @@ export interface ConversationState {
204
241
  * to consolidate all artifacts under the parent agent run.
205
242
  */
206
243
  launch_id?: string;
244
+
245
+ /**
246
+ * The exact app version this run is pinned to, derived from the `@version` on the
247
+ * started interaction ref / the `x-vertesia-app-version` header at start. Persisted on the state
248
+ * so it survives resume, and applied to the activity client (`withAppVersion`) so every app-owned
249
+ * ref the run resolves — interactions, types, processes, tools — targets this version instead of
250
+ * the current/promoted one. Undefined → current/promoted. Resolution-time only; never a stored
251
+ * capability-ref version.
252
+ */
253
+ app_version?: string;
207
254
  }
208
255
 
209
256
  /**
@@ -1,104 +1,42 @@
1
- import type { JSONObject } from '../json.js';
2
1
  import type { WorkflowExecutionPayload, WorkflowRunStatus } from './workflow.js';
3
2
 
4
- export interface PdfToRichtextOptions {
3
+ export interface DocumentPrepOptions {
5
4
  features?: string[];
6
5
  debug?: boolean;
6
+ output_format?: DocProcessorOutputFormat;
7
7
  [key: string]: unknown;
8
8
  }
9
9
 
10
- export interface PdfToRichTextWorkflowPayload extends Omit<WorkflowExecutionPayload, 'vars'> {
11
- vars: PdfToRichtextOptions;
12
- }
13
-
14
- export interface TransformTablesWorkflowPayload extends Omit<WorkflowExecutionPayload, 'vars'> {
15
- vars: AdaptTablesParams;
16
- environment?: string;
10
+ export interface DocumentPrepWorkflowPayload extends Omit<WorkflowExecutionPayload, 'vars'> {
11
+ vars: DocumentPrepOptions;
17
12
  }
18
13
 
19
- export interface TransformTablesWorkflowResult {
20
- result_path: string;
21
- status: string;
22
- table_count: number;
23
- item_count: number;
24
- }
25
-
26
- /**
27
- * Represents a image in a document that has been analyzed
28
- */
29
- export interface DocImage {
30
- id?: string;
31
- page_number?: number;
32
- description?: string;
33
- is_meaningful?: boolean;
34
- width?: number;
35
- height?: number;
36
- }
37
-
38
- /**
39
- * The export type formats for tables.
40
- */
41
- export type ExportTableFormats = 'json' | 'csv' | 'xml';
42
-
43
- /**
44
- * Represents a table in a document that has been analyzed
45
- */
46
- export interface DocTable {
47
- page_number?: number;
48
- table_number?: number;
49
- title?: string;
50
- format: 'application/csv' | 'application/json';
51
- }
52
-
53
- /**
54
- * Represents a table in a document that has been analyzed in CSV format
55
- */
56
- export interface DocTableCsv extends DocTable {
57
- format: 'application/csv';
58
- title?: string;
59
- data: string;
60
- }
61
-
62
- /**
63
- * Represents a table in a document that has been analyzed in JSON format
64
- */
65
- export interface DocTableJson extends DocTable {
66
- format: 'application/json';
67
- title?: string;
68
- data: JSONObject[];
69
- }
70
-
71
- export type DocTableResponse = DocTableCsv | DocTableJson;
14
+ export type DocumentProcessingPhase = 'markdown' | 'grounded_extraction';
72
15
 
73
16
  /**
74
17
  * Output format for document processing workflows
75
18
  */
76
- export type DocProcessorOutputFormat = 'xml' | 'markdown';
19
+ export type DocProcessorOutputFormat = 'markdown';
77
20
 
78
21
  /**
79
22
  * Represents a document analysis run status
80
23
  */
81
24
  export interface DocAnalyzeRunStatusResponse extends WorkflowRunStatus {
25
+ phase?: DocumentProcessingPhase;
82
26
  progress?: DocAnalyzerProgress;
83
- /** The output format being used for processing (markdown or xml) */
27
+ /** The output format being used for processing. */
84
28
  output_format?: DocProcessorOutputFormat;
85
29
  }
86
30
 
87
- export interface DocAnalyzerResultResponse {
88
- document?: string;
89
- tables?: DocTableResponse[];
90
- images?: DocImage[];
91
- annotated?: string | null;
92
- }
93
-
94
31
  export interface DocAnalyzerProgress {
32
+ phase?: DocumentProcessingPhase;
95
33
  pages: DocAnalyzerProgressStatus;
96
34
  images: DocAnalyzerProgressStatus;
97
35
  tables: DocAnalyzerProgressStatus;
98
36
  visuals: DocAnalyzerProgressStatus;
99
37
  started_at?: number;
100
38
  percent: number;
101
- /** The output format being used for processing (markdown or xml) */
39
+ /** The output format being used for processing. */
102
40
  output_format?: DocProcessorOutputFormat;
103
41
  }
104
42
 
@@ -176,7 +114,3 @@ export interface AdaptedTable {
176
114
  }
177
115
 
178
116
  export type AdaptedTableResponse = Record<string, AdaptedTable>;
179
-
180
- export interface AnnotatedPdfResponse {
181
- url: string | null;
182
- }
@@ -48,6 +48,7 @@ export interface DSLWorkflowExecutionPayload extends WorkflowExecutionPayload<Re
48
48
  */
49
49
  export interface DSLActivityOptions {
50
50
  startToCloseTimeout?: DurationValue;
51
+ heartbeatTimeout?: DurationValue;
51
52
  scheduleToStartTimeout?: DurationValue;
52
53
  scheduleToCloseTimeout?: DurationValue;
53
54
  retry?: DSLRetryPolicy;
@@ -0,0 +1,154 @@
1
+ import type { InteractionExecutionConfiguration } from '../interaction.js';
2
+ import type { WorkflowRunStatus } from './workflow.js';
3
+
4
+ /**
5
+ * Document-level trust verdict for a grounded extraction. `good_to_go` means the
6
+ * extracted content (after any review corrections) can be used without a human
7
+ * check; `needs_review` means a human should verify it. This reflects content
8
+ * correctness, not how many citation boxes rendered.
9
+ */
10
+ export type GroundedExtractionVerdict = 'good_to_go' | 'needs_review';
11
+
12
+ /**
13
+ * Canonical workflow id for object-scoped grounded extraction. Every entry point
14
+ * must use this id so Temporal prevents overlapping extraction runs per object.
15
+ */
16
+ export function getGroundedExtractionWorkflowId(accountId: string, objectId: string): string {
17
+ return `${accountId.slice(0, 6)}:workflow_execution_request:${objectId}:grounded`;
18
+ }
19
+
20
+ /**
21
+ * Request body to start a grounded extraction on a content object. All fields are
22
+ * optional: with none set, the object's own content-type schema drives the
23
+ * extraction with default models and settings.
24
+ */
25
+ export interface GroundedExtractionRequest {
26
+ /** JSON schema describing the data to extract. Takes precedence over type_ref. */
27
+ schema?: Record<string, unknown>;
28
+ /** Content type id or catalog ref whose object_schema drives the extraction. */
29
+ type_ref?: string;
30
+ /** Interaction to use. Defaults to sys:ExtractInformationGrounded. */
31
+ interaction_name?: string;
32
+ /** Maximum number of pages to process. */
33
+ max_pages?: number;
34
+ /** Run OCR on every page even when a text layer exists. */
35
+ force_ocr?: boolean;
36
+ /** Re-run OCR on pages that need it instead of restoring the stored OCR result. */
37
+ refresh_ocr?: boolean;
38
+ /** Attach clean page images for layout/semantic context; direct-vision pages also receive checkerboards. */
39
+ use_vision?: boolean;
40
+ /**
41
+ * A1 locate-grid cell size in PDF points for vision pages (drives both the drawn
42
+ * grid and cell→box resolution). Smaller = finer grid / more cells. Default: 14.
43
+ */
44
+ grid_cell_pt?: number;
45
+ /**
46
+ * How to read pages that have no digital text layer (scans / image-only pages).
47
+ * 'vision' (default): read them off the page image with the extraction model,
48
+ * skipping OCR entirely. 'ocr': legacy path — OCR those pages and block-ground on
49
+ * the (lossy) OCR text. Set to 'ocr' to revert to the pre-vision behavior.
50
+ */
51
+ raster_mode?: 'vision' | 'ocr';
52
+ /** Maximum pages per extraction call; larger documents are split into sequential windows. */
53
+ window_pages?: number;
54
+ /**
55
+ * Extract with an autonomous agent (views the whole document at once) instead of
56
+ * the deterministic windowed pipeline. Sidesteps window-boundary splits on long
57
+ * documents. The workflow stages the artifacts into an agent space, runs a
58
+ * conversation agent that writes the extraction, then folds it back.
59
+ */
60
+ agentic_extraction?: boolean;
61
+ /** Agent interaction for agentic_extraction. Defaults to sys:GeneralAgent. */
62
+ extract_agent?: string;
63
+ /** Update the object's properties with the extracted data. Default: true. */
64
+ update_properties?: boolean;
65
+ /** LLM execution configuration (model, environment, ...) for the main pass. */
66
+ config?: InteractionExecutionConfiguration;
67
+ /** Execution configuration used instead of `config` on hard content (scans, handwriting). */
68
+ hard_config?: InteractionExecutionConfiguration;
69
+ /** Hardness score (0..1) at or above which `hard_config` is used. Default: 0.5. */
70
+ hardness_threshold?: number;
71
+ /** Execution configuration for the post-extraction review pass. No review runs when absent. */
72
+ review_config?: InteractionExecutionConfiguration;
73
+ /** Hardness score (0..1) at or above which the review runs. Defaults to hardness_threshold. */
74
+ review_threshold?: number;
75
+ /** Review triggers when any page's citation coverage falls below this floor. Default: 0.2. */
76
+ coverage_review_threshold?: number;
77
+ /** Run the model review even when every citation was digitally verified. Requires review_config. */
78
+ force_review?: boolean;
79
+ /**
80
+ * Free-text operator guidance folded into the extraction prompt to steer a
81
+ * (re-)extraction, e.g. "part numbers are in the third column; some line items
82
+ * wrap onto the next row".
83
+ */
84
+ operator_instructions?: string;
85
+ /** Interactive assistant only: the operator's opening message for the assistant conversation. */
86
+ user_prompt?: string;
87
+ }
88
+
89
+ /**
90
+ * Response from starting the interactive grounded extraction assistant. The agent
91
+ * run + conversation workflow are launched server-side (recordRun -> stage the
92
+ * document into the agent space -> launch the interactive conversation); the
93
+ * client renders the conversation with `agent_run_id`.
94
+ */
95
+ export interface GroundedAssistantResponse {
96
+ /** The AgentRun id to stream/render the conversation. */
97
+ agent_run_id: string;
98
+ /** The conversation workflow id backing the run. */
99
+ workflow_id: string;
100
+ /** The object the assistant is scoped to. */
101
+ object_id: string;
102
+ }
103
+
104
+ /**
105
+ * Status of a grounded extraction workflow. Carries the doc-level verdict once the
106
+ * run has completed and written its result.
107
+ */
108
+ export interface GroundedExtractionRunStatusResponse extends WorkflowRunStatus {
109
+ /** The trust verdict, present once the run has completed. */
110
+ verdict?: GroundedExtractionVerdict;
111
+ }
112
+
113
+ /**
114
+ * How each extracted value was verified. Two kinds, both trustworthy:
115
+ * digitally verified (matched the document's text — digital layer or OCR) and
116
+ * AI verified (the reviewer confirmed it against the page image — the primary
117
+ * signal for scanned or handwritten content, which has no text layer to match).
118
+ */
119
+ export interface GroundedVerificationBreakdown {
120
+ /** Total number of cited values. */
121
+ total: number;
122
+ /** Values matched verbatim against the document's text (digital layer or OCR). */
123
+ digitally_verified: number;
124
+ /** Values the reviewer model confirmed against the page image. */
125
+ ai_verified: number;
126
+ /** Values read from the image but neither text-matched nor reviewer-confirmed. */
127
+ unverified: number;
128
+ }
129
+
130
+ /**
131
+ * Completed grounded extraction result: the extracted data with its trust verdict
132
+ * and verification breakdown, plus a download URL for the full citations artifact.
133
+ */
134
+ export interface GroundedExtractionResultResponse {
135
+ object_id: string;
136
+ /** The extracted data, shaped by the requested schema. */
137
+ data: Record<string, unknown>;
138
+ /** Document-level trust verdict. */
139
+ verdict?: GroundedExtractionVerdict;
140
+ /** One-sentence rationale for the verdict. */
141
+ verdict_reason?: string;
142
+ /** Mean citation confidence in [0,1]. */
143
+ confidence?: number;
144
+ /** Per-value verification breakdown. */
145
+ verification: GroundedVerificationBreakdown;
146
+ /** Review outcome, when a review pass ran. */
147
+ review?: {
148
+ assessment: 'complete' | 'issues_found';
149
+ summary?: string;
150
+ corrections_applied?: number;
151
+ };
152
+ /** Signed download URL for the full grounded-extraction.json (data + citations + boxes). */
153
+ result_url?: string | null;
154
+ }
@@ -6,6 +6,7 @@ export * from './common.js';
6
6
  export * from './conversation-state.js';
7
7
  export * from './doc-analyzer.js';
8
8
  export * from './dsl-workflow.js';
9
+ export * from './grounded-extraction.js';
9
10
  export * from './hive-memory.js';
10
11
  export * from './object-types.js';
11
12
  export * from './process.js';