@vertesia/common 1.5.0-dev.20260714.072725Z → 1.5.0-dev.20260722.120446Z

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (127) hide show
  1. package/lib/apikey.d.ts +1 -0
  2. package/lib/apikey.d.ts.map +1 -1
  3. package/lib/apikey.js.map +1 -1
  4. package/lib/apps.d.ts +500 -34
  5. package/lib/apps.d.ts.map +1 -1
  6. package/lib/apps.js +63 -70
  7. package/lib/apps.js.map +1 -1
  8. package/lib/audit-trail.d.ts +61 -1
  9. package/lib/audit-trail.d.ts.map +1 -1
  10. package/lib/audit-trail.js +15 -0
  11. package/lib/audit-trail.js.map +1 -1
  12. package/lib/data-platform.d.ts +121 -5
  13. package/lib/data-platform.d.ts.map +1 -1
  14. package/lib/index.d.ts +7 -0
  15. package/lib/index.d.ts.map +1 -1
  16. package/lib/index.js +7 -0
  17. package/lib/index.js.map +1 -1
  18. package/lib/interaction.d.ts +19 -1
  19. package/lib/interaction.d.ts.map +1 -1
  20. package/lib/interaction.js.map +1 -1
  21. package/lib/json-schema.d.ts +1 -1
  22. package/lib/json-schema.d.ts.map +1 -1
  23. package/lib/platform-event.d.ts +79 -3
  24. package/lib/platform-event.d.ts.map +1 -1
  25. package/lib/platform-event.js.map +1 -1
  26. package/lib/project.d.ts +119 -20
  27. package/lib/project.d.ts.map +1 -1
  28. package/lib/project.js +65 -0
  29. package/lib/project.js.map +1 -1
  30. package/lib/query.d.ts +6 -0
  31. package/lib/query.d.ts.map +1 -1
  32. package/lib/refs.d.ts +1 -0
  33. package/lib/refs.d.ts.map +1 -1
  34. package/lib/schema-for-extraction.d.ts +21 -0
  35. package/lib/schema-for-extraction.d.ts.map +1 -0
  36. package/lib/schema-for-extraction.js +205 -0
  37. package/lib/schema-for-extraction.js.map +1 -0
  38. package/lib/store/agent-run.d.ts +2 -0
  39. package/lib/store/agent-run.d.ts.map +1 -1
  40. package/lib/store/conversation-state.d.ts +39 -0
  41. package/lib/store/conversation-state.d.ts.map +1 -1
  42. package/lib/store/conversation-state.js +3 -0
  43. package/lib/store/conversation-state.js.map +1 -1
  44. package/lib/store/doc-analyzer.d.ts +10 -67
  45. package/lib/store/doc-analyzer.d.ts.map +1 -1
  46. package/lib/store/dsl-workflow.d.ts +1 -0
  47. package/lib/store/dsl-workflow.d.ts.map +1 -1
  48. package/lib/store/dsl-workflow.js.map +1 -1
  49. package/lib/store/grounded-extraction.d.ts +146 -0
  50. package/lib/store/grounded-extraction.d.ts.map +1 -0
  51. package/lib/store/grounded-extraction.js +8 -0
  52. package/lib/store/grounded-extraction.js.map +1 -0
  53. package/lib/store/index.d.ts +1 -0
  54. package/lib/store/index.d.ts.map +1 -1
  55. package/lib/store/index.js +1 -0
  56. package/lib/store/index.js.map +1 -1
  57. package/lib/store/store.d.ts +307 -1
  58. package/lib/store/store.d.ts.map +1 -1
  59. package/lib/store/store.js +438 -0
  60. package/lib/store/store.js.map +1 -1
  61. package/lib/store/workflow.d.ts +3 -0
  62. package/lib/store/workflow.d.ts.map +1 -1
  63. package/lib/store/workflow.js.map +1 -1
  64. package/lib/user.d.ts +14 -0
  65. package/lib/user.d.ts.map +1 -1
  66. package/lib/user.js +31 -0
  67. package/lib/user.js.map +1 -1
  68. package/lib/vertesia-common.js +2 -2
  69. package/lib/vertesia-common.js.map +1 -1
  70. package/lib/view-configuration-validation.d.ts +15 -0
  71. package/lib/view-configuration-validation.d.ts.map +1 -0
  72. package/lib/view-configuration-validation.js +63 -0
  73. package/lib/view-configuration-validation.js.map +1 -0
  74. package/lib/view-query-validation.d.ts +13 -0
  75. package/lib/view-query-validation.d.ts.map +1 -0
  76. package/lib/view-query-validation.js +266 -0
  77. package/lib/view-query-validation.js.map +1 -0
  78. package/lib/view-validation-helpers.d.ts +15 -0
  79. package/lib/view-validation-helpers.d.ts.map +1 -0
  80. package/lib/view-validation-helpers.js +25 -0
  81. package/lib/view-validation-helpers.js.map +1 -0
  82. package/lib/views-schema.d.ts +1992 -0
  83. package/lib/views-schema.d.ts.map +1 -0
  84. package/lib/views-schema.js +674 -0
  85. package/lib/views-schema.js.map +1 -0
  86. package/lib/views-validation.d.ts +21 -0
  87. package/lib/views-validation.d.ts.map +1 -0
  88. package/lib/views-validation.js +164 -0
  89. package/lib/views-validation.js.map +1 -0
  90. package/lib/views.d.ts +381 -0
  91. package/lib/views.d.ts.map +1 -0
  92. package/lib/views.js +41 -0
  93. package/lib/views.js.map +1 -0
  94. package/package.json +5 -4
  95. package/src/apikey.ts +1 -0
  96. package/src/apps.test.ts +9 -1
  97. package/src/apps.ts +583 -90
  98. package/src/audit-trail.ts +83 -0
  99. package/src/data-platform.ts +129 -5
  100. package/src/index.ts +12 -0
  101. package/src/interaction.ts +20 -1
  102. package/src/json-schema.ts +0 -1
  103. package/src/platform-event.ts +92 -2
  104. package/src/project.test.ts +44 -0
  105. package/src/project.ts +205 -22
  106. package/src/query.ts +6 -0
  107. package/src/refs.ts +1 -0
  108. package/src/roles.test.ts +32 -0
  109. package/src/schema-for-extraction.test.ts +191 -0
  110. package/src/schema-for-extraction.ts +231 -0
  111. package/src/store/agent-run.ts +2 -0
  112. package/src/store/conversation-state.ts +47 -0
  113. package/src/store/doc-analyzer.ts +10 -76
  114. package/src/store/dsl-workflow.ts +1 -0
  115. package/src/store/grounded-extraction.ts +154 -0
  116. package/src/store/index.ts +1 -0
  117. package/src/store/store.ts +778 -1
  118. package/src/store/workflow.ts +3 -0
  119. package/src/user.ts +46 -0
  120. package/src/view-configuration-validation.ts +74 -0
  121. package/src/view-query-validation.test.ts +21 -0
  122. package/src/view-query-validation.ts +319 -0
  123. package/src/view-validation-helpers.ts +28 -0
  124. package/src/views-schema.test.ts +364 -0
  125. package/src/views-schema.ts +689 -0
  126. package/src/views-validation.ts +234 -0
  127. package/src/views.ts +484 -0
@@ -1,4 +1,6 @@
1
+ import type { JSONSchemaType } from 'ajv';
1
2
  import type { ComputedFacetResponse } from '../facets.js';
3
+ import type { InteractionExecutionConfiguration } from '../interaction.js';
2
4
  import type { JSONObject } from '../json.js';
3
5
  import type { SearchPayload } from '../payload.js';
4
6
  import type { SupportedEmbeddingTypes } from '../project.js';
@@ -422,6 +424,12 @@ export interface GenerationRunMetadata {
422
424
  date: string;
423
425
  model: string;
424
426
  target?: string;
427
+ /**
428
+ * Fingerprint of the inputs used by property extraction (content etag, type + its object
429
+ * schema, source, instructions, interaction). Lets a later run skip re-extraction when
430
+ * nothing changed.
431
+ */
432
+ extraction_fingerprint?: string;
425
433
  }
426
434
 
427
435
  // Base rendition interface for document and audio
@@ -448,7 +456,89 @@ export interface ContentMetadata {
448
456
  location?: Location;
449
457
  generation_runs?: GenerationRunMetadata[];
450
458
  etag?: string;
459
+ /** ETag of text materialized from object properties by intake rendering. */
460
+ rendered_text_etag?: string;
451
461
  renditions?: Rendition[];
462
+ /**
463
+ * Embedded/technical metadata harvested from the source file by intake
464
+ * (office docProps, PDF docinfo). Free-form, nature-appropriate keys.
465
+ */
466
+ embedded?: Record<string, unknown>;
467
+ /** Type-detection provenance recorded by the intake sniff pipeline. */
468
+ type_detection?: TypeDetectionMetadata;
469
+ /** Locate-pass provenance: which pages the document map found relevant. */
470
+ locate?: LocateMetadata;
471
+ /** Vision-evidence provenance for the last visual extraction run. */
472
+ vision_evidence?: VisionEvidenceMetadata;
473
+ }
474
+
475
+ /**
476
+ * Provenance persisted at `metadata.locate` when the intake locate (document-map) pass runs.
477
+ * The page list doubles as navigation metadata for the UI.
478
+ */
479
+ export interface LocateMetadata {
480
+ /** Relevant pages proposed by the locate pass, in plan-ranked order (1-based). */
481
+ pages: number[];
482
+ /** Detail profile the plan requested for visual extraction. */
483
+ visual_detail?: 'low' | 'standard' | 'high';
484
+ /** Whether the plan asked for color rendering. */
485
+ needs_color?: boolean;
486
+ /** The model's one-line explanation of the selection. */
487
+ reason?: string;
488
+ page_count?: number;
489
+ /** Pages per contact sheet used for the pass (8 or 16). */
490
+ detail?: number;
491
+ sheet_count?: number;
492
+ located_at: string;
493
+ }
494
+
495
+ /**
496
+ * Provenance persisted at `metadata.vision_evidence` whenever intake prepares scoped page
497
+ * images for visual extraction (design: vision evidence spec — dropped pages are recorded,
498
+ * never silently batched).
499
+ */
500
+ export interface VisionEvidenceMetadata {
501
+ /** Extraction source that requested the evidence. */
502
+ source_requested?: 'auto' | 'text' | 'vision' | 'mixed';
503
+ /** Pages rendered and sent as evidence, in ranked order (1-based). */
504
+ pages_sent: number[];
505
+ /** Resolved detail profile name. */
506
+ detail: 'low' | 'standard' | 'high';
507
+ /** Candidate pages dropped by budget clamping (recorded, not batched). */
508
+ dropped_pages?: number[];
509
+ /** The locate plan's reason, when the plan drove the page selection. */
510
+ plan_reason?: string;
511
+ /** Which clamps fired (page_count, allowed_details, token budget, page caps, payload). */
512
+ clamps_applied?: string[];
513
+ /** Estimated image tokens for the pages sent. */
514
+ est_tokens?: number;
515
+ page_count?: number;
516
+ prepared_at: string;
517
+ }
518
+
519
+ /**
520
+ * Durable provenance persisted at `metadata.type_detection` whenever the intake sniff pipeline
521
+ * runs. `method` records which mechanism decides the type: the sniff itself (high confidence),
522
+ * the post-conversion selector (medium/low/other), or the post-conversion selector because the
523
+ * document was below the small-doc page threshold.
524
+ */
525
+ export interface TypeDetectionMetadata {
526
+ method: 'sniff' | 'post_conversion' | 'post_conversion_small_doc';
527
+ /** Sniffed type id, or 'other'. */
528
+ type?: string;
529
+ type_name?: string;
530
+ /** Sniff confidence, 0..1. */
531
+ confidence?: number;
532
+ band?: 'high' | 'medium' | 'low';
533
+ rationale?: string;
534
+ alternates?: string[];
535
+ /** Which evidence kinds the sniff saw. */
536
+ evidence?: 'text' | 'image' | 'both';
537
+ page_count?: number;
538
+ /** Why the sniff LLM call was skipped (e.g. 'below_min_pages'). */
539
+ skipped_reason?: string;
540
+ min_pages?: number;
541
+ detected_at: string;
452
542
  }
453
543
 
454
544
  // Type-specific metadata interfaces
@@ -491,10 +581,34 @@ export interface DocumentMetadata extends ContentMetadata {
491
581
  image_count?: number;
492
582
  zone_count?: number;
493
583
  needs_ocr_count?: number;
584
+ /** Fingerprint of source+policy used for custom conversion, to skip re-converting unchanged docs. */
585
+ conversion_fingerprint?: string;
494
586
  };
587
+ /**
588
+ * Grounded-extraction trust signal + key data. Written by the grounded pipeline
589
+ * (verdict, confidence, citation counts, review status, source etag, ...) and
590
+ * queryable for list/filter. Open-ended so more grounded key-data can be stored
591
+ * without a type change.
592
+ */
593
+ grounded?: GroundedMetadata;
495
594
  sections?: TextSection[]; // List of sections with descriptions and line indexes
496
595
  }
497
596
 
597
+ /** Grounded-extraction summary stored on document metadata. Additional keys allowed. */
598
+ export interface GroundedMetadata {
599
+ verdict?: string;
600
+ confidence?: number;
601
+ citation_count?: number;
602
+ verified_citations?: number;
603
+ reviewed_at?: string;
604
+ generated_at?: string;
605
+ /** Source PDF content etag used by the grounded extraction. */
606
+ source_content_etag?: string | null;
607
+ /** @deprecated Grounded source identity is tracked by source_content_etag. */
608
+ source_text_etag?: string | null;
609
+ [key: string]: unknown;
610
+ }
611
+
498
612
  export interface Transcript {
499
613
  text?: string;
500
614
  segments?: TranscriptSegment[];
@@ -685,6 +799,12 @@ interface StoredTypeRef {
685
799
  */
686
800
  id: string;
687
801
  name: string;
802
+ /**
803
+ * Display hint from the type's intake policy (`intake.default_view`). Enriched by the
804
+ * API on single-object reads so clients can pick the initial view without fetching the
805
+ * type. Absent on list responses and older servers.
806
+ */
807
+ default_view?: ContentTypeIntakePolicy['default_view'];
688
808
  }
689
809
 
690
810
  interface InCodeTypeRef {
@@ -694,6 +814,12 @@ interface InCodeTypeRef {
694
814
  */
695
815
  id: string;
696
816
  name: string;
817
+ /**
818
+ * Display hint from the type's intake policy (`intake.default_view`). Enriched by the
819
+ * API on single-object reads so clients can pick the initial view without fetching the
820
+ * type. Absent on list responses and older servers.
821
+ */
822
+ default_view?: ContentTypeIntakePolicy['default_view'];
697
823
  }
698
824
 
699
825
  export interface ComplexSearchPayload extends Omit<SearchPayload, 'query'> {
@@ -723,9 +849,651 @@ export interface ColumnLayout {
723
849
  */
724
850
  default?: unknown;
725
851
  }
852
+
853
+ export type ContentObjectTypeStatus = 'active' | 'draft';
854
+
855
+ /** Vision detail level names referenced by intake policies. The rendering profiles behind the
856
+ * names (dpi, max size, quality, color mode) are PLATFORM-defined and project-overridable —
857
+ * a type only ever references a detail name. */
858
+ export type IntakeVisionDetail = 'low' | 'standard' | 'high';
859
+
860
+ /**
861
+ * Named page scope for intake conversion/extraction: everything or the locate-pass result.
862
+ * Static page ranges live in the sibling `page_ranges` field (which wins when set) — kept as
863
+ * a SEPARATE field because scalar-or-collection unions generate unstable API clients.
864
+ */
865
+ export type IntakePageScope = 'all' | 'located';
866
+
867
+ /**
868
+ * Static page ranges: inclusive [start, end] pairs; negative indexes count from the end of
869
+ * the document ([[1, 2], [-1, -1]] = first two pages plus the last page).
870
+ */
871
+ export type IntakePageRanges = [number, number][];
872
+
873
+ /** Rendering settings behind a vision detail name (platform defaults, project-overridable
874
+ * via `configuration.intake.vision_profiles`). */
875
+ export interface IntakeVisionProfileSettings {
876
+ /** Render resolution in dots per inch. */
877
+ dpi: number;
878
+ /** Maximum height/width of the rendered page image in pixels. */
879
+ max_hw: number;
880
+ /** JPEG quality (0-100). */
881
+ quality: number;
882
+ /** grayscale renders gray always; auto keeps color when the plan asks for it. */
883
+ color_mode: 'grayscale' | 'auto';
884
+ }
885
+
886
+ export interface ContentTypeExtractionGroundingReviewPolicy {
887
+ /** Set false to disable an inherited grounding review pass for this type. */
888
+ enabled?: boolean;
889
+ /** Model execution configuration for the review interaction. */
890
+ config?: InteractionExecutionConfiguration;
891
+ /** Hardness score at or above which review runs. Defaults to hardness_threshold. */
892
+ threshold?: number;
893
+ /**
894
+ * Review also runs when any page's citation coverage falls below this
895
+ * floor (evidence of missed content). Default 0.2.
896
+ */
897
+ coverage_threshold?: number;
898
+ /** Run review regardless of hardness. */
899
+ force?: boolean;
900
+ }
901
+
902
+ export interface ContentTypeExtractionGroundingPolicy {
903
+ /** Enable PDF block-level citation grounding for property extraction. */
904
+ enabled?: boolean;
905
+ /** Grounded extraction interaction. Defaults to the system grounded extractor. */
906
+ interaction?: string;
907
+ /** Maximum pages to process. */
908
+ max_pages?: number;
909
+ /** Run OCR on every page even when a text layer exists. */
910
+ force_ocr?: boolean;
911
+ /** Attach instrumented page images to the grounded extraction prompt. */
912
+ use_vision?: boolean;
913
+ /**
914
+ * How to read pages with no digital text layer (scans / image-only pages).
915
+ * 'vision' (default): read them off the page image and skip OCR. 'ocr': legacy
916
+ * path — OCR those pages and block-ground on the (lossy) OCR text.
917
+ */
918
+ raster_mode?: 'vision' | 'ocr';
919
+ /**
920
+ * A1 locate-grid cell size in PDF points for vision pages. Smaller = finer grid
921
+ * (more cells, tighter boxes) but can trip weaker models into over-reading;
922
+ * tune per the model in `config`. Default 15.
923
+ */
924
+ grid_cell_pt?: number;
925
+ /**
926
+ * Drop block bounding boxes from the extraction prompt. Only sound with
927
+ * use_vision (layout comes from the image).
928
+ */
929
+ omit_block_boxes?: boolean;
930
+ /** Maximum pages per grounded extraction call before windowing. */
931
+ window_pages?: number;
932
+ /** Update object properties with grounded extraction data. Default true. */
933
+ update_properties?: boolean;
934
+ /** Model execution configuration for the main grounded extraction interaction. */
935
+ config?: InteractionExecutionConfiguration;
936
+ /** Model execution configuration used for hard-to-read content. */
937
+ hard_config?: InteractionExecutionConfiguration;
938
+ /** Hardness score at or above which hard_config is used. Default 0.5. */
939
+ hardness_threshold?: number;
940
+ /**
941
+ * Minimum citations-per-leaf-value ratio; completions below it retry with
942
+ * escalation. Default 0.3.
943
+ */
944
+ min_citation_density?: number;
945
+ /** Re-run OCR instead of restoring durable OCR artifacts (stale pipeline output). */
946
+ refresh_ocr?: boolean;
947
+ /** Optional post-extraction review pass. */
948
+ review?: ContentTypeExtractionGroundingReviewPolicy;
949
+ }
950
+
951
+ /**
952
+ * Per-content-type policy for the standard intake workflows.
953
+ */
954
+ export interface ContentTypeIntakePolicy {
955
+ /** Intake orchestration mode for this type. */
956
+ mode?: 'programmatic' | 'agentic';
957
+ /** Guidance used when selecting or creating this content type. */
958
+ identification?: {
959
+ guidance?: string;
960
+ distinguish_from?: string;
961
+ examples?: string[];
962
+ };
963
+ /**
964
+ * Document-map ("locate") pass: page thumbnails tiled into labeled contact sheets, one
965
+ * vision call returns which pages matter for THIS type. The result can scope conversion
966
+ * and extraction, and doubles as the vision planner for visual extraction.
967
+ */
968
+ locate?: {
969
+ /** What to look for ("commercial terms, payment schedule, signature pages"). */
970
+ instructions: string;
971
+ /** Pages per contact sheet: 8 = bigger tiles (headings readable). Default 16. */
972
+ detail?: 8 | 16;
973
+ /** Only run when the page count is at least this. Default 8. */
974
+ min_pages?: number;
975
+ };
976
+ /** Controls source-to-text conversion before extraction and embedding. */
977
+ text_conversion?: {
978
+ enabled?: boolean;
979
+ method?: 'auto' | 'basic' | 'llm' | 'custom';
980
+ custom?: {
981
+ interaction?: string;
982
+ agent?: string;
983
+ };
984
+ instructions?: string;
985
+ output_format?: 'markdown' | 'text';
986
+ /** Which pages to convert: everything or the locate result. Default all. */
987
+ scope?: IntakePageScope;
988
+ /** Static page ranges to convert (wins over `scope` when set). */
989
+ page_ranges?: IntakePageRanges;
990
+ };
991
+ /** Controls schema-property extraction after type assignment. */
992
+ extraction?: {
993
+ enabled?: boolean;
994
+ source?: 'auto' | 'text' | 'vision' | 'mixed';
995
+ instructions?: string;
996
+ interaction?: string;
997
+ /** Which pages extraction sees: everything or the locate result. */
998
+ scope?: IntakePageScope;
999
+ /** Static page ranges extraction sees (wins over `scope` when set). */
1000
+ page_ranges?: IntakePageRanges;
1001
+ /** Cap on pages sent to extraction. Default 20. */
1002
+ max_pages?: number;
1003
+ /** Vision evidence budget for visual extraction. Detail names reference platform
1004
+ * profiles; the type never defines dpi/quality/resolution. */
1005
+ vision?: {
1006
+ default_detail?: IntakeVisionDetail;
1007
+ allowed_details?: IntakeVisionDetail[];
1008
+ /** PRIMARY budget: estimated image tokens per extraction call. Default 16000. */
1009
+ max_image_tokens?: number;
1010
+ /** Transport guard in megabytes. Default 16. */
1011
+ max_payload_mb?: number;
1012
+ /** Cap on page images per extraction call. Default 8. */
1013
+ max_pages_per_call?: number;
1014
+ };
1015
+ verification?: {
1016
+ enabled?: boolean;
1017
+ model?: string;
1018
+ environment?: string;
1019
+ materiality?: string;
1020
+ threshold?: number;
1021
+ max_retries?: number;
1022
+ on_fail?: 'flag' | 'block';
1023
+ };
1024
+ /** Controls PDF block-level citation grounding with annotated proof output. */
1025
+ grounding?: ContentTypeExtractionGroundingPolicy;
1026
+ };
1027
+ /** Handlebars template used to materialize extracted properties into object text. */
1028
+ rendering_template?: string;
1029
+ /** Per-type embedding switches. Unspecified values inherit the project policy. */
1030
+ embeddings?: Partial<Record<SupportedEmbeddingTypes, boolean>>;
1031
+ /** Whether intake should generate a table of contents for matching documents. */
1032
+ generate_toc?: boolean;
1033
+ /** Preferred first view for objects of this type. */
1034
+ default_view?: 'auto' | 'text' | 'pdf' | 'image' | 'properties';
1035
+ }
1036
+
1037
+ /** Reusable sub-schema for IntakePageScope ('all' | 'located'). */
1038
+ const IntakePageScopeSchema = {
1039
+ type: 'string',
1040
+ enum: ['all', 'located'],
1041
+ description: "Named pages selection: 'all' or 'located' (the locate-pass result). Default all.",
1042
+ nullable: true,
1043
+ };
1044
+
1045
+ /** Reusable sub-schema for IntakePageRanges (inclusive [start, end] pairs, negative = from end). */
1046
+ const IntakePageRangesSchema = {
1047
+ type: 'array',
1048
+ items: { type: 'array', items: { type: 'integer' }, minItems: 2, maxItems: 2 },
1049
+ description:
1050
+ 'Static inclusive [start, end] page ranges; negative indexes count from the end ' +
1051
+ '([[1,2],[-1,-1]] = first two pages plus the last). Wins over scope when set.',
1052
+ nullable: true,
1053
+ };
1054
+
1055
+ const IntakeExecutionConfigurationSchema = {
1056
+ type: 'object',
1057
+ description: 'Interaction execution configuration such as model, environment, and model options.',
1058
+ nullable: true,
1059
+ required: [],
1060
+ additionalProperties: true,
1061
+ properties: {
1062
+ id: { type: 'string', nullable: true },
1063
+ environment: { type: 'string', nullable: true },
1064
+ model: { type: 'string', nullable: true },
1065
+ do_validate: { type: 'boolean', nullable: true },
1066
+ run_data: { type: 'string', nullable: true },
1067
+ configMode: { type: 'string', nullable: true },
1068
+ model_options: {
1069
+ type: 'object',
1070
+ nullable: true,
1071
+ required: [],
1072
+ additionalProperties: true,
1073
+ },
1074
+ http_timeout: {
1075
+ type: 'object',
1076
+ nullable: true,
1077
+ required: [],
1078
+ additionalProperties: true,
1079
+ },
1080
+ },
1081
+ };
1082
+
1083
+ const ContentTypeExtractionGroundingPolicySchema = {
1084
+ type: 'object',
1085
+ description:
1086
+ 'PDF block-level citation grounding policy. When enabled, property extraction uses the grounded ' +
1087
+ 'child workflow and stores citations plus annotated proof output.',
1088
+ nullable: true,
1089
+ required: [],
1090
+ additionalProperties: false,
1091
+ properties: {
1092
+ enabled: {
1093
+ type: 'boolean',
1094
+ description: 'Enable PDF block-level citation grounding for this type.',
1095
+ nullable: true,
1096
+ },
1097
+ interaction: {
1098
+ type: 'string',
1099
+ description: 'Grounded extraction interaction id. Omit to use the system grounded extractor.',
1100
+ nullable: true,
1101
+ },
1102
+ max_pages: {
1103
+ type: 'integer',
1104
+ minimum: 1,
1105
+ description: 'Maximum pages to process.',
1106
+ nullable: true,
1107
+ },
1108
+ force_ocr: {
1109
+ type: 'boolean',
1110
+ description: 'Run OCR on every page even when a text layer exists.',
1111
+ nullable: true,
1112
+ },
1113
+ use_vision: {
1114
+ type: 'boolean',
1115
+ description: 'Attach instrumented page images to the grounded extraction prompt.',
1116
+ nullable: true,
1117
+ },
1118
+ raster_mode: {
1119
+ type: 'string',
1120
+ enum: ['vision', 'ocr'],
1121
+ description:
1122
+ "How to read pages with no digital text layer (scans). 'vision' (default) reads off the page image and skips OCR; 'ocr' is the legacy OCR-then-block-ground path.",
1123
+ nullable: true,
1124
+ },
1125
+ grid_cell_pt: {
1126
+ type: 'number',
1127
+ minimum: 1,
1128
+ description:
1129
+ 'A1 locate-grid cell size in PDF points for vision pages. Smaller = finer grid; tune per the model in config. Default 14.',
1130
+ nullable: true,
1131
+ },
1132
+ omit_block_boxes: {
1133
+ type: 'boolean',
1134
+ description: 'Drop block bounding boxes from the extraction prompt (only sound with use_vision).',
1135
+ nullable: true,
1136
+ },
1137
+ window_pages: {
1138
+ type: 'integer',
1139
+ minimum: 1,
1140
+ description:
1141
+ 'Maximum pages per extraction call before sequential window completion (later windows append to prior). Default 3.',
1142
+ nullable: true,
1143
+ },
1144
+ update_properties: {
1145
+ type: 'boolean',
1146
+ description: 'Update object properties with grounded extraction data. Default true.',
1147
+ nullable: true,
1148
+ },
1149
+ config: IntakeExecutionConfigurationSchema,
1150
+ hard_config: IntakeExecutionConfigurationSchema,
1151
+ hardness_threshold: {
1152
+ type: 'number',
1153
+ minimum: 0,
1154
+ maximum: 1,
1155
+ description: 'Hardness score at or above which hard_config is used. Default 0.5.',
1156
+ nullable: true,
1157
+ },
1158
+ min_citation_density: {
1159
+ type: 'number',
1160
+ minimum: 0,
1161
+ maximum: 1,
1162
+ description:
1163
+ 'Minimum citations-per-leaf-value ratio; completions below it retry with escalation. Default 0.3.',
1164
+ nullable: true,
1165
+ },
1166
+ refresh_ocr: {
1167
+ type: 'boolean',
1168
+ description: 'Re-run OCR instead of restoring durable OCR artifacts (stale pipeline output).',
1169
+ nullable: true,
1170
+ },
1171
+ review: {
1172
+ type: 'object',
1173
+ description: 'Optional post-extraction review pass for hard content.',
1174
+ nullable: true,
1175
+ required: [],
1176
+ additionalProperties: false,
1177
+ properties: {
1178
+ enabled: {
1179
+ type: 'boolean',
1180
+ description: 'Set false to disable an inherited grounding review pass for this type.',
1181
+ nullable: true,
1182
+ },
1183
+ config: IntakeExecutionConfigurationSchema,
1184
+ threshold: {
1185
+ type: 'number',
1186
+ minimum: 0,
1187
+ maximum: 1,
1188
+ description: 'Hardness score at or above which review runs.',
1189
+ nullable: true,
1190
+ },
1191
+ coverage_threshold: {
1192
+ type: 'number',
1193
+ minimum: 0,
1194
+ maximum: 1,
1195
+ description:
1196
+ "Review also runs when any page's citation coverage falls below this floor. Default 0.2.",
1197
+ nullable: true,
1198
+ },
1199
+ force: {
1200
+ type: 'boolean',
1201
+ description: 'Run review regardless of hardness.',
1202
+ nullable: true,
1203
+ },
1204
+ },
1205
+ },
1206
+ },
1207
+ };
1208
+
1209
+ /** JSON schema for validating ContentTypeIntakePolicy payloads at API/tool boundaries.
1210
+ * NOTE: typed via a cast because AJV's strict `JSONSchemaType` mapping cannot express the
1211
+ * `[number, number]` pair items of `page_ranges` as a uniform-items array. The runtime
1212
+ * schema is compiled (and thus validated) by every consumer and by the schema-acceptance
1213
+ * unit test in packages/workflows. */
1214
+ export const ContentTypeIntakePolicySchema = {
1215
+ type: 'object',
1216
+ description:
1217
+ 'Per-content-type policy for standard intake: type selection, conversion, extraction, rendering, and embeddings.',
1218
+ required: [],
1219
+ additionalProperties: false,
1220
+ properties: {
1221
+ mode: {
1222
+ type: 'string',
1223
+ enum: ['programmatic', 'agentic'],
1224
+ description:
1225
+ 'Intake orchestration mode. Use programmatic unless the user explicitly asks for agentic intake.',
1226
+ nullable: true,
1227
+ },
1228
+ identification: {
1229
+ type: 'object',
1230
+ description: 'Guidance used by automatic type selection to recognize this type before full extraction.',
1231
+ nullable: true,
1232
+ required: [],
1233
+ additionalProperties: false,
1234
+ properties: {
1235
+ guidance: {
1236
+ type: 'string',
1237
+ description: 'Classifier-facing description of what this type is and when it should be selected.',
1238
+ nullable: true,
1239
+ },
1240
+ distinguish_from: {
1241
+ type: 'string',
1242
+ description: 'How to distinguish this type from common look-alike document types.',
1243
+ nullable: true,
1244
+ },
1245
+ examples: {
1246
+ type: 'array',
1247
+ description: 'Object ids of human-confirmed examples for this type.',
1248
+ nullable: true,
1249
+ items: { type: 'string' },
1250
+ },
1251
+ },
1252
+ },
1253
+ locate: {
1254
+ type: 'object',
1255
+ description:
1256
+ 'Document-map pass: labeled page-thumbnail contact sheets and one vision call return which ' +
1257
+ 'pages matter for this type. Scopes conversion/extraction and plans visual extraction.',
1258
+ nullable: true,
1259
+ required: ['instructions'],
1260
+ additionalProperties: false,
1261
+ properties: {
1262
+ instructions: {
1263
+ type: 'string',
1264
+ description: 'What to look for (e.g. "commercial terms, payment schedule, signature pages").',
1265
+ },
1266
+ detail: {
1267
+ type: 'integer',
1268
+ enum: [8, 16],
1269
+ description: 'Pages per contact sheet: 8 = bigger tiles with readable headings. Default 16.',
1270
+ nullable: true,
1271
+ },
1272
+ min_pages: {
1273
+ type: 'integer',
1274
+ minimum: 0,
1275
+ description: 'Only run the locate pass when the page count is at least this. Default 8.',
1276
+ nullable: true,
1277
+ },
1278
+ },
1279
+ },
1280
+ text_conversion: {
1281
+ type: 'object',
1282
+ description: 'Controls source-to-text conversion before extraction, search, and text embeddings.',
1283
+ nullable: true,
1284
+ required: [],
1285
+ additionalProperties: false,
1286
+ properties: {
1287
+ enabled: {
1288
+ type: 'boolean',
1289
+ description: 'Set false for extraction-only types that should not create converted markdown/text.',
1290
+ nullable: true,
1291
+ },
1292
+ method: {
1293
+ type: 'string',
1294
+ enum: ['auto', 'basic', 'llm', 'custom'],
1295
+ description: 'Conversion method. Use auto unless the user asks for a specific converter.',
1296
+ nullable: true,
1297
+ },
1298
+ custom: {
1299
+ type: 'object',
1300
+ description: 'Custom conversion implementation for method=custom.',
1301
+ nullable: true,
1302
+ required: [],
1303
+ additionalProperties: false,
1304
+ properties: {
1305
+ interaction: {
1306
+ type: 'string',
1307
+ description: 'Interaction id to call for custom conversion.',
1308
+ nullable: true,
1309
+ },
1310
+ agent: {
1311
+ type: 'string',
1312
+ description: 'Agent id to launch for custom conversion.',
1313
+ nullable: true,
1314
+ },
1315
+ },
1316
+ },
1317
+ instructions: {
1318
+ type: 'string',
1319
+ description: 'Instructions for what source content to preserve during conversion.',
1320
+ nullable: true,
1321
+ },
1322
+ output_format: {
1323
+ type: 'string',
1324
+ enum: ['markdown', 'text'],
1325
+ description: 'Output format for converted text. Prefer markdown for readable documents.',
1326
+ nullable: true,
1327
+ },
1328
+ scope: IntakePageScopeSchema,
1329
+ page_ranges: IntakePageRangesSchema,
1330
+ },
1331
+ },
1332
+ extraction: {
1333
+ type: 'object',
1334
+ description: 'Controls structured property extraction against the content type object_schema.',
1335
+ nullable: true,
1336
+ required: [],
1337
+ additionalProperties: false,
1338
+ properties: {
1339
+ enabled: {
1340
+ type: 'boolean',
1341
+ description: 'Whether intake should extract structured properties for this type.',
1342
+ nullable: true,
1343
+ },
1344
+ source: {
1345
+ type: 'string',
1346
+ enum: ['auto', 'text', 'vision', 'mixed'],
1347
+ description:
1348
+ 'Evidence source for extraction: auto chooses text or vision, text sends text only, vision sends image/PDF evidence only, mixed sends both.',
1349
+ nullable: true,
1350
+ },
1351
+ instructions: {
1352
+ type: 'string',
1353
+ description: 'Type-specific extraction instructions such as pages or sections to ignore.',
1354
+ nullable: true,
1355
+ },
1356
+ interaction: {
1357
+ type: 'string',
1358
+ description: 'Interaction id used for extraction. Omit to use the system extractor.',
1359
+ nullable: true,
1360
+ },
1361
+ scope: IntakePageScopeSchema,
1362
+ page_ranges: IntakePageRangesSchema,
1363
+ max_pages: {
1364
+ type: 'integer',
1365
+ minimum: 1,
1366
+ description: 'Cap on pages sent to extraction. Default 20.',
1367
+ nullable: true,
1368
+ },
1369
+ vision: {
1370
+ type: 'object',
1371
+ description:
1372
+ 'Vision evidence budget for visual extraction. Detail names reference platform-defined ' +
1373
+ 'profiles; the type never sets dpi, quality, or resolution.',
1374
+ nullable: true,
1375
+ required: [],
1376
+ additionalProperties: false,
1377
+ properties: {
1378
+ default_detail: {
1379
+ type: 'string',
1380
+ enum: ['low', 'standard', 'high'],
1381
+ description: 'Detail profile used when the plan does not request one. Default standard.',
1382
+ nullable: true,
1383
+ },
1384
+ allowed_details: {
1385
+ type: 'array',
1386
+ items: { type: 'string', enum: ['low', 'standard', 'high'] },
1387
+ description: 'Detail profiles the plan may request. Others fall back to default_detail.',
1388
+ nullable: true,
1389
+ },
1390
+ max_image_tokens: {
1391
+ type: 'integer',
1392
+ minimum: 1,
1393
+ description: 'PRIMARY budget: estimated image tokens per extraction call. Default 16000.',
1394
+ nullable: true,
1395
+ },
1396
+ max_payload_mb: {
1397
+ type: 'number',
1398
+ minimum: 1,
1399
+ description: 'Transport guard in megabytes. Default 16.',
1400
+ nullable: true,
1401
+ },
1402
+ max_pages_per_call: {
1403
+ type: 'integer',
1404
+ minimum: 1,
1405
+ description: 'Cap on page images per extraction call. Default 8.',
1406
+ nullable: true,
1407
+ },
1408
+ },
1409
+ },
1410
+ verification: {
1411
+ type: 'object',
1412
+ description: 'Optional safe-mode verification of extracted values against source evidence.',
1413
+ nullable: true,
1414
+ required: [],
1415
+ additionalProperties: false,
1416
+ properties: {
1417
+ enabled: {
1418
+ type: 'boolean',
1419
+ description: 'Whether to verify extracted values after extraction.',
1420
+ nullable: true,
1421
+ },
1422
+ model: { type: 'string', description: 'Verifier model override.', nullable: true },
1423
+ environment: {
1424
+ type: 'string',
1425
+ description: 'Verifier environment override.',
1426
+ nullable: true,
1427
+ },
1428
+ materiality: {
1429
+ type: 'string',
1430
+ description: 'What errors are material for this type.',
1431
+ nullable: true,
1432
+ },
1433
+ threshold: {
1434
+ type: 'number',
1435
+ minimum: 0,
1436
+ maximum: 1,
1437
+ description: 'Minimum verification confidence before flag/block behavior applies.',
1438
+ nullable: true,
1439
+ },
1440
+ max_retries: {
1441
+ type: 'integer',
1442
+ minimum: 0,
1443
+ description: 'Maximum extraction retries when verification fails.',
1444
+ nullable: true,
1445
+ },
1446
+ on_fail: {
1447
+ type: 'string',
1448
+ enum: ['flag', 'block'],
1449
+ description: 'Failure behavior after retries: flag for review or block property write.',
1450
+ nullable: true,
1451
+ },
1452
+ },
1453
+ },
1454
+ grounding: ContentTypeExtractionGroundingPolicySchema,
1455
+ },
1456
+ },
1457
+ rendering_template: {
1458
+ type: 'string',
1459
+ description: 'Handlebars template used to materialize extracted properties into object text.',
1460
+ nullable: true,
1461
+ },
1462
+ embeddings: {
1463
+ type: 'object',
1464
+ description: 'Per-type embedding switches. Omitted fields inherit the project embedding policy.',
1465
+ nullable: true,
1466
+ required: [],
1467
+ additionalProperties: false,
1468
+ properties: {
1469
+ text: { type: 'boolean', description: 'Whether to generate text embeddings.', nullable: true },
1470
+ image: { type: 'boolean', description: 'Whether to generate image embeddings.', nullable: true },
1471
+ properties: {
1472
+ type: 'boolean',
1473
+ description: 'Whether to generate property embeddings.',
1474
+ nullable: true,
1475
+ },
1476
+ },
1477
+ },
1478
+ generate_toc: {
1479
+ type: 'boolean',
1480
+ description: 'Whether intake should generate table-of-contents sections for this type.',
1481
+ nullable: true,
1482
+ },
1483
+ default_view: {
1484
+ type: 'string',
1485
+ enum: ['auto', 'text', 'pdf', 'image', 'properties'],
1486
+ description: 'Preferred first view for objects of this type.',
1487
+ nullable: true,
1488
+ },
1489
+ },
1490
+ } as unknown as JSONSchemaType<ContentTypeIntakePolicy>;
1491
+
726
1492
  export interface ContentObjectType extends ContentObjectTypeItem {}
727
1493
  export interface ContentObjectTypeItem extends BaseObject {
1494
+ status?: ContentObjectTypeStatus;
728
1495
  is_chunkable?: boolean;
1496
+ intake?: ContentTypeIntakePolicy;
729
1497
  /**
730
1498
  * This is only included in ContentObjectTypeItem if explicitly requested
731
1499
  * It is always included in ContentObjectType
@@ -744,7 +1512,16 @@ export interface ContentObjectTypeItem extends BaseObject {
744
1512
  }
745
1513
  export type InCodeTypeDefinition = Pick<
746
1514
  ContentObjectTypeItem,
747
- 'id' | 'name' | 'description' | 'tags' | 'object_schema' | 'table_layout' | 'is_chunkable' | 'strict_mode'
1515
+ | 'id'
1516
+ | 'name'
1517
+ | 'description'
1518
+ | 'tags'
1519
+ | 'object_schema'
1520
+ | 'table_layout'
1521
+ | 'is_chunkable'
1522
+ | 'strict_mode'
1523
+ | 'status'
1524
+ | 'intake'
748
1525
  >;
749
1526
  export interface ContentObjectTypeCatalogEntry extends InCodeTypeDefinition {
750
1527
  updated_by?: string;