onyx-database 2.2.0__tar.gz → 2.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. {onyx_database-2.2.0 → onyx_database-2.4.0}/CHANGELOG.md +29 -0
  2. {onyx_database-2.2.0 → onyx_database-2.4.0}/PKG-INFO +221 -6
  3. {onyx_database-2.2.0 → onyx_database-2.4.0}/README.md +220 -5
  4. {onyx_database-2.2.0 → onyx_database-2.4.0}/onyx.schema.json +1 -1
  5. onyx_database-2.4.0/onyx_database/__init__.py +183 -0
  6. onyx_database-2.4.0/onyx_database/helpers/candidate_search.py +642 -0
  7. onyx_database-2.4.0/onyx_database/helpers/conditions.py +277 -0
  8. {onyx_database-2.2.0 → onyx_database-2.4.0}/onyx_database/onyx.py +66 -8
  9. {onyx_database-2.2.0 → onyx_database-2.4.0}/onyx_database/onyx_async.py +66 -8
  10. onyx_database-2.4.0/onyx_database/py.typed +1 -0
  11. {onyx_database-2.2.0 → onyx_database-2.4.0}/onyx_database/query_builder.py +227 -5
  12. {onyx_database-2.2.0 → onyx_database-2.4.0}/onyx_database/query_builder_async.py +158 -7
  13. onyx_database-2.4.0/onyx_database/types.py +191 -0
  14. {onyx_database-2.2.0 → onyx_database-2.4.0}/pyproject.toml +1 -1
  15. {onyx_database-2.2.0 → onyx_database-2.4.0}/schema/onyx.schema.json +1 -1
  16. {onyx_database-2.2.0 → onyx_database-2.4.0}/scripts/bump-version.sh +1 -1
  17. onyx_database-2.4.0/tests/test_candidate_search.py +558 -0
  18. {onyx_database-2.2.0 → onyx_database-2.4.0}/tests/test_entity_wire.py +69 -0
  19. onyx_database-2.4.0/tests/test_search.py +526 -0
  20. onyx_database-2.2.0/onyx_database/__init__.py +0 -113
  21. onyx_database-2.2.0/onyx_database/helpers/conditions.py +0 -115
  22. onyx_database-2.2.0/onyx_database/types.py +0 -33
  23. onyx_database-2.2.0/tests/test_search.py +0 -78
  24. {onyx_database-2.2.0 → onyx_database-2.4.0}/.github/workflows/ci.yml +0 -0
  25. {onyx_database-2.2.0 → onyx_database-2.4.0}/.github/workflows/publish.yml +0 -0
  26. {onyx_database-2.2.0 → onyx_database-2.4.0}/.gitignore +0 -0
  27. {onyx_database-2.2.0 → onyx_database-2.4.0}/CONTRIBUTING.md +0 -0
  28. {onyx_database-2.2.0 → onyx_database-2.4.0}/LICENSE +0 -0
  29. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/README.md +0 -0
  30. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/__init__.py +0 -0
  31. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/ai/chat.py +0 -0
  32. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/ai/chat_shorthand.py +0 -0
  33. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/ai/chat_shorthand_stream.py +0 -0
  34. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/ai/chat_stream.py +0 -0
  35. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/ai/models.py +0 -0
  36. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/ai/script_approval.py +0 -0
  37. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/ai/streaming.py +0 -0
  38. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/delete/by_id.py +0 -0
  39. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/delete/query.py +0 -0
  40. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/document/save_get_delete_document.py +0 -0
  41. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/query/aggregate_avg.py +0 -0
  42. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/query/aggregates_with_grouping.py +0 -0
  43. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/query/basic.py +0 -0
  44. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/query/basic_inference.py +0 -0
  45. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/query/compound.py +0 -0
  46. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/query/find_by_id.py +0 -0
  47. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/query/first_or_none.py +0 -0
  48. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/query/format.py +0 -0
  49. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/query/full_text_search_all_tables.py +0 -0
  50. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/query/full_text_table_search.py +0 -0
  51. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/query/in_partition.py +0 -0
  52. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/query/inner_query.py +0 -0
  53. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/query/list.py +0 -0
  54. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/query/not_inner_query.py +0 -0
  55. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/query/order_by.py +0 -0
  56. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/query/resolver.py +0 -0
  57. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/query/resolver_with_hints.py +0 -0
  58. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/query/search_by_resolver_fields.py +0 -0
  59. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/query/select_example.py +0 -0
  60. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/query/sorting_and_paging.py +0 -0
  61. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/query/update.py +0 -0
  62. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/save/basic.py +0 -0
  63. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/save/batch_save.py +0 -0
  64. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/save/cascade.py +0 -0
  65. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/save/cascade_builder.py +0 -0
  66. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/schema/basic.py +0 -0
  67. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/secrets/basic.py +0 -0
  68. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/seed.py +0 -0
  69. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/stream/close.py +0 -0
  70. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/stream/create_events.py +0 -0
  71. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/stream/delete_events.py +0 -0
  72. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/stream/query_stream.py +0 -0
  73. {onyx_database-2.2.0 → onyx_database-2.4.0}/examples/stream/update_events.py +0 -0
  74. {onyx_database-2.2.0 → onyx_database-2.4.0}/onyx_database/ai.py +0 -0
  75. {onyx_database-2.2.0 → onyx_database-2.4.0}/onyx_database/config.py +0 -0
  76. {onyx_database-2.2.0 → onyx_database-2.4.0}/onyx_database/entity_wire.py +0 -0
  77. {onyx_database-2.2.0 → onyx_database-2.4.0}/onyx_database/errors.py +0 -0
  78. {onyx_database-2.2.0 → onyx_database-2.4.0}/onyx_database/helpers/__init__.py +0 -0
  79. {onyx_database-2.2.0 → onyx_database-2.4.0}/onyx_database/helpers/aggregates.py +0 -0
  80. {onyx_database-2.2.0 → onyx_database-2.4.0}/onyx_database/helpers/sort.py +0 -0
  81. {onyx_database-2.2.0 → onyx_database-2.4.0}/onyx_database/http.py +0 -0
  82. {onyx_database-2.2.0 → onyx_database-2.4.0}/onyx_database/query_results.py +0 -0
  83. {onyx_database-2.2.0 → onyx_database-2.4.0}/onyx_database/query_results_async.py +0 -0
  84. {onyx_database-2.2.0 → onyx_database-2.4.0}/onyx_database/stream.py +0 -0
  85. {onyx_database-2.2.0 → onyx_database-2.4.0}/plan/acceptance-criteria.md +0 -0
  86. {onyx_database-2.2.0 → onyx_database-2.4.0}/plan/plan.md +0 -0
  87. {onyx_database-2.2.0 → onyx_database-2.4.0}/scripts/release-flow.md +0 -0
  88. {onyx_database-2.2.0 → onyx_database-2.4.0}/scripts/run-examples.sh +0 -0
  89. {onyx_database-2.2.0 → onyx_database-2.4.0}/tasks/01-http-resilience.md +0 -0
  90. {onyx_database-2.2.0 → onyx_database-2.4.0}/tasks/02-async-client.md +0 -0
  91. {onyx_database-2.2.0 → onyx_database-2.4.0}/tasks/03-query-values.md +0 -0
  92. {onyx_database-2.2.0 → onyx_database-2.4.0}/tasks/04-codegen-pydantic.md +0 -0
  93. {onyx_database-2.2.0 → onyx_database-2.4.0}/tasks/05-cli-parity.md +0 -0
  94. {onyx_database-2.2.0 → onyx_database-2.4.0}/tasks/06-packaging-ci.md +0 -0
  95. {onyx_database-2.2.0 → onyx_database-2.4.0}/tasks/07-onyx-cli-python-exports.md +0 -0
  96. {onyx_database-2.2.0 → onyx_database-2.4.0}/tests/fixtures/entity-wire-v1.json +0 -0
  97. {onyx_database-2.2.0 → onyx_database-2.4.0}/tests/fixtures/entity-wire-v1.msgpack.hex +0 -0
  98. {onyx_database-2.2.0 → onyx_database-2.4.0}/tests/test_ai.py +0 -0
  99. {onyx_database-2.2.0 → onyx_database-2.4.0}/tests/test_async.py +0 -0
  100. {onyx_database-2.2.0 → onyx_database-2.4.0}/tests/test_http.py +0 -0
  101. {onyx_database-2.2.0 → onyx_database-2.4.0}/tests/test_query_results.py +0 -0
@@ -1,5 +1,34 @@
1
1
  # Changelog
2
2
 
3
+ ## Unreleased
4
+
5
+ ## 2.4.0 - 2026-08-29
6
+
7
+ - Added one high-level `search` API for lexical, semantic, and hybrid retrieval
8
+ across sync and async table builders and database-wide facades. The new form
9
+ accepts an options mapping or Python keyword arguments, emits the fail-closed
10
+ `SEARCH` wire operator, defaults to hybrid/any retrieval, validates scores in
11
+ `0..1`, and remains composable with structured filters while rejecting
12
+ mutations.
13
+ - Preserved the exact legacy `MATCHES` wire shape for `search(text)` and
14
+ `search(text, min_score)` calls.
15
+ - Added typed `SearchOptions` overloads and a PEP 561 marker while retaining the
16
+ legacy numeric-score overloads on sync and async builders and database-wide
17
+ facades.
18
+ - Added recursive guards for duplicate or mis-targeted `SEARCH`, mixed
19
+ `__full_text__` predicates, and unsupported live query streams. Database-wide
20
+ search no longer inherits a configured default partition.
21
+
22
+ ## 2.3.0 - 2026-08-29
23
+
24
+ - Added typed, hard-bounded native `CANDIDATES`, `SEARCH_CANDIDATES`, and
25
+ `HNSW_CANDIDATES` query helpers to sync and async builders.
26
+ - Added lossless semantic-signature and HNSW wire validation, including signed
27
+ 64-bit calibration identifiers, mixed-radix bucket checks, vector/work bounds,
28
+ and sole-root admission enforcement.
29
+ - Added the missing public `NOT_BETWEEN` operator/helper so the Python SDK now
30
+ covers every Cloud query operator.
31
+
3
32
  ## 2.2.0 - 2026-08-29
4
33
 
5
34
  - Added sync and async `fenced_save`, `fenced_delete_where`, and `fenced_update_where` helpers for the
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: onyx-database
3
- Version: 2.2.0
3
+ Version: 2.4.0
4
4
  Summary: Python client SDK for Onyx Cloud Database (builder-pattern API and schema helpers)
5
5
  Project-URL: Homepage, https://onyx.dev
6
6
  Project-URL: Repository, https://github.com/OnyxDevTools/onyx-database-python
@@ -199,7 +199,7 @@ db = onyx.init(
199
199
 
200
200
  #### Default partition + logging
201
201
 
202
- - `partition` sets a default partition for queries, `find_by_id`, and deletes by primary key.
202
+ - `partition` sets a default partition for table-scoped queries, `find_by_id`, and deletes by primary key; database-wide `db.search(...)` omits it.
203
203
  - Save operations use the partition field on the entity itself (if present).
204
204
  - `request_logging_enabled` logs HTTP requests and JSON bodies.
205
205
  - `response_logging_enabled` logs HTTP responses and JSON bodies.
@@ -491,7 +491,7 @@ Importable helpers for conditions and sort:
491
491
  from onyx_database import (
492
492
  eq, neq, within, not_within,
493
493
  in_op, not_in,
494
- between,
494
+ between, not_between,
495
495
  gt, gte, lt, lte,
496
496
  like, not_like, contains, not_contains,
497
497
  starts_with, not_starts_with, matches, not_matches,
@@ -503,6 +503,62 @@ from onyx_database import (
503
503
  - Prefer `within` / `not_within` for inclusion checks (supports arrays, comma-separated strings, or inner queries).
504
504
  - `in_op` / `not_in` remain available for backward compatibility and are exact aliases.
505
505
 
506
+ For application search, choose lexical, semantic, or hybrid mode directly:
507
+
508
+ ```py
509
+ question = "how do i calculate cost per horse"
510
+
511
+ lexical = db.from_table("ActiveDocumentChunk").search(
512
+ question,
513
+ mode="lexical",
514
+ match="any",
515
+ ).list()
516
+
517
+ semantic = db.from_table("ActiveDocumentChunk").search(
518
+ question,
519
+ mode="semantic",
520
+ ).list()
521
+
522
+ hybrid = db.from_table("ActiveDocumentChunk").search(
523
+ question,
524
+ mode="hybrid",
525
+ ).list()
526
+ ```
527
+
528
+ Native candidate channels remain available as explicit, physically bounded
529
+ low-level APIs:
530
+
531
+ ```py
532
+ from onyx_database import hnsw_search_query
533
+
534
+ lexical = (
535
+ db.from_table("ActiveDocumentChunk")
536
+ .in_partition("revision-7")
537
+ .approximate_search("customer success", max_candidates=128)
538
+ .limit(20)
539
+ .list()
540
+ )
541
+
542
+ semantic = (
543
+ db.from_table("ChunkAttentionHash")
544
+ .in_partition("revision-7")
545
+ .hnsw_candidates(
546
+ hnsw_search_query(
547
+ calibration_id=73,
548
+ vector=prompt_embedding,
549
+ max_candidates=256,
550
+ ef_search=1024,
551
+ )
552
+ )
553
+ .limit(20)
554
+ .list()
555
+ )
556
+ ```
557
+
558
+ `CANDIDATES`, `SEARCH_CANDIDATES`, and `HNSW_CANDIDATES` must be the sole root
559
+ criterion. Candidate results are approximate; exactly rerank them when the
560
+ full-precision vectors are available.
561
+
506
562
  Aggregate / string helpers for `select()` expressions:
507
563
 
508
564
  ```py
@@ -554,9 +610,75 @@ roles_missing_permission = (
554
610
  )
555
611
  ```
556
612
 
557
- ### Native bounded full-text search
613
+ ### Lexical, semantic, and hybrid search
614
+
615
+ The high-level API sends natural-language text to the database when you pass an
616
+ options mapping or search-option keywords. `mode` defaults to `"hybrid"`, `match`
617
+ defaults to `"any"`, `min_score` defaults to `None`, and `max_candidates` defaults
618
+ to `1000`.
619
+
620
+ ```py
621
+ question = "how do i calculate cost per horse"
622
+
623
+ # Match any normalized query term.
624
+ lexical = db.from_table(tables.ActiveDocumentChunk).search(
625
+ question,
626
+ mode="lexical",
627
+ match="any",
628
+ min_score=0.4,
629
+ max_candidates=500,
630
+ ).list()
631
+
632
+ # Let the database embed the query and use semantic retrieval.
633
+ semantic = db.from_table(tables.ActiveDocumentChunk).search(
634
+ question,
635
+ mode="semantic",
636
+ ).list()
637
+
638
+ # Combine lexical and semantic retrieval.
639
+ hybrid = db.from_table(tables.ActiveDocumentChunk).search(
640
+ question,
641
+ mode="hybrid",
642
+ ).list()
643
+
644
+ # A mapping is equivalent; Python snake_case and wire camelCase are accepted.
645
+ same_hybrid = db.from_table(tables.ActiveDocumentChunk).search(
646
+ question,
647
+ {"mode": "hybrid", "match": "any", "max_candidates": 500},
648
+ ).list()
649
+ ```
558
650
 
559
- Use `.search(...)` on a builder or the `search` predicate helper to add a `MATCHES` condition against the `__full_text__` pseudo-field. `db.search(...)` sets `table = "ALL"` and seeds a query builder with that condition (extras like `partition`, `pageSize`, `nextPage` remain querystring params when provided).
651
+ The same options work through `db.search(...)` for database-wide search and on
652
+ the async client. When supplied, `mode` must be `"lexical"`, `"semantic"`, or
653
+ `"hybrid"`; `match` must be `"all"` or `"any"`; `min_score` must be finite and
654
+ between 0 and 1; and `max_candidates` must be between 1 and 5000. Hybrid mode
655
+ requires at least 2 candidates so both lexical and semantic channels receive a
656
+ candidate budget.
657
+ High-level `SEARCH` queries are read-only but may be combined with structured
658
+ filters using `where`, `and_`, or `or_`. A query may contain only one `SEARCH`,
659
+ it cannot contain another `__full_text__` predicate, and high-level/candidate
660
+ search plans cannot be used for live query streams.
661
+
662
+ Table-scoped high-level search spans the table's current partitions by default
663
+ under one global candidate budget; use `in_partition(...)` to constrain it.
664
+ The low-level candidate APIs below still require one concrete partition.
665
+ Database-wide `db.search(...)` searches eligible unpartitioned tables only and
666
+ does not inherit the client's configured default partition.
667
+
668
+ Semantic and hybrid modes require a server-side embedding provider. The provider
669
+ must embed saved searchable text and query text with the same model and vector
670
+ space; marking a field searchable does not select or configure that provider.
671
+ Rows saved before embedding was enabled must be explicitly resaved or backfilled
672
+ before HNSW can retrieve them.
673
+
674
+ ### Legacy vector-managed search
675
+
676
+ Use `.search(...)` on a builder or the `search` predicate helper to add a `MATCHES` condition against the `__full_text__` pseudo-field. `db.search(...)` sets `table = "ALL"` and seeds a query builder with that condition. Paging remains a query parameter; partition is deliberately omitted from `ALL` searches.
677
+
678
+ For compatibility, calls with no options and calls whose only second argument is
679
+ a numeric score retain this legacy wire contract. Pass an options mapping or a
680
+ high-level option such as `mode`, `match`, or `max_candidates` to opt into
681
+ `SEARCH`.
560
682
 
561
683
  ```py
562
684
  from onyx_database import onyx, search, eq
@@ -694,6 +816,99 @@ Example request bodies emitted by the SDK:
694
816
  }
695
817
  ```
696
818
 
819
+ For advanced caller-supplied semantic routing, build a validated
820
+ `VectorSearchQuery`. The helper accepts Python snake_case arguments and emits the
821
+ Cloud API's camelCase wire fields:
822
+
823
+ ```py
824
+ from onyx_database import semantic_vector_signature, vector_search_query
825
+
826
+ signature = semantic_vector_signature(
827
+ calibration_id=-7909761245221418085,
828
+ bucket_id=5,
829
+ cells=[1, 2],
830
+ cell_counts=[2, 3],
831
+ fingerprint=["0x0123456789abcdef"],
832
+ boundary_confidence=0.2,
833
+ )
834
+
835
+ hybrid = vector_search_query(
836
+ text="storm warning",
837
+ semantic=signature,
838
+ min_score=0.15,
839
+ nearby_bucket_radius=2,
840
+ max_candidates=500,
841
+ require_all_terms=False,
842
+ )
843
+
844
+ db.from_table(tables.Document).search(hybrid).limit(20).list()
845
+ ```
846
+
847
+ `VectorSearchQuery` requires text and/or a semantic signature. Its defaults are
848
+ `nearbyBucketRadius=1`, `maxCandidates=1000`, and `requireAllTerms=true`;
849
+ `maxCandidates` is bounded to `1..5000`. Semantic helpers validate the mixed-radix
850
+ bucket identifier, compute or verify exactly four fingerprint bands, keep the
851
+ non-zero signed 64-bit calibration identifier as decimal text, and emit fingerprint
852
+ words as fixed-width unsigned hexadecimal strings.
853
+
854
+ Three explicit candidate APIs provide physically bounded admission for downstream
855
+ reranking:
856
+
857
+ ```py
858
+ from onyx_database import hnsw_search_query
859
+
860
+ # Bounded lexical admission from a SEARCHABLE table.
861
+ lexical = (
862
+ db.from_table(tables.Document)
863
+ .in_partition("corpus-a")
864
+ .approximate_search(
865
+ "storm warning",
866
+ min_score=0.1,
867
+ max_candidates=250,
868
+ require_all_terms=False,
869
+ )
870
+ .limit(20)
871
+ .list()
872
+ )
873
+
874
+ # Native HNSW nearest-neighbor admission.
875
+ hnsw = hnsw_search_query(
876
+ calibration_id=-7909761245221418085,
877
+ vector=[0.25, -0.5, 0.75],
878
+ max_candidates=100,
879
+ ef_search=400,
880
+ min_score=0.2,
881
+ )
882
+ neighbors = (
883
+ db.from_table(tables.Document)
884
+ .in_partition("corpus-a")
885
+ .hnsw_candidates(hnsw)
886
+ .list()
887
+ )
888
+
889
+ # Bounded EQUAL/IN-style admission from an ordinary secondary index.
890
+ routed = (
891
+ db.from_table(tables.Document)
892
+ .in_partition("corpus-a")
893
+ .approximate_candidates("tenantId", ["tenant-a", "tenant-b"], 200)
894
+ .list()
895
+ )
896
+ ```
897
+
898
+ `SEARCH_CANDIDATES`, `HNSW_CANDIDATES`, and `CANDIDATES` are positive, read-only,
899
+ sole-root criteria. The builders reject attempts to combine them with `where`,
900
+ `and_`, `or_`, or `search` in either call order. Partitioned tables require one
901
+ concrete partition. HNSW vectors contain `1..16384` finite values with a non-zero
902
+ norm; `maxCandidates` is `1..5000`, `efSearch` is at least `maxCandidates` and at
903
+ most `20000`, and `minScore` is optional in `[-1, 1]`. The same methods and wire
904
+ validation are available on `AsyncQueryBuilder`.
905
+
906
+ Condition forms are exported as `approximate_search`, `hnsw_candidates`, and
907
+ `approximate_candidates`. Use them only as the sole condition, for example
908
+ `.where(approximate_search("storm", max_candidates=250))`. Database-wide
909
+ `db.search(...)` accepts the same high-level mode, match, score, and candidate
910
+ options as a table builder.
911
+
697
912
  ---
698
913
 
699
914
  ## Usage examples with `User`, `Role`, `Permission`
@@ -985,7 +1200,7 @@ except (OnyxConfigError, OnyxHTTPError) as err:
985
1200
 
986
1201
  A typical release flow for this repository:
987
1202
 
988
- 1. Update the version in `onyx_database/_version.py` (or use your preferred versioning tool).
1203
+ 1. Update the project version in `pyproject.toml`.
989
1204
  2. Build: `python -m build`
990
1205
  3. Publish: `twine upload dist/*`
991
1206
 
@@ -174,7 +174,7 @@ db = onyx.init(
174
174
 
175
175
  #### Default partition + logging
176
176
 
177
- - `partition` sets a default partition for queries, `find_by_id`, and deletes by primary key.
177
+ - `partition` sets a default partition for table-scoped queries, `find_by_id`, and deletes by primary key; database-wide `db.search(...)` omits it.
178
178
  - Save operations use the partition field on the entity itself (if present).
179
179
  - `request_logging_enabled` logs HTTP requests and JSON bodies.
180
180
  - `response_logging_enabled` logs HTTP responses and JSON bodies.
@@ -466,7 +466,7 @@ Importable helpers for conditions and sort:
466
466
  from onyx_database import (
467
467
  eq, neq, within, not_within,
468
468
  in_op, not_in,
469
- between,
469
+ between, not_between,
470
470
  gt, gte, lt, lte,
471
471
  like, not_like, contains, not_contains,
472
472
  starts_with, not_starts_with, matches, not_matches,
@@ -478,6 +478,62 @@ from onyx_database import (
478
478
  - Prefer `within` / `not_within` for inclusion checks (supports arrays, comma-separated strings, or inner queries).
479
479
  - `in_op` / `not_in` remain available for backward compatibility and are exact aliases.
480
480
 
481
+ For application search, choose lexical, semantic, or hybrid mode directly:
482
+
483
+ ```py
484
+ question = "how do i calculate cost per horse"
485
+
486
+ lexical = db.from_table("ActiveDocumentChunk").search(
487
+ question,
488
+ mode="lexical",
489
+ match="any",
490
+ ).list()
491
+
492
+ semantic = db.from_table("ActiveDocumentChunk").search(
493
+ question,
494
+ mode="semantic",
495
+ ).list()
496
+
497
+ hybrid = db.from_table("ActiveDocumentChunk").search(
498
+ question,
499
+ mode="hybrid",
500
+ ).list()
501
+ ```
502
+
503
+ Native candidate channels remain available as explicit, physically bounded
504
+ low-level APIs:
505
+
506
+ ```py
507
+ from onyx_database import hnsw_search_query
508
+
509
+ lexical = (
510
+ db.from_table("ActiveDocumentChunk")
511
+ .in_partition("revision-7")
512
+ .approximate_search("customer success", max_candidates=128)
513
+ .limit(20)
514
+ .list()
515
+ )
516
+
517
+ semantic = (
518
+ db.from_table("ChunkAttentionHash")
519
+ .in_partition("revision-7")
520
+ .hnsw_candidates(
521
+ hnsw_search_query(
522
+ calibration_id=73,
523
+ vector=prompt_embedding,
524
+ max_candidates=256,
525
+ ef_search=1024,
526
+ )
527
+ )
528
+ .limit(20)
529
+ .list()
530
+ )
531
+ ```
532
+
533
+ `CANDIDATES`, `SEARCH_CANDIDATES`, and `HNSW_CANDIDATES` must be the sole root
534
+ criterion. Candidate results are approximate; exactly rerank them when the
535
+ full-precision vectors are available.
536
+
481
537
  Aggregate / string helpers for `select()` expressions:
482
538
 
483
539
  ```py
@@ -529,9 +585,75 @@ roles_missing_permission = (
529
585
  )
530
586
  ```
531
587
 
532
- ### Native bounded full-text search
588
+ ### Lexical, semantic, and hybrid search
589
+
590
+ The high-level API sends natural-language text to the database when you pass an
591
+ options mapping or search-option keywords. `mode` defaults to `"hybrid"`, `match`
592
+ defaults to `"any"`, `min_score` defaults to `None`, and `max_candidates` defaults
593
+ to `1000`.
594
+
595
+ ```py
596
+ question = "how do i calculate cost per horse"
597
+
598
+ # Match any normalized query term.
599
+ lexical = db.from_table(tables.ActiveDocumentChunk).search(
600
+ question,
601
+ mode="lexical",
602
+ match="any",
603
+ min_score=0.4,
604
+ max_candidates=500,
605
+ ).list()
606
+
607
+ # Let the database embed the query and use semantic retrieval.
608
+ semantic = db.from_table(tables.ActiveDocumentChunk).search(
609
+ question,
610
+ mode="semantic",
611
+ ).list()
612
+
613
+ # Combine lexical and semantic retrieval.
614
+ hybrid = db.from_table(tables.ActiveDocumentChunk).search(
615
+ question,
616
+ mode="hybrid",
617
+ ).list()
618
+
619
+ # A mapping is equivalent; Python snake_case and wire camelCase are accepted.
620
+ same_hybrid = db.from_table(tables.ActiveDocumentChunk).search(
621
+ question,
622
+ {"mode": "hybrid", "match": "any", "max_candidates": 500},
623
+ ).list()
624
+ ```
533
625
 
534
- Use `.search(...)` on a builder or the `search` predicate helper to add a `MATCHES` condition against the `__full_text__` pseudo-field. `db.search(...)` sets `table = "ALL"` and seeds a query builder with that condition (extras like `partition`, `pageSize`, `nextPage` remain querystring params when provided).
626
+ The same options work through `db.search(...)` for database-wide search and on
627
+ the async client. When supplied, `mode` must be `"lexical"`, `"semantic"`, or
628
+ `"hybrid"`; `match` must be `"all"` or `"any"`; `min_score` must be finite and
629
+ between 0 and 1; and `max_candidates` must be between 1 and 5000. Hybrid mode
630
+ requires at least 2 candidates so both lexical and semantic channels receive a
631
+ candidate budget.
632
+ High-level `SEARCH` queries are read-only but may be combined with structured
633
+ filters using `where`, `and_`, or `or_`. A query may contain only one `SEARCH`,
634
+ it cannot contain another `__full_text__` predicate, and high-level/candidate
635
+ search plans cannot be used for live query streams.
636
+
637
+ Table-scoped high-level search spans the table's current partitions by default
638
+ under one global candidate budget; use `in_partition(...)` to constrain it.
639
+ The low-level candidate APIs below still require one concrete partition.
640
+ Database-wide `db.search(...)` searches eligible unpartitioned tables only and
641
+ does not inherit the client's configured default partition.
642
+
643
+ Semantic and hybrid modes require a server-side embedding provider. The provider
644
+ must embed saved searchable text and query text with the same model and vector
645
+ space; marking a field searchable does not select or configure that provider.
646
+ Rows saved before embedding was enabled must be explicitly resaved or backfilled
647
+ before HNSW can retrieve them.
648
+
649
+ ### Legacy vector-managed search
650
+
651
+ Use `.search(...)` on a builder or the `search` predicate helper to add a `MATCHES` condition against the `__full_text__` pseudo-field. `db.search(...)` sets `table = "ALL"` and seeds a query builder with that condition. Paging remains a query parameter; partition is deliberately omitted from `ALL` searches.
652
+
653
+ For compatibility, calls with no options and calls whose only second argument is
654
+ a numeric score retain this legacy wire contract. Pass an options mapping or a
655
+ high-level option such as `mode`, `match`, or `max_candidates` to opt into
656
+ `SEARCH`.
535
657
 
536
658
  ```py
537
659
  from onyx_database import onyx, search, eq
@@ -669,6 +791,99 @@ Example request bodies emitted by the SDK:
669
791
  }
670
792
  ```
671
793
 
794
+ For advanced caller-supplied semantic routing, build a validated
795
+ `VectorSearchQuery`. The helper accepts Python snake_case arguments and emits the
796
+ Cloud API's camelCase wire fields:
797
+
798
+ ```py
799
+ from onyx_database import semantic_vector_signature, vector_search_query
800
+
801
+ signature = semantic_vector_signature(
802
+ calibration_id=-7909761245221418085,
803
+ bucket_id=5,
804
+ cells=[1, 2],
805
+ cell_counts=[2, 3],
806
+ fingerprint=["0x0123456789abcdef"],
807
+ boundary_confidence=0.2,
808
+ )
809
+
810
+ hybrid = vector_search_query(
811
+ text="storm warning",
812
+ semantic=signature,
813
+ min_score=0.15,
814
+ nearby_bucket_radius=2,
815
+ max_candidates=500,
816
+ require_all_terms=False,
817
+ )
818
+
819
+ db.from_table(tables.Document).search(hybrid).limit(20).list()
820
+ ```
821
+
822
+ `VectorSearchQuery` requires text and/or a semantic signature. Its defaults are
823
+ `nearbyBucketRadius=1`, `maxCandidates=1000`, and `requireAllTerms=true`;
824
+ `maxCandidates` is bounded to `1..5000`. Semantic helpers validate the mixed-radix
825
+ bucket identifier, compute or verify exactly four fingerprint bands, keep the
826
+ non-zero signed 64-bit calibration identifier as decimal text, and emit fingerprint
827
+ words as fixed-width unsigned hexadecimal strings.
828
+
829
+ Three explicit candidate APIs provide physically bounded admission for downstream
830
+ reranking:
831
+
832
+ ```py
833
+ from onyx_database import hnsw_search_query
834
+
835
+ # Bounded lexical admission from a SEARCHABLE table.
836
+ lexical = (
837
+ db.from_table(tables.Document)
838
+ .in_partition("corpus-a")
839
+ .approximate_search(
840
+ "storm warning",
841
+ min_score=0.1,
842
+ max_candidates=250,
843
+ require_all_terms=False,
844
+ )
845
+ .limit(20)
846
+ .list()
847
+ )
848
+
849
+ # Native HNSW nearest-neighbor admission.
850
+ hnsw = hnsw_search_query(
851
+ calibration_id=-7909761245221418085,
852
+ vector=[0.25, -0.5, 0.75],
853
+ max_candidates=100,
854
+ ef_search=400,
855
+ min_score=0.2,
856
+ )
857
+ neighbors = (
858
+ db.from_table(tables.Document)
859
+ .in_partition("corpus-a")
860
+ .hnsw_candidates(hnsw)
861
+ .list()
862
+ )
863
+
864
+ # Bounded EQUAL/IN-style admission from an ordinary secondary index.
865
+ routed = (
866
+ db.from_table(tables.Document)
867
+ .in_partition("corpus-a")
868
+ .approximate_candidates("tenantId", ["tenant-a", "tenant-b"], 200)
869
+ .list()
870
+ )
871
+ ```
872
+
873
+ `SEARCH_CANDIDATES`, `HNSW_CANDIDATES`, and `CANDIDATES` are positive, read-only,
874
+ sole-root criteria. The builders reject attempts to combine them with `where`,
875
+ `and_`, `or_`, or `search` in either call order. Partitioned tables require one
876
+ concrete partition. HNSW vectors contain `1..16384` finite values with a non-zero
877
+ norm; `maxCandidates` is `1..5000`, `efSearch` is at least `maxCandidates` and at
878
+ most `20000`, and `minScore` is optional in `[-1, 1]`. The same methods and wire
879
+ validation are available on `AsyncQueryBuilder`.
880
+
881
+ Condition forms are exported as `approximate_search`, `hnsw_candidates`, and
882
+ `approximate_candidates`. Use them only as the sole condition, for example
883
+ `.where(approximate_search("storm", max_candidates=250))`. Database-wide
884
+ `db.search(...)` accepts the same high-level mode, match, score, and candidate
885
+ options as a table builder.
886
+
672
887
  ---
673
888
 
674
889
  ## Usage examples with `User`, `Role`, `Permission`
@@ -960,7 +1175,7 @@ except (OnyxConfigError, OnyxHTTPError) as err:
960
1175
 
961
1176
  A typical release flow for this repository:
962
1177
 
963
- 1. Update the version in `onyx_database/_version.py` (or use your preferred versioning tool).
1178
+ 1. Update the project version in `pyproject.toml`.
964
1179
  2. Build: `python -m build`
965
1180
  3. Publish: `twine upload dist/*`
966
1181
 
@@ -1,5 +1,5 @@
1
1
  {
2
- "databaseId": "bbabca0e-82ce-11f0-0000-a2ce78b61b6a",
2
+ "databaseId": "YOUR_DATABASE_ID",
3
3
  "entities": [
4
4
  {
5
5
  "name": "AuditLog",