django-tsearchable 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- django_tsearchable-0.2.0.dist-info/METADATA +147 -0
- django_tsearchable-0.2.0.dist-info/RECORD +22 -0
- django_tsearchable-0.2.0.dist-info/WHEEL +4 -0
- django_tsearchable-0.2.0.dist-info/licenses/LICENSE +16 -0
- tsearchable/__init__.py +18 -0
- tsearchable/admin.py +12 -0
- tsearchable/apps.py +12 -0
- tsearchable/checks.py +62 -0
- tsearchable/drf.py +23 -0
- tsearchable/filters.py +12 -0
- tsearchable/forms.py +21 -0
- tsearchable/management/__init__.py +0 -0
- tsearchable/management/commands/__init__.py +0 -0
- tsearchable/management/commands/rebuild_search_vectors.py +28 -0
- tsearchable/management/commands/tsearch_trigger_sql.py +23 -0
- tsearchable/models.py +156 -0
- tsearchable/operations.py +37 -0
- tsearchable/query.py +117 -0
- tsearchable/querysets.py +95 -0
- tsearchable/signals.py +42 -0
- tsearchable/testing.py +17 -0
- tsearchable/views.py +24 -0
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: django-tsearchable
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Drop-in PostgreSQL full-text search for Django: auto-maintained search_vector, GIN/GiST index, boolean query syntax, admin/DRF/forms integration.
|
|
5
|
+
Project-URL: Homepage, https://github.com/yourname/django-tsearchable
|
|
6
|
+
Project-URL: Issues, https://github.com/yourname/django-tsearchable/issues
|
|
7
|
+
Project-URL: Changelog, https://github.com/yourname/django-tsearchable/blob/main/CHANGELOG.md
|
|
8
|
+
Author-email: Your Name <you@example.com>
|
|
9
|
+
License: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: django,drf,full-text-search,postgresql,search,tsvector
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Framework :: Django
|
|
14
|
+
Classifier: Framework :: Django :: 4.2
|
|
15
|
+
Classifier: Framework :: Django :: 5.0
|
|
16
|
+
Classifier: Framework :: Django :: 5.1
|
|
17
|
+
Classifier: Intended Audience :: Developers
|
|
18
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
19
|
+
Classifier: Programming Language :: Python :: 3
|
|
20
|
+
Classifier: Topic :: Database
|
|
21
|
+
Requires-Python: >=3.9
|
|
22
|
+
Requires-Dist: django>=4.2
|
|
23
|
+
Provides-Extra: dev
|
|
24
|
+
Requires-Dist: build; extra == 'dev'
|
|
25
|
+
Requires-Dist: django-filter; extra == 'dev'
|
|
26
|
+
Requires-Dist: djangorestframework; extra == 'dev'
|
|
27
|
+
Requires-Dist: psycopg[binary]; extra == 'dev'
|
|
28
|
+
Requires-Dist: pytest; extra == 'dev'
|
|
29
|
+
Requires-Dist: pytest-django; extra == 'dev'
|
|
30
|
+
Requires-Dist: twine; extra == 'dev'
|
|
31
|
+
Provides-Extra: drf
|
|
32
|
+
Requires-Dist: djangorestframework>=3.14; extra == 'drf'
|
|
33
|
+
Provides-Extra: filter
|
|
34
|
+
Requires-Dist: django-filter>=23.5; extra == 'filter'
|
|
35
|
+
Description-Content-Type: text/markdown
|
|
36
|
+
|
|
37
|
+
# django-tsearchable
|
|
38
|
+
|
|
39
|
+
**PostgreSQL full-text search for Django in one abstract model.** Subclass `SearchableModel`, list the fields,
|
|
40
|
+
and get an auto-maintained `search_vector`, a GIN/GiST index, and a search syntax your users already understand:
|
|
41
|
+
`python and (django or "rest framework") not php`.
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
from django.db import models
|
|
45
|
+
from tsearchable import SearchableModel
|
|
46
|
+
|
|
47
|
+
class Candidate(SearchableModel):
|
|
48
|
+
first_name = models.CharField(max_length=255)
|
|
49
|
+
last_name = models.CharField(max_length=255)
|
|
50
|
+
email = models.EmailField()
|
|
51
|
+
skills = ArrayField(models.CharField(max_length=50), default=list)
|
|
52
|
+
tags = models.ManyToManyField("Tag")
|
|
53
|
+
|
|
54
|
+
search_fields = {"first_name": "A", "last_name": "A", "email": "B", "skills": "C", "tags__title": "C"}
|
|
55
|
+
search_index = "gin" # "gin" | "gist" | None
|
|
56
|
+
|
|
57
|
+
Candidate.objects.search('python and not java', rank=True)
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
## Features
|
|
61
|
+
|
|
62
|
+
| Area | What you get |
|
|
63
|
+
|---|---|
|
|
64
|
+
| **Model** | Abstract `SearchableModel` adds `search_vector`, builds the index (`gin`/`gist`) in your migrations, weights `A`–`D`, ORM-style paths (`company__name`, `tags__title`), list/array fields |
|
|
65
|
+
| **Query syntax** | implicit AND, `and` / `or` / `not` / `!`, parentheses, `"exact phrase"`, prefix matching, accent- and Turkish-aware (`cagri` finds `Çağrı`). Hostile input can never raise a tsquery syntax error. Or switch to Postgres `websearch` syntax |
|
|
66
|
+
| **Always up to date** | refreshed on `save()` and automatically on M2M changes; or choose `"trigger"` (DB trigger, covers `bulk_create`/`update()`), `"deferred"` (signal → your Celery/RQ task), or `False` |
|
|
67
|
+
| **Search quality** | `rank=True` relevance ordering, `highlight=[...]` (`<mark>` snippets), `fuzzy=True` typo tolerance (pg_trgm), `.suggest()` autocomplete, per-field languages, `search_all()` across models |
|
|
68
|
+
| **Integrations** | Django admin mixin, DRF filter backend, django-filter `SearchFilter`, `SearchForm` + `SearchableListViewMixin` |
|
|
69
|
+
| **Tooling** | `manage.py rebuild_search_vectors`, `manage.py tsearch_trigger_sql`, system checks, `tsearchable.testing` assertions |
|
|
70
|
+
|
|
71
|
+
## Install
|
|
72
|
+
|
|
73
|
+
```bash
|
|
74
|
+
pip install django-tsearchable # + [drf] and/or [filter] extras
|
|
75
|
+
```
|
|
76
|
+
```python
|
|
77
|
+
INSTALLED_APPS = ["django.contrib.postgres", "tsearchable", ...]
|
|
78
|
+
```
|
|
79
|
+
Requires PostgreSQL, Django ≥ 4.2. No `unaccent` extension needed. Then `makemigrations && migrate`
|
|
80
|
+
and backfill existing rows once: `python manage.py rebuild_search_vectors`.
|
|
81
|
+
|
|
82
|
+
## Usage
|
|
83
|
+
|
|
84
|
+
### Configuration (class attributes)
|
|
85
|
+
|
|
86
|
+
| Attribute | Default | Meaning |
|
|
87
|
+
|---|---|---|
|
|
88
|
+
| `search_fields` | `[]` | list of paths, or `{path: "A"}` / `{path: ("A", "turkish")}` (weight, language) |
|
|
89
|
+
| `search_config` | `"simple"` | Postgres text search config |
|
|
90
|
+
| `search_mode` | `"tsquery"` | `"tsquery"` (rich syntax) or `"websearch"` |
|
|
91
|
+
| `search_index` | `"gin"` | `"gin"`, `"gist"` or `None`; `search_index_options` is passed to the index |
|
|
92
|
+
| `search_auto_update` | `True` | `True`, `False`, `"trigger"`, `"deferred"` |
|
|
93
|
+
| `search_fallback_fields` | `[]` | extra `icontains` OR-ed into `.search()` (`search_fallback_unaccent` for `unaccent`) |
|
|
94
|
+
| `search_fuzzy_fields` | `[]` | fields for trigram typo tolerance |
|
|
95
|
+
| `search_normalize` | `True` | lowercase + accent stripping for the `simple` config |
|
|
96
|
+
|
|
97
|
+
### Searching
|
|
98
|
+
```python
|
|
99
|
+
Candidate.objects.search("pyhton", fuzzy=True) # typo tolerant (CREATE EXTENSION pg_trgm)
|
|
100
|
+
Candidate.objects.search("backend", rank=True, highlight=["bio"]) # obj.bio_highlight -> "…<mark>backend</mark>…"
|
|
101
|
+
Candidate.objects.suggest("pyt", field="first_name") # autocomplete
|
|
102
|
+
search_all([Candidate, JobPost], "python") # one relevance-sorted list
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
### Admin / DRF / forms
|
|
106
|
+
```python
|
|
107
|
+
class CandidateAdmin(SearchableAdminMixin, admin.ModelAdmin): ... # tsearchable.admin
|
|
108
|
+
|
|
109
|
+
class CandidateViewSet(ModelViewSet): # tsearchable.drf (?search=...)
|
|
110
|
+
filter_backends = [SearchableFilterBackend]
|
|
111
|
+
|
|
112
|
+
class CandidateList(SearchableListViewMixin, ListView): model = Candidate # tsearchable.views (?q=...)
|
|
113
|
+
# template context: search_form, search_query
|
|
114
|
+
|
|
115
|
+
class CandidateFilter(FilterSet): q = SearchFilter() # tsearchable.filters
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
### Keeping vectors fresh
|
|
119
|
+
* Default: updated on `save()` and on M2M add/remove/clear for paths like `tags__title`.
|
|
120
|
+
* `"trigger"`: local columns only; run `manage.py tsearch_trigger_sql app.Model` and paste the printed
|
|
121
|
+
`SearchTrigger(...)` into a migration. Covers `bulk_create`, `update()` and raw SQL.
|
|
122
|
+
* `"deferred"`: connect `tsearchable.signals.search_vector_refresh_requested` to enqueue
|
|
123
|
+
`instance.refresh_search_vector()` in your task queue.
|
|
124
|
+
* After `bulk_*` with the default mode: `Model.objects.rebuild_search_vectors()`.
|
|
125
|
+
|
|
126
|
+
### Query syntax
|
|
127
|
+
`python django` (AND) · `python or java` · `python not java` · `!java` · `(a or b) and c` · `"senior backend"` (phrase) · `pyth*` / `pyth` (prefix)
|
|
128
|
+
|
|
129
|
+
### System checks
|
|
130
|
+
`manage.py check` validates `search_fields` paths, index type, update mode, trigger limitations, and that the DB is PostgreSQL.
|
|
131
|
+
|
|
132
|
+
### Testing your own code
|
|
133
|
+
```python
|
|
134
|
+
from tsearchable.testing import assert_search_finds, assert_search_misses
|
|
135
|
+
assert_search_finds(Candidate, "python not java", alice)
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
## Limitations
|
|
139
|
+
* Highlighting uses the original text, so accent-insensitive matches (`cagri` → `Çağrı`) match but may not be highlighted.
|
|
140
|
+
* Reverse-FK paths (e.g. `experiences__title`) are indexed on `save()` but not refreshed when the related row changes.
|
|
141
|
+
* `"trigger"` mode only normalises Turkish letters (optionally `unaccent=True` for others) and supports local fields.
|
|
142
|
+
|
|
143
|
+
## Development
|
|
144
|
+
```bash
|
|
145
|
+
pip install -e ".[dev]" && pytest # needs a local PostgreSQL (PGUSER/PGPASSWORD/PGHOST env vars)
|
|
146
|
+
```
|
|
147
|
+
MIT licensed.
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
tsearchable/__init__.py,sha256=ZHJsYEKvgTxRM9cxWVkGb0creCmvYStCIRxNoOrPOW8,636
|
|
2
|
+
tsearchable/admin.py,sha256=wdemPdSWFUoJiwClMuLeh3xvTdGNAOeRLLXrIvQ7XZ4,527
|
|
3
|
+
tsearchable/apps.py,sha256=VLDtw_sqTSgrZVNQI2GL3oKWjLgrCBhcIdHB3A4jdnI,304
|
|
4
|
+
tsearchable/checks.py,sha256=tGgytQ49bxNAB07Ry7R-5HJWUiRF_yKknvf09zVGZTk,2805
|
|
5
|
+
tsearchable/drf.py,sha256=R2GUT4BTR6LEcF4fF4uQLObx4Yp0AHzRos_aUpIRKsA,955
|
|
6
|
+
tsearchable/filters.py,sha256=kOafeT_6P_uEyjTlCgQegTDli-PiEmXIGfFf7Qqf-uQ,369
|
|
7
|
+
tsearchable/forms.py,sha256=2O14UhzYNqUzDuXq72E0rZA5DdWAD2CQAvE0Axy1kxY,707
|
|
8
|
+
tsearchable/models.py,sha256=wvtxP_Z7PDjfdbT97cqMXa_QBoSOQ5EYmpK7JAIjq38,6452
|
|
9
|
+
tsearchable/operations.py,sha256=ktnqPD3vkD_MRJtNLUil8CjUwmhiY7gDeUbruH1eZ9I,1524
|
|
10
|
+
tsearchable/query.py,sha256=upk2C6wQ9fDLjCPiYBk1YCpRa7eEYq4JwJ7tkclRH_0,4320
|
|
11
|
+
tsearchable/querysets.py,sha256=0X9CDZpYmnfUB_CoV87jc41v_NKoMVwwHIYo_NRb0nY,3936
|
|
12
|
+
tsearchable/signals.py,sha256=Tb-IBwLiY8fqS4p07lOu77p-F1NTR2fPB-M-QDqPHzc,1700
|
|
13
|
+
tsearchable/testing.py,sha256=xhbiM4xlNu8j2aROtRiS3gSM7cIsygSSOy42ByzTTIU,765
|
|
14
|
+
tsearchable/views.py,sha256=dRNOXS7OOesQv8rh5Cd9sNRIu8FGAwq7AggZ71s4jZE,848
|
|
15
|
+
tsearchable/management/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
16
|
+
tsearchable/management/commands/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
17
|
+
tsearchable/management/commands/rebuild_search_vectors.py,sha256=yXeHRmivR8_pkqdQXG4QdTvGPOpKMOMzkhCn8-RkSKA,1301
|
|
18
|
+
tsearchable/management/commands/tsearch_trigger_sql.py,sha256=J9iS_VZZGZZwjO5nVU31xQeXMRY3bGyEGgxvpinJTwA,902
|
|
19
|
+
django_tsearchable-0.2.0.dist-info/METADATA,sha256=pLN-hmHXH2GQwWOhmMce0aMDp0l8AB8TzWkWWDKg2JM,7315
|
|
20
|
+
django_tsearchable-0.2.0.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
21
|
+
django_tsearchable-0.2.0.dist-info/licenses/LICENSE,sha256=4iQbqGktDW0Z96miDtfz1OLDwapPbcTaKsJ7W6CCLq4,1066
|
|
22
|
+
django_tsearchable-0.2.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Your Name
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy of this software and associated
|
|
6
|
+
documentation files (the "Software"), to deal in the Software without restriction, including without limitation the
|
|
7
|
+
rights to use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the Software, and to permit
|
|
8
|
+
persons to whom the Software is furnished to do so, subject to the following conditions:
|
|
9
|
+
|
|
10
|
+
The above copyright notice and this permission notice shall be included in all copies or substantial portions of the
|
|
11
|
+
Software.
|
|
12
|
+
|
|
13
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE
|
|
14
|
+
WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
|
15
|
+
COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR
|
|
16
|
+
OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
tsearchable/__init__.py
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
"""django-tsearchable. Django-dependent names load lazily, so `tsearchable.query` works without Django set up."""
|
|
2
|
+
from .query import TSQueryConverter, normalize_text
|
|
3
|
+
|
|
4
|
+
__version__ = "0.2.0"
|
|
5
|
+
__all__ = [
|
|
6
|
+
"SearchableModel", "SearchableQuerySet", "SearchableManager", "search_all",
|
|
7
|
+
"TSQueryConverter", "normalize_text",
|
|
8
|
+
]
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def __getattr__(name):
|
|
12
|
+
if name == "SearchableModel":
|
|
13
|
+
from .models import SearchableModel
|
|
14
|
+
return SearchableModel
|
|
15
|
+
if name in ("SearchableQuerySet", "SearchableManager", "search_all"):
|
|
16
|
+
from . import querysets
|
|
17
|
+
return getattr(querysets, name)
|
|
18
|
+
raise AttributeError(name)
|
tsearchable/admin.py
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
class SearchableAdminMixin:
|
|
2
|
+
"""Admin search box understands `python and not java`, "exact phrase", prefix matching.
|
|
3
|
+
|
|
4
|
+
class CandidateAdmin(SearchableAdminMixin, admin.ModelAdmin): ...
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
search_rank = False
|
|
8
|
+
|
|
9
|
+
def get_search_results(self, request, queryset, search_term):
|
|
10
|
+
if search_term and search_term.strip() and hasattr(queryset, "search"):
|
|
11
|
+
return queryset.search(search_term, rank=self.search_rank), False
|
|
12
|
+
return super().get_search_results(request, queryset, search_term)
|
tsearchable/apps.py
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
from django.apps import AppConfig
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class TSearchableConfig(AppConfig):
|
|
5
|
+
name = "tsearchable"
|
|
6
|
+
verbose_name = "TSearchable"
|
|
7
|
+
|
|
8
|
+
def ready(self):
|
|
9
|
+
from . import checks # noqa: F401 (registers system checks)
|
|
10
|
+
from .signals import connect_m2m_handlers
|
|
11
|
+
|
|
12
|
+
connect_m2m_handlers()
|
tsearchable/checks.py
ADDED
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
from django.apps import apps
|
|
2
|
+
from django.core.checks import Error, Tags, Warning, register
|
|
3
|
+
from django.core.exceptions import FieldDoesNotExist
|
|
4
|
+
from django.db import connections
|
|
5
|
+
|
|
6
|
+
from .models import AUTO_UPDATE_MODES, INDEX_TYPES, SearchableModel
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def _path_ok(model, path):
|
|
10
|
+
parts = path.split("__")
|
|
11
|
+
current = model
|
|
12
|
+
for i, part in enumerate(parts):
|
|
13
|
+
last = i == len(parts) - 1
|
|
14
|
+
try:
|
|
15
|
+
field = current._meta.get_field(part)
|
|
16
|
+
except FieldDoesNotExist:
|
|
17
|
+
return last and hasattr(current, part) # property / method on the model
|
|
18
|
+
if not last:
|
|
19
|
+
if not field.is_relation or field.related_model is None:
|
|
20
|
+
return False
|
|
21
|
+
current = field.related_model
|
|
22
|
+
return True
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def check_model(model):
|
|
26
|
+
errors = []
|
|
27
|
+
label = model._meta.label
|
|
28
|
+
specs = model._search_field_specs()
|
|
29
|
+
if not specs:
|
|
30
|
+
errors.append(Warning(f"{label} has no `search_fields`; search_vector will always be empty.", obj=model, id="tsearchable.W001"))
|
|
31
|
+
for path in specs:
|
|
32
|
+
if not _path_ok(model, path):
|
|
33
|
+
errors.append(Error(f"`search_fields` entry {path!r} does not resolve on {label}.", obj=model, id="tsearchable.E001"))
|
|
34
|
+
if model.search_index and str(model.search_index).lower() not in INDEX_TYPES:
|
|
35
|
+
errors.append(Error(f"`search_index` must be one of {sorted(INDEX_TYPES)} or None, got {model.search_index!r}.", obj=model, id="tsearchable.E002"))
|
|
36
|
+
if model.search_auto_update not in AUTO_UPDATE_MODES:
|
|
37
|
+
errors.append(Error(f"`search_auto_update` must be one of {AUTO_UPDATE_MODES}, got {model.search_auto_update!r}.", obj=model, id="tsearchable.E003"))
|
|
38
|
+
if model.search_auto_update == "trigger" and any("__" in p for p in specs):
|
|
39
|
+
errors.append(Error("`search_auto_update = 'trigger'` only supports local fields (no `__` paths).", obj=model, id="tsearchable.E004"))
|
|
40
|
+
if model.search_mode not in ("tsquery", "websearch"):
|
|
41
|
+
errors.append(Error(f"`search_mode` must be 'tsquery' or 'websearch', got {model.search_mode!r}.", obj=model, id="tsearchable.E005"))
|
|
42
|
+
return errors
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@register(Tags.models)
|
|
46
|
+
def check_searchable_models(app_configs, **kwargs):
|
|
47
|
+
errors = []
|
|
48
|
+
for model in apps.get_models():
|
|
49
|
+
if issubclass(model, SearchableModel):
|
|
50
|
+
errors += check_model(model)
|
|
51
|
+
return errors
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
@register(Tags.database, deploy=False)
|
|
55
|
+
def check_database_vendor(app_configs, databases=None, **kwargs):
|
|
56
|
+
if not any(issubclass(m, SearchableModel) for m in apps.get_models()):
|
|
57
|
+
return []
|
|
58
|
+
errors = []
|
|
59
|
+
for alias in databases or []:
|
|
60
|
+
if connections[alias].vendor != "postgresql":
|
|
61
|
+
errors.append(Error(f"Database {alias!r} is not PostgreSQL; django-tsearchable requires PostgreSQL.", id="tsearchable.E006"))
|
|
62
|
+
return errors
|
tsearchable/drf.py
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
from rest_framework.filters import BaseFilterBackend
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class SearchableFilterBackend(BaseFilterBackend):
|
|
5
|
+
"""DRF filter backend: `?search=python and not java`. Override the param with `search_param` on the view."""
|
|
6
|
+
|
|
7
|
+
search_param = "search"
|
|
8
|
+
|
|
9
|
+
def filter_queryset(self, request, queryset, view):
|
|
10
|
+
param = getattr(view, "search_param", None) or self.search_param
|
|
11
|
+
term = request.query_params.get(param, "").strip()
|
|
12
|
+
if not term or not hasattr(queryset, "search"):
|
|
13
|
+
return queryset
|
|
14
|
+
return queryset.search(term, rank=getattr(view, "search_rank", False))
|
|
15
|
+
|
|
16
|
+
def get_schema_operation_parameters(self, view):
|
|
17
|
+
return [{
|
|
18
|
+
"name": getattr(view, "search_param", None) or self.search_param,
|
|
19
|
+
"required": False,
|
|
20
|
+
"in": "query",
|
|
21
|
+
"description": "Full-text search (supports and / or / not, quotes, parentheses).",
|
|
22
|
+
"schema": {"type": "string"},
|
|
23
|
+
}]
|
tsearchable/filters.py
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
import django_filters
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class SearchFilter(django_filters.CharFilter):
|
|
5
|
+
"""django-filter integration: q = SearchFilter() inside a FilterSet."""
|
|
6
|
+
|
|
7
|
+
def __init__(self, *args, rank=False, **kwargs):
|
|
8
|
+
self.rank = rank
|
|
9
|
+
super().__init__(*args, **kwargs)
|
|
10
|
+
|
|
11
|
+
def filter(self, qs, value):
|
|
12
|
+
return qs.search(value, rank=self.rank) if value else qs
|
tsearchable/forms.py
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
from django import forms
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class SearchForm(forms.Form):
|
|
5
|
+
"""GET search form with a single (optionally renamed) query field."""
|
|
6
|
+
|
|
7
|
+
q = forms.CharField(required=False, strip=True, max_length=200, label="Search")
|
|
8
|
+
|
|
9
|
+
def __init__(self, *args, field_name="q", **kwargs):
|
|
10
|
+
super().__init__(*args, **kwargs)
|
|
11
|
+
self.field_name = field_name
|
|
12
|
+
if field_name != "q":
|
|
13
|
+
self.fields = {field_name: self.fields["q"]}
|
|
14
|
+
|
|
15
|
+
@property
|
|
16
|
+
def query(self):
|
|
17
|
+
return (self.cleaned_data.get(self.field_name) or "") if self.is_valid() else ""
|
|
18
|
+
|
|
19
|
+
def apply(self, queryset, **search_kwargs):
|
|
20
|
+
q = self.query
|
|
21
|
+
return queryset.search(q, **search_kwargs) if q else queryset
|
|
File without changes
|
|
File without changes
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
from django.apps import apps
|
|
2
|
+
from django.core.management.base import BaseCommand, CommandError
|
|
3
|
+
|
|
4
|
+
from tsearchable.models import SearchableModel
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class Command(BaseCommand):
|
|
8
|
+
help = "Recompute search_vector for SearchableModel tables (all of them, or the given app_label.Model)."
|
|
9
|
+
|
|
10
|
+
def add_arguments(self, parser):
|
|
11
|
+
parser.add_argument("models", nargs="*", help="app_label.ModelName (default: every SearchableModel)")
|
|
12
|
+
parser.add_argument("--batch-size", type=int, default=500)
|
|
13
|
+
|
|
14
|
+
def handle(self, *args, models, batch_size, **options):
|
|
15
|
+
if models:
|
|
16
|
+
targets = []
|
|
17
|
+
for label in models:
|
|
18
|
+
try:
|
|
19
|
+
targets.append(apps.get_model(label))
|
|
20
|
+
except (LookupError, ValueError):
|
|
21
|
+
raise CommandError(f"Unknown model {label!r}; use app_label.ModelName")
|
|
22
|
+
else:
|
|
23
|
+
targets = [m for m in apps.get_models() if issubclass(m, SearchableModel)]
|
|
24
|
+
for model in targets:
|
|
25
|
+
if not issubclass(model, SearchableModel):
|
|
26
|
+
raise CommandError(f"{model._meta.label} is not a SearchableModel")
|
|
27
|
+
n = model._default_manager.all().rebuild_search_vectors(chunk_size=batch_size)
|
|
28
|
+
self.stdout.write(self.style.SUCCESS(f"{model._meta.label}: {n} rows"))
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
from django.apps import apps
|
|
2
|
+
from django.core.management.base import BaseCommand, CommandError
|
|
3
|
+
|
|
4
|
+
|
|
5
|
+
class Command(BaseCommand):
|
|
6
|
+
help = "Print a migration snippet that installs the DB trigger for a model using search_auto_update='trigger'."
|
|
7
|
+
|
|
8
|
+
def add_arguments(self, parser):
|
|
9
|
+
parser.add_argument("model", help="app_label.ModelName")
|
|
10
|
+
|
|
11
|
+
def handle(self, *args, model, **options):
|
|
12
|
+
try:
|
|
13
|
+
m = apps.get_model(model)
|
|
14
|
+
except (LookupError, ValueError):
|
|
15
|
+
raise CommandError(f"Unknown model {model!r}")
|
|
16
|
+
specs = m._search_field_specs()
|
|
17
|
+
fields = {p: w for p, (w, _c) in specs.items()}
|
|
18
|
+
self.stdout.write(
|
|
19
|
+
"from tsearchable.operations import SearchTrigger\n\n"
|
|
20
|
+
"operations = [\n"
|
|
21
|
+
f" SearchTrigger(table={m._meta.db_table!r}, fields={fields!r}, config={m.search_config!r}),\n"
|
|
22
|
+
"]"
|
|
23
|
+
)
|
tsearchable/models.py
ADDED
|
@@ -0,0 +1,156 @@
|
|
|
1
|
+
import operator
|
|
2
|
+
from functools import reduce
|
|
3
|
+
|
|
4
|
+
from django.contrib.postgres.indexes import GinIndex, GistIndex
|
|
5
|
+
from django.contrib.postgres.search import SearchVector, SearchVectorField
|
|
6
|
+
from django.core.exceptions import FieldDoesNotExist, ObjectDoesNotExist
|
|
7
|
+
from django.db import models, transaction
|
|
8
|
+
from django.db.models import Value
|
|
9
|
+
from django.db.models.base import ModelBase
|
|
10
|
+
|
|
11
|
+
from .query import normalize_text
|
|
12
|
+
from .querysets import SearchableManager
|
|
13
|
+
from .signals import search_vector_refresh_requested
|
|
14
|
+
|
|
15
|
+
INDEX_TYPES = {"gin": GinIndex, "gist": GistIndex}
|
|
16
|
+
AUTO_UPDATE_MODES = (True, False, "trigger", "deferred")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _get_attr(obj, name):
|
|
20
|
+
try:
|
|
21
|
+
if hasattr(obj, name):
|
|
22
|
+
return getattr(obj, name)
|
|
23
|
+
meta = getattr(obj, "_meta", None)
|
|
24
|
+
if meta is not None: # reverse relation addressed by its query name (ORM style)
|
|
25
|
+
field = meta.get_field(name)
|
|
26
|
+
if field.auto_created and not field.concrete:
|
|
27
|
+
return getattr(obj, field.get_accessor_name(), None)
|
|
28
|
+
except (FieldDoesNotExist, ObjectDoesNotExist, AttributeError):
|
|
29
|
+
pass
|
|
30
|
+
return None
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _values(obj, parts):
|
|
34
|
+
"""Collect text for an ORM-style path (`a__b__c`): FKs, reverse FK / M2M managers, lists, callables."""
|
|
35
|
+
if obj is None:
|
|
36
|
+
return []
|
|
37
|
+
if not parts:
|
|
38
|
+
if isinstance(obj, (list, tuple, set)):
|
|
39
|
+
return [str(x) for x in obj if x not in (None, "")]
|
|
40
|
+
return [str(obj)] if obj != "" else []
|
|
41
|
+
head, *rest = parts
|
|
42
|
+
value = _get_attr(obj, head)
|
|
43
|
+
if callable(value) and not hasattr(value, "all"):
|
|
44
|
+
value = value()
|
|
45
|
+
if hasattr(value, "all"):
|
|
46
|
+
out = []
|
|
47
|
+
for item in value.all():
|
|
48
|
+
out += _values(item, rest)
|
|
49
|
+
return out
|
|
50
|
+
return _values(value, rest)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class SearchableModelBase(ModelBase):
|
|
54
|
+
def __new__(mcs, name, bases, attrs, **kwargs):
|
|
55
|
+
cls = super().__new__(mcs, name, bases, attrs, **kwargs)
|
|
56
|
+
if cls._meta.abstract or not cls.search_index:
|
|
57
|
+
return cls
|
|
58
|
+
index_cls = INDEX_TYPES.get(str(cls.search_index).lower())
|
|
59
|
+
if index_cls is None: # reported by the system check tsearchable.E002
|
|
60
|
+
return cls
|
|
61
|
+
if any("search_vector" in idx.fields for idx in cls._meta.indexes):
|
|
62
|
+
return cls
|
|
63
|
+
index = index_cls(fields=["search_vector"], **cls.search_index_options)
|
|
64
|
+
index.set_name_with_model(cls)
|
|
65
|
+
cls._meta.indexes = [*cls._meta.indexes, index]
|
|
66
|
+
# ModelState.from_model only reads options listed in original_attrs -> needed for makemigrations
|
|
67
|
+
cls._meta.original_attrs["indexes"] = cls._meta.indexes
|
|
68
|
+
return cls
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
class SearchableModel(models.Model, metaclass=SearchableModelBase):
|
|
72
|
+
"""
|
|
73
|
+
class Candidate(SearchableModel):
|
|
74
|
+
search_fields = {"first_name": "A", "tags__title": "C", "title": ("A", "turkish")}
|
|
75
|
+
search_index = "gin" # "gin" | "gist" | None
|
|
76
|
+
"""
|
|
77
|
+
|
|
78
|
+
# list of ORM-style paths, or {path: weight} / {path: (weight, config)}; weight is "A".."D"
|
|
79
|
+
search_fields = []
|
|
80
|
+
search_config = "simple"
|
|
81
|
+
search_mode = "tsquery" # "tsquery" (and/or/not/quotes/prefix) | "websearch" (Postgres websearch syntax)
|
|
82
|
+
search_index = "gin" # "gin" | "gist" | None
|
|
83
|
+
search_index_options = {}
|
|
84
|
+
search_auto_update = True # True | False | "trigger" | "deferred"
|
|
85
|
+
search_fallback_fields = [] # extra icontains OR-ed in by .search()
|
|
86
|
+
search_fallback_unaccent = False # True needs the `unaccent` Postgres extension
|
|
87
|
+
search_fuzzy_fields = [] # trigram-similarity fallback for .search(fuzzy=True); needs `pg_trgm`
|
|
88
|
+
search_normalize = True # lowercase + accent strip (Turkish aware) for fields using the "simple" config
|
|
89
|
+
|
|
90
|
+
search_vector = SearchVectorField(null=True, blank=True, editable=False)
|
|
91
|
+
|
|
92
|
+
objects = SearchableManager()
|
|
93
|
+
|
|
94
|
+
class Meta:
|
|
95
|
+
abstract = True
|
|
96
|
+
|
|
97
|
+
# ---- configuration helpers -----------------------------------------
|
|
98
|
+
@classmethod
|
|
99
|
+
def _search_field_specs(cls):
|
|
100
|
+
"""{path: (weight, config)}"""
|
|
101
|
+
items = cls.search_fields.items() if isinstance(cls.search_fields, dict) else ((f, None) for f in cls.search_fields)
|
|
102
|
+
out = {}
|
|
103
|
+
for path, spec in items:
|
|
104
|
+
weight, config = None, cls.search_config
|
|
105
|
+
if isinstance(spec, (tuple, list)):
|
|
106
|
+
weight = spec[0] if spec else None
|
|
107
|
+
if len(spec) > 1 and spec[1]:
|
|
108
|
+
config = spec[1]
|
|
109
|
+
elif spec:
|
|
110
|
+
weight = spec
|
|
111
|
+
out[path] = (weight, config)
|
|
112
|
+
return out
|
|
113
|
+
|
|
114
|
+
@classmethod
|
|
115
|
+
def search_configs(cls):
|
|
116
|
+
configs = [cls.search_config]
|
|
117
|
+
for _, config in cls._search_field_specs().values():
|
|
118
|
+
if config not in configs:
|
|
119
|
+
configs.append(config)
|
|
120
|
+
return configs
|
|
121
|
+
|
|
122
|
+
# ---- vector maintenance --------------------------------------------
|
|
123
|
+
def build_search_vector_expression(self):
|
|
124
|
+
parts = []
|
|
125
|
+
for path, (weight, config) in self._search_field_specs().items():
|
|
126
|
+
text = " ".join(_values(self, path.split("__")))
|
|
127
|
+
if self.search_normalize and config == "simple":
|
|
128
|
+
text = normalize_text(text)
|
|
129
|
+
if text.strip():
|
|
130
|
+
parts.append(SearchVector(Value(text), weight=weight, config=config))
|
|
131
|
+
if not parts:
|
|
132
|
+
return Value(None, output_field=SearchVectorField())
|
|
133
|
+
return reduce(operator.add, parts)
|
|
134
|
+
|
|
135
|
+
def refresh_search_vector(self):
|
|
136
|
+
if self.pk is None:
|
|
137
|
+
return
|
|
138
|
+
type(self)._base_manager.filter(pk=self.pk).update(search_vector=self.build_search_vector_expression())
|
|
139
|
+
|
|
140
|
+
def schedule_search_refresh(self):
|
|
141
|
+
mode = self.search_auto_update
|
|
142
|
+
if mode is True:
|
|
143
|
+
self.refresh_search_vector()
|
|
144
|
+
elif mode == "deferred":
|
|
145
|
+
cls = type(self)
|
|
146
|
+
transaction.on_commit(lambda: search_vector_refresh_requested.send(sender=cls, instance=self))
|
|
147
|
+
|
|
148
|
+
def save(self, *args, **kwargs):
|
|
149
|
+
with transaction.atomic():
|
|
150
|
+
super().save(*args, **kwargs)
|
|
151
|
+
mode = self.search_auto_update
|
|
152
|
+
if mode is True or mode == "deferred":
|
|
153
|
+
update_fields = kwargs.get("update_fields")
|
|
154
|
+
roots = {p.split("__")[0] for p in self._search_field_specs()}
|
|
155
|
+
if update_fields is None or roots & set(update_fields):
|
|
156
|
+
self.schedule_search_refresh()
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""Migration helper for `search_auto_update = "trigger"` (DB-side vector maintenance, catches bulk operations)."""
|
|
2
|
+
from django.db import migrations
|
|
3
|
+
|
|
4
|
+
_FROM = "İIıÇçĞğÖöŞşÜü"
|
|
5
|
+
_TO = "iiiccggoossuu"
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def _expr(column, unaccent):
|
|
9
|
+
e = f"lower(translate(coalesce(NEW.\"{column}\"::text, ''), '{_FROM}', '{_TO}'))"
|
|
10
|
+
return f"unaccent({e})" if unaccent else e
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def SearchTrigger(table, fields, config="simple", column="search_vector", unaccent=False):
|
|
14
|
+
"""
|
|
15
|
+
fields: {"first_name": "A", "email": "B", "bio": None} (local columns only)
|
|
16
|
+
unaccent=True additionally strips non-Turkish accents (needs `CREATE EXTENSION unaccent`,
|
|
17
|
+
e.g. migrations.RunSQL / django.contrib.postgres.operations.UnaccentExtension()).
|
|
18
|
+
"""
|
|
19
|
+
base = f"tsearch_{table}"[:50]
|
|
20
|
+
fn, trg = f"{base}_fn", f"{base}_trg"
|
|
21
|
+
parts = []
|
|
22
|
+
for col, weight in fields.items():
|
|
23
|
+
vec = f"to_tsvector('{config}', {_expr(col, unaccent)})"
|
|
24
|
+
parts.append(f"setweight({vec}, '{weight}')" if weight else vec)
|
|
25
|
+
body = " || ".join(parts) if parts else "NULL"
|
|
26
|
+
sql = f"""
|
|
27
|
+
CREATE OR REPLACE FUNCTION {fn}() RETURNS trigger AS $$
|
|
28
|
+
BEGIN
|
|
29
|
+
NEW."{column}" := {body};
|
|
30
|
+
RETURN NEW;
|
|
31
|
+
END $$ LANGUAGE plpgsql;
|
|
32
|
+
DROP TRIGGER IF EXISTS {trg} ON "{table}";
|
|
33
|
+
CREATE TRIGGER {trg} BEFORE INSERT OR UPDATE ON "{table}" FOR EACH ROW EXECUTE FUNCTION {fn}();
|
|
34
|
+
UPDATE "{table}" SET "{column}" = NULL;
|
|
35
|
+
"""
|
|
36
|
+
reverse = f'DROP TRIGGER IF EXISTS {trg} ON "{table}"; DROP FUNCTION IF EXISTS {fn}();'
|
|
37
|
+
return migrations.RunSQL(sql, reverse)
|
tsearchable/query.py
ADDED
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
"""User-facing search syntax -> PostgreSQL raw tsquery string. Pure Python, no Django needed."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import re
|
|
5
|
+
import unicodedata
|
|
6
|
+
from typing import List
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def normalize_text(text: str) -> str:
|
|
10
|
+
"""Lowercase + strip accents (Turkish aware). Used for BOTH indexing and querying,
|
|
11
|
+
so the two sides always agree and no `unaccent` DB extension is required."""
|
|
12
|
+
text = text.replace("ı", "i").replace("İ", "I")
|
|
13
|
+
text = unicodedata.normalize("NFKD", text)
|
|
14
|
+
text = "".join(c for c in text if not unicodedata.combining(c))
|
|
15
|
+
return text.lower()
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class TSQueryConverter:
|
|
19
|
+
OPERATORS = {"and": ("&", 2), "&": ("&", 2), "or": ("|", 1), "|": ("|", 1), "not": ("!", 3), "!": ("!", 3)}
|
|
20
|
+
_TOKEN = re.compile(r'"[^"]*"|\(|\)|&|\||!|[^\s()&|!"]+')
|
|
21
|
+
_WORD = re.compile(r"\w+", re.UNICODE)
|
|
22
|
+
|
|
23
|
+
def __init__(self, normalize: bool = True):
|
|
24
|
+
self.normalize = normalize
|
|
25
|
+
|
|
26
|
+
# ---- helpers -------------------------------------------------------
|
|
27
|
+
def _is_op(self, t: str) -> bool:
|
|
28
|
+
return t.lower() in self.OPERATORS
|
|
29
|
+
|
|
30
|
+
def _is_unary(self, t: str) -> bool:
|
|
31
|
+
return t.lower() in ("not", "!")
|
|
32
|
+
|
|
33
|
+
def _prec(self, t: str) -> int:
|
|
34
|
+
return self.OPERATORS[t.lower()][1]
|
|
35
|
+
|
|
36
|
+
def _words(self, text: str) -> List[str]:
|
|
37
|
+
if self.normalize:
|
|
38
|
+
text = normalize_text(text)
|
|
39
|
+
else:
|
|
40
|
+
text = text.lower()
|
|
41
|
+
# \w+ strips every tsquery metacharacter (: ' & | ! ( ) < >) -> no syntax errors / injection
|
|
42
|
+
return self._WORD.findall(text)
|
|
43
|
+
|
|
44
|
+
# ---- pipeline ------------------------------------------------------
|
|
45
|
+
def tokenize(self, query: str) -> List[str]:
|
|
46
|
+
return self._TOKEN.findall(query)
|
|
47
|
+
|
|
48
|
+
def add_implicit_and(self, tokens: List[str]) -> List[str]:
|
|
49
|
+
out: List[str] = []
|
|
50
|
+
for i, t in enumerate(tokens):
|
|
51
|
+
out.append(t)
|
|
52
|
+
if i + 1 == len(tokens):
|
|
53
|
+
break
|
|
54
|
+
n = tokens[i + 1]
|
|
55
|
+
ends_operand = t == ")" or (not self._is_op(t) and t != "(")
|
|
56
|
+
starts_operand = n == "(" or self._is_unary(n) or (not self._is_op(n) and n != ")")
|
|
57
|
+
if ends_operand and starts_operand:
|
|
58
|
+
out.append("and")
|
|
59
|
+
return out
|
|
60
|
+
|
|
61
|
+
def infix_to_postfix(self, tokens: List[str]) -> List[str]:
|
|
62
|
+
out: List[str] = []
|
|
63
|
+
stack: List[str] = []
|
|
64
|
+
for t in tokens:
|
|
65
|
+
if t == "(":
|
|
66
|
+
stack.append(t)
|
|
67
|
+
elif t == ")":
|
|
68
|
+
while stack and stack[-1] != "(":
|
|
69
|
+
out.append(stack.pop())
|
|
70
|
+
if stack:
|
|
71
|
+
stack.pop()
|
|
72
|
+
elif self._is_op(t):
|
|
73
|
+
if not self._is_unary(t):
|
|
74
|
+
while stack and stack[-1] != "(" and self._prec(stack[-1]) >= self._prec(t):
|
|
75
|
+
out.append(stack.pop())
|
|
76
|
+
stack.append(t)
|
|
77
|
+
else:
|
|
78
|
+
out.append(t)
|
|
79
|
+
while stack:
|
|
80
|
+
op = stack.pop()
|
|
81
|
+
if op != "(":
|
|
82
|
+
out.append(op)
|
|
83
|
+
return out
|
|
84
|
+
|
|
85
|
+
def format_term(self, token: str) -> str:
|
|
86
|
+
if token.startswith('"'):
|
|
87
|
+
words = self._words(token)
|
|
88
|
+
if not words:
|
|
89
|
+
return ""
|
|
90
|
+
return words[0] if len(words) == 1 else "(" + " <-> ".join(words) + ")"
|
|
91
|
+
words = self._words(token)
|
|
92
|
+
if not words:
|
|
93
|
+
return ""
|
|
94
|
+
terms = [f"{w}:*" for w in words]
|
|
95
|
+
return terms[0] if len(terms) == 1 else "(" + " & ".join(terms) + ")"
|
|
96
|
+
|
|
97
|
+
def evaluate_postfix(self, postfix: List[str]) -> str:
|
|
98
|
+
stack: List[str] = []
|
|
99
|
+
for t in postfix:
|
|
100
|
+
if self._is_unary(t):
|
|
101
|
+
if stack:
|
|
102
|
+
stack.append(f"!{stack.pop()}")
|
|
103
|
+
elif self._is_op(t):
|
|
104
|
+
if len(stack) >= 2:
|
|
105
|
+
right, left = stack.pop(), stack.pop()
|
|
106
|
+
stack.append(f"({left} {self.OPERATORS[t.lower()][0]} {right})")
|
|
107
|
+
else:
|
|
108
|
+
f = self.format_term(t)
|
|
109
|
+
if f:
|
|
110
|
+
stack.append(f)
|
|
111
|
+
return " & ".join(stack)
|
|
112
|
+
|
|
113
|
+
def convert(self, query: str) -> str:
|
|
114
|
+
if not query or not query.strip():
|
|
115
|
+
return ""
|
|
116
|
+
tokens = self.add_implicit_and(self.tokenize(query.strip()))
|
|
117
|
+
return self.evaluate_postfix(self.infix_to_postfix(tokens))
|
tsearchable/querysets.py
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
from django.contrib.postgres.search import SearchHeadline, SearchQuery, SearchRank
|
|
2
|
+
from django.db import models
|
|
3
|
+
from django.db.models import F, Q
|
|
4
|
+
|
|
5
|
+
from .query import TSQueryConverter, normalize_text
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def _build_queries(model, query, mode):
|
|
9
|
+
"""One SearchQuery per distinct text-search config used by the model."""
|
|
10
|
+
queries = []
|
|
11
|
+
for config in model.search_configs():
|
|
12
|
+
normalize = model.search_normalize and config == "simple"
|
|
13
|
+
if mode == "websearch":
|
|
14
|
+
text = normalize_text(query) if normalize else query
|
|
15
|
+
queries.append(SearchQuery(text, search_type="websearch", config=config))
|
|
16
|
+
else:
|
|
17
|
+
tsquery = TSQueryConverter(normalize=normalize).convert(query)
|
|
18
|
+
if tsquery:
|
|
19
|
+
queries.append(SearchQuery(tsquery, search_type="raw", config=config))
|
|
20
|
+
return queries
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class SearchableQuerySet(models.QuerySet):
|
|
24
|
+
def search(self, query, *, rank=False, fallback=True, fuzzy=False, highlight=None, mode=None):
|
|
25
|
+
"""
|
|
26
|
+
Full-text search.
|
|
27
|
+
|
|
28
|
+
rank order by relevance (adds `search_rank`)
|
|
29
|
+
fallback OR in icontains on `search_fallback_fields`
|
|
30
|
+
fuzzy OR in trigram similarity on `search_fuzzy_fields` (typo tolerance, needs pg_trgm)
|
|
31
|
+
highlight list of text fields -> adds `<field>_highlight` with <mark>..</mark> around matches
|
|
32
|
+
mode override `search_mode` ("tsquery" | "websearch")
|
|
33
|
+
"""
|
|
34
|
+
if not query or not query.strip():
|
|
35
|
+
return self
|
|
36
|
+
model = self.model
|
|
37
|
+
query = query.strip()
|
|
38
|
+
sqs = _build_queries(model, query, mode or model.search_mode)
|
|
39
|
+
cond = Q()
|
|
40
|
+
for sq in sqs:
|
|
41
|
+
cond |= Q(search_vector=sq)
|
|
42
|
+
distinct = False
|
|
43
|
+
if fallback:
|
|
44
|
+
suffix = "__unaccent__icontains" if model.search_fallback_unaccent else "__icontains"
|
|
45
|
+
for field in model.search_fallback_fields:
|
|
46
|
+
cond |= Q(**{field + suffix: query})
|
|
47
|
+
distinct = distinct or "__" in field
|
|
48
|
+
if fuzzy:
|
|
49
|
+
for field in model.search_fuzzy_fields:
|
|
50
|
+
cond |= Q(**{field + "__trigram_similar": query})
|
|
51
|
+
distinct = distinct or "__" in field
|
|
52
|
+
if not cond:
|
|
53
|
+
return self.none()
|
|
54
|
+
qs = self.filter(cond)
|
|
55
|
+
if distinct:
|
|
56
|
+
qs = qs.distinct()
|
|
57
|
+
if sqs:
|
|
58
|
+
if rank:
|
|
59
|
+
qs = qs.annotate(search_rank=SearchRank(F("search_vector"), sqs[0])).order_by("-search_rank")
|
|
60
|
+
for field in highlight or []:
|
|
61
|
+
qs = qs.annotate(
|
|
62
|
+
**{f"{field}_highlight": SearchHeadline(F(field), sqs[0], start_sel="<mark>", stop_sel="</mark>")}
|
|
63
|
+
)
|
|
64
|
+
return qs
|
|
65
|
+
|
|
66
|
+
def suggest(self, prefix, field, limit=10):
|
|
67
|
+
"""Autocomplete: distinct values of `field` starting with `prefix`."""
|
|
68
|
+
prefix = (prefix or "").strip()
|
|
69
|
+
if not prefix:
|
|
70
|
+
return []
|
|
71
|
+
qs = self.filter(**{f"{field}__istartswith": prefix}).order_by(field)
|
|
72
|
+
return list(qs.values_list(field, flat=True).distinct()[:limit])
|
|
73
|
+
|
|
74
|
+
def rebuild_search_vectors(self, chunk_size=500):
|
|
75
|
+
"""Recompute search_vector for every row (backfill, or after bulk_create/bulk_update/update())."""
|
|
76
|
+
n = 0
|
|
77
|
+
for obj in self.iterator(chunk_size=chunk_size):
|
|
78
|
+
obj.refresh_search_vector()
|
|
79
|
+
n += 1
|
|
80
|
+
return n
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
SearchableManager = models.Manager.from_queryset(SearchableQuerySet)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def search_all(models_, query, limit=20, **kwargs):
|
|
87
|
+
"""Search several SearchableModels at once; returns objects merged and sorted by relevance."""
|
|
88
|
+
hits = []
|
|
89
|
+
for model in models_:
|
|
90
|
+
for obj in model._default_manager.search(query, rank=True, **kwargs)[:limit]:
|
|
91
|
+
if not hasattr(obj, "search_rank"):
|
|
92
|
+
obj.search_rank = 0.0
|
|
93
|
+
hits.append(obj)
|
|
94
|
+
hits.sort(key=lambda o: o.search_rank, reverse=True)
|
|
95
|
+
return hits[:limit]
|
tsearchable/signals.py
ADDED
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
from django.apps import apps
|
|
2
|
+
from django.db import transaction
|
|
3
|
+
from django.db.models.signals import m2m_changed
|
|
4
|
+
from django.dispatch import Signal
|
|
5
|
+
|
|
6
|
+
#: Sent (on commit) when `search_auto_update = "deferred"`. Connect a receiver that enqueues
|
|
7
|
+
#: a Celery/RQ/... task calling ``instance.refresh_search_vector()``.
|
|
8
|
+
search_vector_refresh_requested = Signal() # kwargs: sender=model class, instance=obj
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def _make_m2m_handler(model):
|
|
12
|
+
def handler(sender, instance, action, reverse, pk_set=None, **kwargs):
|
|
13
|
+
if action not in ("post_add", "post_remove", "post_clear"):
|
|
14
|
+
return
|
|
15
|
+
if not reverse:
|
|
16
|
+
instance.schedule_search_refresh()
|
|
17
|
+
elif pk_set:
|
|
18
|
+
for obj in model._base_manager.filter(pk__in=pk_set):
|
|
19
|
+
obj.schedule_search_refresh()
|
|
20
|
+
|
|
21
|
+
return handler
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def connect_m2m_handlers():
|
|
25
|
+
"""Refresh vectors automatically when an M2M used in `search_fields` (e.g. `tags__title`) changes."""
|
|
26
|
+
from .models import SearchableModel
|
|
27
|
+
|
|
28
|
+
for model in apps.get_models():
|
|
29
|
+
if not issubclass(model, SearchableModel) or model.search_auto_update not in (True, "deferred"):
|
|
30
|
+
continue
|
|
31
|
+
for path in model._search_field_specs():
|
|
32
|
+
try:
|
|
33
|
+
field = model._meta.get_field(path.split("__")[0])
|
|
34
|
+
except Exception:
|
|
35
|
+
continue
|
|
36
|
+
if getattr(field, "many_to_many", False) and not field.auto_created:
|
|
37
|
+
m2m_changed.connect(
|
|
38
|
+
_make_m2m_handler(model),
|
|
39
|
+
sender=field.remote_field.through,
|
|
40
|
+
weak=False,
|
|
41
|
+
dispatch_uid=f"tsearchable.{model._meta.label_lower}.{field.name}",
|
|
42
|
+
)
|
tsearchable/testing.py
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""Small assertions for your own test-suite: assert_search_finds(Candidate, "python", alice)"""
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
def _qs(target):
|
|
5
|
+
return target._default_manager.all() if hasattr(target, "_default_manager") else target
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def assert_search_finds(target, query, *objs, **search_kwargs):
|
|
9
|
+
found = set(_qs(target).search(query, **search_kwargs).values_list("pk", flat=True))
|
|
10
|
+
missing = [o for o in objs if o.pk not in found]
|
|
11
|
+
assert not missing, f"search({query!r}) did not find: {missing}"
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def assert_search_misses(target, query, *objs, **search_kwargs):
|
|
15
|
+
found = set(_qs(target).search(query, **search_kwargs).values_list("pk", flat=True))
|
|
16
|
+
present = [o for o in objs if o.pk in found]
|
|
17
|
+
assert not present, f"search({query!r}) unexpectedly found: {present}"
|
tsearchable/views.py
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
from .forms import SearchForm
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class SearchableListViewMixin:
|
|
5
|
+
"""Put before ListView: class CandidateList(SearchableListViewMixin, ListView): model = Candidate"""
|
|
6
|
+
|
|
7
|
+
search_param = "q"
|
|
8
|
+
search_rank = False
|
|
9
|
+
search_form_class = SearchForm
|
|
10
|
+
|
|
11
|
+
def get_search_form(self):
|
|
12
|
+
return self.search_form_class(self.request.GET or None, field_name=self.search_param)
|
|
13
|
+
|
|
14
|
+
def get_queryset(self):
|
|
15
|
+
queryset = super().get_queryset()
|
|
16
|
+
self.search_form = self.get_search_form()
|
|
17
|
+
return self.search_form.apply(queryset, rank=self.search_rank)
|
|
18
|
+
|
|
19
|
+
def get_context_data(self, **kwargs):
|
|
20
|
+
context = super().get_context_data(**kwargs)
|
|
21
|
+
form = getattr(self, "search_form", None) or self.get_search_form()
|
|
22
|
+
context["search_form"] = form
|
|
23
|
+
context["search_query"] = form.query
|
|
24
|
+
return context
|