django-grouper 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,32 @@
1
+ Metadata-Version: 2.1
2
+ Name: django-grouper
3
+ Version: 0.1.0
4
+ Summary:
5
+ Author: Birger Schacht
6
+ Requires-Python: >=3.10,<4.0
7
+ Classifier: Programming Language :: Python :: 3
8
+ Classifier: Programming Language :: Python :: 3.10
9
+ Classifier: Programming Language :: Python :: 3.11
10
+ Classifier: Programming Language :: Python :: 3.12
11
+ Requires-Dist: django (>=3)
12
+ Requires-Dist: pandas (>=2.2.2,<3.0.0)
13
+ Requires-Dist: scikit-learn (>=1.5.0,<2.0.0)
14
+ Requires-Dist: sparse-dot-topn (>=1.1.1,<2.0.0)
15
+ Description-Content-Type: text/markdown
16
+
17
+ Group Django model instances on similar fields
18
+
19
+ Based on [this tutorial](https://towardsdatascience.com/group-thousands-of-similar-spreadsheet-text-cells-in-seconds-2493b3ce6d8d)
20
+
21
+ # Installation
22
+
23
+ 1. add `django_grouper` to your `INSTALLED_APPS`
24
+ 2. add `path("", include("django_grouper.urls"))` to your `urlpatterns`
25
+ 3. go to `/grouper` and add the param `group_content_type` to specify which ContentType to group and the params `group_fields` to specify which fields to group by
26
+
27
+ # Configuration
28
+
29
+ If you define a `GROUP_FILTER` in you settings, the queryset in the `/grouper` view will be passed through this filter. The filter is expected to
30
+ be a callable and will receive the queryset and the request as parameters. This means you can pass the queryset together with the request to a
31
+ django-filter filter.
32
+
@@ -0,0 +1,15 @@
1
+ Group Django model instances on similar fields
2
+
3
+ Based on [this tutorial](https://towardsdatascience.com/group-thousands-of-similar-spreadsheet-text-cells-in-seconds-2493b3ce6d8d)
4
+
5
+ # Installation
6
+
7
+ 1. add `django_grouper` to your `INSTALLED_APPS`
8
+ 2. add `path("", include("django_grouper.urls"))` to your `urlpatterns`
9
+ 3. go to `/grouper` and add the param `group_content_type` to specify which ContentType to group and the params `group_fields` to specify which fields to group by
10
+
11
+ # Configuration
12
+
13
+ If you define a `GROUP_FILTER` in you settings, the queryset in the `/grouper` view will be passed through this filter. The filter is expected to
14
+ be a callable and will receive the queryset and the request as parameters. This means you can pass the queryset together with the request to a
15
+ django-filter filter.
File without changes
@@ -0,0 +1,6 @@
1
+ from django.apps import AppConfig
2
+
3
+
4
+ class DjangoGrouperConfig(AppConfig):
5
+ default_auto_field = "django.db.models.AutoField"
6
+ name = "django_grouper"
@@ -0,0 +1,52 @@
1
+ {% extends "webpage/base.html" %}
2
+ {% block content %}
3
+ {% if merged_ids %}
4
+ {% for merge_id in merged_ids %}Merged {{ merge_id }}{% endfor %}
5
+ {% else %}
6
+ <style>
7
+ .checkbox:checked ~ .card{
8
+ background: #ffc107;
9
+ }
10
+ .checkbox {
11
+ position: absolute;
12
+ top: 0;
13
+ left: 0;
14
+ opacity: 0;
15
+ width: 100%;
16
+ height: 100%;
17
+ z-index: 999;
18
+ cursor: pointer;
19
+ }
20
+ .card_area{
21
+ position: relative;
22
+ margin-bottom: 30px;
23
+ }
24
+ </style>
25
+ <div class="container">
26
+ <form class="mt-5" method="post">
27
+ {% csrf_token %}
28
+ <button type="submit" class="btn btn-primary">Merge selected</button>
29
+ <input type="hidden"
30
+ name="content_type"
31
+ value="{{ content_type.app_label }}.{{ content_type.model }}">
32
+ <div class="card-columns mt-5">
33
+ {% for object in object_list %}
34
+ <div class="card_area">
35
+ <input class="checkbox"
36
+ type="checkbox"
37
+ id="to_merge"
38
+ name="to_merge"
39
+ value="{{ object.pk }}" />
40
+ <div class="card border" style="width: 18rem;">
41
+ <div class="card-body">
42
+ <h5 class="card-title">{{ object }}</h5>
43
+ <p class="card-text">Details</p>
44
+ </div>
45
+ </div>
46
+ </div>
47
+ {% endfor %}
48
+ </div>
49
+ </form>
50
+ </div>
51
+ {% endif %}
52
+ {% endblock content %}
@@ -0,0 +1,18 @@
1
+ {% extends "webpage/base.html" %}
2
+ {% block content %}
3
+ <div class="container mt-3">
4
+ <ul>
5
+ {% for group, ids in groups.items %}
6
+ <li>
7
+ <a href="{% url "group" %}?group_content_type={{ content_type.app_label }}.{{ content_type.model }}{% for id in ids %}&ids={{ id }}{% endfor %}">{{ group }}</a>:
8
+ <details style="display: inline;">
9
+ <summary>{{ ids|length }}</summary>
10
+ {% for id in ids %}
11
+ <a href="{% url "apis_core:apis_entities:generic_entities_detail_view" content_type.model id %}">{{ id }}</a>,
12
+ {% endfor %}
13
+ </details>
14
+ </li>
15
+ {% endfor %}
16
+ </ul>
17
+ </div>
18
+ {% endblock content %}
@@ -0,0 +1,8 @@
1
+ from django.urls import path
2
+
3
+ from . import views
4
+
5
+ urlpatterns = [
6
+ path("grouper/", views.Grouper.as_view(), name="grouper"),
7
+ path("group/", views.Group.as_view(), name="group"),
8
+ ]
@@ -0,0 +1,79 @@
1
+ import re
2
+ import pandas as pd
3
+ from sklearn.feature_extraction.text import TfidfVectorizer
4
+ from sparse_dot_topn import awesome_cossim_topn
5
+
6
+
7
+ def group_queryset(queryset, fields=[]) -> dict:
8
+ # Instaniate our lookup hash table
9
+ group_lookup = {}
10
+
11
+ # Write a function for cleaning strings and returning an array of ngrams
12
+ def ngrams_analyzer(string):
13
+ string = re.sub(r"[,-./]", r"", string)
14
+ ngrams = zip(*[string[i:] for i in range(5)]) # N-Gram length is 5
15
+ return ["".join(ngram) for ngram in ngrams]
16
+
17
+ def find_group(row, col):
18
+ # If either the row or the col string have already been given
19
+ # a group, return that group. Otherwise return none
20
+ if row in group_lookup:
21
+ return group_lookup[row]
22
+ elif col in group_lookup:
23
+ return group_lookup[col]
24
+ else:
25
+ return None
26
+
27
+ def add_vals_to_lookup(group, row, col):
28
+ # Once we know the group name, set it as the value
29
+ # for both strings in the group_lookup
30
+ group_lookup[row] = group
31
+ group_lookup[col] = group
32
+
33
+ def add_pair_to_lookup(row, col):
34
+ # in this function we'll add both the row and the col to the lookup
35
+ group = find_group(row, col) # first, see if one has already been added
36
+ if group is not None:
37
+ # if we already know the group, make sure both row and col are in lookup
38
+ add_vals_to_lookup(group, row, col)
39
+ else:
40
+ # if we get here, we need to add a new group.
41
+ # The name is arbitrary, so just make it the row
42
+ add_vals_to_lookup(row, row, col)
43
+
44
+ # Construct your vectorizer for building the TF-IDF matrix
45
+ vectorizer = TfidfVectorizer(analyzer=ngrams_analyzer)
46
+
47
+ if fields:
48
+ allfields = fields + ["pk"]
49
+ values = queryset.values_list(*allfields)
50
+ df = pd.DataFrame(list(values), columns=allfields)
51
+
52
+ df["grouper"] = df[fields.pop(0)].astype(str).str.cat(df[fields].astype(str))
53
+
54
+ # Grab the column you'd like to group, filter out duplicate values
55
+ # and make sure the values are Unicode
56
+ vals = df["grouper"].unique().astype("U")
57
+
58
+ # Build the matrix!!!
59
+ tfidf_matrix = vectorizer.fit_transform(vals)
60
+
61
+ cosine_matrix = awesome_cossim_topn(
62
+ tfidf_matrix, tfidf_matrix.transpose(), vals.size, 0.8
63
+ )
64
+
65
+ # Build a coordinate matrix
66
+ coo_matrix = cosine_matrix.tocoo()
67
+
68
+ # for each row and column in coo_matrix
69
+ # if they're not the same string add them to the group lookup
70
+ for row, col in zip(coo_matrix.row, coo_matrix.col):
71
+ if row != col:
72
+ add_pair_to_lookup(vals[row], vals[col])
73
+
74
+ df["Group"] = df["grouper"].map(group_lookup).fillna(df["grouper"])
75
+
76
+ d = df.groupby("Group")["pk"].apply(list).to_dict()
77
+ ret = {key: value for key, value in d.items() if len(value) > 1}
78
+ return dict(sorted(ret.items(), key=lambda item: len(item[1]), reverse=True))
79
+ return {}
@@ -0,0 +1,63 @@
1
+ from django.conf import settings
2
+ from django.contrib.contenttypes.models import ContentType
3
+ from django.shortcuts import get_object_or_404
4
+ from django.views.generic.base import TemplateView
5
+ from django.http import HttpResponseRedirect
6
+
7
+
8
+ from .utils import group_queryset
9
+
10
+
11
+ class BaseView(TemplateView):
12
+ def dispatch(self, request, *args, **kwargs):
13
+ app_label, model = request.GET.get("group_content_type", ".").split(".")
14
+ self.django_content_type = get_object_or_404(
15
+ ContentType, app_label=app_label, model=model
16
+ )
17
+ return super().dispatch(request, *args, **kwargs)
18
+
19
+
20
+ class Grouper(BaseView):
21
+ template_name = "django_grouper/grouper.html"
22
+
23
+ def get_context_data(self, *args, **kwargs):
24
+ fields = self.request.GET.getlist("group_fields", [])
25
+
26
+ ctx = super().get_context_data(*args, **kwargs)
27
+ ctx["content_type"] = self.django_content_type
28
+ ctx["groups"] = group_queryset(self.get_queryset(), fields)
29
+ return ctx
30
+
31
+ def get_queryset(self):
32
+ qs = self.django_content_type.model_class().objects.all()
33
+ if hasattr(settings, "GROUP_FILTER"):
34
+ qs = settings.GROUP_FILTER(qs, self.request)
35
+ return qs
36
+
37
+
38
+ class Group(BaseView):
39
+ template_name = "django_grouper/group.html"
40
+
41
+ def get_context_data(self, *args, **kwargs):
42
+ ids = self.request.GET.getlist("ids", [])
43
+
44
+ ctx = super().get_context_data(*args, **kwargs)
45
+ ctx["content_type"] = self.django_content_type
46
+ ctx["title"] = self.request.GET.get("group_title")
47
+ ctx["object_list"] = self.django_content_type.model_class().objects.filter(
48
+ pk__in=ids
49
+ )
50
+ return ctx
51
+
52
+ def post(self, request, *args, **kwargs):
53
+ ctx = self.get_context_data()
54
+ ctx["merged_ids"] = []
55
+ newinstance = self.django_content_type.model_class().objects.create()
56
+ for merge_id in request.POST.getlist("to_merge"):
57
+ mergeobject = get_object_or_404(
58
+ self.django_content_type.model_class(), pk=merge_id
59
+ )
60
+ mergeobject.grouped_into = newinstance
61
+ mergeobject.save()
62
+ ctx["merged_ids"].append(merge_id)
63
+ return HttpResponseRedirect(newinstance.get_absolute_url())
@@ -0,0 +1,21 @@
1
+ [tool.poetry]
2
+ name = "django-grouper"
3
+ version = "0.1.0"
4
+ description = ""
5
+ authors = ["Birger Schacht"]
6
+ readme = "README.md"
7
+
8
+ [tool.poetry.dependencies]
9
+ python = "^3.10"
10
+ pandas = "^2.2.2"
11
+ scikit-learn = "^1.5.0"
12
+ sparse-dot-topn = "^1.1.1"
13
+ django = ">=3"
14
+
15
+ [tool.poetry.group.dev.dependencies]
16
+ ruff = "^0.4.8"
17
+ djlint = "^1.31.1"
18
+
19
+ [build-system]
20
+ requires = ["poetry-core"]
21
+ build-backend = "poetry.core.masonry.api"