django-grouper 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- django_grouper-0.1.0/PKG-INFO +32 -0
- django_grouper-0.1.0/README.md +15 -0
- django_grouper-0.1.0/django_grouper/__init__.py +0 -0
- django_grouper-0.1.0/django_grouper/apps.py +6 -0
- django_grouper-0.1.0/django_grouper/templates/django_grouper/group.html +52 -0
- django_grouper-0.1.0/django_grouper/templates/django_grouper/grouper.html +18 -0
- django_grouper-0.1.0/django_grouper/urls.py +8 -0
- django_grouper-0.1.0/django_grouper/utils.py +79 -0
- django_grouper-0.1.0/django_grouper/views.py +63 -0
- django_grouper-0.1.0/pyproject.toml +21 -0
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
Metadata-Version: 2.1
|
|
2
|
+
Name: django-grouper
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary:
|
|
5
|
+
Author: Birger Schacht
|
|
6
|
+
Requires-Python: >=3.10,<4.0
|
|
7
|
+
Classifier: Programming Language :: Python :: 3
|
|
8
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
9
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
10
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
11
|
+
Requires-Dist: django (>=3)
|
|
12
|
+
Requires-Dist: pandas (>=2.2.2,<3.0.0)
|
|
13
|
+
Requires-Dist: scikit-learn (>=1.5.0,<2.0.0)
|
|
14
|
+
Requires-Dist: sparse-dot-topn (>=1.1.1,<2.0.0)
|
|
15
|
+
Description-Content-Type: text/markdown
|
|
16
|
+
|
|
17
|
+
Group Django model instances on similar fields
|
|
18
|
+
|
|
19
|
+
Based on [this tutorial](https://towardsdatascience.com/group-thousands-of-similar-spreadsheet-text-cells-in-seconds-2493b3ce6d8d)
|
|
20
|
+
|
|
21
|
+
# Installation
|
|
22
|
+
|
|
23
|
+
1. add `django_grouper` to your `INSTALLED_APPS`
|
|
24
|
+
2. add `path("", include("django_grouper.urls"))` to your `urlpatterns`
|
|
25
|
+
3. go to `/grouper` and add the param `group_content_type` to specify which ContentType to group and the params `group_fields` to specify which fields to group by
|
|
26
|
+
|
|
27
|
+
# Configuration
|
|
28
|
+
|
|
29
|
+
If you define a `GROUP_FILTER` in you settings, the queryset in the `/grouper` view will be passed through this filter. The filter is expected to
|
|
30
|
+
be a callable and will receive the queryset and the request as parameters. This means you can pass the queryset together with the request to a
|
|
31
|
+
django-filter filter.
|
|
32
|
+
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
Group Django model instances on similar fields
|
|
2
|
+
|
|
3
|
+
Based on [this tutorial](https://towardsdatascience.com/group-thousands-of-similar-spreadsheet-text-cells-in-seconds-2493b3ce6d8d)
|
|
4
|
+
|
|
5
|
+
# Installation
|
|
6
|
+
|
|
7
|
+
1. add `django_grouper` to your `INSTALLED_APPS`
|
|
8
|
+
2. add `path("", include("django_grouper.urls"))` to your `urlpatterns`
|
|
9
|
+
3. go to `/grouper` and add the param `group_content_type` to specify which ContentType to group and the params `group_fields` to specify which fields to group by
|
|
10
|
+
|
|
11
|
+
# Configuration
|
|
12
|
+
|
|
13
|
+
If you define a `GROUP_FILTER` in you settings, the queryset in the `/grouper` view will be passed through this filter. The filter is expected to
|
|
14
|
+
be a callable and will receive the queryset and the request as parameters. This means you can pass the queryset together with the request to a
|
|
15
|
+
django-filter filter.
|
|
File without changes
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
{% extends "webpage/base.html" %}
|
|
2
|
+
{% block content %}
|
|
3
|
+
{% if merged_ids %}
|
|
4
|
+
{% for merge_id in merged_ids %}Merged {{ merge_id }}{% endfor %}
|
|
5
|
+
{% else %}
|
|
6
|
+
<style>
|
|
7
|
+
.checkbox:checked ~ .card{
|
|
8
|
+
background: #ffc107;
|
|
9
|
+
}
|
|
10
|
+
.checkbox {
|
|
11
|
+
position: absolute;
|
|
12
|
+
top: 0;
|
|
13
|
+
left: 0;
|
|
14
|
+
opacity: 0;
|
|
15
|
+
width: 100%;
|
|
16
|
+
height: 100%;
|
|
17
|
+
z-index: 999;
|
|
18
|
+
cursor: pointer;
|
|
19
|
+
}
|
|
20
|
+
.card_area{
|
|
21
|
+
position: relative;
|
|
22
|
+
margin-bottom: 30px;
|
|
23
|
+
}
|
|
24
|
+
</style>
|
|
25
|
+
<div class="container">
|
|
26
|
+
<form class="mt-5" method="post">
|
|
27
|
+
{% csrf_token %}
|
|
28
|
+
<button type="submit" class="btn btn-primary">Merge selected</button>
|
|
29
|
+
<input type="hidden"
|
|
30
|
+
name="content_type"
|
|
31
|
+
value="{{ content_type.app_label }}.{{ content_type.model }}">
|
|
32
|
+
<div class="card-columns mt-5">
|
|
33
|
+
{% for object in object_list %}
|
|
34
|
+
<div class="card_area">
|
|
35
|
+
<input class="checkbox"
|
|
36
|
+
type="checkbox"
|
|
37
|
+
id="to_merge"
|
|
38
|
+
name="to_merge"
|
|
39
|
+
value="{{ object.pk }}" />
|
|
40
|
+
<div class="card border" style="width: 18rem;">
|
|
41
|
+
<div class="card-body">
|
|
42
|
+
<h5 class="card-title">{{ object }}</h5>
|
|
43
|
+
<p class="card-text">Details</p>
|
|
44
|
+
</div>
|
|
45
|
+
</div>
|
|
46
|
+
</div>
|
|
47
|
+
{% endfor %}
|
|
48
|
+
</div>
|
|
49
|
+
</form>
|
|
50
|
+
</div>
|
|
51
|
+
{% endif %}
|
|
52
|
+
{% endblock content %}
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
{% extends "webpage/base.html" %}
|
|
2
|
+
{% block content %}
|
|
3
|
+
<div class="container mt-3">
|
|
4
|
+
<ul>
|
|
5
|
+
{% for group, ids in groups.items %}
|
|
6
|
+
<li>
|
|
7
|
+
<a href="{% url "group" %}?group_content_type={{ content_type.app_label }}.{{ content_type.model }}{% for id in ids %}&ids={{ id }}{% endfor %}">{{ group }}</a>:
|
|
8
|
+
<details style="display: inline;">
|
|
9
|
+
<summary>{{ ids|length }}</summary>
|
|
10
|
+
{% for id in ids %}
|
|
11
|
+
<a href="{% url "apis_core:apis_entities:generic_entities_detail_view" content_type.model id %}">{{ id }}</a>,
|
|
12
|
+
{% endfor %}
|
|
13
|
+
</details>
|
|
14
|
+
</li>
|
|
15
|
+
{% endfor %}
|
|
16
|
+
</ul>
|
|
17
|
+
</div>
|
|
18
|
+
{% endblock content %}
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
import re
|
|
2
|
+
import pandas as pd
|
|
3
|
+
from sklearn.feature_extraction.text import TfidfVectorizer
|
|
4
|
+
from sparse_dot_topn import awesome_cossim_topn
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def group_queryset(queryset, fields=[]) -> dict:
|
|
8
|
+
# Instaniate our lookup hash table
|
|
9
|
+
group_lookup = {}
|
|
10
|
+
|
|
11
|
+
# Write a function for cleaning strings and returning an array of ngrams
|
|
12
|
+
def ngrams_analyzer(string):
|
|
13
|
+
string = re.sub(r"[,-./]", r"", string)
|
|
14
|
+
ngrams = zip(*[string[i:] for i in range(5)]) # N-Gram length is 5
|
|
15
|
+
return ["".join(ngram) for ngram in ngrams]
|
|
16
|
+
|
|
17
|
+
def find_group(row, col):
|
|
18
|
+
# If either the row or the col string have already been given
|
|
19
|
+
# a group, return that group. Otherwise return none
|
|
20
|
+
if row in group_lookup:
|
|
21
|
+
return group_lookup[row]
|
|
22
|
+
elif col in group_lookup:
|
|
23
|
+
return group_lookup[col]
|
|
24
|
+
else:
|
|
25
|
+
return None
|
|
26
|
+
|
|
27
|
+
def add_vals_to_lookup(group, row, col):
|
|
28
|
+
# Once we know the group name, set it as the value
|
|
29
|
+
# for both strings in the group_lookup
|
|
30
|
+
group_lookup[row] = group
|
|
31
|
+
group_lookup[col] = group
|
|
32
|
+
|
|
33
|
+
def add_pair_to_lookup(row, col):
|
|
34
|
+
# in this function we'll add both the row and the col to the lookup
|
|
35
|
+
group = find_group(row, col) # first, see if one has already been added
|
|
36
|
+
if group is not None:
|
|
37
|
+
# if we already know the group, make sure both row and col are in lookup
|
|
38
|
+
add_vals_to_lookup(group, row, col)
|
|
39
|
+
else:
|
|
40
|
+
# if we get here, we need to add a new group.
|
|
41
|
+
# The name is arbitrary, so just make it the row
|
|
42
|
+
add_vals_to_lookup(row, row, col)
|
|
43
|
+
|
|
44
|
+
# Construct your vectorizer for building the TF-IDF matrix
|
|
45
|
+
vectorizer = TfidfVectorizer(analyzer=ngrams_analyzer)
|
|
46
|
+
|
|
47
|
+
if fields:
|
|
48
|
+
allfields = fields + ["pk"]
|
|
49
|
+
values = queryset.values_list(*allfields)
|
|
50
|
+
df = pd.DataFrame(list(values), columns=allfields)
|
|
51
|
+
|
|
52
|
+
df["grouper"] = df[fields.pop(0)].astype(str).str.cat(df[fields].astype(str))
|
|
53
|
+
|
|
54
|
+
# Grab the column you'd like to group, filter out duplicate values
|
|
55
|
+
# and make sure the values are Unicode
|
|
56
|
+
vals = df["grouper"].unique().astype("U")
|
|
57
|
+
|
|
58
|
+
# Build the matrix!!!
|
|
59
|
+
tfidf_matrix = vectorizer.fit_transform(vals)
|
|
60
|
+
|
|
61
|
+
cosine_matrix = awesome_cossim_topn(
|
|
62
|
+
tfidf_matrix, tfidf_matrix.transpose(), vals.size, 0.8
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
# Build a coordinate matrix
|
|
66
|
+
coo_matrix = cosine_matrix.tocoo()
|
|
67
|
+
|
|
68
|
+
# for each row and column in coo_matrix
|
|
69
|
+
# if they're not the same string add them to the group lookup
|
|
70
|
+
for row, col in zip(coo_matrix.row, coo_matrix.col):
|
|
71
|
+
if row != col:
|
|
72
|
+
add_pair_to_lookup(vals[row], vals[col])
|
|
73
|
+
|
|
74
|
+
df["Group"] = df["grouper"].map(group_lookup).fillna(df["grouper"])
|
|
75
|
+
|
|
76
|
+
d = df.groupby("Group")["pk"].apply(list).to_dict()
|
|
77
|
+
ret = {key: value for key, value in d.items() if len(value) > 1}
|
|
78
|
+
return dict(sorted(ret.items(), key=lambda item: len(item[1]), reverse=True))
|
|
79
|
+
return {}
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
from django.conf import settings
|
|
2
|
+
from django.contrib.contenttypes.models import ContentType
|
|
3
|
+
from django.shortcuts import get_object_or_404
|
|
4
|
+
from django.views.generic.base import TemplateView
|
|
5
|
+
from django.http import HttpResponseRedirect
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
from .utils import group_queryset
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class BaseView(TemplateView):
|
|
12
|
+
def dispatch(self, request, *args, **kwargs):
|
|
13
|
+
app_label, model = request.GET.get("group_content_type", ".").split(".")
|
|
14
|
+
self.django_content_type = get_object_or_404(
|
|
15
|
+
ContentType, app_label=app_label, model=model
|
|
16
|
+
)
|
|
17
|
+
return super().dispatch(request, *args, **kwargs)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class Grouper(BaseView):
|
|
21
|
+
template_name = "django_grouper/grouper.html"
|
|
22
|
+
|
|
23
|
+
def get_context_data(self, *args, **kwargs):
|
|
24
|
+
fields = self.request.GET.getlist("group_fields", [])
|
|
25
|
+
|
|
26
|
+
ctx = super().get_context_data(*args, **kwargs)
|
|
27
|
+
ctx["content_type"] = self.django_content_type
|
|
28
|
+
ctx["groups"] = group_queryset(self.get_queryset(), fields)
|
|
29
|
+
return ctx
|
|
30
|
+
|
|
31
|
+
def get_queryset(self):
|
|
32
|
+
qs = self.django_content_type.model_class().objects.all()
|
|
33
|
+
if hasattr(settings, "GROUP_FILTER"):
|
|
34
|
+
qs = settings.GROUP_FILTER(qs, self.request)
|
|
35
|
+
return qs
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class Group(BaseView):
|
|
39
|
+
template_name = "django_grouper/group.html"
|
|
40
|
+
|
|
41
|
+
def get_context_data(self, *args, **kwargs):
|
|
42
|
+
ids = self.request.GET.getlist("ids", [])
|
|
43
|
+
|
|
44
|
+
ctx = super().get_context_data(*args, **kwargs)
|
|
45
|
+
ctx["content_type"] = self.django_content_type
|
|
46
|
+
ctx["title"] = self.request.GET.get("group_title")
|
|
47
|
+
ctx["object_list"] = self.django_content_type.model_class().objects.filter(
|
|
48
|
+
pk__in=ids
|
|
49
|
+
)
|
|
50
|
+
return ctx
|
|
51
|
+
|
|
52
|
+
def post(self, request, *args, **kwargs):
|
|
53
|
+
ctx = self.get_context_data()
|
|
54
|
+
ctx["merged_ids"] = []
|
|
55
|
+
newinstance = self.django_content_type.model_class().objects.create()
|
|
56
|
+
for merge_id in request.POST.getlist("to_merge"):
|
|
57
|
+
mergeobject = get_object_or_404(
|
|
58
|
+
self.django_content_type.model_class(), pk=merge_id
|
|
59
|
+
)
|
|
60
|
+
mergeobject.grouped_into = newinstance
|
|
61
|
+
mergeobject.save()
|
|
62
|
+
ctx["merged_ids"].append(merge_id)
|
|
63
|
+
return HttpResponseRedirect(newinstance.get_absolute_url())
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
[tool.poetry]
|
|
2
|
+
name = "django-grouper"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = ""
|
|
5
|
+
authors = ["Birger Schacht"]
|
|
6
|
+
readme = "README.md"
|
|
7
|
+
|
|
8
|
+
[tool.poetry.dependencies]
|
|
9
|
+
python = "^3.10"
|
|
10
|
+
pandas = "^2.2.2"
|
|
11
|
+
scikit-learn = "^1.5.0"
|
|
12
|
+
sparse-dot-topn = "^1.1.1"
|
|
13
|
+
django = ">=3"
|
|
14
|
+
|
|
15
|
+
[tool.poetry.group.dev.dependencies]
|
|
16
|
+
ruff = "^0.4.8"
|
|
17
|
+
djlint = "^1.31.1"
|
|
18
|
+
|
|
19
|
+
[build-system]
|
|
20
|
+
requires = ["poetry-core"]
|
|
21
|
+
build-backend = "poetry.core.masonry.api"
|