django-robots-manager 6.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,175 @@
1
+ Metadata-Version: 2.4
2
+ Name: django-robots-manager
3
+ Version: 6.2.1
4
+ Summary: Robots exclusion application for Django, complementing Sitemaps. Continuation of django-robots 6.x after a long pause in upstream releases.
5
+ License-Expression: BSD-3-Clause
6
+ License-File: LICENSE.txt
7
+ Author: Artem Fabrikov
8
+ Author-email: a.fabrikov1406@yandex.ru
9
+ Requires-Python: >=3.10
10
+ Classifier: Environment :: Web Environment
11
+ Classifier: Framework :: Django
12
+ Classifier: Framework :: Django :: 4.2
13
+ Classifier: Framework :: Django :: 5.2
14
+ Classifier: Framework :: Django :: 6.0
15
+ Classifier: Framework :: Django :: 6.1
16
+ Classifier: Intended Audience :: Developers
17
+ Classifier: Operating System :: OS Independent
18
+ Classifier: Programming Language :: Python :: 3
19
+ Classifier: Programming Language :: Python :: 3.10
20
+ Classifier: Programming Language :: Python :: 3.11
21
+ Classifier: Programming Language :: Python :: 3.12
22
+ Classifier: Programming Language :: Python :: 3.13
23
+ Classifier: Programming Language :: Python :: 3.14
24
+ Classifier: Topic :: Internet :: WWW/HTTP :: Dynamic Content
25
+ Requires-Dist: Django (>=4.2)
26
+ Project-URL: Homepage, https://github.com/KitKat-ru/django-robots-manager
27
+ Project-URL: Issues, https://github.com/KitKat-ru/django-robots-manager/issues
28
+ Project-URL: Repository, https://github.com/KitKat-ru/django-robots-manager
29
+ Project-URL: Upstream, https://github.com/jazzband/django-robots/
30
+ Description-Content-Type: text/markdown
31
+
32
+ # django-robots-manager
33
+
34
+ [![Tests](https://github.com/KitKat-ru/django-robots-manager/actions/workflows/tests.yml/badge.svg)](https://github.com/KitKat-ru/django-robots-manager/actions/workflows/tests.yml)
35
+ [![GitHub tag](https://img.shields.io/github/v/tag/KitKat-ru/django-robots-manager?sort=semver)](https://github.com/KitKat-ru/django-robots-manager/tags)
36
+ [![Python](https://img.shields.io/badge/python-3.10%20%7C%203.11%20%7C%203.12%20%7C%203.13%20%7C%203.14-blue)](https://github.com/KitKat-ru/django-robots-manager/actions/workflows/tests.yml)
37
+ [![Django](https://img.shields.io/badge/django-4.2%20%7C%205.2%20%7C%206.0%20%7C%206.1-0C4B33)](https://github.com/KitKat-ru/django-robots-manager/actions/workflows/tests.yml)
38
+ [![License](https://img.shields.io/badge/license-BSD--3--Clause-blue)](https://github.com/KitKat-ru/django-robots-manager/blob/main/LICENSE.txt)
39
+ [![Ruff](https://img.shields.io/endpoint?url=https://raw.githubusercontent.com/astral-sh/ruff/main/assets/badge/v2.json)](https://github.com/astral-sh/ruff)
40
+ [![pre-commit](https://img.shields.io/badge/pre--commit-enabled-brightgreen?logo=pre-commit)](https://github.com/pre-commit/pre-commit)
41
+
42
+ A continuation of [django-robots](https://github.com/jazzband/django-robots/),
43
+ picking up from the 6.x series (based on release 6.1), after a long pause in releases
44
+ of the original library.
45
+
46
+ The import path and app label stay `robots`, so it is a drop-in replacement for
47
+ django-robots 5.0 and 6.x: existing tables and migration history are reused.
48
+
49
+ ## Changes from upstream 6.1
50
+
51
+ - `__version__` is read via `importlib.metadata` only; the `pkg_resources` fallback
52
+ and `default_app_config` (Django < 3.2) are removed.
53
+ - The South guard in `robots.migrations` is removed.
54
+ - `RuleAdminForm` rejects a rule whose allowed and disallowed URLs share a pattern.
55
+ - `RuleAdminForm` rejects a second rule for the same robot (case-insensitive) on the
56
+ same site.
57
+ - `ROBOTS_USE_HOST` now defaults to `False`: Yandex
58
+ [stopped using](https://webmaster.yandex.ru/blog/301-y-redirekt-polnostyu-zamenil-direktivu-host)
59
+ the `Host` directive in 2018, and Google
60
+ [never supported it](https://developers.google.com/search/docs/crawling-indexing/robots/robots_txt)
61
+ (it is not part of [RFC 9309](https://www.rfc-editor.org/rfc/rfc9309) either). Set it
62
+ to `True` to keep the old output.
63
+ - `ROBOTS_SITE_BY_REQUEST` looks the site up the same way as Django's sites framework:
64
+ case-insensitively, retrying without the port (`example.com:8000` matches
65
+ `example.com`).
66
+ - `robots.txt` is served as `text/plain; charset=utf-8`
67
+ ([RFC 9309](https://www.rfc-editor.org/rfc/rfc9309) requires UTF-8), with rules sorted
68
+ by robot and URLs by pattern, in a fixed number of queries.
69
+ - `Rule.comment`: an optional single-line note rendered as a `# ...` line above the
70
+ rule's group in `robots.txt`.
71
+ - URL patterns are percent-encoded on output: non-ASCII characters (e.g. Cyrillic),
72
+ whitespace and `#` become `%XX`, as
73
+ [Yandex requires](https://yandex.ru/support/webmaster/ru/controlling-robot/robots-txt).
74
+ Patterns are stored as entered, and raw and encoded forms of the same path count as
75
+ the same pattern when checking allowed/disallowed conflicts.
76
+ - [Clean-param](https://yandex.ru/support/webmaster/ru/robot-workings/clean-param)
77
+ directives (Yandex): see below.
78
+ - System checks for `ROBOTS_SITEMAP_URLS`: `robots.E001` if it is a string instead of
79
+ a list, `robots.W001` for URLs that are not absolute, `robots.W002` for non-ASCII
80
+ domains.
81
+ - Only the `ru` locale is shipped.
82
+
83
+ ## Installation
84
+
85
+ Requires Python 3.10+ and Django 4.2+ (tested with Django 4.2, 5.2, 6.0 and 6.1).
86
+ Projects on older Python or Django versions can stay on django-robots 6.1.
87
+
88
+ ```python
89
+ INSTALLED_APPS = [
90
+ "django.contrib.sites",
91
+ ...
92
+ "robots",
93
+ ]
94
+ ```
95
+
96
+ ```python
97
+ urlpatterns = [
98
+ re_path(r"^robots\.txt", include("robots.urls")),
99
+ ]
100
+ ```
101
+
102
+ Settings (`ROBOTS_SITEMAP_URLS`, `ROBOTS_USE_SITEMAP`, `ROBOTS_USE_HOST`,
103
+ `ROBOTS_CACHE_TIMEOUT`, `ROBOTS_SITE_BY_REQUEST`, `ROBOTS_USE_SCHEME_IN_HOST`,
104
+ `ROBOTS_SITEMAP_VIEW_NAME`) are the same as upstream, except for the changes listed
105
+ above.
106
+
107
+ ### Internationalized domains
108
+
109
+ Store non-ASCII domains in `Site.domain` (and in `ROBOTS_SITEMAP_URLS`) in Punycode,
110
+ e.g. `xn--d1aqf.xn--p1ai` instead of `дом.рф`:
111
+
112
+ - the domain is written to `Sitemap:` (and `Host:`) as stored, and
113
+ [Yandex requires](https://yandex.ru/support/webmaster/ru/controlling-robot/robots-txt)
114
+ Punycode there;
115
+ - with `ROBOTS_SITE_BY_REQUEST = True` the site is looked up by the request `Host`
116
+ header, which is always Punycode, so a site stored as `дом.рф` is not found and
117
+ `robots.txt` responds with an error.
118
+
119
+ ## Clean-param
120
+
121
+ `Clean-param` tells Yandex which URL parameters do not change the page content, so
122
+ `/catalog/?ref=vk` and `/catalog/?sid=1` are crawled and indexed as `/catalog/`. Other
123
+ search engines ignore it.
124
+
125
+ Add directives in the admin under *Clean-param directives* and attach them to sites:
126
+
127
+ | Parameters | Path | Output |
128
+ |---|---|---|
129
+ | `ref&sid` | `/catalog/` | `Clean-param: ref&sid /catalog/` |
130
+ | `sort` | *(empty)* | `Clean-param: sort` (whole site) |
131
+
132
+ The directive is cross-sectional, so it is rendered once per file, next to `Sitemap`,
133
+ not inside a `User-agent` group. Following the Yandex rules, parameter names are
134
+ case-sensitive, the path may contain only `A-Za-z0-9.-/*_` (a leading `/` is added
135
+ if missing, as for URL patterns), and the whole line is limited to 500 characters.
136
+ Unlike `Allow`/`Disallow`, the path is not percent-encoded, so pages with Cyrillic
137
+ paths (e.g. `/о-компании/`) can only be covered by a directive without a path.
138
+
139
+ ## Development
140
+
141
+ `tests/` holds a minimal Django project (SQLite) used both for the test suite and for
142
+ trying the app locally. It is not part of the distributed package. Run the commands
143
+ from the repository root.
144
+
145
+ ```bash
146
+ python -m venv .venv
147
+ . .venv/bin/activate
148
+ pip install -e .
149
+ export DJANGO_SETTINGS_MODULE=tests.settings
150
+
151
+ python -m django test tests # run the test suite
152
+
153
+ pip install "coverage[toml]" # test coverage, as in CI
154
+ python -m coverage run -m django test tests
155
+ python -m coverage report
156
+
157
+ python -m django migrate
158
+ python -m django createsuperuser
159
+ python -m django runserver # http://localhost:8000/admin/, http://localhost:8000/robots.txt
160
+
161
+ python -m django makemigrations robots --check --dry-run # after model changes
162
+ ```
163
+
164
+ The demo project uses `SITE_ID = 1` (`example.com`), so rules must be attached to that
165
+ site to appear in `/robots.txt`.
166
+
167
+ Linting and formatting use [ruff](https://docs.astral.sh/ruff/) via
168
+ [pre-commit](https://pre-commit.com/); neither is a dependency of the package.
169
+
170
+ ```bash
171
+ pip install pre-commit
172
+ pre-commit install # run the hooks on every commit
173
+ pre-commit run --all-files # run them on the whole repository
174
+ ```
175
+
@@ -0,0 +1,23 @@
1
+ robots/__init__.py,sha256=lZfKHv3FyvT128TfuNhk5czZNVgxSnT7PsflmdQ6ulU,87
2
+ robots/admin.py,sha256=KrL7gaKi9M-0XGjU-C6s1QcjNPmkbQpXKvwhQFTGets,1174
3
+ robots/apps.py,sha256=rVl-CjRkAqOs1js6q-D6IlnZUlo0w5X94qJoczrxznI,287
4
+ robots/checks.py,sha256=1JoBMJGVl447YE-KPpVnNUda3x_azkFJxMo1QoikSSw,1411
5
+ robots/forms.py,sha256=wqz_GD7g_n3n2XbxkDVNh-P-LjjLcqvUBaKYulTRy08,2341
6
+ robots/locale/ru/LC_MESSAGES/django.mo,sha256=ZGZHbOSqCVvf-luGkNsrqMyS8KCpsa3g4gKrZXjRZgE,7564
7
+ robots/locale/ru/LC_MESSAGES/django.po,sha256=mljy3_004mdjwS5NyKOWD9HvUs4xjlFejiPdR5GgwdI,8546
8
+ robots/migrations/0001_initial.py,sha256=-61NoEIujKMlW6Nsu3D_Sned-kjIQpnwUVcSr_Ez_mo,4142
9
+ robots/migrations/0002_alter_id_fields.py,sha256=uHED78W8Ol17OLAKjQ6qsN19TQG-wcZwJtkyj547cEw,693
10
+ robots/migrations/0003_rule_comment.py,sha256=x08veYyK6Vu7Stfp_LxbAgBf4ifm8pwe6K5qFA0ojC8,732
11
+ robots/migrations/0004_cleanparam.py,sha256=u0y4ADtrr5Dd54ZO0i3BtjPm1LY-0LNNyUfywcZCjC0,1868
12
+ robots/migrations/0005_alter_rule_disallowed.py,sha256=Kpna96r7I65jh5GuXiby_kLqk_XMZzajUmkL74i11d8,768
13
+ robots/migrations/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
14
+ robots/models.py,sha256=aWTeOZSp29RM4HXH7dGYv8d4bgSmLRnN1bOpatxKCUk,6951
15
+ robots/settings.py,sha256=SLIaTKZqUsSxDzKkpPgr048DqgYuV7c3Br7qv_F-zl0,790
16
+ robots/templates/robots/rule_list.html,sha256=qp42wvWI9LCVYsmBd14p6HC04govRfl8RUDZcdTzLxA,830
17
+ robots/urls.py,sha256=dEIQ_G64GdyCSM142XztOOX9pQ5gKOeLSgP27ISKZvI,136
18
+ robots/validators.py,sha256=KBhkCKNs6s596lV0zQ0nGRTsNoWkuA3eq0HJDyGe5nA,1700
19
+ robots/views.py,sha256=ytjWUxufw6EkgHCqfh97BqDwkv2xMr3LVNia_9SpijE,3980
20
+ django_robots_manager-6.2.1.dist-info/METADATA,sha256=j3HHfaoSC0nz2-2T3loMeQBHBglPsJFREVydYCs3M94,8279
21
+ django_robots_manager-6.2.1.dist-info/WHEEL,sha256=EGEvSphFYqXKs23-kQBeyNoJP1nrT8ZJKQoi5p5DYL8,88
22
+ django_robots_manager-6.2.1.dist-info/licenses/LICENSE.txt,sha256=IBDZ1PU5a4nJOc47Q2U-XMwHENfPB83G-vsFQoPjXBE,1555
23
+ django_robots_manager-6.2.1.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: poetry-core 2.4.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,29 @@
1
+ Copyright (c) 2008-, Jannis Leidel
2
+ Copyright (c) 2026-, Artem Fabrikov
3
+ All rights reserved.
4
+
5
+ Redistribution and use in source and binary forms, with or without
6
+ modification, are permitted provided that the following conditions are
7
+ met:
8
+
9
+ * Redistributions of source code must retain the above copyright
10
+ notice, this list of conditions and the following disclaimer.
11
+ * Redistributions in binary form must reproduce the above
12
+ copyright notice, this list of conditions and the following
13
+ disclaimer in the documentation and/or other materials provided
14
+ with the distribution.
15
+ * Neither the name of the author nor the names of other
16
+ contributors may be used to endorse or promote products derived
17
+ from this software without specific prior written permission.
18
+
19
+ THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
20
+ "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
21
+ LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR
22
+ A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT
23
+ OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
24
+ SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT
25
+ LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
26
+ DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
27
+ THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
28
+ (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
29
+ OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
robots/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ from importlib.metadata import version
2
+
3
+ __version__ = version("django-robots-manager")
robots/admin.py ADDED
@@ -0,0 +1,36 @@
1
+ from django.contrib import admin
2
+ from django.http import HttpRequest
3
+ from django.utils.translation import gettext_lazy as _
4
+
5
+ from robots.forms import RuleAdminForm
6
+ from robots.models import CleanParam, Rule, Url
7
+
8
+
9
+ class RuleAdmin(admin.ModelAdmin):
10
+ form = RuleAdminForm
11
+ fieldsets = (
12
+ (None, {"fields": ("robot", "sites", "comment")}),
13
+ (_("URL patterns"), {"fields": ("allowed", "disallowed")}),
14
+ (
15
+ _("Advanced options"),
16
+ {"classes": ("collapse",), "fields": ("crawl_delay",)},
17
+ ),
18
+ )
19
+ list_filter = ("sites",)
20
+ list_display = ("robot", "allowed_urls", "disallowed_urls")
21
+ search_fields = ("robot", "allowed__pattern", "disallowed__pattern")
22
+ filter_horizontal = ("sites", "allowed", "disallowed")
23
+
24
+ def get_queryset(self, request: HttpRequest):
25
+ return super().get_queryset(request).prefetch_related("allowed", "disallowed")
26
+
27
+
28
+ class CleanParamAdmin(admin.ModelAdmin):
29
+ list_display = ("parameters", "path")
30
+ list_filter = ("sites",)
31
+ filter_horizontal = ("sites",)
32
+
33
+
34
+ admin.site.register(Url)
35
+ admin.site.register(Rule, RuleAdmin)
36
+ admin.site.register(CleanParam, CleanParamAdmin)
robots/apps.py ADDED
@@ -0,0 +1,12 @@
1
+ from django.apps import AppConfig
2
+ from django.core import checks
3
+
4
+ from robots.checks import check_sitemap_urls
5
+
6
+
7
+ class RobotsConfig(AppConfig):
8
+ default_auto_field = "django.db.models.BigAutoField"
9
+ name = "robots"
10
+
11
+ def ready(self):
12
+ checks.register(check_sitemap_urls)
robots/checks.py ADDED
@@ -0,0 +1,39 @@
1
+ from urllib.parse import SplitResult, urlsplit
2
+
3
+ from django.core import checks
4
+
5
+ from robots import settings
6
+
7
+
8
+ def check_sitemap_urls(app_configs, **kwargs):
9
+ """Validate that ROBOTS_SITEMAP_URLS is a list of absolute ASCII URLs."""
10
+ sitemap_urls = settings.SITEMAP_URLS
11
+ if isinstance(sitemap_urls, str):
12
+ return [
13
+ checks.Error(
14
+ "ROBOTS_SITEMAP_URLS must be a list or tuple of URLs, not a string.",
15
+ hint="Wrap the URL in a list: ['https://example.com/sitemap.xml'].",
16
+ id="robots.E001",
17
+ )
18
+ ]
19
+
20
+ messages = []
21
+ for url in sitemap_urls:
22
+ parts: SplitResult = urlsplit(url)
23
+ if parts.scheme not in ("http", "https") or not parts.netloc:
24
+ messages.append(
25
+ checks.Warning(
26
+ f"ROBOTS_SITEMAP_URLS has a URL that is not absolute: {url!r}.",
27
+ hint="Use an absolute URL, e.g. 'https://example.com/sitemap.xml'.",
28
+ id="robots.W001",
29
+ )
30
+ )
31
+ elif not parts.netloc.isascii():
32
+ messages.append(
33
+ checks.Warning(
34
+ f"ROBOTS_SITEMAP_URLS contains a non-ASCII domain: {url!r}.",
35
+ hint="Write the domain in Punycode, e.g. 'xn--d1aqf.xn--p1ai'.",
36
+ id="robots.W002",
37
+ )
38
+ )
39
+ return messages
robots/forms.py ADDED
@@ -0,0 +1,64 @@
1
+ from django import forms
2
+ from django.contrib.sites.models import Site
3
+ from django.utils.translation import gettext_lazy as _
4
+
5
+ from robots.models import Rule, encode_pattern
6
+
7
+
8
+ class RuleAdminForm(forms.ModelForm):
9
+ class Meta:
10
+ model = Rule
11
+ fields = "__all__"
12
+
13
+ def clean(self):
14
+ if not self.cleaned_data.get("disallowed", False) and not self.cleaned_data.get(
15
+ "allowed", False
16
+ ):
17
+ raise forms.ValidationError(
18
+ _("Please specify at least one allowed or disallowed URL.")
19
+ )
20
+
21
+ allowed = self.cleaned_data.get("allowed")
22
+ disallowed = self.cleaned_data.get("disallowed")
23
+ if allowed and disallowed:
24
+ allowed_patterns = {
25
+ encode_pattern(pattern): pattern
26
+ for pattern in allowed.values_list("pattern", flat=True)
27
+ }
28
+ conflicts = allowed_patterns.keys() & {
29
+ encode_pattern(pattern)
30
+ for pattern in disallowed.values_list("pattern", flat=True)
31
+ }
32
+ if conflicts:
33
+ raise forms.ValidationError(
34
+ _(
35
+ "URL patterns cannot be both allowed and disallowed: "
36
+ "%(patterns)s."
37
+ ),
38
+ params={
39
+ "patterns": ", ".join(
40
+ sorted(allowed_patterns[key] for key in conflicts)
41
+ )
42
+ },
43
+ )
44
+
45
+ robot = self.cleaned_data.get("robot")
46
+ sites = self.cleaned_data.get("sites")
47
+ if robot and sites:
48
+ other_rules = Rule.objects.filter(robot__iexact=robot).exclude(
49
+ pk=self.instance.pk
50
+ )
51
+ duplicate_domains = (
52
+ Site.objects.filter(pk__in=sites, rule__in=other_rules)
53
+ .values_list("domain", flat=True)
54
+ .distinct()
55
+ )
56
+ if duplicate_domains:
57
+ raise forms.ValidationError(
58
+ _("A rule for robot %(robot)s already exists on sites: %(sites)s."),
59
+ params={
60
+ "robot": robot,
61
+ "sites": ", ".join(sorted(duplicate_domains)),
62
+ },
63
+ )
64
+ return self.cleaned_data
Binary file
@@ -0,0 +1,188 @@
1
+ # SOME DESCRIPTIVE TITLE.
2
+ # Copyright (C) YEAR THE PACKAGE'S COPYRIGHT HOLDER
3
+ # This file is distributed under the same license as the PACKAGE package.
4
+ #
5
+ # Translators:
6
+ # alekam <alex@kamedov.ru>, 2011
7
+ # Jannis Leidel <jannis@leidel.info>, 2011
8
+ msgid ""
9
+ msgstr ""
10
+ "Project-Id-Version: django-robots\n"
11
+ "Report-Msgid-Bugs-To: \n"
12
+ "POT-Creation-Date: 2011-02-08 12:09+0100\n"
13
+ "PO-Revision-Date: 2013-11-20 09:28+0000\n"
14
+ "Last-Translator: Jannis Leidel <jannis@leidel.info>\n"
15
+ "Language-Team: Russian (http://www.transifex.com/projects/p/django-robots/language/ru/)\n"
16
+ "MIME-Version: 1.0\n"
17
+ "Content-Type: text/plain; charset=UTF-8\n"
18
+ "Content-Transfer-Encoding: 8bit\n"
19
+ "Language: ru\n"
20
+ "Plural-Forms: nplurals=3; plural=(n%10==1 && n%100!=11 ? 0 : n%10>=2 && n%10<=4 && (n%100<10 || n%100>=20) ? 1 : 2);\n"
21
+
22
+ #: admin.py:11
23
+ msgid "URL patterns"
24
+ msgstr "Шаблоны URL"
25
+
26
+ #: admin.py:12
27
+ msgid "Advanced options"
28
+ msgstr "Расширенные настройки"
29
+
30
+ #: forms.py:18
31
+ msgid "Please specify at least one allowed or disallowed URL."
32
+ msgstr "Укажите как минимум один URL-адрес (неважно разрешенный или нет)"
33
+
34
+ #: forms.py:29
35
+ #, python-format
36
+ msgid "URL patterns cannot be both allowed and disallowed: %(patterns)s."
37
+ msgstr "Шаблоны URL не могут быть одновременно разрешёнными и запрещёнными: %(patterns)s."
38
+
39
+ #: forms.py:48
40
+ #, python-format
41
+ msgid "A rule for robot %(robot)s already exists on sites: %(sites)s."
42
+ msgstr "Правило для робота %(robot)s уже существует на сайтах: %(sites)s."
43
+
44
+ #: models.py:11
45
+ msgid "pattern"
46
+ msgstr "шаблон"
47
+
48
+ #: models.py:12
49
+ msgid ""
50
+ "Case-sensitive. A missing trailing slash does also match to files which "
51
+ "start with the name of the pattern, e.g., '/admin' matches /admin.html too. "
52
+ "Some major search engines allow an asterisk (*) as a wildcard and a dollar "
53
+ "sign ($) to match the end of the URL, e.g., '/*.jpg$'."
54
+ msgstr "Внимание: учитывается регистр!<br />Если в конце пропущен слэш, то под шаблон попадут все файлы, путь к которым начинается с таких же символов. Например: под шаблон \"/admin\" так же попадет \"/admin.html\".<br />Часть поисковых систем понимают звездочку (*) как произвольное количество любых символов и знак доллара ($) как символ конца URL. Например: \"/*.jpg$\""
55
+
56
+ #: models.py:19 models.py:20
57
+ msgid "url"
58
+ msgstr "URL-адрес"
59
+
60
+ #: models.py:37
61
+ msgid "robot"
62
+ msgstr "робот"
63
+
64
+ #: models.py:38
65
+ msgid ""
66
+ "This should be a user agent string like 'Googlebot'. Enter an asterisk (*) "
67
+ "for all user agents. For a full list look at the <a target=_blank "
68
+ "href='http://www.robotstxt.org/db.html'> database of Web Robots</a>."
69
+ msgstr "Название робота (User agent). Введите звездочку (*) для применения правил ко всем роботов. Полный список можно посмотреть в <a target=_blank href='http://www.robotstxt.org/db.html'>базе данных веб-ботов</a>."
70
+
71
+ #: models.py:46 models.py:83
72
+ msgid "allowed"
73
+ msgstr "разрешенные URL"
74
+
75
+ #: models.py:47
76
+ msgid "The URLs which are allowed to be accessed by bots."
77
+ msgstr "URL адреса разрешенные для индексации поисковыми роботами."
78
+
79
+ #: models.py:51 models.py:87
80
+ msgid "disallowed"
81
+ msgstr "запрещенные URL"
82
+
83
+ #: models.py:98
84
+ msgid ""
85
+ "The URLs which are not allowed to be accessed by bots. robots.txt is public "
86
+ "and only asks well-behaved crawlers to stay away: it does not protect "
87
+ "content, and listing a URL here makes it visible to everyone."
88
+ msgstr ""
89
+ "URL-адреса, запрещённые для обхода поисковыми роботами. Файл robots.txt "
90
+ "публичный и лишь просит добросовестных роботов не заходить на эти адреса: "
91
+ "он не защищает содержимое, а перечисленные здесь URL увидит любой."
92
+
93
+ #: models.py:55
94
+ msgid "sites"
95
+ msgstr ""
96
+
97
+ #: models.py:57
98
+ msgid "crawl delay"
99
+ msgstr "частота обновления"
100
+
101
+ #: models.py:59
102
+ msgid ""
103
+ "Between 0.1 and 99.0. This field is supported by some search engines and "
104
+ "defines the delay between successive crawler accesses in seconds. If the "
105
+ "crawler rate is a problem for your server, you can set the delay up to 5 or "
106
+ "10 or a comfortable value for your server, but it's suggested to start with "
107
+ "small values (0.5-1), and increase as needed to an acceptable value for your"
108
+ " server. Larger delay values add more delay between successive crawl "
109
+ "accesses and decrease the maximum crawl rate to your web server."
110
+ msgstr "Введите значение между 0.1 и 99.0. Этот параметр поддерживается некоторыми поисковыми системами и определяет задержку в секундах до следующего запроса робота. Если робот обнаруживает проблемы на вашем сервер, вы можите установить задержку от 5, 10 или более подходяще значение для вашего сервера, но лучше начинать с небольших значений (0.5-1), и постепенно увеличивать до достижения оптимальных значений для вашего сервера. Большие значения увеличивают время выгрузки роботом вашего сайта и могут отрицательно влиять на ваш рейтинг в данной системе."
111
+
112
+ #: models.py:75
113
+ msgid "rule"
114
+ msgstr "правило индексации"
115
+
116
+ #: models.py:76
117
+ msgid "rules"
118
+ msgstr "правила индексации"
119
+
120
+ #: models.py:82 models.py:86
121
+ msgid "and"
122
+ msgstr "и"
123
+
124
+ #: models.py:106
125
+ msgid "comment"
126
+ msgstr "комментарий"
127
+
128
+ #: models.py:111
129
+ msgid ""
130
+ "Optional note shown as a '# ...' line above this rule in robots.txt. "
131
+ "robots.txt is public: do not put anything confidential here."
132
+ msgstr ""
133
+ "Необязательная заметка, которая выводится строкой «# ...» перед правилом в "
134
+ "robots.txt. Файл robots.txt публичный: не пишите сюда ничего "
135
+ "конфиденциального."
136
+
137
+ #: validators.py:18
138
+ msgid "Use a single line without line breaks or control characters."
139
+ msgstr "Используйте одну строку без переносов и управляющих символов."
140
+
141
+ #: models.py:162
142
+ msgid "parameters"
143
+ msgstr "параметры"
144
+
145
+ #: models.py:165
146
+ msgid "Parameter names separated by &, e.g. ref&sid. Case-sensitive."
147
+ msgstr "Имена параметров через &, например ref&sid. Регистр учитывается."
148
+
149
+ #: models.py:168
150
+ msgid "path"
151
+ msgstr "путь"
152
+
153
+ #: models.py:173
154
+ msgid ""
155
+ "Optional path prefix, e.g. '/catalog/'. Leave empty to apply to the whole "
156
+ "site."
157
+ msgstr ""
158
+ "Необязательный префикс пути, например '/catalog/'. Оставьте пустым, чтобы "
159
+ "директива действовала на весь сайт."
160
+
161
+ #: models.py:180
162
+ msgid "Clean-param directive"
163
+ msgstr "директива Clean-param"
164
+
165
+ #: models.py:181
166
+ msgid "Clean-param directives"
167
+ msgstr "директивы Clean-param"
168
+
169
+ #: models.py:196
170
+ #, python-format
171
+ msgid "The Clean-param directive must not exceed %(limit)d characters."
172
+ msgstr "Директива Clean-param не должна быть длиннее %(limit)d символов."
173
+
174
+ #: validators.py:33
175
+ msgid ""
176
+ "Enter parameter names separated by \"&\", without spaces, \"=\", \"#\" or "
177
+ "non-ASCII characters."
178
+ msgstr ""
179
+ "Введите имена параметров через «&», без пробелов, «=», «#» и символов вне "
180
+ "ASCII."
181
+
182
+ #: validators.py:45
183
+ msgid ""
184
+ "The path may contain only Latin letters, digits and the characters \".\", "
185
+ "\"-\", \"/\", \"*\" and \"_\"."
186
+ msgstr ""
187
+ "Путь может содержать только латинские буквы, цифры и символы «.», «-», «/», "
188
+ "«*» и «_»."
@@ -0,0 +1,103 @@
1
+ from django.db import migrations, models
2
+
3
+
4
+ class Migration(migrations.Migration):
5
+ dependencies = [
6
+ ("sites", "0001_initial"),
7
+ ]
8
+
9
+ operations = [
10
+ migrations.CreateModel(
11
+ name="Rule",
12
+ fields=[
13
+ (
14
+ "id",
15
+ models.AutoField(
16
+ auto_created=True,
17
+ serialize=False,
18
+ verbose_name="ID",
19
+ primary_key=True,
20
+ ),
21
+ ),
22
+ (
23
+ "robot",
24
+ models.CharField(
25
+ max_length=255,
26
+ help_text="This should be a user agent string like 'Googlebot'. Enter an asterisk (*) for all user agents. For a full list look at the <a target=_blank href='http://www.robotstxt.org/db.html'> database of Web Robots</a>.",
27
+ verbose_name="robot",
28
+ ),
29
+ ),
30
+ (
31
+ "crawl_delay",
32
+ models.DecimalField(
33
+ blank=True,
34
+ help_text="Between 0.1 and 99.0. This field is supported by some search engines and defines the delay between successive crawler accesses in seconds. If the crawler rate is a problem for your server, you can set the delay up to 5 or 10 or a comfortable value for your server, but it's suggested to start with small values (0.5-1), and increase as needed to an acceptable value for your server. Larger delay values add more delay between successive crawl accesses and decrease the maximum crawl rate to your web server.",
35
+ verbose_name="crawl delay",
36
+ decimal_places=1,
37
+ max_digits=3,
38
+ null=True,
39
+ ),
40
+ ),
41
+ (
42
+ "sites",
43
+ models.ManyToManyField(to="sites.Site", verbose_name="sites"),
44
+ ),
45
+ ],
46
+ options={
47
+ "verbose_name_plural": "rules",
48
+ "verbose_name": "rule",
49
+ },
50
+ bases=(models.Model,),
51
+ ),
52
+ migrations.CreateModel(
53
+ name="Url",
54
+ fields=[
55
+ (
56
+ "id",
57
+ models.AutoField(
58
+ auto_created=True,
59
+ serialize=False,
60
+ verbose_name="ID",
61
+ primary_key=True,
62
+ ),
63
+ ),
64
+ (
65
+ "pattern",
66
+ models.CharField(
67
+ max_length=255,
68
+ help_text="Case-sensitive. A missing trailing slash does also match to files which start with the name of the pattern, e.g., '/admin' matches /admin.html too. Some major search engines allow an asterisk (*) as a wildcard and a dollar sign ($) to match the end of the URL, e.g., '/*.jpg$'.",
69
+ verbose_name="pattern",
70
+ ),
71
+ ),
72
+ ],
73
+ options={
74
+ "verbose_name_plural": "url",
75
+ "verbose_name": "url",
76
+ },
77
+ bases=(models.Model,),
78
+ ),
79
+ migrations.AddField(
80
+ model_name="rule",
81
+ name="disallowed",
82
+ field=models.ManyToManyField(
83
+ to="robots.Url",
84
+ blank=True,
85
+ related_name="disallowed",
86
+ verbose_name="disallowed",
87
+ help_text="The URLs which are not allowed to be accessed by bots.",
88
+ ),
89
+ preserve_default=True,
90
+ ),
91
+ migrations.AddField(
92
+ model_name="rule",
93
+ name="allowed",
94
+ field=models.ManyToManyField(
95
+ to="robots.Url",
96
+ blank=True,
97
+ related_name="allowed",
98
+ verbose_name="allowed",
99
+ help_text="The URLs which are allowed to be accessed by bots.",
100
+ ),
101
+ preserve_default=True,
102
+ ),
103
+ ]
@@ -0,0 +1,26 @@
1
+ # Generated by Django 3.2 on 2021-09-23 12:14
2
+
3
+ from django.db import migrations, models
4
+
5
+
6
+ class Migration(migrations.Migration):
7
+ dependencies = [
8
+ ("robots", "0001_initial"),
9
+ ]
10
+
11
+ operations = [
12
+ migrations.AlterField(
13
+ model_name="rule",
14
+ name="id",
15
+ field=models.BigAutoField(
16
+ auto_created=True, primary_key=True, serialize=False, verbose_name="ID"
17
+ ),
18
+ ),
19
+ migrations.AlterField(
20
+ model_name="url",
21
+ name="id",
22
+ field=models.BigAutoField(
23
+ auto_created=True, primary_key=True, serialize=False, verbose_name="ID"
24
+ ),
25
+ ),
26
+ ]
@@ -0,0 +1,25 @@
1
+ # Generated by Django 6.1.1 on 2026-10-06 20:29
2
+
3
+ from django.db import migrations, models
4
+
5
+ import robots.validators
6
+
7
+
8
+ class Migration(migrations.Migration):
9
+ dependencies = [
10
+ ("robots", "0002_alter_id_fields"),
11
+ ]
12
+
13
+ operations = [
14
+ migrations.AddField(
15
+ model_name="rule",
16
+ name="comment",
17
+ field=models.CharField(
18
+ blank=True,
19
+ help_text="Optional note shown as a '# ...' line above this rule in robots.txt. robots.txt is public: do not put anything confidential here.",
20
+ max_length=255,
21
+ validators=[robots.validators.validate_single_line],
22
+ verbose_name="comment",
23
+ ),
24
+ ),
25
+ ]
@@ -0,0 +1,56 @@
1
+ # Generated by Django 6.1.1 on 2026-10-06 21:54
2
+
3
+ from django.db import migrations, models
4
+
5
+ import robots.validators
6
+
7
+
8
+ class Migration(migrations.Migration):
9
+ dependencies = [
10
+ ("robots", "0003_rule_comment"),
11
+ ("sites", "0002_alter_domain_unique"),
12
+ ]
13
+
14
+ operations = [
15
+ migrations.CreateModel(
16
+ name="CleanParam",
17
+ fields=[
18
+ (
19
+ "id",
20
+ models.BigAutoField(
21
+ auto_created=True,
22
+ primary_key=True,
23
+ serialize=False,
24
+ verbose_name="ID",
25
+ ),
26
+ ),
27
+ (
28
+ "parameters",
29
+ models.CharField(
30
+ help_text="Parameter names separated by &, e.g. ref&sid. Case-sensitive.",
31
+ max_length=255,
32
+ validators=[robots.validators.validate_clean_param_parameters],
33
+ verbose_name="parameters",
34
+ ),
35
+ ),
36
+ (
37
+ "path",
38
+ models.CharField(
39
+ blank=True,
40
+ help_text="Optional path prefix, e.g. '/catalog/'. Leave empty to apply to the whole site.",
41
+ max_length=255,
42
+ validators=[robots.validators.validate_clean_param_path],
43
+ verbose_name="path",
44
+ ),
45
+ ),
46
+ (
47
+ "sites",
48
+ models.ManyToManyField(to="sites.site", verbose_name="sites"),
49
+ ),
50
+ ],
51
+ options={
52
+ "verbose_name": "Clean-param directive",
53
+ "verbose_name_plural": "Clean-param directives",
54
+ },
55
+ ),
56
+ ]
@@ -0,0 +1,23 @@
1
+ # Generated by Django 6.1.1 on 2026-10-06 22:43
2
+
3
+ from django.db import migrations, models
4
+
5
+
6
+ class Migration(migrations.Migration):
7
+ dependencies = [
8
+ ("robots", "0004_cleanparam"),
9
+ ]
10
+
11
+ operations = [
12
+ migrations.AlterField(
13
+ model_name="rule",
14
+ name="disallowed",
15
+ field=models.ManyToManyField(
16
+ blank=True,
17
+ help_text="The URLs which are not allowed to be accessed by bots. robots.txt is public and only asks well-behaved crawlers to stay away: it does not protect content, and listing a URL here makes it visible to everyone.",
18
+ related_name="disallowed",
19
+ to="robots.url",
20
+ verbose_name="disallowed",
21
+ ),
22
+ ),
23
+ ]
File without changes
robots/models.py ADDED
@@ -0,0 +1,216 @@
1
+ from urllib.parse import quote
2
+
3
+ from django.contrib import admin
4
+ from django.contrib.sites.models import Site
5
+ from django.core.exceptions import ValidationError
6
+ from django.db import models
7
+ from django.utils.text import get_text_list
8
+ from django.utils.translation import gettext_lazy as _
9
+
10
+ from robots.validators import (
11
+ validate_clean_param_parameters,
12
+ validate_clean_param_path,
13
+ validate_single_line,
14
+ )
15
+
16
+ # Printable ASCII except "#", which starts a comment in robots.txt.
17
+ PATTERN_SAFE_CHARS = "".join(
18
+ chr(code) for code in range(0x21, 0x7F) if chr(code) != "#"
19
+ )
20
+
21
+
22
+ def encode_pattern(pattern):
23
+ """Percent-encode non-ASCII characters, whitespace and "#" for robots.txt."""
24
+ return quote(pattern, safe=PATTERN_SAFE_CHARS)
25
+
26
+
27
+ class Url(models.Model):
28
+ """
29
+ Defines a URL pattern for use with a robot exclusion rule. It's
30
+ case-sensitive and exact, e.g., "/admin" and "/admin/" are different URLs.
31
+ """
32
+
33
+ pattern = models.CharField(
34
+ _("pattern"),
35
+ max_length=255,
36
+ help_text=_(
37
+ "Case-sensitive. A missing trailing slash does al"
38
+ "so match to files which start with the name of "
39
+ "the pattern, e.g., '/admin' matches /admin.html "
40
+ "too. Some major search engines allow an asterisk"
41
+ " (*) as a wildcard and a dollar sign ($) to "
42
+ "match the end of the URL, e.g., '/*.jpg$'."
43
+ ),
44
+ )
45
+
46
+ class Meta:
47
+ verbose_name = _("url")
48
+ verbose_name_plural = _("url")
49
+
50
+ def __str__(self):
51
+ return self.pattern
52
+
53
+ @property
54
+ def encoded_pattern(self):
55
+ return encode_pattern(self.pattern)
56
+
57
+ def save(self, *args, **kwargs):
58
+ if not self.pattern.startswith("/"):
59
+ self.pattern = "/" + self.pattern
60
+ super().save(*args, **kwargs)
61
+
62
+
63
+ class Rule(models.Model):
64
+ """
65
+ Defines an abstract rule which is used to respond to crawling web robots,
66
+ using the robot exclusion standard, a.k.a. robots.txt. It allows or
67
+ disallows the robot identified by its user agent to access the given URLs.
68
+ The Site contrib app is used to enable multiple robots.txt per instance.
69
+ """
70
+
71
+ robot = models.CharField(
72
+ _("robot"),
73
+ max_length=255,
74
+ help_text=_(
75
+ "This should be a user agent string like "
76
+ "'Googlebot'. Enter an asterisk (*) for all "
77
+ "user agents. For a full list look at the "
78
+ "<a target=_blank href='"
79
+ "http://www.robotstxt.org/db.html"
80
+ "'> database of Web Robots</a>."
81
+ ),
82
+ )
83
+
84
+ allowed = models.ManyToManyField(
85
+ Url,
86
+ blank=True,
87
+ related_name="allowed",
88
+ verbose_name=_("allowed"),
89
+ help_text=_("The URLs which are allowed to be accessed by bots."),
90
+ )
91
+
92
+ disallowed = models.ManyToManyField(
93
+ Url,
94
+ blank=True,
95
+ related_name="disallowed",
96
+ verbose_name=_("disallowed"),
97
+ help_text=_(
98
+ "The URLs which are not allowed to be accessed by bots. robots.txt is "
99
+ "public and only asks well-behaved crawlers to stay away: it does not "
100
+ "protect content, and listing a URL here makes it visible to everyone."
101
+ ),
102
+ )
103
+ sites = models.ManyToManyField(Site, verbose_name=_("sites"))
104
+
105
+ crawl_delay = models.DecimalField(
106
+ _("crawl delay"),
107
+ blank=True,
108
+ null=True,
109
+ max_digits=3,
110
+ decimal_places=1,
111
+ help_text=_(
112
+ "Between 0.1 and 99.0. This field is "
113
+ "supported by some search engines and "
114
+ "defines the delay between successive "
115
+ "crawler accesses in seconds. If the "
116
+ "crawler rate is a problem for your "
117
+ "server, you can set the delay up to 5 "
118
+ "or 10 or a comfortable value for your "
119
+ "server, but it's suggested to start "
120
+ "with small values (0.5-1), and "
121
+ "increase as needed to an acceptable "
122
+ "value for your server. Larger delay "
123
+ "values add more delay between "
124
+ "successive crawl accesses and "
125
+ "decrease the maximum crawl rate to "
126
+ "your web server."
127
+ ),
128
+ )
129
+
130
+ comment = models.CharField(
131
+ _("comment"),
132
+ max_length=255,
133
+ blank=True,
134
+ validators=[validate_single_line],
135
+ help_text=_(
136
+ "Optional note shown as a '# ...' line above this rule in robots.txt. "
137
+ "robots.txt is public: do not put anything confidential here."
138
+ ),
139
+ )
140
+
141
+ class Meta:
142
+ verbose_name = _("rule")
143
+ verbose_name_plural = _("rules")
144
+
145
+ def __str__(self):
146
+ return self.robot
147
+
148
+ @admin.display(description=_("allowed"))
149
+ def allowed_urls(self):
150
+ return get_text_list(list(self.allowed.all()), _("and"))
151
+
152
+ @admin.display(description=_("disallowed"))
153
+ def disallowed_urls(self):
154
+ return get_text_list(list(self.disallowed.all()), _("and"))
155
+
156
+
157
+ class CleanParam(models.Model):
158
+ """
159
+ Defines a Yandex Clean-param directive: URL parameters that do not change
160
+ the page content and should be ignored under the given path prefix.
161
+ """
162
+
163
+ MAX_DIRECTIVE_LENGTH = 500
164
+
165
+ parameters = models.CharField(
166
+ _("parameters"),
167
+ max_length=255,
168
+ validators=[validate_clean_param_parameters],
169
+ help_text=_("Parameter names separated by &, e.g. ref&sid. Case-sensitive."),
170
+ )
171
+ path = models.CharField(
172
+ _("path"),
173
+ max_length=255,
174
+ blank=True,
175
+ validators=[validate_clean_param_path],
176
+ help_text=_(
177
+ "Optional path prefix, e.g. '/catalog/'. Leave empty to apply to the "
178
+ "whole site."
179
+ ),
180
+ )
181
+ sites = models.ManyToManyField(Site, verbose_name=_("sites"))
182
+
183
+ class Meta:
184
+ verbose_name = _("Clean-param directive")
185
+ verbose_name_plural = _("Clean-param directives")
186
+
187
+ def __str__(self):
188
+ return self.directive
189
+
190
+ @property
191
+ def directive(self):
192
+ if self.path:
193
+ return f"Clean-param: {self.parameters} {self.path}"
194
+ return f"Clean-param: {self.parameters}"
195
+
196
+ def add_leading_slash(self):
197
+ """Prefix a non-empty path with "/" unless it starts with "/" or "*"."""
198
+ if self.path and not self.path.startswith(("/", "*")):
199
+ self.path = "/" + self.path
200
+
201
+ def clean_fields(self, exclude=None):
202
+ self.add_leading_slash()
203
+ super().clean_fields(exclude=exclude)
204
+
205
+ def clean(self):
206
+ super().clean()
207
+ if len(self.directive) > self.MAX_DIRECTIVE_LENGTH:
208
+ raise ValidationError(
209
+ _("The Clean-param directive must not exceed %(limit)d characters."),
210
+ params={"limit": self.MAX_DIRECTIVE_LENGTH},
211
+ code="max_length",
212
+ )
213
+
214
+ def save(self, *args, **kwargs):
215
+ self.add_leading_slash()
216
+ super().save(*args, **kwargs)
robots/settings.py ADDED
@@ -0,0 +1,24 @@
1
+ import sys
2
+ from typing import ClassVar
3
+
4
+
5
+ class Settings:
6
+ defaults: ClassVar[dict] = {
7
+ #: A list of one or more sitemaps to inform robots about:
8
+ "SITEMAP_URLS": ("ROBOTS_SITEMAP_URLS", []),
9
+ "USE_SITEMAP": ("ROBOTS_USE_SITEMAP", True),
10
+ "USE_HOST": ("ROBOTS_USE_HOST", False),
11
+ "CACHE_TIMEOUT": ("ROBOTS_CACHE_TIMEOUT", None),
12
+ "SITE_BY_REQUEST": ("ROBOTS_SITE_BY_REQUEST", False),
13
+ "USE_SCHEME_IN_HOST": ("ROBOTS_USE_SCHEME_IN_HOST", False),
14
+ "SITEMAP_VIEW_NAME": ("ROBOTS_SITEMAP_VIEW_NAME", False),
15
+ }
16
+
17
+ def __getattr__(self, attribute):
18
+ from django.conf import settings
19
+
20
+ if attribute in self.defaults:
21
+ return getattr(settings, *self.defaults[attribute])
22
+
23
+
24
+ sys.modules[__name__] = Settings()
@@ -0,0 +1,13 @@
1
+ {% load l10n %}{% if rules %}{% for rule in rules %}{% ifchanged rule.robot %}{% if not forloop.first %}
2
+ {% endif %}{% if rule.comment %}# {{ rule.comment|safe }}
3
+ {% endif %}User-agent: {{ rule.robot }}{% endifchanged %}
4
+ {% for url in rule.allowed.all %}Allow: {{ url.encoded_pattern|safe }}
5
+ {% endfor %}{% for url in rule.disallowed.all %}Disallow: {{ url.encoded_pattern|safe }}
6
+ {% endfor %}{% if rule.crawl_delay %}Crawl-delay: {% localize off %}{{ rule.crawl_delay|floatformat:'0' }}{% endlocalize %}
7
+ {% endif %}{% endfor %}{% else %}User-agent: *
8
+ Disallow:
9
+ {% endif %}{% if host or clean_params or sitemap_urls %}
10
+ {% if host %}Host: {{ host }}
11
+ {% endif %}{% for clean_param in clean_params %}{{ clean_param.directive|safe }}
12
+ {% endfor %}{% for sitemap_url in sitemap_urls %}Sitemap: {{ sitemap_url }}
13
+ {% endfor %}{% endif %}
robots/urls.py ADDED
@@ -0,0 +1,7 @@
1
+ from django.urls import path
2
+
3
+ from robots.views import rules_list
4
+
5
+ urlpatterns = [
6
+ path("", rules_list, name="robots_rule_list"),
7
+ ]
robots/validators.py ADDED
@@ -0,0 +1,49 @@
1
+ import re
2
+ import unicodedata
3
+
4
+ from django.core.exceptions import ValidationError
5
+ from django.utils.translation import gettext_lazy as _
6
+
7
+ # Characters Yandex allows in the Clean-param path prefix.
8
+ CLEAN_PARAM_PATH_RE = re.compile(r"[A-Za-z0-9.\-/*_]+")
9
+
10
+ # Control characters and Unicode line/paragraph separators. Tab is allowed: RFC 9309
11
+ # permits whitespace inside comments.
12
+ FORBIDDEN_CATEGORIES = {"Cc", "Zl", "Zp"}
13
+
14
+
15
+ def validate_single_line(value):
16
+ """Reject line breaks and other control characters except tab."""
17
+ if any(
18
+ char != "\t" and unicodedata.category(char) in FORBIDDEN_CATEGORIES
19
+ for char in value
20
+ ):
21
+ raise ValidationError(
22
+ _("Use a single line without line breaks or control characters."),
23
+ code="invalid",
24
+ )
25
+
26
+
27
+ def validate_clean_param_parameters(value):
28
+ """Require "&"-separated parameter names of printable ASCII without "=" and "#"."""
29
+ for name in value.split("&"):
30
+ if not name or any(not "!" <= char <= "~" or char in "=#" for char in name):
31
+ raise ValidationError(
32
+ _(
33
+ 'Enter parameter names separated by "&", without spaces, '
34
+ '"=", "#" or non-ASCII characters.'
35
+ ),
36
+ code="invalid",
37
+ )
38
+
39
+
40
+ def validate_clean_param_path(value):
41
+ """Allow only the characters Yandex accepts in a Clean-param path prefix."""
42
+ if not CLEAN_PARAM_PATH_RE.fullmatch(value):
43
+ raise ValidationError(
44
+ _(
45
+ "The path may contain only Latin letters, digits and the characters "
46
+ '".", "-", "/", "*" and "_".'
47
+ ),
48
+ code="invalid",
49
+ )
robots/views.py ADDED
@@ -0,0 +1,112 @@
1
+ from django.contrib.sitemaps import views as sitemap_views
2
+ from django.contrib.sites.models import Site
3
+ from django.db.models import Prefetch
4
+ from django.http import HttpRequest
5
+ from django.http.request import split_domain_port
6
+ from django.urls import NoReverseMatch, reverse
7
+ from django.views.decorators.cache import cache_page
8
+ from django.views.generic import ListView
9
+
10
+ from robots import settings
11
+ from robots.models import CleanParam, Rule, Url
12
+
13
+
14
+ class RuleList(ListView):
15
+ """
16
+ Returns a generated robots.txt file with correct mimetype (text/plain),
17
+ status code (200 or 404), sitemap url (automatically).
18
+ """
19
+
20
+ model = Rule
21
+ context_object_name = "rules"
22
+ cache_timeout = settings.CACHE_TIMEOUT
23
+
24
+ def get_current_site(self, request: HttpRequest):
25
+ if settings.SITE_BY_REQUEST:
26
+ host = request.get_host()
27
+ try:
28
+ return Site.objects.get(domain__iexact=host)
29
+ except Site.DoesNotExist:
30
+ domain, _port = split_domain_port(host)
31
+ return Site.objects.get(domain__iexact=domain)
32
+ else:
33
+ return Site.objects.get_current()
34
+
35
+ def reverse_sitemap_url(self):
36
+ try:
37
+ if settings.SITEMAP_VIEW_NAME:
38
+ return reverse(settings.SITEMAP_VIEW_NAME)
39
+ else:
40
+ return reverse(sitemap_views.index)
41
+ except NoReverseMatch:
42
+ try:
43
+ return reverse(sitemap_views.sitemap)
44
+ except NoReverseMatch:
45
+ pass
46
+
47
+ def get_domain(self):
48
+ scheme = self.request.is_secure() and "https" or "http"
49
+ if not self.current_site.domain.startswith(("http", "https")):
50
+ return f"{scheme}://{self.current_site.domain}"
51
+ return self.current_site.domain
52
+
53
+ def get_sitemap_urls(self):
54
+ sitemap_urls = list(settings.SITEMAP_URLS)
55
+
56
+ if not sitemap_urls and settings.USE_SITEMAP:
57
+ sitemap_url = self.reverse_sitemap_url()
58
+
59
+ if sitemap_url is not None:
60
+ if not sitemap_url.startswith(("http", "https")):
61
+ sitemap_url = f"{self.get_domain()}{sitemap_url}"
62
+ if sitemap_url not in sitemap_urls:
63
+ sitemap_urls.append(sitemap_url)
64
+
65
+ return sitemap_urls
66
+
67
+ def get_queryset(self):
68
+ urls = Url.objects.order_by("pattern")
69
+ return (
70
+ Rule.objects.filter(sites=self.current_site)
71
+ .order_by("robot")
72
+ .prefetch_related(
73
+ Prefetch("allowed", queryset=urls),
74
+ Prefetch("disallowed", queryset=urls),
75
+ )
76
+ )
77
+
78
+ def get_context_data(self, **kwargs):
79
+ context = super().get_context_data(**kwargs)
80
+ context["sitemap_urls"] = self.get_sitemap_urls()
81
+ context["clean_params"] = CleanParam.objects.filter(
82
+ sites=self.current_site
83
+ ).order_by("path", "parameters")
84
+ if settings.USE_HOST:
85
+ if settings.USE_SCHEME_IN_HOST:
86
+ context["host"] = self.get_domain()
87
+ else:
88
+ context["host"] = self.current_site.domain
89
+ else:
90
+ context["host"] = None
91
+ return context
92
+
93
+ def render_to_response(self, context, **kwargs):
94
+ return super().render_to_response(
95
+ context, content_type="text/plain; charset=utf-8", **kwargs
96
+ )
97
+
98
+ def get_cache_timeout(self):
99
+ return self.cache_timeout
100
+
101
+ def dispatch(self, request: HttpRequest, *args, **kwargs):
102
+ cache_timeout = self.get_cache_timeout()
103
+ self.current_site = self.get_current_site(request)
104
+ super_dispatch = super().dispatch
105
+ if not cache_timeout:
106
+ return super_dispatch(request, *args, **kwargs)
107
+ key_prefix = self.current_site.domain
108
+ cache_decorator = cache_page(cache_timeout, key_prefix=key_prefix)
109
+ return cache_decorator(super_dispatch)(request, *args, **kwargs)
110
+
111
+
112
+ rules_list = RuleList.as_view()