password-finder 2.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 District5
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,17 @@
1
+ # Keep the source distribution in step with the wheel: ship the package, the
2
+ # license, the readme, and the security policy -- but not tests, fixtures, CI
3
+ # config, or build artefacts. (Tests are dev-only and the email fixtures contain
4
+ # password-shaped test data that has no place in a distributed artefact.)
5
+
6
+ include LICENSE
7
+ include README.md
8
+ include SECURITY.md
9
+ include pyproject.toml
10
+
11
+ prune tests
12
+ prune .github
13
+ prune dist
14
+ prune build
15
+
16
+ global-exclude *.py[cod]
17
+ global-exclude __pycache__
@@ -0,0 +1,208 @@
1
+ Metadata-Version: 2.4
2
+ Name: password-finder
3
+ Version: 2.1.1
4
+ Summary: Extract passwords from free-form text such as email bodies (e.g. a password for an encrypted attachment sent in a separate message).
5
+ Author: District5
6
+ License: MIT
7
+ Keywords: password,extract,parse,email,text,regex
8
+ Classifier: License :: OSI Approved :: MIT License
9
+ Classifier: Programming Language :: Python :: 3
10
+ Classifier: Topic :: Text Processing
11
+ Requires-Python: >=3.9
12
+ Description-Content-Type: text/markdown
13
+ License-File: LICENSE
14
+ Provides-Extra: dev
15
+ Requires-Dist: pytest>=7.0; extra == "dev"
16
+ Dynamic: license-file
17
+
18
+ # Password Finder
19
+
20
+ A small Python library that extracts a password out of free-form text.
21
+
22
+ The motivating case: an encrypted attachment arrives in one email, and the
23
+ password to open it arrives in another (or in the body of the same message)
24
+ phrased in plain English — *"the password for the attached document is
25
+ Q3Report2026z"*. This library pulls that password back out.
26
+
27
+ ## Installation
28
+
29
+ ```bash
30
+ pip install password-finder
31
+ ```
32
+
33
+ Requires Python 3.9+. It has no runtime dependencies (standard library only).
34
+
35
+ ## Usage
36
+
37
+ The finder **always returns a list of candidates**, ranked by confidence — it
38
+ never returns a single string, because there is often more than one plausible
39
+ password and the caller is best placed to decide (try each in turn, apply a
40
+ threshold, ask a human). The list is empty when nothing plausible was found.
41
+
42
+ Everything works on a **string** — read a file (or take an email body) and pass
43
+ its contents straight in:
44
+
45
+ ```python
46
+ from password_finder import PasswordFinder
47
+
48
+ finder = PasswordFinder()
49
+
50
+ email_body = open("message.html", encoding="utf-8").read()
51
+
52
+ # Full candidate objects, ranked by confidence (highest first), de-duplicated
53
+ for c in finder.find_all(email_body):
54
+ print(c.password) # "Q3Report2026z"
55
+ print(c.confidence) # 0.96
56
+ print(c.keyword) # "password"
57
+ print(c.context) # "…the password to open the document is: Q3Report2026z…"
58
+
59
+ # Just the strings, same ranking
60
+ passwords = finder.find_passwords(email_body) # ["Q3Report2026z", ...]
61
+
62
+ # The most likely one (guard for the empty case yourself)
63
+ candidates = finder.find_all(email_body)
64
+ best = candidates[0].password if candidates else None
65
+ ```
66
+
67
+ If you don't need to hold on to a configured finder, call the module-level
68
+ helpers — string in, passwords out:
69
+
70
+ ```python
71
+ import password_finder
72
+
73
+ password_finder.find_passwords(open("message.html").read()) # ["Q3Report2026z", ...]
74
+ password_finder.find_all("The password is Hunter2!") # [Candidate(...)]
75
+ ```
76
+
77
+ **HTML is detected automatically.** You never tell the library whether a string
78
+ is HTML or plain text — it decides per input. When a string looks like HTML
79
+ (tags or entities present), `<script>`/`<style>` blocks and comments are
80
+ dropped, tags are stripped and entities decoded before matching, so
81
+ `<b>Rb9TZ4W</b>` and `&amp;` come out correctly; plain text is left untouched.
82
+ You can ask the library directly with `PasswordFinder.looks_like_html(text)`,
83
+ or force plain-text handling with `PasswordFinder(decode_html=False)`.
84
+
85
+ ## How it works
86
+
87
+ The finder scans for password-related keywords — `password`, `passphrase`,
88
+ `passcode`, `access code`, `unlock code`, `one-time code`, `authentication
89
+ code`, `login credentials`, `pwd`, `pin`, `code`, `secret`, localised spellings
90
+ (`kennwort`, `mot de passe`, `contraseña`, …) and more — optionally followed by
91
+ filler words (*"for opening the protected mail content"*) and a connector (`:`,
92
+ `=`, or `is`), then captures the token that follows. It matches four layouts:
93
+
94
+ 1. **strict** — keyword + curated filler + connector,
95
+ 2. **loose** — keyword + free text up to a `:`/`=`,
96
+ 3. **wrapped** — keyword immediately followed by a quoted/bracketed value,
97
+ 4. **nextline** — keyword alone on a line with the value on the next line
98
+ (this also recovers label/value pairs in adjacent HTML table cells).
99
+
100
+ Values wrapped in quotes, `*markdown*`, `` `backticks` `` or `(brackets)` are
101
+ unwrapped automatically. The matching `pattern` and character `span` are exposed
102
+ on each `Candidate` for debugging.
103
+
104
+ Each hit is scored heuristically. Signals that raise confidence:
105
+
106
+ - an explicit `:` or `=` connector,
107
+ - a specific keyword (`password` beats `pin`),
108
+ - a value that is close to the keyword (short distance),
109
+ - high **Shannon entropy** and **character-class diversity** (looks generated),
110
+ - a sensible length (6–40 characters),
111
+ - a value the sender deliberately **quoted/bracketed**.
112
+
113
+ Signals that lower it, or reject the match outright:
114
+
115
+ - lots of filler words / large distance between the keyword and the value,
116
+ - the "password" being a plain word like `attached`, `below`, `reset`, or `is`,
117
+ - a long pure-digit run (looks like a reference/policy number),
118
+ - shapes that are **never** passwords: URLs, email addresses, phone numbers,
119
+ dates, times, and currency amounts — these are rejected outright.
120
+
121
+ This means when a message contains several candidates, the most plausible one
122
+ sorts to the top. Because it is heuristic, always sanity-check
123
+ `.confidence` for your use case rather than trusting the top hit blindly.
124
+
125
+ ## Configuration
126
+
127
+ Basic knobs, shown with their defaults:
128
+
129
+ ```python
130
+ finder = PasswordFinder(
131
+ min_length=3, # ignore very short tokens
132
+ max_length=128, # ignore very long tokens (URLs etc.)
133
+ extra_keywords=None, # add your own trigger words
134
+ decode_html=True, # strip HTML before matching
135
+ )
136
+ ```
137
+
138
+ Raising `min_length` and lowering `max_length` is the quickest way to trade
139
+ recall for precision — e.g. `PasswordFinder(min_length=6, max_length=64)` if you
140
+ know the passwords you care about are never shorter than six characters.
141
+
142
+ Everything the finder uses — keywords and their base scores, the filler/reject
143
+ word lists, and every scoring weight — lives in `FinderConfig` and can be
144
+ overridden without touching the engine:
145
+
146
+ ```python
147
+ from password_finder import PasswordFinder, FinderConfig, Weights
148
+
149
+ config = FinderConfig(
150
+ keywords={"password": 0.75, "kennwort": 0.72, "magicword": 0.9},
151
+ weights=Weights(colon_bonus=0.20, entropy_scale=0.03),
152
+ )
153
+ finder = PasswordFinder(config=config)
154
+ ```
155
+
156
+ `extra_keywords` still works alongside a custom `config` (it merges on top).
157
+
158
+ ## Command-line testing
159
+
160
+ Two helpers are installed with the package for trying the library against real
161
+ messages. They print every candidate, best first:
162
+
163
+ ```bash
164
+ find-password email.txt # from a file
165
+ pbpaste | find-password # pipe from the clipboard (macOS)
166
+ cat message.eml | find-password
167
+
168
+ scan-emails # run over every file in tests/emails/
169
+ scan-emails path/to/dir # ...or a directory of your choosing
170
+ ```
171
+
172
+ You can also invoke them without installing:
173
+
174
+ ```bash
175
+ python -m password_finder.cli email.txt
176
+ ```
177
+
178
+ ## Tests
179
+
180
+ ```bash
181
+ python -m venv .venv && . .venv/bin/activate
182
+ pip install -e ".[dev]"
183
+ pytest
184
+ ```
185
+
186
+ Test emails live in `tests/emails/`, named after the password(s) they contain
187
+ using `-OR-` as the separator (e.g. `{pw1}-OR-{pw2}.html`). `test_email_fixtures.py`
188
+ runs the finder over each one and asserts those passwords are the top-ranked
189
+ candidates, so dropping a new email in there and naming it after its password(s)
190
+ is enough to add a regression test.
191
+
192
+ `tests/emails-special-characters/` holds passwords made of punctuation and
193
+ symbols — including characters a file name cannot contain — so it uses a
194
+ different convention: **the first line of the file is the expected password**,
195
+ and `test_special_character_fixtures.py` strips that line before matching. This
196
+ is where trimming, HTML entity decoding, and the rejection rules get exercised.
197
+
198
+ `tests/emails_negative/` holds the **negative corpus** — realistic emails that
199
+ must *not* yield a convincing password (marketing, statements, and keywords
200
+ sitting next to URLs / dates / phone numbers). `test_negative_corpus.py` asserts
201
+ nothing crosses a confidence threshold, guarding against precision regressions.
202
+
203
+ Each fixture directory has a `README.md` explaining its convention and how to
204
+ anonymise a real message before adding it.
205
+
206
+ ## License
207
+
208
+ MIT — Copyright (c) 2026 District5. See [LICENSE](LICENSE) for the full text.
@@ -0,0 +1,191 @@
1
+ # Password Finder
2
+
3
+ A small Python library that extracts a password out of free-form text.
4
+
5
+ The motivating case: an encrypted attachment arrives in one email, and the
6
+ password to open it arrives in another (or in the body of the same message)
7
+ phrased in plain English — *"the password for the attached document is
8
+ Q3Report2026z"*. This library pulls that password back out.
9
+
10
+ ## Installation
11
+
12
+ ```bash
13
+ pip install password-finder
14
+ ```
15
+
16
+ Requires Python 3.9+. It has no runtime dependencies (standard library only).
17
+
18
+ ## Usage
19
+
20
+ The finder **always returns a list of candidates**, ranked by confidence — it
21
+ never returns a single string, because there is often more than one plausible
22
+ password and the caller is best placed to decide (try each in turn, apply a
23
+ threshold, ask a human). The list is empty when nothing plausible was found.
24
+
25
+ Everything works on a **string** — read a file (or take an email body) and pass
26
+ its contents straight in:
27
+
28
+ ```python
29
+ from password_finder import PasswordFinder
30
+
31
+ finder = PasswordFinder()
32
+
33
+ email_body = open("message.html", encoding="utf-8").read()
34
+
35
+ # Full candidate objects, ranked by confidence (highest first), de-duplicated
36
+ for c in finder.find_all(email_body):
37
+ print(c.password) # "Q3Report2026z"
38
+ print(c.confidence) # 0.96
39
+ print(c.keyword) # "password"
40
+ print(c.context) # "…the password to open the document is: Q3Report2026z…"
41
+
42
+ # Just the strings, same ranking
43
+ passwords = finder.find_passwords(email_body) # ["Q3Report2026z", ...]
44
+
45
+ # The most likely one (guard for the empty case yourself)
46
+ candidates = finder.find_all(email_body)
47
+ best = candidates[0].password if candidates else None
48
+ ```
49
+
50
+ If you don't need to hold on to a configured finder, call the module-level
51
+ helpers — string in, passwords out:
52
+
53
+ ```python
54
+ import password_finder
55
+
56
+ password_finder.find_passwords(open("message.html").read()) # ["Q3Report2026z", ...]
57
+ password_finder.find_all("The password is Hunter2!") # [Candidate(...)]
58
+ ```
59
+
60
+ **HTML is detected automatically.** You never tell the library whether a string
61
+ is HTML or plain text — it decides per input. When a string looks like HTML
62
+ (tags or entities present), `<script>`/`<style>` blocks and comments are
63
+ dropped, tags are stripped and entities decoded before matching, so
64
+ `<b>Rb9TZ4W</b>` and `&amp;` come out correctly; plain text is left untouched.
65
+ You can ask the library directly with `PasswordFinder.looks_like_html(text)`,
66
+ or force plain-text handling with `PasswordFinder(decode_html=False)`.
67
+
68
+ ## How it works
69
+
70
+ The finder scans for password-related keywords — `password`, `passphrase`,
71
+ `passcode`, `access code`, `unlock code`, `one-time code`, `authentication
72
+ code`, `login credentials`, `pwd`, `pin`, `code`, `secret`, localised spellings
73
+ (`kennwort`, `mot de passe`, `contraseña`, …) and more — optionally followed by
74
+ filler words (*"for opening the protected mail content"*) and a connector (`:`,
75
+ `=`, or `is`), then captures the token that follows. It matches four layouts:
76
+
77
+ 1. **strict** — keyword + curated filler + connector,
78
+ 2. **loose** — keyword + free text up to a `:`/`=`,
79
+ 3. **wrapped** — keyword immediately followed by a quoted/bracketed value,
80
+ 4. **nextline** — keyword alone on a line with the value on the next line
81
+ (this also recovers label/value pairs in adjacent HTML table cells).
82
+
83
+ Values wrapped in quotes, `*markdown*`, `` `backticks` `` or `(brackets)` are
84
+ unwrapped automatically. The matching `pattern` and character `span` are exposed
85
+ on each `Candidate` for debugging.
86
+
87
+ Each hit is scored heuristically. Signals that raise confidence:
88
+
89
+ - an explicit `:` or `=` connector,
90
+ - a specific keyword (`password` beats `pin`),
91
+ - a value that is close to the keyword (short distance),
92
+ - high **Shannon entropy** and **character-class diversity** (looks generated),
93
+ - a sensible length (6–40 characters),
94
+ - a value the sender deliberately **quoted/bracketed**.
95
+
96
+ Signals that lower it, or reject the match outright:
97
+
98
+ - lots of filler words / large distance between the keyword and the value,
99
+ - the "password" being a plain word like `attached`, `below`, `reset`, or `is`,
100
+ - a long pure-digit run (looks like a reference/policy number),
101
+ - shapes that are **never** passwords: URLs, email addresses, phone numbers,
102
+ dates, times, and currency amounts — these are rejected outright.
103
+
104
+ This means when a message contains several candidates, the most plausible one
105
+ sorts to the top. Because it is heuristic, always sanity-check
106
+ `.confidence` for your use case rather than trusting the top hit blindly.
107
+
108
+ ## Configuration
109
+
110
+ Basic knobs, shown with their defaults:
111
+
112
+ ```python
113
+ finder = PasswordFinder(
114
+ min_length=3, # ignore very short tokens
115
+ max_length=128, # ignore very long tokens (URLs etc.)
116
+ extra_keywords=None, # add your own trigger words
117
+ decode_html=True, # strip HTML before matching
118
+ )
119
+ ```
120
+
121
+ Raising `min_length` and lowering `max_length` is the quickest way to trade
122
+ recall for precision — e.g. `PasswordFinder(min_length=6, max_length=64)` if you
123
+ know the passwords you care about are never shorter than six characters.
124
+
125
+ Everything the finder uses — keywords and their base scores, the filler/reject
126
+ word lists, and every scoring weight — lives in `FinderConfig` and can be
127
+ overridden without touching the engine:
128
+
129
+ ```python
130
+ from password_finder import PasswordFinder, FinderConfig, Weights
131
+
132
+ config = FinderConfig(
133
+ keywords={"password": 0.75, "kennwort": 0.72, "magicword": 0.9},
134
+ weights=Weights(colon_bonus=0.20, entropy_scale=0.03),
135
+ )
136
+ finder = PasswordFinder(config=config)
137
+ ```
138
+
139
+ `extra_keywords` still works alongside a custom `config` (it merges on top).
140
+
141
+ ## Command-line testing
142
+
143
+ Two helpers are installed with the package for trying the library against real
144
+ messages. They print every candidate, best first:
145
+
146
+ ```bash
147
+ find-password email.txt # from a file
148
+ pbpaste | find-password # pipe from the clipboard (macOS)
149
+ cat message.eml | find-password
150
+
151
+ scan-emails # run over every file in tests/emails/
152
+ scan-emails path/to/dir # ...or a directory of your choosing
153
+ ```
154
+
155
+ You can also invoke them without installing:
156
+
157
+ ```bash
158
+ python -m password_finder.cli email.txt
159
+ ```
160
+
161
+ ## Tests
162
+
163
+ ```bash
164
+ python -m venv .venv && . .venv/bin/activate
165
+ pip install -e ".[dev]"
166
+ pytest
167
+ ```
168
+
169
+ Test emails live in `tests/emails/`, named after the password(s) they contain
170
+ using `-OR-` as the separator (e.g. `{pw1}-OR-{pw2}.html`). `test_email_fixtures.py`
171
+ runs the finder over each one and asserts those passwords are the top-ranked
172
+ candidates, so dropping a new email in there and naming it after its password(s)
173
+ is enough to add a regression test.
174
+
175
+ `tests/emails-special-characters/` holds passwords made of punctuation and
176
+ symbols — including characters a file name cannot contain — so it uses a
177
+ different convention: **the first line of the file is the expected password**,
178
+ and `test_special_character_fixtures.py` strips that line before matching. This
179
+ is where trimming, HTML entity decoding, and the rejection rules get exercised.
180
+
181
+ `tests/emails_negative/` holds the **negative corpus** — realistic emails that
182
+ must *not* yield a convincing password (marketing, statements, and keywords
183
+ sitting next to URLs / dates / phone numbers). `test_negative_corpus.py` asserts
184
+ nothing crosses a confidence threshold, guarding against precision regressions.
185
+
186
+ Each fixture directory has a `README.md` explaining its convention and how to
187
+ anonymise a real message before adding it.
188
+
189
+ ## License
190
+
191
+ MIT — Copyright (c) 2026 District5. See [LICENSE](LICENSE) for the full text.
@@ -0,0 +1,102 @@
1
+ # Security Policy
2
+
3
+ ## Reporting a vulnerability
4
+
5
+ **Please do not open a public issue for a security problem.**
6
+
7
+ Use GitHub's private vulnerability reporting instead: go to the
8
+ [Security tab](https://github.com/district-5/password-finder/security)
9
+ and choose **Report a vulnerability**. That channel is private to you and the
10
+ maintainers.
11
+
12
+ If you cannot use it, open a normal issue containing **only** the sentence
13
+ "I have a security report" — no details, no sample data — and a maintainer will
14
+ arrange a private channel. Do not describe the problem in the public issue.
15
+
16
+ We aim to acknowledge a report within 5 working days, and to agree a disclosure
17
+ timeline with you once we have confirmed it. Please give us a reasonable chance
18
+ to ship a fix before publishing.
19
+
20
+ <!-- Maintainers: if you set up a dedicated security mailbox, add it here as a
21
+ second channel. Do not list an address that nobody monitors. -->
22
+
23
+ ## Never send us a real password
24
+
25
+ This library exists to extract passwords, so reports naturally come with sample
26
+ messages attached. **Sanitise them before sending, even privately.**
27
+
28
+ Replace the real password with an invented one of the **same shape** — same
29
+ length, same mix of upper case, lower case, digits and symbols, same separators
30
+ — and strip names, email addresses, phone numbers, account and reference
31
+ numbers, real domains, and mail headers. `tests/emails/README.md` has worked
32
+ examples with measured confidence scores.
33
+
34
+ If a real password has already been committed, pasted into an issue, or emailed
35
+ anywhere: **rotate it first**, then tell us. Treat it as disclosed. Git objects
36
+ stay reachable in forks, caches, and existing clones long after a force-push,
37
+ so rewriting history is not a remedy on its own.
38
+
39
+ ## Scope
40
+
41
+ The library takes untrusted text and runs regular expressions over it. It has no
42
+ runtime dependencies, opens no network connections, reads no files (the
43
+ `find-password` and `scan-emails` CLI entry points do), and never evaluates or
44
+ deserialises input. So the interesting attack surface is narrow, and it is
45
+ mostly about what a hostile *input string* can do to a process that parses it.
46
+
47
+ In scope:
48
+
49
+ - **Catastrophic backtracking / denial of service.** An input that makes
50
+ matching take pathological time or memory. The matcher compiles patterns from
51
+ a configurable keyword list and applies bounded-repetition groups to arbitrary
52
+ text; an input that defeats those bounds is a valid report. Please include the
53
+ input (sanitised), the wall-clock time, and your `min_length`/`max_length` and
54
+ `FinderConfig` settings.
55
+ - **Unbounded memory growth** on large or adversarial bodies, including in HTML
56
+ normalisation.
57
+ - **Crashes** that a caller cannot defend against — an uncaught exception on
58
+ input that is merely unusual rather than invalid.
59
+ - **Leaks of extracted values** into places a caller did not choose: an
60
+ exception message, a warning, `repr()` output, or a temporary file.
61
+ - **Supply-chain problems**: a compromised or typo-squatted `password-finder`
62
+ release, or a tampered artefact.
63
+
64
+ Out of scope — real reports, but not security ones, so please use the
65
+ [issue templates](https://github.com/district-5/password-finder/issues/new/choose):
66
+
67
+ - The finder **missing** a password, ranking the wrong candidate first, or
68
+ returning a false positive. These are detection bugs. The library is
69
+ explicitly heuristic and returns ranked candidates rather than one answer.
70
+ - A confidence score you disagree with.
71
+ - The library finding a password in a message you consider private — it only
72
+ ever sees the string you pass it.
73
+
74
+ ## Handling extracted passwords safely
75
+
76
+ Some hazards live in the calling code rather than in this library, and they are
77
+ easy to trip over. If you integrate it, note:
78
+
79
+ - **`str(candidate)` returns the bare password.** `Candidate.__str__` is the
80
+ password itself, so `print(f"found {c}")`, `logging.info("%s", c)`, and any
81
+ f-string interpolation write a real secret to your logs.
82
+ - **`candidate.context` contains the password plus roughly 40 characters either
83
+ side.** It exists for debugging and review. Logging it leaks both the secret
84
+ and the surrounding message text.
85
+ - **`candidate.to_dict()` serialises `password` and `context` in full.** Do not
86
+ hand it to a JSON logger, an error tracker, or an analytics pipeline.
87
+ - **The CLI prints candidates to stdout.** Convenient for testing, but that
88
+ means terminal scrollback, shell history, CI logs, and any redirect target.
89
+ Do not run `scan-emails` over real mail in CI.
90
+ - Keep extracted passwords in memory only as long as you need them, and prefer
91
+ passing them straight to the consumer (e.g. the archive being opened) over
92
+ storing them.
93
+
94
+ ## Supported versions
95
+
96
+ | Version | Supported |
97
+ | --- | --- |
98
+ | 2.1.x | Yes |
99
+ | < 2.1 | No — please upgrade |
100
+
101
+ Fixes land on `master` and go out in a new release; there are no long-term
102
+ support branches.
@@ -0,0 +1,32 @@
1
+ [build-system]
2
+ requires = ["setuptools>=61"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "password-finder"
7
+ version = "2.1.1"
8
+ description = "Extract passwords from free-form text such as email bodies (e.g. a password for an encrypted attachment sent in a separate message)."
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ license = { text = "MIT" }
12
+ keywords = ["password", "extract", "parse", "email", "text", "regex"]
13
+ authors = [{ name = "District5" }]
14
+ classifiers = [
15
+ "License :: OSI Approved :: MIT License",
16
+ "Programming Language :: Python :: 3",
17
+ "Topic :: Text Processing",
18
+ ]
19
+ dependencies = []
20
+
21
+ [project.optional-dependencies]
22
+ dev = ["pytest>=7.0"]
23
+
24
+ [project.scripts]
25
+ find-password = "password_finder.cli:find_password"
26
+ scan-emails = "password_finder.cli:scan_emails"
27
+
28
+ [tool.setuptools.packages.find]
29
+ where = ["src"]
30
+
31
+ [tool.pytest.ini_options]
32
+ testpaths = ["tests"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,22 @@
1
+ """Extract passwords from free-form text such as email bodies.
2
+
3
+ The motivating case: an encrypted attachment arrives in one email, and the
4
+ password to open it arrives in another (or in the body of the same message)
5
+ phrased in plain English -- *"the password for the attached document is
6
+ Q3Report2026z"*. This library pulls that password back out.
7
+ """
8
+
9
+ from .candidate import Candidate
10
+ from .config import DEFAULT_KEYWORDS, FinderConfig, Weights
11
+ from .finder import PasswordFinder, find_all, find_passwords
12
+
13
+ __all__ = [
14
+ "Candidate",
15
+ "PasswordFinder",
16
+ "FinderConfig",
17
+ "Weights",
18
+ "DEFAULT_KEYWORDS",
19
+ "find_passwords",
20
+ "find_all",
21
+ ]
22
+ __version__ = "2.1.0"
@@ -0,0 +1,47 @@
1
+ """Value object representing a single extracted password candidate."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+
7
+
8
+ @dataclass(frozen=True)
9
+ class Candidate:
10
+ """A single password candidate extracted from a body of text.
11
+
12
+ Immutable value object. :attr:`confidence` is a heuristic score in the
13
+ range ``0.0`` - ``1.0`` describing how likely this string is to be the
14
+ intended password. It is used to rank candidates when a text contains more
15
+ than one.
16
+
17
+ :param password: The extracted password.
18
+ :param confidence: Heuristic score, ``0.0`` (unlikely) - ``1.0`` (very likely).
19
+ :param keyword: The keyword that triggered the match (e.g. ``"password"``).
20
+ :param context: A short snippet of surrounding text for debugging/review.
21
+ :param offset: Character offset of the password within the normalised text.
22
+ :param pattern: Which match strategy fired: ``strict``/``loose``/``wrapped``/``nextline``.
23
+ :param span: ``(start, end)`` character span of the password in the normalised text.
24
+ """
25
+
26
+ password: str
27
+ confidence: float
28
+ keyword: str
29
+ context: str
30
+ offset: int
31
+ pattern: str = ""
32
+ span: tuple[int, int] = (-1, -1)
33
+
34
+ def __str__(self) -> str:
35
+ return self.password
36
+
37
+ def to_dict(self) -> dict:
38
+ """Return the candidate as a plain dictionary."""
39
+ return {
40
+ "password": self.password,
41
+ "confidence": self.confidence,
42
+ "keyword": self.keyword,
43
+ "context": self.context,
44
+ "offset": self.offset,
45
+ "pattern": self.pattern,
46
+ "span": self.span,
47
+ }