remote-jobs-api 1.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- remote_jobs_api-1.1.0/LICENSE +21 -0
- remote_jobs_api-1.1.0/PKG-INFO +157 -0
- remote_jobs_api-1.1.0/README.md +132 -0
- remote_jobs_api-1.1.0/pyproject.toml +40 -0
- remote_jobs_api-1.1.0/remote_jobs_api/__init__.py +2 -0
- remote_jobs_api-1.1.0/remote_jobs_api/__main__.py +2 -0
- remote_jobs_api-1.1.0/remote_jobs_api/cli.py +74 -0
- remote_jobs_api-1.1.0/remote_jobs_api/feed.py +312 -0
- remote_jobs_api-1.1.0/remote_jobs_api/server.py +131 -0
- remote_jobs_api-1.1.0/remote_jobs_api.egg-info/PKG-INFO +157 -0
- remote_jobs_api-1.1.0/remote_jobs_api.egg-info/SOURCES.txt +14 -0
- remote_jobs_api-1.1.0/remote_jobs_api.egg-info/dependency_links.txt +1 -0
- remote_jobs_api-1.1.0/remote_jobs_api.egg-info/entry_points.txt +2 -0
- remote_jobs_api-1.1.0/remote_jobs_api.egg-info/top_level.txt +1 -0
- remote_jobs_api-1.1.0/setup.cfg +4 -0
- remote_jobs_api-1.1.0/tests/test_feed.py +161 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Nova (earnnova7)
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: remote-jobs-api
|
|
3
|
+
Version: 1.1.0
|
|
4
|
+
Summary: Normalized remote-job data across 5 boards — one endpoint, one schema, skill fit-score
|
|
5
|
+
Author-email: Nova <earnnova@tten.no>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/earnnova-dev/remote-jobs-api
|
|
8
|
+
Project-URL: Documentation, https://earnnova-dev.github.io/remote-jobs-api/
|
|
9
|
+
Keywords: jobs,remote,api,scraping,feed,aggregator,data
|
|
10
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
11
|
+
Classifier: Intended Audience :: Developers
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
13
|
+
Classifier: Operating System :: OS Independent
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Topic :: Internet :: WWW/HTTP
|
|
20
|
+
Classifier: Topic :: Software Development :: Libraries
|
|
21
|
+
Requires-Python: >=3.9
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
License-File: LICENSE
|
|
24
|
+
Dynamic: license-file
|
|
25
|
+
|
|
26
|
+
# Remote Jobs API
|
|
27
|
+
|
|
28
|
+
**Live, normalized remote-job data across 5 boards — one clean endpoint, one schema.**
|
|
29
|
+
|
|
30
|
+
Stop scraping Remotive / RemoteOK / We Work Remotely / Jobicy / Hacker News
|
|
31
|
+
yourself. This API pulls them, normalizes every listing into a single
|
|
32
|
+
schema, dedupes, and hands you structured JSON (or CSV) — with an optional
|
|
33
|
+
**0-100 skill fit-score** for ranking.
|
|
34
|
+
|
|
35
|
+
Built for:
|
|
36
|
+
- Job boards & aggregator sites
|
|
37
|
+
- Recruiter and ATS tooling
|
|
38
|
+
- Lead-gen and market research
|
|
39
|
+
- **AI agents** that need structured job data in one call
|
|
40
|
+
|
|
41
|
+
## Why it's different
|
|
42
|
+
- **One schema, every board.** No per-board adapters on your side.
|
|
43
|
+
- **No session scraping.** Reads public job-board endpoints — your data never
|
|
44
|
+
leaves their servers, no browser, no account.
|
|
45
|
+
- **Self-hostable.** Stdlib-only Python (zero dependencies). Run it on a
|
|
46
|
+
laptop, a $5 VPS, or containerize it with the included Dockerfile.
|
|
47
|
+
- **Deterministic fit scoring.** No hidden LLM, no black box — reproducible
|
|
48
|
+
ranking you can reason about and unit-test.
|
|
49
|
+
|
|
50
|
+
## Endpoints
|
|
51
|
+
|
|
52
|
+
| Method | Path | Description |
|
|
53
|
+
|--------|------|-------------|
|
|
54
|
+
| GET | `/health` | Status + live job count |
|
|
55
|
+
| GET | `/v1/jobs/sources` | Available boards |
|
|
56
|
+
| GET | `/v1/jobs` | Normalized job listings |
|
|
57
|
+
|
|
58
|
+
### GET /v1/jobs
|
|
59
|
+
|
|
60
|
+
Query params:
|
|
61
|
+
|
|
62
|
+
| Param | Type | Notes |
|
|
63
|
+
|-------|------|-------|
|
|
64
|
+
| `skills` | string | Comma-separated keywords → adds `fit_score`, ranks best-first |
|
|
65
|
+
| `source` | string | `remotive` / `remoteok` / `jobicy` / `wwr` / `hn` |
|
|
66
|
+
| `limit` | int | 1-500 (default 50) |
|
|
67
|
+
| `min_score` | int | 0-100, filter by fit (needs `skills`) |
|
|
68
|
+
| `format` | string | `json` (default) or `csv` |
|
|
69
|
+
|
|
70
|
+
### Example
|
|
71
|
+
|
|
72
|
+
```bash
|
|
73
|
+
curl "https://<your-host>/v1/jobs?skills=python,backend,api&limit=3"
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
```json
|
|
77
|
+
{
|
|
78
|
+
"count": 3,
|
|
79
|
+
"generated_at": 1790433192,
|
|
80
|
+
"jobs": [
|
|
81
|
+
{
|
|
82
|
+
"id": "94516b10164d3c70",
|
|
83
|
+
"title": "Senior Backend Developer (Python)",
|
|
84
|
+
"company": "Proxify AB",
|
|
85
|
+
"url": "https://weworkremotely.com/remote-jobs/proxify-ab-senior-backend-developer-python-10",
|
|
86
|
+
"location": "Anywhere in the World",
|
|
87
|
+
"category": "Back-End Programming",
|
|
88
|
+
"source": "wwr",
|
|
89
|
+
"published": "Tue, 15 Sep 2026 09:01:36 +0000",
|
|
90
|
+
"fit_score": 80
|
|
91
|
+
}
|
|
92
|
+
]
|
|
93
|
+
}
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
## Run it yourself
|
|
97
|
+
|
|
98
|
+
Zero runtime dependencies — it's pure Python stdlib.
|
|
99
|
+
|
|
100
|
+
**Option A — CLI (query the feed directly, no server needed):**
|
|
101
|
+
|
|
102
|
+
```bash
|
|
103
|
+
pip install remote-jobs-api
|
|
104
|
+
remote-jobs-api --skills python,api --limit 5
|
|
105
|
+
remote-jobs-api --source wwr --format csv
|
|
106
|
+
remote-jobs-api --help
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
**Option B — HTTP API server (self-host):**
|
|
110
|
+
|
|
111
|
+
```bash
|
|
112
|
+
pip install remote-jobs-api
|
|
113
|
+
remote-jobs-api --serve --port 8321 # or: python -m remote_jobs_api.server
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
Or from source / Docker:
|
|
117
|
+
|
|
118
|
+
```bash
|
|
119
|
+
git clone https://github.com/earnnova-dev/remote-jobs-api
|
|
120
|
+
cd remote-jobs-api
|
|
121
|
+
python3 -m remote_jobs_api.server # listens on :8321
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
```bash
|
|
125
|
+
docker build -t remote-jobs-api .
|
|
126
|
+
docker run -p 8321:8321 remote-jobs-api
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
Env vars:
|
|
130
|
+
- `PORT` (default `8321`)
|
|
131
|
+
- `RJA_CACHE_TTL` (default `300` s) — how long to cache upstream fetches
|
|
132
|
+
|
|
133
|
+
## Testing
|
|
134
|
+
|
|
135
|
+
Offline unit tests (no network): `python -m pytest`.
|
|
136
|
+
|
|
137
|
+
## Pricing (indicative, for a hosted tier)
|
|
138
|
+
|
|
139
|
+
| Plan | Price | Includes |
|
|
140
|
+
|------|-------|----------|
|
|
141
|
+
| **Free** | $0 | 100 calls/mo, all boards, 50 results/req |
|
|
142
|
+
| **Pro** | $19/mo | 10k calls/mo, CSV, min_score, higher limits |
|
|
143
|
+
| **Team** | $49/mo | 100k calls/mo, SLA, webhooks (soon) |
|
|
144
|
+
|
|
145
|
+
> Self-hosted = free, MIT. The paid tier is the hosted, always-on,
|
|
146
|
+
> high-rate-limit version — the same engine, managed for you.
|
|
147
|
+
|
|
148
|
+
## License
|
|
149
|
+
|
|
150
|
+
MIT. See [LICENSE](LICENSE).
|
|
151
|
+
|
|
152
|
+
## Data & compliance
|
|
153
|
+
|
|
154
|
+
This API only reads public, no-auth job-board endpoints and returns data
|
|
155
|
+
exactly as those boards publish it. It does not store your personal data,
|
|
156
|
+
does not scrape authenticated sessions, and does not resell any board's
|
|
157
|
+
proprietary content beyond what is publicly viewable.
|
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
# Remote Jobs API
|
|
2
|
+
|
|
3
|
+
**Live, normalized remote-job data across 5 boards — one clean endpoint, one schema.**
|
|
4
|
+
|
|
5
|
+
Stop scraping Remotive / RemoteOK / We Work Remotely / Jobicy / Hacker News
|
|
6
|
+
yourself. This API pulls them, normalizes every listing into a single
|
|
7
|
+
schema, dedupes, and hands you structured JSON (or CSV) — with an optional
|
|
8
|
+
**0-100 skill fit-score** for ranking.
|
|
9
|
+
|
|
10
|
+
Built for:
|
|
11
|
+
- Job boards & aggregator sites
|
|
12
|
+
- Recruiter and ATS tooling
|
|
13
|
+
- Lead-gen and market research
|
|
14
|
+
- **AI agents** that need structured job data in one call
|
|
15
|
+
|
|
16
|
+
## Why it's different
|
|
17
|
+
- **One schema, every board.** No per-board adapters on your side.
|
|
18
|
+
- **No session scraping.** Reads public job-board endpoints — your data never
|
|
19
|
+
leaves their servers, no browser, no account.
|
|
20
|
+
- **Self-hostable.** Stdlib-only Python (zero dependencies). Run it on a
|
|
21
|
+
laptop, a $5 VPS, or containerize it with the included Dockerfile.
|
|
22
|
+
- **Deterministic fit scoring.** No hidden LLM, no black box — reproducible
|
|
23
|
+
ranking you can reason about and unit-test.
|
|
24
|
+
|
|
25
|
+
## Endpoints
|
|
26
|
+
|
|
27
|
+
| Method | Path | Description |
|
|
28
|
+
|--------|------|-------------|
|
|
29
|
+
| GET | `/health` | Status + live job count |
|
|
30
|
+
| GET | `/v1/jobs/sources` | Available boards |
|
|
31
|
+
| GET | `/v1/jobs` | Normalized job listings |
|
|
32
|
+
|
|
33
|
+
### GET /v1/jobs
|
|
34
|
+
|
|
35
|
+
Query params:
|
|
36
|
+
|
|
37
|
+
| Param | Type | Notes |
|
|
38
|
+
|-------|------|-------|
|
|
39
|
+
| `skills` | string | Comma-separated keywords → adds `fit_score`, ranks best-first |
|
|
40
|
+
| `source` | string | `remotive` / `remoteok` / `jobicy` / `wwr` / `hn` |
|
|
41
|
+
| `limit` | int | 1-500 (default 50) |
|
|
42
|
+
| `min_score` | int | 0-100, filter by fit (needs `skills`) |
|
|
43
|
+
| `format` | string | `json` (default) or `csv` |
|
|
44
|
+
|
|
45
|
+
### Example
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
curl "https://<your-host>/v1/jobs?skills=python,backend,api&limit=3"
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
```json
|
|
52
|
+
{
|
|
53
|
+
"count": 3,
|
|
54
|
+
"generated_at": 1790433192,
|
|
55
|
+
"jobs": [
|
|
56
|
+
{
|
|
57
|
+
"id": "94516b10164d3c70",
|
|
58
|
+
"title": "Senior Backend Developer (Python)",
|
|
59
|
+
"company": "Proxify AB",
|
|
60
|
+
"url": "https://weworkremotely.com/remote-jobs/proxify-ab-senior-backend-developer-python-10",
|
|
61
|
+
"location": "Anywhere in the World",
|
|
62
|
+
"category": "Back-End Programming",
|
|
63
|
+
"source": "wwr",
|
|
64
|
+
"published": "Tue, 15 Sep 2026 09:01:36 +0000",
|
|
65
|
+
"fit_score": 80
|
|
66
|
+
}
|
|
67
|
+
]
|
|
68
|
+
}
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
## Run it yourself
|
|
72
|
+
|
|
73
|
+
Zero runtime dependencies — it's pure Python stdlib.
|
|
74
|
+
|
|
75
|
+
**Option A — CLI (query the feed directly, no server needed):**
|
|
76
|
+
|
|
77
|
+
```bash
|
|
78
|
+
pip install remote-jobs-api
|
|
79
|
+
remote-jobs-api --skills python,api --limit 5
|
|
80
|
+
remote-jobs-api --source wwr --format csv
|
|
81
|
+
remote-jobs-api --help
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
**Option B — HTTP API server (self-host):**
|
|
85
|
+
|
|
86
|
+
```bash
|
|
87
|
+
pip install remote-jobs-api
|
|
88
|
+
remote-jobs-api --serve --port 8321 # or: python -m remote_jobs_api.server
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
Or from source / Docker:
|
|
92
|
+
|
|
93
|
+
```bash
|
|
94
|
+
git clone https://github.com/earnnova-dev/remote-jobs-api
|
|
95
|
+
cd remote-jobs-api
|
|
96
|
+
python3 -m remote_jobs_api.server # listens on :8321
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
```bash
|
|
100
|
+
docker build -t remote-jobs-api .
|
|
101
|
+
docker run -p 8321:8321 remote-jobs-api
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
Env vars:
|
|
105
|
+
- `PORT` (default `8321`)
|
|
106
|
+
- `RJA_CACHE_TTL` (default `300` s) — how long to cache upstream fetches
|
|
107
|
+
|
|
108
|
+
## Testing
|
|
109
|
+
|
|
110
|
+
Offline unit tests (no network): `python -m pytest`.
|
|
111
|
+
|
|
112
|
+
## Pricing (indicative, for a hosted tier)
|
|
113
|
+
|
|
114
|
+
| Plan | Price | Includes |
|
|
115
|
+
|------|-------|----------|
|
|
116
|
+
| **Free** | $0 | 100 calls/mo, all boards, 50 results/req |
|
|
117
|
+
| **Pro** | $19/mo | 10k calls/mo, CSV, min_score, higher limits |
|
|
118
|
+
| **Team** | $49/mo | 100k calls/mo, SLA, webhooks (soon) |
|
|
119
|
+
|
|
120
|
+
> Self-hosted = free, MIT. The paid tier is the hosted, always-on,
|
|
121
|
+
> high-rate-limit version — the same engine, managed for you.
|
|
122
|
+
|
|
123
|
+
## License
|
|
124
|
+
|
|
125
|
+
MIT. See [LICENSE](LICENSE).
|
|
126
|
+
|
|
127
|
+
## Data & compliance
|
|
128
|
+
|
|
129
|
+
This API only reads public, no-auth job-board endpoints and returns data
|
|
130
|
+
exactly as those boards publish it. It does not store your personal data,
|
|
131
|
+
does not scrape authenticated sessions, and does not resell any board's
|
|
132
|
+
proprietary content beyond what is publicly viewable.
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=61"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "remote-jobs-api"
|
|
7
|
+
version = "1.1.0"
|
|
8
|
+
description = "Normalized remote-job data across 5 boards — one endpoint, one schema, skill fit-score"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
authors = [{ name = "Nova", email = "earnnova@tten.no" }]
|
|
13
|
+
keywords = ["jobs", "remote", "api", "scraping", "feed", "aggregator", "data"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 5 - Production/Stable",
|
|
16
|
+
"Intended Audience :: Developers",
|
|
17
|
+
"License :: OSI Approved :: MIT License",
|
|
18
|
+
"Operating System :: OS Independent",
|
|
19
|
+
"Programming Language :: Python :: 3",
|
|
20
|
+
"Programming Language :: Python :: 3.9",
|
|
21
|
+
"Programming Language :: Python :: 3.10",
|
|
22
|
+
"Programming Language :: Python :: 3.11",
|
|
23
|
+
"Programming Language :: Python :: 3.12",
|
|
24
|
+
"Topic :: Internet :: WWW/HTTP",
|
|
25
|
+
"Topic :: Software Development :: Libraries",
|
|
26
|
+
]
|
|
27
|
+
|
|
28
|
+
[project.urls]
|
|
29
|
+
Homepage = "https://github.com/earnnova-dev/remote-jobs-api"
|
|
30
|
+
Documentation = "https://earnnova-dev.github.io/remote-jobs-api/"
|
|
31
|
+
|
|
32
|
+
[project.scripts]
|
|
33
|
+
remote-jobs-api = "remote_jobs_api.cli:main"
|
|
34
|
+
|
|
35
|
+
[tool.setuptools]
|
|
36
|
+
packages = ["remote_jobs_api"]
|
|
37
|
+
|
|
38
|
+
[tool.pytest.ini_options]
|
|
39
|
+
testpaths = ["tests"]
|
|
40
|
+
addopts = "-q"
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
"""remote-jobs-api CLI — query the normalized feed without running a server.
|
|
2
|
+
|
|
3
|
+
Usage:
|
|
4
|
+
remote-jobs-api --skills python,api --limit 5
|
|
5
|
+
remote-jobs-api --source wwr --format csv
|
|
6
|
+
remote-jobs-api --serve --port 8321 # run the HTTP API
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import argparse
|
|
11
|
+
import csv
|
|
12
|
+
import io
|
|
13
|
+
import json
|
|
14
|
+
import sys
|
|
15
|
+
|
|
16
|
+
from . import __version__, feed
|
|
17
|
+
from .feed import SOURCES
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _print_csv(jobs, file):
|
|
21
|
+
buf = io.StringIO()
|
|
22
|
+
cols = ["id", "title", "company", "location", "salary",
|
|
23
|
+
"category", "source", "published", "fit_score", "url"]
|
|
24
|
+
w = csv.DictWriter(buf, fieldnames=cols, extrasaction="ignore")
|
|
25
|
+
w.writeheader()
|
|
26
|
+
for j in jobs:
|
|
27
|
+
w.writerow(j)
|
|
28
|
+
file.write(buf.getvalue())
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _run_query(args) -> int:
|
|
32
|
+
jobs = feed.filter_and_rank(
|
|
33
|
+
feed.collect(),
|
|
34
|
+
skills=args.skills or None,
|
|
35
|
+
source=args.source,
|
|
36
|
+
min_score=args.min_score,
|
|
37
|
+
)[: max(1, min(500, args.limit))]
|
|
38
|
+
|
|
39
|
+
if args.format == "csv":
|
|
40
|
+
_print_csv(jobs, sys.stdout)
|
|
41
|
+
else:
|
|
42
|
+
print(json.dumps(
|
|
43
|
+
{"count": len(jobs), "query": vars(args), "jobs": jobs},
|
|
44
|
+
ensure_ascii=False, indent=2,
|
|
45
|
+
))
|
|
46
|
+
return 0
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def main(argv=None) -> int:
|
|
50
|
+
p = argparse.ArgumentParser(
|
|
51
|
+
prog="remote-jobs-api",
|
|
52
|
+
description="Normalized remote-job data from 5 boards: " + ", ".join(SOURCES),
|
|
53
|
+
)
|
|
54
|
+
p.add_argument("--version", action="version", version=f"remote-jobs-api {__version__}")
|
|
55
|
+
p.add_argument("--skills", help="comma-separated keywords for fit scoring")
|
|
56
|
+
p.add_argument("--source", choices=SOURCES, help="restrict to one board")
|
|
57
|
+
p.add_argument("--limit", type=int, default=50, help="max results (1-500, default 50)")
|
|
58
|
+
p.add_argument("--min-score", type=int, help="minimum fit_score (needs --skills)")
|
|
59
|
+
p.add_argument("--format", choices=["json", "csv"], default="json")
|
|
60
|
+
p.add_argument("--serve", action="store_true", help="run the HTTP API server")
|
|
61
|
+
p.add_argument("--port", type=int, default=8321)
|
|
62
|
+
args = p.parse_args(argv)
|
|
63
|
+
|
|
64
|
+
if args.serve:
|
|
65
|
+
import os
|
|
66
|
+
os.environ["PORT"] = str(args.port)
|
|
67
|
+
from .server import main as serve_main
|
|
68
|
+
serve_main()
|
|
69
|
+
return 0
|
|
70
|
+
return _run_query(args)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
if __name__ == "__main__":
|
|
74
|
+
sys.exit(main())
|
|
@@ -0,0 +1,312 @@
|
|
|
1
|
+
"""Remote-Jobs Data API — normalized feed layer.
|
|
2
|
+
|
|
3
|
+
Stdlib-only. Pulls live remote-job listings from public job-board endpoints
|
|
4
|
+
that require no auth, normalizes them into ONE schema, and exposes them for
|
|
5
|
+
the HTTP API. This is the same feed engine as GigWatch, repackaged as a data
|
|
6
|
+
product.
|
|
7
|
+
|
|
8
|
+
Public sources (no key, no account):
|
|
9
|
+
- remotive https://remotive.com/api/remote-jobs
|
|
10
|
+
- remoteok https://remoteok.com/api
|
|
11
|
+
- jobicy https://jobicy.com/api/v2/remote-jobs
|
|
12
|
+
- wwr https://weworkremotely.com/remote-jobs.rss
|
|
13
|
+
- hn HN "Who is Hiring?" (Algolia, monthly megathread)
|
|
14
|
+
"""
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import hashlib
|
|
18
|
+
import html
|
|
19
|
+
import json
|
|
20
|
+
import re
|
|
21
|
+
import urllib.parse
|
|
22
|
+
import urllib.request
|
|
23
|
+
import xml.etree.ElementTree as ET
|
|
24
|
+
from dataclasses import dataclass, asdict, field
|
|
25
|
+
from typing import Any, Dict, List, Optional
|
|
26
|
+
|
|
27
|
+
UA = "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (compatible; RemoteJobsAPI/1.0)"
|
|
28
|
+
|
|
29
|
+
SOURCES = ["remotive", "remoteok", "jobicy", "wwr", "hn"]
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@dataclass
|
|
33
|
+
class Job:
|
|
34
|
+
id: str
|
|
35
|
+
title: str
|
|
36
|
+
company: str = ""
|
|
37
|
+
url: str = ""
|
|
38
|
+
location: str = ""
|
|
39
|
+
salary: str = ""
|
|
40
|
+
category: str = ""
|
|
41
|
+
tags: List[str] = field(default_factory=list)
|
|
42
|
+
published: str = ""
|
|
43
|
+
source: str = ""
|
|
44
|
+
description: str = ""
|
|
45
|
+
|
|
46
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
47
|
+
d = asdict(self)
|
|
48
|
+
# keep the public schema lean and stable
|
|
49
|
+
return {
|
|
50
|
+
"id": d["id"],
|
|
51
|
+
"title": d["title"],
|
|
52
|
+
"company": d["company"],
|
|
53
|
+
"url": d["url"],
|
|
54
|
+
"location": d["location"],
|
|
55
|
+
"salary": d["salary"],
|
|
56
|
+
"category": d["category"],
|
|
57
|
+
"tags": d["tags"],
|
|
58
|
+
"published": d["published"],
|
|
59
|
+
"source": d["source"],
|
|
60
|
+
"description": d["description"][:500],
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _get(url: str, timeout: int = 20) -> bytes:
|
|
65
|
+
req = urllib.request.Request(url, headers={"User-Agent": UA})
|
|
66
|
+
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
|
67
|
+
return resp.read()
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _clean(text: str) -> str:
|
|
71
|
+
if not text:
|
|
72
|
+
return ""
|
|
73
|
+
text = re.sub(r"<[^>]+>", " ", text)
|
|
74
|
+
text = html.unescape(text)
|
|
75
|
+
return re.sub(r"\s+", " ", text).strip()
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _h(s: str) -> str:
|
|
79
|
+
return hashlib.sha1((s or "").encode("utf-8")).hexdigest()[:16]
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def fetch_remotive(limit: Optional[int] = None) -> List[Job]:
|
|
83
|
+
data = json.loads(_get("https://remotive.com/api/remote-jobs").decode("utf-8"))
|
|
84
|
+
out = []
|
|
85
|
+
for raw in data.get("jobs", []):
|
|
86
|
+
out.append(Job(
|
|
87
|
+
id=str(raw.get("id")),
|
|
88
|
+
title=_clean(raw.get("title", "")),
|
|
89
|
+
company=_clean(raw.get("company_name", "")),
|
|
90
|
+
url=raw.get("url", ""),
|
|
91
|
+
location=_clean(raw.get("candidate_required_location", "")),
|
|
92
|
+
salary=_clean(raw.get("salary", "")),
|
|
93
|
+
category=_clean(raw.get("category", "")),
|
|
94
|
+
tags=[_clean(t) for t in raw.get("tags", []) if t],
|
|
95
|
+
published=raw.get("publication_date", ""),
|
|
96
|
+
source="remotive",
|
|
97
|
+
description=_clean(raw.get("description", "")),
|
|
98
|
+
))
|
|
99
|
+
return out[:limit] if limit else out
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def fetch_remoteok(limit: Optional[int] = None) -> List[Job]:
|
|
103
|
+
data = json.loads(_get("https://remoteok.com/api").decode("utf-8", "replace"))
|
|
104
|
+
out = []
|
|
105
|
+
for raw in data:
|
|
106
|
+
if not isinstance(raw, dict) or not raw.get("position"):
|
|
107
|
+
continue
|
|
108
|
+
lo, hi = raw.get("salary_min"), raw.get("salary_max")
|
|
109
|
+
salary = ""
|
|
110
|
+
if lo and hi:
|
|
111
|
+
salary = f"${int(lo):,}-${int(hi):,}"
|
|
112
|
+
elif lo or hi:
|
|
113
|
+
salary = f"${int(lo or hi):,}"
|
|
114
|
+
out.append(Job(
|
|
115
|
+
id=str(raw.get("id") or _h(raw.get("url", ""))),
|
|
116
|
+
title=_clean(raw.get("position", "")),
|
|
117
|
+
company=_clean(raw.get("company", "")),
|
|
118
|
+
url=raw.get("url", ""),
|
|
119
|
+
location=_clean(raw.get("location", "")),
|
|
120
|
+
salary=salary,
|
|
121
|
+
tags=[_clean(t) for t in raw.get("tags", []) if t],
|
|
122
|
+
published=raw.get("date", ""),
|
|
123
|
+
source="remoteok",
|
|
124
|
+
description=_clean(raw.get("description", "")),
|
|
125
|
+
))
|
|
126
|
+
out = [j for j in out if j.id and j.title]
|
|
127
|
+
return out[:limit] if limit else out
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def fetch_jobicy(limit: Optional[int] = None) -> List[Job]:
|
|
131
|
+
n = limit or 50
|
|
132
|
+
data = json.loads(_get(f"https://jobicy.com/api/v2/remote-jobs?count={n}").decode("utf-8", "replace"))
|
|
133
|
+
out = []
|
|
134
|
+
for raw in data.get("jobs", []):
|
|
135
|
+
if not isinstance(raw, dict):
|
|
136
|
+
continue
|
|
137
|
+
title = _clean(raw.get("jobTitle") or "")
|
|
138
|
+
if not title:
|
|
139
|
+
continue
|
|
140
|
+
tags = []
|
|
141
|
+
for k in ("jobIndustry", "jobType"):
|
|
142
|
+
v = raw.get(k) or []
|
|
143
|
+
if isinstance(v, list):
|
|
144
|
+
tags.extend(str(t) for t in v)
|
|
145
|
+
loc = " / ".join(p for p in (raw.get("jobGeo") or "", raw.get("jobLevel") or "") if p)
|
|
146
|
+
sal = str(raw.get("salary") or "").strip()
|
|
147
|
+
if sal.upper() in ("N/A", "NA"):
|
|
148
|
+
sal = ""
|
|
149
|
+
out.append(Job(
|
|
150
|
+
id=str(raw.get("id") or _h(raw.get("url", ""))),
|
|
151
|
+
title=title,
|
|
152
|
+
company=_clean(raw.get("companyName") or ""),
|
|
153
|
+
url=raw.get("url", ""),
|
|
154
|
+
location=loc,
|
|
155
|
+
salary=sal,
|
|
156
|
+
tags=tags,
|
|
157
|
+
published=raw.get("lastUpdate") or "",
|
|
158
|
+
source="jobicy",
|
|
159
|
+
description=_clean((raw.get("jobExcerpt") or "")[:300]),
|
|
160
|
+
))
|
|
161
|
+
out = [j for j in out if j.id and j.title]
|
|
162
|
+
return out[:limit] if limit else out
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def fetch_wwr(limit: Optional[int] = None) -> List[Job]:
|
|
166
|
+
root = ET.fromstring(_get("https://weworkremotely.com/remote-jobs.rss").decode("utf-8", "replace"))
|
|
167
|
+
out = []
|
|
168
|
+
for it in root.findall(".//item"):
|
|
169
|
+
title = (it.findtext("title") or "").strip()
|
|
170
|
+
link = (it.findtext("link") or "").strip()
|
|
171
|
+
company = ""
|
|
172
|
+
if ": " in title:
|
|
173
|
+
company, title = (p.strip() for p in title.split(": ", 1))
|
|
174
|
+
out.append(Job(
|
|
175
|
+
id=_h(link or title),
|
|
176
|
+
title=title,
|
|
177
|
+
company=company,
|
|
178
|
+
url=link,
|
|
179
|
+
category=_clean(it.findtext("category") or ""),
|
|
180
|
+
location=_clean(it.findtext("region") or ""),
|
|
181
|
+
published=(it.findtext("pubDate") or "").strip(),
|
|
182
|
+
source="wwr",
|
|
183
|
+
description=_clean(it.findtext("description") or ""),
|
|
184
|
+
))
|
|
185
|
+
out = [j for j in out if j.id and j.title]
|
|
186
|
+
return out[:limit] if limit else out
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def fetch_hn(limit: Optional[int] = None) -> List[Job]:
|
|
190
|
+
import time
|
|
191
|
+
now = int(time.time())
|
|
192
|
+
window = now - 35 * 86400
|
|
193
|
+
res = json.loads(_get("https://hn.algolia.com/api/v1/search?%s" % urllib.parse.urlencode({
|
|
194
|
+
"query": "who is hiring", "tags": "story",
|
|
195
|
+
"numericFilters": f"created_at_i>={window}", "hitsPerPage": 20,
|
|
196
|
+
})).decode("utf-8"))
|
|
197
|
+
stories = []
|
|
198
|
+
for h in res.get("hits", []):
|
|
199
|
+
t = (h.get("title") or "").lower()
|
|
200
|
+
if "who is hiring" not in t or "analysis" in t:
|
|
201
|
+
continue
|
|
202
|
+
if "who wants to be hired" in t or "show hn" in t:
|
|
203
|
+
continue
|
|
204
|
+
stories.append(h)
|
|
205
|
+
out = []
|
|
206
|
+
for story in stories:
|
|
207
|
+
item = json.loads(_get(f"https://hn.algolia.com/api/v1/items/{story.get('objectID')}").decode("utf-8"))
|
|
208
|
+
for ch in item.get("children", [])[:100]:
|
|
209
|
+
cid = ch.get("id")
|
|
210
|
+
base = f"https://news.ycombinator.com/item?id={cid}"
|
|
211
|
+
raw = ch.get("text") or ""
|
|
212
|
+
if not raw:
|
|
213
|
+
continue
|
|
214
|
+
# split on the RAW lines first (before any whitespace collapsing),
|
|
215
|
+
# then clean each line individually.
|
|
216
|
+
for line in raw.splitlines():
|
|
217
|
+
s = _clean(line).lstrip("-*• \t").strip()
|
|
218
|
+
if not s or len(s) > 120:
|
|
219
|
+
continue # multi-sentence blocks are not single-job lines
|
|
220
|
+
title = company = ""
|
|
221
|
+
if "|" in s:
|
|
222
|
+
parts = [p.strip() for p in s.split("|")]
|
|
223
|
+
if len(parts) >= 2:
|
|
224
|
+
company, title = parts[0], parts[1]
|
|
225
|
+
else:
|
|
226
|
+
m = re.match(r"^([A-Z][A-Za-z0-9&.'\-]{0,60}?)\s*(?:—|–|:|-)\s+(.{3,120})$", s)
|
|
227
|
+
if m:
|
|
228
|
+
company, title = m.group(1), m.group(2)
|
|
229
|
+
if not title or len(title) < 3 or len(title) > 100:
|
|
230
|
+
continue
|
|
231
|
+
if not company or company.lower() in ("http", "https", "i", "we", "a", "the"):
|
|
232
|
+
continue
|
|
233
|
+
if "http" in title.lower():
|
|
234
|
+
continue # title must not be a URL
|
|
235
|
+
out.append(Job(
|
|
236
|
+
id=_h(f"{cid}|{title}"),
|
|
237
|
+
title=title, company=company, url=base,
|
|
238
|
+
source="hn", description=s[:300],
|
|
239
|
+
))
|
|
240
|
+
out = [j for j in out if j.id and j.title]
|
|
241
|
+
return out[:limit] if limit else out
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
_FETCHERS = {
|
|
245
|
+
"remotive": fetch_remotive,
|
|
246
|
+
"remoteok": fetch_remoteok,
|
|
247
|
+
"jobicy": fetch_jobicy,
|
|
248
|
+
"wwr": fetch_wwr,
|
|
249
|
+
"hn": fetch_hn,
|
|
250
|
+
}
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def collect(sources: Optional[List[str]] = None, limit: Optional[int] = None,
|
|
254
|
+
per_source_limit: int = 100) -> List[Job]:
|
|
255
|
+
"""Fetch from the given sources (default: all), dedupe by URL, return flat list."""
|
|
256
|
+
sources = sources or SOURCES
|
|
257
|
+
seen: set = set()
|
|
258
|
+
out: List[Job] = []
|
|
259
|
+
for src in sources:
|
|
260
|
+
fn = _FETCHERS.get(src)
|
|
261
|
+
if not fn:
|
|
262
|
+
continue
|
|
263
|
+
try:
|
|
264
|
+
for j in fn(limit=per_source_limit):
|
|
265
|
+
key = j.url or j.id
|
|
266
|
+
if key in seen:
|
|
267
|
+
continue
|
|
268
|
+
seen.add(key)
|
|
269
|
+
out.append(j)
|
|
270
|
+
except Exception:
|
|
271
|
+
# a flaky source should not break the whole API
|
|
272
|
+
continue
|
|
273
|
+
# sort: newest-ish first (published desc, fall back to source order)
|
|
274
|
+
out.sort(key=lambda j: (j.published or ""), reverse=True)
|
|
275
|
+
return out[:limit] if limit else out
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
def _match_score(job: Job, skills: List[str]) -> int:
|
|
279
|
+
"""Deterministic 0-100 fit score: title hits weigh more than body hits."""
|
|
280
|
+
if not skills:
|
|
281
|
+
return 50
|
|
282
|
+
title = job.title.lower()
|
|
283
|
+
hay = job.haystack if hasattr(job, "haystack") else " ".join(
|
|
284
|
+
[job.title, job.company, job.category, job.location, job.salary,
|
|
285
|
+
" ".join(job.tags), job.description]).lower()
|
|
286
|
+
score = 0.0
|
|
287
|
+
for s in skills:
|
|
288
|
+
s = s.lower().strip()
|
|
289
|
+
if not s:
|
|
290
|
+
continue
|
|
291
|
+
if s in title:
|
|
292
|
+
score += 40
|
|
293
|
+
elif s in hay:
|
|
294
|
+
score += 15
|
|
295
|
+
return int(max(0, min(100, score)))
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
def filter_and_rank(jobs: List[Job], skills: Optional[List[str]] = None,
|
|
299
|
+
remote_only: bool = False, source: Optional[str] = None,
|
|
300
|
+
min_score: Optional[int] = None) -> List[Dict[str, Any]]:
|
|
301
|
+
out = []
|
|
302
|
+
for j in jobs:
|
|
303
|
+
if source and j.source != source:
|
|
304
|
+
continue
|
|
305
|
+
d = j.to_dict()
|
|
306
|
+
d["fit_score"] = _match_score(j, skills or [])
|
|
307
|
+
if min_score is not None and d["fit_score"] < min_score:
|
|
308
|
+
continue
|
|
309
|
+
out.append(d)
|
|
310
|
+
# best fit first, then newest
|
|
311
|
+
out.sort(key=lambda d: (d["fit_score"], d["published"]), reverse=True)
|
|
312
|
+
return out
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
"""Remote-Jobs Data API — stdlib HTTP server.
|
|
2
|
+
|
|
3
|
+
Endpoints:
|
|
4
|
+
GET /health -> {"status":"ok","sources":[...],"count_live":N}
|
|
5
|
+
GET /v1/jobs -> normalized job listings
|
|
6
|
+
GET /v1/jobs/sources -> {"sources":[...]}
|
|
7
|
+
|
|
8
|
+
Query params on /v1/jobs:
|
|
9
|
+
skills comma-separated keywords (e.g. "python,backend,api") -> fit_score 0-100
|
|
10
|
+
source remotive|remoteok|jobicy|wwr|hn (filter to one board)
|
|
11
|
+
limit max results (default 50, max 500)
|
|
12
|
+
min_score minimum fit_score to return (0-100)
|
|
13
|
+
format json|csv (default json)
|
|
14
|
+
|
|
15
|
+
Design: stdlib-only, no deps. Run: python3 -m remote_jobs_api.server
|
|
16
|
+
"""
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import csv
|
|
20
|
+
import io
|
|
21
|
+
import json
|
|
22
|
+
import os
|
|
23
|
+
import time
|
|
24
|
+
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
|
25
|
+
from urllib.parse import urlparse, parse_qs
|
|
26
|
+
|
|
27
|
+
from . import feed
|
|
28
|
+
from .feed import SOURCES
|
|
29
|
+
|
|
30
|
+
# In-memory cache: fetch is slow + upstreams rate-limit. Cache for N seconds.
|
|
31
|
+
_CACHE_TTL = int(os.environ.get("RJA_CACHE_TTL", "300")) # 5 min
|
|
32
|
+
_cache: dict = {"ts": 0.0, "jobs": []}
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _cached_all(force: bool = False) -> list:
|
|
36
|
+
now = time.time()
|
|
37
|
+
if not force and now - _cache["ts"] < _CACHE_TTL and _cache["jobs"]:
|
|
38
|
+
return _cache["jobs"]
|
|
39
|
+
jobs = feed.collect(per_source_limit=120)
|
|
40
|
+
_cache["ts"] = now
|
|
41
|
+
_cache["jobs"] = jobs
|
|
42
|
+
return jobs
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class Handler(BaseHTTPRequestHandler):
|
|
46
|
+
server_version = "RemoteJobsAPI/1.0"
|
|
47
|
+
|
|
48
|
+
def log_message(self, *a): # quiet
|
|
49
|
+
pass
|
|
50
|
+
|
|
51
|
+
def _send(self, code: int, body: bytes, ctype: str = "application/json"):
|
|
52
|
+
self.send_response(code)
|
|
53
|
+
self.send_header("Content-Type", ctype)
|
|
54
|
+
self.send_header("Content-Length", str(len(body)))
|
|
55
|
+
self.send_header("Access-Control-Allow-Origin", "*")
|
|
56
|
+
self.end_headers()
|
|
57
|
+
self.wfile.write(body)
|
|
58
|
+
|
|
59
|
+
def _json(self, code: int, obj):
|
|
60
|
+
self._send(code, json.dumps(obj, ensure_ascii=False).encode("utf-8"))
|
|
61
|
+
|
|
62
|
+
def do_OPTIONS(self):
|
|
63
|
+
self.send_response(204)
|
|
64
|
+
self.send_header("Access-Control-Allow-Origin", "*")
|
|
65
|
+
self.send_header("Access-Control-Allow-Methods", "GET, OPTIONS")
|
|
66
|
+
self.send_header("Access-Control-Allow-Headers", "*")
|
|
67
|
+
self.end_headers()
|
|
68
|
+
|
|
69
|
+
def do_GET(self):
|
|
70
|
+
parsed = urlparse(self.path)
|
|
71
|
+
path = parsed.path.rstrip("/") or "/"
|
|
72
|
+
q = parse_qs(parsed.query)
|
|
73
|
+
|
|
74
|
+
if path == "/health":
|
|
75
|
+
live = len(_cached_all())
|
|
76
|
+
return self._json(200, {"status": "ok", "sources": SOURCES, "count_live": live})
|
|
77
|
+
|
|
78
|
+
if path == "/v1/jobs/sources":
|
|
79
|
+
return self._json(200, {"sources": SOURCES})
|
|
80
|
+
|
|
81
|
+
if path == "/v1/jobs":
|
|
82
|
+
skills = [s.strip() for s in (q.get("skills", [""])[0]).split(",") if s.strip()]
|
|
83
|
+
source = (q.get("source", [""])[0] or "").strip() or None
|
|
84
|
+
if source and source not in SOURCES:
|
|
85
|
+
return self._json(400, {"error": f"unknown source {source!r}; choose from {SOURCES}"})
|
|
86
|
+
try:
|
|
87
|
+
limit = max(1, min(500, int(q.get("limit", ["50"])[0])))
|
|
88
|
+
except ValueError:
|
|
89
|
+
limit = 50
|
|
90
|
+
min_score = None
|
|
91
|
+
if q.get("min_score", [""])[0].strip():
|
|
92
|
+
try:
|
|
93
|
+
min_score = int(q["min_score"][0])
|
|
94
|
+
except ValueError:
|
|
95
|
+
return self._json(400, {"error": "min_score must be an int"})
|
|
96
|
+
fmt = (q.get("format", ["json"])[0] or "json").lower()
|
|
97
|
+
|
|
98
|
+
jobs = feed.filter_and_rank(
|
|
99
|
+
_cached_all(), skills=skills, source=source, min_score=min_score
|
|
100
|
+
)[:limit]
|
|
101
|
+
|
|
102
|
+
if fmt == "csv":
|
|
103
|
+
buf = io.StringIO()
|
|
104
|
+
cols = ["id", "title", "company", "location", "salary",
|
|
105
|
+
"category", "source", "published", "fit_score", "url"]
|
|
106
|
+
w = csv.DictWriter(buf, fieldnames=cols, extrasaction="ignore")
|
|
107
|
+
w.writeheader()
|
|
108
|
+
for j in jobs:
|
|
109
|
+
w.writerow(j)
|
|
110
|
+
return self._send(200, buf.getvalue().encode("utf-8"), "text/csv")
|
|
111
|
+
|
|
112
|
+
return self._json(200, {
|
|
113
|
+
"count": len(jobs),
|
|
114
|
+
"generated_at": int(time.time()),
|
|
115
|
+
"query": {"skills": skills, "source": source, "min_score": min_score, "limit": limit},
|
|
116
|
+
"jobs": jobs,
|
|
117
|
+
})
|
|
118
|
+
|
|
119
|
+
return self._json(404, {"error": "not found",
|
|
120
|
+
"routes": ["/health", "/v1/jobs", "/v1/jobs/sources"]})
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def main():
|
|
124
|
+
port = int(os.environ.get("PORT", "8321"))
|
|
125
|
+
srv = ThreadingHTTPServer(("0.0.0.0", port), Handler)
|
|
126
|
+
print(f"Remote-Jobs Data API listening on :{port}", flush=True)
|
|
127
|
+
srv.serve_forever()
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
if __name__ == "__main__":
|
|
131
|
+
main()
|
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: remote-jobs-api
|
|
3
|
+
Version: 1.1.0
|
|
4
|
+
Summary: Normalized remote-job data across 5 boards — one endpoint, one schema, skill fit-score
|
|
5
|
+
Author-email: Nova <earnnova@tten.no>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/earnnova-dev/remote-jobs-api
|
|
8
|
+
Project-URL: Documentation, https://earnnova-dev.github.io/remote-jobs-api/
|
|
9
|
+
Keywords: jobs,remote,api,scraping,feed,aggregator,data
|
|
10
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
11
|
+
Classifier: Intended Audience :: Developers
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
13
|
+
Classifier: Operating System :: OS Independent
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Topic :: Internet :: WWW/HTTP
|
|
20
|
+
Classifier: Topic :: Software Development :: Libraries
|
|
21
|
+
Requires-Python: >=3.9
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
License-File: LICENSE
|
|
24
|
+
Dynamic: license-file
|
|
25
|
+
|
|
26
|
+
# Remote Jobs API
|
|
27
|
+
|
|
28
|
+
**Live, normalized remote-job data across 5 boards — one clean endpoint, one schema.**
|
|
29
|
+
|
|
30
|
+
Stop scraping Remotive / RemoteOK / We Work Remotely / Jobicy / Hacker News
|
|
31
|
+
yourself. This API pulls them, normalizes every listing into a single
|
|
32
|
+
schema, dedupes, and hands you structured JSON (or CSV) — with an optional
|
|
33
|
+
**0-100 skill fit-score** for ranking.
|
|
34
|
+
|
|
35
|
+
Built for:
|
|
36
|
+
- Job boards & aggregator sites
|
|
37
|
+
- Recruiter and ATS tooling
|
|
38
|
+
- Lead-gen and market research
|
|
39
|
+
- **AI agents** that need structured job data in one call
|
|
40
|
+
|
|
41
|
+
## Why it's different
|
|
42
|
+
- **One schema, every board.** No per-board adapters on your side.
|
|
43
|
+
- **No session scraping.** Reads public job-board endpoints — your data never
|
|
44
|
+
leaves their servers, no browser, no account.
|
|
45
|
+
- **Self-hostable.** Stdlib-only Python (zero dependencies). Run it on a
|
|
46
|
+
laptop, a $5 VPS, or containerize it with the included Dockerfile.
|
|
47
|
+
- **Deterministic fit scoring.** No hidden LLM, no black box — reproducible
|
|
48
|
+
ranking you can reason about and unit-test.
|
|
49
|
+
|
|
50
|
+
## Endpoints
|
|
51
|
+
|
|
52
|
+
| Method | Path | Description |
|
|
53
|
+
|--------|------|-------------|
|
|
54
|
+
| GET | `/health` | Status + live job count |
|
|
55
|
+
| GET | `/v1/jobs/sources` | Available boards |
|
|
56
|
+
| GET | `/v1/jobs` | Normalized job listings |
|
|
57
|
+
|
|
58
|
+
### GET /v1/jobs
|
|
59
|
+
|
|
60
|
+
Query params:
|
|
61
|
+
|
|
62
|
+
| Param | Type | Notes |
|
|
63
|
+
|-------|------|-------|
|
|
64
|
+
| `skills` | string | Comma-separated keywords → adds `fit_score`, ranks best-first |
|
|
65
|
+
| `source` | string | `remotive` / `remoteok` / `jobicy` / `wwr` / `hn` |
|
|
66
|
+
| `limit` | int | 1-500 (default 50) |
|
|
67
|
+
| `min_score` | int | 0-100, filter by fit (needs `skills`) |
|
|
68
|
+
| `format` | string | `json` (default) or `csv` |
|
|
69
|
+
|
|
70
|
+
### Example
|
|
71
|
+
|
|
72
|
+
```bash
|
|
73
|
+
curl "https://<your-host>/v1/jobs?skills=python,backend,api&limit=3"
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
```json
|
|
77
|
+
{
|
|
78
|
+
"count": 3,
|
|
79
|
+
"generated_at": 1790433192,
|
|
80
|
+
"jobs": [
|
|
81
|
+
{
|
|
82
|
+
"id": "94516b10164d3c70",
|
|
83
|
+
"title": "Senior Backend Developer (Python)",
|
|
84
|
+
"company": "Proxify AB",
|
|
85
|
+
"url": "https://weworkremotely.com/remote-jobs/proxify-ab-senior-backend-developer-python-10",
|
|
86
|
+
"location": "Anywhere in the World",
|
|
87
|
+
"category": "Back-End Programming",
|
|
88
|
+
"source": "wwr",
|
|
89
|
+
"published": "Tue, 15 Sep 2026 09:01:36 +0000",
|
|
90
|
+
"fit_score": 80
|
|
91
|
+
}
|
|
92
|
+
]
|
|
93
|
+
}
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
## Run it yourself
|
|
97
|
+
|
|
98
|
+
Zero runtime dependencies — it's pure Python stdlib.
|
|
99
|
+
|
|
100
|
+
**Option A — CLI (query the feed directly, no server needed):**
|
|
101
|
+
|
|
102
|
+
```bash
|
|
103
|
+
pip install remote-jobs-api
|
|
104
|
+
remote-jobs-api --skills python,api --limit 5
|
|
105
|
+
remote-jobs-api --source wwr --format csv
|
|
106
|
+
remote-jobs-api --help
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
**Option B — HTTP API server (self-host):**
|
|
110
|
+
|
|
111
|
+
```bash
|
|
112
|
+
pip install remote-jobs-api
|
|
113
|
+
remote-jobs-api --serve --port 8321 # or: python -m remote_jobs_api.server
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
Or from source / Docker:
|
|
117
|
+
|
|
118
|
+
```bash
|
|
119
|
+
git clone https://github.com/earnnova-dev/remote-jobs-api
|
|
120
|
+
cd remote-jobs-api
|
|
121
|
+
python3 -m remote_jobs_api.server # listens on :8321
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
```bash
|
|
125
|
+
docker build -t remote-jobs-api .
|
|
126
|
+
docker run -p 8321:8321 remote-jobs-api
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
Env vars:
|
|
130
|
+
- `PORT` (default `8321`)
|
|
131
|
+
- `RJA_CACHE_TTL` (default `300` s) — how long to cache upstream fetches
|
|
132
|
+
|
|
133
|
+
## Testing
|
|
134
|
+
|
|
135
|
+
Offline unit tests (no network): `python -m pytest`.
|
|
136
|
+
|
|
137
|
+
## Pricing (indicative, for a hosted tier)
|
|
138
|
+
|
|
139
|
+
| Plan | Price | Includes |
|
|
140
|
+
|------|-------|----------|
|
|
141
|
+
| **Free** | $0 | 100 calls/mo, all boards, 50 results/req |
|
|
142
|
+
| **Pro** | $19/mo | 10k calls/mo, CSV, min_score, higher limits |
|
|
143
|
+
| **Team** | $49/mo | 100k calls/mo, SLA, webhooks (soon) |
|
|
144
|
+
|
|
145
|
+
> Self-hosted = free, MIT. The paid tier is the hosted, always-on,
|
|
146
|
+
> high-rate-limit version — the same engine, managed for you.
|
|
147
|
+
|
|
148
|
+
## License
|
|
149
|
+
|
|
150
|
+
MIT. See [LICENSE](LICENSE).
|
|
151
|
+
|
|
152
|
+
## Data & compliance
|
|
153
|
+
|
|
154
|
+
This API only reads public, no-auth job-board endpoints and returns data
|
|
155
|
+
exactly as those boards publish it. It does not store your personal data,
|
|
156
|
+
does not scrape authenticated sessions, and does not resell any board's
|
|
157
|
+
proprietary content beyond what is publicly viewable.
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
README.md
|
|
3
|
+
pyproject.toml
|
|
4
|
+
remote_jobs_api/__init__.py
|
|
5
|
+
remote_jobs_api/__main__.py
|
|
6
|
+
remote_jobs_api/cli.py
|
|
7
|
+
remote_jobs_api/feed.py
|
|
8
|
+
remote_jobs_api/server.py
|
|
9
|
+
remote_jobs_api.egg-info/PKG-INFO
|
|
10
|
+
remote_jobs_api.egg-info/SOURCES.txt
|
|
11
|
+
remote_jobs_api.egg-info/dependency_links.txt
|
|
12
|
+
remote_jobs_api.egg-info/entry_points.txt
|
|
13
|
+
remote_jobs_api.egg-info/top_level.txt
|
|
14
|
+
tests/test_feed.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
remote_jobs_api
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
"""Offline unit tests — no network.
|
|
2
|
+
|
|
3
|
+
Covers the normalization helpers, the fit-score, filter_and_rank, and the
|
|
4
|
+
CLI argument parsing, using in-memory fixtures.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import io
|
|
9
|
+
import json
|
|
10
|
+
import sys
|
|
11
|
+
from contextlib import redirect_stdout
|
|
12
|
+
|
|
13
|
+
import pytest
|
|
14
|
+
|
|
15
|
+
from remote_jobs_api import feed
|
|
16
|
+
from remote_jobs_api.feed import Job, _clean, _h, _match_score, filter_and_rank, SOURCES
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
# --- helpers -------------------------------------------------------------
|
|
20
|
+
|
|
21
|
+
def _job(**kw) -> Job:
|
|
22
|
+
base = dict(id="x", title="Senior Backend Developer", company="Acme",
|
|
23
|
+
url="https://example.com/j/x", source="test")
|
|
24
|
+
base.update(kw)
|
|
25
|
+
return Job(**base)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
# --- _clean --------------------------------------------------------------
|
|
29
|
+
|
|
30
|
+
def test_clean_strips_tags_and_whitespace():
|
|
31
|
+
assert _clean("<b> Hello </b>\n World ") == "Hello World"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def test_clean_unescapes_entities():
|
|
35
|
+
assert _clean("A & B") == "A & B"
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def test_clean_empty():
|
|
39
|
+
assert _clean("") == ""
|
|
40
|
+
assert _clean(None) == ""
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
# --- _h deterministic hash -----------------------------------------------
|
|
44
|
+
|
|
45
|
+
def test_hash_stable_and_short():
|
|
46
|
+
a = _h("hello")
|
|
47
|
+
b = _h("hello")
|
|
48
|
+
assert a == b
|
|
49
|
+
assert len(a) == 16
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def test_hash_differs():
|
|
53
|
+
assert _h("a") != _h("b")
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
# --- Job.to_dict lean schema --------------------------------------------
|
|
57
|
+
|
|
58
|
+
def test_to_dict_has_core_fields():
|
|
59
|
+
d = _job().to_dict()
|
|
60
|
+
for k in ("id", "title", "company", "url", "location", "salary",
|
|
61
|
+
"category", "tags", "published", "source"):
|
|
62
|
+
assert k in d
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def test_to_dict_truncates_description():
|
|
66
|
+
d = _job(description="x" * 2000).to_dict()
|
|
67
|
+
assert len(d["description"]) <= 500
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
# --- _match_score --------------------------------------------------------
|
|
71
|
+
|
|
72
|
+
def test_score_title_hit_higher_than_body():
|
|
73
|
+
title_hit = _job(title="Python Backend", description="")
|
|
74
|
+
body_hit = _job(title="Designer", description="we use python a lot")
|
|
75
|
+
assert _match_score(title_hit, ["python"]) > _match_score(body_hit, ["python"])
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def test_score_bounds():
|
|
79
|
+
assert _match_score(_job(), []) == 50
|
|
80
|
+
assert 0 <= _match_score(_job(title="Python Python Python"), ["python"]) <= 100
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def test_score_case_insensitive():
|
|
84
|
+
assert _match_score(_job(title="PYTHON"), ["python"]) >= 40
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
# --- filter_and_rank -----------------------------------------------------
|
|
88
|
+
|
|
89
|
+
def test_filter_source():
|
|
90
|
+
a = _job(id="a", source="remotive")
|
|
91
|
+
b = _job(id="b", source="wwr")
|
|
92
|
+
out = filter_and_rank([a, b], source="wwr")
|
|
93
|
+
assert [d["id"] for d in out] == ["b"]
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def test_filter_min_score():
|
|
97
|
+
high = _job(id="high", title="Python Senior")
|
|
98
|
+
low = _job(id="low", title="Designer")
|
|
99
|
+
out = filter_and_rank([high, low], skills=["python"], min_score=40)
|
|
100
|
+
assert all(d["fit_score"] >= 40 for d in out)
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def test_rank_best_fit_first():
|
|
104
|
+
# A job matching TWO distinct skills outranks one matching only one.
|
|
105
|
+
# (Repeating a single skill does not raise the score — each skill counts once.)
|
|
106
|
+
a = _job(id="a", title="Python Developer")
|
|
107
|
+
b = _job(id="b", title="Senior Python Engineer", tags=["api"])
|
|
108
|
+
out = filter_and_rank([a, b], skills=["python", "api"])
|
|
109
|
+
assert out[0]["id"] == "b"
|
|
110
|
+
assert out[0]["fit_score"] > out[1]["fit_score"]
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
# --- sources registry ----------------------------------------------------
|
|
114
|
+
|
|
115
|
+
def test_sources_known():
|
|
116
|
+
assert set(SOURCES) == {"remotive", "remoteok", "jobicy", "wwr", "hn"}
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
# --- CLI -----------------------------------------------------------------
|
|
120
|
+
|
|
121
|
+
def _run_cli(argv):
|
|
122
|
+
from remote_jobs_api import cli
|
|
123
|
+
buf = io.StringIO()
|
|
124
|
+
with redirect_stdout(buf):
|
|
125
|
+
# monkeypatch collect so the CLI stays offline
|
|
126
|
+
orig = feed.collect
|
|
127
|
+
feed.collect = lambda *a, **k: [
|
|
128
|
+
_job(id="a", title="Python Backend", source="remotive"),
|
|
129
|
+
_job(id="b", title="Growth Designer", source="wwr"),
|
|
130
|
+
]
|
|
131
|
+
try:
|
|
132
|
+
rc = cli.main(argv)
|
|
133
|
+
finally:
|
|
134
|
+
feed.collect = orig
|
|
135
|
+
return rc, buf.getvalue()
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def test_cli_json_output():
|
|
139
|
+
rc, out = _run_cli(["--skills", "python", "--limit", "5"])
|
|
140
|
+
assert rc == 0
|
|
141
|
+
data = json.loads(out)
|
|
142
|
+
assert data["count"] == 2
|
|
143
|
+
assert data["jobs"][0]["fit_score"] >= 40
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def test_cli_source_filter():
|
|
147
|
+
rc, out = _run_cli(["--source", "wwr"])
|
|
148
|
+
data = json.loads(out)
|
|
149
|
+
assert all(j["source"] == "wwr" for j in data["jobs"])
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def test_cli_csv_output():
|
|
153
|
+
rc, out = _run_cli(["--format", "csv"])
|
|
154
|
+
assert rc == 0
|
|
155
|
+
lines = out.strip().splitlines()
|
|
156
|
+
assert lines[0].startswith("id,")
|
|
157
|
+
assert len(lines) == 3 # header + 2 rows
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
if __name__ == "__main__":
|
|
161
|
+
sys.exit(pytest.main([__file__, "-v"]))
|