gitbleed 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gitbleed/__init__.py +0 -0
- gitbleed/main.py +1139 -0
- gitbleed-1.0.0.dist-info/METADATA +76 -0
- gitbleed-1.0.0.dist-info/RECORD +9 -0
- gitbleed-1.0.0.dist-info/WHEEL +5 -0
- gitbleed-1.0.0.dist-info/entry_points.txt +2 -0
- gitbleed-1.0.0.dist-info/licenses/LICENSE +201 -0
- gitbleed-1.0.0.dist-info/licenses/NOTICE +9 -0
- gitbleed-1.0.0.dist-info/top_level.txt +1 -0
gitbleed/main.py
ADDED
|
@@ -0,0 +1,1139 @@
|
|
|
1
|
+
# Copyright 2026 Oliver R. Calazans Jeronimo
|
|
2
|
+
# Modified from the original project git-dumper (Copyright (c) 2017 Maxime Arthaud)
|
|
3
|
+
#
|
|
4
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
5
|
+
# you may not use this file except in compliance with the License.
|
|
6
|
+
# You may obtain a copy of the License at
|
|
7
|
+
#
|
|
8
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
9
|
+
#
|
|
10
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
11
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
12
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
13
|
+
# See the License for the specific language governing permissions and
|
|
14
|
+
# limitations under the License.
|
|
15
|
+
|
|
16
|
+
import argparse
|
|
17
|
+
import multiprocessing
|
|
18
|
+
import os
|
|
19
|
+
import queue
|
|
20
|
+
import re
|
|
21
|
+
import subprocess
|
|
22
|
+
import sys
|
|
23
|
+
import time
|
|
24
|
+
import traceback
|
|
25
|
+
import urllib.parse
|
|
26
|
+
from contextlib import closing
|
|
27
|
+
from dataclasses import dataclass
|
|
28
|
+
from pathlib import Path
|
|
29
|
+
|
|
30
|
+
import bs4
|
|
31
|
+
import dulwich.index
|
|
32
|
+
import dulwich.objects
|
|
33
|
+
import dulwich.pack
|
|
34
|
+
import requests
|
|
35
|
+
import urllib3
|
|
36
|
+
from requests_pkcs12 import Pkcs12Adapter
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class GitBleed:
|
|
41
|
+
|
|
42
|
+
__slots__ = ('_args', '_session', '_response', '_environment')
|
|
43
|
+
|
|
44
|
+
def __init__(self):
|
|
45
|
+
self._args : Arguments = None
|
|
46
|
+
self._session : requests.Session = None
|
|
47
|
+
self._response : requests.Response = None
|
|
48
|
+
self._environment : dict[str, str] = None
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def execute(self):
|
|
53
|
+
self._get_args()
|
|
54
|
+
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
|
|
55
|
+
self._create_session()
|
|
56
|
+
self._find_base_url()
|
|
57
|
+
self._try_to_connect()
|
|
58
|
+
self._valid_response()
|
|
59
|
+
self._setup_env_for_proxy()
|
|
60
|
+
self._try_fast_dump() # if successful, it stops here
|
|
61
|
+
self._fetch_common_files()
|
|
62
|
+
self._discover_references()
|
|
63
|
+
self._fetch_git_packs()
|
|
64
|
+
self._discover_and_fetch_objects()
|
|
65
|
+
self._finalize_checkout()
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _get_args(self):
|
|
70
|
+
parser = Parser()
|
|
71
|
+
parser.parse()
|
|
72
|
+
self._args = parser.get_args()
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _create_session(self):
|
|
77
|
+
self._session = requests.Session()
|
|
78
|
+
self._session.verify = False
|
|
79
|
+
|
|
80
|
+
self._session.headers.update(self._args.http_headers)
|
|
81
|
+
self._session.headers.pop("Accept-Encoding", None)
|
|
82
|
+
self._configure_session()
|
|
83
|
+
|
|
84
|
+
if os.listdir(self._args.directory):
|
|
85
|
+
Display.warning(f"Destination '{self._args.directory}' is not empty")
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _configure_session(self):
|
|
90
|
+
if self._args.proxy:
|
|
91
|
+
self._session.proxies = {
|
|
92
|
+
"http": self._args.proxy,
|
|
93
|
+
"https": self._args.proxy,
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
if not self._args.cert_p12:
|
|
97
|
+
self._session.mount(
|
|
98
|
+
self._args.url,
|
|
99
|
+
requests.adapters.HTTPAdapter(max_retries=self._args.retry),
|
|
100
|
+
)
|
|
101
|
+
return
|
|
102
|
+
|
|
103
|
+
self._session.mount(
|
|
104
|
+
self._args.url,
|
|
105
|
+
Pkcs12Adapter(
|
|
106
|
+
pkcs12_filename=self._args.cert_p12,
|
|
107
|
+
pkcs12_password=self._args.cert_p12_pwd,
|
|
108
|
+
max_retries=self._args.retry,
|
|
109
|
+
),
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _find_base_url(self):
|
|
115
|
+
url: str = self._args.url
|
|
116
|
+
|
|
117
|
+
url = url.rstrip("/")
|
|
118
|
+
if url.endswith("HEAD"):
|
|
119
|
+
url = url[:-4]
|
|
120
|
+
|
|
121
|
+
url = url.rstrip("/")
|
|
122
|
+
if url.endswith(".git"):
|
|
123
|
+
url = url[:-4]
|
|
124
|
+
|
|
125
|
+
self._args.url = url.rstrip("/")
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _try_to_connect(self):
|
|
130
|
+
try:
|
|
131
|
+
self._response = self._session.get(
|
|
132
|
+
f"{self._args.url}/.git/HEAD",
|
|
133
|
+
timeout=self._args.timeout,
|
|
134
|
+
)
|
|
135
|
+
time.sleep(self._args.delay)
|
|
136
|
+
except Exception as e:
|
|
137
|
+
Display.fatal(f"Unable to connect to {self._args.url}. Error: {e}")
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _valid_response(self):
|
|
142
|
+
valid, _, error_msg = verify_response(self._response)
|
|
143
|
+
|
|
144
|
+
if self._response.status_code >= 400:
|
|
145
|
+
Display.fatal("Target unreachable. Dumping stopped")
|
|
146
|
+
|
|
147
|
+
elif self._response.status_code >= 300:
|
|
148
|
+
Display.warning(f"Redirection required to {self._response.headers['Location']}")
|
|
149
|
+
sys.exit(0)
|
|
150
|
+
|
|
151
|
+
if not valid:
|
|
152
|
+
Display.fatal(f"Invalid response from {self._response.url}: {error_msg}")
|
|
153
|
+
|
|
154
|
+
if not re.match(r"^(ref:.*|[0-9a-f]{40}$)", self._response.text.strip()):
|
|
155
|
+
Display.fatal(f"{self._response.url} is not a git HEAD file")
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _setup_env_for_proxy(self):
|
|
160
|
+
self._environment = os.environ.copy()
|
|
161
|
+
|
|
162
|
+
if not self._args.proxy:
|
|
163
|
+
return
|
|
164
|
+
|
|
165
|
+
self._environment["ALL_PROXY"] = self._args.proxy
|
|
166
|
+
self._environment["all_proxy"] = self._args.proxy
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def _try_fast_dump(self):
|
|
171
|
+
Display.section("Trying fast dumping")
|
|
172
|
+
|
|
173
|
+
response = self._session.get(f"{self._args.url}/.git/", allow_redirects=False)
|
|
174
|
+
time.sleep(self._args.delay)
|
|
175
|
+
|
|
176
|
+
Display.response(response)
|
|
177
|
+
|
|
178
|
+
if (
|
|
179
|
+
response.status_code != 200
|
|
180
|
+
or not is_html(response)
|
|
181
|
+
or "HEAD" not in get_indexed_files(response)
|
|
182
|
+
):
|
|
183
|
+
return
|
|
184
|
+
|
|
185
|
+
Display.section("Fetching .git recursively")
|
|
186
|
+
self._process_tasks([".git/", ".gitignore"], RecursiveDownloadWorker)
|
|
187
|
+
|
|
188
|
+
self._finalize_checkout()
|
|
189
|
+
sys.exit(0)
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def _process_tasks(self, initial_tasks: list, worker, tasks_done=None):
|
|
194
|
+
if not initial_tasks:
|
|
195
|
+
return
|
|
196
|
+
|
|
197
|
+
tasks_seen = set(tasks_done) if tasks_done else set()
|
|
198
|
+
pending_tasks = multiprocessing.Queue()
|
|
199
|
+
tasks_done = multiprocessing.Queue()
|
|
200
|
+
num_pending_tasks = 0
|
|
201
|
+
|
|
202
|
+
for task in initial_tasks:
|
|
203
|
+
assert task is not None
|
|
204
|
+
if task not in tasks_seen:
|
|
205
|
+
pending_tasks.put(task)
|
|
206
|
+
num_pending_tasks += 1
|
|
207
|
+
tasks_seen.add(task)
|
|
208
|
+
|
|
209
|
+
processes = [worker(pending_tasks, tasks_done, self._args) for _ in range(self._args.jobs)]
|
|
210
|
+
|
|
211
|
+
for p in processes:
|
|
212
|
+
p.start()
|
|
213
|
+
|
|
214
|
+
timed_out = False
|
|
215
|
+
|
|
216
|
+
while num_pending_tasks > 0:
|
|
217
|
+
try:
|
|
218
|
+
task_result = tasks_done.get(block=True, timeout=60)
|
|
219
|
+
except queue.Empty:
|
|
220
|
+
Display.warning("Timeout waiting for workers; aborting")
|
|
221
|
+
timed_out = True
|
|
222
|
+
break
|
|
223
|
+
|
|
224
|
+
num_pending_tasks -= 1
|
|
225
|
+
|
|
226
|
+
for task in task_result:
|
|
227
|
+
assert task is not None
|
|
228
|
+
if task not in tasks_seen:
|
|
229
|
+
pending_tasks.put(task)
|
|
230
|
+
num_pending_tasks += 1
|
|
231
|
+
tasks_seen.add(task)
|
|
232
|
+
|
|
233
|
+
if timed_out:
|
|
234
|
+
for p in processes: p.terminate()
|
|
235
|
+
for p in processes: p.join()
|
|
236
|
+
return
|
|
237
|
+
|
|
238
|
+
for _ in range(self._args.jobs):
|
|
239
|
+
pending_tasks.put(None)
|
|
240
|
+
|
|
241
|
+
for p in processes:
|
|
242
|
+
p.join()
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
UNSAFE = r"^\s*fsmonitor|sshcommand|askpass|editor|pager"
|
|
247
|
+
|
|
248
|
+
def _sanitize_file(self, filepath=".git/config"):
|
|
249
|
+
if not os.path.isfile(filepath):
|
|
250
|
+
Display.warning(".git/config not found")
|
|
251
|
+
return
|
|
252
|
+
|
|
253
|
+
Display.section("Sanitizing .git/config")
|
|
254
|
+
|
|
255
|
+
with open(filepath, 'r+') as f:
|
|
256
|
+
content = f.read()
|
|
257
|
+
modified_content = re.sub(self.UNSAFE, r'# \g<0>', content, flags=re.IGNORECASE)
|
|
258
|
+
|
|
259
|
+
if content != modified_content:
|
|
260
|
+
Display.warning(f"'{filepath}' file was altered")
|
|
261
|
+
f.seek(0)
|
|
262
|
+
f.write(modified_content)
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def _fetch_common_files(self):
|
|
267
|
+
Display.section("Fetching common files")
|
|
268
|
+
|
|
269
|
+
TASKS = [
|
|
270
|
+
".gitignore",
|
|
271
|
+
".git/COMMIT_EDITMSG",
|
|
272
|
+
".git/description",
|
|
273
|
+
".git/hooks/applypatch-msg.sample",
|
|
274
|
+
".git/hooks/commit-msg.sample",
|
|
275
|
+
".git/hooks/post-commit.sample",
|
|
276
|
+
".git/hooks/post-receive.sample",
|
|
277
|
+
".git/hooks/post-update.sample",
|
|
278
|
+
".git/hooks/pre-applypatch.sample",
|
|
279
|
+
".git/hooks/pre-commit.sample",
|
|
280
|
+
".git/hooks/pre-push.sample",
|
|
281
|
+
".git/hooks/pre-rebase.sample",
|
|
282
|
+
".git/hooks/pre-receive.sample",
|
|
283
|
+
".git/hooks/prepare-commit-msg.sample",
|
|
284
|
+
".git/hooks/update.sample",
|
|
285
|
+
".git/index",
|
|
286
|
+
".git/info/exclude",
|
|
287
|
+
".git/objects/info/packs",
|
|
288
|
+
]
|
|
289
|
+
|
|
290
|
+
self._process_tasks(TASKS, DownloadWorker)
|
|
291
|
+
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
def _discover_references(self):
|
|
295
|
+
Display.section("Finding refs/")
|
|
296
|
+
|
|
297
|
+
TASKS = [
|
|
298
|
+
".git/FETCH_HEAD",
|
|
299
|
+
".git/HEAD",
|
|
300
|
+
".git/ORIG_HEAD",
|
|
301
|
+
".git/config",
|
|
302
|
+
".git/info/refs",
|
|
303
|
+
".git/logs/HEAD",
|
|
304
|
+
".git/logs/refs/heads/main",
|
|
305
|
+
".git/logs/refs/heads/master",
|
|
306
|
+
".git/logs/refs/heads/staging",
|
|
307
|
+
".git/logs/refs/heads/production",
|
|
308
|
+
".git/logs/refs/heads/development",
|
|
309
|
+
".git/logs/refs/remotes/origin/HEAD",
|
|
310
|
+
".git/logs/refs/remotes/origin/main",
|
|
311
|
+
".git/logs/refs/remotes/origin/master",
|
|
312
|
+
".git/logs/refs/remotes/origin/staging",
|
|
313
|
+
".git/logs/refs/remotes/origin/production",
|
|
314
|
+
".git/logs/refs/remotes/origin/development",
|
|
315
|
+
".git/logs/refs/stash",
|
|
316
|
+
".git/packed-refs",
|
|
317
|
+
".git/refs/heads/main",
|
|
318
|
+
".git/refs/heads/master",
|
|
319
|
+
".git/refs/heads/staging",
|
|
320
|
+
".git/refs/heads/production",
|
|
321
|
+
".git/refs/heads/development",
|
|
322
|
+
".git/refs/remotes/origin/HEAD",
|
|
323
|
+
".git/refs/remotes/origin/main",
|
|
324
|
+
".git/refs/remotes/origin/master",
|
|
325
|
+
".git/refs/remotes/origin/staging",
|
|
326
|
+
".git/refs/remotes/origin/production",
|
|
327
|
+
".git/refs/remotes/origin/development",
|
|
328
|
+
".git/refs/stash",
|
|
329
|
+
".git/refs/wip/wtree/refs/heads/main",
|
|
330
|
+
".git/refs/wip/wtree/refs/heads/master",
|
|
331
|
+
".git/refs/wip/wtree/refs/heads/staging",
|
|
332
|
+
".git/refs/wip/wtree/refs/heads/production",
|
|
333
|
+
".git/refs/wip/wtree/refs/heads/development",
|
|
334
|
+
".git/refs/wip/index/refs/heads/main",
|
|
335
|
+
".git/refs/wip/index/refs/heads/master",
|
|
336
|
+
".git/refs/wip/index/refs/heads/staging",
|
|
337
|
+
".git/refs/wip/index/refs/heads/production",
|
|
338
|
+
".git/refs/wip/index/refs/heads/development",
|
|
339
|
+
]
|
|
340
|
+
|
|
341
|
+
self._add_user_specified_branches(TASKS)
|
|
342
|
+
self._process_tasks(TASKS, FindRefsWorker)
|
|
343
|
+
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
def _add_user_specified_branches(self, tasks: list):
|
|
347
|
+
if not self._args.branches:
|
|
348
|
+
return
|
|
349
|
+
|
|
350
|
+
for branch in self._args.branches:
|
|
351
|
+
if not re.match(r'^[A-Za-z0-9\-\._]+$', branch):
|
|
352
|
+
Display.warning(f"Ignoring invalid branch name '{branch}'")
|
|
353
|
+
continue
|
|
354
|
+
|
|
355
|
+
tasks.extend([
|
|
356
|
+
f".git/logs/refs/heads/{branch}",
|
|
357
|
+
f".git/refs/heads/{branch}",
|
|
358
|
+
f".git/logs/refs/remotes/origin/{branch}",
|
|
359
|
+
f".git/refs/remotes/origin/{branch}",
|
|
360
|
+
f".git/refs/wip/wtree/refs/heads/{branch}",
|
|
361
|
+
f".git/refs/wip/index/refs/heads/{branch}",
|
|
362
|
+
])
|
|
363
|
+
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
def _fetch_git_packs(self):
|
|
367
|
+
Display.section("Finding packs")
|
|
368
|
+
tasks = []
|
|
369
|
+
|
|
370
|
+
info_packs_path = os.path.join(
|
|
371
|
+
self._args.directory, ".git", "objects", "info", "packs"
|
|
372
|
+
)
|
|
373
|
+
|
|
374
|
+
if os.path.exists(info_packs_path):
|
|
375
|
+
with open(info_packs_path, "r") as f:
|
|
376
|
+
info_packs = f.read()
|
|
377
|
+
|
|
378
|
+
for sha1 in re.findall(r"pack-([a-f0-9]{40})\.pack", info_packs):
|
|
379
|
+
tasks.append(".git/objects/pack/pack-%s.idx" % sha1)
|
|
380
|
+
tasks.append(".git/objects/pack/pack-%s.pack" % sha1)
|
|
381
|
+
|
|
382
|
+
self._process_tasks(tasks, DownloadWorker)
|
|
383
|
+
|
|
384
|
+
|
|
385
|
+
|
|
386
|
+
def _discover_and_fetch_objects(self):
|
|
387
|
+
Display.section("Finding objects")
|
|
388
|
+
objs = set()
|
|
389
|
+
packed_objs = set()
|
|
390
|
+
|
|
391
|
+
files = [
|
|
392
|
+
os.path.join(self._args.directory, ".git", "packed-refs"),
|
|
393
|
+
os.path.join(self._args.directory, ".git", "info", "refs"),
|
|
394
|
+
os.path.join(self._args.directory, ".git", "FETCH_HEAD"),
|
|
395
|
+
os.path.join(self._args.directory, ".git", "ORIG_HEAD"),
|
|
396
|
+
]
|
|
397
|
+
|
|
398
|
+
self._collect_file_paths_from_git_subdir(files, "refs")
|
|
399
|
+
self._collect_file_paths_from_git_subdir(files, "logs")
|
|
400
|
+
self._extract_sha1_hashes_from_files(files, objs)
|
|
401
|
+
self._parse_staging_area(objs)
|
|
402
|
+
self._process_pack_files(packed_objs, objs)
|
|
403
|
+
|
|
404
|
+
Display.section("Fetching objects")
|
|
405
|
+
self._process_tasks(objs, FindObjectsWorker, tasks_done=packed_objs)
|
|
406
|
+
|
|
407
|
+
|
|
408
|
+
|
|
409
|
+
def _collect_file_paths_from_git_subdir(self, files: list, subpath: str):
|
|
410
|
+
base_path = os.path.join(self._args.directory, ".git", subpath)
|
|
411
|
+
|
|
412
|
+
if not os.path.isdir(base_path):
|
|
413
|
+
return
|
|
414
|
+
|
|
415
|
+
for dirpath, _, filenames in os.walk(base_path):
|
|
416
|
+
for filename in filenames:
|
|
417
|
+
files.append(os.path.join(dirpath, filename))
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
|
|
421
|
+
@staticmethod
|
|
422
|
+
def _extract_sha1_hashes_from_files(files: list, objs: set):
|
|
423
|
+
for filepath in files:
|
|
424
|
+
if not os.path.isfile(filepath):
|
|
425
|
+
continue
|
|
426
|
+
|
|
427
|
+
try:
|
|
428
|
+
with open(filepath, "r", encoding='utf-8', errors='ignore') as f:
|
|
429
|
+
content = f.read()
|
|
430
|
+
except Exception:
|
|
431
|
+
continue
|
|
432
|
+
|
|
433
|
+
for match in re.findall(r"(^|\s)([a-f0-9]{40})($|\s)", content):
|
|
434
|
+
objs.add(match[1])
|
|
435
|
+
|
|
436
|
+
|
|
437
|
+
|
|
438
|
+
def _parse_staging_area(self, objs: set):
|
|
439
|
+
index_path = os.path.join(self._args.directory, ".git", "index")
|
|
440
|
+
|
|
441
|
+
if not os.path.exists(index_path):
|
|
442
|
+
return
|
|
443
|
+
|
|
444
|
+
try:
|
|
445
|
+
index = dulwich.index.Index(index_path)
|
|
446
|
+
for entry in index.iterobjects():
|
|
447
|
+
objs.add(entry[1].decode())
|
|
448
|
+
except Exception:
|
|
449
|
+
pass
|
|
450
|
+
|
|
451
|
+
|
|
452
|
+
|
|
453
|
+
def _process_pack_files(self, packed_objs: set, objs: set):
|
|
454
|
+
pack_file_dir = os.path.join(self._args.directory, ".git", "objects", "pack")
|
|
455
|
+
|
|
456
|
+
if not os.path.isdir(pack_file_dir):
|
|
457
|
+
return
|
|
458
|
+
|
|
459
|
+
for filename in os.listdir(pack_file_dir):
|
|
460
|
+
if not filename.startswith("pack-") or not filename.endswith(".pack"):
|
|
461
|
+
continue
|
|
462
|
+
|
|
463
|
+
try:
|
|
464
|
+
pack_data_path = os.path.join(pack_file_dir, filename)
|
|
465
|
+
pack_idx_path = os.path.join(pack_file_dir, filename[:-5] + ".idx")
|
|
466
|
+
pack_data = dulwich.pack.PackData(pack_data_path)
|
|
467
|
+
pack_idx = dulwich.pack.load_pack_index(pack_idx_path)
|
|
468
|
+
pack = dulwich.pack.Pack.from_objects(pack_data, pack_idx)
|
|
469
|
+
|
|
470
|
+
for obj_file in pack.iterobjects():
|
|
471
|
+
packed_objs.add(obj_file.sha().hexdigest())
|
|
472
|
+
objs |= set(get_referenced_sha1(obj_file))
|
|
473
|
+
|
|
474
|
+
except Exception:
|
|
475
|
+
continue
|
|
476
|
+
|
|
477
|
+
|
|
478
|
+
|
|
479
|
+
def _finalize_checkout(self):
|
|
480
|
+
Display.section("Running git checkout")
|
|
481
|
+
|
|
482
|
+
cwd = os.getcwd()
|
|
483
|
+
try:
|
|
484
|
+
os.chdir(self._args.directory)
|
|
485
|
+
self._sanitize_file()
|
|
486
|
+
|
|
487
|
+
subprocess.call(
|
|
488
|
+
["git", "checkout", "."],
|
|
489
|
+
stderr=subprocess.DEVNULL,
|
|
490
|
+
env=self._environment,
|
|
491
|
+
)
|
|
492
|
+
|
|
493
|
+
finally:
|
|
494
|
+
os.chdir(cwd)
|
|
495
|
+
Display.success(f"Repository saved to {self._args.directory}")
|
|
496
|
+
|
|
497
|
+
|
|
498
|
+
|
|
499
|
+
|
|
500
|
+
|
|
501
|
+
@dataclass(slots=True)
|
|
502
|
+
class Arguments:
|
|
503
|
+
cert_p12_pwd : str
|
|
504
|
+
cert_p12 : str
|
|
505
|
+
url : str
|
|
506
|
+
directory : str
|
|
507
|
+
proxy : str
|
|
508
|
+
jobs : int
|
|
509
|
+
retry : int
|
|
510
|
+
timeout : int
|
|
511
|
+
http_headers : dict[str, str]
|
|
512
|
+
branches : list[str]
|
|
513
|
+
delay : float
|
|
514
|
+
|
|
515
|
+
|
|
516
|
+
|
|
517
|
+
class Parser:
|
|
518
|
+
|
|
519
|
+
__slots__ = ('_args', '_parser')
|
|
520
|
+
|
|
521
|
+
def __init__(self):
|
|
522
|
+
self._args : argparse.Namespace = None
|
|
523
|
+
self._parser : argparse.ArgumentParser = None
|
|
524
|
+
|
|
525
|
+
|
|
526
|
+
|
|
527
|
+
def parse(self):
|
|
528
|
+
self._create_args()
|
|
529
|
+
self._args = self._parser.parse_args()
|
|
530
|
+
self._valid_jobs()
|
|
531
|
+
self._valid_retry()
|
|
532
|
+
self._valid_timeout()
|
|
533
|
+
self._valid_proxy()
|
|
534
|
+
self._valid_certificate()
|
|
535
|
+
self._valid_delay()
|
|
536
|
+
self._create_dir()
|
|
537
|
+
|
|
538
|
+
|
|
539
|
+
|
|
540
|
+
def _create_args(self):
|
|
541
|
+
self._parser = argparse.ArgumentParser(
|
|
542
|
+
usage="gitbleed --url example.com <ARGS>",
|
|
543
|
+
description="Dump a git repository from a website.",
|
|
544
|
+
formatter_class=lambda prog: argparse.HelpFormatter(
|
|
545
|
+
prog,
|
|
546
|
+
max_help_position=40,
|
|
547
|
+
width=100,
|
|
548
|
+
),
|
|
549
|
+
)
|
|
550
|
+
self._parser.add_argument(
|
|
551
|
+
"--url", type=str, required=True,
|
|
552
|
+
help="URL of the git repository to dump"
|
|
553
|
+
)
|
|
554
|
+
self._parser.add_argument("--proxy", help="Proxy URL, e.g. `socks5://127.0.0.1:9050`")
|
|
555
|
+
self._parser.add_argument("--cert-p12", help="Client certificate in PKCS#12")
|
|
556
|
+
self._parser.add_argument("--cert-p12-pwd", help="Password for the client certificate")
|
|
557
|
+
self._parser.add_argument(
|
|
558
|
+
"-d", "--delay", type=float, default=0,
|
|
559
|
+
help="Delay in seconds between requests"
|
|
560
|
+
)
|
|
561
|
+
self._parser.add_argument(
|
|
562
|
+
"-o", "--output", type=str, default="",
|
|
563
|
+
help="Directory where the repository will be saved",
|
|
564
|
+
)
|
|
565
|
+
self._parser.add_argument(
|
|
566
|
+
"-j", "--jobs", type=int, default=10,
|
|
567
|
+
help="Number of simultaneous requests",
|
|
568
|
+
)
|
|
569
|
+
self._parser.add_argument(
|
|
570
|
+
"-r", "--retry", type=int, default=3,
|
|
571
|
+
help="Number of request attempts before giving up",
|
|
572
|
+
)
|
|
573
|
+
self._parser.add_argument(
|
|
574
|
+
"-t", "--timeout", type=int, default=3,
|
|
575
|
+
help="Maximum time in seconds per request",
|
|
576
|
+
)
|
|
577
|
+
self._parser.add_argument(
|
|
578
|
+
"-u", "--user-agent", type=str,
|
|
579
|
+
default="Mozilla/5.0 (Windows NT 10.0; rv:78.0) Gecko/20100101 Firefox/78.0",
|
|
580
|
+
help="User-agent to use for requests",
|
|
581
|
+
)
|
|
582
|
+
self._parser.add_argument(
|
|
583
|
+
"-H", "--header", type=str, action="append",
|
|
584
|
+
help="Additional http headers, e.g `NAME=VALUE`",
|
|
585
|
+
)
|
|
586
|
+
self._parser.add_argument(
|
|
587
|
+
"-b", "--branch", dest="branches", action="append",
|
|
588
|
+
help=(
|
|
589
|
+
"Additional branch names to check for, e.g. `-b dev -b prod`. "
|
|
590
|
+
"The default branches (`main`, `master`, `staging`, `production`, "
|
|
591
|
+
"`development`) are always checked."
|
|
592
|
+
),
|
|
593
|
+
)
|
|
594
|
+
|
|
595
|
+
|
|
596
|
+
def _valid_jobs(self):
|
|
597
|
+
if self._args.jobs < 1:
|
|
598
|
+
self._parser.error(f"Invalid number of jobs, got {self._args.jobs}")
|
|
599
|
+
|
|
600
|
+
|
|
601
|
+
def _valid_retry(self):
|
|
602
|
+
if self._args.retry < 1:
|
|
603
|
+
self._parser.error(f"Invalid number of retries, got {self._args.retry}")
|
|
604
|
+
|
|
605
|
+
|
|
606
|
+
def _valid_timeout(self):
|
|
607
|
+
if self._args.timeout < 1:
|
|
608
|
+
self._parser.error(f"Invalid timeout, got {self._args.timeout}")
|
|
609
|
+
|
|
610
|
+
|
|
611
|
+
def _valid_delay(self):
|
|
612
|
+
if self._args.delay < 0:
|
|
613
|
+
self._parser.error(f"Delay value cannot be negative. Got {self._args.delay}")
|
|
614
|
+
|
|
615
|
+
|
|
616
|
+
def _valid_proxy(self):
|
|
617
|
+
if not self._args.proxy:
|
|
618
|
+
return
|
|
619
|
+
|
|
620
|
+
raw = self._args.proxy
|
|
621
|
+
if "://" not in raw:
|
|
622
|
+
raw = "socks5://" + raw
|
|
623
|
+
|
|
624
|
+
try:
|
|
625
|
+
parsed = urllib.parse.urlparse(raw)
|
|
626
|
+
except ValueError as e:
|
|
627
|
+
self._parser.error(f"Invalid proxy, got {self._args.proxy}: {e}")
|
|
628
|
+
|
|
629
|
+
if parsed.scheme not in ("socks5", "socks5h", "socks4", "socks4a", "http", "https"):
|
|
630
|
+
self._parser.error(
|
|
631
|
+
f"Unsupported proxy scheme '{parsed.scheme}' in {self._args.proxy}"
|
|
632
|
+
)
|
|
633
|
+
|
|
634
|
+
if not parsed.hostname or not parsed.port:
|
|
635
|
+
self._parser.error(f"Invalid proxy, got {self._args.proxy}")
|
|
636
|
+
|
|
637
|
+
self._args.proxy = f"{parsed.scheme}://{parsed.netloc}"
|
|
638
|
+
|
|
639
|
+
|
|
640
|
+
|
|
641
|
+
def _create_dir(self):
|
|
642
|
+
if self._args.output == "":
|
|
643
|
+
self._args.output = Path(__file__).resolve().parent
|
|
644
|
+
|
|
645
|
+
if not os.path.isdir(self._args.output):
|
|
646
|
+
self._parser.error(f"{self._args.output} is not a directory")
|
|
647
|
+
|
|
648
|
+
url = re.sub(r"^https?://", "", self._args.url)
|
|
649
|
+
self._args.output = os.path.join(self._args.output, url)
|
|
650
|
+
|
|
651
|
+
if not os.path.exists(self._args.output):
|
|
652
|
+
os.makedirs(self._args.output)
|
|
653
|
+
|
|
654
|
+
|
|
655
|
+
|
|
656
|
+
def _valid_certificate(self):
|
|
657
|
+
if not self._args.cert_p12:
|
|
658
|
+
return
|
|
659
|
+
|
|
660
|
+
if not os.path.exists(self._args.cert_p12):
|
|
661
|
+
self._parser.error(
|
|
662
|
+
f"Client certificate {self._args.cert_p12} does not exist"
|
|
663
|
+
)
|
|
664
|
+
|
|
665
|
+
if not os.path.isfile(self._args.cert_p12):
|
|
666
|
+
self._parser.error(
|
|
667
|
+
f"Client certificate {self._args.cert_p12} is not a file"
|
|
668
|
+
)
|
|
669
|
+
|
|
670
|
+
if not self._args.cert_p12_pwd:
|
|
671
|
+
self._parser.error("Client certificate password is required")
|
|
672
|
+
|
|
673
|
+
|
|
674
|
+
|
|
675
|
+
def _valid_headers(self) -> dict:
|
|
676
|
+
http_headers = {
|
|
677
|
+
"User-Agent": self._args.user_agent,
|
|
678
|
+
"Accept": "*/*",
|
|
679
|
+
}
|
|
680
|
+
|
|
681
|
+
if not self._args.header:
|
|
682
|
+
return http_headers
|
|
683
|
+
|
|
684
|
+
for header in self._args.header:
|
|
685
|
+
tokens: list[str] = header.split("=", maxsplit=1)
|
|
686
|
+
|
|
687
|
+
if len(tokens) != 2:
|
|
688
|
+
self._parser.error(
|
|
689
|
+
f"HTTP header must have the form NAME=VALUE, got {header}"
|
|
690
|
+
)
|
|
691
|
+
|
|
692
|
+
name, value = tokens
|
|
693
|
+
http_headers[name.strip()] = value.strip()
|
|
694
|
+
|
|
695
|
+
return http_headers
|
|
696
|
+
|
|
697
|
+
|
|
698
|
+
|
|
699
|
+
def get_args(self) -> Arguments:
|
|
700
|
+
return Arguments(
|
|
701
|
+
cert_p12_pwd = self._args.cert_p12_pwd,
|
|
702
|
+
cert_p12 = self._args.cert_p12,
|
|
703
|
+
directory = self._args.output,
|
|
704
|
+
proxy = self._args.proxy,
|
|
705
|
+
url = self._args.url,
|
|
706
|
+
jobs = self._args.jobs,
|
|
707
|
+
retry = self._args.retry,
|
|
708
|
+
timeout = self._args.timeout,
|
|
709
|
+
http_headers = self._valid_headers(),
|
|
710
|
+
branches = self._args.branches,
|
|
711
|
+
delay = self._args.delay,
|
|
712
|
+
)
|
|
713
|
+
|
|
714
|
+
|
|
715
|
+
|
|
716
|
+
|
|
717
|
+
class Display:
|
|
718
|
+
|
|
719
|
+
HTML: str = " [\033[34mHTML\033[0m] "
|
|
720
|
+
|
|
721
|
+
@staticmethod
|
|
722
|
+
def response(response: requests.Response):
|
|
723
|
+
code = response.status_code
|
|
724
|
+
|
|
725
|
+
if code >= 400: x = f"\033[31m{code}\033[0m"
|
|
726
|
+
elif code >= 300: x = f"\033[33m{code}\033[0m"
|
|
727
|
+
elif code >= 200: x = f"\033[32m{code}\033[0m"
|
|
728
|
+
else: x = f"{code}"
|
|
729
|
+
|
|
730
|
+
z = Display.HTML if is_html(response) else ' '
|
|
731
|
+
|
|
732
|
+
print(f"[{x}]{z}{response.url}", flush=True)
|
|
733
|
+
|
|
734
|
+
|
|
735
|
+
@staticmethod
|
|
736
|
+
def section(text: str):
|
|
737
|
+
print(f"[###] {text}", flush=True)
|
|
738
|
+
|
|
739
|
+
|
|
740
|
+
@staticmethod
|
|
741
|
+
def warning(text: str):
|
|
742
|
+
print(f"[\033[33m{'!!!'}\033[0m] {text}", flush=True)
|
|
743
|
+
|
|
744
|
+
|
|
745
|
+
@staticmethod
|
|
746
|
+
def fatal(text: str):
|
|
747
|
+
print(f"[\033[31m{'ERR'}\033[0m] {text}", flush=True)
|
|
748
|
+
sys.exit(1)
|
|
749
|
+
|
|
750
|
+
|
|
751
|
+
@staticmethod
|
|
752
|
+
def success(text: str):
|
|
753
|
+
print(f"[\033[32m{'OK!'}\033[0m] {text}", flush=True)
|
|
754
|
+
|
|
755
|
+
|
|
756
|
+
|
|
757
|
+
|
|
758
|
+
def is_html(response: requests.Response) -> bool:
|
|
759
|
+
return (
|
|
760
|
+
"Content-Type" in response.headers
|
|
761
|
+
and "text/html" in response.headers["Content-Type"]
|
|
762
|
+
)
|
|
763
|
+
|
|
764
|
+
|
|
765
|
+
|
|
766
|
+
def is_safe_path(path: str) -> bool:
|
|
767
|
+
""" Prevents directory traversal attacks """
|
|
768
|
+
if path.startswith("/"):
|
|
769
|
+
return False
|
|
770
|
+
|
|
771
|
+
safe_path = os.path.expanduser("~")
|
|
772
|
+
try:
|
|
773
|
+
real = os.path.realpath(os.path.join(safe_path, path))
|
|
774
|
+
return os.path.commonpath((real, safe_path)) == safe_path
|
|
775
|
+
|
|
776
|
+
except ValueError:
|
|
777
|
+
return False
|
|
778
|
+
|
|
779
|
+
|
|
780
|
+
|
|
781
|
+
def get_indexed_files(response: requests.Response) -> list:
|
|
782
|
+
html = bs4.BeautifulSoup(response.text, "html.parser")
|
|
783
|
+
files = []
|
|
784
|
+
|
|
785
|
+
for link in html.find_all("a"):
|
|
786
|
+
url = urllib.parse.urlparse(link.get("href"))
|
|
787
|
+
|
|
788
|
+
if (
|
|
789
|
+
url.path
|
|
790
|
+
and is_safe_path(url.path)
|
|
791
|
+
and not url.scheme
|
|
792
|
+
and not url.netloc
|
|
793
|
+
):
|
|
794
|
+
files.append(url.path)
|
|
795
|
+
|
|
796
|
+
return files
|
|
797
|
+
|
|
798
|
+
|
|
799
|
+
|
|
800
|
+
def verify_response(response: requests.Response) -> tuple[bool, bool, str]:
|
|
801
|
+
Display.response(response)
|
|
802
|
+
|
|
803
|
+
if response.status_code >= 400:
|
|
804
|
+
return False, False, None
|
|
805
|
+
|
|
806
|
+
elif response.status_code >= 300 and "Location" in response.headers:
|
|
807
|
+
return False, True, (
|
|
808
|
+
f"Moved to {response.headers['Location']}. Code {response.status_code}"
|
|
809
|
+
)
|
|
810
|
+
|
|
811
|
+
elif (
|
|
812
|
+
"Content-Length" in response.headers
|
|
813
|
+
and response.headers["Content-Length"] == 0
|
|
814
|
+
):
|
|
815
|
+
return False, True, "Responded with a zero-length body"
|
|
816
|
+
|
|
817
|
+
elif is_html(response):
|
|
818
|
+
return False, True, f"{response.url} responded with a HTML"
|
|
819
|
+
|
|
820
|
+
else:
|
|
821
|
+
return True, False, None
|
|
822
|
+
|
|
823
|
+
|
|
824
|
+
|
|
825
|
+
def create_intermediate_dirs(path: str):
|
|
826
|
+
dirname, basename = os.path.split(path)
|
|
827
|
+
|
|
828
|
+
if dirname and not os.path.exists(dirname):
|
|
829
|
+
try:
|
|
830
|
+
os.makedirs(dirname)
|
|
831
|
+
except FileExistsError:
|
|
832
|
+
pass
|
|
833
|
+
|
|
834
|
+
|
|
835
|
+
|
|
836
|
+
def get_referenced_sha1(obj_file):
|
|
837
|
+
""" Return all the referenced SHA1 in the given object file """
|
|
838
|
+
objs = []
|
|
839
|
+
|
|
840
|
+
if isinstance(obj_file, dulwich.objects.Commit):
|
|
841
|
+
objs.append(obj_file.tree.decode())
|
|
842
|
+
for parent in obj_file.parents:
|
|
843
|
+
objs.append(parent.decode())
|
|
844
|
+
|
|
845
|
+
elif isinstance(obj_file, dulwich.objects.Tree):
|
|
846
|
+
for item in obj_file.iteritems():
|
|
847
|
+
objs.append(item.sha.decode())
|
|
848
|
+
|
|
849
|
+
elif isinstance(obj_file, dulwich.objects.Tag):
|
|
850
|
+
objs.append(obj_file.object[1].decode())
|
|
851
|
+
|
|
852
|
+
elif isinstance(obj_file, dulwich.objects.Blob):
|
|
853
|
+
pass
|
|
854
|
+
|
|
855
|
+
else:
|
|
856
|
+
Display.warning(f"Unexpected object type: {obj_file}")
|
|
857
|
+
|
|
858
|
+
return objs
|
|
859
|
+
|
|
860
|
+
|
|
861
|
+
|
|
862
|
+
|
|
863
|
+
class Worker(multiprocessing.Process):
|
|
864
|
+
|
|
865
|
+
def __init__(
|
|
866
|
+
self,
|
|
867
|
+
pending_tasks : multiprocessing.Queue,
|
|
868
|
+
tasks_done : multiprocessing.Queue,
|
|
869
|
+
args : Arguments,
|
|
870
|
+
):
|
|
871
|
+
super().__init__()
|
|
872
|
+
self.daemon : bool = True
|
|
873
|
+
self.pending_tasks : multiprocessing.Queue = pending_tasks
|
|
874
|
+
self.tasks_done : multiprocessing.Queue = tasks_done
|
|
875
|
+
self.args : Arguments = args
|
|
876
|
+
|
|
877
|
+
|
|
878
|
+
|
|
879
|
+
def run(self):
|
|
880
|
+
self.init(self.args)
|
|
881
|
+
|
|
882
|
+
while True:
|
|
883
|
+
task = self.pending_tasks.get(block=True)
|
|
884
|
+
|
|
885
|
+
if task is None: # end signal
|
|
886
|
+
return
|
|
887
|
+
|
|
888
|
+
try:
|
|
889
|
+
result = self.do_task(task, self.args)
|
|
890
|
+
except Exception:
|
|
891
|
+
Display.warning(f"Task {task} raised exception:")
|
|
892
|
+
traceback.print_exc()
|
|
893
|
+
result = []
|
|
894
|
+
|
|
895
|
+
if not isinstance(result, list):
|
|
896
|
+
Display.warning(
|
|
897
|
+
f"do_task returned {type(result).__name__}, expected list"
|
|
898
|
+
)
|
|
899
|
+
result = []
|
|
900
|
+
|
|
901
|
+
self.tasks_done.put(result)
|
|
902
|
+
|
|
903
|
+
def init(self, args: Arguments):
|
|
904
|
+
raise NotImplementedError
|
|
905
|
+
|
|
906
|
+
def do_task(self, task, args: Arguments) -> list:
|
|
907
|
+
raise NotImplementedError
|
|
908
|
+
|
|
909
|
+
|
|
910
|
+
|
|
911
|
+
|
|
912
|
+
class DownloadWorker(Worker):
|
|
913
|
+
|
|
914
|
+
def init(self, args: Arguments):
|
|
915
|
+
if hasattr(self, '_session'):
|
|
916
|
+
return
|
|
917
|
+
|
|
918
|
+
self._delay : float = args.delay
|
|
919
|
+
self._session : requests.Session = requests.Session()
|
|
920
|
+
|
|
921
|
+
self._session.verify = False
|
|
922
|
+
self._configure_session(args)
|
|
923
|
+
|
|
924
|
+
|
|
925
|
+
|
|
926
|
+
def _configure_session(self, args: Arguments):
|
|
927
|
+
self._session.headers.clear()
|
|
928
|
+
self._session.headers.update(args.http_headers)
|
|
929
|
+
self._session.headers.pop("Accept-Encoding", None)
|
|
930
|
+
|
|
931
|
+
if args.proxy:
|
|
932
|
+
self._session.proxies = {
|
|
933
|
+
"http": args.proxy,
|
|
934
|
+
"https": args.proxy,
|
|
935
|
+
}
|
|
936
|
+
|
|
937
|
+
if not args.cert_p12:
|
|
938
|
+
self._session.mount(
|
|
939
|
+
args.url,
|
|
940
|
+
requests.adapters.HTTPAdapter(max_retries=args.retry),
|
|
941
|
+
)
|
|
942
|
+
return
|
|
943
|
+
|
|
944
|
+
self._session.mount(
|
|
945
|
+
args.url,
|
|
946
|
+
Pkcs12Adapter(
|
|
947
|
+
pkcs12_filename=args.cert_p12,
|
|
948
|
+
pkcs12_password=args.cert_p12_pwd,
|
|
949
|
+
max_retries=args.retry,
|
|
950
|
+
),
|
|
951
|
+
)
|
|
952
|
+
|
|
953
|
+
|
|
954
|
+
|
|
955
|
+
def _request(self, url: str, **kwargs) -> requests.Response:
|
|
956
|
+
response = self._session.get(url, **kwargs)
|
|
957
|
+
|
|
958
|
+
if self._delay > 0:
|
|
959
|
+
time.sleep(self._delay)
|
|
960
|
+
|
|
961
|
+
return response
|
|
962
|
+
|
|
963
|
+
|
|
964
|
+
|
|
965
|
+
def do_task(self, filepath: str, args: Arguments) -> list:
|
|
966
|
+
if os.path.isfile(os.path.join(args.directory, filepath)):
|
|
967
|
+
print(f"[---] Already downloaded {args.url}/{filepath}", flush=True)
|
|
968
|
+
return []
|
|
969
|
+
|
|
970
|
+
with closing(
|
|
971
|
+
self._request(
|
|
972
|
+
f"{args.url}/{filepath}",
|
|
973
|
+
allow_redirects=False,
|
|
974
|
+
stream=True,
|
|
975
|
+
timeout=args.timeout,
|
|
976
|
+
)
|
|
977
|
+
) as response:
|
|
978
|
+
valid, display, error_msg = verify_response(response)
|
|
979
|
+
|
|
980
|
+
if not valid:
|
|
981
|
+
if display:
|
|
982
|
+
Display.warning(f"Invalid response from {response.url}: {error_msg}")
|
|
983
|
+
return []
|
|
984
|
+
|
|
985
|
+
abspath = os.path.abspath(os.path.join(args.directory, filepath))
|
|
986
|
+
create_intermediate_dirs(abspath)
|
|
987
|
+
|
|
988
|
+
with open(abspath, "wb") as f:
|
|
989
|
+
for chunk in response.iter_content(4096):
|
|
990
|
+
f.write(chunk)
|
|
991
|
+
|
|
992
|
+
return []
|
|
993
|
+
|
|
994
|
+
|
|
995
|
+
|
|
996
|
+
|
|
997
|
+
class RecursiveDownloadWorker(DownloadWorker):
|
|
998
|
+
|
|
999
|
+
def do_task(self, filepath: str, args: Arguments) -> list:
|
|
1000
|
+
if os.path.isfile(os.path.join(args.directory, filepath)):
|
|
1001
|
+
print(f"[---] Already downloaded {args.url}/{filepath}", flush=True)
|
|
1002
|
+
return []
|
|
1003
|
+
|
|
1004
|
+
with closing(
|
|
1005
|
+
self._request(
|
|
1006
|
+
f"{args.url}/{filepath}",
|
|
1007
|
+
allow_redirects=False,
|
|
1008
|
+
stream=True,
|
|
1009
|
+
timeout=args.timeout,
|
|
1010
|
+
)
|
|
1011
|
+
) as response:
|
|
1012
|
+
Display.response(response)
|
|
1013
|
+
|
|
1014
|
+
if (
|
|
1015
|
+
response.status_code in (301, 302)
|
|
1016
|
+
and "Location" in response.headers
|
|
1017
|
+
and response.headers["Location"].endswith(filepath + "/")
|
|
1018
|
+
):
|
|
1019
|
+
return [filepath + "/"]
|
|
1020
|
+
|
|
1021
|
+
if filepath.endswith("/"): # directory index
|
|
1022
|
+
if not is_html(response):
|
|
1023
|
+
Display.warning(f"Expected HTML index at {response.url}")
|
|
1024
|
+
return []
|
|
1025
|
+
|
|
1026
|
+
return [
|
|
1027
|
+
filepath + filename
|
|
1028
|
+
for filename in get_indexed_files(response)
|
|
1029
|
+
]
|
|
1030
|
+
|
|
1031
|
+
valid, display, error_msg = verify_response(response)
|
|
1032
|
+
|
|
1033
|
+
if not valid:
|
|
1034
|
+
if display:
|
|
1035
|
+
Display.warning(f"Invalid response from {response.url}: {error_msg}")
|
|
1036
|
+
return []
|
|
1037
|
+
|
|
1038
|
+
abspath = os.path.abspath(os.path.join(args.directory, filepath))
|
|
1039
|
+
create_intermediate_dirs(abspath)
|
|
1040
|
+
|
|
1041
|
+
with open(abspath, "wb") as f:
|
|
1042
|
+
for chunk in response.iter_content(4096):
|
|
1043
|
+
f.write(chunk)
|
|
1044
|
+
|
|
1045
|
+
return []
|
|
1046
|
+
|
|
1047
|
+
|
|
1048
|
+
|
|
1049
|
+
|
|
1050
|
+
class FindRefsWorker(DownloadWorker):
|
|
1051
|
+
|
|
1052
|
+
def do_task(self, filepath: str, args: Arguments) -> list:
|
|
1053
|
+
response = self._request(
|
|
1054
|
+
f"{args.url}/{filepath}",
|
|
1055
|
+
allow_redirects=False,
|
|
1056
|
+
timeout=args.timeout,
|
|
1057
|
+
)
|
|
1058
|
+
|
|
1059
|
+
valid, display, error_msg = verify_response(response)
|
|
1060
|
+
|
|
1061
|
+
if not valid:
|
|
1062
|
+
if display:
|
|
1063
|
+
Display.warning(f"Invalid response from {response.url}: {error_msg}")
|
|
1064
|
+
return []
|
|
1065
|
+
|
|
1066
|
+
abspath = os.path.abspath(os.path.join(args.directory, filepath))
|
|
1067
|
+
create_intermediate_dirs(abspath)
|
|
1068
|
+
|
|
1069
|
+
with open(abspath, "w") as f:
|
|
1070
|
+
f.write(response.text)
|
|
1071
|
+
|
|
1072
|
+
tasks = []
|
|
1073
|
+
|
|
1074
|
+
for ref in re.findall(
|
|
1075
|
+
r"(refs(/[a-zA-Z0-9\-\.\_\*]+)+)", response.text
|
|
1076
|
+
):
|
|
1077
|
+
ref = ref[0]
|
|
1078
|
+
if not ref.endswith("*") and is_safe_path(ref):
|
|
1079
|
+
tasks.append(f".git/{ref}")
|
|
1080
|
+
tasks.append(f".git/logs/{ref}")
|
|
1081
|
+
|
|
1082
|
+
return tasks
|
|
1083
|
+
|
|
1084
|
+
|
|
1085
|
+
|
|
1086
|
+
|
|
1087
|
+
class FindObjectsWorker(DownloadWorker):
|
|
1088
|
+
|
|
1089
|
+
def do_task(self, obj, args: Arguments) -> list:
|
|
1090
|
+
filepath = f".git/objects/{obj[:2]}/{obj[2:]}"
|
|
1091
|
+
|
|
1092
|
+
if os.path.isfile(os.path.join(args.directory, filepath)):
|
|
1093
|
+
print(f"[---] Already downloaded {args.url}/{filepath}", flush=True)
|
|
1094
|
+
else:
|
|
1095
|
+
if not self._get_obj(filepath, args):
|
|
1096
|
+
return []
|
|
1097
|
+
|
|
1098
|
+
try:
|
|
1099
|
+
abspath = os.path.abspath(os.path.join(args.directory, filepath))
|
|
1100
|
+
obj_file = dulwich.objects.ShaFile.from_path(abspath)
|
|
1101
|
+
return get_referenced_sha1(obj_file)
|
|
1102
|
+
except Exception as e:
|
|
1103
|
+
Display.warning(f"Error while parsing file {filepath}: {e}")
|
|
1104
|
+
return []
|
|
1105
|
+
|
|
1106
|
+
def _get_obj(self, filepath: str, args: Arguments) -> bool:
|
|
1107
|
+
response = self._request(
|
|
1108
|
+
f"{args.url}/{filepath}",
|
|
1109
|
+
allow_redirects=False,
|
|
1110
|
+
timeout=args.timeout,
|
|
1111
|
+
)
|
|
1112
|
+
|
|
1113
|
+
valid, display, error_msg = verify_response(response)
|
|
1114
|
+
|
|
1115
|
+
if not valid:
|
|
1116
|
+
if display:
|
|
1117
|
+
Display.warning(f"Invalid response from {response.url}: {error_msg}")
|
|
1118
|
+
return False
|
|
1119
|
+
|
|
1120
|
+
abspath = os.path.abspath(os.path.join(args.directory, filepath))
|
|
1121
|
+
create_intermediate_dirs(abspath)
|
|
1122
|
+
|
|
1123
|
+
with open(abspath, "wb") as f:
|
|
1124
|
+
f.write(response.content)
|
|
1125
|
+
|
|
1126
|
+
return True
|
|
1127
|
+
|
|
1128
|
+
|
|
1129
|
+
|
|
1130
|
+
|
|
1131
|
+
if __name__ == "__main__":
|
|
1132
|
+
try:
|
|
1133
|
+
git_looter = GitBleed()
|
|
1134
|
+
git_looter.execute()
|
|
1135
|
+
sys.exit(0)
|
|
1136
|
+
except KeyboardInterrupt:
|
|
1137
|
+
pass
|
|
1138
|
+
except Exception as e:
|
|
1139
|
+
Display.fatal(f'{e}')
|