gitbleed 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
gitbleed/main.py ADDED
@@ -0,0 +1,1139 @@
1
+ # Copyright 2026 Oliver R. Calazans Jeronimo
2
+ # Modified from the original project git-dumper (Copyright (c) 2017 Maxime Arthaud)
3
+ #
4
+ # Licensed under the Apache License, Version 2.0 (the "License");
5
+ # you may not use this file except in compliance with the License.
6
+ # You may obtain a copy of the License at
7
+ #
8
+ # http://www.apache.org/licenses/LICENSE-2.0
9
+ #
10
+ # Unless required by applicable law or agreed to in writing, software
11
+ # distributed under the License is distributed on an "AS IS" BASIS,
12
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13
+ # See the License for the specific language governing permissions and
14
+ # limitations under the License.
15
+
16
+ import argparse
17
+ import multiprocessing
18
+ import os
19
+ import queue
20
+ import re
21
+ import subprocess
22
+ import sys
23
+ import time
24
+ import traceback
25
+ import urllib.parse
26
+ from contextlib import closing
27
+ from dataclasses import dataclass
28
+ from pathlib import Path
29
+
30
+ import bs4
31
+ import dulwich.index
32
+ import dulwich.objects
33
+ import dulwich.pack
34
+ import requests
35
+ import urllib3
36
+ from requests_pkcs12 import Pkcs12Adapter
37
+
38
+
39
+
40
+ class GitBleed:
41
+
42
+ __slots__ = ('_args', '_session', '_response', '_environment')
43
+
44
+ def __init__(self):
45
+ self._args : Arguments = None
46
+ self._session : requests.Session = None
47
+ self._response : requests.Response = None
48
+ self._environment : dict[str, str] = None
49
+
50
+
51
+
52
+ def execute(self):
53
+ self._get_args()
54
+ urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
55
+ self._create_session()
56
+ self._find_base_url()
57
+ self._try_to_connect()
58
+ self._valid_response()
59
+ self._setup_env_for_proxy()
60
+ self._try_fast_dump() # if successful, it stops here
61
+ self._fetch_common_files()
62
+ self._discover_references()
63
+ self._fetch_git_packs()
64
+ self._discover_and_fetch_objects()
65
+ self._finalize_checkout()
66
+
67
+
68
+
69
+ def _get_args(self):
70
+ parser = Parser()
71
+ parser.parse()
72
+ self._args = parser.get_args()
73
+
74
+
75
+
76
+ def _create_session(self):
77
+ self._session = requests.Session()
78
+ self._session.verify = False
79
+
80
+ self._session.headers.update(self._args.http_headers)
81
+ self._session.headers.pop("Accept-Encoding", None)
82
+ self._configure_session()
83
+
84
+ if os.listdir(self._args.directory):
85
+ Display.warning(f"Destination '{self._args.directory}' is not empty")
86
+
87
+
88
+
89
+ def _configure_session(self):
90
+ if self._args.proxy:
91
+ self._session.proxies = {
92
+ "http": self._args.proxy,
93
+ "https": self._args.proxy,
94
+ }
95
+
96
+ if not self._args.cert_p12:
97
+ self._session.mount(
98
+ self._args.url,
99
+ requests.adapters.HTTPAdapter(max_retries=self._args.retry),
100
+ )
101
+ return
102
+
103
+ self._session.mount(
104
+ self._args.url,
105
+ Pkcs12Adapter(
106
+ pkcs12_filename=self._args.cert_p12,
107
+ pkcs12_password=self._args.cert_p12_pwd,
108
+ max_retries=self._args.retry,
109
+ ),
110
+ )
111
+
112
+
113
+
114
+ def _find_base_url(self):
115
+ url: str = self._args.url
116
+
117
+ url = url.rstrip("/")
118
+ if url.endswith("HEAD"):
119
+ url = url[:-4]
120
+
121
+ url = url.rstrip("/")
122
+ if url.endswith(".git"):
123
+ url = url[:-4]
124
+
125
+ self._args.url = url.rstrip("/")
126
+
127
+
128
+
129
+ def _try_to_connect(self):
130
+ try:
131
+ self._response = self._session.get(
132
+ f"{self._args.url}/.git/HEAD",
133
+ timeout=self._args.timeout,
134
+ )
135
+ time.sleep(self._args.delay)
136
+ except Exception as e:
137
+ Display.fatal(f"Unable to connect to {self._args.url}. Error: {e}")
138
+
139
+
140
+
141
+ def _valid_response(self):
142
+ valid, _, error_msg = verify_response(self._response)
143
+
144
+ if self._response.status_code >= 400:
145
+ Display.fatal("Target unreachable. Dumping stopped")
146
+
147
+ elif self._response.status_code >= 300:
148
+ Display.warning(f"Redirection required to {self._response.headers['Location']}")
149
+ sys.exit(0)
150
+
151
+ if not valid:
152
+ Display.fatal(f"Invalid response from {self._response.url}: {error_msg}")
153
+
154
+ if not re.match(r"^(ref:.*|[0-9a-f]{40}$)", self._response.text.strip()):
155
+ Display.fatal(f"{self._response.url} is not a git HEAD file")
156
+
157
+
158
+
159
+ def _setup_env_for_proxy(self):
160
+ self._environment = os.environ.copy()
161
+
162
+ if not self._args.proxy:
163
+ return
164
+
165
+ self._environment["ALL_PROXY"] = self._args.proxy
166
+ self._environment["all_proxy"] = self._args.proxy
167
+
168
+
169
+
170
+ def _try_fast_dump(self):
171
+ Display.section("Trying fast dumping")
172
+
173
+ response = self._session.get(f"{self._args.url}/.git/", allow_redirects=False)
174
+ time.sleep(self._args.delay)
175
+
176
+ Display.response(response)
177
+
178
+ if (
179
+ response.status_code != 200
180
+ or not is_html(response)
181
+ or "HEAD" not in get_indexed_files(response)
182
+ ):
183
+ return
184
+
185
+ Display.section("Fetching .git recursively")
186
+ self._process_tasks([".git/", ".gitignore"], RecursiveDownloadWorker)
187
+
188
+ self._finalize_checkout()
189
+ sys.exit(0)
190
+
191
+
192
+
193
+ def _process_tasks(self, initial_tasks: list, worker, tasks_done=None):
194
+ if not initial_tasks:
195
+ return
196
+
197
+ tasks_seen = set(tasks_done) if tasks_done else set()
198
+ pending_tasks = multiprocessing.Queue()
199
+ tasks_done = multiprocessing.Queue()
200
+ num_pending_tasks = 0
201
+
202
+ for task in initial_tasks:
203
+ assert task is not None
204
+ if task not in tasks_seen:
205
+ pending_tasks.put(task)
206
+ num_pending_tasks += 1
207
+ tasks_seen.add(task)
208
+
209
+ processes = [worker(pending_tasks, tasks_done, self._args) for _ in range(self._args.jobs)]
210
+
211
+ for p in processes:
212
+ p.start()
213
+
214
+ timed_out = False
215
+
216
+ while num_pending_tasks > 0:
217
+ try:
218
+ task_result = tasks_done.get(block=True, timeout=60)
219
+ except queue.Empty:
220
+ Display.warning("Timeout waiting for workers; aborting")
221
+ timed_out = True
222
+ break
223
+
224
+ num_pending_tasks -= 1
225
+
226
+ for task in task_result:
227
+ assert task is not None
228
+ if task not in tasks_seen:
229
+ pending_tasks.put(task)
230
+ num_pending_tasks += 1
231
+ tasks_seen.add(task)
232
+
233
+ if timed_out:
234
+ for p in processes: p.terminate()
235
+ for p in processes: p.join()
236
+ return
237
+
238
+ for _ in range(self._args.jobs):
239
+ pending_tasks.put(None)
240
+
241
+ for p in processes:
242
+ p.join()
243
+
244
+
245
+
246
+ UNSAFE = r"^\s*fsmonitor|sshcommand|askpass|editor|pager"
247
+
248
+ def _sanitize_file(self, filepath=".git/config"):
249
+ if not os.path.isfile(filepath):
250
+ Display.warning(".git/config not found")
251
+ return
252
+
253
+ Display.section("Sanitizing .git/config")
254
+
255
+ with open(filepath, 'r+') as f:
256
+ content = f.read()
257
+ modified_content = re.sub(self.UNSAFE, r'# \g<0>', content, flags=re.IGNORECASE)
258
+
259
+ if content != modified_content:
260
+ Display.warning(f"'{filepath}' file was altered")
261
+ f.seek(0)
262
+ f.write(modified_content)
263
+
264
+
265
+
266
+ def _fetch_common_files(self):
267
+ Display.section("Fetching common files")
268
+
269
+ TASKS = [
270
+ ".gitignore",
271
+ ".git/COMMIT_EDITMSG",
272
+ ".git/description",
273
+ ".git/hooks/applypatch-msg.sample",
274
+ ".git/hooks/commit-msg.sample",
275
+ ".git/hooks/post-commit.sample",
276
+ ".git/hooks/post-receive.sample",
277
+ ".git/hooks/post-update.sample",
278
+ ".git/hooks/pre-applypatch.sample",
279
+ ".git/hooks/pre-commit.sample",
280
+ ".git/hooks/pre-push.sample",
281
+ ".git/hooks/pre-rebase.sample",
282
+ ".git/hooks/pre-receive.sample",
283
+ ".git/hooks/prepare-commit-msg.sample",
284
+ ".git/hooks/update.sample",
285
+ ".git/index",
286
+ ".git/info/exclude",
287
+ ".git/objects/info/packs",
288
+ ]
289
+
290
+ self._process_tasks(TASKS, DownloadWorker)
291
+
292
+
293
+
294
+ def _discover_references(self):
295
+ Display.section("Finding refs/")
296
+
297
+ TASKS = [
298
+ ".git/FETCH_HEAD",
299
+ ".git/HEAD",
300
+ ".git/ORIG_HEAD",
301
+ ".git/config",
302
+ ".git/info/refs",
303
+ ".git/logs/HEAD",
304
+ ".git/logs/refs/heads/main",
305
+ ".git/logs/refs/heads/master",
306
+ ".git/logs/refs/heads/staging",
307
+ ".git/logs/refs/heads/production",
308
+ ".git/logs/refs/heads/development",
309
+ ".git/logs/refs/remotes/origin/HEAD",
310
+ ".git/logs/refs/remotes/origin/main",
311
+ ".git/logs/refs/remotes/origin/master",
312
+ ".git/logs/refs/remotes/origin/staging",
313
+ ".git/logs/refs/remotes/origin/production",
314
+ ".git/logs/refs/remotes/origin/development",
315
+ ".git/logs/refs/stash",
316
+ ".git/packed-refs",
317
+ ".git/refs/heads/main",
318
+ ".git/refs/heads/master",
319
+ ".git/refs/heads/staging",
320
+ ".git/refs/heads/production",
321
+ ".git/refs/heads/development",
322
+ ".git/refs/remotes/origin/HEAD",
323
+ ".git/refs/remotes/origin/main",
324
+ ".git/refs/remotes/origin/master",
325
+ ".git/refs/remotes/origin/staging",
326
+ ".git/refs/remotes/origin/production",
327
+ ".git/refs/remotes/origin/development",
328
+ ".git/refs/stash",
329
+ ".git/refs/wip/wtree/refs/heads/main",
330
+ ".git/refs/wip/wtree/refs/heads/master",
331
+ ".git/refs/wip/wtree/refs/heads/staging",
332
+ ".git/refs/wip/wtree/refs/heads/production",
333
+ ".git/refs/wip/wtree/refs/heads/development",
334
+ ".git/refs/wip/index/refs/heads/main",
335
+ ".git/refs/wip/index/refs/heads/master",
336
+ ".git/refs/wip/index/refs/heads/staging",
337
+ ".git/refs/wip/index/refs/heads/production",
338
+ ".git/refs/wip/index/refs/heads/development",
339
+ ]
340
+
341
+ self._add_user_specified_branches(TASKS)
342
+ self._process_tasks(TASKS, FindRefsWorker)
343
+
344
+
345
+
346
+ def _add_user_specified_branches(self, tasks: list):
347
+ if not self._args.branches:
348
+ return
349
+
350
+ for branch in self._args.branches:
351
+ if not re.match(r'^[A-Za-z0-9\-\._]+$', branch):
352
+ Display.warning(f"Ignoring invalid branch name '{branch}'")
353
+ continue
354
+
355
+ tasks.extend([
356
+ f".git/logs/refs/heads/{branch}",
357
+ f".git/refs/heads/{branch}",
358
+ f".git/logs/refs/remotes/origin/{branch}",
359
+ f".git/refs/remotes/origin/{branch}",
360
+ f".git/refs/wip/wtree/refs/heads/{branch}",
361
+ f".git/refs/wip/index/refs/heads/{branch}",
362
+ ])
363
+
364
+
365
+
366
+ def _fetch_git_packs(self):
367
+ Display.section("Finding packs")
368
+ tasks = []
369
+
370
+ info_packs_path = os.path.join(
371
+ self._args.directory, ".git", "objects", "info", "packs"
372
+ )
373
+
374
+ if os.path.exists(info_packs_path):
375
+ with open(info_packs_path, "r") as f:
376
+ info_packs = f.read()
377
+
378
+ for sha1 in re.findall(r"pack-([a-f0-9]{40})\.pack", info_packs):
379
+ tasks.append(".git/objects/pack/pack-%s.idx" % sha1)
380
+ tasks.append(".git/objects/pack/pack-%s.pack" % sha1)
381
+
382
+ self._process_tasks(tasks, DownloadWorker)
383
+
384
+
385
+
386
+ def _discover_and_fetch_objects(self):
387
+ Display.section("Finding objects")
388
+ objs = set()
389
+ packed_objs = set()
390
+
391
+ files = [
392
+ os.path.join(self._args.directory, ".git", "packed-refs"),
393
+ os.path.join(self._args.directory, ".git", "info", "refs"),
394
+ os.path.join(self._args.directory, ".git", "FETCH_HEAD"),
395
+ os.path.join(self._args.directory, ".git", "ORIG_HEAD"),
396
+ ]
397
+
398
+ self._collect_file_paths_from_git_subdir(files, "refs")
399
+ self._collect_file_paths_from_git_subdir(files, "logs")
400
+ self._extract_sha1_hashes_from_files(files, objs)
401
+ self._parse_staging_area(objs)
402
+ self._process_pack_files(packed_objs, objs)
403
+
404
+ Display.section("Fetching objects")
405
+ self._process_tasks(objs, FindObjectsWorker, tasks_done=packed_objs)
406
+
407
+
408
+
409
+ def _collect_file_paths_from_git_subdir(self, files: list, subpath: str):
410
+ base_path = os.path.join(self._args.directory, ".git", subpath)
411
+
412
+ if not os.path.isdir(base_path):
413
+ return
414
+
415
+ for dirpath, _, filenames in os.walk(base_path):
416
+ for filename in filenames:
417
+ files.append(os.path.join(dirpath, filename))
418
+
419
+
420
+
421
+ @staticmethod
422
+ def _extract_sha1_hashes_from_files(files: list, objs: set):
423
+ for filepath in files:
424
+ if not os.path.isfile(filepath):
425
+ continue
426
+
427
+ try:
428
+ with open(filepath, "r", encoding='utf-8', errors='ignore') as f:
429
+ content = f.read()
430
+ except Exception:
431
+ continue
432
+
433
+ for match in re.findall(r"(^|\s)([a-f0-9]{40})($|\s)", content):
434
+ objs.add(match[1])
435
+
436
+
437
+
438
+ def _parse_staging_area(self, objs: set):
439
+ index_path = os.path.join(self._args.directory, ".git", "index")
440
+
441
+ if not os.path.exists(index_path):
442
+ return
443
+
444
+ try:
445
+ index = dulwich.index.Index(index_path)
446
+ for entry in index.iterobjects():
447
+ objs.add(entry[1].decode())
448
+ except Exception:
449
+ pass
450
+
451
+
452
+
453
+ def _process_pack_files(self, packed_objs: set, objs: set):
454
+ pack_file_dir = os.path.join(self._args.directory, ".git", "objects", "pack")
455
+
456
+ if not os.path.isdir(pack_file_dir):
457
+ return
458
+
459
+ for filename in os.listdir(pack_file_dir):
460
+ if not filename.startswith("pack-") or not filename.endswith(".pack"):
461
+ continue
462
+
463
+ try:
464
+ pack_data_path = os.path.join(pack_file_dir, filename)
465
+ pack_idx_path = os.path.join(pack_file_dir, filename[:-5] + ".idx")
466
+ pack_data = dulwich.pack.PackData(pack_data_path)
467
+ pack_idx = dulwich.pack.load_pack_index(pack_idx_path)
468
+ pack = dulwich.pack.Pack.from_objects(pack_data, pack_idx)
469
+
470
+ for obj_file in pack.iterobjects():
471
+ packed_objs.add(obj_file.sha().hexdigest())
472
+ objs |= set(get_referenced_sha1(obj_file))
473
+
474
+ except Exception:
475
+ continue
476
+
477
+
478
+
479
+ def _finalize_checkout(self):
480
+ Display.section("Running git checkout")
481
+
482
+ cwd = os.getcwd()
483
+ try:
484
+ os.chdir(self._args.directory)
485
+ self._sanitize_file()
486
+
487
+ subprocess.call(
488
+ ["git", "checkout", "."],
489
+ stderr=subprocess.DEVNULL,
490
+ env=self._environment,
491
+ )
492
+
493
+ finally:
494
+ os.chdir(cwd)
495
+ Display.success(f"Repository saved to {self._args.directory}")
496
+
497
+
498
+
499
+
500
+
501
+ @dataclass(slots=True)
502
+ class Arguments:
503
+ cert_p12_pwd : str
504
+ cert_p12 : str
505
+ url : str
506
+ directory : str
507
+ proxy : str
508
+ jobs : int
509
+ retry : int
510
+ timeout : int
511
+ http_headers : dict[str, str]
512
+ branches : list[str]
513
+ delay : float
514
+
515
+
516
+
517
+ class Parser:
518
+
519
+ __slots__ = ('_args', '_parser')
520
+
521
+ def __init__(self):
522
+ self._args : argparse.Namespace = None
523
+ self._parser : argparse.ArgumentParser = None
524
+
525
+
526
+
527
+ def parse(self):
528
+ self._create_args()
529
+ self._args = self._parser.parse_args()
530
+ self._valid_jobs()
531
+ self._valid_retry()
532
+ self._valid_timeout()
533
+ self._valid_proxy()
534
+ self._valid_certificate()
535
+ self._valid_delay()
536
+ self._create_dir()
537
+
538
+
539
+
540
+ def _create_args(self):
541
+ self._parser = argparse.ArgumentParser(
542
+ usage="gitbleed --url example.com <ARGS>",
543
+ description="Dump a git repository from a website.",
544
+ formatter_class=lambda prog: argparse.HelpFormatter(
545
+ prog,
546
+ max_help_position=40,
547
+ width=100,
548
+ ),
549
+ )
550
+ self._parser.add_argument(
551
+ "--url", type=str, required=True,
552
+ help="URL of the git repository to dump"
553
+ )
554
+ self._parser.add_argument("--proxy", help="Proxy URL, e.g. `socks5://127.0.0.1:9050`")
555
+ self._parser.add_argument("--cert-p12", help="Client certificate in PKCS#12")
556
+ self._parser.add_argument("--cert-p12-pwd", help="Password for the client certificate")
557
+ self._parser.add_argument(
558
+ "-d", "--delay", type=float, default=0,
559
+ help="Delay in seconds between requests"
560
+ )
561
+ self._parser.add_argument(
562
+ "-o", "--output", type=str, default="",
563
+ help="Directory where the repository will be saved",
564
+ )
565
+ self._parser.add_argument(
566
+ "-j", "--jobs", type=int, default=10,
567
+ help="Number of simultaneous requests",
568
+ )
569
+ self._parser.add_argument(
570
+ "-r", "--retry", type=int, default=3,
571
+ help="Number of request attempts before giving up",
572
+ )
573
+ self._parser.add_argument(
574
+ "-t", "--timeout", type=int, default=3,
575
+ help="Maximum time in seconds per request",
576
+ )
577
+ self._parser.add_argument(
578
+ "-u", "--user-agent", type=str,
579
+ default="Mozilla/5.0 (Windows NT 10.0; rv:78.0) Gecko/20100101 Firefox/78.0",
580
+ help="User-agent to use for requests",
581
+ )
582
+ self._parser.add_argument(
583
+ "-H", "--header", type=str, action="append",
584
+ help="Additional http headers, e.g `NAME=VALUE`",
585
+ )
586
+ self._parser.add_argument(
587
+ "-b", "--branch", dest="branches", action="append",
588
+ help=(
589
+ "Additional branch names to check for, e.g. `-b dev -b prod`. "
590
+ "The default branches (`main`, `master`, `staging`, `production`, "
591
+ "`development`) are always checked."
592
+ ),
593
+ )
594
+
595
+
596
+ def _valid_jobs(self):
597
+ if self._args.jobs < 1:
598
+ self._parser.error(f"Invalid number of jobs, got {self._args.jobs}")
599
+
600
+
601
+ def _valid_retry(self):
602
+ if self._args.retry < 1:
603
+ self._parser.error(f"Invalid number of retries, got {self._args.retry}")
604
+
605
+
606
+ def _valid_timeout(self):
607
+ if self._args.timeout < 1:
608
+ self._parser.error(f"Invalid timeout, got {self._args.timeout}")
609
+
610
+
611
+ def _valid_delay(self):
612
+ if self._args.delay < 0:
613
+ self._parser.error(f"Delay value cannot be negative. Got {self._args.delay}")
614
+
615
+
616
+ def _valid_proxy(self):
617
+ if not self._args.proxy:
618
+ return
619
+
620
+ raw = self._args.proxy
621
+ if "://" not in raw:
622
+ raw = "socks5://" + raw
623
+
624
+ try:
625
+ parsed = urllib.parse.urlparse(raw)
626
+ except ValueError as e:
627
+ self._parser.error(f"Invalid proxy, got {self._args.proxy}: {e}")
628
+
629
+ if parsed.scheme not in ("socks5", "socks5h", "socks4", "socks4a", "http", "https"):
630
+ self._parser.error(
631
+ f"Unsupported proxy scheme '{parsed.scheme}' in {self._args.proxy}"
632
+ )
633
+
634
+ if not parsed.hostname or not parsed.port:
635
+ self._parser.error(f"Invalid proxy, got {self._args.proxy}")
636
+
637
+ self._args.proxy = f"{parsed.scheme}://{parsed.netloc}"
638
+
639
+
640
+
641
+ def _create_dir(self):
642
+ if self._args.output == "":
643
+ self._args.output = Path(__file__).resolve().parent
644
+
645
+ if not os.path.isdir(self._args.output):
646
+ self._parser.error(f"{self._args.output} is not a directory")
647
+
648
+ url = re.sub(r"^https?://", "", self._args.url)
649
+ self._args.output = os.path.join(self._args.output, url)
650
+
651
+ if not os.path.exists(self._args.output):
652
+ os.makedirs(self._args.output)
653
+
654
+
655
+
656
+ def _valid_certificate(self):
657
+ if not self._args.cert_p12:
658
+ return
659
+
660
+ if not os.path.exists(self._args.cert_p12):
661
+ self._parser.error(
662
+ f"Client certificate {self._args.cert_p12} does not exist"
663
+ )
664
+
665
+ if not os.path.isfile(self._args.cert_p12):
666
+ self._parser.error(
667
+ f"Client certificate {self._args.cert_p12} is not a file"
668
+ )
669
+
670
+ if not self._args.cert_p12_pwd:
671
+ self._parser.error("Client certificate password is required")
672
+
673
+
674
+
675
+ def _valid_headers(self) -> dict:
676
+ http_headers = {
677
+ "User-Agent": self._args.user_agent,
678
+ "Accept": "*/*",
679
+ }
680
+
681
+ if not self._args.header:
682
+ return http_headers
683
+
684
+ for header in self._args.header:
685
+ tokens: list[str] = header.split("=", maxsplit=1)
686
+
687
+ if len(tokens) != 2:
688
+ self._parser.error(
689
+ f"HTTP header must have the form NAME=VALUE, got {header}"
690
+ )
691
+
692
+ name, value = tokens
693
+ http_headers[name.strip()] = value.strip()
694
+
695
+ return http_headers
696
+
697
+
698
+
699
+ def get_args(self) -> Arguments:
700
+ return Arguments(
701
+ cert_p12_pwd = self._args.cert_p12_pwd,
702
+ cert_p12 = self._args.cert_p12,
703
+ directory = self._args.output,
704
+ proxy = self._args.proxy,
705
+ url = self._args.url,
706
+ jobs = self._args.jobs,
707
+ retry = self._args.retry,
708
+ timeout = self._args.timeout,
709
+ http_headers = self._valid_headers(),
710
+ branches = self._args.branches,
711
+ delay = self._args.delay,
712
+ )
713
+
714
+
715
+
716
+
717
+ class Display:
718
+
719
+ HTML: str = " [\033[34mHTML\033[0m] "
720
+
721
+ @staticmethod
722
+ def response(response: requests.Response):
723
+ code = response.status_code
724
+
725
+ if code >= 400: x = f"\033[31m{code}\033[0m"
726
+ elif code >= 300: x = f"\033[33m{code}\033[0m"
727
+ elif code >= 200: x = f"\033[32m{code}\033[0m"
728
+ else: x = f"{code}"
729
+
730
+ z = Display.HTML if is_html(response) else ' '
731
+
732
+ print(f"[{x}]{z}{response.url}", flush=True)
733
+
734
+
735
+ @staticmethod
736
+ def section(text: str):
737
+ print(f"[###] {text}", flush=True)
738
+
739
+
740
+ @staticmethod
741
+ def warning(text: str):
742
+ print(f"[\033[33m{'!!!'}\033[0m] {text}", flush=True)
743
+
744
+
745
+ @staticmethod
746
+ def fatal(text: str):
747
+ print(f"[\033[31m{'ERR'}\033[0m] {text}", flush=True)
748
+ sys.exit(1)
749
+
750
+
751
+ @staticmethod
752
+ def success(text: str):
753
+ print(f"[\033[32m{'OK!'}\033[0m] {text}", flush=True)
754
+
755
+
756
+
757
+
758
+ def is_html(response: requests.Response) -> bool:
759
+ return (
760
+ "Content-Type" in response.headers
761
+ and "text/html" in response.headers["Content-Type"]
762
+ )
763
+
764
+
765
+
766
+ def is_safe_path(path: str) -> bool:
767
+ """ Prevents directory traversal attacks """
768
+ if path.startswith("/"):
769
+ return False
770
+
771
+ safe_path = os.path.expanduser("~")
772
+ try:
773
+ real = os.path.realpath(os.path.join(safe_path, path))
774
+ return os.path.commonpath((real, safe_path)) == safe_path
775
+
776
+ except ValueError:
777
+ return False
778
+
779
+
780
+
781
+ def get_indexed_files(response: requests.Response) -> list:
782
+ html = bs4.BeautifulSoup(response.text, "html.parser")
783
+ files = []
784
+
785
+ for link in html.find_all("a"):
786
+ url = urllib.parse.urlparse(link.get("href"))
787
+
788
+ if (
789
+ url.path
790
+ and is_safe_path(url.path)
791
+ and not url.scheme
792
+ and not url.netloc
793
+ ):
794
+ files.append(url.path)
795
+
796
+ return files
797
+
798
+
799
+
800
+ def verify_response(response: requests.Response) -> tuple[bool, bool, str]:
801
+ Display.response(response)
802
+
803
+ if response.status_code >= 400:
804
+ return False, False, None
805
+
806
+ elif response.status_code >= 300 and "Location" in response.headers:
807
+ return False, True, (
808
+ f"Moved to {response.headers['Location']}. Code {response.status_code}"
809
+ )
810
+
811
+ elif (
812
+ "Content-Length" in response.headers
813
+ and response.headers["Content-Length"] == 0
814
+ ):
815
+ return False, True, "Responded with a zero-length body"
816
+
817
+ elif is_html(response):
818
+ return False, True, f"{response.url} responded with a HTML"
819
+
820
+ else:
821
+ return True, False, None
822
+
823
+
824
+
825
+ def create_intermediate_dirs(path: str):
826
+ dirname, basename = os.path.split(path)
827
+
828
+ if dirname and not os.path.exists(dirname):
829
+ try:
830
+ os.makedirs(dirname)
831
+ except FileExistsError:
832
+ pass
833
+
834
+
835
+
836
+ def get_referenced_sha1(obj_file):
837
+ """ Return all the referenced SHA1 in the given object file """
838
+ objs = []
839
+
840
+ if isinstance(obj_file, dulwich.objects.Commit):
841
+ objs.append(obj_file.tree.decode())
842
+ for parent in obj_file.parents:
843
+ objs.append(parent.decode())
844
+
845
+ elif isinstance(obj_file, dulwich.objects.Tree):
846
+ for item in obj_file.iteritems():
847
+ objs.append(item.sha.decode())
848
+
849
+ elif isinstance(obj_file, dulwich.objects.Tag):
850
+ objs.append(obj_file.object[1].decode())
851
+
852
+ elif isinstance(obj_file, dulwich.objects.Blob):
853
+ pass
854
+
855
+ else:
856
+ Display.warning(f"Unexpected object type: {obj_file}")
857
+
858
+ return objs
859
+
860
+
861
+
862
+
863
+ class Worker(multiprocessing.Process):
864
+
865
+ def __init__(
866
+ self,
867
+ pending_tasks : multiprocessing.Queue,
868
+ tasks_done : multiprocessing.Queue,
869
+ args : Arguments,
870
+ ):
871
+ super().__init__()
872
+ self.daemon : bool = True
873
+ self.pending_tasks : multiprocessing.Queue = pending_tasks
874
+ self.tasks_done : multiprocessing.Queue = tasks_done
875
+ self.args : Arguments = args
876
+
877
+
878
+
879
+ def run(self):
880
+ self.init(self.args)
881
+
882
+ while True:
883
+ task = self.pending_tasks.get(block=True)
884
+
885
+ if task is None: # end signal
886
+ return
887
+
888
+ try:
889
+ result = self.do_task(task, self.args)
890
+ except Exception:
891
+ Display.warning(f"Task {task} raised exception:")
892
+ traceback.print_exc()
893
+ result = []
894
+
895
+ if not isinstance(result, list):
896
+ Display.warning(
897
+ f"do_task returned {type(result).__name__}, expected list"
898
+ )
899
+ result = []
900
+
901
+ self.tasks_done.put(result)
902
+
903
+ def init(self, args: Arguments):
904
+ raise NotImplementedError
905
+
906
+ def do_task(self, task, args: Arguments) -> list:
907
+ raise NotImplementedError
908
+
909
+
910
+
911
+
912
+ class DownloadWorker(Worker):
913
+
914
+ def init(self, args: Arguments):
915
+ if hasattr(self, '_session'):
916
+ return
917
+
918
+ self._delay : float = args.delay
919
+ self._session : requests.Session = requests.Session()
920
+
921
+ self._session.verify = False
922
+ self._configure_session(args)
923
+
924
+
925
+
926
+ def _configure_session(self, args: Arguments):
927
+ self._session.headers.clear()
928
+ self._session.headers.update(args.http_headers)
929
+ self._session.headers.pop("Accept-Encoding", None)
930
+
931
+ if args.proxy:
932
+ self._session.proxies = {
933
+ "http": args.proxy,
934
+ "https": args.proxy,
935
+ }
936
+
937
+ if not args.cert_p12:
938
+ self._session.mount(
939
+ args.url,
940
+ requests.adapters.HTTPAdapter(max_retries=args.retry),
941
+ )
942
+ return
943
+
944
+ self._session.mount(
945
+ args.url,
946
+ Pkcs12Adapter(
947
+ pkcs12_filename=args.cert_p12,
948
+ pkcs12_password=args.cert_p12_pwd,
949
+ max_retries=args.retry,
950
+ ),
951
+ )
952
+
953
+
954
+
955
+ def _request(self, url: str, **kwargs) -> requests.Response:
956
+ response = self._session.get(url, **kwargs)
957
+
958
+ if self._delay > 0:
959
+ time.sleep(self._delay)
960
+
961
+ return response
962
+
963
+
964
+
965
+ def do_task(self, filepath: str, args: Arguments) -> list:
966
+ if os.path.isfile(os.path.join(args.directory, filepath)):
967
+ print(f"[---] Already downloaded {args.url}/{filepath}", flush=True)
968
+ return []
969
+
970
+ with closing(
971
+ self._request(
972
+ f"{args.url}/{filepath}",
973
+ allow_redirects=False,
974
+ stream=True,
975
+ timeout=args.timeout,
976
+ )
977
+ ) as response:
978
+ valid, display, error_msg = verify_response(response)
979
+
980
+ if not valid:
981
+ if display:
982
+ Display.warning(f"Invalid response from {response.url}: {error_msg}")
983
+ return []
984
+
985
+ abspath = os.path.abspath(os.path.join(args.directory, filepath))
986
+ create_intermediate_dirs(abspath)
987
+
988
+ with open(abspath, "wb") as f:
989
+ for chunk in response.iter_content(4096):
990
+ f.write(chunk)
991
+
992
+ return []
993
+
994
+
995
+
996
+
997
+ class RecursiveDownloadWorker(DownloadWorker):
998
+
999
+ def do_task(self, filepath: str, args: Arguments) -> list:
1000
+ if os.path.isfile(os.path.join(args.directory, filepath)):
1001
+ print(f"[---] Already downloaded {args.url}/{filepath}", flush=True)
1002
+ return []
1003
+
1004
+ with closing(
1005
+ self._request(
1006
+ f"{args.url}/{filepath}",
1007
+ allow_redirects=False,
1008
+ stream=True,
1009
+ timeout=args.timeout,
1010
+ )
1011
+ ) as response:
1012
+ Display.response(response)
1013
+
1014
+ if (
1015
+ response.status_code in (301, 302)
1016
+ and "Location" in response.headers
1017
+ and response.headers["Location"].endswith(filepath + "/")
1018
+ ):
1019
+ return [filepath + "/"]
1020
+
1021
+ if filepath.endswith("/"): # directory index
1022
+ if not is_html(response):
1023
+ Display.warning(f"Expected HTML index at {response.url}")
1024
+ return []
1025
+
1026
+ return [
1027
+ filepath + filename
1028
+ for filename in get_indexed_files(response)
1029
+ ]
1030
+
1031
+ valid, display, error_msg = verify_response(response)
1032
+
1033
+ if not valid:
1034
+ if display:
1035
+ Display.warning(f"Invalid response from {response.url}: {error_msg}")
1036
+ return []
1037
+
1038
+ abspath = os.path.abspath(os.path.join(args.directory, filepath))
1039
+ create_intermediate_dirs(abspath)
1040
+
1041
+ with open(abspath, "wb") as f:
1042
+ for chunk in response.iter_content(4096):
1043
+ f.write(chunk)
1044
+
1045
+ return []
1046
+
1047
+
1048
+
1049
+
1050
+ class FindRefsWorker(DownloadWorker):
1051
+
1052
+ def do_task(self, filepath: str, args: Arguments) -> list:
1053
+ response = self._request(
1054
+ f"{args.url}/{filepath}",
1055
+ allow_redirects=False,
1056
+ timeout=args.timeout,
1057
+ )
1058
+
1059
+ valid, display, error_msg = verify_response(response)
1060
+
1061
+ if not valid:
1062
+ if display:
1063
+ Display.warning(f"Invalid response from {response.url}: {error_msg}")
1064
+ return []
1065
+
1066
+ abspath = os.path.abspath(os.path.join(args.directory, filepath))
1067
+ create_intermediate_dirs(abspath)
1068
+
1069
+ with open(abspath, "w") as f:
1070
+ f.write(response.text)
1071
+
1072
+ tasks = []
1073
+
1074
+ for ref in re.findall(
1075
+ r"(refs(/[a-zA-Z0-9\-\.\_\*]+)+)", response.text
1076
+ ):
1077
+ ref = ref[0]
1078
+ if not ref.endswith("*") and is_safe_path(ref):
1079
+ tasks.append(f".git/{ref}")
1080
+ tasks.append(f".git/logs/{ref}")
1081
+
1082
+ return tasks
1083
+
1084
+
1085
+
1086
+
1087
+ class FindObjectsWorker(DownloadWorker):
1088
+
1089
+ def do_task(self, obj, args: Arguments) -> list:
1090
+ filepath = f".git/objects/{obj[:2]}/{obj[2:]}"
1091
+
1092
+ if os.path.isfile(os.path.join(args.directory, filepath)):
1093
+ print(f"[---] Already downloaded {args.url}/{filepath}", flush=True)
1094
+ else:
1095
+ if not self._get_obj(filepath, args):
1096
+ return []
1097
+
1098
+ try:
1099
+ abspath = os.path.abspath(os.path.join(args.directory, filepath))
1100
+ obj_file = dulwich.objects.ShaFile.from_path(abspath)
1101
+ return get_referenced_sha1(obj_file)
1102
+ except Exception as e:
1103
+ Display.warning(f"Error while parsing file {filepath}: {e}")
1104
+ return []
1105
+
1106
+ def _get_obj(self, filepath: str, args: Arguments) -> bool:
1107
+ response = self._request(
1108
+ f"{args.url}/{filepath}",
1109
+ allow_redirects=False,
1110
+ timeout=args.timeout,
1111
+ )
1112
+
1113
+ valid, display, error_msg = verify_response(response)
1114
+
1115
+ if not valid:
1116
+ if display:
1117
+ Display.warning(f"Invalid response from {response.url}: {error_msg}")
1118
+ return False
1119
+
1120
+ abspath = os.path.abspath(os.path.join(args.directory, filepath))
1121
+ create_intermediate_dirs(abspath)
1122
+
1123
+ with open(abspath, "wb") as f:
1124
+ f.write(response.content)
1125
+
1126
+ return True
1127
+
1128
+
1129
+
1130
+
1131
+ if __name__ == "__main__":
1132
+ try:
1133
+ git_looter = GitBleed()
1134
+ git_looter.execute()
1135
+ sys.exit(0)
1136
+ except KeyboardInterrupt:
1137
+ pass
1138
+ except Exception as e:
1139
+ Display.fatal(f'{e}')