mcrit 1.6.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mcrit/SingleJobWorker.py +105 -0
- mcrit/SpawningWorker.py +151 -0
- mcrit/Worker.py +539 -0
- mcrit/__init__.py +0 -0
- mcrit/__main__.py +126 -0
- mcrit/cache/logbuckets.json +1 -0
- mcrit/client/McritClient.py +805 -0
- mcrit/client/McritConsole.py +640 -0
- mcrit/client/__init__.py +0 -0
- mcrit/config/ConfigInterface.py +30 -0
- mcrit/config/GunicornConfig.py +17 -0
- mcrit/config/McritConfig.py +34 -0
- mcrit/config/MinHashConfig.py +49 -0
- mcrit/config/QueueConfig.py +25 -0
- mcrit/config/ShinglerConfig.py +36 -0
- mcrit/config/StorageConfig.py +70 -0
- mcrit/config/__init__.py +0 -0
- mcrit/index/MinHashIndex.py +687 -0
- mcrit/index/SearchCursor.py +109 -0
- mcrit/index/SearchQueryParser.py +161 -0
- mcrit/index/SearchQueryTree.py +210 -0
- mcrit/index/__init__.py +0 -0
- mcrit/libs/__init__.py +0 -0
- mcrit/libs/graph.py +26 -0
- mcrit/libs/mongoqueue.py +738 -0
- mcrit/libs/parallel.py +7 -0
- mcrit/libs/pymmh3.py +449 -0
- mcrit/libs/utility.py +75 -0
- mcrit/matchers/FunctionCfgMatcher.py +215 -0
- mcrit/matchers/MatcherCross.py +138 -0
- mcrit/matchers/MatcherFlags.py +5 -0
- mcrit/matchers/MatcherInterface.py +677 -0
- mcrit/matchers/MatcherQuery.py +62 -0
- mcrit/matchers/MatcherQueryFunction.py +66 -0
- mcrit/matchers/MatcherSample.py +23 -0
- mcrit/matchers/MatcherVs.py +52 -0
- mcrit/matchers/MatcherVsGroup.py +70 -0
- mcrit/matchers/__init__.py +0 -0
- mcrit/minhash/MinHash.py +103 -0
- mcrit/minhash/MinHasher.py +196 -0
- mcrit/minhash/ShingleLoader.py +71 -0
- mcrit/minhash/__init__.py +0 -0
- mcrit/queue/JobCollection.py +47 -0
- mcrit/queue/LocalQueue.py +574 -0
- mcrit/queue/QueueFactory.py +28 -0
- mcrit/queue/QueueRemoteCalls.py +490 -0
- mcrit/queue/__init__.py +0 -0
- mcrit/server/BlocksResource.py +27 -0
- mcrit/server/FamilyResource.py +122 -0
- mcrit/server/FunctionResource.py +91 -0
- mcrit/server/JobResource.py +168 -0
- mcrit/server/MatchResource.py +77 -0
- mcrit/server/QueryResource.py +173 -0
- mcrit/server/SampleResource.py +230 -0
- mcrit/server/StatusResource.py +141 -0
- mcrit/server/__init__.py +0 -0
- mcrit/server/application_routes.py +157 -0
- mcrit/server/utils.py +59 -0
- mcrit/server/wsgi.py +3 -0
- mcrit/shinglers/AbstractShingler.py +69 -0
- mcrit/shinglers/EscapedBlockShingler.py +61 -0
- mcrit/shinglers/FuzzyStatPairShingler.py +97 -0
- mcrit/shinglers/LogBucket.py +119 -0
- mcrit/shinglers/__init__.py +7 -0
- mcrit/storage/FamilyEntry.py +57 -0
- mcrit/storage/FunctionEntry.py +133 -0
- mcrit/storage/FunctionLabelEntry.py +42 -0
- mcrit/storage/MatchedFunctionEntry.py +70 -0
- mcrit/storage/MatchedSampleEntry.py +140 -0
- mcrit/storage/MatchingCache.py +142 -0
- mcrit/storage/MatchingResult.py +677 -0
- mcrit/storage/MemoryStorage.py +973 -0
- mcrit/storage/MongoDbStorage.py +1592 -0
- mcrit/storage/SampleEntry.py +109 -0
- mcrit/storage/StorageFactory.py +13 -0
- mcrit/storage/StorageInterface.py +755 -0
- mcrit/storage/UniqueBlocksResult.py +143 -0
- mcrit/storage/__init__.py +0 -0
- mcrit-1.6.0.data/data/LICENSE +674 -0
- mcrit-1.6.0.data/data/requirements.txt +21 -0
- mcrit-1.6.0.dist-info/METADATA +324 -0
- mcrit-1.6.0.dist-info/RECORD +86 -0
- mcrit-1.6.0.dist-info/WHEEL +5 -0
- mcrit-1.6.0.dist-info/entry_points.txt +2 -0
- mcrit-1.6.0.dist-info/licenses/LICENSE +674 -0
- mcrit-1.6.0.dist-info/top_level.txt +1 -0
mcrit/SingleJobWorker.py
ADDED
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
|
|
3
|
+
import logging
|
|
4
|
+
import os
|
|
5
|
+
import uuid
|
|
6
|
+
from typing import TYPE_CHECKING, Optional
|
|
7
|
+
|
|
8
|
+
from pymongo import ReturnDocument
|
|
9
|
+
|
|
10
|
+
from mcrit.config.McritConfig import McritConfig
|
|
11
|
+
from mcrit.minhash.MinHasher import MinHasher
|
|
12
|
+
from mcrit.queue.QueueFactory import QueueFactory
|
|
13
|
+
from mcrit.queue.QueueRemoteCalls import JobProgressReporter
|
|
14
|
+
from mcrit.storage.StorageFactory import StorageFactory
|
|
15
|
+
from mcrit.Worker import Worker
|
|
16
|
+
|
|
17
|
+
if TYPE_CHECKING:
|
|
18
|
+
from mcrit.storage.StorageInterface import StorageInterface
|
|
19
|
+
|
|
20
|
+
logging.basicConfig(level=logging.INFO)
|
|
21
|
+
LOGGER = logging.getLogger(__name__)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class SingleJobWorker(Worker):
|
|
25
|
+
def __init__(self, job_id, queue=None, config=None, storage: Optional["StorageInterface"] = None, profiling=False):
|
|
26
|
+
self.job_id = job_id
|
|
27
|
+
self._worker_id = f"Worker-{uuid.uuid4()}"
|
|
28
|
+
LOGGER.info(f"Starting as worker: {self._worker_id}")
|
|
29
|
+
if config is None:
|
|
30
|
+
config = McritConfig()
|
|
31
|
+
|
|
32
|
+
if not queue:
|
|
33
|
+
queue = QueueFactory().getQueue(config, consumer_id=self._worker_id)
|
|
34
|
+
|
|
35
|
+
if profiling:
|
|
36
|
+
print("[!] Running as profiled application.")
|
|
37
|
+
profiling_path = os.path.abspath(os.path.join(os.path.dirname(os.path.abspath(__file__)), "..", "profiler"))
|
|
38
|
+
os.makedirs(profiling_path, exist_ok=True)
|
|
39
|
+
else:
|
|
40
|
+
profiling_path = None
|
|
41
|
+
super().__init__(queue=queue, config=config, storage=storage, profiling=profiling)
|
|
42
|
+
|
|
43
|
+
self.config = config
|
|
44
|
+
self._storage_config = config.STORAGE_CONFIG
|
|
45
|
+
self._minhash_config = config.MINHASH_CONFIG
|
|
46
|
+
self._shingler_config = config.SHINGLER_CONFIG
|
|
47
|
+
self._queue_config = config.QUEUE_CONFIG
|
|
48
|
+
self.minhasher = MinHasher(config.MINHASH_CONFIG, config.SHINGLER_CONFIG)
|
|
49
|
+
if storage:
|
|
50
|
+
self._storage = storage
|
|
51
|
+
else:
|
|
52
|
+
self._storage = StorageFactory.getStorage(config)
|
|
53
|
+
|
|
54
|
+
def __enter__(self):
|
|
55
|
+
return self
|
|
56
|
+
|
|
57
|
+
def __exit__(self, *args):
|
|
58
|
+
# TODO unregister our worker_id from all in-progress jobs found in the queue
|
|
59
|
+
self.queue.unregisterWorker()
|
|
60
|
+
self.queue.release_all_jobs()
|
|
61
|
+
|
|
62
|
+
#### Overwrite inherited methods to achive execution of a single job ####
|
|
63
|
+
|
|
64
|
+
def _executeJobPayload(self, job_payload, job):
|
|
65
|
+
LOGGER.debug("DECODE JOB: %s", job_payload)
|
|
66
|
+
method, params, kwparams = self._decodeJobPayload(job_payload)
|
|
67
|
+
# Add progress reporter if necessary:
|
|
68
|
+
if method.progressor:
|
|
69
|
+
LOGGER.debug("kwparams: %s", kwparams)
|
|
70
|
+
kwparams["progress_reporter"] = JobProgressReporter(job, 0.1)
|
|
71
|
+
LOGGER.debug("EXECUTE JOB: %s", job_payload)
|
|
72
|
+
result = method(*params, **kwparams)
|
|
73
|
+
LOGGER.debug("FINISHED JOB: %s", job_payload)
|
|
74
|
+
return result
|
|
75
|
+
|
|
76
|
+
def _executeJob(self, job):
|
|
77
|
+
try:
|
|
78
|
+
with job as j:
|
|
79
|
+
LOGGER.info("Processing Remote Job: %s", job)
|
|
80
|
+
result = self._executeJobPayload(j["payload"], job)
|
|
81
|
+
# LOGGER.debug("Remote Job Result: %s", result)
|
|
82
|
+
# ensure we always have a job_id for finished job payloads
|
|
83
|
+
result_id = self.queue._dicts_to_grid(result, metadata={"result": True, "job": job.job_id})
|
|
84
|
+
# update result directly from single job to ensure we don't loose it
|
|
85
|
+
LOGGER.info("Updating job %s with result %s", job.job_id, result_id)
|
|
86
|
+
updated_job = self.queue.collection.find_one_and_update(
|
|
87
|
+
filter={"_id": job.job_id}, update={"$set": {"result": result_id, "progress": 1}}, return_document=ReturnDocument.AFTER
|
|
88
|
+
)
|
|
89
|
+
# LOGGER.info(updated_job)
|
|
90
|
+
if updated_job is None:
|
|
91
|
+
raise RuntimeError(f"Failed to update job {job.job_id} with result {result_id} in database.")
|
|
92
|
+
job.result = result_id
|
|
93
|
+
LOGGER.info("Finished Remote Job producing result_id: %s", result_id)
|
|
94
|
+
print(result_id)
|
|
95
|
+
except Exception as exc:
|
|
96
|
+
LOGGER.error("Job %s failed with exception: %s", job.job_id, exc, exc_info=True)
|
|
97
|
+
|
|
98
|
+
def run(self):
|
|
99
|
+
self._alive = True
|
|
100
|
+
job = self.queue.get_job(self.job_id)
|
|
101
|
+
LOGGER.debug("Found job")
|
|
102
|
+
self._executeJob(job)
|
|
103
|
+
|
|
104
|
+
def terminate(self):
|
|
105
|
+
self._alive = False
|
mcrit/SpawningWorker.py
ADDED
|
@@ -0,0 +1,151 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
|
|
3
|
+
import logging
|
|
4
|
+
import os
|
|
5
|
+
import re
|
|
6
|
+
import subprocess
|
|
7
|
+
import threading
|
|
8
|
+
import time
|
|
9
|
+
import uuid
|
|
10
|
+
from typing import TYPE_CHECKING, Optional
|
|
11
|
+
|
|
12
|
+
from mcrit.config.McritConfig import McritConfig
|
|
13
|
+
from mcrit.minhash.MinHasher import MinHasher
|
|
14
|
+
from mcrit.queue.QueueFactory import QueueFactory
|
|
15
|
+
from mcrit.storage.StorageFactory import StorageFactory
|
|
16
|
+
from mcrit.Worker import Worker
|
|
17
|
+
|
|
18
|
+
if TYPE_CHECKING:
|
|
19
|
+
from mcrit.storage.StorageInterface import StorageInterface
|
|
20
|
+
|
|
21
|
+
logging.basicConfig(level=logging.INFO)
|
|
22
|
+
LOGGER = logging.getLogger(__name__)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class SpawningWorker(Worker):
|
|
26
|
+
def __init__(self, queue=None, config=None, storage: Optional["StorageInterface"] = None, profiling=False):
|
|
27
|
+
self._worker_id = f"Worker-{uuid.uuid4()}"
|
|
28
|
+
LOGGER.info(f"Starting as spawning worker: {self._worker_id}")
|
|
29
|
+
if config is None:
|
|
30
|
+
config = McritConfig()
|
|
31
|
+
|
|
32
|
+
if not queue:
|
|
33
|
+
queue = QueueFactory().getQueue(config, consumer_id=self._worker_id)
|
|
34
|
+
|
|
35
|
+
if profiling:
|
|
36
|
+
print("[!] Running as profiled application.")
|
|
37
|
+
profiling_path = os.path.abspath(os.path.join(os.path.dirname(os.path.abspath(__file__)), "..", "profiler"))
|
|
38
|
+
os.makedirs(profiling_path, exist_ok=True)
|
|
39
|
+
else:
|
|
40
|
+
profiling_path = None
|
|
41
|
+
super().__init__(queue=queue, config=config, storage=storage, profiling=profiling)
|
|
42
|
+
|
|
43
|
+
self.config = config
|
|
44
|
+
self._storage_config = config.STORAGE_CONFIG
|
|
45
|
+
self._minhash_config = config.MINHASH_CONFIG
|
|
46
|
+
self._shingler_config = config.SHINGLER_CONFIG
|
|
47
|
+
self._queue_config = config.QUEUE_CONFIG
|
|
48
|
+
self.minhasher = MinHasher(config.MINHASH_CONFIG, config.SHINGLER_CONFIG)
|
|
49
|
+
if storage:
|
|
50
|
+
self._storage = storage
|
|
51
|
+
else:
|
|
52
|
+
self._storage = StorageFactory.getStorage(config)
|
|
53
|
+
|
|
54
|
+
def __enter__(self):
|
|
55
|
+
return self
|
|
56
|
+
|
|
57
|
+
def __exit__(self, *args):
|
|
58
|
+
# TODO unregister our worker_id from all in-progress jobs found in the queue
|
|
59
|
+
self.queue.unregisterWorker()
|
|
60
|
+
self.queue.release_all_jobs()
|
|
61
|
+
|
|
62
|
+
#### NO REDIRECTION: SPAWM SINGLE JOB WORKERS INSTEAD ###
|
|
63
|
+
|
|
64
|
+
def _executeJobPayload(self, job_payload, job):
|
|
65
|
+
# instead of execution within our own context, spawn a new process as worker for this job payload
|
|
66
|
+
console_handle = subprocess.Popen(["python", "-m", "mcrit", "singlejobworker", "--job_id", str(job.job_id)], stdout=subprocess.PIPE, stderr=subprocess.PIPE)
|
|
67
|
+
# extract result_id from console_output
|
|
68
|
+
result_id = None
|
|
69
|
+
stdout_lines = []
|
|
70
|
+
|
|
71
|
+
def reader(pipe, label, accum):
|
|
72
|
+
try:
|
|
73
|
+
for line in iter(pipe.readline, b""):
|
|
74
|
+
decoded_line = line.decode("utf-8", errors="replace").rstrip()
|
|
75
|
+
if decoded_line:
|
|
76
|
+
LOGGER.info("%s logs from subprocess: %s", label, decoded_line)
|
|
77
|
+
if accum is not None:
|
|
78
|
+
accum.append(decoded_line)
|
|
79
|
+
except Exception:
|
|
80
|
+
LOGGER.exception("Exception in subprocess reader thread")
|
|
81
|
+
finally:
|
|
82
|
+
pipe.close()
|
|
83
|
+
|
|
84
|
+
t1 = threading.Thread(target=reader, args=(console_handle.stdout, "STDOUT", stdout_lines))
|
|
85
|
+
t2 = threading.Thread(target=reader, args=(console_handle.stderr, "STDERR", None))
|
|
86
|
+
t1.daemon = True
|
|
87
|
+
t2.daemon = True
|
|
88
|
+
t1.start()
|
|
89
|
+
t2.start()
|
|
90
|
+
|
|
91
|
+
try:
|
|
92
|
+
stdout_result, stderr_result = console_handle.communicate(timeout=self._queue_config.QUEUE_SPAWNINGWORKER_CHILDREN_TIMEOUT)
|
|
93
|
+
stdout_result = stdout_result.strip().decode("utf-8")
|
|
94
|
+
# TODO: log output from subprocess in the order it arrived
|
|
95
|
+
# instead of the split to stdout, stderr
|
|
96
|
+
if stdout_result:
|
|
97
|
+
LOGGER.info("STDOUT logs from subprocess: %s", stdout_result)
|
|
98
|
+
if stderr_result:
|
|
99
|
+
stderr_result = stderr_result.strip().decode("utf-8")
|
|
100
|
+
LOGGER.info("STDERR logs from subprocess: %s", stderr_result)
|
|
101
|
+
|
|
102
|
+
last_line = stdout_result.split("\n")[-1]
|
|
103
|
+
# successful output should be just the result_id in a single line
|
|
104
|
+
match = re.match("(?P<result_id>[0-9a-fA-F]{24})", last_line)
|
|
105
|
+
if match:
|
|
106
|
+
result_id = match.group("result_id")
|
|
107
|
+
except subprocess.TimeoutExpired:
|
|
108
|
+
LOGGER.error(f"Job {str(job.job_id)} running as child from SpawningWorker timed out during processing.")
|
|
109
|
+
console_handle.kill()
|
|
110
|
+
console_handle.wait()
|
|
111
|
+
|
|
112
|
+
t1.join()
|
|
113
|
+
t2.join()
|
|
114
|
+
|
|
115
|
+
if stdout_lines:
|
|
116
|
+
# Search backwards for result_id in case there are trailing empty lines or other output
|
|
117
|
+
for line in reversed(stdout_lines):
|
|
118
|
+
if line.strip():
|
|
119
|
+
match = re.match("(?P<result_id>[0-9a-fA-F]{24})", line.strip())
|
|
120
|
+
if match:
|
|
121
|
+
result_id = match.group("result_id")
|
|
122
|
+
break
|
|
123
|
+
return result_id
|
|
124
|
+
|
|
125
|
+
def _executeJob(self, job):
|
|
126
|
+
if time.time() - self.t_last_cleanup >= self.queue.clean_interval:
|
|
127
|
+
self.queue.clean()
|
|
128
|
+
self.t_last_cleanup = time.time()
|
|
129
|
+
try:
|
|
130
|
+
result_id = None
|
|
131
|
+
with job as j:
|
|
132
|
+
LOGGER.info("Processing Remote Job: %s", job)
|
|
133
|
+
result_id = self._executeJobPayload(j["payload"], job)
|
|
134
|
+
if result_id:
|
|
135
|
+
# result should have already been persisted by the child process, we repeat it here to close the job for the queue
|
|
136
|
+
job.result = result_id
|
|
137
|
+
LOGGER.info("Finished Remote Job with result_id: %s", result_id)
|
|
138
|
+
else:
|
|
139
|
+
LOGGER.info("Failed Running Remote Job: %s", job)
|
|
140
|
+
except Exception:
|
|
141
|
+
LOGGER.error("Error occurred while executing job: %s", job, exc_info=True)
|
|
142
|
+
|
|
143
|
+
def run(self):
|
|
144
|
+
self._alive = True
|
|
145
|
+
while self._alive:
|
|
146
|
+
job = self.queue.next()
|
|
147
|
+
if job:
|
|
148
|
+
LOGGER.debug("Found job")
|
|
149
|
+
self._executeJob(job)
|
|
150
|
+
else:
|
|
151
|
+
time.sleep(0.1)
|