closeyourit-ruby 0.10.0 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +16 -1
  3. data/lib/closeyourit/background_worker.rb +14 -14
  4. data/lib/closeyourit/breadcrumb.rb +2 -2
  5. data/lib/closeyourit/breadcrumb_buffer.rb +2 -2
  6. data/lib/closeyourit/client.rb +24 -24
  7. data/lib/closeyourit/configuration.rb +104 -103
  8. data/lib/closeyourit/event.rb +6 -6
  9. data/lib/closeyourit/events/error_event.rb +16 -16
  10. data/lib/closeyourit/events/job_metric_event.rb +11 -8
  11. data/lib/closeyourit/events/log_event.rb +17 -17
  12. data/lib/closeyourit/events/message_event.rb +4 -4
  13. data/lib/closeyourit/events/performance_issue_event.rb +4 -4
  14. data/lib/closeyourit/events/slow_method_event.rb +6 -6
  15. data/lib/closeyourit/events/slow_query_event.rb +4 -4
  16. data/lib/closeyourit/instrumenter.rb +6 -6
  17. data/lib/closeyourit/job_context.rb +61 -0
  18. data/lib/closeyourit/line_cache.rb +4 -4
  19. data/lib/closeyourit/log_buffer.rb +10 -10
  20. data/lib/closeyourit/log_device.rb +18 -18
  21. data/lib/closeyourit/monitor.rb +2 -2
  22. data/lib/closeyourit/performance/request_profile.rb +5 -5
  23. data/lib/closeyourit/performance/rollup.rb +3 -3
  24. data/lib/closeyourit/rails/active_job_extension.rb +36 -31
  25. data/lib/closeyourit/rails/capture_exceptions.rb +3 -3
  26. data/lib/closeyourit/rails/error_subscriber.rb +4 -4
  27. data/lib/closeyourit/rails/log_broadcast.rb +12 -12
  28. data/lib/closeyourit/rails/net_http_patch.rb +22 -22
  29. data/lib/closeyourit/rails/query_source.rb +3 -3
  30. data/lib/closeyourit/rails/railtie.rb +29 -65
  31. data/lib/closeyourit/rails/request_body.rb +7 -7
  32. data/lib/closeyourit/rails/request_context.rb +27 -27
  33. data/lib/closeyourit/scope.rb +29 -29
  34. data/lib/closeyourit/scrubber.rb +50 -50
  35. data/lib/closeyourit/sidekiq/error_handler.rb +2 -2
  36. data/lib/closeyourit/sidekiq/job_metrics_middleware.rb +39 -17
  37. data/lib/closeyourit/stats.rb +8 -8
  38. data/lib/closeyourit/subscribers/job_performance.rb +102 -24
  39. data/lib/closeyourit/subscribers/request_performance.rb +4 -4
  40. data/lib/closeyourit/subscribers/slow_query.rb +42 -20
  41. data/lib/closeyourit/trace_context.rb +22 -22
  42. data/lib/closeyourit/transport.rb +22 -22
  43. data/lib/closeyourit/usage_registry.rb +17 -16
  44. data/lib/closeyourit/version.rb +1 -1
  45. data/lib/closeyourit-ruby.rb +125 -125
  46. metadata +2 -1
@@ -4,29 +4,29 @@ require_relative "breadcrumb_buffer"
4
4
  require_relative "performance/request_profile"
5
5
 
6
6
  module CloseYourIt
7
- # Contesto per-richiesta (o per-job) isolato per execution-context (Fiber storage):
8
- # user/tags/extra/contexts/request. Letto da ErrorEvent#to_h sul thread chiamante (sincrono)
9
- # → il worker di invio non lo vede mai e lo scope non cola tra richieste.
7
+ # Per-request (or per-job) context isolated per execution context (Fiber storage):
8
+ # user/tags/extra/contexts/request. Read by ErrorEvent#to_h on the calling thread (synchronous)
9
+ # → the send worker never sees it and the scope does not leak between requests.
10
10
  class Scope
11
11
  STORAGE_KEY = :__closeyourit_scope
12
12
 
13
13
  class << self
14
- # Scope dell'execution-context corrente. Usa `ActiveSupport::IsolatedExecutionState` quando
15
- # presente (rispetta isolation_level: thread su Puma, fiber su Falcon), altrimenti
16
- # `Thread.current` (thread-local puro, NON ereditato dai thread figli → niente bleed).
14
+ # Scope of the current execution context. Uses `ActiveSupport::IsolatedExecutionState` when
15
+ # present (honors isolation_level: thread on Puma, fiber on Falcon), otherwise
16
+ # `Thread.current` (pure thread-local, NOT inherited by child threads → no bleed).
17
17
  def current
18
18
  store[STORAGE_KEY] ||= new
19
19
  end
20
20
 
21
- # Re-installa uno scope salvato in precedenza. Serve a `after_discard` (ActiveJob 7.1+) per
22
- # riprendere lo scope arricchito durante il `perform` — tag/contesti/breadcrumb, incluse le query —
23
- # dopo che il reset di fine `perform` l'ha sganciato dallo storage (CYRB-19). Simmetrico a `reset!`.
21
+ # Reinstalls a previously saved scope. Used by `after_discard` (ActiveJob 7.1+) to get back the
22
+ # scope enriched during `perform` — tags/contexts/breadcrumbs, queries included — after the
23
+ # end-of-`perform` reset detached it from storage (CYRB-19). Symmetric to `reset!`.
24
24
  def current=(scope)
25
25
  store[STORAGE_KEY] = scope
26
26
  end
27
27
 
28
- # Azzera lo scope corrente — chiamato in `ensure` da middleware e job (su Puma il
29
- # thread è riusato: senza reset lo scope colerebbe nella richiesta successiva).
28
+ # Clears the current scope — called in `ensure` by middleware and jobs (on Puma the
29
+ # thread is reused: without a reset the scope would leak into the next request).
30
30
  def reset!
31
31
  store[STORAGE_KEY] = nil
32
32
  end
@@ -73,8 +73,8 @@ module CloseYourIt
73
73
  @breadcrumbs.add(breadcrumb)
74
74
  end
75
75
 
76
- # Profilo di performance per-richiesta (query + HTTP esterne). Lazy: creato al primo accesso,
77
- # azzerato da #clear a fine richiesta. Il verdetto lo calcola Subscribers::RequestPerformance.
76
+ # Per-request performance profile (queries + external HTTP). Lazy: created on first access,
77
+ # reset by #clear at the end of the request. Subscribers::RequestPerformance computes the verdict.
78
78
  def performance_profile
79
79
  @performance_profile ||= Performance::RequestProfile.new
80
80
  end
@@ -88,17 +88,17 @@ module CloseYourIt
88
88
  @rack_env = nil
89
89
  @trace_id = nil
90
90
  @replay_session_id = nil
91
- # Contesto di trace W3C della richiesta (CloseYourIt::TraceContext): popolato solo con la
92
- # propagazione opt-in attiva, consumato dal patch Net::HTTP per gli header d'uscita.
91
+ # W3C trace context of the request (CloseYourIt::TraceContext): populated only when opt-in
92
+ # propagation is on, consumed by the Net::HTTP patch for outgoing headers.
93
93
  @trace_context = nil
94
94
  @breadcrumbs = BreadcrumbBuffer.new(CloseYourIt.configuration.max_breadcrumbs)
95
95
  @performance_profile = nil
96
96
  end
97
97
 
98
- # Sottoinsieme non vuoto in forma evento Sentry (user/tags/extra/contexts/request),
99
- # fuso nel payload da ErrorEvent#to_h. tags/extra/contexts passano dallo Scrubber (denylist
100
- # ricorsiva per chiave): il backend NON li ri-scruba (Errors::Ingest::Normalize li conserva
101
- # verbatim), quindi questa è l'unica rete di sicurezza contro le chiavi sensibili lì — R2.
98
+ # Non-empty subset in Sentry event shape (user/tags/extra/contexts/request), merged into the
99
+ # payload by ErrorEvent#to_h. tags/extra/contexts go through the Scrubber (recursive denylist by
100
+ # key): the backend does NOT scrub them again (Errors::Ingest::Normalize keeps them verbatim),
101
+ # so this is the only safety net against sensitive keys there — R2.
102
102
  def to_event_hash
103
103
  {
104
104
  "user" => serialize_user,
@@ -112,26 +112,26 @@ module CloseYourIt
112
112
 
113
113
  private
114
114
 
115
- # contexts utente + `replay.replay_id` (session replay) quando presente sullo scope: il
116
- # backend legge contexts.replay.replay_id per legare l'errore server al video (stesso punto
117
- # del percorso JS). replay/replay_id non sono chiavi sensibili → sopravvivono allo Scrubber.
115
+ # User contexts + `replay.replay_id` (session replay) when present on the scope: the backend
116
+ # reads contexts.replay.replay_id to link the server error to the video (same spot as the JS
117
+ # path). replay/replay_id are not sensitive keys → they survive the Scrubber.
118
118
  def contexts_payload
119
119
  merged = @contexts.dup
120
120
  merged["replay"] = { "replay_id" => @replay_session_id } if @replay_session_id
121
121
  scrub(presence(merged))
122
122
  end
123
123
 
124
- # Redige i valori delle chiavi sensibili preservando la struttura (es. contexts.runtime resta
125
- # intatto, solo i valori sotto chiavi sensibili diventano [FILTERED]). Riusa lo Scrubber della
126
- # configurazione, lo stesso percorso di breadcrumb.data e degli attributi di log.
124
+ # Redacts the values of sensitive keys preserving the structure (e.g. contexts.runtime stays
125
+ # intact, only values under sensitive keys become [FILTERED]). Reuses the configuration's
126
+ # Scrubber, the same path as breadcrumb.data and log attributes.
127
127
  def scrub(hash)
128
128
  return hash if hash.nil?
129
129
 
130
130
  Scrubber.new(CloseYourIt.configuration).filter_params(hash)
131
131
  end
132
132
 
133
- # Request context + body params (`request.data`) estratti LAZY qui — cioè solo quando un
134
- # evento viene davvero costruito, mai sul percorso felice della richiesta.
133
+ # Request context + body params (`request.data`) extracted LAZILY here — i.e. only when an
134
+ # event is actually built, never on the request's happy path.
135
135
  def request_payload
136
136
  return nil if @request.nil?
137
137
 
@@ -148,8 +148,8 @@ module CloseYourIt
148
148
  nil
149
149
  end
150
150
 
151
- # `user.id` sempre; email/ip_address/username solo con `send_pii` (il backend li strippa
152
- # comunque — difesa in profondità).
151
+ # `user.id` always; email/ip_address/username only with `send_pii` (the backend strips them
152
+ # anyway — defense in depth).
153
153
  def serialize_user
154
154
  return nil if @user.empty?
155
155
  return @user if CloseYourIt.configuration.send_pii
@@ -1,83 +1,83 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module CloseYourIt
4
- # Rimozione PII dai payload: filtro chiavi sensibili, normalizzazione SQL, scrub messaggi.
5
- # Privacy-by-default — vedi PDR §9.
4
+ # PII removal from payloads: sensitive key filter, SQL normalization, message scrubbing.
5
+ # Privacy-by-default — see PDR §9.
6
6
  class Scrubber
7
7
  FILTERED = "[FILTERED]"
8
8
 
9
- # Token di chiavi sempre redatti (match per sottostringa, normalizzato). Allineato al regex
10
- # SENSITIVE_KEY di backend e client Dart (parità client-side) — vedi
11
- # Errors/Logs::Ingest::Normalize::SENSITIVE_KEY. Copre credenziali/segreti e PII:
9
+ # Key tokens that are always redacted (substring match, normalized). Aligned with the
10
+ # SENSITIVE_KEY regex of the backend and the Dart client (client-side parity) — see
11
+ # Errors/Logs::Ingest::Normalize::SENSITIVE_KEY. Covers credentials/secrets and PII:
12
12
  # /pass|secret|token|api[_-]?key|apikey|authorization|cookie|csrf|credit|card|cvv|ssn|iban|
13
13
  # email|phone|telephone|mobile|dob|birth|passport|bearer|session|pin|pan/i
14
- # `pass` copre password/passwd/pass_code/passkey/passphrase; `cookie` copre set-cookie;
15
- # `credit`+`card` coprono credit_card; `phone` copre telephone/smartphone; `birth` copre
16
- # date_of_birth. Match per sottostringa → privilegia l'over-redaction (privacy-by-default):
17
- # p.es. `company_name` (contiene `pan`) o `shipping` (contiene `pin`) vengono redatti — è
18
- # accettabile, meglio redigere troppo che perdere PII.
19
- # CYRB-3: la lista ometteva email/phone/dob → i bind delle query lente li leakavano nel pannello.
14
+ # `pass` covers password/passwd/pass_code/passkey/passphrase; `cookie` covers set-cookie;
15
+ # `credit`+`card` cover credit_card; `phone` covers telephone/smartphone; `birth` covers
16
+ # date_of_birth. Substring match → favors over-redaction (privacy-by-default):
17
+ # e.g. `company_name` (contains `pan`) or `shipping` (contains `pin`) get redacted — that is
18
+ # acceptable, better to redact too much than to leak PII.
19
+ # CYRB-3: the list omitted email/phone/dob → slow query binds leaked them into the panel.
20
20
  DENYLIST = %w[
21
21
  pass secret token api_key apikey authorization
22
22
  cookie csrf credit card cvv ssn iban
23
23
  email phone telephone mobile dob birth passport bearer session pin pan
24
24
  ].freeze
25
25
 
26
- # CYRB-23 — redazione INTEGRATA delle credenziali di autenticazione nel testo libero.
27
- # `filter_params`/`filter_value` coprono gli header STRUTTURATI (la chiave `authorization` è in
28
- # DENYLIST); un `Authorization: Bearer <token>` incollato dentro `exception.message` o dentro la
29
- # riga di un log è testo, non una chiave, e prima di questo ticket partiva in chiaro perché
30
- # `scrub_message` applicava solo i pattern configurati dall'utente (default `[]`).
26
+ # CYRB-23 — BUILT-IN redaction of authentication credentials in free text.
27
+ # `filter_params`/`filter_value` cover STRUCTURED headers (the `authorization` key is in
28
+ # DENYLIST); an `Authorization: Bearer <token>` pasted into `exception.message` or into a log
29
+ # line is text, not a key, and before this ticket it was sent in clear because `scrub_message`
30
+ # applied only the user-configured patterns (default `[]`).
31
31
  #
32
- # Forma del testo redatto (decisa qui una volta per tutti gli SDK — gemelli CYDA-26/CYJS-38/
33
- # CYPY-15): chiave e schema restano leggibili, sparisce SOLO la credenziale →
34
- # `Authorization: Bearer [FILTERED]`. Il messaggio resta diagnosticamente utile (si vede CHE
35
- # c'era un header e con quale schema) senza trasportare il segreto.
32
+ # Shape of the redacted text (decided here once for all SDKs — twins CYDA-26/CYJS-38/
33
+ # CYPY-15): key and scheme stay readable, ONLY the credential disappears →
34
+ # `Authorization: Bearer [FILTERED]`. The message stays useful for diagnosis (you see THAT
35
+ # there was a header and with which scheme) without carrying the secret.
36
36
  AUTH_SCHEME = /(?i:Bearer|Basic|Token)/
37
37
 
38
- # `token68` (RFC 7235) con padding `=` finale: la forma di una credenziale reale. Minimo 4
39
- # caratteri — un `Basic` di credenziali brevi è corto (`dTpw` = `u:p`); a filtrare la prosa
40
- # pensa NOT_PROSE, non la lunghezza.
38
+ # `token68` (RFC 7235) with trailing `=` padding: the shape of a real credential. Minimum 4
39
+ # characters — a `Basic` with short credentials is short (`dTpw` = `u:p`); filtering out prose
40
+ # is NOT_PROSE's job, not the length's.
41
41
  CREDENTIAL_TOKEN = %r{[A-Za-z0-9\-._~+/]{4,}={0,2}}
42
42
 
43
- # Guardia anti-falso-positivo per lo schema NUDO: quello che segue NON è una credenziale se ha
44
- # la forma di una parola scritta da un umano, cioè sole lettere in UNA di tre forme —
45
- # `minuscole`, `Capitalizzata`, `MAIUSCOLE` — e più corta di 20 caratteri. Salva "the bearer of
46
- # bad news", "invalid bearer credentials", "Basic HTTP authentication", "Bearer Token expired".
47
- # Tutto il resto è credenziale: entropia (cifre, `_`, `-`, `.`), maiuscole interne miste
48
- # (`dTpw`, base64) o lunghezza ≥ 20 (nessuna parola è così lunga, mentre un token opaco di sole
49
- # minuscole sì).
50
- # Trade-off accettato: un CamelCase subito dopo lo schema (`Token MyAppName`) viene redatto.
43
+ # False-positive guard for the BARE scheme: what follows is NOT a credential if it has the shape
44
+ # of a human-written word, i.e. letters only in ONE of three forms — `lowercase`, `Capitalized`,
45
+ # `UPPERCASE` — and shorter than 20 characters. It spares "the bearer of bad news",
46
+ # "invalid bearer credentials", "Basic HTTP authentication", "Bearer Token expired".
47
+ # Everything else is a credential: entropy (digits, `_`, `-`, `.`), mixed inner capitals
48
+ # (`dTpw`, base64) or length ≥ 20 (no word is that long, while an opaque lowercase-only token is).
49
+ # Accepted trade-off: a CamelCase word right after the scheme (`Token MyAppName`) gets redacted.
51
50
  PROSE_WORD = /(?:[a-z]{1,19}|[A-Z][a-z]{0,18}|[A-Z]{1,19})/
52
51
  NOT_PROSE = %r{(?!\[FILTERED\])(?!#{PROSE_WORD}(?![A-Za-z0-9\-._~+/=]))}
53
52
 
54
- # Valore di un header incollato nel testo: sequenza di pezzi non delimitatori, dove una stringa
55
- # fra virgolette conta come UN pezzo. NON `\S+`, che si mangerebbe la `)` di
56
- # "(Authorization: Bearer <token>)"; ma nemmeno una classe secca, che si fermerebbe alla prima
57
- # virgoletta lasciando in chiaro il segreto di `Token token="<token>"` (sintassi RFC 7235 e
58
- # `authenticate_with_http_token` di Rails) e di un frammento JSON incollato nel messaggio.
53
+ # Value of a header pasted into text: a sequence of non-delimiter pieces, where a quoted string
54
+ # counts as ONE piece. NOT `\S+`, which would eat the `)` of "(Authorization: Bearer <token>)";
55
+ # but not a plain class either, which would stop at the first quote leaving in clear the secret
56
+ # of `Token token="<token>"` (RFC 7235 syntax and Rails' `authenticate_with_http_token`) and of a
57
+ # JSON fragment pasted into the message.
59
58
  QUOTED = /"[^"\n]*"|'[^'\n]*'/
60
59
  HEADER_VALUE = /(?:#{QUOTED}|[^\s,;)\]}"'<>])+/
61
60
 
62
- # `Authorization:`/`authorization=` (anche `Proxy-Authorization`, anche `authorization_header`)
63
- # seguito dal valore: qui la chiave è esplicita, quindi si redige senza euristica di prosa —
64
- # over-redaction voluta, coerente con la DENYLIST. `[ \t]` e non `\s`: mai oltre il newline.
65
- # Il lookahead copre lo schema OPZIONALE: senza, su un testo già redatto il motore farebbe
66
- # backtracking sul ramo "senza schema" e redigerebbe la parola `Bearer` stessa (non idempotente).
61
+ # `Authorization:`/`authorization=` (also `Proxy-Authorization`, also `authorization_header`)
62
+ # followed by the value: here the key is explicit, so it is redacted without prose heuristics —
63
+ # intentional over-redaction, consistent with the DENYLIST. `[ \t]` not `\s`: never past newline.
64
+ # The lookahead covers the OPTIONAL scheme: without it, on already redacted text the engine would
65
+ # backtrack onto the "no scheme" branch and redact the word `Bearer` itself (not idempotent).
67
66
  AUTH_HEADER = /((?i:(?:proxy[_-]?)?authorization)[\w-]*["']?[ \t]*[:=][ \t]*)(?!(?:#{AUTH_SCHEME}[ \t]+)?\[FILTERED\])(#{AUTH_SCHEME}[ \t]+)?#{HEADER_VALUE}/
68
67
 
69
- # Schema nudo (`Bearer <token>` senza il nome dell'header) seguito da qualcosa che ha la forma
70
- # di una credenziale: qui la guardia di prosa serve, la parola "bearer" ricorre nei messaggi.
68
+ # Bare scheme (`Bearer <token>` without the header name) followed by something shaped like a
69
+ # credential: here the prose guard is needed, the word "bearer" shows up in messages.
71
70
  AUTH_CREDENTIAL = /(\b#{AUTH_SCHEME}[ \t]+)#{NOT_PROSE}#{CREDENTIAL_TOKEN}/
72
71
 
72
+ DOLLAR_LITERAL = /\$([A-Za-z_][A-Za-z0-9_]*|)\$.*?(?:\$\1\$|\z)/m
73
73
  STRING_LITERAL = /'(?:[^']|'')*'/
74
- NUMERIC_LITERAL = /\b\d+(?:\.\d+)?\b/
74
+ NUMERIC_LITERAL = /(?<![\w$])\d+(?:\.\d+)?\b/
75
75
 
76
76
  def initialize(configuration)
77
77
  @configuration = configuration
78
78
  end
79
79
 
80
- # Filtra ricorsivamente Hash/Array sostituendo i valori delle chiavi sensibili.
80
+ # Recursively filters Hash/Array replacing the values of sensitive keys.
81
81
  def filter_params(value)
82
82
  case value
83
83
  when Hash
@@ -91,15 +91,15 @@ module CloseYourIt
91
91
  end
92
92
  end
93
93
 
94
- # Maschera i literal (stringa/numerici) nello SQL, preservando la struttura.
94
+ # Masks the (string/numeric) literals in SQL, preserving the structure.
95
95
  def obfuscate_sql(sql)
96
96
  return sql if sql.nil? || !@configuration.obfuscate_sql
97
97
 
98
- sql.to_s.gsub(STRING_LITERAL, "?").gsub(NUMERIC_LITERAL, "?")
98
+ sql.to_s.gsub(DOLLAR_LITERAL, "?").gsub(STRING_LITERAL, "?").gsub(NUMERIC_LITERAL, "?")
99
99
  end
100
100
 
101
- # Regola integrata (schemi di autenticazione) PRIMA dei pattern utente: le due protezioni si
102
- # sommano, la prima non è disattivabile — privacy-by-default (CYRB-23).
101
+ # Built-in rule (authentication schemes) BEFORE the user patterns: the two protections add up,
102
+ # the first cannot be disabled — privacy-by-default (CYRB-23).
103
103
  def scrub_message(message)
104
104
  return message if message.nil?
105
105
 
@@ -112,7 +112,7 @@ module CloseYourIt
112
112
  end
113
113
  end
114
114
 
115
- # Valore di un singolo bind/argomento: redatto se il nome (colonna/parametro) è sensibile.
115
+ # Value of a single bind/argument: redacted when the name (column/parameter) is sensitive.
116
116
  def filter_value(key, value)
117
117
  sensitive_key?(key) ? FILTERED : value
118
118
  end
@@ -2,8 +2,8 @@
2
2
 
3
3
  module CloseYourIt
4
4
  module Sidekiq
5
- # Error handler Sidekiq (registrato dal railtie solo se Sidekiq è presente). Sidekiq invoca
6
- # `call(exception, context, config)` e NON ri-solleva → qui catturiamo e basta.
5
+ # Sidekiq error handler (registered by the railtie only when Sidekiq is present). Sidekiq calls
6
+ # `call(exception, context, config)` and does NOT re-raise → here we just capture.
7
7
  class ErrorHandler
8
8
  def call(exception, context, _config = nil)
9
9
  apply_job_scope(context)
@@ -4,40 +4,62 @@ require_relative "../subscribers/job_performance"
4
4
 
5
5
  module CloseYourIt
6
6
  module Sidekiq
7
- # Server middleware Sidekiq (registrato dal railtie solo se Sidekiq è presente) che misura la durata
8
- # di esecuzione e l'attesa in coda del job, poi delega a Subscribers::JobPerformance l'emissione
9
- # delle metriche oltre soglia. Non altera il job: cronometra attorno allo `yield` e ri-solleva
10
- # qualunque errore invariato (la cattura degli errori è dell'ErrorHandler). La misurazione avviene
11
- # nell'`ensure`, così è presa anche per i job che sollevano; la telemetria è isolata (un errore nel
12
- # nostro codice non disturba mai il job ospite). No-op effettivo se `monitor_jobs` è OFF (la
13
- # decisione vive in #record).
7
+ # Sidekiq server middleware (registered by the railtie only when Sidekiq is present) that measures
8
+ # the job's execution duration and queue wait, then delegates emitting over-threshold metrics to
9
+ # Subscribers::JobPerformance. It does not alter the job: it times around `yield` and re-raises
10
+ # any error unchanged (error capture belongs to the ErrorHandler). Measurement happens in `ensure`,
11
+ # so it is taken for jobs that raise too; telemetry is isolated (an error in our code never
12
+ # disturbs the host job). Effectively a no-op when `monitor_jobs` is OFF (the decision lives in #record).
14
13
  class JobMetricsMiddleware
15
14
  def initialize(subscriber = nil)
16
15
  @subscriber = subscriber
17
16
  end
18
17
 
19
18
  def call(_worker, job, queue)
19
+ return yield if active_job_owned?(job)
20
+
20
21
  started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
21
- # L'attesa in coda si conosce all'INIZIO (now - enqueued_at); Sidekiq mette enqueued_at come
22
- # epoch in secondi. Calcolata prima dello yield per non includere la durata del job.
23
- latency = Subscribers::JobPerformance.latency_ms(job["enqueued_at"], now: Time.now.utc)
24
- yield
22
+ context = observe(job, queue)
23
+ value = yield
24
+ succeeded = true
25
+ value
25
26
  ensure
26
- emit(job, queue, started, latency)
27
+ emit(job, queue, started, context, succeeded) if started
27
28
  end
28
29
 
29
30
  private
30
31
 
31
- def emit(job, queue, started, latency)
32
+ def active_job_owned?(job)
33
+ %w[ActiveJob::QueueAdapters::SidekiqAdapter::JobWrapper Sidekiq::ActiveJob::Wrapper].include?(job["class"]) &&
34
+ Subscribers::JobPerformance.active_job_installed?
35
+ end
36
+
37
+ def observe(job, queue)
38
+ # Sidekiq 8 changed the timestamp wire unit; do not infer it from magnitude.
39
+ version = ::Sidekiq::VERSION if defined?(::Sidekiq::VERSION)
40
+ enqueued = JobContext.sidekiq_time(job["enqueued_at"], version: version)
41
+ JobContext.build(framework: "sidekiq", name: job["wrapped"] || job["class"],
42
+ queue: queue || job["queue"], job_id: job["jid"], attempt: attempt(job), retry_count: job["retry_count"],
43
+ enqueued_at: enqueued, started_at: Time.now.utc)
44
+ rescue StandardError
45
+ nil
46
+ end
47
+
48
+ def emit(job, queue, started, context, succeeded)
32
49
  duration_ms = (Process.clock_gettime(Process::CLOCK_MONOTONIC) - started) * 1000.0
50
+ if context
51
+ context = context.merge("duration_ms" => duration_ms, "attempt_outcome" => succeeded ? "success" : "error",
52
+ "terminal" => succeeded ? true : nil)
53
+ JobContext.log(context, CloseYourIt.configuration)
54
+ end
33
55
  subscriber.record(
34
56
  job_class: job["wrapped"] || job["class"],
35
57
  queue: queue || job["queue"],
36
58
  adapter: "sidekiq",
37
59
  duration_ms: duration_ms,
38
- queue_latency_ms: latency,
60
+ queue_latency_ms: context && context["queue_wait_ms"],
39
61
  attempt: attempt(job),
40
- trace_id: job["jid"]
62
+ trace_id: job["jid"], job_context: context
41
63
  )
42
64
  rescue StandardError => e
43
65
  CloseYourIt.internal_logger.error("CloseYourIt job metrics: #{e.class}: #{e.message}")
@@ -47,8 +69,8 @@ module CloseYourIt
47
69
  @subscriber ||= Subscribers::JobPerformance.new
48
70
  end
49
71
 
50
- # Numero di esecuzione 1-based. Sidekiq NON imposta `retry_count` al primo run (nil), lo porta a 0
51
- # al primo retry, 1 al secondo, ... → attempt = retry_count + 2 quando presente, 1 al primo run.
72
+ # 1-based execution number. Sidekiq does NOT set `retry_count` on the first run (nil), sets it to 0
73
+ # on the first retry, 1 on the second, ... → attempt = retry_count + 2 when present, 1 on the first run.
52
74
  def attempt(job)
53
75
  count = job["retry_count"]
54
76
  count.nil? ? 1 : count + 2
@@ -3,11 +3,11 @@
3
3
  require "concurrent"
4
4
 
5
5
  module CloseYourIt
6
- # Contatori diagnostici thread-safe del client: quanti eventi sono stati accodati,
7
- # scartati (coda piena / before_send / sampling), spediti con successo, falliti (rete o status
8
- # non-2xx) e, tra i falliti, quanti per timeout di rete. Rendono visibili i fallimenti silenziosi
9
- # del trasporto fire-and-forget. `timeout` è un sotto-conteggio di `failed` (un timeout resta un
10
- # fallimento d'invio): li teniamo distinti per isolare i problemi di connettività dai non-2xx.
6
+ # Thread-safe client diagnostic counters: how many events were enqueued, dropped (full queue /
7
+ # before_send / sampling), sent successfully, failed (network or non-2xx status) and, among the
8
+ # failed, how many by network timeout. They make the silent failures of the fire-and-forget
9
+ # transport visible. `timeout` is a sub-count of `failed` (a timeout is still a send failure):
10
+ # we keep them apart to isolate connectivity problems from non-2xx.
11
11
  #
12
12
  # CloseYourIt.stats.to_h # => { enqueued: 12, dropped: 0, sent: 11, failed: 1, timeout: 1 }
13
13
  class Stats
@@ -31,9 +31,9 @@ module CloseYourIt
31
31
  @counters.transform_values(&:value)
32
32
  end
33
33
 
34
- # Fotografia thread-safe dei contatori (ogni valore letto atomicamente). È lo stesso Hash di
35
- # `to_h`, con un nome esplicito per il caso d'uso "leggo la diagnostica locale" (CYRB-12): pura
36
- # lettura in-memory, non invia mai telemetria.
34
+ # Thread-safe snapshot of the counters (each value read atomically). It is the same Hash as
35
+ # `to_h`, with an explicit name for the "read the local diagnostics" use case (CYRB-12): a pure
36
+ # in-memory read, it never sends telemetry.
37
37
  alias_method :snapshot, :to_h
38
38
 
39
39
  def reset!
@@ -2,32 +2,66 @@
2
2
 
3
3
  require "time"
4
4
  require_relative "../events/job_metric_event"
5
+ require_relative "../job_context"
5
6
 
6
7
  module CloseYourIt
7
8
  module Subscribers
8
- # Misura durata di esecuzione e attesa in coda (queue latency) dei background job — ActiveJob e
9
- # Sidekiq — e, oltre le soglie configurate, emette metriche performance_issue (subtype `slow_job`
10
- # e `job_queue_latency`). Logica PURA e SENZA STATO condiviso: tutto arriva per parametri, quindi
11
- # job concorrenti sullo stesso thread/processo non si contaminano. Il wiring ad
12
- # ActiveSupport::Notifications e al middleware Sidekiq vive altrove (Railtie / JobMetricsMiddleware).
13
- # Rispetta il master switch `monitor_jobs`, le soglie e il `jobs_sample_rate`.
9
+ # Measures execution duration and queue wait (queue latency) of background jobs — ActiveJob and
10
+ # Sidekiq — and, beyond the configured thresholds, emits performance_issue metrics (subtypes
11
+ # `slow_job` and `job_queue_latency`). PURE logic WITHOUT shared state: everything comes in as
12
+ # parameters, so concurrent jobs on the same thread/process do not contaminate each other. The
13
+ # wiring to ActiveSupport::Notifications and the Sidekiq middleware lives elsewhere (Railtie /
14
+ # JobMetricsMiddleware). Honors the `monitor_jobs` master switch, the thresholds and `jobs_sample_rate`.
14
15
  class JobPerformance
16
+ REGISTRY_LOCK = Mutex.new
17
+ ACTIVE_JOB_SUBSCRIBERS = []
18
+
19
+ def self.active_job_installed?
20
+ REGISTRY_LOCK.synchronize { !ACTIVE_JOB_SUBSCRIBERS.empty? }
21
+ end
15
22
  def initialize(configuration = nil)
16
23
  @configuration = configuration
17
24
  end
18
25
 
19
- # Punto unico di decisione: dai valori misurati (durata e/o attesa) costruisce 0..2 metriche,
20
- # applica le soglie (stretto `>`, così X non genera rumore e X+1 sì) e il sampling, poi le spedisce
21
- # fire-and-forget. `duration_ms` e `queue_latency_ms` sono opzionali: ActiveJob li fornisce da due
22
- # hook distinti (perform_start → attesa, perform → durata), Sidekiq entrambi in una sola chiamata.
23
- # Il sampling è applicato SOLO ai candidati già oltre soglia (i job normali non consumano né
24
- # generano nulla).
26
+ def install
27
+ REGISTRY_LOCK.synchronize do
28
+ return self if @subscriptions || !ACTIVE_JOB_SUBSCRIBERS.empty?
29
+
30
+ @subscriptions = []
31
+ { "enqueue_retry" => :retry, "retry_stopped" => :exhausted, "discard" => :discarded }.each do |name, outcome|
32
+ subscribe(name) { |event| event.payload[:job]&.instance_variable_set(:@__closeyourit_job_outcome, outcome) }
33
+ end
34
+ subscribe("perform_start") { |event| active_job_started(event.payload[:job]) if event.payload[:job] }
35
+ subscribe("perform") do |event|
36
+ job = event.payload[:job]
37
+ if job
38
+ CloseYourIt.usage_registry.record("job", job.class.name) if CloseYourIt.enabled?
39
+ active_job_performed(job, event.duration, payload: event.payload)
40
+ end
41
+ end
42
+ ACTIVE_JOB_SUBSCRIBERS << self
43
+ self
44
+ end
45
+ end
46
+
47
+ def uninstall
48
+ Array(@subscriptions).each { |subscription| ActiveSupport::Notifications.unsubscribe(subscription) }
49
+ @subscriptions = nil
50
+ REGISTRY_LOCK.synchronize { ACTIVE_JOB_SUBSCRIBERS.delete(self) }
51
+ end
52
+
53
+ # Single decision point: from the measured values (duration and/or wait) it builds 0..2 metrics,
54
+ # applies the thresholds (strict `>`, so X makes no noise and X+1 does) and sampling, then sends
55
+ # them fire-and-forget. `duration_ms` and `queue_latency_ms` are optional: ActiveJob provides them
56
+ # from two separate hooks (perform_start → wait, perform → duration), Sidekiq both in one call.
57
+ # Sampling is applied ONLY to candidates already beyond the threshold (normal jobs neither
58
+ # consume nor generate anything).
25
59
  def record(job_class:, queue: nil, adapter: nil, duration_ms: nil, queue_latency_ms: nil,
26
- attempt: nil, trace_id: nil)
60
+ attempt: nil, trace_id: nil, job_context: nil)
27
61
  config = configuration
28
62
  return unless config.monitor_jobs
29
63
 
30
- common = { job_class: job_class, queue: queue, adapter: adapter, attempt: attempt, trace_id: trace_id }
64
+ common = { job_class: job_class, queue: queue, adapter: adapter, attempt: attempt, trace_id: trace_id, job_context: job_context }
31
65
  events = []
32
66
  events << build(config, "slow_job", duration_ms, common) if slow?(config, duration_ms)
33
67
  events << build(config, "job_queue_latency", queue_latency_ms, common) if waited?(config, queue_latency_ms)
@@ -38,34 +72,60 @@ module CloseYourIt
38
72
  nil
39
73
  end
40
74
 
41
- # Hook `perform_start.active_job`: l'attesa in coda è nota appena il job parte (now - enqueued_at).
75
+ # `perform_start.active_job` hook: the queue wait is known as soon as the job starts (now - the
76
+ # moment it was due to run).
42
77
  def active_job_started(job, now: Time.now.utc)
78
+ context = JobContext.build(framework: "active_job", name: job.class.name,
79
+ queue: (job.queue_name if job.respond_to?(:queue_name)), job_id: (job.job_id if job.respond_to?(:job_id)),
80
+ attempt: (job.executions if job.respond_to?(:executions)), enqueued_at: enqueued_at(job),
81
+ scheduled_at: (job.scheduled_at if job.respond_to?(:scheduled_at)), started_at: now)
82
+ job.instance_variable_set(:@__closeyourit_job_context, context)
43
83
  record(
44
84
  job_class: job.class.name,
45
85
  queue: (job.queue_name if job.respond_to?(:queue_name)),
46
86
  adapter: "active_job",
47
- queue_latency_ms: self.class.latency_ms(enqueued_at(job), now: now),
87
+ queue_latency_ms: context["queue_wait_ms"],
48
88
  attempt: (job.executions if job.respond_to?(:executions)),
49
- trace_id: (job.job_id if job.respond_to?(:job_id))
89
+ trace_id: (job.job_id if job.respond_to?(:job_id)), job_context: context
50
90
  )
51
91
  end
52
92
 
53
- # Hook `perform.active_job`: a fine esecuzione la durata è `event.duration` (ms).
54
- def active_job_performed(job, duration_ms)
93
+ # `perform.active_job` hook: at the end of execution the duration is `event.duration` (ms).
94
+ def active_job_performed(job, duration_ms, payload: {})
95
+ context = job.instance_variable_get(:@__closeyourit_job_context)
96
+ if context
97
+ outcome, terminal = active_job_outcome(job, payload)
98
+ context = context.merge("duration_ms" => duration_ms, "attempt_outcome" => outcome, "terminal" => terminal)
99
+ JobContext.log(context, configuration)
100
+ end
55
101
  record(
56
102
  job_class: job.class.name,
57
103
  queue: (job.queue_name if job.respond_to?(:queue_name)),
58
104
  adapter: "active_job",
59
105
  duration_ms: duration_ms,
60
106
  attempt: (job.executions if job.respond_to?(:executions)),
61
- trace_id: (job.job_id if job.respond_to?(:job_id))
107
+ trace_id: (job.job_id if job.respond_to?(:job_id)), job_context: context
62
108
  )
109
+ ensure
110
+ job.remove_instance_variable(:@__closeyourit_job_context) if job.instance_variable_defined?(:@__closeyourit_job_context)
111
+ job.remove_instance_variable(:@__closeyourit_job_outcome) if job.instance_variable_defined?(:@__closeyourit_job_outcome)
63
112
  end
64
113
 
65
- # Normalizza `enqueued_at` (Time, epoch Numerico in secondi come Sidekiq, o String ISO8601) in
66
- # attesa (ms) rispetto a `now`. nil o non parsabile → nil: nessuna metrica di attesa (adapter che
67
- # non popola l'istante di enqueue). Clamp a 0 se negativa (clock skew, enqueue "nel futuro"): lo
68
- # schema di ingest esige `duration_ms >= 0`.
114
+ def active_job_outcome(job, payload)
115
+ outcome = job.instance_variable_get(:@__closeyourit_job_outcome)
116
+ return [ "error", nil ] if payload[:exception_object] || payload[:exception]
117
+ return [ "retry", false ] if outcome == :retry
118
+ return [ "discarded", true ] if outcome == :discarded
119
+ return [ "error", nil ] if outcome == :exhausted
120
+ return [ "unknown", nil ] if outcome == :raised || payload[:aborted]
121
+
122
+ [ "success", true ]
123
+ end
124
+
125
+ # Normalizes `enqueued_at` (Time, Numeric epoch in seconds like Sidekiq, or ISO8601 String) into
126
+ # a wait (ms) relative to `now`. nil or unparsable → nil: no wait metric (adapter that does not
127
+ # populate the enqueue instant). Clamped to 0 when negative (clock skew, enqueue "in the future"):
128
+ # the ingest schema requires `duration_ms >= 0`.
69
129
  def self.latency_ms(enqueued_at, now:)
70
130
  started = to_time(enqueued_at)
71
131
  return nil if started.nil?
@@ -90,10 +150,28 @@ module CloseYourIt
90
150
 
91
151
  private
92
152
 
153
+ def subscribe(name, &callback)
154
+ @subscriptions << ActiveSupport::Notifications.monotonic_subscribe("#{name}.active_job") do |*args|
155
+ callback.call(ActiveSupport::Notifications::Event.new(*args))
156
+ rescue StandardError
157
+ CloseYourIt.internal_logger.warn("CloseYourIt job observation failed")
158
+ end
159
+ end
160
+
93
161
  def enqueued_at(job)
94
162
  job.enqueued_at if job.respond_to?(:enqueued_at)
95
163
  end
96
164
 
165
+ # A job delayed on purpose (set(wait:), retry_on backoff) is due at scheduled_at, not when it was
166
+ # enqueued: the planned delay is not queue wait (CYRB-27). No enqueued_at still means no metric.
167
+ def due_at(job)
168
+ enqueued = self.class.to_time(enqueued_at(job))
169
+ scheduled = self.class.to_time(job.scheduled_at) if job.respond_to?(:scheduled_at)
170
+ return enqueued if enqueued.nil? || scheduled.nil?
171
+
172
+ [ enqueued, scheduled ].max
173
+ end
174
+
97
175
  def configuration
98
176
  @configuration || CloseYourIt.configuration
99
177
  end