modelflowib 2.73__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
modelnewton.py ADDED
@@ -0,0 +1,2178 @@
1
+ # -*- coding: utf-8 -*-
2
+ """
3
+ Created on Fri Jun 19 19:49:50 2020
4
+
5
+ @author: IBH
6
+
7
+ Module which handles model differentiation, construction of jacobi matrizex,
8
+ companion matrices, eigenvalues and creates dense and sparse solving functions.
9
+
10
+ """
11
+ import matplotlib.pyplot as plt
12
+ import matplotlib as mpl
13
+
14
+ import pandas as pd
15
+ from sympy import sympify ,Symbol,Function
16
+
17
+ from collections import defaultdict, namedtuple
18
+ import itertools
19
+
20
+ import numpy as np
21
+ import scipy as sp
22
+ import os
23
+ import sys
24
+ from subprocess import run
25
+ import seaborn as sns
26
+ import ipywidgets as ip
27
+ import inspect
28
+ from itertools import chain, zip_longest
29
+ import fnmatch
30
+ from IPython.display import display,Latex, Markdown, HTML
31
+ from itertools import chain, zip_longest
32
+ from tqdm.notebook import tqdm
33
+ from pathlib import Path
34
+
35
+
36
+ from dataclasses import dataclass, field, asdict
37
+ from functools import lru_cache,cached_property
38
+
39
+ import re
40
+
41
+
42
+ import modelpattern as pt
43
+ # from modelclass import model, ttimer, insertModelVar
44
+
45
+
46
+ from modelmanipulation import split_frml,udtryk_parse,pastestring,stripstring
47
+ #import modeljupyter as mj
48
+
49
+ from modelhelp import tovarlag, ttimer, insertModelVar
50
+ import modeljupyter as mj
51
+ from modelhelp import debug_var
52
+
53
+ @dataclass
54
+ class diff_value_base:
55
+ ''' class define columns in database with values from differentiation'''
56
+
57
+ var : str # lhs var
58
+ pvar : str # rhs var
59
+ lag : int # lag of rhs var
60
+ var_plac : int # placement of lhs in array of endogeneous
61
+ pvar_plac : int # placement of lhs in array of endogeneous
62
+ pvar_endo : bool # is pvar an endogeneous variable
63
+ pvar_exo_plac : int # placement of lhs in array of endogeneous
64
+
65
+ @dataclass(unsafe_hash=True)
66
+ class diff_value_col(diff_value_base):
67
+ ''' The hash able class which can be used as pandas columns'''
68
+
69
+
70
+ @dataclass
71
+ class diff_value(diff_value_base):
72
+ ''' class to contain values from differentiation'''
73
+
74
+ number : int = field(default=0) # index relativ to start in current_per
75
+ date : any = field(default=0) # index in dataframe
76
+
77
+
78
+ class newton_diff():
79
+ '''
80
+ Class to handle Newton solving for un-normalized or normalized models, i.e., models of the form:
81
+
82
+ 0 = G(y, x)
83
+ y = F(y, x)
84
+
85
+ This class provides functionalities to differentiate model equations, analyze eigenvalues and eigenvectors,
86
+ and solve dynamic systems through various approaches.
87
+
88
+ **Provided Functions:**
89
+
90
+ - **eigenvector_plot**: Plot eigenvectors for a specified period.
91
+ - **eigplot**: Plot eigenvalues for a specific period in polar coordinates.
92
+ - **eigplot_all**: Plot all eigenvalues for specified periods in polar coordinates.
93
+ - **eigplot_all0**: Alternative method to plot all eigenvalues in polar coordinates.
94
+ - **get_df_comp_dict**: Get a dictionary of DataFrames representing companion matrices.
95
+ - **get_df_eigen_dict**: Get a dictionary of DataFrames containing eigenvalues and eigenvectors.
96
+ - **get_diff_df_1per**: Get a DataFrame of derivatives for one period.
97
+ - **get_diff_df_tot**: Get a DataFrame of stacked Jacobian matrices for the entire period range.
98
+ - **get_diff_mat_1per**: Get a dictionary of sparse matrices representing the Jacobian for one period.
99
+ - **get_diff_mat_all_1per**: Get a dictionary of all derivative matrices for one period, including all lags.
100
+ - **get_diff_mat_tot**: Get a sparse matrix representing the stacked Jacobian for the entire period range.
101
+ - **get_diff_melted**: Get a "melted" DataFrame of derivatives suitable for creating sparse matrices.
102
+ - **get_diff_melted_var**: Get a melted DataFrame of derivatives including variable information.
103
+ - **get_diff_values_all**: Get all derivative values in a structured format.
104
+ - **get_diffmodel**: Generate a model to calculate the partial derivatives of the original model.
105
+ - **get_eigen_jackknife**: Perform a jackknife analysis by computing eigenvalues with each variable excluded one at a time.
106
+ - **get_eigen_jackknife_abs**: Compute the absolute values of the largest eigenvalues from the jackknife analysis.
107
+ - **get_eigen_jackknife_abs_select**: Summarize the absolute largest eigenvalues for a specific year from the jackknife analysis.
108
+ - **get_eigen_jackknife_df**: Convert jackknife eigenvalue data into a DataFrame.
109
+ - **get_eigenvalues**: Calculate and return eigenvectors based on the companion matrix for a dynamic system.
110
+ - **get_feedback**: Static method to return feedback on the max absolute eigenvector and its sign.
111
+ - **get_solve1per**: Get a solving function for one period using precomputed Jacobian matrices.
112
+ - **get_solve1perlu**: Get a LU decomposition-based solving function for one period.
113
+ - **get_solvestacked**: Get a solving function for the stacked system of equations.
114
+ - **get_solvestacked_it**: Get an iterative solving function for the stacked system using a specified solver.
115
+ - **modeldiff**: Differentiate model equations with respect to endogenous variables.
116
+ - **show_diff**: Display expressions for differential coefficients for specified variables.
117
+ - **show_diff_latex**: Display LaTeX-formatted differential expressions and possibly their values.
118
+ - **show_stacked_diff**: Show selected rows and columns of the stacked Jacobian as a DataFrame.
119
+ '''
120
+
121
+
122
+ # def __init__(self, mmodel, df=None, endovar=None, onlyendocur=False,
123
+ # timeit=False, silent=True, forcenum=False, per='', ljit=0, nchunk=None, endoandexo=False):
124
+ # pass
125
+ # # [Methods implementations]
126
+
127
+ def __init__(self, mmodel, df = None , endovar = None,onlyendocur=False,
128
+ timeit=False, silent = True, forcenum=False,per='',ljit=0,nchunk=None,endoandexo=False,ng=0):
129
+ """
130
+
131
+
132
+ Args:
133
+ mmodel (TYPE): Model to analyze.
134
+ df (TYPE, optional): Dataframe. if None mmodel.lastdf will be used
135
+ endovar (TYPE, optional): if set defines which endogeneous to include . Defaults to None.
136
+ onlyendocur (TYPE, optional): Only calculate for the curren endogeneous variables. Defaults to False.
137
+ timeit (TYPE, optional): writeout time informations . Defaults to False.
138
+ silent (TYPE, optional): Defaults to True.
139
+ forcenum (TYPE, optional): Force differentiation to be numeric else try sumbolic (slower) Defaults to False.
140
+ per (TYPE, optional): Period for which to calculte the jacobi . Defaults to ''.
141
+ ljit (TYPE, optional): Trigger just in time compilation of the differential coiefficient. Defaults to 0.
142
+ nchunk (TYPE, optional): Chunks for which the model is written - relevant if ljit == True. Defaults to None (30 if ljit is set).
143
+ endoandexo (TYPE, optional): Calculate for both endogeneous and exogeneous . Defaults to False.
144
+ ng (TYPE, optional): Evaluate the derivatives model with the ng solver (res_ng) instead of res. Defaults to 0.
145
+
146
+ Returns:
147
+ None.
148
+
149
+ """
150
+ self.df = df if type(df) == pd.DataFrame else mmodel.lastdf
151
+ self.endovar = sorted(mmodel.endogene) if endovar == None else endovar
152
+ self.endoandexo = endoandexo
153
+ self.mmodel = mmodel
154
+ self.onlyendocur = onlyendocur
155
+ self.silent = silent
156
+ self.maxdif = 9999999999999
157
+ self.forcenum = forcenum
158
+ self.timeit= timeit
159
+ self.per=per
160
+ self.ljit=ljit
161
+ # jitting one huge unchunked function makes numba compilation explode,
162
+ # so when ljit is on the generated code is always chunked
163
+ self.nchunk = nchunk if nchunk is not None else (30 if ljit else None)
164
+ self.ng = ng
165
+ if not self.silent: print(f'Prepare model for calculate derivatives for Newton solver')
166
+ # for all equations list either left hand variable or ENDO in frml name
167
+ self.declared_endo_list0 = [pt.kw_frml_name(self.mmodel.allvar[v]['frmlname'], 'ENDO',v)
168
+ for v in self.endovar]
169
+ self.declared_endo_list = [v[:-6] if v.endswith('___RES') else v for v in self.declared_endo_list0] # real endogeneous variables we want to find not the ___res variables
170
+ self.declared_endo_set = set(self.declared_endo_list)
171
+ assert len(self.declared_endo_list) == len(self.declared_endo_set)
172
+ self.placdic = {v : i for i,v in enumerate(self.endovar)}
173
+ self.newmodel = mmodel.__class__
174
+
175
+ if self.endoandexo:
176
+ self.exovar = [v for v in sorted(mmodel.exogene) if not v in self.declared_endo_set]
177
+ self.exoplacdic = {v : i for i,v in enumerate(self.exovar)}
178
+ else:
179
+ self.exoplacdic = {}
180
+ # breakpoint()
181
+ self.diffendocur = self.modeldiff()
182
+ self.diff_model = self.get_diffmodel()
183
+
184
+
185
+ def modeldiff(self):
186
+ ''' Differentiate relations for self.enovar with respect to endogeneous variable
187
+ The result is placed in a dictory in the model instanse: model.diffendocur
188
+ '''
189
+
190
+ def numdif(model,v,rhv,delta = 0.005,silent=True) :
191
+ # print('**',model.allvar[v]['terms']['frml'])
192
+ def tout(t):
193
+ if t.lag:
194
+ return f'{t.var}({t.lag})'
195
+ return t.op+t.number+t.var
196
+
197
+ # breakpoint()
198
+ nt = model.allvar[v]['terms']
199
+ assignpos = nt.index(model.aequalterm) # find the position of =
200
+ rhsterms = nt[assignpos+1:-1]
201
+ vterm = udtryk_parse(rhv)[0]
202
+ plusterm = udtryk_parse(f'({rhv}+{delta/2})',funks=model.funks)
203
+ minusterm = udtryk_parse(f'({rhv}-{delta/2})',funks=model.funks)
204
+ plus = itertools.chain.from_iterable([plusterm if t == vterm else [t] for t in rhsterms])
205
+ minus = itertools.chain.from_iterable([minusterm if t == vterm else [t] for t in rhsterms])
206
+ eplus = f'({"".join(tout(t) for t in plus)})'
207
+ eminus = f'({"".join(tout(t) for t in minus)})'
208
+ expression = f'({eplus}-{eminus})/{delta}'
209
+ if (not silent) and False:
210
+ print(expression)
211
+ return expression
212
+
213
+
214
+ def findallvar(model,v):
215
+ '''Finds all endogenous variables which is on the right side of = in the expresion for variable v
216
+ lagged variables are included if self.onlyendocur == False '''
217
+ # print(v)
218
+ terms= self.mmodel.allvar[v]['terms'][model.allvar[v]['assigpos']:-1]
219
+ if self.endoandexo:
220
+ rhsvar={(nt.var+('('+nt.lag+')' if nt.lag != '' else '')) for nt in terms if nt.var}
221
+ rhsvar={tovarlag(nt.var,nt.lag) for nt in terms if nt.var}
222
+ else:
223
+ if self.onlyendocur :
224
+ rhsvar={tovarlag(nt.var,nt.lag) for nt in terms if nt.var and nt.lag == '' and nt.var in self.declared_endo_set}
225
+
226
+ else:
227
+ rhsvar={tovarlag(nt.var,nt.lag) for nt in terms if nt.var and nt.var in self.declared_endo_set}
228
+ var2=sorted(list(rhsvar))
229
+ return var2
230
+
231
+ with ttimer('Find espressions for partial derivatives',self.timeit):
232
+ clash = {var : Symbol(var) for var in self.mmodel.allvar.keys()}
233
+ # clash = {var :None for var in self.mmodel.allvar.keys()}
234
+ diffendocur={} #defaultdict(defaultdict) #here we wanmt to store the derivativs
235
+ i=0
236
+ for nvar,v in enumerate(self.endovar):
237
+ if nvar >= self.maxdif:
238
+ break
239
+ if not self.silent and 0:
240
+ print(f'Now differentiating {v} {nvar}')
241
+
242
+ endocur = findallvar(self.mmodel,v)
243
+
244
+ diffendocur[v]={}
245
+ t=self.mmodel.allvar[v]['frml'].upper()
246
+ a,fr,n,udtryk=split_frml(t)
247
+ udtryk=udtryk
248
+ udtryk=re.sub(r'LOG\(','log(',udtryk) # sympy uses lover case for log and exp
249
+ udtryk=re.sub(r'EXP\(','exp(',udtryk)
250
+ lhs,rhs=udtryk.split('=',1)
251
+ post = '_____XXYY'
252
+ try:
253
+ if not self.forcenum:
254
+ # kat=sympify(rhs[0:-1], md._clash) # we take the the $ out _clash1 makes I is not taken as imiganary
255
+ lookat = pastestring(rhs[0:-1], post,onlylags=True,funks= self.mmodel.funks)
256
+ lookat=re.sub(r'LOG\(','log(',lookat) # sympy uses lover case for log and exp
257
+ lookat=re.sub(r'EXP\(','exp(',lookat)
258
+
259
+ kat=sympify(lookat,clash) # we take the the $ out _clash1 makes I is not taken as imiganary
260
+ # debug_var(lookat)
261
+ except Exception as inst:
262
+ # breakpoint()
263
+ print(inst)
264
+ e = sys.exc_info()[0]
265
+ print(e)
266
+ print(lookat)
267
+ # print({c:type(s) for c,s in clash.items()})
268
+ print('* Problem sympify ',lhs,'=',rhs[0:-1],'\n')
269
+ for rhv in endocur:
270
+ try:
271
+ if not self.forcenum:
272
+ try:
273
+ ud=str(kat.diff(sympify(pastestring(rhv, post,funks=self.mmodel.funks,onlylags=True ),clash)))
274
+ ud = stripstring(ud,post,self.mmodel.funks)
275
+ ud = re.sub(pt.namepat+r'(?:(\()([0-9]*)(\)))',r'\g<1>\g<2>+\g<3>\g<4>',ud)
276
+ except:
277
+ ud = numdif(self.mmodel,v,rhv,silent=self.silent)
278
+
279
+ if 'DERIVATIVE(' in ud.upper() :
280
+ # sympy could not differentiate symbolically -> fall back to numeric
281
+ ud = numdif(self.mmodel,v,rhv,silent=self.silent)
282
+ if not self.silent and 0: print('numdif of {rhv}')
283
+ else:
284
+ # forcenum: differentiate numerically
285
+ ud = numdif(self.mmodel,v,rhv,silent=self.silent)
286
+ diffendocur[v.upper()][rhv.upper()]=ud
287
+
288
+ except Exception as e:
289
+ print(e,'\nwe have a serious problem deriving:',lhs,'|',rhv,'\n',lhs,'=',rhs)
290
+ # breakpoint()
291
+
292
+ i+=1
293
+ if not self.silent:
294
+ print('Model :',self.mmodel.name)
295
+ print('Number of endogeneus variables :',len(diffendocur))
296
+ print('Number of derivatives :',i)
297
+ return diffendocur
298
+
299
+ def show_diff(self,pat='*'):
300
+ ''' Displays espressions for differential koifficients for a variable
301
+ if var ends with * all matchning variables are displayes'''
302
+ l=self.mmodel.maxnavlen
303
+ xx = self.get_diff_values_all()
304
+ for v in [var for p in pat.split() for var in fnmatch.filter(self.declared_endo_set,p)]:
305
+ # breakpoint()
306
+ thisvar = v if v in self.mmodel.endogene else v+'___RES'
307
+ print(self.mmodel.allvar[thisvar]['frml'])
308
+ for e in self.diffendocur[thisvar]:
309
+ print(f'd{v}/d( {e} ) = {self.diffendocur[thisvar][e]}')
310
+ print(f'& = & {self.diffvalues[thisvar][e].iloc[:,:3]}')
311
+ print(' ')
312
+
313
+ def show_stacked_diff(self,time=None, lhs='',rhs='',dec=2,show=True):
314
+ '''
315
+
316
+
317
+ Parameters
318
+ ----------
319
+ time : list, optional
320
+ DESCRIPTION. The default is None. Time for which to retrieve stacked jacobi
321
+ lhs : string, optional
322
+ DESCRIPTION. The default is ''. Left hand side variables
323
+ rhs : TYPE, optional
324
+ DESCRIPTION. The default is ''. Right hand side variabnles
325
+ dec : TYPE, optional
326
+ DESCRIPTION. The default is 2.
327
+ show : TYPE, optional
328
+ DESCRIPTION. The default is True.
329
+
330
+ Returns
331
+ -------
332
+ selected rows and columns of stacked jacobi as dataframe .
333
+
334
+ '''
335
+
336
+ idx = pd.IndexSlice
337
+
338
+ stacked_df_all = self.get_diff_df_tot()
339
+ if type(time)== type(None) :
340
+ perslice = slice(None) # [stacked_df_all.index[0][0],stacked_df_all.index[0][-1]]
341
+ else:
342
+ perslice = time
343
+
344
+ if lhs:
345
+ lhsslice = lhs.upper().split()
346
+ else:
347
+ lhsslice =slice(None) # [stacked_df_all.index[0][1],stacked_df_all.index[0][-1]]
348
+
349
+ if rhs:
350
+ rhsslice = rhs.upper().split()
351
+ else:
352
+ rhsslice =slice(None) # [stacked_df_all.index[0][1],stacked_df_all.index[0][-1]]
353
+
354
+
355
+ # breakpoint()
356
+ # slices = idx[perslice,varslice]
357
+ stacked_df = stacked_df_all.loc[(perslice,lhsslice),
358
+ (perslice,rhsslice)]
359
+
360
+ if show:
361
+ sdec = str(dec)
362
+ display( HTML(stacked_df.map(lambda x:f'{x:,.{sdec}f}' if x != 0.0 else ' ').to_html()))
363
+ return stacked_df
364
+
365
+ def show_diff_latex(self,pat='*',show_expression=True,show_values=True,maxper=5):
366
+ varpat = r'(?P<var>[a-zA-Z_]\w*)\((?P<lag>[+-][0-9]+)\)'
367
+ # varlatex = '\g<var>_{t\g<lag>}'
368
+
369
+
370
+ def partial_to_latex(v,k):
371
+ udtryk=r'\frac{\partial '+ mj.an_expression_to_latex(v)+r'}{\partial '+mj.an_expression_to_latex(k)+'}'
372
+ return udtryk
373
+
374
+ if show_values: _ = self.get_diff_values_all()
375
+
376
+
377
+ for v in [var for p in pat.split() for var in fnmatch.filter(self.declared_endo_set,p)]:
378
+ thisvar = v if v in self.mmodel.endogene else v+'___RES'
379
+
380
+ _ = f'{mj.frml_as_latex(self.mmodel.allvar[thisvar]["frml"],self.mmodel.funks,name=False)}'
381
+ # display(Latex(r'$'+frmlud+r'$'))
382
+
383
+
384
+ if show_expression:
385
+ totud = [ f'{partial_to_latex(thisvar,i)} & = & {mj.an_expression_to_latex(expression)}'
386
+ for i,expression in self.diffendocur[thisvar].items()]
387
+ ud=r'\\'.join(totud)
388
+ display(Latex(r'\begin{eqnarray*}'+ud+r'\end{eqnarray*} '))
389
+ #display(Latex(f'{ud}'))
390
+
391
+
392
+ if show_values:
393
+ # breakpoint()
394
+ if len(self.diffvalues[thisvar].values()):
395
+ resdf = pd.concat([row for row in self.diffvalues[thisvar].values()]).iloc[:,:maxper]
396
+ resdf.index = ['$'+partial_to_latex(thisvar,k)+'$' for k in self.diffvalues[thisvar].keys()]
397
+ markout = resdf.iloc[:,:].to_markdown()
398
+ display(Markdown(markout))
399
+ # print( (r'\begin{eqnarray}'+ud+r'\end{eqnarray} '))
400
+
401
+ def get_diffmodel(self):
402
+ ''' Returns a model which calculates the partial derivatives of a model'''
403
+
404
+ def makelag(var):
405
+ vterm = udtryk_parse(var)[0]
406
+ if vterm.lag:
407
+ if vterm.lag[0] == '-':
408
+ return f'{vterm.var}___lag___{vterm.lag[1:]}'
409
+ elif vterm.lag[0] == '+':
410
+ return f'{vterm.var}___lead___{vterm.lag[1:]}'
411
+ else:
412
+ return f'{vterm.var}___per___{vterm.lag}'
413
+ else:
414
+ return f'{vterm.var}___lag___0'
415
+
416
+ with ttimer('Generates a model which calculatews the derivatives for a model',self.timeit):
417
+ out = '\n'.join([f'{lhsvar}__P__{makelag(rhsvar)} = {self.diffendocur[lhsvar][rhsvar]} '
418
+ for lhsvar in sorted(self.diffendocur)
419
+ for rhsvar in sorted(self.diffendocur[lhsvar])
420
+ ] )
421
+ dmodel = self.newmodel(out,funks=self.mmodel.funks,straight=True,
422
+ modelname=self.mmodel.name +' Derivatives '+ (' no lags and leads' if self.onlyendocur else ' all lags and leads'))
423
+ return dmodel
424
+
425
+
426
+
427
+ def get_diff_melted(self,periode=None,df=None):
428
+ '''returns a tall matrix with all values to construct jacobimatrix(es) '''
429
+
430
+ def get_lagnr(l):
431
+ ''' extract lag/lead from variable name and returns a signed lag (leads are positive'''
432
+ # breakpoint()
433
+ return int('-'*(l.split('___')[0]=='AG') + l.split('___')[1])
434
+
435
+
436
+ def get_elm(vartuples,i):
437
+ ''' returns a list of lags list of tupels '''
438
+ return [v[i] for v in vartuples]
439
+
440
+ _per_first = periode if type(periode) != type(None) else self.mmodel.current_per
441
+
442
+ if hasattr(_per_first,'__iter__'):
443
+ _per = _per_first
444
+ else:
445
+ _per = [_per_first]
446
+
447
+ _df = self.df if type(df) != pd.DataFrame else df
448
+ self.df = _df
449
+ _df = _df.pipe(lambda df0: df0.rename(columns={c: c.upper() for c in df0.columns}))
450
+ # add the diff-model output columns BEFORE the call: otherwise every
451
+ # call looks like new data to is_newdata and the evaluator is
452
+ # re-exec'ed / recompiled on every Jacobian refresh
453
+ _df = insertModelVar(_df, self.diff_model)
454
+
455
+ self.diff_model.current_per = _per
456
+ # breakpoint()
457
+ with ttimer('calculate derivatives',self.timeit):
458
+ reseval = self.diff_model.res_ng if self.ng else self.diff_model.res
459
+ self.difres = reseval(_df,silent=self.silent,stats=0,ljit=self.ljit,
460
+ chunk=self.nchunk).loc[_per,sorted(self.diff_model.endogene)].fillna(0.0)
461
+ with ttimer('Prepare wide input to sparse matrix',self.timeit):
462
+ # breakpoint()
463
+ cname = namedtuple('cname','var,pvar,lag')
464
+ self.coltup = [cname(i.split('__P__',1)[0],
465
+ i.split('__P__',1)[1].split('___L',1)[0],
466
+ get_lagnr(i.split('__P__',1)[1].split('___L',1)[1]))
467
+ for i in self.difres.columns]
468
+ # breakpoint()
469
+ self.coltupnum = [(self.placdic[var],self.placdic[pvar+'___RES' if (pvar+'___RES' in self.mmodel.endogene) else pvar],lag)
470
+ for var,pvar,lag in self.coltup]
471
+
472
+ self.difres.columns = self.coltupnum
473
+ self.numbers = [i for i,n in enumerate(self.difres.index)]
474
+ self.maxnumber = max(self.numbers)
475
+ self.numbers_to_date = {i:n for i,n in enumerate(self.difres.index)}
476
+ self.nvar = len(self.endovar)
477
+ self.difres.loc[:,'number'] = self.numbers
478
+
479
+ with ttimer('melt the wide input to sparse matrix',self.timeit):
480
+ dmelt = self.difres.melt(id_vars='number')
481
+ dmelt.loc[:,'value']=dmelt['value'].astype('float')
482
+
483
+
484
+ with ttimer('assign tall input to sparse matrix',self.timeit):
485
+ # breakpoint()
486
+ dmelt = dmelt.assign(var = lambda x: get_elm(x.variable,0),
487
+ pvar = lambda x: get_elm(x.variable,1),
488
+ lag = lambda x: get_elm(x.variable,2))
489
+ return dmelt
490
+
491
+
492
+
493
+ def get_diff_mat_tot(self, df=None):
494
+ """
495
+ Generate a stacked Jacobian matrix for the entire model across the current period.
496
+
497
+ This function constructs a sparse matrix representing the Jacobian for the specified
498
+ period range, either normalized or unnormalized, depending on the model's configuration.
499
+
500
+ Args:
501
+ df (pandas.DataFrame, optional): Input DataFrame for the calculations. If not provided,
502
+ the model's internal DataFrame (`self.df`) is used.
503
+
504
+ Returns:
505
+ scipy.sparse.csc_matrix: A sparse matrix of the stacked Jacobian for all periods in
506
+ the model's `current_per`.
507
+
508
+ Notes:
509
+ - The function uses melted data to filter and build the row and column indices for
510
+ constructing the sparse matrix.
511
+ - If the model is normalized, an identity matrix is subtracted from the raw Jacobian.
512
+ """
513
+ dmelt = self.get_diff_melted(periode=None,df=df)
514
+ dmelt = dmelt.eval('''\
515
+ keep = (@self.maxnumber >= lag+number) & (lag+number >=0)
516
+ row = number * @self.nvar + var
517
+ col = (number+lag) *@self.nvar +pvar ''')
518
+ dmelt = dmelt.query('keep')
519
+
520
+ #csc_matrix((data, (row_ind, col_ind)), [shape=(M, N)])
521
+ size = self.nvar*len(self.numbers)
522
+ values = dmelt.value.values
523
+ indicies = (dmelt.row,dmelt.col)
524
+
525
+ raw = self.stacked = sp.sparse.csc_matrix((values,indicies ),shape=(size, size))
526
+
527
+ if self.mmodel.normalized:
528
+ this = raw - sp.sparse.identity(size,format='csc')
529
+ else:
530
+ this = raw
531
+ return this
532
+
533
+ def get_diff_df_tot(self,periode=None,df=None):
534
+ """
535
+ Generate a DataFrame representation of the stacked Jacobian matrix for the entire model across the specified period.
536
+
537
+ This function constructs a dense matrix representing the Jacobian for the specified
538
+ period range, with variables and periods as multi-indexed rows and columns.
539
+
540
+ Args:
541
+ periode (iterable, optional): The period range for which the Jacobian is calculated.
542
+ Defaults to the model's `current_per` if not provided.
543
+ df (pandas.DataFrame, optional): Input DataFrame for the calculations. If not provided,
544
+ the model's internal DataFrame (`self.df`) is used.
545
+
546
+ Returns:
547
+ pandas.DataFrame: A DataFrame of the stacked Jacobian, with a multi-index of
548
+ (period, variable) for both rows and columns.
549
+
550
+ Notes:
551
+ - The function first calculates the sparse stacked Jacobian matrix using
552
+ `get_diff_mat_tot` and then converts it to a dense format.
553
+ - The resulting DataFrame includes all endogenous variables across all periods.
554
+ - The structure is useful for visualization or further manipulations where dense matrices
555
+ are required.
556
+
557
+ """
558
+ stacked_mat = self.get_diff_mat_tot(df=df).toarray()
559
+ colindex = pd.MultiIndex.from_product([self.mmodel.current_per,self.declared_endo_list],names=['per','var'])
560
+ rowindex = pd.MultiIndex.from_product([self.mmodel.current_per,self.declared_endo_list],names=['per','var'])
561
+ out = pd.DataFrame(stacked_mat,index=rowindex,columns=colindex)
562
+ return out
563
+
564
+
565
+ def get_diff_mat_1per(self,periode=None,df=None):
566
+ """
567
+ Generate a dictionary of sparse Jacobian matrices for a single period.
568
+
569
+ This function computes the Jacobian matrix for the specified period, returning
570
+ a dictionary where each key represents a period, and the corresponding value
571
+ is a sparse matrix representing the Jacobian for that period.
572
+
573
+ Args:
574
+ periode (iterable, optional): The period(s) for which the Jacobian is calculated.
575
+ Defaults to the model's `current_per` if not provided.
576
+ df (pandas.DataFrame, optional): Input DataFrame for the calculations. If not provided,
577
+ the model's internal DataFrame (`self.df`) is used.
578
+
579
+ Returns:
580
+ dict: A dictionary where keys are periods (as timestamps) and values are sparse
581
+ Jacobian matrices (`scipy.sparse.csc_matrix`) for each period.
582
+
583
+ Notes:
584
+ - The Jacobian is calculated for endogenous variables only, with rows and columns
585
+ corresponding to these variables.
586
+ - If the model is normalized, the identity matrix is subtracted from the raw Jacobian.
587
+ - The resulting sparse matrices are memory-efficient and suitable for solving
588
+ systems of equations.
589
+ """
590
+ # breakpoint()
591
+ dmelt = self.get_diff_melted(periode=periode,df=df)
592
+
593
+ dmelt = dmelt.eval('''\
594
+ keep = lag == 0
595
+ row = var
596
+ col = pvar ''')
597
+ outdic = {}
598
+ dmelt = dmelt.query('keep')
599
+ grouped = dmelt.groupby(by='number')
600
+ for per,df in grouped:
601
+ values = df.value.values
602
+ indicies = (df.row.values,df.col.values)
603
+ raw = sp.sparse.csc_matrix((values,indicies ), shape=(self.nvar, self.nvar))
604
+ # breakpoint()
605
+ #csc_matrix((data, (row_ind, col_ind)), [shape=(M, N)])
606
+
607
+ if self.mmodel.normalized:
608
+ this = raw -sp.sparse.identity(self.nvar,format='csc')
609
+ else:
610
+ this = raw
611
+ outdic[self.numbers_to_date[per]] = this
612
+
613
+ return outdic
614
+
615
+
616
+
617
+ def get_diff_df_1per(self,df=None,periode=None):
618
+ self.jacsparsedic = self.get_diff_mat_1per(df=df,periode=periode)
619
+ self._jacorgsparsedic = self.jacsparsedic
620
+ return self.jacdfdic
621
+
622
+ @property
623
+ def jacdfdic(self):
624
+ '''Dense DataFrame view of the per-period Jacobians (``jacsparsedic``),
625
+ materialized on demand. Inspection aid -- rebuilt on every access, so
626
+ bind it to a name before working with it repeatedly.'''
627
+ return {p: pd.DataFrame(jac.toarray(),columns=self.endovar,index=self.endovar)
628
+ for p,jac in self.jacsparsedic.items()}
629
+
630
+ @property
631
+ def jacorgdfdic(self):
632
+ '''Dense DataFrame view of the raw per-period Jacobians (before the
633
+ residual-row mask is applied), materialized on demand.'''
634
+ return {p: pd.DataFrame(jac.toarray(),columns=self.endovar,index=self.endovar)
635
+ for p,jac in self._jacorgsparsedic.items()}
636
+
637
+
638
+
639
+
640
+ def get_solve1perlu(self,df='',periode=''):
641
+ # if update or not hasattr(self,'stacked'):
642
+ self.jacsparsedic = self.get_diff_mat_1per(df=df,periode=periode)
643
+ self.ludic = {p : sp.linalg.lu_factor(jac.toarray()) for p,jac in self.jacsparsedic.items()}
644
+ self.solveludic = {p: lambda distance : sp.linalg.lu_solve(lu,distance) for p,lu in self.ludic.items()}
645
+ return self.solveludic
646
+
647
+ # def get_solve1per(self,df=None,periode=None):
648
+ # # if update or not hasattr(self,'stacked'):
649
+ # # breakpoint()
650
+ # self.jacsparsedic = self.get_diff_mat_1per(df=df,periode=periode)
651
+ # self.solvelusparsedic = {p: sp.sparse.linalg.factorized(jac) for p,jac in self.jacsparsedic.items()}
652
+ # return self.solvelusparsedic
653
+
654
+
655
+ def get_solve1per(self,df=None,periode=None,is_residual_eq=None):
656
+ # if update or not hasattr(self,'stacked'):
657
+ # breakpoint()
658
+ if is_residual_eq is not None and not self.mmodel.normalized :
659
+ diag_mask = sp.sparse.diags((~is_residual_eq).astype(float))
660
+ temp = self.get_diff_mat_1per(df=df,periode=periode)
661
+ self.jacsparsedic = { p: jac - diag_mask for p,jac in temp.items() }
662
+ else:
663
+ temp = self.get_diff_mat_1per(df=df,periode=periode)
664
+ self.jacsparsedic = temp
665
+
666
+ # dense DataFrame views (jacorgdfdic / jacdfdic) are lazy properties --
667
+ # materializing n x n frames for every period on each nonlin refresh
668
+ # dominated the refresh cost for large models
669
+ self._jacorgsparsedic = temp
670
+
671
+ # every period's Jacobian shares one sparsity pattern, so with
672
+ # factorizer='umfpack' the symbolic analysis is done once and reused
673
+ # across all periods (and later nonlin refreshes)
674
+ if getattr(self,'factorizer',None) == 'umfpack':
675
+ with ttimer('umfpack factorize per-period jacobians',self.timeit):
676
+ self.solvelusparsedic = {p: self._factorize_umfpack(jac, timeit=False)
677
+ for p,jac in self.jacsparsedic.items()}
678
+ return self.solvelusparsedic
679
+ # the per-period Jacobian pattern is the model's dependency graph, not
680
+ # a band, so scipy's default ordering (COLAMD) is kept unless a
681
+ # permc_spec is set explicitly
682
+ if getattr(self,'permc_spec',None):
683
+ self.solvelusparsedic = {p: sp.sparse.linalg.splu(jac,permc_spec=self.permc_spec).solve
684
+ for p,jac in self.jacsparsedic.items()}
685
+ else:
686
+ self.solvelusparsedic = {p: sp.sparse.linalg.factorized(jac) for p,jac in self.jacsparsedic.items()}
687
+ return self.solvelusparsedic
688
+
689
+
690
+ def get_solvestacked(self,df='',is_residual_eq=None):
691
+ # if update or not hasattr(self,'stacked'):
692
+
693
+ # breakpoint()
694
+ if is_residual_eq is not None and not self.mmodel.normalized :
695
+ diag_mask = sp.sparse.diags((~is_residual_eq).astype(float))
696
+ self.stacked = self.get_diff_mat_tot(df=df) - diag_mask
697
+ else:
698
+ self.stacked = self.get_diff_mat_tot(df=df)
699
+ if getattr(self,'factorizer',None) == 'umfpack':
700
+ self.solvestacked = self._factorize_umfpack(self.stacked)
701
+ else:
702
+ with ttimer('factorize stacked jacobian',self.timeit):
703
+ # 'NATURAL' preserves the block-band structure of the stacked
704
+ # matrix (period blocks coupled over a narrow lag/lead range);
705
+ # fill-reducing orderings like COLAMD scramble the band and can
706
+ # explode fill (13s -> 1.2s on FRB/US MCE). Any SuperLU
707
+ # ordering can be selected via the permc_spec attribute.
708
+ permc_spec = getattr(self,'permc_spec',None) or 'NATURAL'
709
+ self.solvestacked = sp.sparse.linalg.splu(self.stacked,permc_spec=permc_spec).solve
710
+ return self.solvestacked
711
+
712
+ def _factorize_umfpack(self, jac, timeit=None):
713
+ """LU factorization through cvxopt's UMFPACK, reusing the symbolic analysis.
714
+
715
+ The Jacobian keeps the same sparsity pattern between factorizations --
716
+ across rebuilds of the stacked matrix, and across periods and nonlin
717
+ refreshes of the per-period matrices -- only the values change, so the
718
+ fill-reducing symbolic analysis is done once and cached on the
719
+ instance; later factorizations redo only the numeric phase. The
720
+ cached analysis is invalidated if the pattern does change (different
721
+ period range or model). Returns a solve callable like ``factorized``.
722
+ """
723
+ from cvxopt import matrix, spmatrix
724
+ from cvxopt import umfpack
725
+
726
+ timeit = self.timeit if timeit is None else timeit
727
+ coo = jac.tocoo()
728
+ with ttimer('umfpack build spmatrix',timeit):
729
+ A = spmatrix(matrix(coo.data), coo.row.tolist(), coo.col.tolist(), coo.shape)
730
+
731
+ same_pattern = (getattr(self,'_umf_pattern',None) is not None
732
+ and np.array_equal(self._umf_pattern[0], jac.indptr)
733
+ and np.array_equal(self._umf_pattern[1], jac.indices))
734
+ if not same_pattern:
735
+ with ttimer('umfpack symbolic analysis',timeit):
736
+ self._umf_symbolic = umfpack.symbolic(A)
737
+ self._umf_pattern = (jac.indptr.copy(), jac.indices.copy())
738
+ with ttimer('umfpack numeric factorization',timeit):
739
+ numeric = umfpack.numeric(A, self._umf_symbolic)
740
+
741
+ def solvejac(b):
742
+ x = matrix(np.asarray(b, dtype='float64'))
743
+ umfpack.solve(A, numeric, x)
744
+ return np.asarray(x).ravel()
745
+
746
+ return solvejac
747
+
748
+ def get_solvestacked_it(self,df='',solver = sp.sparse.linalg.bicg):
749
+ # if update or not hasattr(self,'stacked'):
750
+ self.stacked = self.get_diff_mat_tot(df=df)
751
+
752
+ def solvestacked_it(b):
753
+ return solver(self.stacked,b)[0]
754
+
755
+ return solvestacked_it
756
+
757
+ def get_diff_melted_var(self,periode=None,df=None):
758
+ '''makes dict with all derivative matrices for all lags '''
759
+
760
+ def get_lagnr(l):
761
+ ''' extract lag/lead from variable name and returns a signed lag (leads are positive'''
762
+ #breakpoint()
763
+ return int('-'*(l.split('___')[0]=='AG') + l.split('___')[1])
764
+
765
+
766
+ def get_elm(vartuples,i):
767
+ ''' returns a list of lags list of tupels '''
768
+ return [v[i] for v in vartuples]
769
+
770
+ _per_first = periode if type(periode) != type(None) else self.mmodel.current_per
771
+
772
+ if hasattr(_per_first,'__iter__'):
773
+ _per = _per_first
774
+ else:
775
+ _per = [_per_first]
776
+
777
+ _df = self.df if type(df) != pd.DataFrame else df
778
+ _df = _df.pipe(lambda df0: df0.rename(columns={c: c.upper() for c in df0.columns}))
779
+ # see get_diff_melted: expand before the call so is_newdata
780
+ # recognizes the databank and the evaluator is reused
781
+ _df = insertModelVar(_df, self.diff_model)
782
+
783
+ self.diff_model.current_per = _per
784
+ # breakpoint()
785
+ reseval = self.diff_model.res_ng if self.ng else self.diff_model.res
786
+ difres = reseval(_df,silent=self.silent,stats=0,ljit=self.ljit,chunk=self.nchunk
787
+ ).loc[_per,sorted(self.diff_model.endogene)].fillna(0.0).astype('float')
788
+
789
+ cname = namedtuple('cname','var,pvar,lag')
790
+ col_vars = [cname(i.rsplit('__P__',1)[0],
791
+ i.rsplit('__P__',1)[1].split('___L',1)[0],
792
+ get_lagnr(i.rsplit('__P__',1)[1].split('___L',1)[1]))
793
+ for i in difres.columns]
794
+
795
+ col_ident = [diff_value_col(**i._asdict(), var_plac=self.placdic[i.var],
796
+ pvar_plac=self.placdic.get(i.pvar+'___RES' if (i.pvar+'___RES' in self.mmodel.endogene) else i.pvar, 0),
797
+ pvar_endo = i.pvar in self.mmodel.endogene or i.pvar+'___RES' in self.mmodel.endogene,
798
+ pvar_exo_plac = self.exoplacdic.get(i.pvar, 0) ) for i in col_vars]
799
+
800
+ difres.columns = col_ident
801
+ difres = difres.copy()
802
+ difres.loc[:,'dates'] = difres.index
803
+ dmelt = difres.melt(id_vars='dates')
804
+ unfolded = pd.DataFrame( [asdict(i) for i in dmelt.variable.values])
805
+ totalmelt = pd.concat([dmelt[['dates','value']],unfolded],axis=1)
806
+
807
+ # breakpoint()
808
+
809
+
810
+ return totalmelt
811
+
812
+ def get_diff_mat_all_1per(self,periode=None,df=None,asdf=False):
813
+ dmelt = self.get_diff_melted_var(periode=periode,df=df)
814
+ with ttimer('Prepare numpy input to sparse matrix',self.timeit):
815
+ outdic = defaultdict(lambda: defaultdict(dict))
816
+ grouped = dmelt.groupby(by=['pvar_endo','dates','lag'])
817
+ for (endo,date,lag),df in grouped:
818
+ values = df.value.values
819
+ # breakpoint()
820
+ # #csc_matrix((data, (row_ind, col_ind)), [shape=(M, N)])
821
+ # print(f'endo:{endo} ,date:{date}, lag:{lag}, \n df')
822
+ if endo:
823
+ indicies = (df.var_plac.values,df.pvar_plac.values)
824
+ this = sp.sparse.csc_matrix((values,indicies ),
825
+ shape=(len(self.declared_endo_list), len(self.declared_endo_list)))
826
+
827
+ if asdf:
828
+ outdic[date]['endo'][f'lag={lag}'] = pd.DataFrame(this.toarray(),
829
+ columns=self.declared_endo_list,index=self.declared_endo_list)
830
+ else:
831
+ outdic[date]['endo'][f'lag={lag}'] = this
832
+ else:
833
+ indicies = (df.var_plac.values,df.pvar_exo_plac.values)
834
+ this = sp.sparse.csc_matrix((values,indicies ),
835
+ shape=(len(self.endovar), len(self.exovar)))
836
+ if asdf:
837
+ outdic[date]['exo'][f'lag={lag}']= pd.DataFrame(this.toarray(), columns=self.exovar,index=self.declared_endo_list)
838
+ else:
839
+ outdic[date]['exo'][f'lag={lag}'] = this
840
+
841
+ return outdic
842
+
843
+ def get_mixed_structure(self):
844
+ """
845
+ Returns tuple:
846
+ - is_residual_row: bool array (len = n_rows) where rows ending with '___RES' are True
847
+ - row_to_col_idx: int array mapping each row’s base variable to declared_endo_list index
848
+ - col_indexer: dict name->col index for declared_endo_list
849
+ """
850
+ row_names = np.array(self.endovar, dtype=str)
851
+ is_residual_row = np.char.endswith(row_names, '___RES')
852
+
853
+ def base_name(r):
854
+ return r[:-6] if r.endswith('___RES') else r
855
+
856
+ declared = list(self.declared_endo_list)
857
+ col_pos = {v: i for i, v in enumerate(declared)}
858
+
859
+ row_to_col_idx = np.array([col_pos.get(base_name(r), -1) for r in row_names], dtype=int)
860
+ if (row_to_col_idx < 0).any():
861
+ bad = [row_names[i] for i in np.where(row_to_col_idx < 0)[0]]
862
+ raise Exception(f'Row(s) not mappable to declared endo: {bad}')
863
+
864
+ return is_residual_row, row_to_col_idx, col_pos
865
+
866
+
867
+ def get_diff_mat_1per_mixed(self, periode=None, df=None):
868
+ """
869
+ Build J for one period:
870
+ rows = all equations (normalized + ___RES)
871
+ cols = declared_endo_list (true unknowns)
872
+ Subtract -I only on normalized rows.
873
+ """
874
+ dmelt = self.get_diff_melted(periode=periode, df=df).copy()
875
+ dmelt = dmelt.eval('keep = (lag == 0)').query('keep')
876
+
877
+ n_rows = len(self.endovar)
878
+ n_cols = len(self.declared_endo_list)
879
+ col_pos = {v: i for i, v in enumerate(self.declared_endo_list)}
880
+
881
+ def base_name(v):
882
+ return v[:-6] if v.endswith('___RES') else v
883
+
884
+ dmelt = dmelt.assign(
885
+ row=lambda x: x['var'],
886
+ col=lambda x: x['pvar'].apply(lambda p: col_pos.get(base_name(p), -1))
887
+ )
888
+ if (dmelt['col'] < 0).any():
889
+ bad = dmelt.loc[dmelt['col'] < 0, 'pvar'].unique().tolist()
890
+ raise Exception(f'Unmapped derivative columns in mixed J: {bad}')
891
+
892
+ values = dmelt['value'].astype(float).values
893
+ rows = dmelt['row'].values
894
+ cols = dmelt['col'].values
895
+ raw = sp.sparse.csc_matrix((values, (rows, cols)), shape=(n_rows, n_cols))
896
+
897
+ is_residual_row, row_to_col_idx, _ = self.get_mixed_structure()
898
+ diag_rows = np.where(~is_residual_row)[0]
899
+ diag_cols = row_to_col_idx[~is_residual_row]
900
+ correction = sp.sparse.coo_matrix(
901
+ (np.ones_like(diag_rows, dtype=float), (diag_rows, diag_cols)),
902
+ shape=(n_rows, n_cols)
903
+ ).tocsc()
904
+
905
+ mixed = raw - correction
906
+ per = self.mmodel.current_per if periode is None else periode
907
+ if hasattr(per, '__iter__') and len(per):
908
+ key = per[0]
909
+ else:
910
+ key = per
911
+ return {key: mixed}
912
+
913
+
914
+ def get_solve1per_mixed(self, df=None, periode=None):
915
+ """Factorized solver for mixed Jacobian (rows=all eqs, cols=declared endos)."""
916
+ self.jacsparsedic_mixed = self.get_diff_mat_1per_mixed(df=df, periode=periode)
917
+ self.solvelusparsedic_mixed = {
918
+ p: sp.sparse.linalg.factorized(jac) for p, jac in self.jacsparsedic_mixed.items()
919
+ }
920
+ return self.solvelusparsedic_mixed
921
+
922
+
923
+
924
+ def get_diff_values_all(self,periode=None,df=None,asdf=False):
925
+ ''' stuff the values of derivatives into nested dic '''
926
+ dmelt = self.get_diff_melted_var(periode=periode,df=df)
927
+ with ttimer('Prepare numpy input to sparse matrix',self.timeit):
928
+ self.diffvalues = defaultdict(lambda: defaultdict(dict))
929
+ grouped = dmelt.groupby(by=['var','pvar','lag'])
930
+ for (var,pvar,lag),df in grouped:
931
+ res = df.pivot(index='pvar',columns='dates',values='value')
932
+ pvar_name = tovarlag(pvar,int(lag))
933
+ #reakpoint()
934
+ # #csc_matrix((data, (row_ind, col_ind)), [shape=(M, N)])
935
+ # print(f'endo:{endo} ,date:{date}, lag:{lag}, \n df')
936
+ self.diffvalues[var][pvar_name]=res
937
+ return self.diffvalues
938
+
939
+
940
+
941
+ def get_eigenvalues (self, periode=None, asdf=True, filnan=False, silent=False, dropvar=None, dropvar_nr=0,progressbar=False):
942
+ """
943
+ Calculate and return the eigenvectors based on the companion matrix for a dynamic system.
944
+
945
+ This method computes the eigenvectors of a dynamic system represented by a companion matrix derived from Jacobian matrices for different periods. The computation involves handling missing values, optionally filling NaN values with zero, and the ability to drop specific variables from the calculation.
946
+
947
+ Parameters:
948
+ - periode (optional): The period for which the eigenvectors are to be calculated. If None, defaults to the entire range.
949
+ - asdf (bool, optional): Determines the format of the matrices (DataFrame or sparse matrix). Defaults to True (DataFrame).
950
+ - filnan (bool, optional): If True, fills NaN values in the Jacobian matrices with zero. Defaults to False.
951
+ - silent (bool, optional): If False, prints detailed information about NaN values and other relevant details during the computation. Defaults to False.
952
+ - dropvar (optional): Specifies variables to be dropped from the calculation. If None, no variables are dropped. Defaults to None.
953
+ - dropvar_nr (int, optional): The number of variables to drop. Defaults to 0.
954
+
955
+ Returns:
956
+ dict: A dictionary with keys as dates and values as eigenvectors for each period, derived from the companion matrix of the system.
957
+
958
+ The function performs several steps:
959
+ - Computes the Jacobian matrices for the given period.
960
+ - Handles NaN values based on the 'filnan' parameter.
961
+ - Optionally drops specified variables from the calculation.
962
+ - Constructs the companion matrix from the modified Jacobian matrices.
963
+ - Calculates the eigenvectors from the companion matrix.
964
+
965
+ Note:
966
+ The companion matrix is crucial in analyzing the stability and dynamics of the system. The eigenvectors provide insights into the system's behavior over time.
967
+ """
968
+ ...
969
+
970
+ first_element = lambda dic: dic[list(dic.keys())[0]] # first element in a dict
971
+ # print('Calculate derivatives')
972
+ jacobiall = self.get_diff_mat_all_1per(periode,asdf=asdf)
973
+ # breakpoint()
974
+ if not silent:
975
+ for date,content in jacobiall.items():
976
+ for lag,df in content['endo'].items():
977
+ if not (this := df.loc[df.isna().any(axis=1),df.isna().any(axis=0)]).empty:
978
+ if filnan:
979
+ print(f'{date} {lag} These elements in the jacobi is set to zero',this,'\n')
980
+ else:
981
+ print(f'{date} {lag} These elements in the jacobi contains NaN, You can use filnan = True to set values equal to 0',this,'\n')
982
+
983
+ pat = ' '.join(this.index)
984
+ self.show_diff_latex(pat)
985
+
986
+
987
+
988
+ A_dic_gross0 = {date : {lag : df.fillna(0) if filnan else df for lag,df in content['endo'].items()}
989
+ for date,content in jacobiall.items()}
990
+
991
+ # vnow to get lag=-1 leftmost and so om
992
+ A_dic_gross = {date: {lag : content[lag]
993
+ for lag in sorted(content.keys())}
994
+
995
+ for date,content in A_dic_gross0.items()}
996
+
997
+
998
+ if type(dropvar)== type(None):
999
+ A_dic = {date: {lag: df for lag,df in a_year.items() } for date,a_year in A_dic_gross.items() }
1000
+
1001
+ else:
1002
+ A_dic = {date: {lag: df.drop(index=dropvar,columns=dropvar) for lag,df in a_year.items() } for date,a_year in A_dic_gross.items() }
1003
+ # print(f'{dropvar_nr} {dropvar}')
1004
+ # breakpoint()
1005
+ xlags = sorted([lag for lag in first_element(A_dic).keys() if lag !='lag=0'],key=lambda lag:int(lag.split('=')[1]),reverse=True)
1006
+ number=len(xlags)
1007
+ # dim = len(self.endovar)
1008
+ first_A_dict = first_element(A_dic) # The first dict of A's
1009
+ first_A = first_A_dict['lag=0']
1010
+ self.varnames = first_A.columns
1011
+ dim = len(first_A)
1012
+
1013
+ if asdf:
1014
+ np_to_df = lambda nparray: pd.DataFrame(nparray,
1015
+ index = self.varnames,columns=self.varnames)
1016
+ lib = np
1017
+ values = lambda df: df.values
1018
+ calc_eig = lib.linalg.eig
1019
+ calc_eig_reserve = lambda sparse_matrix : sp.linalg.eig(sparse_matrix.toarray())
1020
+
1021
+ else:
1022
+ np_to_df = lambda sparse_matrix : sparse_matrix
1023
+ lib = sp.sparse
1024
+ values = lambda sparse_matrix : sparse_matrix
1025
+ calc_eig = lambda sparse_matrix : lib.linalg.eigs(sparse_matrix)
1026
+ calc_eig_reserve = lambda sparse_matrix : sp.linalg.eig(sparse_matrix.toarray())
1027
+
1028
+ I=lib.eye(dim)*1.0000
1029
+
1030
+ zeros = lib.zeros((dim,dim))
1031
+ self.A_dic = A_dic
1032
+
1033
+ # a idendity matrix
1034
+ AINV_dic = {date: np_to_df(lib.linalg.inv(I-A['lag=0']))
1035
+ for date,A in (tqdm(A_dic.items(),'Invert (I-A)') if progressbar else A_dic.items()) }
1036
+
1037
+ C_dic = {date: {lag : AINV_dic[date] @ A[lag] for lag,Alag in A.items() if lag!='lag=0'}
1038
+ for date,A in A_dic.items()} # calculate A**-1*A(lag)
1039
+
1040
+ # top=lib.eye((number-1)*dim,(number)*dim,dim)
1041
+ newbottom = lib.eye((number-1)*dim,(number)*dim)
1042
+ # top=lib.eye((number)*dim,(number+1)*dim,dim)
1043
+ # breakpoint()
1044
+ top_dic = {date: lib.hstack([values(thisC) for thisC in C.values()]) for
1045
+ date,C in C_dic.items()}
1046
+
1047
+ comp_dic = {}
1048
+ for date,top in top_dic.items():
1049
+ comp_dic[date] = lib.vstack([top,newbottom])
1050
+
1051
+ try:
1052
+ eigen_values_and_vectors = {date : calc_eig(comp) for date,comp in ( tqdm(comp_dic.items(),'Calculate Eigenvalues') if progressbar else comp_dic.items()) }
1053
+ except:
1054
+ print(f'Using reserve calculatioon of eigenvalues {dropvar_nr=} {dropvar=}')
1055
+
1056
+ eigen_values_and_vectors = {date : calc_eig_reserve(comp) for date,comp in tqdm(comp_dic.items())}
1057
+
1058
+
1059
+ eig_dic = {date : both[0] for date,both in eigen_values_and_vectors.items() }
1060
+
1061
+ for i,date in enumerate(eig_dic.keys() ):
1062
+ ...
1063
+ if 0:
1064
+ if i == 0:
1065
+ print('here')
1066
+ print( sum(abs(eig_dic[date])))
1067
+
1068
+ self.A_dic = A_dic
1069
+ self.comp_dic = comp_dic
1070
+ self.eigen_values_and_vectors = eigen_values_and_vectors
1071
+ self.eig_dic = eig_dic
1072
+ return eig_dic
1073
+
1074
+ @lru_cache(maxsize=None)
1075
+ def get_eigen_jackknife(self,maxnames = 200_000,periode=None,progressbar=True):
1076
+ """
1077
+ Compute and cache eigenvalues for matrices with each variable excluded one at a time, up to a maximum number.
1078
+
1079
+ This function uses the jackknife technique to evaluate the impact of each variable on the system's stability by excluding each variable one by one from the eigenvector calculation. It caches the results for efficient repeated access.
1080
+
1081
+ Parameters:
1082
+ - maxnames (int, optional): The maximum number of variables to consider for exclusion in the jackknife process. Defaults to 20.
1083
+
1084
+ Returns:
1085
+ dict: A dictionary where keys are the names of variables excluded (or 'ALL' for no exclusion) and values are the corresponding eigenvectors.
1086
+
1087
+ Note:
1088
+ The function is computationally intensive and can take significant time for larger systems.
1089
+ """
1090
+ if not hasattr(self, 'eig_dic'):
1091
+ _ = self.get_eigenvalues (filnan = True,silent=False,asdf=1)
1092
+
1093
+ name_to_loop =[n for i,n in enumerate(self.varnames) if i < maxnames and not n.endswith('_FITTED') ]
1094
+ print(f'Calculating eigenvalues of {len(name_to_loop)} different matrices takes time, so make cup of coffee and a take a short nap')
1095
+ jackknife_dict = {f'{name}': self.get_eigenvalues (dropvar=name,periode=periode)
1096
+ for name in (tqdm(name_to_loop) if progressbar else name_to_loop)}
1097
+
1098
+ base_dict = {'NONE' : self.get_eigenvalues (dropvar=None,periode=periode )} # we læeave the properties clean
1099
+
1100
+
1101
+ return {**base_dict, **jackknife_dict}
1102
+
1103
+ @lru_cache(maxsize=None)
1104
+ def get_eigen_jackknife_df(self,maxnames = 200_000,progressbar=True,periode=None,filecache=True,refresh=False):
1105
+ """
1106
+ Convert the eigenvalue data obtained from a jackknife analysis into a pandas DataFrame, including additional columns for the absolute length, real, and imaginary parts of the eigenvalues.
1107
+
1108
+ The jackknife analysis is performed by computing and caching eigenvalues for matrices with each variable excluded one at a time, up to a specified maximum number. This method uses the jackknife technique to evaluate the impact of each variable on the system's stability by excluding each variable one by one from the eigenvector calculation. The results are then flattened and transformed into a DataFrame for further analysis.
1109
+
1110
+ Parameters:
1111
+ - maxnames (int, optional): The maximum number of variables to consider for exclusion in the jackknife process. Defaults to 200,000.
1112
+ - progressbar (bool, optional): If True, displays a progress bar during the computation of eigenvalues. Defaults to True.
1113
+
1114
+ Returns:
1115
+ pandas.DataFrame: A DataFrame containing the eigenvalues with additional columns for length, real, and imaginary parts. Each row represents an eigenvalue for a specific variable exclusion, year, and index.
1116
+
1117
+ Note:
1118
+ - The function is computationally intensive and can take significant time for larger systems.
1119
+ - A progress bar can be displayed for monitoring the computation progress.
1120
+ - The function is especially useful for detailed analysis and visualization of the eigenvalues obtained from the jackknife analysis.
1121
+ """
1122
+ jackfile = Path('jackdf.csv')
1123
+ if (not refresh) and jackfile.exists() and filecache:
1124
+ df = pd.read_csv('jackdf.csv',index_col=0)
1125
+ print('Jackdf read from file')
1126
+ return df
1127
+
1128
+ jackdict = self.get_eigen_jackknife(maxnames = maxnames,progressbar=progressbar,periode=periode)
1129
+
1130
+ flattened_data = [{'excluded': scenario, 'year': year, 'index': i, 'value': value}
1131
+ for scenario, years in jackdict.items()
1132
+ for year, values in years.items()
1133
+ for i, value in enumerate(values)]
1134
+
1135
+ # Creating the DataFrame
1136
+ df = pd.DataFrame.from_dict(flattened_data)
1137
+
1138
+ # Applying transformations
1139
+ df['length'] = df['value'].apply(lambda x: abs(x))
1140
+ df['realvalue'] = df['value'].apply(lambda x: x.real)
1141
+ df['imagvalue'] = df['value'].apply(lambda x: x.imag)
1142
+
1143
+ vardict = {**{'NONE':'whole model'}, **{k : f'{k} {v}' for k,v in self.mmodel.var_description.items() }}
1144
+ df = df.assign(excluded_description=df['excluded'].map(lambda x: vardict.get(x, x)))
1145
+
1146
+ if filecache:
1147
+ df.to_csv('jackdf.csv')
1148
+ print('Jackdf written to file')
1149
+
1150
+
1151
+
1152
+ return df
1153
+
1154
+
1155
+ def get_eigen_jackknife_abs(self,largest=20,maxnames = 200_000):
1156
+ """
1157
+ Compute the absolute values of the largest eigenvalues from the jackknife eigenvalue analysis.
1158
+
1159
+ This function calculates the absolute values of the largest eigenvalues for each set of eigenvalues obtained from the `get_eigen_jackknife` method. It focuses on the largest eigenvalues to understand the most significant influences on the system's stability.
1160
+
1161
+ Parameters:
1162
+ - largest (int, optional): The number of largest eigenvalues to consider. Defaults to 20.
1163
+ - maxnames (int, optional): The maximum number of variables to exclude in the jackknife process. Defaults to 20.
1164
+
1165
+ Returns:
1166
+ dict: A dictionary with the absolute values of the largest eigenvalues for each variable exclusion scenario.
1167
+
1168
+ Note:
1169
+ This method helps in identifying the most impactful variables on the system's stability by focusing on the largest eigenvalues.
1170
+ """
1171
+
1172
+
1173
+ base = self.get_eigen_jackknife(maxnames=maxnames)
1174
+ res = {name: {date: np.sort(np.partition(np.abs(eigenvalues), -largest)[-largest:])[::-1]
1175
+
1176
+ for date,eigenvalues in alldates.items()}
1177
+ for name,alldates in base.items()}
1178
+ return res
1179
+
1180
+
1181
+ @staticmethod
1182
+ def jack_largest_reduction(jackdf, eigenvalue_row=0, periode=None,imag_only=False):
1183
+ """
1184
+ Identifies the largest reduction in eigenvalue magnitude for a specified period
1185
+ and optionally focuses on eigenvalues with imaginary parts if imag_only is True.
1186
+
1187
+ Parameters:
1188
+ - jackdf (DataFrame): A DataFrame containing jackknife analysis results,
1189
+ including eigenvalues, their real and imaginary parts, and descriptions of
1190
+ exclusions.
1191
+ - eigenvalue_row (int, optional): The row index of the eigenvalue to analyze.
1192
+ Defaults to 0, which typically represents the largest magnitude eigenvalue.
1193
+ - periode (int/str, optional): The specific period (year) to analyze. If None,
1194
+ the function processes the first year found in the DataFrame. Defaults to None.
1195
+ - imag_only (bool, optional): If True, only considers eigenvalues with non-zero
1196
+ imaginary parts for analysis. Defaults to False.
1197
+
1198
+ Returns:
1199
+ DataFrame: A sorted DataFrame with the nth largest length (eigenvalue magnitude)
1200
+ for each excluded variable or condition, including the year, exclusion identifier,
1201
+ length (magnitude of the eigenvalue), description of the exclusion, and the real
1202
+ and imaginary parts of the eigenvalue. The row for 'excluded == "NONE"' is moved to the front.
1203
+
1204
+ Raises:
1205
+ Exception: If the specified period is not found in the DataFrame's years.
1206
+ """
1207
+ years = jackdf.year.unique()
1208
+
1209
+ # Determine the year to analyze based on the input
1210
+ if years[0] == periode or type(periode) == type(None):
1211
+ year = years[0]
1212
+ elif periode in years:
1213
+ year = periode
1214
+ else:
1215
+ raise Exception('No such year')
1216
+
1217
+ # Filter DataFrame based on the specified year and condition
1218
+ df = jackdf.query('year == @year & imagvalue != 0.0 ') if imag_only else jackdf.query('year == @year')
1219
+
1220
+ df_sorted = df.sort_values(by=['year', 'excluded', 'length'], ascending=[True, True, False])
1221
+ nth_largest_length = df_sorted.groupby(['year', 'excluded']).nth(eigenvalue_row).reset_index()
1222
+
1223
+ # Further refine and sort the resulting DataFrame
1224
+ nth_largest_length = nth_largest_length[['year', 'excluded', 'length', 'excluded_description', 'realvalue', 'imagvalue']].sort_values('length')
1225
+
1226
+ # Move the row where 'excluded == "NONE"' to the front
1227
+ none_row = nth_largest_length.query('excluded == "NONE" ')
1228
+ others = nth_largest_length.query('excluded != "NONE" ')
1229
+ new_nth_largest_length = pd.concat([none_row, others]).reset_index(drop=True)
1230
+
1231
+ return new_nth_largest_length
1232
+
1233
+
1234
+
1235
+ def jack_largest_reduction_plot(self,jackdf, eigenvalue_row=0, periode=None,imag_only=False):
1236
+ """
1237
+ Creates an interactive Plotly plot to visualize the reduction in eigenvalue magnitude across different exclusions,
1238
+ highlighting the 'NONE' exclusion category with a distinct color and displaying detailed information on hover.
1239
+ Optionally focuses on eigenvalues with imaginary parts if imag_only is True.
1240
+
1241
+ Parameters:
1242
+ - jackdf (DataFrame): A DataFrame containing the results of a jackknife analysis, including eigenvalues and their
1243
+ descriptions. The DataFrame is expected to have at least the columns 'excluded', 'length', 'excluded_description',
1244
+ 'realvalue', and 'imagvalue'.
1245
+ - eigenvalue_row (int, optional): Specifies the row index of the eigenvalue to analyze. Defaults to 0, which typically
1246
+ corresponds to the largest magnitude eigenvalue.
1247
+ - periode (int/str, optional): The specific period (year) to analyze. If None, the function processes data without
1248
+ filtering by period. Defaults to None.
1249
+ - imag_only (bool, optional): If True, only considers eigenvalues with non-zero imaginary parts for analysis.
1250
+ Defaults to False.
1251
+
1252
+ The function processes the input DataFrame to highlight the 'NONE' category in red and all other categories in blue.
1253
+ It then creates a Plotly FigureWidget to plot these data points as markers on a scatter plot. The y-axis tick labels are
1254
+ hidden to emphasize the data points rather than the categorical labels.
1255
+
1256
+ An HTML widget is used to display detailed information about a data point (exclusion description, length, and imaginary
1257
+ part of the eigenvalue) when the user hovers over it. This interactive feature provides a deeper insight into the impact
1258
+ of each exclusion on the eigenvalue magnitude.
1259
+
1260
+ Displays:
1261
+ - An interactive Plotly scatter plot within the Jupyter notebook.
1262
+ - A dynamic HTML widget that updates with detailed information about the hovered data point.
1263
+ """
1264
+ import plotly.graph_objs as go
1265
+ from ipywidgets import VBox, HTML, Textarea, Layout
1266
+
1267
+ result_df = __class__.jack_largest_reduction(jackdf, eigenvalue_row=eigenvalue_row, periode=periode,imag_only=imag_only)
1268
+ #print(result_df)
1269
+
1270
+ # Assign highlight colors based on the 'excluded' category
1271
+ result_df['Highlight'] = result_df['excluded'].apply(lambda x: 'NONE' if x == 'NONE' else 'Other')
1272
+ # Create a combined highlight column based on 'excluded' and 'imagvalue'
1273
+ def get_highlight(row):
1274
+ if row['excluded'] == 'NONE' and row['imagvalue'] != 0.0:
1275
+ return 'NONE (Non-zero Imag)'
1276
+ elif row['excluded'] == 'NONE' and row['imagvalue'] == 0.0:
1277
+ return 'NONE (Zero Imag)'
1278
+ elif row['imagvalue'] != 0.0:
1279
+ return 'Other (Non-zero Imag)'
1280
+ else:
1281
+ return 'Other (Zero Imag)'
1282
+
1283
+ result_df['CombinedHighlight'] = result_df.apply(get_highlight, axis=1)
1284
+ # Add a column for marker size based on the 'CombinedHighlight' or any condition
1285
+ result_df['marker_size'] = result_df['CombinedHighlight'].apply(lambda x: 20 if 'NONE' in x else 10)
1286
+
1287
+
1288
+ # Define color map for the combined highlight
1289
+ color_map = {
1290
+ 'NONE (Non-zero Imag)': 'red',
1291
+ 'NONE (Zero Imag)': 'red',
1292
+ 'Other (Non-zero Imag)': 'green',
1293
+ 'Other (Zero Imag)': 'blue'
1294
+ }
1295
+
1296
+ # Create the Plotly FigureWidget using the combined highlight for color mapping
1297
+ fig = go.FigureWidget(data=[
1298
+ go.Scatter(
1299
+ x=result_df['length'],
1300
+ y=result_df['excluded'],
1301
+ mode='markers',
1302
+ marker=dict(
1303
+ color=result_df['CombinedHighlight'].map(color_map),
1304
+ size=result_df['marker_size'] # Use the marker_size column here
1305
+
1306
+ )
1307
+ )
1308
+ ])
1309
+
1310
+
1311
+ # Create the Plotly FigureWidget with customized marker colors
1312
+ # fig = go.FigureWidget(data=[
1313
+ # go.Scatter(x=result_df['length'], y=result_df['excluded'], mode='markers',
1314
+ # marker=dict(color=result_df['Highlight'].map({'NONE': 'red', 'Other': 'blue'})))
1315
+ # ])
1316
+
1317
+ # Update layout to hide y-axis tick labels for a cleaner presentation
1318
+ fig.update_layout(
1319
+ title={
1320
+ 'text': f"Eigenvalue Lengths by excluded equation. {eigenvalue_row}'th largest eigenvalues, {'Only eigenvalues with imaginary values' if imag_only else''}",
1321
+ 'y':0.9,
1322
+ 'x':0.5,
1323
+ 'xanchor': 'center',
1324
+ 'yanchor': 'top'
1325
+ },
1326
+ yaxis=dict(showticklabels=False)
1327
+ )
1328
+
1329
+ # Initialize the HTML widget for displaying hover information
1330
+ textarea = Textarea(
1331
+ value="",
1332
+ placeholder='',
1333
+ description='Info:',
1334
+ disabled=False,
1335
+ layout = {'width': '95%', 'height': '100px'} ,
1336
+ style = {'description_width': '5%'},
1337
+ # layout={'width': '100%', 'height': '100px'} # Adjust the size as needed
1338
+ )
1339
+ info2 = f"Length: {result_df.iloc[0]['length']:.2f} Imag: {result_df.iloc[0]['imagvalue']:.2f}"+'\n'
1340
+
1341
+ textarea.value = 'The whole model, no equation excluded\n' + info2
1342
+
1343
+
1344
+
1345
+ # Define the function to update the info widget upon hovering over a data point
1346
+ def update_info(trace, points, state):
1347
+ if points.point_inds:
1348
+ ind = points.point_inds[0] # Index of the hovered point
1349
+ # Format the information to display
1350
+
1351
+ info1 = f"Excluded equation: {result_df.iloc[ind]['excluded_description']}"+'\n'
1352
+ info2 = f"Length: {result_df.iloc[ind]['length']:.2f} Imag: {result_df.iloc[ind]['imagvalue']:.2f}"+'\n'
1353
+
1354
+ if result_df.iloc[ind]['excluded'] == 'NONE':
1355
+ textarea.value = 'The whole model, no equation excluded\n' + info2
1356
+ else:
1357
+ textarea.value = info1 + info2 + self.mmodel.allvar[result_df.iloc[ind]['excluded']]['frml']
1358
+
1359
+ # Bind the on_hover event to the update_info function for each trace
1360
+
1361
+ for trace in fig.data:
1362
+ trace.on_hover(update_info)
1363
+
1364
+ # Combine the plot and info widget in a vertical layout and display them
1365
+ display(VBox([fig,textarea]))
1366
+
1367
+
1368
+
1369
+
1370
+
1371
+ def get_eigen_jackknife_abs_select(self,year=2023,largest=20,maxnames = 200_000):
1372
+ """
1373
+ Select and summarize the absolute largest eigenvalues for a specific year from the jackknife analysis.
1374
+
1375
+ This function focuses on a specific year and extracts the sum of the absolute largest eigenvalues obtained from the `get_eigen_jackknife_abs` method. It helps in understanding the aggregate impact of variable exclusions on the system's stability for a particular year.
1376
+
1377
+ Parameters:
1378
+ - year (int, optional): The specific year to focus on. Defaults to 2023.
1379
+ - largest (int, optional): The number of largest eigenvalues to consider. Defaults to 20.
1380
+ - maxnames (int, optional): The maximum number of variables to exclude in the jackknife process. Defaults to 20.
1381
+
1382
+ Returns:
1383
+ pandas.Series: A series sorted by the sum of the absolute largest eigenvalues for each variable exclusion scenario in the specified year.
1384
+
1385
+ Note:
1386
+ This method is useful for temporal analysis of the system's stability, focusing on the contributions of each variable in a specific year.
1387
+ """
1388
+
1389
+ xx = {v: sum(d[year]) for v,d in self.get_eigen_jackknife_abs(maxnames=maxnames,largest=largest).items() }
1390
+ return pd.Series(xx).sort_values()
1391
+
1392
+ def get_df_eigen_dict(self):
1393
+ rownames = [f'{c}{("("+str(l)+")") if l != 0 else ""}' for l in range(-1,self.mmodel.maxlag-1,-1)
1394
+ for c in self.varnames]
1395
+ # breakpoint()
1396
+ values_and_vectors = {per: [pd.DataFrame(vv[0],columns = ['Eigenvalues']).T,
1397
+ pd.DataFrame(vv[1],index = rownames) ]
1398
+ for per,vv in self.eigen_values_and_vectors.items()}
1399
+
1400
+ combined = {per : pd.concat(vandv)
1401
+ for per,vandv in values_and_vectors.items()}
1402
+
1403
+ return combined
1404
+
1405
+ def get_df_comp_dict(self):
1406
+ rownames = [f'{c}{("("+str(l)+")") if l != 0 else ""}' for l in range(-1,self.mmodel.maxlag-1,-1)
1407
+ for c in self.varnames]
1408
+ df_dict = {k : pd.DataFrame(v,columns=rownames,index=rownames)
1409
+ for k,v in self.comp_dic.items()}
1410
+ return df_dict
1411
+
1412
+
1413
+ def eigplot(self, eig_dic=None,per=None,size=(4,3),top=0.9):
1414
+ import matplotlib.pyplot as plt
1415
+ plt.close('all')
1416
+ if type(eig_dic) == type(None):
1417
+ this_eig_dic = self.eig_dic
1418
+ else:
1419
+ this_eig_dic = eig_dic
1420
+ # breakpoint()
1421
+ if type(per) == type(None):
1422
+ first_key = list(this_eig_dic.keys())[0]
1423
+ else:
1424
+ first_key = per
1425
+
1426
+ w = this_eig_dic[first_key]
1427
+
1428
+ fig, ax = plt.subplots(figsize=size,subplot_kw={'projection': 'polar'}) #A4
1429
+ fig.suptitle(f'Eigen values in {first_key}\n',fontsize=20)
1430
+
1431
+ for x in w:
1432
+ ax.plot([0,np.angle(x)],[0,np.abs(x)],marker='o')
1433
+ ax.set_rticks([0.5, 1, 1.5])
1434
+ fig.subplots_adjust(top=0.8)
1435
+ return fig
1436
+
1437
+ def eigenvector_plot(self,per=None,size=(4,3),top=0.9):
1438
+ import matplotlib.pyplot as plt
1439
+
1440
+ this_eig_dic = self.eig_dic
1441
+ # breakpoint()
1442
+ if type(per) == type(None):
1443
+ first_key = list(this_eig_dic.keys())[0]
1444
+ else:
1445
+ first_key = per
1446
+
1447
+ w = this_eig_dic[first_key]
1448
+
1449
+ fig, ax = plt.subplots(figsize=size,subplot_kw={'projection': 'polar'}) #A4
1450
+ fig.suptitle(f'Eigen values in {first_key}\n',fontsize=20)
1451
+
1452
+ for x in w:
1453
+ ax.plot([0,np.angle(x)],[0,np.abs(x)],marker='o')
1454
+ ax.set_rticks([0.5, 1, 1.5])
1455
+ fig.subplots_adjust(top=0.8)
1456
+ return fig
1457
+
1458
+ @staticmethod
1459
+ def get_feedback(eig_dic,per=None):
1460
+ '''Returns a dict of max abs eigenvector and the sign '''
1461
+
1462
+ return {per: max(abs(eigen_vector)) * (-1 if sum(abs(eigen_vector.imag)) else 1)
1463
+ for per,eigen_vector in eig_dic.items()}
1464
+
1465
+
1466
+ def eigplot_all0(self,eig_dic,size=(4,3)):
1467
+ colrows = 4
1468
+ ncols = min(colrows,len(eig_dic))
1469
+ nrows=-((-len(eig_dic))//ncols)
1470
+ fig, axis = plt.subplots(nrows=nrows,ncols=ncols,figsize=(3*ncols,3*nrows),
1471
+ subplot_kw={'projection': 'polar'},constrained_layout=True)
1472
+ # breakpoint()
1473
+ laxis = axis.flatten()
1474
+ for i,(ax,(key,w)) in enumerate(zip(laxis,eig_dic.items())):
1475
+ for x in w:
1476
+ ax.plot([0,np.angle(x)],[0,np.abs(x)],marker='o')
1477
+ ax.set_rticks([0.5, 1, 1.5])
1478
+ ax.set_title(f'{key}',loc='right')
1479
+
1480
+
1481
+ return fig
1482
+
1483
+ def eigplot_all(self,eig_dic,periode=None,size=(4,3),maxfig=6):
1484
+ """
1485
+ Plots the eigenvalues for specified periods in polar coordinates.
1486
+
1487
+ This method takes a dictionary of eigenvalues, optionally filters them by specified periods,
1488
+ and plots each eigenvalue on polar plots. The number of plots can be limited by `maxfig`.
1489
+ If `periode` is not specified, it defaults to the current period defined in the model object.
1490
+ The plots are arranged in a grid, with a maximum of two columns.
1491
+
1492
+ Parameters
1493
+ ----------
1494
+ eig_dic : dict
1495
+ A dictionary where keys are period identifiers and values are iterables of complex numbers
1496
+ representing eigenvalues.
1497
+ periode : iterable, optional
1498
+ An iterable of period identifiers to plot. If `None` (default), eigenvalues for the current
1499
+ period in the model are plotted.
1500
+ size : tuple of int, optional
1501
+ The size of each subplot in inches. Default is (4, 3).
1502
+ maxfig : int, optional
1503
+ The maximum number of figures to display. Default is 6.
1504
+
1505
+ Returns
1506
+ -------
1507
+ matplotlib.figure.Figure
1508
+ A matplotlib Figure object containing the generated plots.
1509
+
1510
+
1511
+
1512
+ """
1513
+
1514
+ plt.close('all')
1515
+ plt.ioff()
1516
+
1517
+ _per_first = periode if type(periode) != type(None) else self.mmodel.current_per
1518
+
1519
+ if hasattr(_per_first,'__iter__'):
1520
+ _per = _per_first
1521
+ else:
1522
+ _per = [_per_first]
1523
+
1524
+ plot_dic = {p : v for p,v in eig_dic.items() if p in _per }
1525
+
1526
+
1527
+ maxaxes = min(maxfig,len(plot_dic))
1528
+ colrow = 2
1529
+ ncols = min(colrow,maxaxes)
1530
+ nrows=-((-maxaxes)//ncols)
1531
+
1532
+ fig = plt.figure(figsize=(9*ncols,10*nrows),constrained_layout=True)
1533
+ spec = mpl.gridspec.GridSpec(ncols=ncols,nrows=nrows,figure=fig)
1534
+ # breakpoint()
1535
+ fig.suptitle('Eigenvalues',fontsize=20)
1536
+
1537
+ for i,(key,w) in enumerate(plot_dic.items()):
1538
+ if i >= maxaxes:
1539
+ break
1540
+ col = i%colrow
1541
+ row = i//colrow
1542
+ # print(i,row,col)
1543
+ ax = fig.add_subplot(spec[row, col],projection='polar')
1544
+ for x in w:
1545
+ ax.plot([0,np.angle(x)],[0,np.abs(x)],marker='o')
1546
+ ax.set_rticks([0.5, 1, 1.5])
1547
+ ax.set_title(f'{key}',loc='right')
1548
+
1549
+ plt.show()
1550
+ return fig
1551
+
1552
+ def eigplot_all(self, eig_dic, periode=None, size=(4, 3), maxfig=6):
1553
+ plt.close('all')
1554
+ plt.ioff()
1555
+
1556
+ _per_first = periode if periode is not None else self.mmodel.current_per
1557
+
1558
+ if hasattr(_per_first, '__iter__'):
1559
+ _per = _per_first
1560
+ else:
1561
+ _per = [_per_first]
1562
+
1563
+ plot_dic = {p: v for p, v in eig_dic.items() if p in _per}
1564
+
1565
+ maxaxes = min(maxfig, len(plot_dic))
1566
+ colrow = 2
1567
+ ncols = min(colrow, maxaxes)
1568
+ nrows = -((-maxaxes) // ncols)
1569
+
1570
+ # Dynamically calculate figure size based on subplot size and layout
1571
+ fig_width = size[0] * ncols # Adjusted to consider 'size' parameter for width
1572
+ fig_height = size[1] * nrows # Adjusted to consider 'size' parameter for height
1573
+
1574
+ fig = plt.figure(figsize=(fig_width, fig_height), constrained_layout=True)
1575
+ spec = mpl.gridspec.GridSpec(ncols=ncols, nrows=nrows, figure=fig)
1576
+
1577
+ fig.suptitle('Eigenvalues', fontsize=20)
1578
+
1579
+ for i, (key, w) in enumerate(plot_dic.items()):
1580
+ if i >= maxaxes:
1581
+ break
1582
+ col = i % colrow
1583
+ row = i // colrow
1584
+ ax = fig.add_subplot(spec[row, col], projection='polar')
1585
+ for x in w:
1586
+ ax.plot([0, np.angle(x)], [0, np.abs(x)], marker='o')
1587
+ ax.set_rticks([0.5, 1, 1.5])
1588
+ ax.set_title(f'{key}', loc='right')
1589
+
1590
+ plt.show()
1591
+ return fig
1592
+
1593
+ def plot_eigenvalues_polar_old(self,eig_dic):
1594
+ import plotly.graph_objects as go
1595
+ import numpy as np
1596
+ from ipywidgets import Dropdown, Output, VBox, HTML
1597
+ from IPython.display import display
1598
+
1599
+
1600
+
1601
+ year_dropdown = Dropdown(options=list(eig_dic.keys()), description='Time:')
1602
+ info_box = HTML(value="Hover over a point to see details.")
1603
+ plot_output = Output()
1604
+
1605
+ def plot_eigenvalues_polar_vectors(year):
1606
+ eigenvalues = eig_dic[year]
1607
+
1608
+ with plot_output:
1609
+ plot_output.clear_output(wait=True)
1610
+ r_values = [np.abs(ev) for ev in eigenvalues] # Magnitudes for the polar plot
1611
+ theta_values = [np.angle(ev, deg=True) for ev in eigenvalues] # Angles for the polar plot
1612
+
1613
+ # Prepare data for the plot, repeating magnitude and angle for each eigenvalue, and adding breaks (None)
1614
+ r_plot_values = [val for pair in zip([0]*len(r_values), r_values, [None]*len(r_values)) for val in pair]
1615
+ theta_plot_values = [val for pair in zip(theta_values, theta_values, [None]*len(theta_values)) for val in pair]
1616
+
1617
+ fig = go.FigureWidget(go.Scatterpolar(
1618
+ r=r_plot_values,
1619
+ theta=theta_plot_values,
1620
+ mode='lines+markers',
1621
+ marker=dict(color='blue', size=5),
1622
+ line=dict(color='blue')
1623
+ ))
1624
+
1625
+ fig.update_layout(
1626
+ title=f'Polar Plot of Eigenvalue Vectors for Year {year}',
1627
+ polar=dict(
1628
+ radialaxis=dict(visible=True, range=[0, max(r_values) * 1.1]),
1629
+ angularaxis=dict(direction='clockwise', thetaunit='degrees', rotation=0)
1630
+ ),
1631
+ showlegend=False
1632
+ )
1633
+
1634
+ def update_info(trace, points, _):
1635
+ if points.point_inds:
1636
+ # Each eigenvalue is represented by 3 points in the plot data (start, end, None), calculate the index accordingly
1637
+ ind = points.point_inds[0] // 3
1638
+ ev = eigenvalues[ind] # Get the eigenvalue
1639
+
1640
+ for trace in fig.data:
1641
+ trace.on_hover(update_info)
1642
+
1643
+ display(fig)
1644
+
1645
+ def on_year_change(change):
1646
+ plot_eigenvalues_polar_vectors(change.new)
1647
+
1648
+ year_dropdown.observe(on_year_change, names='value')
1649
+ display(VBox([year_dropdown, plot_output]))
1650
+ plot_eigenvalues_polar_vectors(year_dropdown.value)
1651
+
1652
+ def eigenvalues_show(self,eig_dic):
1653
+ """
1654
+ Generates and displays a polar plot of eigenvalues and their corresponding eigenvectors
1655
+ for a selected year from a dictionary of DataFrames. Each DataFrame contains eigenvalues
1656
+ and eigenvectors of the companion matrix for that year. The first row of the DataFrame
1657
+ consists of eigenvalues, and the subsequent rows contain the corresponding eigenvectors.
1658
+ The user can interact with the plot via a dropdown for year selection, a slider for eigenvalue
1659
+ selection, and a button to toggle additional plot details. The polar plot dynamically updates
1660
+ to reflect the selected eigenvalue, displaying its magnitude and phase. Additional information
1661
+ about the selected eigenvector is also displayed, facilitating a detailed temporal analysis
1662
+ of the eigenvalues.
1663
+
1664
+ Parameters:
1665
+ - eig_dic (dict): A dictionary where keys are years (or time periods) and values are DataFrames
1666
+ containing the first row as eigenvalues and the subsequent rows as the corresponding
1667
+ eigenvectors of the companion matrix for each year.
1668
+
1669
+ Side Effects:
1670
+ - Displays interactive widgets including a dropdown for year selection, a plot output area,
1671
+ and a slider for selecting specific eigenvalues. Additionally, displays textual information
1672
+ about the selected eigenvalue and its eigenvectors.
1673
+ - Utilizes Plotly for generating the polar plot and ipywidgets for interactive controls.
1674
+ - The method defines and uses several inner functions to handle events like year change,
1675
+ eigenvalue selection, and other interactions.
1676
+
1677
+ Returns:
1678
+ - None. The method's primary function is to display interactive widgets and plots within a Jupyter
1679
+ notebook environment.
1680
+
1681
+ Note:
1682
+ - This method is designed for use within a Jupyter notebook as it relies on IPython.display
1683
+ for rendering and ipywidgets for interactivity.
1684
+ - The actual plotting and widget setup are accomplished through several nested functions within
1685
+ this method, making use of closures and nonlocal variables for state management.
1686
+ """
1687
+
1688
+ import plotly.graph_objects as go
1689
+ import numpy as np
1690
+ from ipywidgets import Dropdown, Output, VBox, HTML, HBox, IntSlider,Select,Textarea,Checkbox,Button
1691
+ from IPython.display import display
1692
+
1693
+
1694
+
1695
+
1696
+
1697
+
1698
+ year_dropdown = Dropdown(options=list(eig_dic.keys()), description='Time:')
1699
+ plot_output = Output()
1700
+
1701
+ def get_a_eigenvalue_vector(eigenvalues_vectors,year):
1702
+ """
1703
+ Extracts and processes eigenvalues and their corresponding eigenvectors for a given year.
1704
+ Filters and sorts eigenvalues by their magnitude, returning the significant eigenvalues and
1705
+ their associated eigenvectors.
1706
+
1707
+ Parameters:
1708
+ - eigenvalues_vectors (dict): Dictionary containing DataFrames of eigenvalues and eigenvectors
1709
+ indexed by year.
1710
+ - year (str/int): The year for which to extract and process the eigenvalue vector.
1711
+
1712
+ Returns:
1713
+ - tuple: A tuple containing the significant eigenvalue and a DataFrame of the corresponding
1714
+ eigenvectors after processing.
1715
+ """
1716
+
1717
+
1718
+ compabs = lambda complex: [abs(value) for index,value in complex.items()]
1719
+
1720
+ eig_gt = (eigenvalues_vectors[year].
1721
+ T. # Transpose as we query and eval on columns
1722
+ eval('absolute_value=@compabs(Eigenvalues)'). # calculate the absolute value of the eigenvalue
1723
+ query('absolute_value > 0.01').
1724
+ sort_values(by='absolute_value',ascending=False).reset_index(drop=True). # Transpose again.
1725
+ drop('absolute_value',axis=1) # We dont need the absolute value anymore, so the column is dropped.
1726
+ )
1727
+
1728
+ return eig_gt.iloc[:,0],eig_gt.iloc[:,1:].abs()
1729
+
1730
+
1731
+
1732
+ def plot_eigenvalues_polar_vectors(year):
1733
+ eigenvalues,eigenvectors = get_a_eigenvalue_vector(eig_dic,year )
1734
+ valueslider = IntSlider(
1735
+ value=1,
1736
+ min=0,
1737
+ max=len(eigenvalues)-1,
1738
+ step=1,
1739
+ description='Number:',
1740
+ disabled=False,
1741
+ continuous_update=False,
1742
+ orientation='vertical',
1743
+ readout=True,
1744
+ readout_format='d'
1745
+ )
1746
+
1747
+ var_info = Textarea(
1748
+ value="",
1749
+ placeholder='',
1750
+ description='Info:',
1751
+ disabled=False,
1752
+ layout = {'width': '95%', 'height': '100px'} ,
1753
+ style = {'description_width': '5%'},
1754
+ # layout={'width': '100%', 'height': '100px'} # Adjust the size as needed
1755
+ )
1756
+
1757
+ wopenplot = Button(value=True,description = 'Open plot widget',disabled=False,icon='check',
1758
+ layout={'width':'95%'} ,style={'description_width':'70%'})
1759
+
1760
+ with plot_output:
1761
+ plot_output.clear_output(wait=True)
1762
+ r_values = [np.abs(ev) for ev in eigenvalues] # Magnitudes for the polar plot
1763
+ theta_values = [np.angle(ev, deg=True) for ev in eigenvalues] # Angles for the polar plot
1764
+
1765
+ # Prepare data for the plot, repeating magnitude and angle for each eigenvalue, and adding breaks (None)
1766
+ r_plot_values = [val for pair in zip([0]*len(r_values), r_values, [None]*len(r_values)) for val in pair]
1767
+ theta_plot_values = [val for pair in zip(theta_values, theta_values, [None]*len(theta_values)) for val in pair]
1768
+
1769
+ fig = go.FigureWidget(go.Scatterpolar(
1770
+ r=r_plot_values,
1771
+ theta=theta_plot_values,
1772
+ mode='lines+markers',
1773
+ marker=dict(color='blue', size=5),
1774
+ line=dict(color='blue')
1775
+ ))
1776
+ fig.update_layout(
1777
+ title_text='', # Ensure no title is set
1778
+ margin=dict(l=20, r=20, t=20, b=20) # Adjust margins around the plot
1779
+ )
1780
+
1781
+ fig.update_layout(
1782
+ # title=f'Polar Plot of Eigenvalue Vectors for Year {year}',
1783
+ polar=dict(
1784
+ radialaxis=dict(visible=True, range=[0, max(r_values) * 1.1]),
1785
+ angularaxis=dict(direction='clockwise', thetaunit='degrees', rotation=0)
1786
+ ),
1787
+ showlegend=False
1788
+ )
1789
+ v_dropdown_description = HTML(value="<strong>Eigenvalue :</strong> <br><strong>Eigenvectors:</strong>", placeholder='',description='',)
1790
+ v_dropdown = Select(description='',rows=14,layout = {'width': '95%'})
1791
+ box = VBox([HBox([fig,valueslider,VBox([v_dropdown_description,v_dropdown,wopenplot])]),var_info])
1792
+
1793
+ lopenplot = False
1794
+
1795
+ def vector_info(selected_index):
1796
+ """
1797
+ Updates the displayed information for a selected eigenvector, including its magnitude,
1798
+ phase, and detailed component values. Dynamically updates dropdown options to reflect
1799
+ the components of the selected eigenvector.
1800
+
1801
+ Parameters:
1802
+ - selected_index (int): Index of the selected eigenvalue/eigenvector.
1803
+ """
1804
+
1805
+ nonlocal lopenplot
1806
+ this_vector = eigenvectors.iloc[selected_index,:].sort_values(ascending=False)
1807
+ eigenvalue = eigenvalues[selected_index]
1808
+
1809
+ select_options = [(f"{index} - {value:.2f}", f"{index}") for index, value in this_vector.items() if value >= 0.01]
1810
+ v_dropdown.options = select_options
1811
+
1812
+ v_dropdown_description.value=f"<strong>Eigenvalue #: {selected_index}</strong> <br>Length: {abs(eigenvalue):.2f}<br>Real: {eigenvalue.real:.2f}<br>Imag: {eigenvalue.imag:.2f} <br><strong>Eigenvectors:</strong>"
1813
+ # gross_varnames = [v.split('(')[0] for t,v in select_options]
1814
+ # varnames = ' '.join(list(dict.fromkeys(gross_varnames)))
1815
+ # keepbox = self.mmodel.keep_show(use_var_groups=False,selectfrom = varnames,use_smpl= True)
1816
+ if lopenplot :
1817
+ plot_output.clear_output(wait=True)
1818
+ # box = VBox([HBox([fig,valueslider,VBox([v_dropdown_description,v_dropdown,wopenplot])]),var_info,keepbox.datawidget])
1819
+ box = VBox([HBox([fig,valueslider,VBox([v_dropdown_description,v_dropdown,wopenplot])]),var_info])
1820
+ display(box)
1821
+ lopenplot = False
1822
+
1823
+
1824
+
1825
+ vector_info(0)
1826
+
1827
+ def on_openplot(change):
1828
+ """
1829
+ Handles the event triggered by clicking the 'Open plot widget' button. It refreshes the plot
1830
+ and optionally includes additional widgets for further data exploration.
1831
+
1832
+ Parameters:
1833
+ - change (dict): Contains details of the button click event. Not used in the function body
1834
+ but necessary for event handler signature.
1835
+ """
1836
+
1837
+ nonlocal lopenplot
1838
+ lopenplot = True
1839
+ gross_varnames = [v.split('(')[0] for t,v in v_dropdown.options]
1840
+ varnames = ' '.join(list(dict.fromkeys(gross_varnames)))
1841
+ keepbox = self.mmodel.keep_show(use_var_groups=False,selectfrom = varnames,use_smpl= True,init_dif=True)
1842
+ box = VBox([HBox([fig,valueslider,VBox([v_dropdown_description,v_dropdown,wopenplot])]),var_info,keepbox.datawidget])
1843
+ plot_output.clear_output(wait=True)
1844
+
1845
+ display(box)
1846
+
1847
+ def info_show(ind):
1848
+ """
1849
+ Displays information about the eigenvalue and its corresponding eigenvectors for a given index.
1850
+ This function is designed to update the displayed information based on user selection or interaction.
1851
+
1852
+ Parameters:
1853
+ - ind (int): The index of the selected eigenvalue to display information for.
1854
+ """
1855
+
1856
+ ev = eigenvalues[ind] # Get the eigenvalue
1857
+ vector_info(ind)
1858
+
1859
+ def update_info(trace, points, _):
1860
+ """
1861
+ Callback function to handle hover events on the plot. It identifies the eigenvalue
1862
+ corresponding to the hovered point and updates the information displayed to the user.
1863
+
1864
+ Parameters:
1865
+ - trace: The trace object associated with the hover event. Not used in the function body.
1866
+ - points: The points object containing information about the hovered point.
1867
+ - _: Placeholder for additional arguments. Not used in the function body.
1868
+ """
1869
+
1870
+ nonlocal lopenplot
1871
+ if points.point_inds:
1872
+ if lopenplot :
1873
+ plot_output.clear_output(wait=True)
1874
+ # box = VBox([HBox([fig,valueslider,VBox([v_dropdown_description,v_dropdown,wopenplot])]),var_info,keepbox.datawidget])
1875
+ box = VBox([HBox([fig,valueslider,VBox([v_dropdown_description,v_dropdown,wopenplot])]),var_info])
1876
+ display(box)
1877
+ lopenplot = False
1878
+
1879
+ # Each eigenvalue is represented by 3 points in the plot data (start, end, None), calculate the index accordingly
1880
+ ind = points.point_inds[0] // 3
1881
+ valueslider.value = ind
1882
+ info_show(ind)
1883
+
1884
+
1885
+
1886
+
1887
+ def on_v_dropdown_change(change):
1888
+ """
1889
+ Handles changes in the eigenvector selection dropdown. Updates the displayed information
1890
+ related to the selected eigenvector, including its description and formula.
1891
+
1892
+ Parameters:
1893
+ - change (dict): Contains details of the selection change in the dropdown widget.
1894
+ """
1895
+
1896
+ # print(f'{change=}')
1897
+ # print(f'{change.owner=}')
1898
+ # print(f'{change.owner.options=}')
1899
+ # print(f'{change.new=}' + '\n')
1900
+ if type(change.new) == dict and len(change.new ):
1901
+ index=change.new['index']
1902
+ name = change.owner.options[index][1]
1903
+ varname = name.split('(')[0]
1904
+ info1 = f'{varname}: {self.mmodel.var_description[varname]}' +'\n'
1905
+ info2 = self.mmodel.allvar[varname]['frml']
1906
+ var_info.value= info1 + info2
1907
+
1908
+
1909
+
1910
+
1911
+
1912
+ for trace in fig.data:
1913
+ trace.on_hover(update_info)
1914
+
1915
+
1916
+
1917
+
1918
+ def on_slide_change(change):
1919
+ """
1920
+ Responds to changes in the eigenvalue selection slider. Updates the plot to highlight
1921
+ the selected eigenvalue and its corresponding eigenvectors, and updates displayed information.
1922
+
1923
+ Parameters:
1924
+ - change (dict): Contains details of the slider value change.
1925
+ """
1926
+
1927
+ # print(f'{change.new=} ')
1928
+ if type(change.new) == int:
1929
+ selected_index = change.new
1930
+ else:
1931
+ if len(change.new):
1932
+ selected_index = change.new['value']
1933
+ else:
1934
+ return
1935
+ # print(f'{change.new=} {selected_index=}')
1936
+ colors = ['red' if i == selected_index else 'green' for i in range(len(eigenvalues))]
1937
+ fig.data[0].marker.color = [color for pair in zip(colors, colors, colors) for color in pair]
1938
+ sizes = [15 if i == selected_index else 5 for i in range(len(eigenvalues))]
1939
+ fig.data[0].marker.size = [size for pair in zip(sizes,sizes,sizes) for size in pair]
1940
+
1941
+ info_show(selected_index)
1942
+
1943
+ # update_info(None,SimulatedPoints(0),None)
1944
+
1945
+
1946
+ wopenplot.on_click(on_openplot)
1947
+
1948
+ valueslider.observe(on_slide_change)
1949
+
1950
+ v_dropdown.observe(on_v_dropdown_change)
1951
+
1952
+ valueslider.value= 0
1953
+
1954
+ display(box)
1955
+
1956
+ def on_year_change(change):
1957
+ """
1958
+ Callback function for handling changes in the year selection dropdown. Updates the polar plot
1959
+ to display the eigenvalues and eigenvectors for the newly selected year.
1960
+
1961
+ Parameters:
1962
+ - change (dict): Contains details about the change event in the year dropdown.
1963
+ """
1964
+
1965
+ plot_eigenvalues_polar_vectors(change.new)
1966
+
1967
+ year_dropdown.observe(on_year_change, names='value')
1968
+ display(VBox([year_dropdown, plot_output]))
1969
+ plot_eigenvalues_polar_vectors(year_dropdown.value)
1970
+
1971
+ def analyze_jacobian_df(
1972
+ J: pd.DataFrame,
1973
+ tol: float = 1e-12,
1974
+ top_k: int = 20,
1975
+ compute_svd: bool = True,
1976
+ max_svd_dim: int = 2000,
1977
+ similarity_threshold: float = 0.999999,
1978
+ ) -> dict:
1979
+ """
1980
+ Analyze a Jacobian stored as a pandas DataFrame with variable names in both
1981
+ rows and columns (index + columns).
1982
+
1983
+ Returns a dict with:
1984
+ - basic stats
1985
+ - zero rows/cols (by name)
1986
+ - duplicate rows/cols (exact)
1987
+ - "near-duplicate" row pairs (cosine similarity above threshold)
1988
+ - rank/conditioning via SVD (optional, size-limited)
1989
+ - strongest column per row summary (helps spot unmatched structure)
1990
+
1991
+ Parameters
1992
+ ----------
1993
+ J : pd.DataFrame
1994
+ Jacobian matrix, index=equation/row variable names, columns=variable names.
1995
+ tol : float
1996
+ Tolerance used to treat values as zero (|a| < tol).
1997
+ top_k : int
1998
+ How many items to show in top lists (for reporting).
1999
+ compute_svd : bool
2000
+ Whether to compute SVD-based diagnostics (dense).
2001
+ max_svd_dim : int
2002
+ Only compute SVD if both dimensions <= this threshold.
2003
+ similarity_threshold : float
2004
+ Threshold for cosine similarity to flag near-duplicate row pairs.
2005
+ """
2006
+ if not isinstance(J, pd.DataFrame):
2007
+ raise TypeError("J must be a pandas DataFrame")
2008
+
2009
+ if J.shape[0] == 0 or J.shape[1] == 0:
2010
+ raise ValueError("J is empty")
2011
+
2012
+ # ensure numeric
2013
+ Jnum = J.apply(pd.to_numeric, errors="coerce")
2014
+ nonfinite_mask = ~np.isfinite(Jnum.to_numpy(dtype=float))
2015
+ nonfinite_count = int(nonfinite_mask.sum())
2016
+
2017
+ A = Jnum.to_numpy(dtype=float)
2018
+ absA = np.abs(A)
2019
+
2020
+ # basic counts
2021
+ nnz = int((absA > tol).sum())
2022
+ density = nnz / (A.size)
2023
+
2024
+ # zero rows / cols
2025
+ zero_rows = Jnum.index[(absA < tol).all(axis=1)].tolist()
2026
+ zero_cols = Jnum.columns[(absA < tol).all(axis=0)].tolist()
2027
+
2028
+ # exact duplicates
2029
+ dup_row_names = Jnum.index[Jnum.duplicated(keep=False)].tolist()
2030
+ dup_col_names = Jnum.columns[Jnum.T.duplicated(keep=False)].tolist()
2031
+
2032
+ # strongest column per row
2033
+ absdf = Jnum.abs()
2034
+ strongest_col = absdf.idxmax(axis=1) # column name for each row
2035
+ strongest_val = absdf.max(axis=1)
2036
+ strongest_counts = strongest_col.value_counts()
2037
+
2038
+ # top "weak" rows (small max derivative)
2039
+ weak_rows = strongest_val.sort_values().head(top_k)
2040
+
2041
+ # near-duplicate rows via cosine similarity (skip if huge)
2042
+ near_dup_pairs = []
2043
+ if J.shape[0] <= 5000: # safety guard
2044
+ row_norm = np.sqrt((A * A).sum(axis=1))
2045
+ safe_norm = np.where(row_norm == 0, 1.0, row_norm)
2046
+ Jr = (A.T / safe_norm).T # row-normalized
2047
+ # cosine similarity matrix
2048
+ S = Jr @ Jr.T
2049
+ np.fill_diagonal(S, 0.0)
2050
+ pairs = np.argwhere(np.abs(S) >= similarity_threshold)
2051
+ # keep only i<j to avoid duplicates
2052
+ pairs = [(int(i), int(j), float(S[i, j])) for i, j in pairs if i < j]
2053
+ # sort by similarity magnitude
2054
+ pairs.sort(key=lambda x: abs(x[2]), reverse=True)
2055
+ for i, j, sim in pairs[:top_k]:
2056
+ near_dup_pairs.append((Jnum.index[i], Jnum.index[j], sim))
2057
+
2058
+ # SVD-based diagnostics (dense) if not too big
2059
+ svd_info = None
2060
+ if compute_svd and (J.shape[0] <= max_svd_dim and J.shape[1] <= max_svd_dim):
2061
+ try:
2062
+ s = np.linalg.svd(A, compute_uv=False)
2063
+ smin = float(s.min()) if s.size else np.nan
2064
+ smax = float(s.max()) if s.size else np.nan
2065
+ cond = (smax / smin) if smin not in (0.0, np.nan) else np.inf
2066
+ # near-zero singular values (relative criterion)
2067
+ rel = s / smax if smax not in (0.0, np.nan) else s
2068
+ near_zero = int((rel < 1e-12).sum()) # heuristic
2069
+ svd_info = {
2070
+ "min_singular_value": smin,
2071
+ "max_singular_value": smax,
2072
+ "condition_number": float(cond) if np.isfinite(cond) else np.inf,
2073
+ "near_zero_singular_values_count(rel<1e-12)": near_zero,
2074
+ }
2075
+ except Exception as e:
2076
+ svd_info = {"error": repr(e), "note": "SVD failed (possibly ill-conditioned or NaNs)."}
2077
+
2078
+ report = {
2079
+ "shape": J.shape,
2080
+ "tol": tol,
2081
+ "nonfinite_entries_count": nonfinite_count,
2082
+ "nnz(|a|>tol)": nnz,
2083
+ "density": density,
2084
+ "zero_rows": zero_rows,
2085
+ "zero_cols": zero_cols,
2086
+ "duplicate_rows_exact": dup_row_names,
2087
+ "duplicate_cols_exact": dup_col_names,
2088
+ "strongest_col_per_row_counts_top": strongest_counts.head(top_k).to_dict(),
2089
+ "weak_rows_smallest_row_max_abs_top": weak_rows.to_dict(), # row -> max_abs_value
2090
+ "near_duplicate_row_pairs_top": near_dup_pairs, # (row_i, row_j, cosine_sim)
2091
+ "svd_info": svd_info,
2092
+ }
2093
+ return report
2094
+
2095
+
2096
+ def print_jacobian_report(report: dict, top_k: int = 20) -> None:
2097
+ """Pretty-print key parts of analyze_jacobian_df() output."""
2098
+ print(f"Shape: {report.get('shape')}, tol={report.get('tol')}")
2099
+ print(f"Non-finite entries: {report.get('nonfinite_entries_count')}")
2100
+ print(f"NNZ(|a|>tol): {report.get('nnz(|a|>tol)')}, density={report.get('density'):.6g}")
2101
+
2102
+ zr = report.get("zero_rows", [])
2103
+ zc = report.get("zero_cols", [])
2104
+ print(f"\nZero rows: {len(zr)}")
2105
+ if zr:
2106
+ print(" examples:", zr[:top_k])
2107
+
2108
+ print(f"\nZero cols: {len(zc)}")
2109
+ if zc:
2110
+ print(" examples:", zc[:top_k])
2111
+
2112
+ dr = report.get("duplicate_rows_exact", [])
2113
+ dc = report.get("duplicate_cols_exact", [])
2114
+ print(f"\nExact duplicate rows: {len(dr)}")
2115
+ if dr:
2116
+ print(" examples:", dr[:top_k])
2117
+
2118
+ print(f"\nExact duplicate cols: {len(dc)}")
2119
+ if dc:
2120
+ print(" examples:", dc[:top_k])
2121
+
2122
+ print("\nStrongest column per row (top):")
2123
+ sc = report.get("strongest_col_per_row_counts_top", {})
2124
+ for k, v in list(sc.items())[:top_k]:
2125
+ print(f" {k}: {v}")
2126
+
2127
+ print("\nWeak rows (smallest row max |derivative|):")
2128
+ wr = report.get("weak_rows_smallest_row_max_abs_top", {})
2129
+ for k, v in list(wr.items())[:top_k]:
2130
+ print(f" {k}: {v:g}")
2131
+
2132
+ nd = report.get("near_duplicate_row_pairs_top", [])
2133
+ print(f"\nNear-duplicate row pairs (cosine sim thresholded): {len(nd)}")
2134
+ for a, b, sim in nd[:top_k]:
2135
+ print(f" {a} <-> {b} sim={sim:+.6f}")
2136
+
2137
+ svd = report.get("svd_info")
2138
+ if svd:
2139
+ print("\nSVD info:")
2140
+ for k, v in svd.items():
2141
+ print(f" {k}: {v}")
2142
+
2143
+ #%%
2144
+ if __name__ == '__main__':
2145
+ #%% testing
2146
+ os.environ['PYTHONBREAKPOINT'] = '1'
2147
+ from modelclass import model
2148
+ fsolow = '''\
2149
+ Y = a * k**alfa * l **(1-alfa)
2150
+ C = (1-SAVING_RATIO) * Y(-1)
2151
+ I = Y - C
2152
+ diff(K) = I-depreciates_rate * K(-1)
2153
+ diff(l) = labor_growth * (L(-1)+l(-2))/2
2154
+ K_intense = K/L '''
2155
+ msolow = model.from_eq(fsolow)
2156
+ #print(msolow.equations)
2157
+ N = 32
2158
+ df = pd.DataFrame({'L':[100]*N,'K':[100]*N},index =[i+2000 for i in range(N)])
2159
+ df.loc[:,'ALFA'] = 0.5
2160
+ df.loc[:,'DEPRECIATES_RATE'] = 0.01
2161
+ df.loc[:,'LABOR_GROWTH'] = 0.01
2162
+ df.loc[:,'SAVING_RATIO'] = 0.10
2163
+ msolow(df,silent=1,ljit=0,transpile_reset=1)
2164
+ msolow.normalized = True
2165
+
2166
+ newton_all = newton_diff(msolow,forcenum=0,per=2002)
2167
+ dif__model = newton_all.diff_model.equations
2168
+ melt = newton_all.get_diff_melted_var()
2169
+ tt = newton_all.get_diff_mat_all_1per(2002,asdf=True)
2170
+ #newton_all.show_diff()
2171
+ cc = newton_all.get_eigenvalues (asdf=True,periode=2010)
2172
+ fig= newton_all.eigplot_all(cc,maxfig=3)
2173
+ #%% more testing
2174
+ if 1:
2175
+ newton = newton_diff(msolow)
2176
+ pdic = newton.get_diff_df_1per()
2177
+ longdf = newton.get_diff_melted()
2178
+