import%20marimo%0A%0A__generated_with%20%3D%20%220.25.0%22%0Aapp%20%3D%20marimo.App()%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20marimo%20as%20mo%0A%0A%20%20%20%20return%20(mo%2C)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20METADATA%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%22id%22%3A%20%22modern_tcn%22%2C%0A%20%20%20%20%20%20%20%20%22name%22%3A%20%22ModernTCN%22%2C%0A%20%20%20%20%20%20%20%20%22types%22%3A%20%5B%22architecture%22%5D%2C%0A%20%20%20%20%20%20%20%20%22families%22%3A%20%5B%22neural_networks%22%2C%20%22convolutional%22%2C%20%22sequential%22%5D%2C%0A%20%20%20%20%20%20%20%20%22tasks%22%3A%20%5B%22forecasting%22%2C%20%22classification%22%2C%20%22anomaly_detection%22%2C%20%22representation%22%5D%2C%0A%20%20%20%20%20%20%20%20%22data%22%3A%20%5B%22time_series%22%2C%20%22sequences%22%5D%2C%0A%20%20%20%20%20%20%20%20%22learning%22%3A%20%5B%22supervised%22%5D%2C%0A%20%20%20%20%20%20%20%20%22capacity%22%3A%20%22parametric%22%2C%0A%20%20%20%20%20%20%20%20%22mechanisms%22%3A%20%5B%22depthwise_convolution%22%2C%20%22large_kernels%22%2C%20%22patching%22%2C%20%22backpropagation%22%5D%2C%0A%20%20%20%20%20%20%20%20%22properties%22%3A%20%5B%22nonlinear%22%2C%20%22representation_learning%22%5D%2C%0A%20%20%20%20%20%20%20%20%22constraints%22%3A%20%5B%22requires_large_data%22%2C%20%22requires_scaling%22%2C%20%22sensitive_to_tuning%22%5D%2C%0A%20%20%20%20%20%20%20%20%22difficulty%22%3A%20%22advanced%22%2C%0A%20%20%20%20%20%20%20%20%22status%22%3A%20%22documented%22%2C%0A%20%20%20%20%20%20%20%20%22explainability%22%3A%20%22low%22%2C%0A%20%20%20%20%20%20%20%20%22training_cost%22%3A%20%22high%22%2C%0A%20%20%20%20%20%20%20%20%22inference_cost%22%3A%20%22medium%22%2C%0A%20%20%20%20%20%20%20%20%22data_appetite%22%3A%20%22high%22%2C%0A%20%20%20%20%7D%0A%0A%20%20%20%20mo.md(f%22%23%20%7BMETADATA%5B'name'%5D%7D%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20In%20one%20sentence%0A%0A%20%20%20%20ModernTCN%20is%20a%20pure%20convolutional%20architecture%20that%20uses%20patches%2C%20large%20depthwise%20kernels%2C%20and%20separate%20temporal%20and%20cross-variable%20mixing%20for%20general%20time-series%20analysis.%0A%0A%20%20%20%20%23%23%20Mental%20model%0A%0A%20%20%20%20Treat%20a%20multivariate%20series%20like%20a%20grid%3A%20time%20runs%20horizontally%20and%20variables%20run%20vertically.%20ModernTCN%20first%20compresses%20nearby%20observations%20into%20patches%2C%20then%20uses%20large%20convolutions%20to%20discover%20long%20temporal%20motifs.%20A%20separate%20pointwise%20feed-forward%20stage%20mixes%20information%20across%20variables.%0A%0A%20%20%20%20This%20separation%20is%20the%20main%20idea%3A%20**learn%20temporal%20structure%20without%20immediately%20blending%20every%20channel%2C%20then%20learn%20how%20channels%20interact**.%0A%0A%20%20%20%20%23%23%20How%20it%20differs%20from%20a%20classic%20TCN%0A%0A%20%20%20%20A%20classic%20TCN%20usually%20expands%20history%20through%20causal%20dilations.%20ModernTCN%20takes%20inspiration%20from%20modern%20ConvNet%20design%3A%20large%20depthwise%20kernels%2C%20inverted%20bottlenecks%2C%20residual%20connections%2C%20and%20patching.%20Its%20receptive%20field%20comes%20mainly%20from%20kernel%20width%20and%20depth%20rather%20than%20an%20exponential%20dilation%20schedule.%0A%0A%20%20%20%20The%20official%20ICLR%202024%20work%20evaluates%20one%20architecture%20across%20long-%20and%20short-horizon%20forecasting%2C%20imputation%2C%20classification%2C%20and%20anomaly%20detection.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20Input%20and%20output%0A%0A%20%20%20%20-%20**Input%3A**%20a%20regularly%20sampled%20window%20with%20time%20steps%20by%20variables.%0A%20%20%20%20-%20**Output%3A**%20a%20future%20horizon%2C%20class%2C%20reconstruction%2C%20or%20anomaly%20score%20depending%20on%20the%20head.%0A%20%20%20%20-%20**Learns%3A**%20patch%20embeddings%2C%20per-variable%20temporal%20filters%2C%20cross-variable%20mixing%2C%20and%20the%20prediction%20head.%0A%0A%20%20%20%20%23%23%20The%20ModernTCN%20block%0A%0A%20%20%20%20For%20a%20depthwise%20temporal%20convolution%2C%20each%20variable%20initially%20receives%20its%20own%20kernel%3A%0A%0A%20%20%20%20%24%24%0A%20%20%20%20z_%7Bc%2Ct%7D%3D%5Csum_%7Bj%3D0%7D%5E%7Bk-1%7Dw_%7Bc%2Cj%7Dx_%7Bc%2Ct-j%7D.%0A%20%20%20%20%24%24%0A%0A%20%20%20%20The%20operation%20captures%20long%20patterns%20efficiently%20because%20it%20does%20not%20create%20every%20input-output%20channel%20pair%20inside%20the%20large%20kernel.%20A%20later%20pointwise%20transformation%20can%20then%20mix%20the%20channel%20representations.%0A%0A%20%20%20%20%23%23%20When%20to%20use%20it%0A%0A%20%20%20%20-%20You%20have%20many%20related%2C%20regularly%20sampled%20series.%0A%20%20%20%20-%20Local%20and%20medium-range%20motifs%20are%20plausible.%0A%20%20%20%20-%20Parallel%20training%20and%20inference%20matter.%0A%20%20%20%20-%20You%20can%20train%20and%20backtest%20a%20domain-specific%20model.%0A%0A%20%20%20%20%23%23%20When%20to%20avoid%20it%0A%0A%20%20%20%20-%20Data%20is%20scarce%20and%20a%20seasonal%20baseline%20is%20already%20competitive.%0A%20%20%20%20-%20Irregular%20timing%20is%20not%20encoded%20explicitly.%0A%20%20%20%20-%20The%20relevant%20dependency%20is%20longer%20than%20the%20effective%20receptive%20field.%0A%20%20%20%20-%20You%20need%20a%20ready-to-use%20zero-shot%20model%20rather%20than%20an%20architecture%20to%20train.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20---%0A%0A%20%20%20%20%23%23%20Notebook%20%E2%80%94%20what%20a%20wide%20depthwise%20kernel%20can%20see%0A%0A%20%20%20%20This%20is%20a%20structural%20illustration%2C%20not%20a%20reimplementation%20of%20the%20full%20training%20code.%20We%20create%20three%20related%20variables%20and%20apply%20the%20same%20**kind**%20of%20per-variable%20wide%20temporal%20filtering%20used%20to%20build%20a%20large%20effective%20receptive%20field.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20numpy%20as%20np%0A%20%20%20%20import%20matplotlib.pyplot%20as%20plt%0A%0A%20%20%20%20rng%20%3D%20np.random.default_rng(7)%0A%20%20%20%20t%20%3D%20np.arange(240)%0A%20%20%20%20signal%20%3D%20np.vstack(%0A%20%20%20%20%20%20%20%20%5B%0A%20%20%20%20%20%20%20%20%20%20%20%20np.sin(2%20*%20np.pi%20*%20t%20%2F%2024)%20%2B%200.25%20*%20np.sin(2%20*%20np.pi%20*%20t%20%2F%207)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%200.8%20*%20np.sin(2%20*%20np.pi%20*%20(t%20-%204)%20%2F%2024)%20%2B%200.003%20*%20t%2C%0A%20%20%20%20%20%20%20%20%20%20%20%200.5%20*%20np.sin(2%20*%20np.pi%20*%20t%20%2F%2048)%20%2B%200.2%20*%20np.cos(2%20*%20np.pi%20*%20t%20%2F%2012)%2C%0A%20%20%20%20%20%20%20%20%5D%0A%20%20%20%20)%0A%20%20%20%20signal%20%2B%3D%20rng.normal(scale%3D0.16%2C%20size%3Dsignal.shape)%0A%20%20%20%20return%20np%2C%20plt%2C%20signal%2C%20t%0A%0A%0A%40app.cell%0Adef%20_(np%2C%20plt%2C%20signal%2C%20t)%3A%0A%20%20%20%20def%20depthwise_average(values%2C%20kernel_size)%3A%0A%20%20%20%20%20%20%20%20kernel%20%3D%20np.ones(kernel_size)%20%2F%20kernel_size%0A%20%20%20%20%20%20%20%20return%20np.vstack(%0A%20%20%20%20%20%20%20%20%20%20%20%20%5Bnp.convolve(channel%2C%20kernel%2C%20mode%3D%22same%22)%20for%20channel%20in%20values%5D%0A%20%20%20%20%20%20%20%20)%0A%0A%20%20%20%20narrow%20%3D%20depthwise_average(signal%2C%20kernel_size%3D5)%0A%20%20%20%20wide%20%3D%20depthwise_average(signal%2C%20kernel_size%3D31)%0A%0A%20%20%20%20fig_filters%2C%20axes%20%3D%20plt.subplots(3%2C%201%2C%20figsize%3D(11%2C%207)%2C%20sharex%3DTrue)%0A%20%20%20%20for%20channel%2C%20axis%20in%20enumerate(axes)%3A%0A%20%20%20%20%20%20%20%20axis.plot(t%2C%20signal%5Bchannel%5D%2C%20color%3D%22%23a8b2ac%22%2C%20linewidth%3D1%2C%20label%3D%22observed%22)%0A%20%20%20%20%20%20%20%20axis.plot(t%2C%20narrow%5Bchannel%5D%2C%20color%3D%22%232b78b8%22%2C%20label%3D%22kernel%20%3D%205%22)%0A%20%20%20%20%20%20%20%20axis.plot(t%2C%20wide%5Bchannel%5D%2C%20color%3D%22%23e18719%22%2C%20linewidth%3D2%2C%20label%3D%22kernel%20%3D%2031%22)%0A%20%20%20%20%20%20%20%20axis.set_ylabel(f%22variable%20%7Bchannel%20%2B%201%7D%22)%0A%20%20%20%20axes%5B0%5D.legend(ncol%3D3%2C%20frameon%3DFalse)%0A%20%20%20%20axes%5B-1%5D.set_xlabel(%22time%22)%0A%20%20%20%20fig_filters.suptitle(%22Wide%20per-variable%20kernels%20expose%20slower%20temporal%20structure%22)%0A%20%20%20%20fig_filters.tight_layout()%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20A%20larger%20kernel%20does%20more%20than%20smooth%3A%20when%20its%20weights%20are%20learned%20instead%20of%20fixed%2C%20it%20can%20detect%20a%20long%20motif%20directly.%20Stacking%20blocks%20expands%20the%20reachable%20history%20further.%20For%20stride-one%20convolutions%20with%20kernel%20size%20%24k%24%20and%20%24L%24%20layers%2C%20a%20simple%20upper-level%20calculation%20is%3A%0A%0A%20%20%20%20%24%24R%3D1%2BL(k-1).%24%24%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(np%2C%20plt)%3A%0A%20%20%20%20depths%20%3D%20np.arange(1%2C%2013)%0A%20%20%20%20fig_receptive%2C%20receptive_axis%20%3D%20plt.subplots(figsize%3D(9%2C%204))%0A%20%20%20%20for%20kernel_size%2C%20color%20in%20%5B(3%2C%20%22%2378909c%22)%2C%20(7%2C%20%22%232b78b8%22)%2C%20(31%2C%20%22%23e18719%22)%5D%3A%0A%20%20%20%20%20%20%20%20receptive_axis.plot(%0A%20%20%20%20%20%20%20%20%20%20%20%20depths%2C%0A%20%20%20%20%20%20%20%20%20%20%20%201%20%2B%20depths%20*%20(kernel_size%20-%201)%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20marker%3D%22o%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20label%3Df%22kernel%20%3D%20%7Bkernel_size%7D%22%2C%0A%20%20%20%20%20%20%20%20%20%20%20%20color%3Dcolor%2C%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20receptive_axis.set(%0A%20%20%20%20%20%20%20%20xlabel%3D%22stacked%20convolution%20blocks%22%2C%0A%20%20%20%20%20%20%20%20ylabel%3D%22nominal%20receptive%20field%22%2C%0A%20%20%20%20%20%20%20%20title%3D%22Kernel%20width%20and%20depth%20determine%20reachable%20history%22%2C%0A%20%20%20%20)%0A%20%20%20%20receptive_axis.legend(frameon%3DFalse)%0A%20%20%20%20receptive_axis.grid(alpha%3D0.2)%0A%20%20%20%20fig_receptive.tight_layout()%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20---%0A%0A%20%20%20%20%23%23%20Evaluation%20and%20diagnosis%0A%0A%20%20%20%20Use%20rolling-origin%20backtests.%20Compare%20against%20last-value%2C%20seasonal-naive%2C%20linear%2C%20and%20established%20neural%20baselines.%20Ablate%20context%20length%20and%20kernel%20width%3A%20if%20neither%20changes%20performance%2C%20the%20architecture%20may%20not%20be%20using%20its%20advertised%20history.%0A%0A%20%20%20%20Check%20every%20transformation%20for%20leakage%2C%20especially%20normalization%20across%20the%20full%20series.%20Report%20the%20training%20budget%20as%20well%20as%20accuracy%20because%20ModernTCN's%20value%20proposition%20includes%20convolutional%20efficiency.%0A%0A%20%20%20%20%23%23%20Related%20models%0A%0A%20%20%20%20-%20**Temporal%20Convolutional%20Network%3A**%20the%20causal%2C%20dilated%20predecessor%20and%20a%20clearer%20starting%20point%20for%20receptive%20fields.%0A%20%20%20%20-%20**Chronos%3A**%20a%20pretrained%20alternative%20when%20zero-shot%20transfer%20is%20more%20valuable%20than%20training%20a%20domain%20model.%0A%0A%20%20%20%20%23%23%20Primary%20sources%0A%0A%20%20%20%20-%20%5BModernTCN%20paper%5D(https%3A%2F%2Fopenreview.net%2Fforum%3Fid%3DvpJMJerXHU)%0A%20%20%20%20-%20%5BOfficial%20ModernTCN%20implementation%5D(https%3A%2F%2Fgithub.com%2Fluodhhh%2FModernTCN)%0A%0A%20%20%20%20%23%23%20Practical%20takeaway%0A%0A%20%20%20%20Choose%20ModernTCN%20when%20you%20have%20enough%20domain%20data%20to%20learn%20a%20specialized%20multiscale%20convolutional%20forecaster%20and%20can%20verify%20that%20its%20larger%20receptive%20field%20beats%20simpler%20temporal%20baselines.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0Aif%20__name__%20%3D%3D%20%22__main__%22%3A%0A%20%20%20%20app.run()%0A
6decdf3263e16515dd63c16dad468695