import%20marimo%0A%0A__generated_with%20%3D%20%220.23.16%22%0Aapp%20%3D%20marimo.App()%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20marimo%20as%20mo%0A%0A%20%20%20%20return%20(mo%2C)%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20METADATA%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%22id%22%3A%20%22tcn%22%2C%0A%20%20%20%20%20%20%20%20%22name%22%3A%20%22Temporal%20Convolutional%20Network%22%2C%0A%20%20%20%20%20%20%20%20%22types%22%3A%20%5B%22architecture%22%5D%2C%0A%20%20%20%20%20%20%20%20%22families%22%3A%20%5B%22neural_networks%22%2C%20%22convolutional%22%2C%20%22sequential%22%5D%2C%0A%20%20%20%20%20%20%20%20%22tasks%22%3A%20%5B%22classification%22%2C%20%22regression%22%2C%20%22forecasting%22%2C%20%22representation%22%5D%2C%0A%20%20%20%20%20%20%20%20%22data%22%3A%20%5B%22time_series%22%2C%20%22sequences%22%2C%20%22audio%22%5D%2C%0A%20%20%20%20%20%20%20%20%22learning%22%3A%20%5B%22supervised%22%2C%20%22self_supervised%22%5D%2C%0A%20%20%20%20%20%20%20%20%22capacity%22%3A%20%22parametric%22%2C%0A%20%20%20%20%20%20%20%20%22mechanisms%22%3A%20%5B%22causal_convolution%22%2C%20%22dilation%22%2C%20%22backpropagation%22%5D%2C%0A%20%20%20%20%20%20%20%20%22properties%22%3A%20%5B%22nonlinear%22%2C%20%22representation_learning%22%5D%2C%0A%20%20%20%20%20%20%20%20%22constraints%22%3A%20%5B%22requires_large_data%22%2C%20%22requires_scaling%22%2C%20%22sensitive_to_tuning%22%5D%2C%0A%20%20%20%20%20%20%20%20%22difficulty%22%3A%20%22intermediate%22%2C%0A%20%20%20%20%20%20%20%20%22status%22%3A%20%22complete%22%2C%0A%20%20%20%20%20%20%20%20%22explainability%22%3A%20%22low%22%2C%0A%20%20%20%20%20%20%20%20%22training_cost%22%3A%20%22high%22%2C%0A%20%20%20%20%20%20%20%20%22inference_cost%22%3A%20%22medium%22%2C%0A%20%20%20%20%20%20%20%20%22data_appetite%22%3A%20%22high%22%2C%0A%20%20%20%20%7D%0A%0A%20%20%20%20mo.md(f%22%23%20%7BMETADATA%5B'name'%5D%7D%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20In%20one%20sentence%0A%0A%20%20%20%20A%20TCN%20uses%20causal%2C%20usually%20dilated%20one-dimensional%20convolutions%20to%20model%20sequences%20with%20a%20large%20and%20controllable%20history.%0A%0A%20%20%20%20%23%23%20Mental%20model%0A%0A%20%20%20%20Instead%20of%20reading%20a%20sequence%20one%20step%20at%20a%20time%2C%20a%20TCN%20slides%20pattern%20detectors%20across%20many%20positions%20in%20parallel.%20Deeper%20layers%20combine%20short%20motifs%20into%20longer%20patterns.%0A%0A%20%20%20%20Dilation%20creates%20gaps%20between%20sampled%20positions.%20A%20layer%20with%20dilation%201%20sees%20nearby%20values%3B%20later%20layers%20with%20dilation%202%2C%204%2C%20and%208%20reach%20farther%20back%20without%20needing%20huge%20kernels.%0A%0A%20%20%20%20%23%23%20Why%20it%20belongs%20to%20several%20categories%0A%0A%20%20%20%20TCN%20is%20a%20neural-network%20architecture%20because%20it%20learns%20layered%20representations%3B%20convolutional%20because%20its%20main%20operation%20is%201D%20convolution%3B%20sequential%20because%20it%20models%20ordered%20data%3B%20and%20compatible%20with%20supervised%20or%20self-supervised%20objectives.%0A%0A%20%20%20%20%23%23%20Input%20and%20output%0A%0A%20%20%20%20-%20**Input%3A**%20sequences%20shaped%20as%20time%20steps%20by%20channels%20or%20features.%0A%20%20%20%20-%20**Output%3A**%20one%20prediction%20per%20sequence%2C%20one%20per%20time%20step%2C%20or%20a%20future%20horizon.%0A%20%20%20%20-%20**Learns%3A**%20temporal%20convolution%20filters%20and%20usually%20residual%20transformations.%0A%0A%20%20%20%20%23%23%20Core%20ingredients%0A%0A%20%20%20%20**Causal%20convolution**%20%E2%80%94%20the%20output%20at%20time%20%24t%24%20depends%20only%20on%20time%20%24t%24%20and%20earlier%20positions.%20Padding%20must%20be%20designed%20carefully%3B%20ordinary%20symmetric%20padding%20can%20leak%20future%20information.%0A%0A%20%20%20%20**Dilation**%20%E2%80%94%20dilated%20kernels%20expand%20context%20efficiently.%20For%20kernel%20size%20%24k%24%20and%20dilations%20%24d_l%24%2C%20the%20receptive%20field%20of%20a%20simple%20stack%20is%3A%0A%0A%20%20%20%20%24%24R%20%3D%201%20%2B%20%5Csum_l%20(k-1)d_l%24%24%0A%0A%20%20%20%20**Residual%20blocks**%20%E2%80%94%20skip%20connections%20help%20gradients%20and%20let%20deeper%20stacks%20refine%20rather%20than%20completely%20replace%20representations.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20%23%23%20A%20practical%20example%0A%0A%20%20%20%20For%20machine-vibration%20monitoring%2C%20early%20filters%20can%20detect%20short%20oscillations%20while%20dilated%20layers%20connect%20those%20motifs%20across%20longer%20operating%20cycles.%20The%20architecture%20can%20classify%20a%20window%2C%20predict%20future%20readings%2C%20or%20flag%20unusual%20patterns.%0A%0A%20%20%20%20%23%23%20When%20to%20use%20it%0A%0A%20%20%20%20-%20Local%20motifs%20at%20multiple%20temporal%20scales%20are%20plausible.%0A%20%20%20%20-%20Predictions%20must%20be%20causal.%0A%20%20%20%20-%20Parallel%20training%20matters.%0A%20%20%20%20-%20You%20want%20explicit%20control%20over%20maximum%20context.%0A%20%20%20%20-%20Sequences%20are%20long%20enough%20to%20challenge%20simple%20dense%20models%20but%20attention%20is%20unnecessary%20or%20costly.%0A%0A%20%20%20%20%23%23%20When%20to%20avoid%20it%0A%0A%20%20%20%20-%20Relevant%20context%20is%20longer%20than%20the%20designed%20receptive%20field.%0A%20%20%20%20-%20Irregular%20event%20timing%20is%20not%20represented%20properly.%0A%20%20%20%20-%20Direct%20content-based%20interaction%20between%20arbitrary%20positions%20is%20central.%0A%20%20%20%20-%20Data%20is%20too%20small%20to%20justify%20a%20learned%20sequence%20architecture.%0A%20%20%20%20-%20A%20naive%20seasonal%20or%20classical%20model%20already%20solves%20the%20forecasting%20task.%0A%0A%20%20%20%20%23%23%20Data%20preparation%0A%0A%20%20%20%20Split%20by%20time%20or%20entity%20before%20creating%20overlapping%20windows.%20Fit%20scaling%20only%20on%20training%20periods.%20Include%20masks%20or%20elapsed-time%20features%20for%20irregular%20sampling.%20Confirm%20that%20every%20feature%20in%20a%20window%20would%20exist%20at%20prediction%20time.%0A%0A%20%20%20%20%23%23%20Important%20controls%0A%0A%20%20%20%20%7C%20Control%20%7C%20Role%20%7C%0A%20%20%20%20%7C---------%7C------%7C%0A%20%20%20%20%7C%20Kernel%20size%20%7C%20Local%20pattern%20width%20%7C%0A%20%20%20%20%7C%20Dilation%20schedule%20%7C%20How%20quickly%20history%20expands%20%7C%0A%20%20%20%20%7C%20Number%20of%20blocks%2Fchannels%20%7C%20Capacity%20%7C%0A%20%20%20%20%7C%20Dropout%20and%20weight%20decay%20%7C%20Regularization%20%7C%0A%20%20%20%20%7C%20Receptive%20field%20%7C%20Must%20cover%20plausible%20dependencies%20%7C%0A%0A%20%20%20%20---%0A%0A%20%20%20%20%23%23%20Notebook%20%E2%80%94%20seeing%20the%20receptive%20field%0A%0A%20%20%20%20**Question%3A**%20how%20do%20causal%20dilation%20and%20depth%20decide%20which%20history%20can%20affect%20a%20prediction%3F%0A%0A%20%20%20%20This%20notebook%20isolates%20the%20architecture's%20connectivity.%20It%20does%20not%20train%20a%20large%20network.%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_()%3A%0A%20%20%20%20import%20numpy%20as%20np%0A%20%20%20%20import%20matplotlib.pyplot%20as%20plt%0A%0A%20%20%20%20def%20receptive_positions(output_position%2C%20kernel_size%2C%20dilations)%3A%0A%20%20%20%20%20%20%20%20_positions%20%3D%20%7Boutput_position%7D%0A%20%20%20%20%20%20%20%20history%20%3D%20%5B_positions%5D%0A%20%20%20%20%20%20%20%20for%20dilation%20in%20reversed(dilations)%3A%0A%20%20%20%20%20%20%20%20%20%20%20%20_positions%20%3D%20%7B%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20position%20-%20offset%20*%20dilation%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20for%20position%20in%20_positions%0A%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20%20for%20offset%20in%20range(kernel_size)%0A%20%20%20%20%20%20%20%20%20%20%20%20%7D%0A%20%20%20%20%20%20%20%20%20%20%20%20history.append(_positions)%0A%20%20%20%20%20%20%20%20return%20list(reversed(history))%0A%0A%20%20%20%20return%20plt%2C%20receptive_positions%0A%0A%0A%40app.cell%0Adef%20_(plt%2C%20receptive_positions)%3A%0A%20%20%20%20kernel_size%20%3D%203%0A%20%20%20%20dilations%20%3D%20%5B1%2C%202%2C%204%2C%208%5D%0A%20%20%20%20history%20%3D%20receptive_positions(output_position%3D30%2C%20kernel_size%3Dkernel_size%2C%20dilations%3Ddilations)%0A%20%20%20%20fig%2C%20ax%20%3D%20plt.subplots(figsize%3D(11%2C%204))%0A%20%20%20%20for%20layer%2C%20_positions%20in%20enumerate(history)%3A%0A%20%20%20%20%20%20%20%20ax.scatter(sorted(_positions)%2C%20%5Blayer%5D%20*%20len(_positions)%2C%20s%3D35)%0A%20%20%20%20ax.set(%0A%20%20%20%20%20%20%20%20yticks%3Drange(len(history))%2C%0A%20%20%20%20%20%20%20%20yticklabels%3D%5Bf%22layer%20%7Bi%7D%22%20for%20i%20in%20range(len(history))%5D%2C%0A%20%20%20%20)%0A%20%20%20%20ax.set(xlabel%3D%22sequence%20position%22%2C%20title%3D%22Positions%20connected%20to%20output%20at%20t%3D30%22)%0A%20%20%20%20ax.invert_yaxis()%0A%20%20%20%20return%0A%0A%0A%40app.cell%0Adef%20_(receptive_positions)%3A%0A%20%20%20%20for%20schedule%20in%20(%5B1%2C%201%2C%201%2C%201%5D%2C%20%5B1%2C%202%2C%204%2C%208%5D%2C%20%5B1%2C%202%2C%204%2C%208%2C%2016%5D)%3A%0A%20%20%20%20%20%20%20%20_positions%20%3D%20receptive_positions(100%2C%20kernel_size%3D3%2C%20dilations%3Dschedule)%5B0%5D%0A%20%20%20%20%20%20%20%20print(%0A%20%20%20%20%20%20%20%20%20%20%20%20f%22dilations%3D%7Bschedule%7D%3A%20receptive%20field%20spans%20%22%0A%20%20%20%20%20%20%20%20%20%20%20%20f%22%7Bmax(_positions)%20-%20min(_positions)%20%2B%201%7D%20steps%22%0A%20%20%20%20%20%20%20%20)%0A%20%20%20%20return%0A%0A%0A%40app.cell(hide_code%3DTrue)%0Adef%20_(mo)%3A%0A%20%20%20%20mo.md(r%22%22%22%0A%20%20%20%20---%0A%0A%20%20%20%20%23%23%20Evaluation%20and%20diagnosis%0A%0A%20%20%20%20Compare%20against%20last-value%2C%20seasonal%2C%20linear%2C%20and%20tree-based%20baselines%20where%20appropriate.%20Evaluate%20across%20multiple%20future%20windows%20and%20regimes.%20Ablate%20context%20length%3A%20if%20performance%20does%20not%20change%2C%20the%20large%20receptive%20field%20may%20not%20be%20useful.%0A%0A%20%20%20%20Check%20boundary%20padding%2C%20latency%20for%20streaming%20use%2C%20and%20whether%20predictions%20shift%20when%20unavailable%20future%20values%20are%20deliberately%20perturbed.%20That%20last%20test%20can%20expose%20leakage.%0A%0A%20%20%20%20%23%23%20Cost%20profile%0A%0A%20%20%20%20Training%20parallelizes%20across%20sequence%20positions%20better%20than%20recurrent%20networks.%20Inference%20can%20be%20efficient%2C%20especially%20with%20caching%2C%20but%20naive%20recomputation%20over%20long%20windows%20wastes%20work.%0A%0A%20%20%20%20%23%23%20Related%20models%0A%0A%20%20%20%20-%20CNNs%20share%20convolutional%20filters%20but%20usually%20target%20spatial%20rather%20than%20temporal%20structure.%0A%20%20%20%20-%20RNN%2C%20LSTM%2C%20and%20GRU%20compress%20history%20into%20recurrent%20state.%0A%20%20%20%20-%20**Transformer**%20uses%20attention%20for%20flexible%20position-to-position%20interaction.%0A%20%20%20%20-%20State-space%20models%20offer%20another%20route%20to%20long%20efficient%20context.%0A%0A%20%20%20%20%23%23%20Practical%20takeaway%0A%0A%20%20%20%20Choose%20a%20TCN%20when%20%22patterns%20across%20known%20temporal%20scales%22%20is%20a%20better%20description%20of%20the%20problem%20than%20%22every%20position%20may%20need%20to%20look%20directly%20at%20every%20other%20position.%22%0A%20%20%20%20%22%22%22)%0A%20%20%20%20return%0A%0A%0Aif%20__name__%20%3D%3D%20%22__main__%22%3A%0A%20%20%20%20app.run()%0A
d266eb8120188611c63c374fb96d1d7d