yushuang88 commited on
Commit
30d9258
·
verified ·
1 Parent(s): 07a5280

Add model configuration

Browse files
Files changed (1) hide show
  1. config.json +128 -0
config.json ADDED
@@ -0,0 +1,128 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "FNO2d"
4
+ ],
5
+ "framework": "pytorch",
6
+ "model_name": "FNO",
7
+ "model_type": "fno",
8
+ "source_config": "config/config.yaml",
9
+ "paper": {
10
+ "title": "Fourier Neural Operator for Parametric Partial Differential Equations",
11
+ "arxiv": "2010.08895",
12
+ "experiment": "FNO-2D Navier-Stokes, nu=1e-5, T=20",
13
+ "viscosity": 1e-05,
14
+ "reference_relative_l2": 0.1556,
15
+ "reference_parameter_count": 414517,
16
+ "reference_epoch_seconds_v100": 127.8
17
+ },
18
+ "data": {
19
+ "root": "/public/share/sugonhpcapp01/onestore/onedatasets/FNO_data",
20
+ "file": "NavierStokes_V1e-5_N1200_T20.mat",
21
+ "key": "u",
22
+ "layout": "N,H,W,T",
23
+ "dtype": "float32",
24
+ "expected_shape": [
25
+ 1200,
26
+ 64,
27
+ 64,
28
+ 20
29
+ ],
30
+ "resolution": [
31
+ 64,
32
+ 64
33
+ ],
34
+ "ntrain": 1000,
35
+ "ntest": 200,
36
+ "train_start": 0,
37
+ "test_start": 1000,
38
+ "history": 10,
39
+ "horizon": 10,
40
+ "recording_interval": 1.0,
41
+ "future_times": [
42
+ 11,
43
+ 12,
44
+ 13,
45
+ 14,
46
+ 15,
47
+ 16,
48
+ 17,
49
+ 18,
50
+ 19,
51
+ 20
52
+ ],
53
+ "normalization": "none"
54
+ },
55
+ "model": {
56
+ "name": "FNO2d",
57
+ "input_channels": 10,
58
+ "output_channels": 1,
59
+ "use_grid": true,
60
+ "grid_channels": 2,
61
+ "grid_include_endpoint": false,
62
+ "width": 32,
63
+ "modes1": 12,
64
+ "modes2": 12,
65
+ "num_layers": 4,
66
+ "projection_width": 128,
67
+ "activation": "relu",
68
+ "normalization": "batch_norm",
69
+ "block_order": "relu(batch_norm(spectral_plus_pointwise))",
70
+ "fft_norm": "backward",
71
+ "spectral_init": "scaled_uniform_complex"
72
+ },
73
+ "training": {
74
+ "epochs": 500,
75
+ "batch_size": 20,
76
+ "optimizer": "adam",
77
+ "learning_rate": 0.001,
78
+ "weight_decay": 0.0,
79
+ "scheduler": "step_lr",
80
+ "scheduler_step_size": 100,
81
+ "scheduler_gamma": 0.5,
82
+ "seed": 0,
83
+ "dtype": "float32",
84
+ "amp": false,
85
+ "gradient_clipping": null,
86
+ "ema": false,
87
+ "distributed": "single",
88
+ "num_workers": 0,
89
+ "pin_memory": true,
90
+ "deterministic": true,
91
+ "relative_l2_epsilon": 1e-12,
92
+ "train_rollout_steps": 10,
93
+ "evaluation_rollout_steps": 10,
94
+ "checkpoint_monitor": "train_full_relative_l2",
95
+ "checkpoint_mode": "min",
96
+ "evaluate_test_every_epoch": true
97
+ },
98
+ "inference": {
99
+ "batch_size": 20,
100
+ "seed": 0,
101
+ "dtype": "float32",
102
+ "rollout_steps": 10
103
+ },
104
+ "paths": {
105
+ "checkpoint": "weight/best_model.pth",
106
+ "results_dir": "results",
107
+ "train_history": "results/train_history.json",
108
+ "predictions": "results/predictions.npz",
109
+ "metrics": "results/metrics.json",
110
+ "per_sample_metrics": "results/per_sample_metrics.csv",
111
+ "training_curves": "results/training_curves.png",
112
+ "rollout_figure": "results/sample_000_rollout.png",
113
+ "run_metadata": "results/run_metadata.json",
114
+ "summary": "results/summary.md"
115
+ },
116
+ "assumptions": [
117
+ "The paper does not specify a validation split; the best checkpoint is selected using train full-trajectory relative L2, never test error.",
118
+ "The paper does not specify batch size or seed; batch_size=20 and seed=0 are explicit engineering assumptions.",
119
+ "The paper does not define the exact relative-L2 reduction; ratios are computed per sample and then averaged with epsilon=1e-12.",
120
+ "The paper does not specify the projection hidden width; projection_width=128 is configurable.",
121
+ "Coordinate-grid input is configurable and enabled; the periodic grid excludes the duplicated endpoint.",
122
+ "The block ordering is ReLU(BatchNorm(spectral + pointwise)); the paper states ReLU and batch normalization but not their exact order.",
123
+ "Adam uses weight_decay=0.0 because the paper does not state an additional weight-decay regularizer."
124
+ ],
125
+ "conflicts": [
126
+ "The paper states d_v=32 but reports 414,517 parameters without enough connection details to reproduce both uniquely. Width 32 takes precedence and the actual parameter count must be reported."
127
+ ]
128
+ }