3 BackProp+Regularization
3 BackProp+Regularization
Deep Learning
Forward Propagation
Computing Derivatives
Backward Propagation
Bias/Variance Tradeoff
1
𝑎𝑎1
𝑥𝑥1 = 𝑎𝑎1
0
1
𝑎𝑎2
2
𝑦𝑦� 𝑋𝑋 = 𝑥𝑥 (1) 𝑥𝑥 (2) … 𝑥𝑥 (𝑚𝑚)
𝑥𝑥2 = 𝑎𝑎2
0
𝑎𝑎3
1
𝑎𝑎1
𝑥𝑥3 = 𝑎𝑎3
0
1
𝑎𝑎4
𝑌𝑌 = 𝑦𝑦 1 , 𝑦𝑦 2 , … , 𝑦𝑦 𝑚𝑚
…
1
𝑎𝑎1
𝑥𝑥1 = 0
𝑎𝑎1 1
𝑎𝑎2 1 1 𝑇𝑇 [1] [1] 1 1
𝑥𝑥2 = 𝑎𝑎2
0 𝑎𝑎1
2
𝑦𝑦� 𝑧𝑧4 = 𝑤𝑤4 𝑎𝑎[0] + 𝑏𝑏4 , 𝑎𝑎 4 = 𝑔𝑔[1] (𝑧𝑧1 ) = 𝜎𝜎(𝑧𝑧4 )
1
𝑎𝑎3
𝑥𝑥3 = 𝑎𝑎3
0
1
𝑎𝑎4
𝑥𝑥2 = 𝑎𝑎2
0
1
2
𝑎𝑎1 𝑦𝑦� 1
𝑧𝑧3 = 𝑤𝑤3
1 𝑇𝑇 [1]
𝑎𝑎[0] + 𝑏𝑏3 ,
[1] 1
𝑎𝑎 3 = 𝜎𝜎(𝑧𝑧3 )
𝑎𝑎3 1 1 𝑇𝑇 [1] [1] 1
𝑥𝑥3 = 𝑎𝑎3
0 𝑧𝑧4 = 𝑤𝑤4 𝑎𝑎[0] + 𝑏𝑏4 , 𝑎𝑎 4 = 𝜎𝜎(𝑧𝑧4 )
1
𝑎𝑎4
𝑍𝑍 1 = 𝑊𝑊 1 𝑋𝑋 + 𝑏𝑏 1
𝐴𝐴 1 = 𝜎𝜎(𝑍𝑍 1 )
A[1] = 𝑎𝑎[1](1) 𝑎𝑎[1](2) … 𝑎𝑎[1](𝑚𝑚) 𝑍𝑍 2 = 𝑊𝑊 2
𝐴𝐴 1 + 𝑏𝑏 2
𝐴𝐴 2 = 𝜎𝜎(𝑍𝑍 2 )
Instructor: David C. Anastasiu CSEN 342: Deep Learning 11
Computing Derivatives
𝑔𝑔 𝑘𝑘 𝑧𝑧 𝑘𝑘
𝑎𝑎[𝑘𝑘]
𝜕𝜕𝑎𝑎[𝑘𝑘]
Local gradient
𝜕𝜕𝑎𝑎[𝑘𝑘−1]
𝑑𝑑𝑑𝑑 1 𝑑𝑑𝑑𝑑 1
𝑓𝑓 𝑥𝑥 = 𝑒𝑒 𝑥𝑥 ⇒ = 𝑒𝑒 𝑥𝑥 𝑓𝑓 𝑥𝑥 = ⇒ =− 2
𝑑𝑑𝑑𝑑 𝑥𝑥 𝑑𝑑𝑑𝑑 𝑥𝑥
𝑑𝑑𝑑𝑑 𝑑𝑑𝑑𝑑
𝑓𝑓 𝑥𝑥 = 𝑎𝑎𝑎𝑎 ⇒ = 𝑎𝑎 𝑓𝑓 𝑥𝑥 = 𝑐𝑐 + 𝑥𝑥 ⇒ =1
𝑑𝑑𝑑𝑑 𝑑𝑑𝑑𝑑
Local Upstream
gradient * gradient
1/𝑥𝑥 ′ = −1/𝑥𝑥 2
1
− ∗ 1 = −0.53
1.372
Source: Stanford 231n
A detailed example 1
𝑓𝑓 𝑥𝑥, 𝑤𝑤 =
1 + exp − 𝑤𝑤0 𝑥𝑥0 + 𝑤𝑤1 𝑥𝑥1 + 𝑤𝑤2
𝑑𝑑𝑑𝑑 1 𝑑𝑑𝑑𝑑 1
𝑓𝑓 𝑥𝑥 = 𝑒𝑒 𝑥𝑥 ⇒ = 𝑒𝑒 𝑥𝑥 𝑓𝑓 𝑥𝑥 = ⇒ =− 2
𝑑𝑑𝑑𝑑 𝑥𝑥 𝑑𝑑𝑑𝑑 𝑥𝑥
𝑑𝑑𝑑𝑑 𝑑𝑑𝑑𝑑
𝑓𝑓 𝑥𝑥 = 𝑎𝑎𝑎𝑎 ⇒ = 𝑎𝑎 𝑓𝑓 𝑥𝑥 = 𝑐𝑐 + 𝑥𝑥 ⇒ =1
𝑑𝑑𝑑𝑑 𝑑𝑑𝑑𝑑
Local Upstream
gradient * gradient
𝑑𝑑𝑑𝑑 1 𝑑𝑑𝑑𝑑 1
𝑓𝑓 𝑥𝑥 = 𝑒𝑒 𝑥𝑥 ⇒ = 𝑒𝑒 𝑥𝑥 𝑓𝑓 𝑥𝑥 = ⇒ =− 2
𝑑𝑑𝑑𝑑 𝑥𝑥 𝑑𝑑𝑑𝑑 𝑥𝑥
𝑑𝑑𝑑𝑑 𝑑𝑑𝑑𝑑
𝑓𝑓 𝑥𝑥 = 𝑎𝑎𝑎𝑎 ⇒ = 𝑎𝑎 𝑓𝑓 𝑥𝑥 = 𝑐𝑐 + 𝑥𝑥 ⇒ =1
𝑑𝑑𝑑𝑑 𝑑𝑑𝑑𝑑
Local Upstream
gradient * gradient
𝑑𝑑𝑑𝑑 1 𝑑𝑑𝑑𝑑 1
𝑓𝑓 𝑥𝑥 = 𝑒𝑒 𝑥𝑥 ⇒ = 𝑒𝑒 𝑥𝑥 𝑓𝑓 𝑥𝑥 = ⇒ =− 2
𝑑𝑑𝑑𝑑 𝑥𝑥 𝑑𝑑𝑑𝑑 𝑥𝑥
𝑑𝑑𝑑𝑑 𝑑𝑑𝑑𝑑
𝑓𝑓 𝑥𝑥 = 𝑎𝑎𝑎𝑎 ⇒ = 𝑎𝑎 𝑓𝑓 𝑥𝑥 = 𝑐𝑐 + 𝑥𝑥 ⇒ =1
𝑑𝑑𝑑𝑑 𝑑𝑑𝑑𝑑
Local Upstream
gradient * gradient
𝑑𝑑𝑑𝑑 1 𝑑𝑑𝑑𝑑 1
𝑓𝑓 𝑥𝑥 = 𝑒𝑒 𝑥𝑥 ⇒ = 𝑒𝑒 𝑥𝑥 𝑓𝑓 𝑥𝑥 = ⇒ =− 2
𝑑𝑑𝑑𝑑 𝑥𝑥 𝑑𝑑𝑑𝑑 𝑥𝑥
𝑑𝑑𝑑𝑑 𝑑𝑑𝑑𝑑
𝑓𝑓 𝑥𝑥 = 𝑎𝑎𝑎𝑎 ⇒ = 𝑎𝑎 𝑓𝑓 𝑥𝑥 = 𝑐𝑐 + 𝑥𝑥 ⇒ =1
𝑑𝑑𝑑𝑑 𝑑𝑑𝑑𝑑
Local Upstream
gradient * gradient
𝑑𝑑𝑑𝑑 1 𝑑𝑑𝑑𝑑 1
𝑓𝑓 𝑥𝑥 = 𝑒𝑒 𝑥𝑥 ⇒ = 𝑒𝑒 𝑥𝑥 𝑓𝑓 𝑥𝑥 = ⇒ =− 2
𝑑𝑑𝑑𝑑 𝑥𝑥 𝑑𝑑𝑑𝑑 𝑥𝑥
𝑑𝑑𝑑𝑑 𝑑𝑑𝑑𝑑
𝑓𝑓 𝑥𝑥 = 𝑎𝑎𝑎𝑎 ⇒ = 𝑎𝑎 𝑓𝑓 𝑥𝑥 = 𝑐𝑐 + 𝑥𝑥 ⇒ =1
𝑑𝑑𝑑𝑑 𝑑𝑑𝑑𝑑
Local Upstream
gradient * gradient
𝑑𝑑𝑑𝑑 1 𝑑𝑑𝑑𝑑 1
𝑓𝑓 𝑥𝑥 = 𝑒𝑒 𝑥𝑥 ⇒ = 𝑒𝑒 𝑥𝑥 𝑓𝑓 𝑥𝑥 = ⇒ =− 2
𝑑𝑑𝑑𝑑 𝑥𝑥 𝑑𝑑𝑑𝑑 𝑥𝑥
𝑑𝑑𝑑𝑑 𝑑𝑑𝑑𝑑
𝑓𝑓 𝑥𝑥 = 𝑎𝑎𝑎𝑎 ⇒ = 𝑎𝑎 𝑓𝑓 𝑥𝑥 = 𝑐𝑐 + 𝑥𝑥 ⇒ =1
𝑑𝑑𝑑𝑑 𝑑𝑑𝑑𝑑
𝑑𝑑𝑑𝑑 1 𝑑𝑑𝑑𝑑 1
𝑓𝑓 𝑥𝑥 = 𝑒𝑒 𝑥𝑥 ⇒ = 𝑒𝑒 𝑥𝑥 𝑓𝑓 𝑥𝑥 = ⇒ =− 2
𝑑𝑑𝑑𝑑 𝑥𝑥 𝑑𝑑𝑑𝑑 𝑥𝑥
𝑑𝑑𝑑𝑑 𝑑𝑑𝑑𝑑
𝑓𝑓 𝑥𝑥 = 𝑎𝑎𝑎𝑎 ⇒ = 𝑎𝑎 𝑓𝑓 𝑥𝑥 = 𝑐𝑐 + 𝑥𝑥 ⇒ =1
𝑑𝑑𝑑𝑑 𝑑𝑑𝑑𝑑
𝑑𝑑𝑑𝑑 1 𝑑𝑑𝑑𝑑 1
𝑓𝑓 𝑥𝑥 = 𝑒𝑒 𝑥𝑥 ⇒ = 𝑒𝑒 𝑥𝑥 𝑓𝑓 𝑥𝑥 = ⇒ =− 2
𝑑𝑑𝑑𝑑 𝑥𝑥 𝑑𝑑𝑑𝑑 𝑥𝑥
𝑑𝑑𝑑𝑑 𝑑𝑑𝑑𝑑
𝑓𝑓 𝑥𝑥 = 𝑎𝑎𝑎𝑎 ⇒ = 𝑎𝑎 𝑓𝑓 𝑥𝑥 = 𝑐𝑐 + 𝑥𝑥 ⇒ =1
𝑑𝑑𝑑𝑑 𝑑𝑑𝑑𝑑
-0.60
𝑑𝑑𝑑𝑑 1 𝑑𝑑𝑑𝑑 1
𝑓𝑓 𝑥𝑥 = 𝑒𝑒 𝑥𝑥 ⇒ = 𝑒𝑒 𝑥𝑥 𝑓𝑓 𝑥𝑥 = ⇒ =− 2
𝑑𝑑𝑑𝑑 𝑥𝑥 𝑑𝑑𝑑𝑑 𝑥𝑥
𝑑𝑑𝑑𝑑 𝑑𝑑𝑑𝑑
𝑓𝑓 𝑥𝑥 = 𝑎𝑎𝑎𝑎 ⇒ = 𝑎𝑎 𝑓𝑓 𝑥𝑥 = 𝑐𝑐 + 𝑥𝑥 ⇒ =1
𝑑𝑑𝑑𝑑 𝑑𝑑𝑑𝑑
-0.40
-0.60
𝑑𝑑𝑑𝑑 1 𝑑𝑑𝑑𝑑 1
𝑓𝑓 𝑥𝑥 = 𝑒𝑒 𝑥𝑥 ⇒ = 𝑒𝑒 𝑥𝑥 𝑓𝑓 𝑥𝑥 = ⇒ =− 2
𝑑𝑑𝑑𝑑 𝑥𝑥 𝑑𝑑𝑑𝑑 𝑥𝑥
𝑑𝑑𝑑𝑑 𝑑𝑑𝑑𝑑
0.40 𝑓𝑓 𝑥𝑥 = 𝑎𝑎𝑎𝑎 ⇒ = 𝑎𝑎 𝑓𝑓 𝑥𝑥 = 𝑐𝑐 + 𝑥𝑥 ⇒ =1
𝑑𝑑𝑑𝑑 𝑑𝑑𝑑𝑑
-0.40
-0.60
𝑑𝑑𝑑𝑑 1 𝑑𝑑𝑑𝑑 1
𝑓𝑓 𝑥𝑥 = 𝑒𝑒 𝑥𝑥 ⇒ = 𝑒𝑒 𝑥𝑥 𝑓𝑓 𝑥𝑥 = ⇒ =− 2
𝑑𝑑𝑑𝑑 𝑥𝑥 𝑑𝑑𝑑𝑑 𝑥𝑥
𝑑𝑑𝑑𝑑 𝑑𝑑𝑑𝑑
0.40 𝑓𝑓 𝑥𝑥 = 𝑎𝑎𝑎𝑎 ⇒ = 𝑎𝑎 𝑓𝑓 𝑥𝑥 = 𝑐𝑐 + 𝑥𝑥 ⇒ =1
𝑑𝑑𝑑𝑑 𝑑𝑑𝑑𝑑
-0.40
-0.60
-0.40
-0.60
Sigmoid gate 𝜎𝜎 𝑥𝑥
𝜎𝜎 ′ 𝑥𝑥 = 𝜎𝜎 𝑥𝑥 1 − 𝜎𝜎 𝑥𝑥
𝜎𝜎 1 1 − 𝜎𝜎 1 = 0.73 ∗ 1 − 0.73 = 0.20
𝑏𝑏 1 𝑏𝑏 2
Backward Pass
1 2 𝑇𝑇 2
𝑑𝑑𝑑𝑑 1 = 𝑑𝑑𝑑𝑑 1 𝑎𝑎 0 𝑇𝑇 𝑑𝑑𝑑𝑑 = 𝑊𝑊 𝑑𝑑𝑑𝑑 ∗ 𝑔𝑔 1 ′ (𝑧𝑧 1
) 𝑑𝑑𝑑𝑑 2 = 𝑎𝑎 2 − 𝑦𝑦
𝑑𝑑𝑑𝑑 2 = 𝑑𝑑𝑑𝑑 2 𝑎𝑎 1 𝑇𝑇
𝑑𝑑𝑑𝑑 1 = 𝑑𝑑𝑑𝑑 1
𝑑𝑑𝑑𝑑 2 = 𝑑𝑑𝑑𝑑 2
Update
𝑥𝑥 𝑓𝑓(𝑥𝑥) 𝑧𝑧
1 × 𝑀𝑀 𝜕𝜕𝜕𝜕 𝜕𝜕𝜕𝜕 𝜕𝜕𝜕𝜕 𝜕𝜕𝜕𝜕 1 × 𝑁𝑁
=
𝜕𝜕𝜕𝜕 𝜕𝜕𝜕𝜕 𝜕𝜕𝜕𝜕
𝜕𝜕𝜕𝜕
1 × 𝑀𝑀 1 × 𝑁𝑁𝑁𝑁 × 𝑀𝑀 1 × 𝑁𝑁
Vectorized Gradient Computation
𝑑𝑑𝑧𝑧 [2] = 𝑎𝑎[2] − 𝑦𝑦 𝑑𝑑𝑍𝑍 [2] = 𝐴𝐴[2] − 𝑌𝑌
1
𝑑𝑑𝑊𝑊 [2] = 𝑑𝑑𝑧𝑧 [2] 𝑎𝑎 1 𝑇𝑇 𝑑𝑑𝑊𝑊 [2] = 𝑑𝑑𝑍𝑍 [2] 𝐴𝐴 1 𝑇𝑇
𝑚𝑚
1
𝑑𝑑𝑏𝑏 [2] = 𝑑𝑑𝑧𝑧 [2] 𝑑𝑑𝑏𝑏 [2] = 𝑛𝑛𝑛𝑛. 𝑠𝑠𝑠𝑠𝑠𝑠(𝑑𝑑𝑍𝑍 2 , 𝑎𝑎𝑎𝑎𝑎𝑎𝑎𝑎 = 1, 𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘 = 𝑇𝑇𝑇𝑇𝑇𝑇𝑇𝑇)
𝑚𝑚
𝑑𝑑𝑧𝑧 [1] = 𝑊𝑊 2 𝑇𝑇 𝑑𝑑𝑧𝑧 [2] ∗ 𝑔𝑔[1] ′(z 1 ) 𝑑𝑑𝑍𝑍 [1] = 𝑊𝑊 2 𝑇𝑇 𝑑𝑑𝑍𝑍 [2] ∗ 𝑔𝑔[1] ′(Z 1 )
1
𝑑𝑑𝑊𝑊 [1] = 𝑑𝑑𝑧𝑧 [1] 𝑥𝑥 𝑇𝑇 𝑑𝑑𝑊𝑊 [1] = 𝑑𝑑𝑍𝑍 [1] 𝑋𝑋 𝑇𝑇
𝑚𝑚
1
𝑑𝑑𝑏𝑏 [1] = 𝑑𝑑𝑧𝑧 [1] 𝑑𝑑𝑏𝑏 [1] = 𝑛𝑛𝑛𝑛. 𝑠𝑠𝑠𝑠𝑠𝑠(𝑑𝑑𝑍𝑍 1 , 𝑎𝑎𝑎𝑎𝑎𝑎𝑎𝑎 = 1, 𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘 = 𝑇𝑇𝑇𝑇𝑇𝑇𝑇𝑇)
𝑚𝑚
Andrew Ng
Generalization to L Layers
1
𝑎𝑎1
𝐿𝐿−1
𝑥𝑥1 = 𝑎𝑎1
0
1
𝑎𝑎2
𝑎𝑎1
𝐿𝐿
𝑎𝑎1 𝑦𝑦�
[… ] 𝐿𝐿−1
𝑥𝑥2 = 0
𝑎𝑎2 1
𝑎𝑎3
𝑎𝑎2
𝑥𝑥3 = 𝑎𝑎3
0
1
𝐿𝐿−1
𝑎𝑎3
𝑎𝑎4
𝑑𝑑𝑑𝑑 [𝐿𝐿] = 𝐴𝐴[𝐿𝐿] − 𝑌𝑌
1 𝑇𝑇
𝑑𝑑𝑊𝑊 [𝐿𝐿] = 𝑑𝑑𝑍𝑍 𝐿𝐿 𝐴𝐴 𝐿𝐿
𝑚𝑚
𝑍𝑍 [1] = 𝑊𝑊 [1] 𝑋𝑋 + 𝑏𝑏 [1] 1
𝑑𝑑𝑏𝑏 [𝐿𝐿] = 𝑛𝑛𝑛𝑛. sum(d𝑍𝑍 𝐿𝐿 , 𝑎𝑎𝑎𝑎𝑎𝑎𝑎𝑎 = 1, 𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘 = 𝑇𝑇𝑇𝑇𝑇𝑇𝑇𝑇)
𝐴𝐴[1] = 𝑔𝑔 1 (𝑍𝑍 1 ) 𝑚𝑚 𝑇𝑇 𝐿𝐿
𝑑𝑑𝑑𝑑 [𝐿𝐿−1] = 𝑑𝑑𝑊𝑊 𝐿𝐿 𝑑𝑑𝑍𝑍 𝐿𝐿 𝑔𝑔′ (𝑍𝑍 𝐿𝐿−1 )
𝑍𝑍 [2] = 𝑊𝑊 [2] 𝐴𝐴[1] + 𝑏𝑏 [2]
…
𝑇𝑇 1
𝐴𝐴[2] = 𝑔𝑔 2 (𝑍𝑍 2 ) 𝑑𝑑𝑑𝑑 [1] = 𝑑𝑑𝑊𝑊 𝐿𝐿 𝑑𝑑𝑍𝑍 2 𝑔𝑔′ (𝑍𝑍 1 )
1 𝑇𝑇
…
𝑑𝑑𝑊𝑊 = 𝑑𝑑𝑍𝑍 1 𝐴𝐴 1
[1]
𝐴𝐴[𝐿𝐿] = 𝑔𝑔 𝐿𝐿 𝑍𝑍 𝐿𝐿
= 𝑌𝑌� 𝑚𝑚
1
𝑑𝑑𝑏𝑏 = 𝑛𝑛𝑛𝑛. sum(d𝑍𝑍 1 , 𝑎𝑎𝑎𝑎𝑎𝑎𝑎𝑎 = 1, 𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘𝑘 = 𝑇𝑇𝑇𝑇𝑇𝑇𝑇𝑇)
[1]
𝑚𝑚
Bias/Variance Tradeoff
𝑥𝑥 𝑥𝑥 𝑥𝑥
Overfitting: A model that is too complex (too many features) may learn a
function that fits our training data so well that it fails to generalize to data
outside out training set (predict 𝑦𝑦 on new samples).
Instructor: David C. Anastasiu CSEN 342: Deep Learning 38
Overfitting: Logistic Regression Example
𝑥𝑥2 𝑥𝑥2 𝑥𝑥2
Overfitting: A model that is too complex (too many features) may learn a
function that fits our training data so well that it fails to generalize to data
outside out training set (predict 𝑦𝑦 on new samples).
Instructor: David C. Anastasiu CSEN 342: Deep Learning 39
Regularization Intuition
𝑦𝑦
𝑥𝑥
• Data augmentation
• Early stopping
• Dropout
Svetlana Lazebnik
Data augmentation
• Introduce transformations not adequately sampled in the training data
• Geometric: flipping, rotation, shearing, multiple crops
• Photometric: color transformations
Image source
Svetlana Lazebnik
Data augmentation
• Introduce transformations not adequately sampled in the training data
• Geometric: flipping, rotation, shearing, multiple crops
• Photometric: color transformations
• Other: add noise, compression artifacts, lens distortions, etc.
Image source
Svetlana Lazebnik
Data augmentation
• Introduce transformations not adequately sampled in the training data
• Limited only by your imagination and time/memory constraints!
• Avoid introducing artifacts
Image source
Svetlana Lazebnik
Data augmentation
• Introduce transformations not adequately sampled in the training data
• Limited only by your imagination and time/memory constraints!
• Avoid introducing artifacts
• Automatic augmentation strategies: AutoAugment, RandAugment
Svetlana Lazebnik
Beyond Training Error
Iteration Iteration
Stop training the model when accuracy on the validation set decreases
Or train for a long time, but always keep track of the model snapshot
that worked best on val
In common use:
L2 regularization (Weight decay)
L1 regularization
Elastic net (L1 + L2)