diff --git a/assets/x8.xml b/assets/x8.xml
index 3368f7b..67442de 100644
--- a/assets/x8.xml
+++ b/assets/x8.xml
@@ -68,12 +68,12 @@
0
-0.18
- 0.8
- 0.5
- 0.1
- 8.0
- 0.85
- 0
+ 1.8
+ 1.5
+ .01
+ 4.0
+ 1.15
+ 360
NONE
0
@@ -83,12 +83,12 @@
-85.1
-0.18
- 0.8
- 0.5
- 0.1
- 4.0
+ 1.8
+ 1.5
+ 0.01
+ 2.0
0.95
- 0.0
+ 360
LEFT
0
@@ -98,12 +98,12 @@
85.1
-0.18
- 0.8
- 0.5
- 0.1
- 4.0
+ 1.8
+ 1.5
+ 0.01
+ 2.0
0.95
- 0.0
+ 360
RIGHT
0
@@ -111,16 +111,6 @@
-
- 0.0
- 0.0
- 0.0
-
-
- 0.0
- 0.0
- 0.0
-
0.0
@@ -145,7 +135,7 @@
attitude/heading-true-rad
- 0.003
+ 0.0
@@ -264,7 +254,7 @@
Alpha independent lift coefficient
aero/qbar-area
- 0.0867
+ 0.04
@@ -273,7 +263,7 @@
aero/qbar-area
aero/alpha-rad
- 4.0203
+ 2.0203
@@ -468,12 +458,12 @@
- Pitch moment due to alpha\
+ Pitch momet due to alpha\
aero/qbar-area
metrics/cbarw-ft
aero/alpha-rad
- -0.4629
+ -0.6629
@@ -494,7 +484,7 @@
aero/qbar-area
metrics/cbarw-ft
fcs/elevator-pos-rad
- -0.2292
+ -1.1220
diff --git a/data/cross_entropy/stats.txt b/data/cross_entropy/stats.txt
new file mode 100644
index 0000000..fb36156
--- /dev/null
+++ b/data/cross_entropy/stats.txt
@@ -0,0 +1,126 @@
+Generation #1:
+Best, Median, and Worst Learner: (142, 35, 157)
+Best, Median, and Worst Reward: (2982.999081533402, 972.1300675626844, 588.3324620518833)
+
+
+
+Generation #2:
+Best, Median, and Worst Learner: (80, 49, 173)
+Best, Median, and Worst Reward: (2992.656533884816, 978.6449913680553, 594.0445281118155)
+
+
+
+Generation #3:
+Best, Median, and Worst Learner: (178, 56, 16)
+Best, Median, and Worst Reward: (2993.3571543405997, 1445.3874917849898, 604.7305159419775)
+
+
+
+Generation #4:
+Best, Median, and Worst Learner: (84, 193, 37)
+Best, Median, and Worst Reward: (2990.2669111620635, 1959.0998122617602, 602.2353224325925)
+
+
+
+Generation #5:
+Best, Median, and Worst Learner: (70, 185, 88)
+Best, Median, and Worst Reward: (2992.6985498089343, 1972.4182319762185, 653.0758294676198)
+
+
+
+Generation #6:
+Best, Median, and Worst Learner: (103, 70, 184)
+Best, Median, and Worst Reward: (3993.390440976944, 1999.3610292344797, 937.2859918950126)
+
+
+
+Generation #7: (good)
+Best, Median, and Worst Learner: (162, 149, 44)
+Best, Median, and Worst Reward: (3989.6084612680133, 2981.2333327531815, 966.7899975357577)
+
+
+
+Generation #8:
+Best, Median, and Worst Learner: (188, 45, 105)
+Best, Median, and Worst Reward: (2992.4165739994496, 2983.244331255555, 978.5750615522265)
+
+
+
+Generation #9:
+Best, Median, and Worst Learner: (78, 168, 71)
+Best, Median, and Worst Reward: (2993.2526850610448, 2987.396712552756, 968.8540844265372)
+
+
+
+Generation #10:
+Best, Median, and Worst Learner: (91, 177, 88)
+Best, Median, and Worst Reward: (3993.3897130363766, 2988.8644929472357, 974.3769715242088)
+
+
+
+Generation #11:
+Best, Median, and Worst Learner: (89, 154, 52)
+Best, Median, and Worst Reward: (2992.881334366277, 2988.9321229401976, 962.7465795613825)
+
+
+
+Generation #12:
+Best, Median, and Worst Learner: (27, 150, 156)
+Best, Median, and Worst Reward: (3993.390440976944, 2989.2162777222693, 968.5438468381763)
+
+
+
+Generation #13:
+Best, Median, and Worst Learner: (93, 167, 13)
+Best, Median, and Worst Reward: (3993.379415991084, 2989.711838164367, 969.5627298392355)
+
+
+
+Generation #14:
+Best, Median, and Worst Learner: (67, 170, 181)
+Best, Median, and Worst Reward: (3984.084361743182, 2990.3426594454795, 980.2484985329211)
+
+
+
+Generation #15:
+Best, Median, and Worst Learner: (199, 77, 112)
+Best, Median, and Worst Reward: (2993.251348318532, 2989.871223669499, 959.5475498940796)
+
+
+
+Generation #16:
+Best, Median, and Worst Learner: (38, 47, 102)
+Best, Median, and Worst Reward: (2993.230045834556, 2991.1996988034807, 977.2491851244122)
+
+
+
+Generation #17:
+Best, Median, and Worst Learner: (80, 14, 37)
+Best, Median, and Worst Reward: (2993.272437831387, 2991.9086525086313, 980.539373550564)
+
+
+
+Generation #18:
+Best, Median, and Worst Learner: (133, 195, 32)
+Best, Median, and Worst Reward: (2993.77225545235, 2992.6561779286712, 985.9263532422483)
+
+
+
+Generation #19:
+Best, Median, and Worst Learner: (84, 31, 7)
+Best, Median, and Worst Reward: (2993.7730519939214, 2992.7406542636454, 979.8817010223866)
+
+
+
+Generation #20:
+Best, Median, and Worst Learner: (79, 100, 102)
+Best, Median, and Worst Reward: (2993.7877100314945, 2992.7477019503713, 987.0250741429627)
+
+
+
+Generation #21:
+Best, Median, and Worst Learner: (119, 82, 108)
+Best, Median, and Worst Reward: (2993.3186951242387, 2992.9422285705805, 982.8713495023549)
+
+
+
diff --git a/data/ppo/all_data.pkl b/data/ppo/all_data.pkl
new file mode 100644
index 0000000..51bc33e
Binary files /dev/null and b/data/ppo/all_data.pkl differ
diff --git a/data/ppo/policies/learner#37.pth b/data/ppo/policies/learner#37.pth
new file mode 100644
index 0000000..86d49e5
Binary files /dev/null and b/data/ppo/policies/learner#37.pth differ
diff --git a/data/ppo/policies/learner#56.pth b/data/ppo/policies/learner#56.pth
new file mode 100644
index 0000000..bae2083
Binary files /dev/null and b/data/ppo/policies/learner#56.pth differ
diff --git a/data/ppo/stats.txt b/data/ppo/stats.txt
new file mode 100644
index 0000000..09608d3
--- /dev/null
+++ b/data/ppo/stats.txt
@@ -0,0 +1,744 @@
+Policy Number #0:
+ Average Time Steps: 64.346
+ Average Reward: tensor(10.2306)
+ Best Reward: tensor(185.7999)
+ Worst Reward: tensor(-19.8913)
+ Deterministic Reward Evaluation: tensor(73.3813)
+ Time: 11m 18
+
+Policy Number #1:
+ Average Time Steps: 71.666
+ Average Reward: tensor(17.5971)
+ Best Reward: tensor(231.1814)
+ Worst Reward: tensor(-20.6358)
+ Deterministic Reward Evaluation: tensor(82.1320)
+ Time: 12m 33
+
+Policy Number #2:
+ Average Time Steps: 69.852
+ Average Reward: tensor(15.1622)
+ Best Reward: tensor(179.2707)
+ Worst Reward: tensor(-21.3903)
+ Deterministic Reward Evaluation: tensor(56.0810)
+ Time: 12m 9
+
+Policy Number #3:
+ Average Time Steps: 69.142
+ Average Reward: tensor(11.4814)
+ Best Reward: tensor(174.7796)
+ Worst Reward: tensor(-21.0693)
+ Deterministic Reward Evaluation: tensor(153.0145)
+ Time: 12m 4
+
+Policy Number #4:
+ Average Time Steps: 78.09
+ Average Reward: tensor(17.9504)
+ Best Reward: tensor(236.0611)
+ Worst Reward: tensor(-19.4090)
+ Deterministic Reward Evaluation: tensor(79.9471)
+ Time: 13m 47
+
+Policy Number #5:
+ Average Time Steps: 84.284
+ Average Reward: tensor(19.2010)
+ Best Reward: tensor(232.2858)
+ Worst Reward: tensor(-25.4860)
+ Deterministic Reward Evaluation: tensor(69.4026)
+ Time: 14m 36
+
+Policy Number #6:
+ Average Time Steps: 87.118
+ Average Reward: tensor(18.3251)
+ Best Reward: tensor(189.0960)
+ Worst Reward: tensor(-33.6494)
+ Deterministic Reward Evaluation: tensor(89.9409)
+ Time: 15m 10
+
+Policy Number #7:
+ Average Time Steps: 89.856
+ Average Reward: tensor(20.8215)
+ Best Reward: tensor(230.0491)
+ Worst Reward: tensor(-27.7497)
+ Deterministic Reward Evaluation: tensor(66.6348)
+ Time: 15m 38
+
+Policy Number #8:
+ Average Time Steps: 93.718
+ Average Reward: tensor(23.6071)
+ Best Reward: tensor(185.6656)
+ Worst Reward: tensor(-35.7423)
+ Deterministic Reward Evaluation: tensor(91.8381)
+ Time: 16m 12
+
+Policy Number #9:
+ Average Time Steps: 98.414
+ Average Reward: tensor(28.2783)
+ Best Reward: tensor(193.8371)
+ Worst Reward: tensor(-32.6535)
+ Deterministic Reward Evaluation: tensor(74.0798)
+ Time: 17m 58
+
+Policy Number #10:
+ Average Time Steps: 99.896
+ Average Reward: tensor(27.8321)
+ Best Reward: tensor(187.7788)
+ Worst Reward: tensor(-36.7093)
+ Deterministic Reward Evaluation: tensor(154.9915)
+ Time: 17m 42
+
+Policy Number #11:
+ Average Time Steps: 100.044
+ Average Reward: tensor(29.7991)
+ Best Reward: tensor(191.8518)
+ Worst Reward: tensor(-37.2897)
+ Deterministic Reward Evaluation: tensor(80.5144)
+ Time: 17m 51
+
+Policy Number #12:
+ Average Time Steps: 102.034
+ Average Reward: tensor(30.3864)
+ Best Reward: tensor(183.3815)
+ Worst Reward: tensor(-39.6892)
+ Deterministic Reward Evaluation: tensor(169.1824)
+ Time: 18m 14
+
+Policy Number #13:
+ Average Time Steps: 104.628
+ Average Reward: tensor(34.1632)
+ Best Reward: tensor(251.1391)
+ Worst Reward: tensor(-33.9023)
+ Deterministic Reward Evaluation: tensor(86.0289)
+ Time: 18m 46
+
+Policy Number #14:
+ Average Time Steps: 110.006
+ Average Reward: tensor(42.6081)
+ Best Reward: tensor(191.4285)
+ Worst Reward: tensor(-31.7138)
+ Deterministic Reward Evaluation: tensor(175.6570)
+ Time: 19m 16
+
+Policy Number #15:
+ Average Time Steps: 109.534
+ Average Reward: tensor(40.7986)
+ Best Reward: tensor(234.2542)
+ Worst Reward: tensor(-37.6262)
+ Deterministic Reward Evaluation: tensor(90.3845)
+ Time: 19m 14
+
+Policy Number #16:
+ Average Time Steps: 110.512
+ Average Reward: tensor(44.6231)
+ Best Reward: tensor(191.1996)
+ Worst Reward: tensor(-34.8936)
+ Deterministic Reward Evaluation: tensor(175.0175)
+ Time: 19m 28
+
+Policy Number #17:
+ Average Time Steps: 114.738
+ Average Reward: tensor(49.3819)
+ Best Reward: tensor(244.9445)
+ Worst Reward: tensor(-37.5522)
+ Deterministic Reward Evaluation: tensor(92.9966)
+ Time: 20m 6
+
+Policy Number #18:
+ Average Time Steps: 116.088
+ Average Reward: tensor(51.4508)
+ Best Reward: tensor(191.6422)
+ Worst Reward: tensor(-30.1739)
+ Deterministic Reward Evaluation: tensor(177.4778)
+ Time: 20m 43
+
+Policy Number #19:
+ Average Time Steps: 114.136
+ Average Reward: tensor(49.0583)
+ Best Reward: tensor(257.6959)
+ Worst Reward: tensor(-36.2068)
+ Deterministic Reward Evaluation: tensor(94.4184)
+ Time: 20m 34
+
+Policy Number #20:
+ Average Time Steps: 112.668
+ Average Reward: tensor(48.4769)
+ Best Reward: tensor(194.9408)
+ Worst Reward: tensor(-27.5304)
+ Deterministic Reward Evaluation: tensor(183.7022)
+ Time: 20m 10
+
+Policy Number #21:
+ Average Time Steps: 113.768
+ Average Reward: tensor(49.4974)
+ Best Reward: tensor(261.3761)
+ Worst Reward: tensor(-23.3093)
+ Deterministic Reward Evaluation: tensor(178.9386)
+ Time: 20m 13
+
+Policy Number #22:
+ Average Time Steps: 114.216
+ Average Reward: tensor(48.2928)
+ Best Reward: tensor(260.9836)
+ Worst Reward: tensor(-22.7130)
+ Deterministic Reward Evaluation: tensor(190.6056)
+ Time: 20m 13
+
+Policy Number #23:
+ Average Time Steps: 119.686
+ Average Reward: tensor(54.0260)
+ Best Reward: tensor(260.8665)
+ Worst Reward: tensor(-25.2288)
+ Deterministic Reward Evaluation: tensor(189.8301)
+ Time: 20m 56
+
+Policy Number #24:
+ Average Time Steps: 126.428
+ Average Reward: tensor(66.5046)
+ Best Reward: tensor(265.2766)
+ Worst Reward: tensor(-23.4465)
+ Deterministic Reward Evaluation: tensor(255.4890)
+ Time: 22m 21
+
+Policy Number #25:
+ Average Time Steps: 121.938
+ Average Reward: tensor(60.5520)
+ Best Reward: tensor(266.0904)
+ Worst Reward: tensor(-25.7212)
+ Deterministic Reward Evaluation: tensor(190.3151)
+ Time: 21m 11
+
+Policy Number #26:
+ Average Time Steps: 128.098
+ Average Reward: tensor(68.2748)
+ Best Reward: tensor(265.3528)
+ Worst Reward: tensor(-24.7772)
+ Deterministic Reward Evaluation: tensor(255.7828)
+ Time: 22m 45
+
+Policy Number #27:
+ Average Time Steps: 124.944
+ Average Reward: tensor(58.9219)
+ Best Reward: tensor(268.6259)
+ Worst Reward: tensor(-24.3695)
+ Deterministic Reward Evaluation: tensor(256.2360)
+ Time: 21m 59
+
+Policy Number #28:
+ Average Time Steps: 127.924
+ Average Reward: tensor(65.5043)
+ Best Reward: tensor(267.5689)
+ Worst Reward: tensor(-25.4167)
+ Deterministic Reward Evaluation: tensor(193.2034)
+ Time: 22m 9
+
+Policy Number #29:
+ Average Time Steps: 135.126
+ Average Reward: tensor(71.1464)
+ Best Reward: tensor(265.4519)
+ Worst Reward: tensor(-24.5174)
+ Deterministic Reward Evaluation: tensor(236.9192)
+ Time: 23m 37
+
+Policy Number #30:
+ Average Time Steps: 128.228
+ Average Reward: tensor(66.2589)
+ Best Reward: tensor(265.2388)
+ Worst Reward: tensor(-30.4472)
+ Deterministic Reward Evaluation: tensor(193.2180)
+ Time: 22m 23
+
+Policy Number #31:
+ Average Time Steps: 128.792
+ Average Reward: tensor(67.7751)
+ Best Reward: tensor(266.7886)
+ Worst Reward: tensor(-25.1603)
+ Deterministic Reward Evaluation: tensor(257.2142)
+ Time: 22m 15
+
+Policy Number #32:
+ Average Time Steps: 121.346
+ Average Reward: tensor(57.0712)
+ Best Reward: tensor(271.5403)
+ Worst Reward: tensor(-27.3279)
+ Deterministic Reward Evaluation: tensor(193.0520)
+ Time: 20m 42
+
+Policy Number #33:
+ Average Time Steps: 127.398
+ Average Reward: tensor(63.5723)
+ Best Reward: tensor(263.8064)
+ Worst Reward: tensor(-28.2051)
+ Deterministic Reward Evaluation: tensor(260.6054)
+ Time: 22m 9
+
+Policy Number #34:
+ Average Time Steps: 130.22
+ Average Reward: tensor(67.9811)
+ Best Reward: tensor(269.6071)
+ Worst Reward: tensor(-23.1876)
+ Deterministic Reward Evaluation: tensor(191.9536)
+ Time: 22m 37
+
+Policy Number #35:
+ Average Time Steps: 135.046
+ Average Reward: tensor(74.6586)
+ Best Reward: tensor(262.8191)
+ Worst Reward: tensor(-25.6290)
+ Deterministic Reward Evaluation: tensor(238.1738)
+ Time: 23m 24
+
+Policy Number #36:
+ Average Time Steps: 137.858
+ Average Reward: tensor(77.2089)
+ Best Reward: tensor(270.2506)
+ Worst Reward: tensor(-22.4746)
+ Deterministic Reward Evaluation: tensor(192.1620)
+ Time: 23m 30
+
+Policy Number #37:
+ Average Time Steps: 133.138
+ Average Reward: tensor(71.1700)
+ Best Reward: tensor(267.4257)
+ Worst Reward: tensor(-18.9972)
+ Deterministic Reward Evaluation: tensor(256.1472)
+ Time: 23m 17
+
+Policy Number #38:
+ Average Time Steps: 134.324
+ Average Reward: tensor(73.0359)
+ Best Reward: tensor(268.9559)
+ Worst Reward: tensor(-16.3622)
+ Deterministic Reward Evaluation: tensor(255.8402)
+ Time: 23m 27
+
+Policy Number #39:
+ Average Time Steps: 132.25
+ Average Reward: tensor(69.5822)
+ Best Reward: tensor(267.5328)
+ Worst Reward: tensor(-19.3233)
+ Deterministic Reward Evaluation: tensor(263.2840)
+ Time: 23m 12
+
+Policy Number #40:
+ Average Time Steps: 134.078
+ Average Reward: tensor(71.7353)
+ Best Reward: tensor(340.4381)
+ Worst Reward: tensor(-17.0706)
+ Deterministic Reward Evaluation: tensor(192.7283)
+ Time: 23m 11
+
+Policy Number #41:
+ Average Time Steps: 133.934
+ Average Reward: tensor(74.7601)
+ Best Reward: tensor(267.8867)
+ Worst Reward: tensor(-18.4499)
+ Deterministic Reward Evaluation: tensor(264.1262)
+ Time: 22m 43
+
+Policy Number #42:
+ Average Time Steps: 142.242
+ Average Reward: tensor(85.3738)
+ Best Reward: tensor(271.6832)
+ Worst Reward: tensor(-15.0875)
+ Deterministic Reward Evaluation: tensor(255.4032)
+ Time: 24m 20
+
+Policy Number #43:
+ Average Time Steps: 138.308
+ Average Reward: tensor(76.4722)
+ Best Reward: tensor(268.8214)
+ Worst Reward: tensor(-17.0959)
+ Deterministic Reward Evaluation: tensor(235.4065)
+ Time: 24m 1
+
+Policy Number #44:
+ Average Time Steps: 141.928
+ Average Reward: tensor(82.8199)
+ Best Reward: tensor(340.0343)
+ Worst Reward: tensor(-13.9999)
+ Deterministic Reward Evaluation: tensor(255.5421)
+ Time: 24m 43
+
+Policy Number #45:
+ Average Time Steps: 133.008
+ Average Reward: tensor(74.4591)
+ Best Reward: tensor(270.1023)
+ Worst Reward: tensor(-12.4596)
+ Deterministic Reward Evaluation: tensor(267.4449)
+ Time: 22m 55
+
+Policy Number #46:
+ Average Time Steps: 133.67
+ Average Reward: tensor(75.2076)
+ Best Reward: tensor(332.1006)
+ Worst Reward: tensor(-14.0184)
+ Deterministic Reward Evaluation: tensor(237.4824)
+ Time: 22m 51
+
+Policy Number #47:
+ Average Time Steps: 141.036
+ Average Reward: tensor(81.5390)
+ Best Reward: tensor(271.2858)
+ Worst Reward: tensor(-15.7507)
+ Deterministic Reward Evaluation: tensor(241.7335)
+ Time: 24m 26
+
+Policy Number #48:
+ Average Time Steps: 144.012
+ Average Reward: tensor(87.9991)
+ Best Reward: tensor(340.8509)
+ Worst Reward: tensor(-11.3358)
+ Deterministic Reward Evaluation: tensor(255.5053)
+ Time: 25m 6
+
+Policy Number #49:
+ Average Time Steps: 148.598
+ Average Reward: tensor(91.7507)
+ Best Reward: tensor(342.6179)
+ Worst Reward: tensor(-10.0001)
+ Deterministic Reward Evaluation: tensor(255.6338)
+ Time: 25m 47
+
+Policy Number #50:
+ Average Time Steps: 139.876
+ Average Reward: tensor(81.4515)
+ Best Reward: tensor(277.8659)
+ Worst Reward: tensor(-11.6284)
+ Deterministic Reward Evaluation: tensor(258.1944)
+ Time: 24m 7
+
+Policy Number #51:
+ Average Time Steps: 138.864
+ Average Reward: tensor(81.4313)
+ Best Reward: tensor(275.2506)
+ Worst Reward: tensor(-9.3933)
+ Deterministic Reward Evaluation: tensor(256.0809)
+ Time: 24m 12
+
+Policy Number #52:
+ Average Time Steps: 142.4
+ Average Reward: tensor(87.3715)
+ Best Reward: tensor(341.8409)
+ Worst Reward: tensor(-8.9317)
+ Deterministic Reward Evaluation: tensor(257.9247)
+ Time: 25m 3
+
+Policy Number #53:
+ Average Time Steps: 137.916
+ Average Reward: tensor(84.8605)
+ Best Reward: tensor(338.6187)
+ Worst Reward: tensor(-8.1720)
+ Deterministic Reward Evaluation: tensor(258.1808)
+ Time: 24m 12
+
+Policy Number #54:
+ Average Time Steps: 138.66
+ Average Reward: tensor(82.6870)
+ Best Reward: tensor(340.8794)
+ Worst Reward: tensor(-6.3355)
+ Deterministic Reward Evaluation: tensor(259.0560)
+ Time: 24m 24
+
+Policy Number #55:
+ Average Time Steps: 139.012
+ Average Reward: tensor(85.6565)
+ Best Reward: tensor(351.3536)
+ Worst Reward: tensor(-2.2574)
+ Deterministic Reward Evaluation: tensor(257.5242)
+ Time: 24m 24
+
+Policy Number #0:
+ Average Time Steps: 145.214
+ Average Reward: tensor(-7.4743)
+ Best Reward: tensor(173.4915)
+ Worst Reward: tensor(-440.9451)
+ Deterministic Reward Evaluation: tensor(30.4683)
+ Time: 27m 56
+
+Policy Number #1:
+ Average Time Steps: 145.582
+ Average Reward: tensor(-0.7723)
+ Best Reward: tensor(215.1253)
+ Worst Reward: tensor(-391.1151)
+ Deterministic Reward Evaluation: tensor(-22.3279)
+ Time: 28m 58
+
+Policy Number #2:
+ Average Time Steps: 145.954
+ Average Reward: tensor(0.6155)
+ Best Reward: tensor(210.1866)
+ Worst Reward: tensor(-435.8777)
+ Deterministic Reward Evaluation: tensor(82.1853)
+ Time: 29m 13
+
+Policy Number #3:
+ Average Time Steps: 138.51
+ Average Reward: tensor(30.2718)
+ Best Reward: tensor(264.7868)
+ Worst Reward: tensor(-378.7314)
+ Deterministic Reward Evaluation: tensor(140.9172)
+ Time: 55m 48
+
+Policy Number #4:
+ Average Time Steps: 149.772
+ Average Reward: tensor(35.2392)
+ Best Reward: tensor(255.4454)
+ Worst Reward: tensor(-270.3944)
+ Deterministic Reward Evaluation: tensor(141.3073)
+ Time: 32m 53
+
+Policy Number #5:
+ Average Time Steps: 158.652
+ Average Reward: tensor(42.5355)
+ Best Reward: tensor(257.1924)
+ Worst Reward: tensor(-331.6367)
+ Deterministic Reward Evaluation: tensor(152.9267)
+ Time: 47m 11
+
+Policy Number #6:
+ Average Time Steps: 148.986
+ Average Reward: tensor(39.1698)
+ Best Reward: tensor(266.3555)
+ Worst Reward: tensor(-320.5730)
+ Deterministic Reward Evaluation: tensor(204.8249)
+ Time: 34m 35
+
+Policy Number #7:
+ Average Time Steps: 147.516
+ Average Reward: tensor(40.0819)
+ Best Reward: tensor(266.3633)
+ Worst Reward: tensor(-256.6609)
+ Deterministic Reward Evaluation: tensor(203.0340)
+ Time: 43m 27
+
+Policy Number #8:
+ Average Time Steps: 151.824
+ Average Reward: tensor(43.2991)
+ Best Reward: tensor(262.1538)
+ Worst Reward: tensor(-313.0535)
+ Deterministic Reward Evaluation: tensor(203.9398)
+ Time: 27m 11
+
+Policy Number #9:
+ Average Time Steps: 155.31
+ Average Reward: tensor(31.8352)
+ Best Reward: tensor(298.2962)
+ Worst Reward: tensor(-447.9696)
+ Deterministic Reward Evaluation: tensor(204.3748)
+ Time: 29m 21
+
+Policy Number #10:
+ Average Time Steps: 157.236
+ Average Reward: tensor(46.1791)
+ Best Reward: tensor(313.9221)
+ Worst Reward: tensor(-278.9856)
+ Deterministic Reward Evaluation: tensor(154.5100)
+ Time: 29m 13
+
+Policy Number #11:
+ Average Time Steps: 160.574
+ Average Reward: tensor(46.0216)
+ Best Reward: tensor(260.2269)
+ Worst Reward: tensor(-292.5306)
+ Deterministic Reward Evaluation: tensor(207.6479)
+ Time: 29m 33
+
+Policy Number #12:
+ Average Time Steps: 202.89
+ Average Reward: tensor(42.6369)
+ Best Reward: tensor(249.6396)
+ Worst Reward: tensor(-456.7430)
+ Deterministic Reward Evaluation: tensor(172.2585)
+ Time: 37m 25
+
+Policy Number #13:
+ Average Time Steps: 197.002
+ Average Reward: tensor(48.0090)
+ Best Reward: tensor(274.5155)
+ Worst Reward: tensor(-400.7305)
+ Deterministic Reward Evaluation: tensor(252.7695)
+ Time: 36m 11
+
+Policy Number #14:
+ Average Time Steps: 198.448
+ Average Reward: tensor(45.2531)
+ Best Reward: tensor(305.1073)
+ Worst Reward: tensor(-516.1928)
+ Deterministic Reward Evaluation: tensor(236.9256)
+ Time: 36m 22
+
+Policy Number #15:
+ Average Time Steps: 200.884
+ Average Reward: tensor(56.1958)
+ Best Reward: tensor(296.3640)
+ Worst Reward: tensor(-414.7282)
+ Deterministic Reward Evaluation: tensor(241.8818)
+ Time: 36m 37
+
+Policy Number #16:
+ Average Time Steps: 194.67
+ Average Reward: tensor(76.6455)
+ Best Reward: tensor(312.9175)
+ Worst Reward: tensor(-30.9069)
+ Deterministic Reward Evaluation: tensor(245.0537)
+ Time: 33m 27
+
+Policy Number #17:
+ Average Time Steps: 195.402
+ Average Reward: tensor(71.6556)
+ Best Reward: tensor(362.4413)
+ Worst Reward: tensor(-55.9550)
+ Deterministic Reward Evaluation: tensor(247.2440)
+ Time: 33m 49
+
+Policy Number #18:
+ Average Time Steps: 199.25
+ Average Reward: tensor(73.5901)
+ Best Reward: tensor(362.8979)
+ Worst Reward: tensor(-60.0774)
+ Deterministic Reward Evaluation: tensor(249.0204)
+ Time: 34m 47
+
+Policy Number #19:
+ Average Time Steps: 213.658
+ Average Reward: tensor(82.6966)
+ Best Reward: tensor(365.0646)
+ Worst Reward: tensor(-68.5418)
+ Deterministic Reward Evaluation: tensor(244.1972)
+ Time: 37m 16
+
+Policy Number #20:
+ Average Time Steps: 219.208
+ Average Reward: tensor(84.3355)
+ Best Reward: tensor(363.4833)
+ Worst Reward: tensor(-90.5358)
+ Deterministic Reward Evaluation: tensor(249.0656)
+ Time: 38m 11
+
+Policy Number #21:
+ Average Time Steps: 232.402
+ Average Reward: tensor(87.9588)
+ Best Reward: tensor(320.4059)
+ Worst Reward: tensor(-79.3426)
+ Deterministic Reward Evaluation: tensor(255.5874)
+ Time: 40m 19
+
+Policy Number #22:
+ Average Time Steps: 254.748
+ Average Reward: tensor(102.8290)
+ Best Reward: tensor(362.6034)
+ Worst Reward: tensor(-58.5688)
+ Deterministic Reward Evaluation: tensor(251.6808)
+ Time: 43m 50
+
+Policy Number #23:
+ Average Time Steps: 255.688
+ Average Reward: tensor(100.7968)
+ Best Reward: tensor(369.8500)
+ Worst Reward: tensor(-139.9914)
+ Deterministic Reward Evaluation: tensor(315.7679)
+ Time: 43m 55
+
+Policy Number #24:
+ Average Time Steps: 266.752
+ Average Reward: tensor(107.9183)
+ Best Reward: tensor(366.4199)
+ Worst Reward: tensor(-80.5194)
+ Deterministic Reward Evaluation: tensor(328.9924)
+ Time: 46m 4
+
+Policy Number #25:
+ Average Time Steps: 283.872
+ Average Reward: tensor(118.1459)
+ Best Reward: tensor(389.3860)
+ Worst Reward: tensor(-85.2897)
+ Deterministic Reward Evaluation: tensor(303.6195)
+ Time: 48m 48
+
+Policy Number #26:
+ Average Time Steps: 293.844
+ Average Reward: tensor(120.4765)
+ Best Reward: tensor(388.1024)
+ Worst Reward: tensor(-69.2030)
+ Deterministic Reward Evaluation: tensor(275.8389)
+ Time: 50m 35
+
+Policy Number #27:
+ Average Time Steps: 302.354
+ Average Reward: tensor(124.9276)
+ Best Reward: tensor(398.6852)
+ Worst Reward: tensor(-72.2664)
+ Deterministic Reward Evaluation: tensor(338.9153)
+ Time: 52m 37
+
+Policy Number #28:
+ Average Time Steps: 308.248
+ Average Reward: tensor(124.9630)
+ Best Reward: tensor(401.7516)
+ Worst Reward: tensor(-77.2973)
+ Deterministic Reward Evaluation: tensor(283.0857)
+ Time: 104m 53
+
+Policy Number #29:
+ Average Time Steps: 323.362
+ Average Reward: tensor(143.7176)
+ Best Reward: tensor(456.0289)
+ Worst Reward: tensor(-78.3056)
+ Deterministic Reward Evaluation: tensor(339.2289)
+ Time: 53m 46
+
+Policy Number #30:
+ Average Time Steps: 321.636
+ Average Reward: tensor(140.1091)
+ Best Reward: tensor(441.2586)
+ Worst Reward: tensor(-75.3051)
+ Deterministic Reward Evaluation: tensor(347.6123)
+ Time: 54m 55
+
+Policy Number #31:
+ Average Time Steps: 353.61
+ Average Reward: tensor(173.7844)
+ Best Reward: tensor(455.0985)
+ Worst Reward: tensor(-72.6004)
+ Deterministic Reward Evaluation: tensor(343.6443)
+ Time: 59m 30
+
+Policy Number #32:
+ Average Time Steps: 344.968
+ Average Reward: tensor(180.3106)
+ Best Reward: tensor(450.9888)
+ Worst Reward: tensor(-70.7332)
+ Deterministic Reward Evaluation: tensor(351.3824)
+ Time: 58m 47
+
+Policy Number #33:
+ Average Time Steps: 346.722
+ Average Reward: tensor(192.6161)
+ Best Reward: tensor(463.1447)
+ Worst Reward: tensor(-11.2309)
+ Deterministic Reward Evaluation: tensor(350.2022)
+ Time: 58m 55
+
+Policy Number #34:
+ Average Time Steps: 349.434
+ Average Reward: tensor(199.6353)
+ Best Reward: tensor(470.0817)
+ Worst Reward: tensor(-11.9648)
+ Deterministic Reward Evaluation: tensor(356.5537)
+ Time: 59m 12
+
+Policy Number #35:
+ Average Time Steps: 337.252
+ Average Reward: tensor(198.2566)
+ Best Reward: tensor(474.5852)
+ Worst Reward: tensor(-69.1619)
+ Deterministic Reward Evaluation: tensor(356.8981)
+ Time: 58m 4
+
+Policy Number #36:
+ Average Time Steps: 348.766
+ Average Reward: tensor(213.1413)
+ Best Reward: tensor(477.8542)
+ Worst Reward: tensor(-13.5714)
+ Deterministic Reward Evaluation: tensor(357.7015)
+ Time: 59m 19
+
diff --git a/data/ppo/trajectories/best_rollout#37.pkl b/data/ppo/trajectories/best_rollout#37.pkl
new file mode 100644
index 0000000..e6c284c
Binary files /dev/null and b/data/ppo/trajectories/best_rollout#37.pkl differ
diff --git a/data/ppo/trajectories/best_rollout#56.pkl b/data/ppo/trajectories/best_rollout#56.pkl
new file mode 100644
index 0000000..c0c77a9
Binary files /dev/null and b/data/ppo/trajectories/best_rollout#56.pkl differ
diff --git a/data/ppo/trajectories/worst_rollout#37.pkl b/data/ppo/trajectories/worst_rollout#37.pkl
new file mode 100644
index 0000000..128619a
Binary files /dev/null and b/data/ppo/trajectories/worst_rollout#37.pkl differ
diff --git a/data/ppo/trajectories/worst_rollout#56.pkl b/data/ppo/trajectories/worst_rollout#56.pkl
new file mode 100644
index 0000000..53aa41d
Binary files /dev/null and b/data/ppo/trajectories/worst_rollout#56.pkl differ
diff --git a/learning/autopilot.py b/learning/autopilot.py
index 5cd3f47..dd478fa 100644
--- a/learning/autopilot.py
+++ b/learning/autopilot.py
@@ -4,8 +4,13 @@
from tensordict.nn.distributions import NormalParamExtractor
from tensordict.nn import TensorDictModule, InteractionType
from torchrl.modules import ProbabilisticActor, TanhNormal, ValueOperator
+from torch.distributions import Categorical
+from tensordict.nn import CompositeDistribution
+from learning.utils import CategoricalControlsExtractor
import os
+from shared import ELEVATOR_CLAMP
+
"""
An autopilot learner. It takes the form of a policy network that outputs
actions for a given state.
@@ -20,7 +25,7 @@ def __init__(self):
- 3 angular velocities: w_roll, w_pitch, w_yaw
- 3d relative position of next waypoint: wx, wy, wz
"""
- self.inputs = 13
+ self.inputs = 15
"""
Action:
@@ -36,7 +41,7 @@ def __init__(self):
nn.Linear(self.inputs, self.inputs),
nn.ReLU(),
nn.Linear(self.inputs, self.outputs),
- nn.Sigmoid(),
+ nn.Tanh(),
)
# Returns the action selected and 0, representing the log-prob of the
@@ -50,10 +55,10 @@ def get_control(self, action):
Clamps various control outputs and sets the mean for control surfaces to 0.
Assumes [action] is a 4-item tensor of throttle, aileron cmd, elevator cmd, rudder cmd.
"""
- action[0] = 0.5 * action[0]
- action[1] = 0.1 * (action[1] - 0.5)
- action[2] = 0.5 * (action[2] - 0.5)
- action[3] = 0.5 * (action[3] - 0.5)
+ action[0] = 0.8 * (0.5*(action[0] + 1))
+ action[1] = 0.1 * action[1]
+ action[2] = ELEVATOR_CLAMP * action[2]
+ action[3] = 0.1 * action[3]
return action
# flattened_params = flattened dx1 numpy array of all params to init from
@@ -82,12 +87,11 @@ def init_from_params(self, flattened_params):
layer1,
nn.ReLU(),
layer2,
- nn.Sigmoid(),
+ nn.Tanh()
)
- # Loads the network from dir/name.pth
- def init_from_saved(self, dir, name):
- path = os.path.join(dir, name + '.pth')
+ # Loads the network from path
+ def init_from_saved(self, path):
self.policy_network = torch.load(path)
# Saves the network to dir/name.pth
@@ -125,8 +129,8 @@ def transform_from_deterministic_learner(self):
self.policy_network[-2] = nn.Linear(self.inputs, self.outputs * 2)
# Update the output layer to include the original weights and biases
- self.policy_network[-2].weight = nn.Parameter(torch.cat((w, torch.ones(w.shape) * 100), 0))
- self.policy_network[-2].bias = nn.Parameter(torch.cat((b, torch.ones(b.shape) * 100), 0))
+ self.policy_network[-2].weight = nn.Parameter(torch.cat((w, torch.zeros(w.shape)), 0))
+ self.policy_network[-2].bias = nn.Parameter(torch.cat((b, torch.zeros(b.shape)), 0))
# Add a normal param extractor to the network to extract (means, sigmas) tuple
self.policy_network.append(NormalParamExtractor())
@@ -141,10 +145,6 @@ def transform_from_deterministic_learner(self):
module=policy_module,
in_keys=["loc", "scale"],
distribution_class=TanhNormal,
- distribution_kwargs={
- "min": 0, # minimum control
- "max": 1, # maximum control
- },
default_interaction_type=InteractionType.RANDOM,
return_log_prob=True,
)
@@ -153,7 +153,6 @@ def transform_from_deterministic_learner(self):
def get_action(self, observation):
data = TensorDict({"observation": observation}, [])
policy_forward = self.policy_module(data)
- print("action", self.policy_network(observation))
return policy_forward["action"], policy_forward["sample_log_prob"]
# NOTE: This initializes from *deterministic* learner parameters and picks
@@ -173,10 +172,120 @@ def init_from_saved(self, path):
module=policy_module,
in_keys=["loc", "scale"],
distribution_class=TanhNormal,
- distribution_kwargs={
- "min": 0, # minimum control
- "max": 1, # maximum control
- },
default_interaction_type=InteractionType.RANDOM,
return_log_prob=True,
)
+
+"""
+Slew rate autopilot learner.
+Each control (throttle/aileron/elevator/rudder) has three options:
+stay constant, go down, or go up.
+"""
+class SlewRateAutopilotLearner:
+ def __init__(self):
+ self.inputs = 15
+ self.outputs = 4
+
+ # Slew rates are wrt sim clock
+ self.throttle_slew_rate = 0.0005
+ self.aileron_slew_rate = 0.0001
+ self.elevator_slew_rate = 0.0001
+ self.rudder_slew_rate = 0.0001
+
+ self.policy_network = nn.Sequential(
+ nn.Linear(self.inputs, self.inputs),
+ nn.ReLU(),
+ nn.Linear(self.inputs, 3 * self.outputs),
+ nn.Sigmoid(),
+ CategoricalControlsExtractor()
+ )
+
+ self.instantiate_policy_module()
+
+ def instantiate_policy_module(self):
+ policy_module = TensorDictModule(self.policy_network, in_keys=["observation"], out_keys=[("params", "throttle", "probs"),("params", "aileron", "probs"),("params", "elevator", "probs"),("params", "rudder", "probs")])
+ self.policy_module = policy_module = ProbabilisticActor(
+ module=policy_module,
+ in_keys=["params"],
+ distribution_class=CompositeDistribution,
+ distribution_kwargs={
+ "distribution_map": {
+ "throttle": Categorical,
+ "aileron": Categorical,
+ "elevator": Categorical,
+ "rudder": Categorical,
+ }
+ },
+ default_interaction_type=InteractionType.RANDOM,
+ return_log_prob=True
+ )
+
+ # Returns the action selected and the log_prob of that action
+ def get_action(self, observation):
+ data = TensorDict({"observation": observation}, [])
+ throttle_probs, aileron_probs, elevator_probs, rudder_probs = self.policy_network(observation)
+ policy_forward = self.policy_module(data)
+ # print("aileron pos", observation[12], "aileron probs", aileron_probs / torch.sum(aileron_probs))
+
+ action = torch.Tensor([policy_forward['throttle'], policy_forward['aileron'], policy_forward['elevator'], policy_forward['rudder']])
+ return action, policy_forward["sample_log_prob"]
+
+ # Always samples the mode of the output for each control
+ def get_deterministic_action(self, observation):
+ throttle_probs, aileron_probs, elevator_probs, rudder_probs = self.policy_network(observation)
+ action = torch.Tensor([torch.argmax(throttle_probs), torch.argmax(aileron_probs), torch.argmax(elevator_probs), torch.argmax(rudder_probs)])
+ # print("aileron pos", observation[12], "aileron probs", aileron_probs / torch.sum(aileron_probs))
+ w = self.policy_network[2].weight
+ # print(f"Reward min: {torch.min(w)}, mean: {torch.mean(w)}, median: {torch.median(w)}, max: {torch.max(w)}, std: {torch.std(w)}" )
+ return action, 0
+
+ # Apply a -1 transformation to the action to create control tensor such that:
+ # -1 means go down, 0 means stay same, and +1 means go up
+ def get_control(self, action):
+ return action - 1
+
+ # flattened_params = flattened dx1 numpy array of all params to init from
+ # NOTE: the way the params are broken up into the weights/biases of each layer
+ # would need to be manually edited for changes in network architecture
+ def init_from_params(self, flattened_params):
+ flattened_params = torch.from_numpy(flattened_params).to(torch.float32)
+
+ pl, pr = 0, 0
+ layer1 = nn.Linear(self.inputs, self.inputs)
+ pr += layer1.weight.nelement()
+ layer1.weight = nn.Parameter(flattened_params[pl:pr].reshape(layer1.weight.shape))
+ pl = pr
+ pr += layer1.bias.nelement()
+ layer1.bias = nn.Parameter(flattened_params[pl:pr].reshape(layer1.bias.shape))
+
+ layer2 = nn.Linear(self.inputs, 3*self.outputs)
+ pl = pr
+ pr += layer2.weight.nelement()
+ layer2.weight = nn.Parameter(flattened_params[pl:pr].reshape(layer2.weight.shape))
+ pl = pr
+ pr += layer2.bias.nelement()
+ layer2.bias = nn.Parameter(flattened_params[pl:pr].reshape(layer2.bias.shape))
+
+ self.policy_network = nn.Sequential(
+ layer1,
+ nn.ReLU(),
+ layer2,
+ nn.Sigmoid(),
+ CategoricalControlsExtractor()
+ )
+ self.instantiate_policy_module()
+
+ # Loads the network from path
+ def init_from_saved(self, path):
+ self.policy_network = torch.load(path)
+ self.instantiate_policy_module()
+
+ # Saves the network to dir/name.pth
+ def save(self, dir, name):
+ path = os.path.join(dir, name + '.pth')
+ torch.save(self.policy_network, path)
+
+
+
+
+
diff --git a/learning/crossentropy.py b/learning/crossentropy.py
index 677fcf1..a7ab64a 100644
--- a/learning/crossentropy.py
+++ b/learning/crossentropy.py
@@ -1,6 +1,6 @@
import torch
import numpy as np
-from learning.autopilot import AutopilotLearner
+from learning.autopilot import AutopilotLearner, SlewRateAutopilotLearner
from simulation.simulate import FullIntegratedSim
from simulation.jsbsim_aircraft import x8
import os
@@ -25,7 +25,7 @@ def __init__(self, learners, num_params):
def init_using_torch_default(generation_size, num_params):
learners = []
for i in range(generation_size):
- learners.append(AutopilotLearner())
+ learners.append(SlewRateAutopilotLearner())
return Generation(learners, num_params)
# Utilizes [rewards], which contains the reward obtained by each learner as a
@@ -89,17 +89,16 @@ def make_new_generation(mean, cov, generation_size, num_params):
# Generate each learner from params
learners = []
for param_list in selected_params:
- l = AutopilotLearner()
+ l = SlewRateAutopilotLearner()
l.init_from_params(param_list)
learners.append(l)
return Generation(learners, num_params)
-def cross_entropy_train(epochs, generation_size, num_survive, num_params=238, sim_time=60.0, save_dir='cross_entropy'):
+def cross_entropy_train(epochs, generation_size, num_survive, num_params=432, sim_time=60.0, save_dir='cross_entropy'):
# Create save_dir (and if one already exists, rename it with some rand int)
- if os.path.exists(os.path.join('data', save_dir)):
- os.rename(os.path.join('data', save_dir), os.path.join('data', save_dir + '_old' + str(randint(0, 100000))))
- os.mkdir(os.path.join('data', save_dir))
- stats_file = open(os.path.join('data', save_dir, 'stats.txt'), 'w')
+ # if os.path.exists(os.path.join('data', save_dir)):
+ # os.rename(os.path.join('data', save_dir), os.path.join('data', save_dir + '_old' + str(randint(0, 100000))))
+ # os.mkdir(os.path.join('data', save_dir))
# Baseline to be updated after first generation
mean = np.zeros((num_params))
@@ -117,7 +116,7 @@ def cross_entropy_train(epochs, generation_size, num_survive, num_params=238, si
# Evaluate generation through rollouts
rewards = []
for i in range(len(generation.learners)):
- id = str(100*(epoch+1) + (i+1))
+ id = str(1000*(epoch+1) + (i+1))
learner = generation.learners[i]
# Run simulation to evaluate learner
@@ -127,7 +126,7 @@ def cross_entropy_train(epochs, generation_size, num_survive, num_params=238, si
integrated_sim.simulation_loop()
# Acquire/save data
- integrated_sim.mdp_data_collector.save(os.path.join(save_dir, 'generation' + str(epoch+1)), 'trajectory_learner#' + str(i+1))
+ #integrated_sim.mdp_data_collector.save(os.path.join(save_dir, 'generation' + str(epoch+1)), 'trajectory_learner#' + str(i+1))
rewards.append(integrated_sim.mdp_data_collector.get_cum_reward())
print('Reward for Learner #', id, ': ', integrated_sim.mdp_data_collector.get_cum_reward())
@@ -140,13 +139,15 @@ def cross_entropy_train(epochs, generation_size, num_survive, num_params=238, si
cov += 0.01 * np.identity(mean.shape[0])
# Save important info in the save_dir stats file
+ stats_file = open(os.path.join('data', save_dir, 'stats.txt'), 'a')
stats_file.write('Generation #' + str(epoch+1) + ':\n')
stats_file.write('Best, Median, and Worst Learner: ' + str(ids) + '\n')
stats_file.write('Best, Median, and Worst Reward: ' + str(rew) + '\n')
stats_file.write('\n\n\n')
+ stats_file.close()
if __name__ == "__main__":
os.environ["JSBSIM_DEBUG"]=str(0)
# epochs, generation_size, num_survive
- cross_entropy_train(100, 99, 50)
\ No newline at end of file
+ cross_entropy_train(100, 200, 50)
\ No newline at end of file
diff --git a/learning/ppo.py b/learning/ppo.py
index e012f67..cf3851c 100644
--- a/learning/ppo.py
+++ b/learning/ppo.py
@@ -7,11 +7,18 @@
from torchrl.data.replay_buffers.storages import LazyTensorStorage
from torchrl.objectives import ClipPPOLoss
from torchrl.objectives.value import GAE
-from learning.autopilot import StochasticAutopilotLearner
+from learning.autopilot import StochasticAutopilotLearner, SlewRateAutopilotLearner
from simulation.simulate import FullIntegratedSim
from simulation.jsbsim_aircraft import x8
+import numpy as np
import os
+import time
+import statistics
+import beepy as beep
+device = "cpu" if not torch.has_cuda else "cuda:0"
+
+start_id = 36
"""
Gathers rollout data and returns it in the way the Proximal Policy Optimization loss_module expects
"""
@@ -24,36 +31,88 @@ def gather_rollout_data(autopilot_learner, policy_num, num_trajectories=100, sim
sample_log_probs = torch.empty(0)
rewards = torch.empty(0)
dones = torch.empty(0, dtype=torch.bool)
+ best_cum_reward = -float('inf')
+ worst_cum_reward = float('inf')
+ total_cum_reward = 0
+ total_timesteps = 0
for t in range(num_trajectories):
- integrated_sim = FullIntegratedSim(x8, autopilot_learner, sim_time)
+ # Reset distribution
+ print(f"Trajectory #{t+1}")
+ rand = np.random.rand()
+ if rand <= 0.25:
+ in_flight_reset = 0
+ elif rand <= 0.5:
+ in_flight_reset = 3
+ elif rand <= 0.7:
+ in_flight_reset = 4
+ elif rand <= 0.85:
+ in_flight_reset = 5
+ else:
+ in_flight_reset = 6
+ #in_flight_reset = 0
+ # Run a sim with a stochastic version of the autopilot so it can explore
+ integrated_sim = FullIntegratedSim(x8, autopilot_learner, sim_time, in_flight_reset=in_flight_reset, auto_deterministic=False)
integrated_sim.simulation_loop()
# Acquire data
observation, next_observation, action, sample_log_prob, reward, done = integrated_sim.mdp_data_collector.get_trajectory_data()
+ cum_reward = integrated_sim.mdp_data_collector.get_cum_reward()
+ total_cum_reward += cum_reward
+ total_timesteps += reward.shape[0]
- # Save data
- # integrated_sim.mdp_data_collector.save(os.path.join('ppo', 'trajectories'), 'rollout' + str(policy_num * num_trajectories + t))
+ # Save trajectory if worst or best so far
+ if cum_reward > best_cum_reward:
+ integrated_sim.mdp_data_collector.save(os.path.join('ppo', 'trajectories'), 'best_rollout#' + str(policy_num))
+ best_cum_reward = cum_reward
+ if cum_reward < worst_cum_reward:
+ integrated_sim.mdp_data_collector.save(os.path.join('ppo', 'trajectories'), 'worst_rollout#' + str(policy_num))
+ worst_cum_reward = cum_reward
# Add to the data
observations = torch.cat((observations, observation))
next_observations = torch.cat((next_observations, next_observation))
actions = torch.cat((actions, action))
sample_log_probs = torch.cat((sample_log_probs,sample_log_prob))
- rewards = torch.cat((rewards, reward))
+ rewards = torch.cat((rewards, reward))
dones = torch.cat((dones, done))
data_size += observation.shape[0]
-
+
+ # we divide by std for reward scaling
+ print("Reward scale", torch.std(rewards))
+ rewards /= torch.std(rewards)
+ print(f"Reward min: {torch.min(rewards)}, mean: {torch.mean(rewards)}, median: {torch.median(rewards)}, max: {torch.max(rewards)}" )
# Each entry tensor should be data_size x d where d is the dimension of
# that entry for one step in a rollout.
data = TensorDict({
"observation": observations,
- "action": actions.detach(),
+ "action": TensorDict({
+ "throttle": actions[:,0].detach(),
+ "aileron": actions[:,1].detach(),
+ "elevator": actions[:,2].detach(),
+ "rudder": actions[:,3].detach(),
+ # "throttle_log_prob": torch.zeros(data_size).detach(),
+ # "aileron_log_prob": torch.zeros(data_size).detach(),
+ # "elevator_log_prob": torch.zeros(data_size).detach(),
+ # "rudder_log_prob": torch.zeros(data_size).detach(),
+ }, [data_size,]),
"sample_log_prob": sample_log_probs.detach(), # log probability that each action was selected
+ # "sample_log_prob": TensorDict({
+ # }, [data_size,]), # log probability that each action was selected
("next", "done"): dones,
("next", "terminated"): dones,
("next", "reward"): rewards,
("next", "observation"): next_observations,
}, [data_size,])
+ torch.save(data, os.path.join('data', 'ppo', 'all_data.pkl'))
+
+ # Write to stats file
+ stats_file = open(os.path.join('data', 'ppo', 'stats.txt'), 'a')
+ stats_file.write('Policy Number #' + str(policy_num) + ':\n')
+ stats_file.write('\tAverage Time Steps: ' + str(total_timesteps/num_trajectories) + '\n')
+ stats_file.write('\tAverage Reward: ' + str(total_cum_reward/num_trajectories) + '\n')
+ stats_file.write('\tBest Reward: ' + str(best_cum_reward) + '\n')
+ stats_file.write('\tWorst Reward: ' + str(worst_cum_reward) + '\n')
+ stats_file.close()
return data, data_size
@@ -64,12 +123,12 @@ def gather_rollout_data(autopilot_learner, policy_num, num_trajectories=100, sim
action taken during PPO learning.
"""
def make_value_estimator_module(n_obs):
+ global device
+
value_net = nn.Sequential(
- nn.Linear(n_obs, n_obs),
+ nn.Linear(n_obs, 2*n_obs, device=device),
nn.Tanh(),
- nn.Linear(n_obs, n_obs),
- nn.Tanh(),
- nn.Linear(n_obs, 1), # one value is computed for the state
+ nn.Linear(2*n_obs, 1, device=device), # one value is computed for the state
)
return ValueOperator(
@@ -95,7 +154,12 @@ def make_value_estimator_module(n_obs):
def train_ppo_once(policy_num, autopilot_learner, loss_module, advantage_module, optimizer, num_trajectories, num_epochs, batch_size):
# Rollout policy to gather trajectories
dataset, data_size = gather_rollout_data(autopilot_learner, policy_num, num_trajectories)
-
+ dataset = torch.load(os.path.join("data","ppo","all_data.pkl"))
+ print(dataset)
+ # data_size = 56114
+ # print(max(dataset.get(("next", "reward"))))
+ # print(dataset)
+ # data_size = 9484
# Define replay buffer to store trajectory data in during training
replay_buffer = ReplayBuffer(
storage=LazyTensorStorage(data_size),
@@ -103,7 +167,7 @@ def train_ppo_once(policy_num, autopilot_learner, loss_module, advantage_module,
)
for epoch in range(num_epochs):
- print("Training Epoch", epoch+1)
+ # print("Training Epoch", epoch+1)
# Re-compute advantage at each epoch as its value depends on the value
# network which is updated in the inner loop
with torch.no_grad():
@@ -112,36 +176,75 @@ def train_ppo_once(policy_num, autopilot_learner, loss_module, advantage_module,
# Gather the data into a replay buffer for sampling
data_view = dataset.reshape(-1)
replay_buffer.extend(data_view.cpu())
+ batch_loss = 0
+ batch_obj_loss = 0
+ batch_critic_loss = 0
for b in range(0, data_size // batch_size):
# Gather batch and calculate PPO loss on it
batch = replay_buffer.sample(batch_size)
- loss_vals = loss_module(batch.to("cpu"))
+ loss_vals = loss_module(batch.to(device))
loss_value = (
loss_vals["loss_objective"]
+ loss_vals["loss_critic"]
- + loss_vals["loss_entropy"]
)
+ batch_loss += loss_value * batch_size
+ batch_obj_loss += loss_vals["loss_objective"] * batch_size
+ batch_critic_loss += loss_vals["loss_critic"] * batch_size
# Optimize via gradient descent with the optimizer
loss_value.backward()
+
+ torch.nn.utils.clip_grad_norm_(loss_module.parameters(), 1.0)
optimizer.step()
optimizer.zero_grad()
+ if (epoch+1) % 10 == 0:
+ print('loss_value in epoch #', epoch+1, ':', float(batch_loss/ 1000), '(objective:', float(batch_obj_loss/ 1000), ' critic: ', float(batch_critic_loss/ 1000), ')')
if __name__ == "__main__":
+
# Parameters (see https://pytorch.org/rl/tutorials/coding_ppo.html#ppo-parameters)
- batch_size = 100
- num_epochs = 10
- clip_epsilon = 0.2 # for PPO loss
- gamma = 1.0
+ batch_size = 256
+ num_epochs = 100
+ device = "cpu" if not torch.has_cuda else "cuda:0"
+
+ clip_epsilon = 0.2 #0.08 # for PPO loss
+ gamma = 0.99 # keep 1.0
lmbda = 0.95
- entropy_eps = 1e-4
- lr = 3e-4
- num_trajectories = 100
+ # entropy_eps = 1e-4
+ lr = 5e-5
+
+ # lr = 5e-5
+# loss_value in epoch # 1 : 67.5849609375 (objective: -0.002501344308257103 critic: 67.58744049072266 )
+# loss_value in epoch # 2 : 66.79192352294922 (objective: -0.022511614486575127 critic: 66.81440734863281 )
+# loss_value in epoch # 3 : 65.99295806884766 (objective: -0.023394417017698288 critic: 66.01636505126953 )
+# loss_value in epoch # 10 : 63.2775764465332 (objective: -0.03471797704696655 critic: 63.31227493286133 )
+# loss_value in epoch # 20 : 58.33335876464844 (objective: -0.03877483680844307 critic: 58.37212371826172 )
+# loss_value in epoch # 30 : 53.69935607910156 (objective: -0.04969186335802078 critic: 53.749061584472656 )
+# loss_value in epoch # 40 : 48.16702651977539 (objective: -0.048812463879585266 critic: 48.21586227416992 )
+# loss_value in epoch # 50 : 42.11222839355469 (objective: -0.06517862528562546 critic: 42.17741394042969 )
+# loss_value in epoch # 60 : 36.78435134887695 (objective: -0.06043088063597679 critic: 36.844749450683594 )
+# loss_value in epoch # 70 : 33.0183219909668 (objective: -0.062020089477300644 critic: 33.08034133911133 )
+# loss_value in epoch # 80 : 32.66770935058594 (objective: -0.05988125503063202 critic: 32.72758865356445 )
+# loss_value in epoch # 90 : 32.728004455566406 (objective: -0.04434733837842941 critic: 32.772335052490234 )
+# loss_value in epoch # 100 : 32.90699768066406 (objective: -0.03433294966816902 critic: 32.94131851196289 )
+
+
+
+ # 3e-5: roughly 160 (27.3, 133)
+ # 1e-4: roughly 190 (27, 133)
+ # 1e-4: roughly 190 (27, 133)
+
+ num_trajectories = 500
num_policy_iterations = 100
# Build the modules
- autopilot_learner = StochasticAutopilotLearner()
+ #autopilot_learner = StochasticAutopilotLearner()
+ autopilot_learner = SlewRateAutopilotLearner()
+ autopilot_learner.init_from_saved(os.path.join("data", "ppo", "policies", f"learner#{start_id}.pth"))
+ #autopilot_learner.init_from_saved(os.path.join("data", "ppo", "policies", "learner#56.pth"))
+ #autopilot_learner.init_from_saved(os.path.join("data", "cross_entropy", "generation14", "learner#67.pth"))
+ # autopilot_learner.init_from_params(np.random.normal(0, 1, 350))
value_module = make_value_estimator_module(autopilot_learner.inputs)
advantage_module = GAE(
gamma=gamma, lmbda=lmbda, value_network=value_module, average_gae=True
@@ -150,8 +253,8 @@ def train_ppo_once(policy_num, autopilot_learner, loss_module, advantage_module,
actor=autopilot_learner.policy_module,
critic=value_module,
clip_epsilon=clip_epsilon,
- entropy_bonus=bool(entropy_eps),
- entropy_coef=entropy_eps,
+ entropy_bonus=False,
+ entropy_coef=0,
value_target_key=advantage_module.value_target_key,
critic_coef=1.0,
gamma=gamma,
@@ -159,7 +262,21 @@ def train_ppo_once(policy_num, autopilot_learner, loss_module, advantage_module,
)
optimizer = torch.optim.Adam(loss_module.parameters(), lr)
- autopilot_learner.save(os.path.join('data', 'ppo', 'policies'), 'learner#0')
- for i in range(num_policy_iterations):
+ #autopilot_learner.save(os.path.join('data', 'ppo', 'policies'), 'learner#0')
+ for i in range(start_id, num_policy_iterations):
+ start = time.time()
train_ppo_once(i, autopilot_learner, loss_module, advantage_module, optimizer, num_trajectories, num_epochs, batch_size)
- autopilot_learner.save(os.path.join('data', 'ppo', 'policies'), 'learner#' + str(i+1))
\ No newline at end of file
+ autopilot_learner.save(os.path.join('data', 'ppo', 'policies'), 'learner#' + str(i+1))
+
+ # Evaluate the learner deterministically
+ integrated_sim = FullIntegratedSim(x8, autopilot_learner, 60.0, in_flight_reset=0, auto_deterministic=True)
+ integrated_sim.simulation_loop()
+ cum_reward = integrated_sim.mdp_data_collector.get_cum_reward()
+ stats_file = open(os.path.join('data', 'ppo', 'stats.txt'), 'a')
+ stats_file.write('\tDeterministic Reward Evaluation: ' + str(cum_reward))
+ end = time.time()
+ duration = end - start
+ stats_file.write(f'\n\tTime: {str(int(duration//60))}m {str(int(duration) %60)}')
+ stats_file.write('\n\n')
+ stats_file.close()
+ beep.beep(2)
\ No newline at end of file
diff --git a/learning/utils.py b/learning/utils.py
new file mode 100644
index 0000000..a09368f
--- /dev/null
+++ b/learning/utils.py
@@ -0,0 +1,12 @@
+import torch
+import torch.nn as nn
+
+class CategoricalControlsExtractor(nn.Module):
+ def __init__(self):
+ super().__init__()
+
+ def forward(self, *tensors: torch.Tensor) -> tuple[torch.Tensor, ...]:
+ tensor, *others = tensors
+ throttle, aileron, elevator, rudder = tensor.chunk(4, -1)
+ return (throttle, aileron, elevator, rudder, *others)
+ #return (nn.functional.softmax(throttle), nn.functional.softmax(aileron), nn.functional.softmax(elevator), nn.functional.softmax(rudder), *others)
diff --git a/scripts/replay_trajectory.py b/scripts/replay_trajectory.py
index 80eda45..9f57178 100644
--- a/scripts/replay_trajectory.py
+++ b/scripts/replay_trajectory.py
@@ -12,7 +12,7 @@
import sys
import os
import torch
-from learning.autopilot import AutopilotLearner
+from learning.autopilot import AutopilotLearner, SlewRateAutopilotLearner
from simulation.simulate import FullIntegratedSim
from simulation.jsbsim_aircraft import x8
@@ -24,9 +24,15 @@
# Get agent interaction frequency (manually entered, must be correct for same trajectory)
agent_freq = int(sys.argv[2])
+# Get in-flight reset
+in_flight_reset = int(sys.argv[3])
+
# Load states, actions, rewards
states, actions, rewards = torch.load(file)
+print('Altitude', states[:,0])
+print('WP Dist', states[:,8])
+#print('Init Controls', states[0,11:])
# Play sim
-integrated_sim = FullIntegratedSim(x8, AutopilotLearner(), 60.0, agent_interaction_frequency=agent_freq)
+integrated_sim = FullIntegratedSim(x8, SlewRateAutopilotLearner(), 60.0, agent_interaction_frequency=agent_freq, in_flight_reset=in_flight_reset)
integrated_sim.simulation_replay(actions)
\ No newline at end of file
diff --git a/scripts/sim_autopilot.py b/scripts/sim_autopilot.py
index c54daef..5aa8e61 100644
--- a/scripts/sim_autopilot.py
+++ b/scripts/sim_autopilot.py
@@ -6,7 +6,7 @@
and rolls out a trajectory in sim for that autopilot learner.
"""
-from learning.autopilot import AutopilotLearner
+from learning.autopilot import AutopilotLearner, SlewRateAutopilotLearner
from simulation.simulate import FullIntegratedSim
from simulation.jsbsim_aircraft import x8
import sys
@@ -16,10 +16,13 @@
policy_file_path = sys.argv[1]
file = os.path.join(*(["data"] + policy_file_path.split('/')))
+# Get in-flight reset
+in_flight_reset = int(sys.argv[2])
+
# Initialize autopilot
-autopilot = AutopilotLearner()
+autopilot = SlewRateAutopilotLearner()
autopilot.init_from_saved(file)
# Play sim
-integrated_sim = FullIntegratedSim(x8, autopilot, 60.0)
+integrated_sim = FullIntegratedSim(x8, autopilot, 60.0, auto_deterministic=True, in_flight_reset=in_flight_reset)
integrated_sim.simulation_loop()
\ No newline at end of file
diff --git a/shared.py b/shared.py
index 7a6b613..e506330 100644
--- a/shared.py
+++ b/shared.py
@@ -1,5 +1,10 @@
import sys, os
+THROTTLE_CLAMP = 0.5
+AILERON_CLAMP = 0.1
+ELEVATOR_CLAMP = 0.1
+RUDDER_CLAMP = 0.1
+
"""
Context manager object that helps prevent print outputs from 3rd party software.
"""
diff --git a/simulation/jsbsim_aircraft.py b/simulation/jsbsim_aircraft.py
index 98f607e..e41e3f8 100644
--- a/simulation/jsbsim_aircraft.py
+++ b/simulation/jsbsim_aircraft.py
@@ -22,4 +22,4 @@ def get_cruise_speed_fps(self) -> float:
x8 = Aircraft('x8', 'Skywalker x8', 20)
-# x8 = Aircraft('c172p', 'Cessna172P', 120)
+#x8 = Aircraft('c172p', 'Cessna172P', 120)
diff --git a/simulation/jsbsim_simulator.py b/simulation/jsbsim_simulator.py
index 9cc5a11..b7755bc 100644
--- a/simulation/jsbsim_simulator.py
+++ b/simulation/jsbsim_simulator.py
@@ -5,6 +5,7 @@
from typing import Dict, Union
import simulation.jsbsim_properties as prp
from simulation.jsbsim_aircraft import Aircraft, x8
+import numpy as np
import math
import csv
from shared import HidePrints
@@ -83,19 +84,26 @@ def __init__(self,
sim_frequency_hz: float = 60.0,
aircraft: Aircraft = x8,
init_conditions: Dict[prp.Property, float] = None,
+ in_flight_reset: int = 0, # 0 means no in-flight reset, positive integer means one of the reset distribution options
debug_level: int = 0):
- self.fdm = jsbsim.FGFDMExec(root_dir=self.ROOT_DIR)
+ with HidePrints():
+ self.fdm = jsbsim.FGFDMExec(root_dir=self.ROOT_DIR)
self.fdm.set_debug_level(debug_level)
self.sim_dt = 1.0 / sim_frequency_hz
self.aircraft = aircraft
self.client = self.airsim_connect()
- self.initialize(self.sim_dt, self.aircraft.jsbsim_id, init_conditions)
+ self.initialize(self.sim_dt, self.aircraft.jsbsim_id, in_flight_reset)
self.fdm.disable_output()
- self.wall_clock_dt = None
+ self.wall_clock_dt = None # 0.001 is the best for visualization
self.update_airsim(ignore_collisions=True)
- self.waypoint_id = 0
- self.waypoint_threshold = 2
- self.waypoint_rewarded = True
+ self.waypoint_id = in_flight_reset # in_flight_reset should corresponds to the waypoint we start going towards
+ self.waypoint_threshold = 3 * 0.45/0.12 # multiplied by recent scale factor
+ # "ViewMode": "NoDisplay",
+
+ self.takeoff_rewarded = in_flight_reset > 0 # rewarded if already in flight
+ self.completed_takeoff = False
+ self.waypoint_entered = False
+ self.waypoint_reward = False
self.waypoints = []
with open("waypoints.csv", 'r') as file:
csvreader = csv.reader(file)
@@ -142,18 +150,22 @@ def get_loaded_model_name(self) -> str:
else:
return None
- def initialize(self, dt: float, model_name: str, init_conditions: Dict['prp.Property', float] = None) -> None:
+ def initialize(self, dt: float, model_name: str, in_flight_reset: int) -> None:
"""
Start JSBSim with custom initial conditions
:param dt: simulation rate [s]
:param model_name: the aircraft model used
- :param init_conditions: initial simulation conditions
+ :param in_flight_reset: whether to simulate from an in-flight reset
:return: None
"""
# Hardcoded currently, meaning init_conditions argument is overriden
- ic_file = 'basic_ic.xml'
+ if in_flight_reset == 0:
+ ic_file = 'basic_ic.xml'
+ else:
+ ic_file = f'reset_ic{in_flight_reset}.xml'
+
ic_path = os.path.join(os.path.dirname(os.path.abspath(__file__)), ic_file)
self.fdm.load_ic(ic_path, useStoredPath=False)
self.load_model(model_name)
@@ -241,8 +253,9 @@ def airsim_connect() -> airsim.VehicleClient:
:return: the airsim client object
"""
- client = airsim.VehicleClient()
- client.confirmConnection()
+ with HidePrints():
+ client = airsim.VehicleClient()
+ client.confirmConnection()
return client
def update_airsim(self, ignore_collisions = False) -> None:
diff --git a/simulation/mdp.py b/simulation/mdp.py
index 58a292f..f9400ad 100644
--- a/simulation/mdp.py
+++ b/simulation/mdp.py
@@ -4,62 +4,140 @@
states/observations, actions, and rewards.
"""
+import math
import torch
import simulation.jsbsim_properties as prp
import numpy as np
import os
+from shared import THROTTLE_CLAMP, AILERON_CLAMP, ELEVATOR_CLAMP, RUDDER_CLAMP
"""
Extracts agent state data from the sim.
"""
-def state_from_sim(sim, debug=False):
- state = torch.zeros(13,)
+def state_from_sim(sim):
+ state = torch.zeros(15,)
FT_TO_M = 0.3048
# altitude
state[0] = sim[prp.altitude_sl_ft] * FT_TO_M # z
- # velocity
- state[1] = sim[prp.v_north_fps] * FT_TO_M # x velocity
- state[2] = sim[prp.v_east_fps] * FT_TO_M # y velocity
- state[3] = -sim[prp.v_down_fps] * FT_TO_M # z velocity
-
- # angles
- state[4] = sim[prp.roll_rad] # roll
- state[5] = sim[prp.pitch_rad] # pitch
- state[6] = sim[prp.heading_rad] # yaw
+ # speed
+ state[1] = FT_TO_M * np.linalg.norm(np.array([sim[prp.v_north_fps], sim[prp.v_east_fps], -sim[prp.v_down_fps]]))
+
+ # roll
+ state[2] = sim[prp.roll_rad]
+ # pitch
+ state[3] = sim[prp.pitch_rad]
+
# angle rates
- state[7] = sim[prp.p_radps] # roll rate
- state[8] = sim[prp.q_radps] # pitch rate
- state[9] = sim[prp.r_radps] # yaw rate
+ state[4] = sim[prp.p_radps] # roll rate
+ state[5] = sim[prp.q_radps] # pitch rate
+ state[6] = sim[prp.r_radps] # yaw rate
- # next waypoint (relative)
+ state[7] = -sim[prp.v_down_fps] * FT_TO_M # z velocity
+
+ # calculate next waypoint (relative)
position = np.array(sim.get_local_position())
waypoint = np.array(sim.waypoints[sim.waypoint_id])
-
displacement = waypoint - position
+
+ # if sim.waypoint_id == 3:
+ # input()
+ # print("lat", sim[prp.lat_geod_deg])
+ # print("lon", sim[prp.lng_geoc_deg])
+ # print("u_fps", sim[prp.u_fps])
+ # print("v_fps", sim[prp.v_fps])
+ # print("w_fps", sim[prp.w_fps])
+ # print("roll_rad", sim[prp.roll_rad])
+ # print("pitch_rad", sim[prp.pitch_rad])
+ # print("heading_rad", sim[prp.heading_rad])
+ # print("altitude", sim[prp.altitude_sl_ft])
+
+ # determine whether next waypoint needs to be updated if we're currently inside one
if np.linalg.norm(displacement) <= sim.waypoint_threshold:
- print("Waypoint Hit!")
+ print(f"\t\t\t\t\t\t\t\t\t\tWaypoint {sim.waypoint_id} Hit!")
sim.waypoint_id += 1
- sim.waypoint_rewarded = False
+ sim.waypoint_entered = True
+ sim.waypoint_reward = True
+ if sim.waypoint_id == 7:
+ raise Exception(f"Finish! :{50}")
+ # raise Exception(f"Unhealthy state, do better. Penalty :{penalty}")
- state[10] = displacement[0]
- state[11] = displacement[1]
- state[12] = displacement[2]
- if debug:
- print('State!')
- print('Altitude:', state[0])
- print('Velocity: (', state[1], state[2], state[3], ')')
- print('Roll:', state[4], '; Pitch:', state[5], '; Yaw:', state[6])
- print('RollRate:', state[7], '; PitchRate:', state[8], '; YawRate:', state[9])
- print('Relative WP: (', state[10], state[11], state[12], ')')
+ # elif sim.waypoint_entered:
+ # prev_waypoint = np.array(sim.waypoints[sim.waypoint_id-1])
+ # prev_displacement = prev_waypoint - position
+ # if np.linalg.norm(prev_displacement) > sim.waypoint_threshold:
+ # # if we've left the previous waypoint, give reward for it
+ # print(f"\t\t\t\t\t\t\t\t\t\tWaypoint {sim.waypoint_id-1} Rewarded!")
+ # sim.waypoint_reward = True
+ # sim.waypoint_entered = False
+
+ waypoint = np.array(sim.waypoints[sim.waypoint_id])
+ displacement = waypoint - position
+
+ # waypoint ground distance
+ state[8] = np.linalg.norm(np.array(displacement[0:2]))
+
+ # waypoint altitude distance
+ state[9] = displacement[2]
+
+ # heading diff from waypoint
+ waypoint_heading = np.arctan2(displacement[1], displacement[0])
+ heading = sim[prp.heading_rad] # yaw
+ state[10] = waypoint_heading - heading
+ if state[10] < -math.pi:
+ state[10] += 2 * math.pi
+ elif state[10] > math.pi:
+ state[10] -= 2* math.pi
+
+
+ # print("\t\t\t\t\t\t\t\t\t\t YAW", heading/math.pi*180)
+ # print("\t\t\t\t\t\t\t\t\t\t waypoint_heading", waypoint_heading/math.pi*180)
+
+
+ # print("\t\t\t\t\t\t\t\t\t\theading", state[10]/math.pi*180)
+ # print("\t\t\t\t\t\t\t\t\t\tdist", state[8])
+ # print("\t\t\t\t\t\t\t\t\t\theight", state[9])
+
+
+ # Controls state
+ state[11] = sim[prp.throttle_cmd] / THROTTLE_CLAMP
+ state[12] = sim[prp.aileron_cmd] / AILERON_CLAMP
+ state[13] = sim[prp.elevator_cmd] / ELEVATOR_CLAMP
+ state[14] = sim[prp.rudder_cmd] / RUDDER_CLAMP
+
+
+ # Record whether takeoff completed
+ if state[0] >= 2:
+ if not sim.completed_takeoff:
+ print("\t\t\t\t\t\t\t\t\t\tTake off!")
+ sim.completed_takeoff = True
+
+ unhealthy_state, penalty = is_unhealthy_state(state)
+ if unhealthy_state:
+ raise Exception(f"Unhealthy state, do better. Penalty :{penalty}")
+
return state
+"""
+Returns a bool for whether the state is unhealthy
+"""
+def is_unhealthy_state(state):
+ MAX_BANK = math.pi / 3
+ MAX_PITCH = math.pi / 4
+ if np.cos(state[10]) < 0:
+ return True, -10
+ if not -MAX_BANK < state[2] < MAX_BANK:
+ return True, -50
+ if not -MAX_PITCH < state[3] < MAX_PITCH:
+ return True, -25
+ return False, 0
+
"""
Updates sim according to a control, assumes [control] is a 4-item tensor of
@@ -72,6 +150,52 @@ def update_sim_from_control(sim, control, debug=False):
sim[prp.rudder_cmd] = control[3]
if debug:
print('Control Taken:', control)
+
+
+# Get the state/action/log_prob and control from the slewrate autopilot
+count = 0
+def query_slewrate_autopilot(sim, autopilot, deterministic=False):
+ global count
+ state = state_from_sim(sim)
+ action, log_prob = autopilot.get_deterministic_action(state) if deterministic else autopilot.get_action(state)
+ control = autopilot.get_control(action)
+
+ #print('state', state)
+ #print(state[10:])
+ #print('elevator', state[13])
+ """
+ if state[1] <= 18 and not sim.completed_takeoff:
+ print('throttle', state[11])
+ control = torch.Tensor([1, 0, 0, 0])
+ elif state[1] <= 20 and not sim.completed_takeoff:
+ print('throttle down', state[11])
+ control = torch.Tensor([-1, 0, 0, 0])
+ elif count < 20:
+ #print('throttle', state[11])
+ print('holding pitch', state[13])
+ count += 1
+ control = torch.Tensor([0, 0, 0, 0])
+ elif count < 40:
+ print('pitch up')
+ count += 1
+ control = torch.Tensor([0, 0, -1, 0])
+ elif count < 400:
+ count += 1
+ control = torch.Tensor([0, 0, 0, 0])
+ else:
+ control = torch.Tensor([0, 0, 1, 0])
+ """
+
+
+ return state, action, log_prob, control
+
+# Called every single sim step to enact slewrate
+# Assumes autopilot is a SlewRateAutopilotLearner
+def update_sim_from_slewrate_control(sim, control, autopilot):
+ sim[prp.throttle_cmd] = np.clip(sim[prp.throttle_cmd] + control[0] * autopilot.throttle_slew_rate, 0, THROTTLE_CLAMP)
+ sim[prp.aileron_cmd] = np.clip(sim[prp.aileron_cmd] + control[1] * autopilot.aileron_slew_rate, -AILERON_CLAMP, AILERON_CLAMP)
+ sim[prp.elevator_cmd] = np.clip(sim[prp.elevator_cmd] + control[2] * autopilot.elevator_slew_rate, -ELEVATOR_CLAMP, ELEVATOR_CLAMP)
+ sim[prp.rudder_cmd] = np.clip(sim[prp.rudder_cmd] + control[3] * autopilot.rudder_slew_rate, -RUDDER_CLAMP, RUDDER_CLAMP)
"""
Follows a predetermined sequence of controls, instead of using autopilot.
@@ -115,18 +239,27 @@ def enact_predetermined_controls(sim, autopilot):
Basically just updates sim throttle / control surfaces according to the autopilot.
"""
def enact_autopilot(sim, autopilot):
- state = state_from_sim(sim, debug=False)
+ state = state_from_sim(sim)
action, log_prob = autopilot.get_action(state)
-
update_sim_from_control(sim, autopilot.get_control(action))
return state, action, log_prob
+
# Takes in the action outputted directly from the network and outputs the
# normalized quadratic action cost from 0-1
def quadratic_action_cost(action):
- action[1:4] = 2*action[1:4] - 1 # convert control surfaces to [-1, 1]
- return float(torch.dot(action, action).detach()) / 4 # divide by 4 to be 0-1
+ action_cost_weights = torch.tensor([3.0, 10.0, 5.0, 1.0])
+ action[0] = 0.5 * (action[0] + 1) # converts throttle to be 0-1
+ return float(torch.dot(action ** 2, action_cost_weights).detach() / sum(action_cost_weights)) # divide by 4 to be 0-1
+
+# Takes in the action outputted directly from the network and outputs the
+# normalized quadratic action cost from 0-1
+def quadratic_control_cost(control):
+ # control_cost_weights = torch.tensor([1.0, 10.0, 5.0, 1.0])
+ # 0.05882352941 0.5882352941 0.2941176471 0.05882352941
+ control_cost_weights = torch.tensor([0.6, 10.0, 0.3, 0.06])
+ return float(torch.dot(control ** 2, control_cost_weights).detach()) # divide by 4 to be 0-1
"""
The reward function for the bb autopilots. Since they won't know how to fly,
@@ -134,6 +267,7 @@ def quadratic_action_cost(action):
+ 1 if moving with some velocity threshold vel_reward_threshold
+ alt_reward_threshold if flying above ground with some threshold alt_reward_threshold
- action_coeff * the quadratic control cost
+ !!! WARNING: Deprecated
"""
def bb_reward(action, next_state, collided, alt_reward_coeff=10, action_coeff=1, alt_reward_threshold=2, vel_reward_threshold=1):
moving_reward = 1 if (next_state[3]**2 + next_state[4]**2 + next_state[5]**2) > vel_reward_threshold else 0
@@ -145,6 +279,7 @@ def bb_reward(action, next_state, collided, alt_reward_coeff=10, action_coeff=1,
A reward function that tries to incentivize the plane to go towards the waypoint
at each timestep, while also rewarding for being above ground and penalizing
for high control effort
+ !!! WARNING: Deprecated
"""
def new_init_wp_reward(action, next_state, collided, wp_coeff=1, action_coeff=1, alt_reward_threshold=5):
alt_reward = 1 if next_state[2] > alt_reward_threshold else 0
@@ -152,29 +287,49 @@ def new_init_wp_reward(action, next_state, collided, wp_coeff=1, action_coeff=1,
waypoint_rel_unit = next_state[10:13] / torch.norm(next_state[10:13])
vel = next_state[1:4]
- toward_waypoint_reward = wp_coeff * float(torch.dot((vel ** 2 * torch.sign(vel)), waypoint_rel_unit).detach())
+ toward_waypoint_reward = wp_coeff * float(torch.dot(vel, waypoint_rel_unit).detach())
return toward_waypoint_reward + alt_reward - action_cost if not collided else 0
"""
-A reward function.
+A reward function getter.
"""
+a = 0
def get_wp_reward(sim):
- def wp_reward(action, next_state, collided, wp_coeff=0.01, action_coeff=0.01, alt_reward_threshold=10):
- if not sim.waypoint_rewarded:
- sim.waypoint_rewarded = True
- wp_reward = 1_000_000
+ def wp_reward(action, next_state, collided, wp_coeff=0.5, action_coeff=0.5, alt_coeff=0.5, roll_coeff=0.9):
+ global a
+ if sim.waypoint_reward:
+ sim.waypoint_reward = False
+ wp_reward = 50
else: wp_reward = 0
- # print("altitude", next_state[2] )
- alt_reward = 0.1 if next_state[2] > alt_reward_threshold else 0
- action_cost = action_coeff * quadratic_action_cost(action)
-
- waypoint_rel_unit = next_state[10:13] / torch.norm(next_state[10:13])
- vel = next_state[1:4]
- toward_waypoint_reward = wp_coeff * float(torch.dot((vel ** 2 * torch.sign(vel)), waypoint_rel_unit).detach())
- # print("wp_reward: ", wp_reward, " toward_waypoint_reward: ", toward_waypoint_reward)
- # print("alt_reward: ", alt_reward, " action_cost: ", -action_cost)
- # print("\t reward", wp_reward + toward_waypoint_reward + alt_reward - action_cost if not collided else 0)
- return wp_reward + toward_waypoint_reward + alt_reward - action_cost if not collided else 0
+ action_cost = action_coeff * quadratic_control_cost(next_state[11:15])
+ #print('\t\t\t\t\t\tACTION COST:', action_cost, next_state[11:15])
+ # waypoint_rel_unit = torch.nn.functional.normalize(next_state[10:13], dim=0)
+ # vel = next_state[1:4]
+ # toward_waypoint_reward = wp_coeff * float(torch.dot(vel, waypoint_rel_unit).detach())
+ # cosine(relative heading) * ground speed
+ #toward_waypoint_reward = min(wp_coeff * torch.cos(next_state[10]) * np.sqrt(next_state[1] ** 2 - next_state[7] ** 2), 4)
+ away_waypoint_cost = wp_coeff * abs(next_state[10])
+ alitude_diff_cost = alt_coeff * abs(np.arctan2(next_state[9], next_state[8]) - next_state[3]) # pitch "error": relative pitch between current pitch and straight line to waypoint
+ roll_cost = roll_coeff * (abs(next_state[2]))**3
+ #print('\t\t\t\t\t\AWAY WP COST:', away_waypoint_cost)
+ #print('\t\t\t\t\t\ALT COST:', alitude_diff_cost)
+ if not sim.takeoff_rewarded and sim.completed_takeoff:
+ takeoff_reward = 25
+ sim.takeoff_rewarded = True
+ else:
+ takeoff_reward = 0
+
+ staying_alive_reward = 0.5
+ # toward_waypoint_reward = 0
+ rewards = staying_alive_reward + wp_reward + takeoff_reward
+ costs = action_cost + away_waypoint_cost + alitude_diff_cost + roll_cost
+ # if a % 100 == 0:
+ # print("\t\t\t\t\t\t\t\taway_waypoint_cost", away_waypoint_cost, "alitude_diff_cost",alitude_diff_cost)
+ # print("\t\t\t\t\t\t\t\troll_cost", roll_cost, "action_cost",action_cost, "action",next_state[11:15])
+ # print("\t\t\t\t\t\t\t\tRewards:",rewards, "Costs:", -costs)
+ # print("\t\t\t\t\t\t\t\tNet Reward", rewards - costs if not collided else 0)
+ # toward_waypoint_reward = torch.min(torch.tensor(10), 0.01 * 1 / torch.max(torch.tensor(0.1), torch.sum(next_state[10:13]**2)))
+ return rewards - costs if not collided else 0
return wp_reward
"""
@@ -182,7 +337,7 @@ def wp_reward(action, next_state, collided, wp_coeff=0.01, action_coeff=0.01, al
rollout, including states/actions/rewards the agent experienced.
"""
class MDPDataCollector:
- def __init__(self, sim, reward_fn, expected_trajectory_length, state_dim=13, action_dim=4):
+ def __init__(self, sim, reward_fn, expected_trajectory_length, state_dim=15, action_dim=4):
# parameters
self.sim = sim
self.reward_fn = reward_fn
@@ -197,7 +352,7 @@ def __init__(self, sim, reward_fn, expected_trajectory_length, state_dim=13, act
self.cum_reward = 0
# t = timestep of the current state-action pair, in [0, expected_trajectory_length-1]
- def update(self, t, state, action, log_prob, next_state, collided):
+ def update(self, t, state, action, log_prob, next_state, collided, unhealthy_penalty=0.0):
self.states[t, :] = torch.transpose(state, 0, -1)
self.next_states[t, :] = torch.transpose(next_state, 0, -1)
self.actions[t, :] = torch.transpose(action, 0, -1)
@@ -206,7 +361,7 @@ def update(self, t, state, action, log_prob, next_state, collided):
# this function should only be used when sim is still running, so not done
# and not collided
self.dones[t] = 0
- reward = self.reward_fn(action, next_state, collided)
+ reward = self.reward_fn(action, next_state, collided) + unhealthy_penalty
self.rewards[t] = reward
self.cum_reward += reward
diff --git a/simulation/reset_ic3.xml b/simulation/reset_ic3.xml
new file mode 100644
index 0000000..5f3394f
--- /dev/null
+++ b/simulation/reset_ic3.xml
@@ -0,0 +1,17 @@
+
+
+
+ 96.0
+ 0.0
+ 3.16
+ -0.00000180221070782819
+ 0.0022388192006667883
+ -0.00033960315510837
+ 5.89360846566800000
+ 0.030266855814653545
+ 25.397382017225027
+ 2.1
+ 90.0
+ 0.0
+ 0
+
\ No newline at end of file
diff --git a/simulation/reset_ic4.xml b/simulation/reset_ic4.xml
new file mode 100644
index 0000000..ce78a24
--- /dev/null
+++ b/simulation/reset_ic4.xml
@@ -0,0 +1,17 @@
+
+
+
+ 85.7973202721795
+ 0.0
+ 2.8924357765532296
+ 0.00000047068495677246515
+ 0.0035561556674120506
+ -0.00111435048832493
+ 5.72140628231839000
+ -0.030266855814653545
+ 101.0717219337821
+ 2.1
+ 90.0
+ 0.0
+ 0
+
\ No newline at end of file
diff --git a/simulation/reset_ic5.xml b/simulation/reset_ic5.xml
new file mode 100644
index 0000000..cdba4a7
--- /dev/null
+++ b/simulation/reset_ic5.xml
@@ -0,0 +1,17 @@
+
+
+
+ 90.0
+ 0.0
+ 3.0
+ 0.0
+ 0.0049
+ 0.0
+ 8.721406282318400
+ 0.0
+ 170.0
+ 2.1
+ 90.0
+ 0.0
+ 0
+
\ No newline at end of file
diff --git a/simulation/reset_ic6.xml b/simulation/reset_ic6.xml
new file mode 100644
index 0000000..3097c05
--- /dev/null
+++ b/simulation/reset_ic6.xml
@@ -0,0 +1,17 @@
+
+
+
+ 90.0
+ 0.0
+ 3.0
+ 0.0001
+ 0.0061
+ -10.0011143504883249324
+ 7.121406282318400
+ 0.0
+ 225.0
+ 2.1
+ 90.0
+ 0.0
+ 0
+
\ No newline at end of file
diff --git a/simulation/simulate.py b/simulation/simulate.py
index b0ab2a3..26655fa 100644
--- a/simulation/simulate.py
+++ b/simulation/simulate.py
@@ -1,10 +1,13 @@
+from shared import HidePrints
from simulation.jsbsim_simulator import Simulation
from simulation.jsbsim_aircraft import Aircraft, x8
import simulation.jsbsim_properties as prp
-from learning.autopilot import AutopilotLearner
+from learning.autopilot import AutopilotLearner, SlewRateAutopilotLearner
+import torch
import simulation.mdp as mdp
import os
import numpy as np
+from shared import THROTTLE_CLAMP, AILERON_CLAMP, ELEVATOR_CLAMP, RUDDER_CLAMP
"""
A class to integrate JSBSim and AirSim to roll-out a full trajectory for an
@@ -13,20 +16,23 @@
class FullIntegratedSim:
def __init__(self,
aircraft: Aircraft,
- autopilot: AutopilotLearner,
+ autopilot: SlewRateAutopilotLearner,
sim_time: float,
display_graphics: bool = True,
- agent_interaction_frequency: int = 1,
+ agent_interaction_frequency: int = 15,
airsim_frequency_hz: float = 392.0,
sim_frequency_hz: float = 240.0,
- init_conditions: bool = None,
+ in_flight_reset: int = 0, # nonzero if we initialize from a non-takeoff reset distribution
+ auto_deterministic: bool = True, # whether the autopilot picks its mode action (deterministic) or samples
debug_level: int = 0):
# Aircraft and autopilot
self.aircraft = aircraft
self.autopilot = autopilot
+ self.auto_deterministic = auto_deterministic
# Sim params
- self.sim: Simulation = Simulation(sim_frequency_hz, aircraft, init_conditions, debug_level)
+ self.in_flight_reset = in_flight_reset
+ self.sim: Simulation = Simulation(sim_frequency_hz, aircraft, in_flight_reset=in_flight_reset, debug_level=debug_level)
self.sim_time = sim_time
self.display_graphics = display_graphics
self.sim_frequency_hz = sim_frequency_hz
@@ -44,7 +50,12 @@ def __init__(self,
# Triggered when sim is complete
self.done: bool = False
+ #
+ self.unhealthy_termination: bool = False
+ self.unhealthy_penalty: float = 0.0
+
self.initial_collision = False
+
"""
Run loop for one simulation.
"""
@@ -59,31 +70,60 @@ def simulation_loop(self):
pose = self.sim.client.simGetVehiclePose()
# Experimentally determined, in UE4 coordinate system
- ic_position = np.array([0, 0, -1.3411200046539307])
+ if self.in_flight_reset == 0:
+ ic_position = [0, 0, -1.3411200046539307]
+ elif self.in_flight_reset == 3:
+ ic_position = [ 2.50183762e+02, -1.58594549e-01, -8.38120174e+00]
+ elif self.in_flight_reset == 4:
+ ic_position = [ 3.97393585e+02, 4.14202549e-02, -3.14456406e+01]
+ elif self.in_flight_reset == 5:
+ ic_position = [ 5.47565369e+02, -7.68899895e-08, -5.24536057e+01]
+ elif self.in_flight_reset == 6:
+ ic_position = [ 681.66052246, 8.80005455, -69.21742249]
+
+ ic_position = np.array(ic_position)
current_position = np.array([pose.position.x_val, pose.position.y_val, pose.position.z_val])
retry_period = 10
retry_counter = 0
- while (np.abs(ic_position - current_position) > np.finfo(float).eps).all():
+ while (np.abs(ic_position - current_position)/ np.linalg.norm(current_position) > 0.00001).all():
if retry_counter % retry_period == 0:
self.sim.reinitialize()
retry_counter += 1
+ # print("cur pos", current_position)
pose = self.sim.client.simGetVehiclePose()
current_position = np.array([pose.position.x_val, pose.position.y_val, pose.position.z_val])
-
+ # Initialize to max throttle; the agent then learns when/how to decrease for cruise throttle
+ self.sim[prp.throttle_cmd] = THROTTLE_CLAMP
+
+ # If we're initializing in flight, randomly set controls
+ if self.in_flight_reset > 0:
+ self.sim[prp.throttle_cmd] = np.clip(np.random.normal(0.35,THROTTLE_CLAMP/10), 0, THROTTLE_CLAMP)
+ self.sim[prp.aileron_cmd] = np.clip(np.random.normal(0, AILERON_CLAMP/200), -AILERON_CLAMP, AILERON_CLAMP)
+ self.sim[prp.elevator_cmd] = np.clip(np.random.normal(0, ELEVATOR_CLAMP/20), -ELEVATOR_CLAMP, ELEVATOR_CLAMP)
+ self.sim[prp.rudder_cmd] = np.clip(np.random.normal(0, RUDDER_CLAMP/20), -RUDDER_CLAMP, RUDDER_CLAMP)
+
i = 0
while i < update_num:
# Do autopilot controls
try:
- state, action, log_prob = mdp.enact_autopilot(self.sim, self.autopilot)
+ #state, action, log_prob = mdp.enact_autopilot(self.sim, self.autopilot)
+ state, action, log_prob, control = mdp.query_slewrate_autopilot(self.sim, self.autopilot, deterministic=self.auto_deterministic)
+ if torch.isnan(state).any():
+ break
except Exception as e:
print(e)
+ self.unhealthy_penalty = float(str(e).split(":")[-1])
+ # print("penalty", self.unhealthy_penalty)
+ self.unhealthy_termination = True
# If enacting the autopilot fails, end the simulation immediately
break
# Update sim while waiting for next agent interaction
while True:
+ mdp.update_sim_from_slewrate_control(self.sim, control, self.autopilot)
+
# Run another sim step
self.sim.run()
@@ -100,28 +140,35 @@ def simulation_loop(self):
# Check for collisions via airsim and terminate if there is one
if self.sim.get_collision_info().has_collided:
if self.initial_collision:
- print('Aircraft has collided.')
+ # print('Aircraft has collided.')
self.done = True
self.sim.reinitialize()
else:
- print("Aircraft completed initial landing")
+ # print("Aircraft completed initial landing")
self.initial_collision = True
# Exit if sim is over or it's time for another agent interaction
if self.done or i % self.agent_interaction_frequency == 0:
break
-
+
# Get new state
try:
next_state = mdp.state_from_sim(self.sim)
- except:
+ if torch.isnan(next_state).any():
+ next_state = state
+ self.done = True
+ except Exception as e:
# If we couldn't acquire the state, something crashed with jsbsim
# We treat that as the end of the simulation and don't update the state
+ print(f"\t\t\t\t\t\t\t\t\t\t{e}")
next_state = state
self.done = True
+ self.unhealthy_termination = True
+ self.unhealthy_penalty = float(str(e).split(":")[-1])
# Data collection update for this step
- self.mdp_data_collector.update(int(i/self.agent_interaction_frequency)-1, state, action, log_prob, next_state, self.done)
+ self.mdp_data_collector.update(int(i/self.agent_interaction_frequency)-1, state, action,
+ log_prob, next_state, self.unhealthy_termination, self.unhealthy_penalty)
# End if collided
if self.done == True:
@@ -130,18 +177,29 @@ def simulation_loop(self):
self.done = True
self.mdp_data_collector.terminate(int(i/self.agent_interaction_frequency))
print('Simulation complete.')
+ print('\t\t\t\t\t\t\t\t\t\tCum reward:', self.mdp_data_collector.cum_reward)
+
"""
Replays a simulation
"""
def simulation_replay(self, actions):
+ # THESE ARE HARCODED by reading from prints of the initial controls state. TODO: make this a possible input
+ self.sim[prp.throttle_cmd] = 0.5372 * THROTTLE_CLAMP
+ self.sim[prp.aileron_cmd] = -0.0024 * AILERON_CLAMP
+ self.sim[prp.elevator_cmd] = -0.0352 * ELEVATOR_CLAMP
+ self.sim[prp.rudder_cmd] = -0.0080 * RUDDER_CLAMP
+
i = 0
for action in actions:
+ state = mdp.state_from_sim(self.sim)
# Do the control
- mdp.update_sim_from_control(self.autopilot.get_control(action))
-
+ # mdp.update_sim_from_control(self.autopilot.get_control(action))
+ control = self.autopilot.get_control(action)
while True:
# Run another sim step
+ mdp.update_sim_from_slewrate_control(self.sim, control, self.autopilot)
+ #print('control', control)
self.sim.run()
i += 1
@@ -151,7 +209,15 @@ def simulation_replay(self, actions):
# NOTE: for replays, the agent interaction frequency must match what
# it was when the trajectory was created
if i % self.agent_interaction_frequency == 0:
+ print("Step")
break
+ try:
+ next_state = mdp.state_from_sim(self.sim)
+ self.mdp_data_collector.update(int(i/self.agent_interaction_frequency)-1, state, action,
+ 0, next_state, self.unhealthy_termination, self.unhealthy_penalty)
+ except:
+ break
+ print('cum reward', self.mdp_data_collector.cum_reward)
if __name__ == "__main__":
# A one-minute simulation with an untrained autopilot
diff --git a/waypoints.csv b/waypoints.csv
index 7b87c80..bc9e55e 100644
--- a/waypoints.csv
+++ b/waypoints.csv
@@ -1,7 +1,8 @@
waypoint_id,x,y,z,relative_x,relative_y,relative_z
-1,-4461,-417.960938,90,21.5,-0.2,0.89
-2,-701,-417.960938,120,59.1,-0.2,1.19
-3,2869,-57.960938,450,94.8,3.4,4.49
-4,6509,-967.960938,1230,131.2,-5.7,12.29
-5,9469,-477.960938,1480,160.8,-0.8,14.79
-6,15619,-397.960938,2920,222.3,0,29.19
\ No newline at end of file
+1,-11341,-417.960938,70,38.5,-0.2,0.69
+2,-741,-397.960938,110,144.5,0,1.09
+3,9466,-621,1286,246.57,-2.23039062,12.85
+4,24279,-437.960938,2570,394.7,-0.4,25.69
+5,39319,-257.960938,5370,545.1,1.4,53.69
+6,52959,292.039062,7340,681.5,6.9,73.39
+7,72839,-377.960938,11710,880.3,0.2,117.09
\ No newline at end of file