diff --git a/assets/x8.xml b/assets/x8.xml index 3368f7b..67442de 100644 --- a/assets/x8.xml +++ b/assets/x8.xml @@ -68,12 +68,12 @@ 0 -0.18 - 0.8 - 0.5 - 0.1 - 8.0 - 0.85 - 0 + 1.8 + 1.5 + .01 + 4.0 + 1.15 + 360 NONE 0 @@ -83,12 +83,12 @@ -85.1 -0.18 - 0.8 - 0.5 - 0.1 - 4.0 + 1.8 + 1.5 + 0.01 + 2.0 0.95 - 0.0 + 360 LEFT 0 @@ -98,12 +98,12 @@ 85.1 -0.18 - 0.8 - 0.5 - 0.1 - 4.0 + 1.8 + 1.5 + 0.01 + 2.0 0.95 - 0.0 + 360 RIGHT 0 @@ -111,16 +111,6 @@ - - 0.0 - 0.0 - 0.0 - - - 0.0 - 0.0 - 0.0 - 0.0 @@ -145,7 +135,7 @@ attitude/heading-true-rad - 0.003 + 0.0 @@ -264,7 +254,7 @@ Alpha independent lift coefficient aero/qbar-area - 0.0867 + 0.04 @@ -273,7 +263,7 @@ aero/qbar-area aero/alpha-rad - 4.0203 + 2.0203 @@ -468,12 +458,12 @@ - Pitch moment due to alpha\ + Pitch momet due to alpha\ aero/qbar-area metrics/cbarw-ft aero/alpha-rad - -0.4629 + -0.6629 @@ -494,7 +484,7 @@ aero/qbar-area metrics/cbarw-ft fcs/elevator-pos-rad - -0.2292 + -1.1220 diff --git a/data/cross_entropy/stats.txt b/data/cross_entropy/stats.txt new file mode 100644 index 0000000..fb36156 --- /dev/null +++ b/data/cross_entropy/stats.txt @@ -0,0 +1,126 @@ +Generation #1: +Best, Median, and Worst Learner: (142, 35, 157) +Best, Median, and Worst Reward: (2982.999081533402, 972.1300675626844, 588.3324620518833) + + + +Generation #2: +Best, Median, and Worst Learner: (80, 49, 173) +Best, Median, and Worst Reward: (2992.656533884816, 978.6449913680553, 594.0445281118155) + + + +Generation #3: +Best, Median, and Worst Learner: (178, 56, 16) +Best, Median, and Worst Reward: (2993.3571543405997, 1445.3874917849898, 604.7305159419775) + + + +Generation #4: +Best, Median, and Worst Learner: (84, 193, 37) +Best, Median, and Worst Reward: (2990.2669111620635, 1959.0998122617602, 602.2353224325925) + + + +Generation #5: +Best, Median, and Worst Learner: (70, 185, 88) +Best, Median, and Worst Reward: (2992.6985498089343, 1972.4182319762185, 653.0758294676198) + + + +Generation #6: +Best, Median, and Worst Learner: (103, 70, 184) +Best, Median, and Worst Reward: (3993.390440976944, 1999.3610292344797, 937.2859918950126) + + + +Generation #7: (good) +Best, Median, and Worst Learner: (162, 149, 44) +Best, Median, and Worst Reward: (3989.6084612680133, 2981.2333327531815, 966.7899975357577) + + + +Generation #8: +Best, Median, and Worst Learner: (188, 45, 105) +Best, Median, and Worst Reward: (2992.4165739994496, 2983.244331255555, 978.5750615522265) + + + +Generation #9: +Best, Median, and Worst Learner: (78, 168, 71) +Best, Median, and Worst Reward: (2993.2526850610448, 2987.396712552756, 968.8540844265372) + + + +Generation #10: +Best, Median, and Worst Learner: (91, 177, 88) +Best, Median, and Worst Reward: (3993.3897130363766, 2988.8644929472357, 974.3769715242088) + + + +Generation #11: +Best, Median, and Worst Learner: (89, 154, 52) +Best, Median, and Worst Reward: (2992.881334366277, 2988.9321229401976, 962.7465795613825) + + + +Generation #12: +Best, Median, and Worst Learner: (27, 150, 156) +Best, Median, and Worst Reward: (3993.390440976944, 2989.2162777222693, 968.5438468381763) + + + +Generation #13: +Best, Median, and Worst Learner: (93, 167, 13) +Best, Median, and Worst Reward: (3993.379415991084, 2989.711838164367, 969.5627298392355) + + + +Generation #14: +Best, Median, and Worst Learner: (67, 170, 181) +Best, Median, and Worst Reward: (3984.084361743182, 2990.3426594454795, 980.2484985329211) + + + +Generation #15: +Best, Median, and Worst Learner: (199, 77, 112) +Best, Median, and Worst Reward: (2993.251348318532, 2989.871223669499, 959.5475498940796) + + + +Generation #16: +Best, Median, and Worst Learner: (38, 47, 102) +Best, Median, and Worst Reward: (2993.230045834556, 2991.1996988034807, 977.2491851244122) + + + +Generation #17: +Best, Median, and Worst Learner: (80, 14, 37) +Best, Median, and Worst Reward: (2993.272437831387, 2991.9086525086313, 980.539373550564) + + + +Generation #18: +Best, Median, and Worst Learner: (133, 195, 32) +Best, Median, and Worst Reward: (2993.77225545235, 2992.6561779286712, 985.9263532422483) + + + +Generation #19: +Best, Median, and Worst Learner: (84, 31, 7) +Best, Median, and Worst Reward: (2993.7730519939214, 2992.7406542636454, 979.8817010223866) + + + +Generation #20: +Best, Median, and Worst Learner: (79, 100, 102) +Best, Median, and Worst Reward: (2993.7877100314945, 2992.7477019503713, 987.0250741429627) + + + +Generation #21: +Best, Median, and Worst Learner: (119, 82, 108) +Best, Median, and Worst Reward: (2993.3186951242387, 2992.9422285705805, 982.8713495023549) + + + diff --git a/data/ppo/all_data.pkl b/data/ppo/all_data.pkl new file mode 100644 index 0000000..51bc33e Binary files /dev/null and b/data/ppo/all_data.pkl differ diff --git a/data/ppo/policies/learner#37.pth b/data/ppo/policies/learner#37.pth new file mode 100644 index 0000000..86d49e5 Binary files /dev/null and b/data/ppo/policies/learner#37.pth differ diff --git a/data/ppo/policies/learner#56.pth b/data/ppo/policies/learner#56.pth new file mode 100644 index 0000000..bae2083 Binary files /dev/null and b/data/ppo/policies/learner#56.pth differ diff --git a/data/ppo/stats.txt b/data/ppo/stats.txt new file mode 100644 index 0000000..09608d3 --- /dev/null +++ b/data/ppo/stats.txt @@ -0,0 +1,744 @@ +Policy Number #0: + Average Time Steps: 64.346 + Average Reward: tensor(10.2306) + Best Reward: tensor(185.7999) + Worst Reward: tensor(-19.8913) + Deterministic Reward Evaluation: tensor(73.3813) + Time: 11m 18 + +Policy Number #1: + Average Time Steps: 71.666 + Average Reward: tensor(17.5971) + Best Reward: tensor(231.1814) + Worst Reward: tensor(-20.6358) + Deterministic Reward Evaluation: tensor(82.1320) + Time: 12m 33 + +Policy Number #2: + Average Time Steps: 69.852 + Average Reward: tensor(15.1622) + Best Reward: tensor(179.2707) + Worst Reward: tensor(-21.3903) + Deterministic Reward Evaluation: tensor(56.0810) + Time: 12m 9 + +Policy Number #3: + Average Time Steps: 69.142 + Average Reward: tensor(11.4814) + Best Reward: tensor(174.7796) + Worst Reward: tensor(-21.0693) + Deterministic Reward Evaluation: tensor(153.0145) + Time: 12m 4 + +Policy Number #4: + Average Time Steps: 78.09 + Average Reward: tensor(17.9504) + Best Reward: tensor(236.0611) + Worst Reward: tensor(-19.4090) + Deterministic Reward Evaluation: tensor(79.9471) + Time: 13m 47 + +Policy Number #5: + Average Time Steps: 84.284 + Average Reward: tensor(19.2010) + Best Reward: tensor(232.2858) + Worst Reward: tensor(-25.4860) + Deterministic Reward Evaluation: tensor(69.4026) + Time: 14m 36 + +Policy Number #6: + Average Time Steps: 87.118 + Average Reward: tensor(18.3251) + Best Reward: tensor(189.0960) + Worst Reward: tensor(-33.6494) + Deterministic Reward Evaluation: tensor(89.9409) + Time: 15m 10 + +Policy Number #7: + Average Time Steps: 89.856 + Average Reward: tensor(20.8215) + Best Reward: tensor(230.0491) + Worst Reward: tensor(-27.7497) + Deterministic Reward Evaluation: tensor(66.6348) + Time: 15m 38 + +Policy Number #8: + Average Time Steps: 93.718 + Average Reward: tensor(23.6071) + Best Reward: tensor(185.6656) + Worst Reward: tensor(-35.7423) + Deterministic Reward Evaluation: tensor(91.8381) + Time: 16m 12 + +Policy Number #9: + Average Time Steps: 98.414 + Average Reward: tensor(28.2783) + Best Reward: tensor(193.8371) + Worst Reward: tensor(-32.6535) + Deterministic Reward Evaluation: tensor(74.0798) + Time: 17m 58 + +Policy Number #10: + Average Time Steps: 99.896 + Average Reward: tensor(27.8321) + Best Reward: tensor(187.7788) + Worst Reward: tensor(-36.7093) + Deterministic Reward Evaluation: tensor(154.9915) + Time: 17m 42 + +Policy Number #11: + Average Time Steps: 100.044 + Average Reward: tensor(29.7991) + Best Reward: tensor(191.8518) + Worst Reward: tensor(-37.2897) + Deterministic Reward Evaluation: tensor(80.5144) + Time: 17m 51 + +Policy Number #12: + Average Time Steps: 102.034 + Average Reward: tensor(30.3864) + Best Reward: tensor(183.3815) + Worst Reward: tensor(-39.6892) + Deterministic Reward Evaluation: tensor(169.1824) + Time: 18m 14 + +Policy Number #13: + Average Time Steps: 104.628 + Average Reward: tensor(34.1632) + Best Reward: tensor(251.1391) + Worst Reward: tensor(-33.9023) + Deterministic Reward Evaluation: tensor(86.0289) + Time: 18m 46 + +Policy Number #14: + Average Time Steps: 110.006 + Average Reward: tensor(42.6081) + Best Reward: tensor(191.4285) + Worst Reward: tensor(-31.7138) + Deterministic Reward Evaluation: tensor(175.6570) + Time: 19m 16 + +Policy Number #15: + Average Time Steps: 109.534 + Average Reward: tensor(40.7986) + Best Reward: tensor(234.2542) + Worst Reward: tensor(-37.6262) + Deterministic Reward Evaluation: tensor(90.3845) + Time: 19m 14 + +Policy Number #16: + Average Time Steps: 110.512 + Average Reward: tensor(44.6231) + Best Reward: tensor(191.1996) + Worst Reward: tensor(-34.8936) + Deterministic Reward Evaluation: tensor(175.0175) + Time: 19m 28 + +Policy Number #17: + Average Time Steps: 114.738 + Average Reward: tensor(49.3819) + Best Reward: tensor(244.9445) + Worst Reward: tensor(-37.5522) + Deterministic Reward Evaluation: tensor(92.9966) + Time: 20m 6 + +Policy Number #18: + Average Time Steps: 116.088 + Average Reward: tensor(51.4508) + Best Reward: tensor(191.6422) + Worst Reward: tensor(-30.1739) + Deterministic Reward Evaluation: tensor(177.4778) + Time: 20m 43 + +Policy Number #19: + Average Time Steps: 114.136 + Average Reward: tensor(49.0583) + Best Reward: tensor(257.6959) + Worst Reward: tensor(-36.2068) + Deterministic Reward Evaluation: tensor(94.4184) + Time: 20m 34 + +Policy Number #20: + Average Time Steps: 112.668 + Average Reward: tensor(48.4769) + Best Reward: tensor(194.9408) + Worst Reward: tensor(-27.5304) + Deterministic Reward Evaluation: tensor(183.7022) + Time: 20m 10 + +Policy Number #21: + Average Time Steps: 113.768 + Average Reward: tensor(49.4974) + Best Reward: tensor(261.3761) + Worst Reward: tensor(-23.3093) + Deterministic Reward Evaluation: tensor(178.9386) + Time: 20m 13 + +Policy Number #22: + Average Time Steps: 114.216 + Average Reward: tensor(48.2928) + Best Reward: tensor(260.9836) + Worst Reward: tensor(-22.7130) + Deterministic Reward Evaluation: tensor(190.6056) + Time: 20m 13 + +Policy Number #23: + Average Time Steps: 119.686 + Average Reward: tensor(54.0260) + Best Reward: tensor(260.8665) + Worst Reward: tensor(-25.2288) + Deterministic Reward Evaluation: tensor(189.8301) + Time: 20m 56 + +Policy Number #24: + Average Time Steps: 126.428 + Average Reward: tensor(66.5046) + Best Reward: tensor(265.2766) + Worst Reward: tensor(-23.4465) + Deterministic Reward Evaluation: tensor(255.4890) + Time: 22m 21 + +Policy Number #25: + Average Time Steps: 121.938 + Average Reward: tensor(60.5520) + Best Reward: tensor(266.0904) + Worst Reward: tensor(-25.7212) + Deterministic Reward Evaluation: tensor(190.3151) + Time: 21m 11 + +Policy Number #26: + Average Time Steps: 128.098 + Average Reward: tensor(68.2748) + Best Reward: tensor(265.3528) + Worst Reward: tensor(-24.7772) + Deterministic Reward Evaluation: tensor(255.7828) + Time: 22m 45 + +Policy Number #27: + Average Time Steps: 124.944 + Average Reward: tensor(58.9219) + Best Reward: tensor(268.6259) + Worst Reward: tensor(-24.3695) + Deterministic Reward Evaluation: tensor(256.2360) + Time: 21m 59 + +Policy Number #28: + Average Time Steps: 127.924 + Average Reward: tensor(65.5043) + Best Reward: tensor(267.5689) + Worst Reward: tensor(-25.4167) + Deterministic Reward Evaluation: tensor(193.2034) + Time: 22m 9 + +Policy Number #29: + Average Time Steps: 135.126 + Average Reward: tensor(71.1464) + Best Reward: tensor(265.4519) + Worst Reward: tensor(-24.5174) + Deterministic Reward Evaluation: tensor(236.9192) + Time: 23m 37 + +Policy Number #30: + Average Time Steps: 128.228 + Average Reward: tensor(66.2589) + Best Reward: tensor(265.2388) + Worst Reward: tensor(-30.4472) + Deterministic Reward Evaluation: tensor(193.2180) + Time: 22m 23 + +Policy Number #31: + Average Time Steps: 128.792 + Average Reward: tensor(67.7751) + Best Reward: tensor(266.7886) + Worst Reward: tensor(-25.1603) + Deterministic Reward Evaluation: tensor(257.2142) + Time: 22m 15 + +Policy Number #32: + Average Time Steps: 121.346 + Average Reward: tensor(57.0712) + Best Reward: tensor(271.5403) + Worst Reward: tensor(-27.3279) + Deterministic Reward Evaluation: tensor(193.0520) + Time: 20m 42 + +Policy Number #33: + Average Time Steps: 127.398 + Average Reward: tensor(63.5723) + Best Reward: tensor(263.8064) + Worst Reward: tensor(-28.2051) + Deterministic Reward Evaluation: tensor(260.6054) + Time: 22m 9 + +Policy Number #34: + Average Time Steps: 130.22 + Average Reward: tensor(67.9811) + Best Reward: tensor(269.6071) + Worst Reward: tensor(-23.1876) + Deterministic Reward Evaluation: tensor(191.9536) + Time: 22m 37 + +Policy Number #35: + Average Time Steps: 135.046 + Average Reward: tensor(74.6586) + Best Reward: tensor(262.8191) + Worst Reward: tensor(-25.6290) + Deterministic Reward Evaluation: tensor(238.1738) + Time: 23m 24 + +Policy Number #36: + Average Time Steps: 137.858 + Average Reward: tensor(77.2089) + Best Reward: tensor(270.2506) + Worst Reward: tensor(-22.4746) + Deterministic Reward Evaluation: tensor(192.1620) + Time: 23m 30 + +Policy Number #37: + Average Time Steps: 133.138 + Average Reward: tensor(71.1700) + Best Reward: tensor(267.4257) + Worst Reward: tensor(-18.9972) + Deterministic Reward Evaluation: tensor(256.1472) + Time: 23m 17 + +Policy Number #38: + Average Time Steps: 134.324 + Average Reward: tensor(73.0359) + Best Reward: tensor(268.9559) + Worst Reward: tensor(-16.3622) + Deterministic Reward Evaluation: tensor(255.8402) + Time: 23m 27 + +Policy Number #39: + Average Time Steps: 132.25 + Average Reward: tensor(69.5822) + Best Reward: tensor(267.5328) + Worst Reward: tensor(-19.3233) + Deterministic Reward Evaluation: tensor(263.2840) + Time: 23m 12 + +Policy Number #40: + Average Time Steps: 134.078 + Average Reward: tensor(71.7353) + Best Reward: tensor(340.4381) + Worst Reward: tensor(-17.0706) + Deterministic Reward Evaluation: tensor(192.7283) + Time: 23m 11 + +Policy Number #41: + Average Time Steps: 133.934 + Average Reward: tensor(74.7601) + Best Reward: tensor(267.8867) + Worst Reward: tensor(-18.4499) + Deterministic Reward Evaluation: tensor(264.1262) + Time: 22m 43 + +Policy Number #42: + Average Time Steps: 142.242 + Average Reward: tensor(85.3738) + Best Reward: tensor(271.6832) + Worst Reward: tensor(-15.0875) + Deterministic Reward Evaluation: tensor(255.4032) + Time: 24m 20 + +Policy Number #43: + Average Time Steps: 138.308 + Average Reward: tensor(76.4722) + Best Reward: tensor(268.8214) + Worst Reward: tensor(-17.0959) + Deterministic Reward Evaluation: tensor(235.4065) + Time: 24m 1 + +Policy Number #44: + Average Time Steps: 141.928 + Average Reward: tensor(82.8199) + Best Reward: tensor(340.0343) + Worst Reward: tensor(-13.9999) + Deterministic Reward Evaluation: tensor(255.5421) + Time: 24m 43 + +Policy Number #45: + Average Time Steps: 133.008 + Average Reward: tensor(74.4591) + Best Reward: tensor(270.1023) + Worst Reward: tensor(-12.4596) + Deterministic Reward Evaluation: tensor(267.4449) + Time: 22m 55 + +Policy Number #46: + Average Time Steps: 133.67 + Average Reward: tensor(75.2076) + Best Reward: tensor(332.1006) + Worst Reward: tensor(-14.0184) + Deterministic Reward Evaluation: tensor(237.4824) + Time: 22m 51 + +Policy Number #47: + Average Time Steps: 141.036 + Average Reward: tensor(81.5390) + Best Reward: tensor(271.2858) + Worst Reward: tensor(-15.7507) + Deterministic Reward Evaluation: tensor(241.7335) + Time: 24m 26 + +Policy Number #48: + Average Time Steps: 144.012 + Average Reward: tensor(87.9991) + Best Reward: tensor(340.8509) + Worst Reward: tensor(-11.3358) + Deterministic Reward Evaluation: tensor(255.5053) + Time: 25m 6 + +Policy Number #49: + Average Time Steps: 148.598 + Average Reward: tensor(91.7507) + Best Reward: tensor(342.6179) + Worst Reward: tensor(-10.0001) + Deterministic Reward Evaluation: tensor(255.6338) + Time: 25m 47 + +Policy Number #50: + Average Time Steps: 139.876 + Average Reward: tensor(81.4515) + Best Reward: tensor(277.8659) + Worst Reward: tensor(-11.6284) + Deterministic Reward Evaluation: tensor(258.1944) + Time: 24m 7 + +Policy Number #51: + Average Time Steps: 138.864 + Average Reward: tensor(81.4313) + Best Reward: tensor(275.2506) + Worst Reward: tensor(-9.3933) + Deterministic Reward Evaluation: tensor(256.0809) + Time: 24m 12 + +Policy Number #52: + Average Time Steps: 142.4 + Average Reward: tensor(87.3715) + Best Reward: tensor(341.8409) + Worst Reward: tensor(-8.9317) + Deterministic Reward Evaluation: tensor(257.9247) + Time: 25m 3 + +Policy Number #53: + Average Time Steps: 137.916 + Average Reward: tensor(84.8605) + Best Reward: tensor(338.6187) + Worst Reward: tensor(-8.1720) + Deterministic Reward Evaluation: tensor(258.1808) + Time: 24m 12 + +Policy Number #54: + Average Time Steps: 138.66 + Average Reward: tensor(82.6870) + Best Reward: tensor(340.8794) + Worst Reward: tensor(-6.3355) + Deterministic Reward Evaluation: tensor(259.0560) + Time: 24m 24 + +Policy Number #55: + Average Time Steps: 139.012 + Average Reward: tensor(85.6565) + Best Reward: tensor(351.3536) + Worst Reward: tensor(-2.2574) + Deterministic Reward Evaluation: tensor(257.5242) + Time: 24m 24 + +Policy Number #0: + Average Time Steps: 145.214 + Average Reward: tensor(-7.4743) + Best Reward: tensor(173.4915) + Worst Reward: tensor(-440.9451) + Deterministic Reward Evaluation: tensor(30.4683) + Time: 27m 56 + +Policy Number #1: + Average Time Steps: 145.582 + Average Reward: tensor(-0.7723) + Best Reward: tensor(215.1253) + Worst Reward: tensor(-391.1151) + Deterministic Reward Evaluation: tensor(-22.3279) + Time: 28m 58 + +Policy Number #2: + Average Time Steps: 145.954 + Average Reward: tensor(0.6155) + Best Reward: tensor(210.1866) + Worst Reward: tensor(-435.8777) + Deterministic Reward Evaluation: tensor(82.1853) + Time: 29m 13 + +Policy Number #3: + Average Time Steps: 138.51 + Average Reward: tensor(30.2718) + Best Reward: tensor(264.7868) + Worst Reward: tensor(-378.7314) + Deterministic Reward Evaluation: tensor(140.9172) + Time: 55m 48 + +Policy Number #4: + Average Time Steps: 149.772 + Average Reward: tensor(35.2392) + Best Reward: tensor(255.4454) + Worst Reward: tensor(-270.3944) + Deterministic Reward Evaluation: tensor(141.3073) + Time: 32m 53 + +Policy Number #5: + Average Time Steps: 158.652 + Average Reward: tensor(42.5355) + Best Reward: tensor(257.1924) + Worst Reward: tensor(-331.6367) + Deterministic Reward Evaluation: tensor(152.9267) + Time: 47m 11 + +Policy Number #6: + Average Time Steps: 148.986 + Average Reward: tensor(39.1698) + Best Reward: tensor(266.3555) + Worst Reward: tensor(-320.5730) + Deterministic Reward Evaluation: tensor(204.8249) + Time: 34m 35 + +Policy Number #7: + Average Time Steps: 147.516 + Average Reward: tensor(40.0819) + Best Reward: tensor(266.3633) + Worst Reward: tensor(-256.6609) + Deterministic Reward Evaluation: tensor(203.0340) + Time: 43m 27 + +Policy Number #8: + Average Time Steps: 151.824 + Average Reward: tensor(43.2991) + Best Reward: tensor(262.1538) + Worst Reward: tensor(-313.0535) + Deterministic Reward Evaluation: tensor(203.9398) + Time: 27m 11 + +Policy Number #9: + Average Time Steps: 155.31 + Average Reward: tensor(31.8352) + Best Reward: tensor(298.2962) + Worst Reward: tensor(-447.9696) + Deterministic Reward Evaluation: tensor(204.3748) + Time: 29m 21 + +Policy Number #10: + Average Time Steps: 157.236 + Average Reward: tensor(46.1791) + Best Reward: tensor(313.9221) + Worst Reward: tensor(-278.9856) + Deterministic Reward Evaluation: tensor(154.5100) + Time: 29m 13 + +Policy Number #11: + Average Time Steps: 160.574 + Average Reward: tensor(46.0216) + Best Reward: tensor(260.2269) + Worst Reward: tensor(-292.5306) + Deterministic Reward Evaluation: tensor(207.6479) + Time: 29m 33 + +Policy Number #12: + Average Time Steps: 202.89 + Average Reward: tensor(42.6369) + Best Reward: tensor(249.6396) + Worst Reward: tensor(-456.7430) + Deterministic Reward Evaluation: tensor(172.2585) + Time: 37m 25 + +Policy Number #13: + Average Time Steps: 197.002 + Average Reward: tensor(48.0090) + Best Reward: tensor(274.5155) + Worst Reward: tensor(-400.7305) + Deterministic Reward Evaluation: tensor(252.7695) + Time: 36m 11 + +Policy Number #14: + Average Time Steps: 198.448 + Average Reward: tensor(45.2531) + Best Reward: tensor(305.1073) + Worst Reward: tensor(-516.1928) + Deterministic Reward Evaluation: tensor(236.9256) + Time: 36m 22 + +Policy Number #15: + Average Time Steps: 200.884 + Average Reward: tensor(56.1958) + Best Reward: tensor(296.3640) + Worst Reward: tensor(-414.7282) + Deterministic Reward Evaluation: tensor(241.8818) + Time: 36m 37 + +Policy Number #16: + Average Time Steps: 194.67 + Average Reward: tensor(76.6455) + Best Reward: tensor(312.9175) + Worst Reward: tensor(-30.9069) + Deterministic Reward Evaluation: tensor(245.0537) + Time: 33m 27 + +Policy Number #17: + Average Time Steps: 195.402 + Average Reward: tensor(71.6556) + Best Reward: tensor(362.4413) + Worst Reward: tensor(-55.9550) + Deterministic Reward Evaluation: tensor(247.2440) + Time: 33m 49 + +Policy Number #18: + Average Time Steps: 199.25 + Average Reward: tensor(73.5901) + Best Reward: tensor(362.8979) + Worst Reward: tensor(-60.0774) + Deterministic Reward Evaluation: tensor(249.0204) + Time: 34m 47 + +Policy Number #19: + Average Time Steps: 213.658 + Average Reward: tensor(82.6966) + Best Reward: tensor(365.0646) + Worst Reward: tensor(-68.5418) + Deterministic Reward Evaluation: tensor(244.1972) + Time: 37m 16 + +Policy Number #20: + Average Time Steps: 219.208 + Average Reward: tensor(84.3355) + Best Reward: tensor(363.4833) + Worst Reward: tensor(-90.5358) + Deterministic Reward Evaluation: tensor(249.0656) + Time: 38m 11 + +Policy Number #21: + Average Time Steps: 232.402 + Average Reward: tensor(87.9588) + Best Reward: tensor(320.4059) + Worst Reward: tensor(-79.3426) + Deterministic Reward Evaluation: tensor(255.5874) + Time: 40m 19 + +Policy Number #22: + Average Time Steps: 254.748 + Average Reward: tensor(102.8290) + Best Reward: tensor(362.6034) + Worst Reward: tensor(-58.5688) + Deterministic Reward Evaluation: tensor(251.6808) + Time: 43m 50 + +Policy Number #23: + Average Time Steps: 255.688 + Average Reward: tensor(100.7968) + Best Reward: tensor(369.8500) + Worst Reward: tensor(-139.9914) + Deterministic Reward Evaluation: tensor(315.7679) + Time: 43m 55 + +Policy Number #24: + Average Time Steps: 266.752 + Average Reward: tensor(107.9183) + Best Reward: tensor(366.4199) + Worst Reward: tensor(-80.5194) + Deterministic Reward Evaluation: tensor(328.9924) + Time: 46m 4 + +Policy Number #25: + Average Time Steps: 283.872 + Average Reward: tensor(118.1459) + Best Reward: tensor(389.3860) + Worst Reward: tensor(-85.2897) + Deterministic Reward Evaluation: tensor(303.6195) + Time: 48m 48 + +Policy Number #26: + Average Time Steps: 293.844 + Average Reward: tensor(120.4765) + Best Reward: tensor(388.1024) + Worst Reward: tensor(-69.2030) + Deterministic Reward Evaluation: tensor(275.8389) + Time: 50m 35 + +Policy Number #27: + Average Time Steps: 302.354 + Average Reward: tensor(124.9276) + Best Reward: tensor(398.6852) + Worst Reward: tensor(-72.2664) + Deterministic Reward Evaluation: tensor(338.9153) + Time: 52m 37 + +Policy Number #28: + Average Time Steps: 308.248 + Average Reward: tensor(124.9630) + Best Reward: tensor(401.7516) + Worst Reward: tensor(-77.2973) + Deterministic Reward Evaluation: tensor(283.0857) + Time: 104m 53 + +Policy Number #29: + Average Time Steps: 323.362 + Average Reward: tensor(143.7176) + Best Reward: tensor(456.0289) + Worst Reward: tensor(-78.3056) + Deterministic Reward Evaluation: tensor(339.2289) + Time: 53m 46 + +Policy Number #30: + Average Time Steps: 321.636 + Average Reward: tensor(140.1091) + Best Reward: tensor(441.2586) + Worst Reward: tensor(-75.3051) + Deterministic Reward Evaluation: tensor(347.6123) + Time: 54m 55 + +Policy Number #31: + Average Time Steps: 353.61 + Average Reward: tensor(173.7844) + Best Reward: tensor(455.0985) + Worst Reward: tensor(-72.6004) + Deterministic Reward Evaluation: tensor(343.6443) + Time: 59m 30 + +Policy Number #32: + Average Time Steps: 344.968 + Average Reward: tensor(180.3106) + Best Reward: tensor(450.9888) + Worst Reward: tensor(-70.7332) + Deterministic Reward Evaluation: tensor(351.3824) + Time: 58m 47 + +Policy Number #33: + Average Time Steps: 346.722 + Average Reward: tensor(192.6161) + Best Reward: tensor(463.1447) + Worst Reward: tensor(-11.2309) + Deterministic Reward Evaluation: tensor(350.2022) + Time: 58m 55 + +Policy Number #34: + Average Time Steps: 349.434 + Average Reward: tensor(199.6353) + Best Reward: tensor(470.0817) + Worst Reward: tensor(-11.9648) + Deterministic Reward Evaluation: tensor(356.5537) + Time: 59m 12 + +Policy Number #35: + Average Time Steps: 337.252 + Average Reward: tensor(198.2566) + Best Reward: tensor(474.5852) + Worst Reward: tensor(-69.1619) + Deterministic Reward Evaluation: tensor(356.8981) + Time: 58m 4 + +Policy Number #36: + Average Time Steps: 348.766 + Average Reward: tensor(213.1413) + Best Reward: tensor(477.8542) + Worst Reward: tensor(-13.5714) + Deterministic Reward Evaluation: tensor(357.7015) + Time: 59m 19 + diff --git a/data/ppo/trajectories/best_rollout#37.pkl b/data/ppo/trajectories/best_rollout#37.pkl new file mode 100644 index 0000000..e6c284c Binary files /dev/null and b/data/ppo/trajectories/best_rollout#37.pkl differ diff --git a/data/ppo/trajectories/best_rollout#56.pkl b/data/ppo/trajectories/best_rollout#56.pkl new file mode 100644 index 0000000..c0c77a9 Binary files /dev/null and b/data/ppo/trajectories/best_rollout#56.pkl differ diff --git a/data/ppo/trajectories/worst_rollout#37.pkl b/data/ppo/trajectories/worst_rollout#37.pkl new file mode 100644 index 0000000..128619a Binary files /dev/null and b/data/ppo/trajectories/worst_rollout#37.pkl differ diff --git a/data/ppo/trajectories/worst_rollout#56.pkl b/data/ppo/trajectories/worst_rollout#56.pkl new file mode 100644 index 0000000..53aa41d Binary files /dev/null and b/data/ppo/trajectories/worst_rollout#56.pkl differ diff --git a/learning/autopilot.py b/learning/autopilot.py index 5cd3f47..dd478fa 100644 --- a/learning/autopilot.py +++ b/learning/autopilot.py @@ -4,8 +4,13 @@ from tensordict.nn.distributions import NormalParamExtractor from tensordict.nn import TensorDictModule, InteractionType from torchrl.modules import ProbabilisticActor, TanhNormal, ValueOperator +from torch.distributions import Categorical +from tensordict.nn import CompositeDistribution +from learning.utils import CategoricalControlsExtractor import os +from shared import ELEVATOR_CLAMP + """ An autopilot learner. It takes the form of a policy network that outputs actions for a given state. @@ -20,7 +25,7 @@ def __init__(self): - 3 angular velocities: w_roll, w_pitch, w_yaw - 3d relative position of next waypoint: wx, wy, wz """ - self.inputs = 13 + self.inputs = 15 """ Action: @@ -36,7 +41,7 @@ def __init__(self): nn.Linear(self.inputs, self.inputs), nn.ReLU(), nn.Linear(self.inputs, self.outputs), - nn.Sigmoid(), + nn.Tanh(), ) # Returns the action selected and 0, representing the log-prob of the @@ -50,10 +55,10 @@ def get_control(self, action): Clamps various control outputs and sets the mean for control surfaces to 0. Assumes [action] is a 4-item tensor of throttle, aileron cmd, elevator cmd, rudder cmd. """ - action[0] = 0.5 * action[0] - action[1] = 0.1 * (action[1] - 0.5) - action[2] = 0.5 * (action[2] - 0.5) - action[3] = 0.5 * (action[3] - 0.5) + action[0] = 0.8 * (0.5*(action[0] + 1)) + action[1] = 0.1 * action[1] + action[2] = ELEVATOR_CLAMP * action[2] + action[3] = 0.1 * action[3] return action # flattened_params = flattened dx1 numpy array of all params to init from @@ -82,12 +87,11 @@ def init_from_params(self, flattened_params): layer1, nn.ReLU(), layer2, - nn.Sigmoid(), + nn.Tanh() ) - # Loads the network from dir/name.pth - def init_from_saved(self, dir, name): - path = os.path.join(dir, name + '.pth') + # Loads the network from path + def init_from_saved(self, path): self.policy_network = torch.load(path) # Saves the network to dir/name.pth @@ -125,8 +129,8 @@ def transform_from_deterministic_learner(self): self.policy_network[-2] = nn.Linear(self.inputs, self.outputs * 2) # Update the output layer to include the original weights and biases - self.policy_network[-2].weight = nn.Parameter(torch.cat((w, torch.ones(w.shape) * 100), 0)) - self.policy_network[-2].bias = nn.Parameter(torch.cat((b, torch.ones(b.shape) * 100), 0)) + self.policy_network[-2].weight = nn.Parameter(torch.cat((w, torch.zeros(w.shape)), 0)) + self.policy_network[-2].bias = nn.Parameter(torch.cat((b, torch.zeros(b.shape)), 0)) # Add a normal param extractor to the network to extract (means, sigmas) tuple self.policy_network.append(NormalParamExtractor()) @@ -141,10 +145,6 @@ def transform_from_deterministic_learner(self): module=policy_module, in_keys=["loc", "scale"], distribution_class=TanhNormal, - distribution_kwargs={ - "min": 0, # minimum control - "max": 1, # maximum control - }, default_interaction_type=InteractionType.RANDOM, return_log_prob=True, ) @@ -153,7 +153,6 @@ def transform_from_deterministic_learner(self): def get_action(self, observation): data = TensorDict({"observation": observation}, []) policy_forward = self.policy_module(data) - print("action", self.policy_network(observation)) return policy_forward["action"], policy_forward["sample_log_prob"] # NOTE: This initializes from *deterministic* learner parameters and picks @@ -173,10 +172,120 @@ def init_from_saved(self, path): module=policy_module, in_keys=["loc", "scale"], distribution_class=TanhNormal, - distribution_kwargs={ - "min": 0, # minimum control - "max": 1, # maximum control - }, default_interaction_type=InteractionType.RANDOM, return_log_prob=True, ) + +""" +Slew rate autopilot learner. +Each control (throttle/aileron/elevator/rudder) has three options: +stay constant, go down, or go up. +""" +class SlewRateAutopilotLearner: + def __init__(self): + self.inputs = 15 + self.outputs = 4 + + # Slew rates are wrt sim clock + self.throttle_slew_rate = 0.0005 + self.aileron_slew_rate = 0.0001 + self.elevator_slew_rate = 0.0001 + self.rudder_slew_rate = 0.0001 + + self.policy_network = nn.Sequential( + nn.Linear(self.inputs, self.inputs), + nn.ReLU(), + nn.Linear(self.inputs, 3 * self.outputs), + nn.Sigmoid(), + CategoricalControlsExtractor() + ) + + self.instantiate_policy_module() + + def instantiate_policy_module(self): + policy_module = TensorDictModule(self.policy_network, in_keys=["observation"], out_keys=[("params", "throttle", "probs"),("params", "aileron", "probs"),("params", "elevator", "probs"),("params", "rudder", "probs")]) + self.policy_module = policy_module = ProbabilisticActor( + module=policy_module, + in_keys=["params"], + distribution_class=CompositeDistribution, + distribution_kwargs={ + "distribution_map": { + "throttle": Categorical, + "aileron": Categorical, + "elevator": Categorical, + "rudder": Categorical, + } + }, + default_interaction_type=InteractionType.RANDOM, + return_log_prob=True + ) + + # Returns the action selected and the log_prob of that action + def get_action(self, observation): + data = TensorDict({"observation": observation}, []) + throttle_probs, aileron_probs, elevator_probs, rudder_probs = self.policy_network(observation) + policy_forward = self.policy_module(data) + # print("aileron pos", observation[12], "aileron probs", aileron_probs / torch.sum(aileron_probs)) + + action = torch.Tensor([policy_forward['throttle'], policy_forward['aileron'], policy_forward['elevator'], policy_forward['rudder']]) + return action, policy_forward["sample_log_prob"] + + # Always samples the mode of the output for each control + def get_deterministic_action(self, observation): + throttle_probs, aileron_probs, elevator_probs, rudder_probs = self.policy_network(observation) + action = torch.Tensor([torch.argmax(throttle_probs), torch.argmax(aileron_probs), torch.argmax(elevator_probs), torch.argmax(rudder_probs)]) + # print("aileron pos", observation[12], "aileron probs", aileron_probs / torch.sum(aileron_probs)) + w = self.policy_network[2].weight + # print(f"Reward min: {torch.min(w)}, mean: {torch.mean(w)}, median: {torch.median(w)}, max: {torch.max(w)}, std: {torch.std(w)}" ) + return action, 0 + + # Apply a -1 transformation to the action to create control tensor such that: + # -1 means go down, 0 means stay same, and +1 means go up + def get_control(self, action): + return action - 1 + + # flattened_params = flattened dx1 numpy array of all params to init from + # NOTE: the way the params are broken up into the weights/biases of each layer + # would need to be manually edited for changes in network architecture + def init_from_params(self, flattened_params): + flattened_params = torch.from_numpy(flattened_params).to(torch.float32) + + pl, pr = 0, 0 + layer1 = nn.Linear(self.inputs, self.inputs) + pr += layer1.weight.nelement() + layer1.weight = nn.Parameter(flattened_params[pl:pr].reshape(layer1.weight.shape)) + pl = pr + pr += layer1.bias.nelement() + layer1.bias = nn.Parameter(flattened_params[pl:pr].reshape(layer1.bias.shape)) + + layer2 = nn.Linear(self.inputs, 3*self.outputs) + pl = pr + pr += layer2.weight.nelement() + layer2.weight = nn.Parameter(flattened_params[pl:pr].reshape(layer2.weight.shape)) + pl = pr + pr += layer2.bias.nelement() + layer2.bias = nn.Parameter(flattened_params[pl:pr].reshape(layer2.bias.shape)) + + self.policy_network = nn.Sequential( + layer1, + nn.ReLU(), + layer2, + nn.Sigmoid(), + CategoricalControlsExtractor() + ) + self.instantiate_policy_module() + + # Loads the network from path + def init_from_saved(self, path): + self.policy_network = torch.load(path) + self.instantiate_policy_module() + + # Saves the network to dir/name.pth + def save(self, dir, name): + path = os.path.join(dir, name + '.pth') + torch.save(self.policy_network, path) + + + + + diff --git a/learning/crossentropy.py b/learning/crossentropy.py index 677fcf1..a7ab64a 100644 --- a/learning/crossentropy.py +++ b/learning/crossentropy.py @@ -1,6 +1,6 @@ import torch import numpy as np -from learning.autopilot import AutopilotLearner +from learning.autopilot import AutopilotLearner, SlewRateAutopilotLearner from simulation.simulate import FullIntegratedSim from simulation.jsbsim_aircraft import x8 import os @@ -25,7 +25,7 @@ def __init__(self, learners, num_params): def init_using_torch_default(generation_size, num_params): learners = [] for i in range(generation_size): - learners.append(AutopilotLearner()) + learners.append(SlewRateAutopilotLearner()) return Generation(learners, num_params) # Utilizes [rewards], which contains the reward obtained by each learner as a @@ -89,17 +89,16 @@ def make_new_generation(mean, cov, generation_size, num_params): # Generate each learner from params learners = [] for param_list in selected_params: - l = AutopilotLearner() + l = SlewRateAutopilotLearner() l.init_from_params(param_list) learners.append(l) return Generation(learners, num_params) -def cross_entropy_train(epochs, generation_size, num_survive, num_params=238, sim_time=60.0, save_dir='cross_entropy'): +def cross_entropy_train(epochs, generation_size, num_survive, num_params=432, sim_time=60.0, save_dir='cross_entropy'): # Create save_dir (and if one already exists, rename it with some rand int) - if os.path.exists(os.path.join('data', save_dir)): - os.rename(os.path.join('data', save_dir), os.path.join('data', save_dir + '_old' + str(randint(0, 100000)))) - os.mkdir(os.path.join('data', save_dir)) - stats_file = open(os.path.join('data', save_dir, 'stats.txt'), 'w') + # if os.path.exists(os.path.join('data', save_dir)): + # os.rename(os.path.join('data', save_dir), os.path.join('data', save_dir + '_old' + str(randint(0, 100000)))) + # os.mkdir(os.path.join('data', save_dir)) # Baseline to be updated after first generation mean = np.zeros((num_params)) @@ -117,7 +116,7 @@ def cross_entropy_train(epochs, generation_size, num_survive, num_params=238, si # Evaluate generation through rollouts rewards = [] for i in range(len(generation.learners)): - id = str(100*(epoch+1) + (i+1)) + id = str(1000*(epoch+1) + (i+1)) learner = generation.learners[i] # Run simulation to evaluate learner @@ -127,7 +126,7 @@ def cross_entropy_train(epochs, generation_size, num_survive, num_params=238, si integrated_sim.simulation_loop() # Acquire/save data - integrated_sim.mdp_data_collector.save(os.path.join(save_dir, 'generation' + str(epoch+1)), 'trajectory_learner#' + str(i+1)) + #integrated_sim.mdp_data_collector.save(os.path.join(save_dir, 'generation' + str(epoch+1)), 'trajectory_learner#' + str(i+1)) rewards.append(integrated_sim.mdp_data_collector.get_cum_reward()) print('Reward for Learner #', id, ': ', integrated_sim.mdp_data_collector.get_cum_reward()) @@ -140,13 +139,15 @@ def cross_entropy_train(epochs, generation_size, num_survive, num_params=238, si cov += 0.01 * np.identity(mean.shape[0]) # Save important info in the save_dir stats file + stats_file = open(os.path.join('data', save_dir, 'stats.txt'), 'a') stats_file.write('Generation #' + str(epoch+1) + ':\n') stats_file.write('Best, Median, and Worst Learner: ' + str(ids) + '\n') stats_file.write('Best, Median, and Worst Reward: ' + str(rew) + '\n') stats_file.write('\n\n\n') + stats_file.close() if __name__ == "__main__": os.environ["JSBSIM_DEBUG"]=str(0) # epochs, generation_size, num_survive - cross_entropy_train(100, 99, 50) \ No newline at end of file + cross_entropy_train(100, 200, 50) \ No newline at end of file diff --git a/learning/ppo.py b/learning/ppo.py index e012f67..cf3851c 100644 --- a/learning/ppo.py +++ b/learning/ppo.py @@ -7,11 +7,18 @@ from torchrl.data.replay_buffers.storages import LazyTensorStorage from torchrl.objectives import ClipPPOLoss from torchrl.objectives.value import GAE -from learning.autopilot import StochasticAutopilotLearner +from learning.autopilot import StochasticAutopilotLearner, SlewRateAutopilotLearner from simulation.simulate import FullIntegratedSim from simulation.jsbsim_aircraft import x8 +import numpy as np import os +import time +import statistics +import beepy as beep +device = "cpu" if not torch.has_cuda else "cuda:0" + +start_id = 36 """ Gathers rollout data and returns it in the way the Proximal Policy Optimization loss_module expects """ @@ -24,36 +31,88 @@ def gather_rollout_data(autopilot_learner, policy_num, num_trajectories=100, sim sample_log_probs = torch.empty(0) rewards = torch.empty(0) dones = torch.empty(0, dtype=torch.bool) + best_cum_reward = -float('inf') + worst_cum_reward = float('inf') + total_cum_reward = 0 + total_timesteps = 0 for t in range(num_trajectories): - integrated_sim = FullIntegratedSim(x8, autopilot_learner, sim_time) + # Reset distribution + print(f"Trajectory #{t+1}") + rand = np.random.rand() + if rand <= 0.25: + in_flight_reset = 0 + elif rand <= 0.5: + in_flight_reset = 3 + elif rand <= 0.7: + in_flight_reset = 4 + elif rand <= 0.85: + in_flight_reset = 5 + else: + in_flight_reset = 6 + #in_flight_reset = 0 + # Run a sim with a stochastic version of the autopilot so it can explore + integrated_sim = FullIntegratedSim(x8, autopilot_learner, sim_time, in_flight_reset=in_flight_reset, auto_deterministic=False) integrated_sim.simulation_loop() # Acquire data observation, next_observation, action, sample_log_prob, reward, done = integrated_sim.mdp_data_collector.get_trajectory_data() + cum_reward = integrated_sim.mdp_data_collector.get_cum_reward() + total_cum_reward += cum_reward + total_timesteps += reward.shape[0] - # Save data - # integrated_sim.mdp_data_collector.save(os.path.join('ppo', 'trajectories'), 'rollout' + str(policy_num * num_trajectories + t)) + # Save trajectory if worst or best so far + if cum_reward > best_cum_reward: + integrated_sim.mdp_data_collector.save(os.path.join('ppo', 'trajectories'), 'best_rollout#' + str(policy_num)) + best_cum_reward = cum_reward + if cum_reward < worst_cum_reward: + integrated_sim.mdp_data_collector.save(os.path.join('ppo', 'trajectories'), 'worst_rollout#' + str(policy_num)) + worst_cum_reward = cum_reward # Add to the data observations = torch.cat((observations, observation)) next_observations = torch.cat((next_observations, next_observation)) actions = torch.cat((actions, action)) sample_log_probs = torch.cat((sample_log_probs,sample_log_prob)) - rewards = torch.cat((rewards, reward)) + rewards = torch.cat((rewards, reward)) dones = torch.cat((dones, done)) data_size += observation.shape[0] - + + # we divide by std for reward scaling + print("Reward scale", torch.std(rewards)) + rewards /= torch.std(rewards) + print(f"Reward min: {torch.min(rewards)}, mean: {torch.mean(rewards)}, median: {torch.median(rewards)}, max: {torch.max(rewards)}" ) # Each entry tensor should be data_size x d where d is the dimension of # that entry for one step in a rollout. data = TensorDict({ "observation": observations, - "action": actions.detach(), + "action": TensorDict({ + "throttle": actions[:,0].detach(), + "aileron": actions[:,1].detach(), + "elevator": actions[:,2].detach(), + "rudder": actions[:,3].detach(), + # "throttle_log_prob": torch.zeros(data_size).detach(), + # "aileron_log_prob": torch.zeros(data_size).detach(), + # "elevator_log_prob": torch.zeros(data_size).detach(), + # "rudder_log_prob": torch.zeros(data_size).detach(), + }, [data_size,]), "sample_log_prob": sample_log_probs.detach(), # log probability that each action was selected + # "sample_log_prob": TensorDict({ + # }, [data_size,]), # log probability that each action was selected ("next", "done"): dones, ("next", "terminated"): dones, ("next", "reward"): rewards, ("next", "observation"): next_observations, }, [data_size,]) + torch.save(data, os.path.join('data', 'ppo', 'all_data.pkl')) + + # Write to stats file + stats_file = open(os.path.join('data', 'ppo', 'stats.txt'), 'a') + stats_file.write('Policy Number #' + str(policy_num) + ':\n') + stats_file.write('\tAverage Time Steps: ' + str(total_timesteps/num_trajectories) + '\n') + stats_file.write('\tAverage Reward: ' + str(total_cum_reward/num_trajectories) + '\n') + stats_file.write('\tBest Reward: ' + str(best_cum_reward) + '\n') + stats_file.write('\tWorst Reward: ' + str(worst_cum_reward) + '\n') + stats_file.close() return data, data_size @@ -64,12 +123,12 @@ def gather_rollout_data(autopilot_learner, policy_num, num_trajectories=100, sim action taken during PPO learning. """ def make_value_estimator_module(n_obs): + global device + value_net = nn.Sequential( - nn.Linear(n_obs, n_obs), + nn.Linear(n_obs, 2*n_obs, device=device), nn.Tanh(), - nn.Linear(n_obs, n_obs), - nn.Tanh(), - nn.Linear(n_obs, 1), # one value is computed for the state + nn.Linear(2*n_obs, 1, device=device), # one value is computed for the state ) return ValueOperator( @@ -95,7 +154,12 @@ def make_value_estimator_module(n_obs): def train_ppo_once(policy_num, autopilot_learner, loss_module, advantage_module, optimizer, num_trajectories, num_epochs, batch_size): # Rollout policy to gather trajectories dataset, data_size = gather_rollout_data(autopilot_learner, policy_num, num_trajectories) - + dataset = torch.load(os.path.join("data","ppo","all_data.pkl")) + print(dataset) + # data_size = 56114 + # print(max(dataset.get(("next", "reward")))) + # print(dataset) + # data_size = 9484 # Define replay buffer to store trajectory data in during training replay_buffer = ReplayBuffer( storage=LazyTensorStorage(data_size), @@ -103,7 +167,7 @@ def train_ppo_once(policy_num, autopilot_learner, loss_module, advantage_module, ) for epoch in range(num_epochs): - print("Training Epoch", epoch+1) + # print("Training Epoch", epoch+1) # Re-compute advantage at each epoch as its value depends on the value # network which is updated in the inner loop with torch.no_grad(): @@ -112,36 +176,75 @@ def train_ppo_once(policy_num, autopilot_learner, loss_module, advantage_module, # Gather the data into a replay buffer for sampling data_view = dataset.reshape(-1) replay_buffer.extend(data_view.cpu()) + batch_loss = 0 + batch_obj_loss = 0 + batch_critic_loss = 0 for b in range(0, data_size // batch_size): # Gather batch and calculate PPO loss on it batch = replay_buffer.sample(batch_size) - loss_vals = loss_module(batch.to("cpu")) + loss_vals = loss_module(batch.to(device)) loss_value = ( loss_vals["loss_objective"] + loss_vals["loss_critic"] - + loss_vals["loss_entropy"] ) + batch_loss += loss_value * batch_size + batch_obj_loss += loss_vals["loss_objective"] * batch_size + batch_critic_loss += loss_vals["loss_critic"] * batch_size # Optimize via gradient descent with the optimizer loss_value.backward() + + torch.nn.utils.clip_grad_norm_(loss_module.parameters(), 1.0) optimizer.step() optimizer.zero_grad() + if (epoch+1) % 10 == 0: + print('loss_value in epoch #', epoch+1, ':', float(batch_loss/ 1000), '(objective:', float(batch_obj_loss/ 1000), ' critic: ', float(batch_critic_loss/ 1000), ')') if __name__ == "__main__": + # Parameters (see https://pytorch.org/rl/tutorials/coding_ppo.html#ppo-parameters) - batch_size = 100 - num_epochs = 10 - clip_epsilon = 0.2 # for PPO loss - gamma = 1.0 + batch_size = 256 + num_epochs = 100 + device = "cpu" if not torch.has_cuda else "cuda:0" + + clip_epsilon = 0.2 #0.08 # for PPO loss + gamma = 0.99 # keep 1.0 lmbda = 0.95 - entropy_eps = 1e-4 - lr = 3e-4 - num_trajectories = 100 + # entropy_eps = 1e-4 + lr = 5e-5 + + # lr = 5e-5 +# loss_value in epoch # 1 : 67.5849609375 (objective: -0.002501344308257103 critic: 67.58744049072266 ) +# loss_value in epoch # 2 : 66.79192352294922 (objective: -0.022511614486575127 critic: 66.81440734863281 ) +# loss_value in epoch # 3 : 65.99295806884766 (objective: -0.023394417017698288 critic: 66.01636505126953 ) +# loss_value in epoch # 10 : 63.2775764465332 (objective: -0.03471797704696655 critic: 63.31227493286133 ) +# loss_value in epoch # 20 : 58.33335876464844 (objective: -0.03877483680844307 critic: 58.37212371826172 ) +# loss_value in epoch # 30 : 53.69935607910156 (objective: -0.04969186335802078 critic: 53.749061584472656 ) +# loss_value in epoch # 40 : 48.16702651977539 (objective: -0.048812463879585266 critic: 48.21586227416992 ) +# loss_value in epoch # 50 : 42.11222839355469 (objective: -0.06517862528562546 critic: 42.17741394042969 ) +# loss_value in epoch # 60 : 36.78435134887695 (objective: -0.06043088063597679 critic: 36.844749450683594 ) +# loss_value in epoch # 70 : 33.0183219909668 (objective: -0.062020089477300644 critic: 33.08034133911133 ) +# loss_value in epoch # 80 : 32.66770935058594 (objective: -0.05988125503063202 critic: 32.72758865356445 ) +# loss_value in epoch # 90 : 32.728004455566406 (objective: -0.04434733837842941 critic: 32.772335052490234 ) +# loss_value in epoch # 100 : 32.90699768066406 (objective: -0.03433294966816902 critic: 32.94131851196289 ) + + + + # 3e-5: roughly 160 (27.3, 133) + # 1e-4: roughly 190 (27, 133) + # 1e-4: roughly 190 (27, 133) + + num_trajectories = 500 num_policy_iterations = 100 # Build the modules - autopilot_learner = StochasticAutopilotLearner() + #autopilot_learner = StochasticAutopilotLearner() + autopilot_learner = SlewRateAutopilotLearner() + autopilot_learner.init_from_saved(os.path.join("data", "ppo", "policies", f"learner#{start_id}.pth")) + #autopilot_learner.init_from_saved(os.path.join("data", "ppo", "policies", "learner#56.pth")) + #autopilot_learner.init_from_saved(os.path.join("data", "cross_entropy", "generation14", "learner#67.pth")) + # autopilot_learner.init_from_params(np.random.normal(0, 1, 350)) value_module = make_value_estimator_module(autopilot_learner.inputs) advantage_module = GAE( gamma=gamma, lmbda=lmbda, value_network=value_module, average_gae=True @@ -150,8 +253,8 @@ def train_ppo_once(policy_num, autopilot_learner, loss_module, advantage_module, actor=autopilot_learner.policy_module, critic=value_module, clip_epsilon=clip_epsilon, - entropy_bonus=bool(entropy_eps), - entropy_coef=entropy_eps, + entropy_bonus=False, + entropy_coef=0, value_target_key=advantage_module.value_target_key, critic_coef=1.0, gamma=gamma, @@ -159,7 +262,21 @@ def train_ppo_once(policy_num, autopilot_learner, loss_module, advantage_module, ) optimizer = torch.optim.Adam(loss_module.parameters(), lr) - autopilot_learner.save(os.path.join('data', 'ppo', 'policies'), 'learner#0') - for i in range(num_policy_iterations): + #autopilot_learner.save(os.path.join('data', 'ppo', 'policies'), 'learner#0') + for i in range(start_id, num_policy_iterations): + start = time.time() train_ppo_once(i, autopilot_learner, loss_module, advantage_module, optimizer, num_trajectories, num_epochs, batch_size) - autopilot_learner.save(os.path.join('data', 'ppo', 'policies'), 'learner#' + str(i+1)) \ No newline at end of file + autopilot_learner.save(os.path.join('data', 'ppo', 'policies'), 'learner#' + str(i+1)) + + # Evaluate the learner deterministically + integrated_sim = FullIntegratedSim(x8, autopilot_learner, 60.0, in_flight_reset=0, auto_deterministic=True) + integrated_sim.simulation_loop() + cum_reward = integrated_sim.mdp_data_collector.get_cum_reward() + stats_file = open(os.path.join('data', 'ppo', 'stats.txt'), 'a') + stats_file.write('\tDeterministic Reward Evaluation: ' + str(cum_reward)) + end = time.time() + duration = end - start + stats_file.write(f'\n\tTime: {str(int(duration//60))}m {str(int(duration) %60)}') + stats_file.write('\n\n') + stats_file.close() + beep.beep(2) \ No newline at end of file diff --git a/learning/utils.py b/learning/utils.py new file mode 100644 index 0000000..a09368f --- /dev/null +++ b/learning/utils.py @@ -0,0 +1,12 @@ +import torch +import torch.nn as nn + +class CategoricalControlsExtractor(nn.Module): + def __init__(self): + super().__init__() + + def forward(self, *tensors: torch.Tensor) -> tuple[torch.Tensor, ...]: + tensor, *others = tensors + throttle, aileron, elevator, rudder = tensor.chunk(4, -1) + return (throttle, aileron, elevator, rudder, *others) + #return (nn.functional.softmax(throttle), nn.functional.softmax(aileron), nn.functional.softmax(elevator), nn.functional.softmax(rudder), *others) diff --git a/scripts/replay_trajectory.py b/scripts/replay_trajectory.py index 80eda45..9f57178 100644 --- a/scripts/replay_trajectory.py +++ b/scripts/replay_trajectory.py @@ -12,7 +12,7 @@ import sys import os import torch -from learning.autopilot import AutopilotLearner +from learning.autopilot import AutopilotLearner, SlewRateAutopilotLearner from simulation.simulate import FullIntegratedSim from simulation.jsbsim_aircraft import x8 @@ -24,9 +24,15 @@ # Get agent interaction frequency (manually entered, must be correct for same trajectory) agent_freq = int(sys.argv[2]) +# Get in-flight reset +in_flight_reset = int(sys.argv[3]) + # Load states, actions, rewards states, actions, rewards = torch.load(file) +print('Altitude', states[:,0]) +print('WP Dist', states[:,8]) +#print('Init Controls', states[0,11:]) # Play sim -integrated_sim = FullIntegratedSim(x8, AutopilotLearner(), 60.0, agent_interaction_frequency=agent_freq) +integrated_sim = FullIntegratedSim(x8, SlewRateAutopilotLearner(), 60.0, agent_interaction_frequency=agent_freq, in_flight_reset=in_flight_reset) integrated_sim.simulation_replay(actions) \ No newline at end of file diff --git a/scripts/sim_autopilot.py b/scripts/sim_autopilot.py index c54daef..5aa8e61 100644 --- a/scripts/sim_autopilot.py +++ b/scripts/sim_autopilot.py @@ -6,7 +6,7 @@ and rolls out a trajectory in sim for that autopilot learner. """ -from learning.autopilot import AutopilotLearner +from learning.autopilot import AutopilotLearner, SlewRateAutopilotLearner from simulation.simulate import FullIntegratedSim from simulation.jsbsim_aircraft import x8 import sys @@ -16,10 +16,13 @@ policy_file_path = sys.argv[1] file = os.path.join(*(["data"] + policy_file_path.split('/'))) +# Get in-flight reset +in_flight_reset = int(sys.argv[2]) + # Initialize autopilot -autopilot = AutopilotLearner() +autopilot = SlewRateAutopilotLearner() autopilot.init_from_saved(file) # Play sim -integrated_sim = FullIntegratedSim(x8, autopilot, 60.0) +integrated_sim = FullIntegratedSim(x8, autopilot, 60.0, auto_deterministic=True, in_flight_reset=in_flight_reset) integrated_sim.simulation_loop() \ No newline at end of file diff --git a/shared.py b/shared.py index 7a6b613..e506330 100644 --- a/shared.py +++ b/shared.py @@ -1,5 +1,10 @@ import sys, os +THROTTLE_CLAMP = 0.5 +AILERON_CLAMP = 0.1 +ELEVATOR_CLAMP = 0.1 +RUDDER_CLAMP = 0.1 + """ Context manager object that helps prevent print outputs from 3rd party software. """ diff --git a/simulation/jsbsim_aircraft.py b/simulation/jsbsim_aircraft.py index 98f607e..e41e3f8 100644 --- a/simulation/jsbsim_aircraft.py +++ b/simulation/jsbsim_aircraft.py @@ -22,4 +22,4 @@ def get_cruise_speed_fps(self) -> float: x8 = Aircraft('x8', 'Skywalker x8', 20) -# x8 = Aircraft('c172p', 'Cessna172P', 120) +#x8 = Aircraft('c172p', 'Cessna172P', 120) diff --git a/simulation/jsbsim_simulator.py b/simulation/jsbsim_simulator.py index 9cc5a11..b7755bc 100644 --- a/simulation/jsbsim_simulator.py +++ b/simulation/jsbsim_simulator.py @@ -5,6 +5,7 @@ from typing import Dict, Union import simulation.jsbsim_properties as prp from simulation.jsbsim_aircraft import Aircraft, x8 +import numpy as np import math import csv from shared import HidePrints @@ -83,19 +84,26 @@ def __init__(self, sim_frequency_hz: float = 60.0, aircraft: Aircraft = x8, init_conditions: Dict[prp.Property, float] = None, + in_flight_reset: int = 0, # 0 means no in-flight reset, positive integer means one of the reset distribution options debug_level: int = 0): - self.fdm = jsbsim.FGFDMExec(root_dir=self.ROOT_DIR) + with HidePrints(): + self.fdm = jsbsim.FGFDMExec(root_dir=self.ROOT_DIR) self.fdm.set_debug_level(debug_level) self.sim_dt = 1.0 / sim_frequency_hz self.aircraft = aircraft self.client = self.airsim_connect() - self.initialize(self.sim_dt, self.aircraft.jsbsim_id, init_conditions) + self.initialize(self.sim_dt, self.aircraft.jsbsim_id, in_flight_reset) self.fdm.disable_output() - self.wall_clock_dt = None + self.wall_clock_dt = None # 0.001 is the best for visualization self.update_airsim(ignore_collisions=True) - self.waypoint_id = 0 - self.waypoint_threshold = 2 - self.waypoint_rewarded = True + self.waypoint_id = in_flight_reset # in_flight_reset should corresponds to the waypoint we start going towards + self.waypoint_threshold = 3 * 0.45/0.12 # multiplied by recent scale factor + # "ViewMode": "NoDisplay", + + self.takeoff_rewarded = in_flight_reset > 0 # rewarded if already in flight + self.completed_takeoff = False + self.waypoint_entered = False + self.waypoint_reward = False self.waypoints = [] with open("waypoints.csv", 'r') as file: csvreader = csv.reader(file) @@ -142,18 +150,22 @@ def get_loaded_model_name(self) -> str: else: return None - def initialize(self, dt: float, model_name: str, init_conditions: Dict['prp.Property', float] = None) -> None: + def initialize(self, dt: float, model_name: str, in_flight_reset: int) -> None: """ Start JSBSim with custom initial conditions :param dt: simulation rate [s] :param model_name: the aircraft model used - :param init_conditions: initial simulation conditions + :param in_flight_reset: whether to simulate from an in-flight reset :return: None """ # Hardcoded currently, meaning init_conditions argument is overriden - ic_file = 'basic_ic.xml' + if in_flight_reset == 0: + ic_file = 'basic_ic.xml' + else: + ic_file = f'reset_ic{in_flight_reset}.xml' + ic_path = os.path.join(os.path.dirname(os.path.abspath(__file__)), ic_file) self.fdm.load_ic(ic_path, useStoredPath=False) self.load_model(model_name) @@ -241,8 +253,9 @@ def airsim_connect() -> airsim.VehicleClient: :return: the airsim client object """ - client = airsim.VehicleClient() - client.confirmConnection() + with HidePrints(): + client = airsim.VehicleClient() + client.confirmConnection() return client def update_airsim(self, ignore_collisions = False) -> None: diff --git a/simulation/mdp.py b/simulation/mdp.py index 58a292f..f9400ad 100644 --- a/simulation/mdp.py +++ b/simulation/mdp.py @@ -4,62 +4,140 @@ states/observations, actions, and rewards. """ +import math import torch import simulation.jsbsim_properties as prp import numpy as np import os +from shared import THROTTLE_CLAMP, AILERON_CLAMP, ELEVATOR_CLAMP, RUDDER_CLAMP """ Extracts agent state data from the sim. """ -def state_from_sim(sim, debug=False): - state = torch.zeros(13,) +def state_from_sim(sim): + state = torch.zeros(15,) FT_TO_M = 0.3048 # altitude state[0] = sim[prp.altitude_sl_ft] * FT_TO_M # z - # velocity - state[1] = sim[prp.v_north_fps] * FT_TO_M # x velocity - state[2] = sim[prp.v_east_fps] * FT_TO_M # y velocity - state[3] = -sim[prp.v_down_fps] * FT_TO_M # z velocity - - # angles - state[4] = sim[prp.roll_rad] # roll - state[5] = sim[prp.pitch_rad] # pitch - state[6] = sim[prp.heading_rad] # yaw + # speed + state[1] = FT_TO_M * np.linalg.norm(np.array([sim[prp.v_north_fps], sim[prp.v_east_fps], -sim[prp.v_down_fps]])) + + # roll + state[2] = sim[prp.roll_rad] + # pitch + state[3] = sim[prp.pitch_rad] + # angle rates - state[7] = sim[prp.p_radps] # roll rate - state[8] = sim[prp.q_radps] # pitch rate - state[9] = sim[prp.r_radps] # yaw rate + state[4] = sim[prp.p_radps] # roll rate + state[5] = sim[prp.q_radps] # pitch rate + state[6] = sim[prp.r_radps] # yaw rate - # next waypoint (relative) + state[7] = -sim[prp.v_down_fps] * FT_TO_M # z velocity + + # calculate next waypoint (relative) position = np.array(sim.get_local_position()) waypoint = np.array(sim.waypoints[sim.waypoint_id]) - displacement = waypoint - position + + # if sim.waypoint_id == 3: + # input() + # print("lat", sim[prp.lat_geod_deg]) + # print("lon", sim[prp.lng_geoc_deg]) + # print("u_fps", sim[prp.u_fps]) + # print("v_fps", sim[prp.v_fps]) + # print("w_fps", sim[prp.w_fps]) + # print("roll_rad", sim[prp.roll_rad]) + # print("pitch_rad", sim[prp.pitch_rad]) + # print("heading_rad", sim[prp.heading_rad]) + # print("altitude", sim[prp.altitude_sl_ft]) + + # determine whether next waypoint needs to be updated if we're currently inside one if np.linalg.norm(displacement) <= sim.waypoint_threshold: - print("Waypoint Hit!") + print(f"\t\t\t\t\t\t\t\t\t\tWaypoint {sim.waypoint_id} Hit!") sim.waypoint_id += 1 - sim.waypoint_rewarded = False + sim.waypoint_entered = True + sim.waypoint_reward = True + if sim.waypoint_id == 7: + raise Exception(f"Finish! :{50}") + # raise Exception(f"Unhealthy state, do better. Penalty :{penalty}") - state[10] = displacement[0] - state[11] = displacement[1] - state[12] = displacement[2] - if debug: - print('State!') - print('Altitude:', state[0]) - print('Velocity: (', state[1], state[2], state[3], ')') - print('Roll:', state[4], '; Pitch:', state[5], '; Yaw:', state[6]) - print('RollRate:', state[7], '; PitchRate:', state[8], '; YawRate:', state[9]) - print('Relative WP: (', state[10], state[11], state[12], ')') + # elif sim.waypoint_entered: + # prev_waypoint = np.array(sim.waypoints[sim.waypoint_id-1]) + # prev_displacement = prev_waypoint - position + # if np.linalg.norm(prev_displacement) > sim.waypoint_threshold: + # # if we've left the previous waypoint, give reward for it + # print(f"\t\t\t\t\t\t\t\t\t\tWaypoint {sim.waypoint_id-1} Rewarded!") + # sim.waypoint_reward = True + # sim.waypoint_entered = False + + waypoint = np.array(sim.waypoints[sim.waypoint_id]) + displacement = waypoint - position + + # waypoint ground distance + state[8] = np.linalg.norm(np.array(displacement[0:2])) + + # waypoint altitude distance + state[9] = displacement[2] + + # heading diff from waypoint + waypoint_heading = np.arctan2(displacement[1], displacement[0]) + heading = sim[prp.heading_rad] # yaw + state[10] = waypoint_heading - heading + if state[10] < -math.pi: + state[10] += 2 * math.pi + elif state[10] > math.pi: + state[10] -= 2* math.pi + + + # print("\t\t\t\t\t\t\t\t\t\t YAW", heading/math.pi*180) + # print("\t\t\t\t\t\t\t\t\t\t waypoint_heading", waypoint_heading/math.pi*180) + + + # print("\t\t\t\t\t\t\t\t\t\theading", state[10]/math.pi*180) + # print("\t\t\t\t\t\t\t\t\t\tdist", state[8]) + # print("\t\t\t\t\t\t\t\t\t\theight", state[9]) + + + # Controls state + state[11] = sim[prp.throttle_cmd] / THROTTLE_CLAMP + state[12] = sim[prp.aileron_cmd] / AILERON_CLAMP + state[13] = sim[prp.elevator_cmd] / ELEVATOR_CLAMP + state[14] = sim[prp.rudder_cmd] / RUDDER_CLAMP + + + # Record whether takeoff completed + if state[0] >= 2: + if not sim.completed_takeoff: + print("\t\t\t\t\t\t\t\t\t\tTake off!") + sim.completed_takeoff = True + + unhealthy_state, penalty = is_unhealthy_state(state) + if unhealthy_state: + raise Exception(f"Unhealthy state, do better. Penalty :{penalty}") + return state +""" +Returns a bool for whether the state is unhealthy +""" +def is_unhealthy_state(state): + MAX_BANK = math.pi / 3 + MAX_PITCH = math.pi / 4 + if np.cos(state[10]) < 0: + return True, -10 + if not -MAX_BANK < state[2] < MAX_BANK: + return True, -50 + if not -MAX_PITCH < state[3] < MAX_PITCH: + return True, -25 + return False, 0 + """ Updates sim according to a control, assumes [control] is a 4-item tensor of @@ -72,6 +150,52 @@ def update_sim_from_control(sim, control, debug=False): sim[prp.rudder_cmd] = control[3] if debug: print('Control Taken:', control) + + +# Get the state/action/log_prob and control from the slewrate autopilot +count = 0 +def query_slewrate_autopilot(sim, autopilot, deterministic=False): + global count + state = state_from_sim(sim) + action, log_prob = autopilot.get_deterministic_action(state) if deterministic else autopilot.get_action(state) + control = autopilot.get_control(action) + + #print('state', state) + #print(state[10:]) + #print('elevator', state[13]) + """ + if state[1] <= 18 and not sim.completed_takeoff: + print('throttle', state[11]) + control = torch.Tensor([1, 0, 0, 0]) + elif state[1] <= 20 and not sim.completed_takeoff: + print('throttle down', state[11]) + control = torch.Tensor([-1, 0, 0, 0]) + elif count < 20: + #print('throttle', state[11]) + print('holding pitch', state[13]) + count += 1 + control = torch.Tensor([0, 0, 0, 0]) + elif count < 40: + print('pitch up') + count += 1 + control = torch.Tensor([0, 0, -1, 0]) + elif count < 400: + count += 1 + control = torch.Tensor([0, 0, 0, 0]) + else: + control = torch.Tensor([0, 0, 1, 0]) + """ + + + return state, action, log_prob, control + +# Called every single sim step to enact slewrate +# Assumes autopilot is a SlewRateAutopilotLearner +def update_sim_from_slewrate_control(sim, control, autopilot): + sim[prp.throttle_cmd] = np.clip(sim[prp.throttle_cmd] + control[0] * autopilot.throttle_slew_rate, 0, THROTTLE_CLAMP) + sim[prp.aileron_cmd] = np.clip(sim[prp.aileron_cmd] + control[1] * autopilot.aileron_slew_rate, -AILERON_CLAMP, AILERON_CLAMP) + sim[prp.elevator_cmd] = np.clip(sim[prp.elevator_cmd] + control[2] * autopilot.elevator_slew_rate, -ELEVATOR_CLAMP, ELEVATOR_CLAMP) + sim[prp.rudder_cmd] = np.clip(sim[prp.rudder_cmd] + control[3] * autopilot.rudder_slew_rate, -RUDDER_CLAMP, RUDDER_CLAMP) """ Follows a predetermined sequence of controls, instead of using autopilot. @@ -115,18 +239,27 @@ def enact_predetermined_controls(sim, autopilot): Basically just updates sim throttle / control surfaces according to the autopilot. """ def enact_autopilot(sim, autopilot): - state = state_from_sim(sim, debug=False) + state = state_from_sim(sim) action, log_prob = autopilot.get_action(state) - update_sim_from_control(sim, autopilot.get_control(action)) return state, action, log_prob + # Takes in the action outputted directly from the network and outputs the # normalized quadratic action cost from 0-1 def quadratic_action_cost(action): - action[1:4] = 2*action[1:4] - 1 # convert control surfaces to [-1, 1] - return float(torch.dot(action, action).detach()) / 4 # divide by 4 to be 0-1 + action_cost_weights = torch.tensor([3.0, 10.0, 5.0, 1.0]) + action[0] = 0.5 * (action[0] + 1) # converts throttle to be 0-1 + return float(torch.dot(action ** 2, action_cost_weights).detach() / sum(action_cost_weights)) # divide by 4 to be 0-1 + +# Takes in the action outputted directly from the network and outputs the +# normalized quadratic action cost from 0-1 +def quadratic_control_cost(control): + # control_cost_weights = torch.tensor([1.0, 10.0, 5.0, 1.0]) + # 0.05882352941 0.5882352941 0.2941176471 0.05882352941 + control_cost_weights = torch.tensor([0.6, 10.0, 0.3, 0.06]) + return float(torch.dot(control ** 2, control_cost_weights).detach()) # divide by 4 to be 0-1 """ The reward function for the bb autopilots. Since they won't know how to fly, @@ -134,6 +267,7 @@ def quadratic_action_cost(action): + 1 if moving with some velocity threshold vel_reward_threshold + alt_reward_threshold if flying above ground with some threshold alt_reward_threshold - action_coeff * the quadratic control cost + !!! WARNING: Deprecated """ def bb_reward(action, next_state, collided, alt_reward_coeff=10, action_coeff=1, alt_reward_threshold=2, vel_reward_threshold=1): moving_reward = 1 if (next_state[3]**2 + next_state[4]**2 + next_state[5]**2) > vel_reward_threshold else 0 @@ -145,6 +279,7 @@ def bb_reward(action, next_state, collided, alt_reward_coeff=10, action_coeff=1, A reward function that tries to incentivize the plane to go towards the waypoint at each timestep, while also rewarding for being above ground and penalizing for high control effort + !!! WARNING: Deprecated """ def new_init_wp_reward(action, next_state, collided, wp_coeff=1, action_coeff=1, alt_reward_threshold=5): alt_reward = 1 if next_state[2] > alt_reward_threshold else 0 @@ -152,29 +287,49 @@ def new_init_wp_reward(action, next_state, collided, wp_coeff=1, action_coeff=1, waypoint_rel_unit = next_state[10:13] / torch.norm(next_state[10:13]) vel = next_state[1:4] - toward_waypoint_reward = wp_coeff * float(torch.dot((vel ** 2 * torch.sign(vel)), waypoint_rel_unit).detach()) + toward_waypoint_reward = wp_coeff * float(torch.dot(vel, waypoint_rel_unit).detach()) return toward_waypoint_reward + alt_reward - action_cost if not collided else 0 """ -A reward function. +A reward function getter. """ +a = 0 def get_wp_reward(sim): - def wp_reward(action, next_state, collided, wp_coeff=0.01, action_coeff=0.01, alt_reward_threshold=10): - if not sim.waypoint_rewarded: - sim.waypoint_rewarded = True - wp_reward = 1_000_000 + def wp_reward(action, next_state, collided, wp_coeff=0.5, action_coeff=0.5, alt_coeff=0.5, roll_coeff=0.9): + global a + if sim.waypoint_reward: + sim.waypoint_reward = False + wp_reward = 50 else: wp_reward = 0 - # print("altitude", next_state[2] ) - alt_reward = 0.1 if next_state[2] > alt_reward_threshold else 0 - action_cost = action_coeff * quadratic_action_cost(action) - - waypoint_rel_unit = next_state[10:13] / torch.norm(next_state[10:13]) - vel = next_state[1:4] - toward_waypoint_reward = wp_coeff * float(torch.dot((vel ** 2 * torch.sign(vel)), waypoint_rel_unit).detach()) - # print("wp_reward: ", wp_reward, " toward_waypoint_reward: ", toward_waypoint_reward) - # print("alt_reward: ", alt_reward, " action_cost: ", -action_cost) - # print("\t reward", wp_reward + toward_waypoint_reward + alt_reward - action_cost if not collided else 0) - return wp_reward + toward_waypoint_reward + alt_reward - action_cost if not collided else 0 + action_cost = action_coeff * quadratic_control_cost(next_state[11:15]) + #print('\t\t\t\t\t\tACTION COST:', action_cost, next_state[11:15]) + # waypoint_rel_unit = torch.nn.functional.normalize(next_state[10:13], dim=0) + # vel = next_state[1:4] + # toward_waypoint_reward = wp_coeff * float(torch.dot(vel, waypoint_rel_unit).detach()) + # cosine(relative heading) * ground speed + #toward_waypoint_reward = min(wp_coeff * torch.cos(next_state[10]) * np.sqrt(next_state[1] ** 2 - next_state[7] ** 2), 4) + away_waypoint_cost = wp_coeff * abs(next_state[10]) + alitude_diff_cost = alt_coeff * abs(np.arctan2(next_state[9], next_state[8]) - next_state[3]) # pitch "error": relative pitch between current pitch and straight line to waypoint + roll_cost = roll_coeff * (abs(next_state[2]))**3 + #print('\t\t\t\t\t\AWAY WP COST:', away_waypoint_cost) + #print('\t\t\t\t\t\ALT COST:', alitude_diff_cost) + if not sim.takeoff_rewarded and sim.completed_takeoff: + takeoff_reward = 25 + sim.takeoff_rewarded = True + else: + takeoff_reward = 0 + + staying_alive_reward = 0.5 + # toward_waypoint_reward = 0 + rewards = staying_alive_reward + wp_reward + takeoff_reward + costs = action_cost + away_waypoint_cost + alitude_diff_cost + roll_cost + # if a % 100 == 0: + # print("\t\t\t\t\t\t\t\taway_waypoint_cost", away_waypoint_cost, "alitude_diff_cost",alitude_diff_cost) + # print("\t\t\t\t\t\t\t\troll_cost", roll_cost, "action_cost",action_cost, "action",next_state[11:15]) + # print("\t\t\t\t\t\t\t\tRewards:",rewards, "Costs:", -costs) + # print("\t\t\t\t\t\t\t\tNet Reward", rewards - costs if not collided else 0) + # toward_waypoint_reward = torch.min(torch.tensor(10), 0.01 * 1 / torch.max(torch.tensor(0.1), torch.sum(next_state[10:13]**2))) + return rewards - costs if not collided else 0 return wp_reward """ @@ -182,7 +337,7 @@ def wp_reward(action, next_state, collided, wp_coeff=0.01, action_coeff=0.01, al rollout, including states/actions/rewards the agent experienced. """ class MDPDataCollector: - def __init__(self, sim, reward_fn, expected_trajectory_length, state_dim=13, action_dim=4): + def __init__(self, sim, reward_fn, expected_trajectory_length, state_dim=15, action_dim=4): # parameters self.sim = sim self.reward_fn = reward_fn @@ -197,7 +352,7 @@ def __init__(self, sim, reward_fn, expected_trajectory_length, state_dim=13, act self.cum_reward = 0 # t = timestep of the current state-action pair, in [0, expected_trajectory_length-1] - def update(self, t, state, action, log_prob, next_state, collided): + def update(self, t, state, action, log_prob, next_state, collided, unhealthy_penalty=0.0): self.states[t, :] = torch.transpose(state, 0, -1) self.next_states[t, :] = torch.transpose(next_state, 0, -1) self.actions[t, :] = torch.transpose(action, 0, -1) @@ -206,7 +361,7 @@ def update(self, t, state, action, log_prob, next_state, collided): # this function should only be used when sim is still running, so not done # and not collided self.dones[t] = 0 - reward = self.reward_fn(action, next_state, collided) + reward = self.reward_fn(action, next_state, collided) + unhealthy_penalty self.rewards[t] = reward self.cum_reward += reward diff --git a/simulation/reset_ic3.xml b/simulation/reset_ic3.xml new file mode 100644 index 0000000..5f3394f --- /dev/null +++ b/simulation/reset_ic3.xml @@ -0,0 +1,17 @@ + + + + 96.0 + 0.0 + 3.16 + -0.00000180221070782819 + 0.0022388192006667883 + -0.00033960315510837 + 5.89360846566800000 + 0.030266855814653545 + 25.397382017225027 + 2.1 + 90.0 + 0.0 + 0 + \ No newline at end of file diff --git a/simulation/reset_ic4.xml b/simulation/reset_ic4.xml new file mode 100644 index 0000000..ce78a24 --- /dev/null +++ b/simulation/reset_ic4.xml @@ -0,0 +1,17 @@ + + + + 85.7973202721795 + 0.0 + 2.8924357765532296 + 0.00000047068495677246515 + 0.0035561556674120506 + -0.00111435048832493 + 5.72140628231839000 + -0.030266855814653545 + 101.0717219337821 + 2.1 + 90.0 + 0.0 + 0 + \ No newline at end of file diff --git a/simulation/reset_ic5.xml b/simulation/reset_ic5.xml new file mode 100644 index 0000000..cdba4a7 --- /dev/null +++ b/simulation/reset_ic5.xml @@ -0,0 +1,17 @@ + + + + 90.0 + 0.0 + 3.0 + 0.0 + 0.0049 + 0.0 + 8.721406282318400 + 0.0 + 170.0 + 2.1 + 90.0 + 0.0 + 0 + \ No newline at end of file diff --git a/simulation/reset_ic6.xml b/simulation/reset_ic6.xml new file mode 100644 index 0000000..3097c05 --- /dev/null +++ b/simulation/reset_ic6.xml @@ -0,0 +1,17 @@ + + + + 90.0 + 0.0 + 3.0 + 0.0001 + 0.0061 + -10.0011143504883249324 + 7.121406282318400 + 0.0 + 225.0 + 2.1 + 90.0 + 0.0 + 0 + \ No newline at end of file diff --git a/simulation/simulate.py b/simulation/simulate.py index b0ab2a3..26655fa 100644 --- a/simulation/simulate.py +++ b/simulation/simulate.py @@ -1,10 +1,13 @@ +from shared import HidePrints from simulation.jsbsim_simulator import Simulation from simulation.jsbsim_aircraft import Aircraft, x8 import simulation.jsbsim_properties as prp -from learning.autopilot import AutopilotLearner +from learning.autopilot import AutopilotLearner, SlewRateAutopilotLearner +import torch import simulation.mdp as mdp import os import numpy as np +from shared import THROTTLE_CLAMP, AILERON_CLAMP, ELEVATOR_CLAMP, RUDDER_CLAMP """ A class to integrate JSBSim and AirSim to roll-out a full trajectory for an @@ -13,20 +16,23 @@ class FullIntegratedSim: def __init__(self, aircraft: Aircraft, - autopilot: AutopilotLearner, + autopilot: SlewRateAutopilotLearner, sim_time: float, display_graphics: bool = True, - agent_interaction_frequency: int = 1, + agent_interaction_frequency: int = 15, airsim_frequency_hz: float = 392.0, sim_frequency_hz: float = 240.0, - init_conditions: bool = None, + in_flight_reset: int = 0, # nonzero if we initialize from a non-takeoff reset distribution + auto_deterministic: bool = True, # whether the autopilot picks its mode action (deterministic) or samples debug_level: int = 0): # Aircraft and autopilot self.aircraft = aircraft self.autopilot = autopilot + self.auto_deterministic = auto_deterministic # Sim params - self.sim: Simulation = Simulation(sim_frequency_hz, aircraft, init_conditions, debug_level) + self.in_flight_reset = in_flight_reset + self.sim: Simulation = Simulation(sim_frequency_hz, aircraft, in_flight_reset=in_flight_reset, debug_level=debug_level) self.sim_time = sim_time self.display_graphics = display_graphics self.sim_frequency_hz = sim_frequency_hz @@ -44,7 +50,12 @@ def __init__(self, # Triggered when sim is complete self.done: bool = False + # + self.unhealthy_termination: bool = False + self.unhealthy_penalty: float = 0.0 + self.initial_collision = False + """ Run loop for one simulation. """ @@ -59,31 +70,60 @@ def simulation_loop(self): pose = self.sim.client.simGetVehiclePose() # Experimentally determined, in UE4 coordinate system - ic_position = np.array([0, 0, -1.3411200046539307]) + if self.in_flight_reset == 0: + ic_position = [0, 0, -1.3411200046539307] + elif self.in_flight_reset == 3: + ic_position = [ 2.50183762e+02, -1.58594549e-01, -8.38120174e+00] + elif self.in_flight_reset == 4: + ic_position = [ 3.97393585e+02, 4.14202549e-02, -3.14456406e+01] + elif self.in_flight_reset == 5: + ic_position = [ 5.47565369e+02, -7.68899895e-08, -5.24536057e+01] + elif self.in_flight_reset == 6: + ic_position = [ 681.66052246, 8.80005455, -69.21742249] + + ic_position = np.array(ic_position) current_position = np.array([pose.position.x_val, pose.position.y_val, pose.position.z_val]) retry_period = 10 retry_counter = 0 - while (np.abs(ic_position - current_position) > np.finfo(float).eps).all(): + while (np.abs(ic_position - current_position)/ np.linalg.norm(current_position) > 0.00001).all(): if retry_counter % retry_period == 0: self.sim.reinitialize() retry_counter += 1 + # print("cur pos", current_position) pose = self.sim.client.simGetVehiclePose() current_position = np.array([pose.position.x_val, pose.position.y_val, pose.position.z_val]) - + # Initialize to max throttle; the agent then learns when/how to decrease for cruise throttle + self.sim[prp.throttle_cmd] = THROTTLE_CLAMP + + # If we're initializing in flight, randomly set controls + if self.in_flight_reset > 0: + self.sim[prp.throttle_cmd] = np.clip(np.random.normal(0.35,THROTTLE_CLAMP/10), 0, THROTTLE_CLAMP) + self.sim[prp.aileron_cmd] = np.clip(np.random.normal(0, AILERON_CLAMP/200), -AILERON_CLAMP, AILERON_CLAMP) + self.sim[prp.elevator_cmd] = np.clip(np.random.normal(0, ELEVATOR_CLAMP/20), -ELEVATOR_CLAMP, ELEVATOR_CLAMP) + self.sim[prp.rudder_cmd] = np.clip(np.random.normal(0, RUDDER_CLAMP/20), -RUDDER_CLAMP, RUDDER_CLAMP) + i = 0 while i < update_num: # Do autopilot controls try: - state, action, log_prob = mdp.enact_autopilot(self.sim, self.autopilot) + #state, action, log_prob = mdp.enact_autopilot(self.sim, self.autopilot) + state, action, log_prob, control = mdp.query_slewrate_autopilot(self.sim, self.autopilot, deterministic=self.auto_deterministic) + if torch.isnan(state).any(): + break except Exception as e: print(e) + self.unhealthy_penalty = float(str(e).split(":")[-1]) + # print("penalty", self.unhealthy_penalty) + self.unhealthy_termination = True # If enacting the autopilot fails, end the simulation immediately break # Update sim while waiting for next agent interaction while True: + mdp.update_sim_from_slewrate_control(self.sim, control, self.autopilot) + # Run another sim step self.sim.run() @@ -100,28 +140,35 @@ def simulation_loop(self): # Check for collisions via airsim and terminate if there is one if self.sim.get_collision_info().has_collided: if self.initial_collision: - print('Aircraft has collided.') + # print('Aircraft has collided.') self.done = True self.sim.reinitialize() else: - print("Aircraft completed initial landing") + # print("Aircraft completed initial landing") self.initial_collision = True # Exit if sim is over or it's time for another agent interaction if self.done or i % self.agent_interaction_frequency == 0: break - + # Get new state try: next_state = mdp.state_from_sim(self.sim) - except: + if torch.isnan(next_state).any(): + next_state = state + self.done = True + except Exception as e: # If we couldn't acquire the state, something crashed with jsbsim # We treat that as the end of the simulation and don't update the state + print(f"\t\t\t\t\t\t\t\t\t\t{e}") next_state = state self.done = True + self.unhealthy_termination = True + self.unhealthy_penalty = float(str(e).split(":")[-1]) # Data collection update for this step - self.mdp_data_collector.update(int(i/self.agent_interaction_frequency)-1, state, action, log_prob, next_state, self.done) + self.mdp_data_collector.update(int(i/self.agent_interaction_frequency)-1, state, action, + log_prob, next_state, self.unhealthy_termination, self.unhealthy_penalty) # End if collided if self.done == True: @@ -130,18 +177,29 @@ def simulation_loop(self): self.done = True self.mdp_data_collector.terminate(int(i/self.agent_interaction_frequency)) print('Simulation complete.') + print('\t\t\t\t\t\t\t\t\t\tCum reward:', self.mdp_data_collector.cum_reward) + """ Replays a simulation """ def simulation_replay(self, actions): + # THESE ARE HARCODED by reading from prints of the initial controls state. TODO: make this a possible input + self.sim[prp.throttle_cmd] = 0.5372 * THROTTLE_CLAMP + self.sim[prp.aileron_cmd] = -0.0024 * AILERON_CLAMP + self.sim[prp.elevator_cmd] = -0.0352 * ELEVATOR_CLAMP + self.sim[prp.rudder_cmd] = -0.0080 * RUDDER_CLAMP + i = 0 for action in actions: + state = mdp.state_from_sim(self.sim) # Do the control - mdp.update_sim_from_control(self.autopilot.get_control(action)) - + # mdp.update_sim_from_control(self.autopilot.get_control(action)) + control = self.autopilot.get_control(action) while True: # Run another sim step + mdp.update_sim_from_slewrate_control(self.sim, control, self.autopilot) + #print('control', control) self.sim.run() i += 1 @@ -151,7 +209,15 @@ def simulation_replay(self, actions): # NOTE: for replays, the agent interaction frequency must match what # it was when the trajectory was created if i % self.agent_interaction_frequency == 0: + print("Step") break + try: + next_state = mdp.state_from_sim(self.sim) + self.mdp_data_collector.update(int(i/self.agent_interaction_frequency)-1, state, action, + 0, next_state, self.unhealthy_termination, self.unhealthy_penalty) + except: + break + print('cum reward', self.mdp_data_collector.cum_reward) if __name__ == "__main__": # A one-minute simulation with an untrained autopilot diff --git a/waypoints.csv b/waypoints.csv index 7b87c80..bc9e55e 100644 --- a/waypoints.csv +++ b/waypoints.csv @@ -1,7 +1,8 @@ waypoint_id,x,y,z,relative_x,relative_y,relative_z -1,-4461,-417.960938,90,21.5,-0.2,0.89 -2,-701,-417.960938,120,59.1,-0.2,1.19 -3,2869,-57.960938,450,94.8,3.4,4.49 -4,6509,-967.960938,1230,131.2,-5.7,12.29 -5,9469,-477.960938,1480,160.8,-0.8,14.79 -6,15619,-397.960938,2920,222.3,0,29.19 \ No newline at end of file +1,-11341,-417.960938,70,38.5,-0.2,0.69 +2,-741,-397.960938,110,144.5,0,1.09 +3,9466,-621,1286,246.57,-2.23039062,12.85 +4,24279,-437.960938,2570,394.7,-0.4,25.69 +5,39319,-257.960938,5370,545.1,1.4,53.69 +6,52959,292.039062,7340,681.5,6.9,73.39 +7,72839,-377.960938,11710,880.3,0.2,117.09 \ No newline at end of file