-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathreferences.lib
More file actions
76 lines (69 loc) · 2.74 KB
/
Copy pathreferences.lib
File metadata and controls
76 lines (69 loc) · 2.74 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
@misc{schulman2017trustregionpolicyoptimization,
title={Trust Region Policy Optimization},
author={John Schulman and Sergey Levine and Philipp Moritz and Michael I. Jordan and Pieter Abbeel},
year={2017},
eprint={1502.05477},
archivePrefix={arXiv},
primaryClass={cs.LG},
url={https://arxiv.org/abs/1502.05477},
}
@misc{schulman2017proximalpolicyoptimizationalgorithms,
title={Proximal Policy Optimization Algorithms},
author={John Schulman and Filip Wolski and Prafulla Dhariwal and Alec Radford and Oleg Klimov},
year={2017},
eprint={1707.06347},
archivePrefix={arXiv},
primaryClass={cs.LG},
url={https://arxiv.org/abs/1707.06347},
}
@misc{pathak2017curiositydrivenexplorationselfsupervisedprediction,
title={Curiosity-driven Exploration by Self-supervised Prediction},
author={Deepak Pathak and Pulkit Agrawal and Alexei A. Efros and Trevor Darrell},
year={2017},
eprint={1705.05363},
archivePrefix={arXiv},
primaryClass={cs.LG},
url={https://arxiv.org/abs/1705.05363},
}
@misc{wang2024prioritizedgenerativereplay,
title={Prioritized Generative Replay},
author={Renhao Wang and Kevin Frans and Pieter Abbeel and Sergey Levine and Alexei A. Efros},
year={2024},
eprint={2410.18082},
archivePrefix={arXiv},
primaryClass={cs.LG},
url={https://arxiv.org/abs/2410.18082},
}
@inproceedings{Schmidhuber1991APF,
title={A possibility for implementing curiosity and boredom in model-building neural controllers},
author={J{\"u}rgen Schmidhuber},
year={1991},
url={https://api.semanticscholar.org/CorpusID:18060048}
}
@misc{roostaie2021entrpotrustregionpolicy,
title={EnTRPO: Trust Region Policy Optimization Method with Entropy Regularization},
author={Sahar Roostaie and Mohammad Mehdi Ebadzadeh},
year={2021},
eprint={2110.13373},
archivePrefix={arXiv},
primaryClass={cs.LG},
url={https://arxiv.org/abs/2110.13373},
}
@misc{plappert2018parameterspacenoiseexploration,
title={Parameter Space Noise for Exploration},
author={Matthias Plappert and Rein Houthooft and Prafulla Dhariwal and Szymon Sidor and Richard Y. Chen and Xi Chen and Tamim Asfour and Pieter Abbeel and Marcin Andrychowicz},
year={2018},
eprint={1706.01905},
archivePrefix={arXiv},
primaryClass={cs.LG},
url={https://arxiv.org/abs/1706.01905},
}
@misc{wang2020reinforcementlearningperturbedrewards,
title={Reinforcement Learning with Perturbed Rewards},
author={Jingkang Wang and Yang Liu and Bo Li},
year={2020},
eprint={1810.01032},
archivePrefix={arXiv},
primaryClass={cs.LG},
url={https://arxiv.org/abs/1810.01032},
}