-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathreferences.bib
More file actions
306 lines (306 loc) · 31.1 KB
/
Copy pathreferences.bib
File metadata and controls
306 lines (306 loc) · 31.1 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
@article{Mania2018,
abstract = {A common belief in model-free reinforcement learning is that methods based on random search in the parameter space of policies exhibit significantly worse sample complexity than those that explore the space of actions. We dispel such beliefs by introducing a random search method for training static, linear policies for continuous control problems, matching state-of-the-art sample efficiency on the benchmark MuJoCo locomotion tasks. Our method also finds a nearly optimal controller for a challenging instance of the Linear Quadratic Regulator, a classical problem in control theory, when the dynamics are not known. Computationally, our random search algorithm is at least 15 times more efficient than the fastest competing model-free methods on these benchmarks. We take advantage of this computational efficiency to evaluate the performance of our method over hundreds of random seeds and many different hyperparameter configurations for each benchmark task. Our simulations highlight a high variability in performance in these benchmark tasks, suggesting that commonly used estimations of sample efficiency do not adequately evaluate the performance of RL algorithms.},
archivePrefix = {arXiv},
arxivId = {1803.07055},
author = {Mania, Horia and Guy, Aurelia and Recht, Benjamin},
eprint = {1803.07055},
file = {:home/haakonrr/OneDrive/articles/ml/simple-random-search.pdf:pdf},
pages = {1--22},
title = {{Simple random search provides a competitive approach to reinforcement learning}},
url = {http://arxiv.org/abs/1803.07055},
year = {2018}
}
@article{Antonelli2015,
author = {Antonelli, Gianluca and Moe, Signe and Pettersen, Kristin Y},
file = {:home/haakonrr/OneDrive/articles/autonomous/set-based-control-experimental-results.pdf:pdf},
isbn = {9781479999354},
keywords = {Industrial Robotics,Intelligent Control},
pages = {1132--1137},
title = {{Incorporating set-based control within the singularity-robust multiple task-priority inverse kinematics}},
year = {2015}
}
@article{Moe2018,
abstract = {IEEE In this paper, a method is presented for lowering the energy consumption and/or increasing the speed of a standard manipulator spray painting a surface. The approach is based on the observation that a small angle between the spray direction and the surface normal does not affect the quality of the paint job. Recent results in set-based kinematic control are utilized to develop a switched control system, where this angle is defined as a set-based task with a maximum allowed limit. Four different set-based methods are implemented and tested on a UR5 manipulator from Universal Robots. Experimental results verify the correctness of the method, and demonstrate that the set-based approaches can substantially lower the paint time and energy consumption compared to the current standard solution.},
archivePrefix = {arXiv},
arxivId = {1612.09105},
author = {Moe, Signe and Gravdahl, Jan Tommy and Pettersen, Kristin Y.},
doi = {10.1109/TASE.2018.2801382},
eprint = {1612.09105},
file = {:home/haakonrr/OneDrive/articles/autonomous/set-based-control -for-auto-spray-painting.pdf:pdf},
issn = {15455955},
journal = {IEEE Transactions on Automation Science and Engineering},
keywords = {Autonomous robots,Jacobian matrices,Kinematics,Paints,Robots,Surface treatment,Task analysis,Trajectory,industrial control,paints,robot control,robot kinematics.},
pages = {1--12},
title = {{Set-Based Control for Autonomous Spray Painting}},
year = {2018}
}
@article{Kuwata2014,
abstract = {This paper presents an autonomous motion planning algorithm for unmanned surface vehicles (USVs) to navigate safely in dynamic, cluttered environments. The algorithm not only addresses hazard avoidance (HA) for stationary and moving hazards, but also applies the International Regulations for Preventing Collisions at Sea (known as COLREGS, for COLlision REGulationS). The COLREGS rules specify, for example, which vessel is responsible for giving way to the other and to which side of the 'stand-on' vessel to maneuver. Three primary COLREGS rules are considered in this paper: crossing, overtaking, and head-on situations. For autonomous USVs to be safely deployed in environments with other traffic boats, it is imperative that the USV's navigation algorithm obeys COLREGS. Furthermore, when other boats disregard their responsibility under COLREGS, the USV must fall back to its HA algorithms to prevent a collision. The proposed approach is based on velocity obstacles (VO) method, which generates a cone-shaped obstacle in the velocity space. Because VOs also specify on which side of the obstacle the vehicle will pass during the avoidance maneuver, COLREGS are encoded in the velocity space in a natural way. Results from several experiments involving up to four vessels are presented, in what we believe is the first on-water demonstration of autonomous COLREGS maneuvers without explicit intervehicle communication. We also show an application of this motion planner to a target trailing task, where a strategic planner commands USV waypoints based on high-level objectives, and the local motion planner ensures hazard avoidance and compliance with COLREGS during a traverse. {\textcopyright} 2013 IEEE.},
author = {Kuwata, Yoshiaki and Wolf, Michael T. and Zarzhitsky, Dimitri and Huntsberger, Terrance L.},
doi = {10.1109/JOE.2013.2254214},
file = {:home/haakonrr/OneDrive/articles/autonomous/safe-maritime-nav-COLREGS.pdf:pdf},
isbn = {9781612844541},
issn = {03649059},
journal = {IEEE Journal of Oceanic Engineering},
keywords = {COLREGS,unmanned surface vehicle (USV),velocity obstacles (VOs)},
number = {1},
pages = {110--119},
title = {{Safe maritime autonomous navigation with COLREGS, using velocity obstacles}},
volume = {39},
year = {2014}
}
@article{Antonelli2015a,
author = {Antonelli, Gianluca and Moe, Signe and Pettersen, Kristin Y},
file = {:home/haakonrr/OneDrive/articles/autonomous/incorporating-set-based-control.pdf:pdf},
isbn = {9781479999354},
pages = {1132--1137},
title = {{Incorporating set-based control within the singularity-robust multiple task-priority inverse kinematics}},
year = {2015}
}
@article{Moe2017,
abstract = {— A cornerstone ability of an autonomous unmanned surface vessel (USV) is to avoid collisions with stationary obstacles and other moving vehicles while following a predefined path. USVs are typically underactuated, and this paper extends recent results in set-based guidance theory to an underactuated surface vessel, resulting in a switched guidance system with a path following mode and a collision avoidance mode. This system can be used with any combination of path following and collision avoidance guidance laws. Furthermore, a specific guidance law for collision avoidance is suggested that ensures tracking of a safe radius about a moving obstacle. The guidance law is specifically designed to assure collision avoidance while abiding by the International Regulations for Preventing Colli-sions at Sea (COLREGs). It is proven that the USV successfully circumvents the obstacles in a COLREGs compliant manner and that path following is achieved in path following mode. Simulations results confirm the effectiveness of the proposed approach.},
author = {Moe, Signe and Pettersen, Kristin Y.},
doi = {10.1109/CCTA.2017.8062470},
file = {:home/haakonrr/OneDrive/articles/autonomous/2016{\_}2{\_}MED{\_}2.pdf:pdf},
isbn = {9781509021826},
journal = {1st Annual IEEE Conference on Control Technology and Applications, CCTA 2017},
keywords = {Intelligent control systems,Marine control,Nonlinear control},
pages = {241--248},
title = {{Set-Based line-of-sight (LOS) path following with collision avoidance for underactuated unmanned surface vessels under the influence of ocean currents}},
volume = {2017-January},
year = {2017}
}
@article{Moe2016,
abstract = {— An essential ability of autonomous unmanned surface vessels (USVs) and autonomous underwater vehicles (AUVs) moving in a horizontal plane is to follow a general two-dimensional path in the presence of unknown ocean currents. This paper presents a method to achieve this. The proposed guidance and control system only requires absolute velocity measurements for feedback, thereby foregoing the need for expensive sensors to measure relative velocities. The closed-loop system consists of a guidance law and an adaptive feedback linearizing controller combined with sliding mode, and is shown to render the path cross-track error dynamics UGAS and USGES. Simulation results are presented to verify the theoretical results.},
author = {Moe, Signe and Pettersen, Kristin Y. and Fossen, Thor I. and Gravdahl, Jan T.},
doi = {10.1109/MED.2016.7536018},
file = {:home/haakonrr/OneDrive/articles/autonomous/2016{\_}2{\_}MED{\_}1.pdf:pdf},
isbn = {9781467383455},
journal = {24th Mediterranean Conference on Control and Automation, MED 2016},
keywords = {Marine control,Nonlinear control,Unmanned systems},
number = {2},
pages = {38--45},
title = {{Line-of-sight curved path following for underactuated USVs and AUVs in the horizontal plane under the influence of ocean currents}},
year = {2016}
}
@article{Moe2015,
author = {Moe, S and Teel, A and Antonelli, G and Pettersen, K Y},
file = {:home/haakonrr/OneDrive/articles/autonomous/stability-analysis-set-based-control.pdf:pdf},
isbn = {9781479978854},
number = {Cdc},
title = {{Stability Analysis for Set-based Control within the Singularity-robust Multiple Task-priority Inverse Kinematics Framework$\backslash$footnote{\{}$\backslash$url{\{}https://www.dropbox.com/s/mzmz92ematkdo7f/CDC15{\_}0120{\_}MS.pdf?dl=0{\}}{\}}}},
year = {2015}
}
@article{Lehman2011,
abstract = {By synthesising a growing body of work in search$\backslash$nprocesses that are not driven by explicit objectives,$\backslash$nthis paper advances the hypothesis that there is a$\backslash$nfundamental problem with the dominant paradigm of$\backslash$nobjective-based search in evolutionary computation and$\backslash$ngenetic programming: Most ambitious objectives do not$\backslash$nilluminate a path to themselves. That is, the gradient$\backslash$nof improvement induced by ambitious objectives tends to$\backslash$nlead not to the objective itself but instead to dead$\backslash$nend local optima. Indirectly supporting this$\backslash$nhypothesis, great discoveries often are not the result$\backslash$nof objective-driven search. For example, the major$\backslash$ninspiration for both evolutionary computation and$\backslash$ngenetic programming, natural evolution, innovates$\backslash$nthrough an open-ended process that lacks a final$\backslash$nobjective. Similarly, large-scale cultural evolutionary$\backslash$nprocesses, such as the evolution of technology,$\backslash$nmathematics, and art, lack a unified fixed goal. In$\backslash$naddition, direct evidence for this hypothesis is$\backslash$npresented from a recently-introduced search algorithm$\backslash$ncalled novelty search. Though ignorant of the ultimate$\backslash$nobjective of search, in many instances novelty search$\backslash$nhas counter-intuitively outperformed searching directly$\backslash$nfor the objective, including a wide variety of$\backslash$nrandomly-generated problems introduced in an experiment$\backslash$nin this chapter. Thus a new understanding is beginning$\backslash$nto emerge that suggests that searching for a fixed$\backslash$nobjective, which is the reigning paradigm in$\backslash$nevolutionary computation and even machine learning as a$\backslash$nwhole, may ultimately limit what can be achieved. Yet$\backslash$nthe liberating implication of this hypothesis argued in$\backslash$nthis paper is that by embracing search processes that$\backslash$nare not driven by explicit objectives, the breadth and$\backslash$ndepth of what is reachable through evolutionary methods$\backslash$nsuch as genetic programming may be greatly expanded.},
author = {Lehman, Joel and Stanley, Kenneth O.},
doi = {10.1007/978-1-4614-1770-5_3},
file = {:home/haakonrr/OneDrive/articles/rl/novelty-search.pdf:pdf},
isbn = {9781461417705},
pages = {37--56},
title = {{Novelty Search and the Problem with Objectives}},
url = {http://link.springer.com/10.1007/978-1-4614-1770-5{\_}3},
year = {2011}
}
@article{Graves2013,
abstract = {This paper shows how Long Short-term Memory recurrent neural networks can be used to generate complex sequences with long-range structure, simply by predicting one data point at a time. The approach is demonstrated for text (where the data are discrete) and online handwriting (where the data are real-valued). It is then extended to handwriting synthesis by allowing the network to condition its predictions on a text sequence. The resulting system is able to generate highly realistic cursive handwriting in a wide variety of styles.},
archivePrefix = {arXiv},
arxivId = {1308.0850},
author = {Graves, Alex},
doi = {10.1145/2661829.2661935},
eprint = {1308.0850},
file = {:home/haakonrr/OneDrive/articles/ml/generating-sequences-with-RNNs.pdf:pdf},
isbn = {2000201075},
issn = {18792782},
pages = {1--43},
pmid = {23459267},
title = {{Generating Sequences With Recurrent Neural Networks}},
url = {http://arxiv.org/abs/1308.0850},
year = {2013}
}
@article{Graves2016,
abstract = {Modern computers separate computation and memory. Computation is performed by a processor, which can use an addressable memory to bring operands in and out of play. This confers two important benefits: the use of extensible storage to write new information and the ability to treat the contents of memory as variables. Variables are critical to algorithm generality: to perform the same procedure on one datum or another, an algorithm merely has to change the address it reads from. In contrast to computers, the computational and memory resources of artificial neural networks are mixed together in the network weights and neuron activity. This is a major liability: as the memory demands of a task increase, these networks cannot allocate new storage dynam-ically, nor easily learn algorithms that act independently of the values realized by the task variables. Although recent breakthroughs demonstrate that neural networks are remarkably adept at sensory processing 1 , sequence learning 2,3 and reinforcement learning 4 , cognitive scientists and neuroscientists have argued that neural networks are limited in their ability to represent variables and data structures 5–9 , and to store data over long timescales without interference 10,11 . We aim to combine the advantages of neu-ral and computational processing by providing a neural network with read–write access to external memory. The access is narrowly focused, minimizing interference among memoranda and enabling long-term storage 12,13 . The whole system is differentiable, and can therefore be trained end-to-end with gradient descent, allowing the network to learn how to operate and organize the memory in a goal-directed manner.},
archivePrefix = {arXiv},
arxivId = {arXiv:1410.5401v2},
author = {Graves, Alex and Wayne, Greg and Reynolds, Malcolm and Harley, Tim and Danihelka, Ivo and Grabska-Barwi{\'{n}}ska, Agnieszka and Colmenarejo, Sergio G{\'{o}}mez and Grefenstette, Edward and Ramalho, Tiago and Agapiou, John and Badia, Adri{\`{a}} Puigdom{\`{e}}nech and Hermann, Karl Moritz and Zwols, Yori and Ostrovski, Georg and Cain, Adam and King, Helen and Summerfield, Christopher and Blunsom, Phil and Kavukcuoglu, Koray and Hassabis, Demis},
doi = {10.1038/nature20101},
eprint = {arXiv:1410.5401v2},
file = {:home/haakonrr/OneDrive/articles/ml/dnc-graves.pdf:pdf},
isbn = {0896-6273},
issn = {14764687},
journal = {Nature},
number = {7626},
pages = {471--476},
pmid = {26774160},
publisher = {Nature Publishing Group},
title = {{Hybrid computing using a neural network with dynamic external memory}},
url = {http://dx.doi.org/10.1038/nature20101},
volume = {538},
year = {2016}
}
@article{Such2017,
abstract = {Deep artificial neural networks (DNNs) are typically trained via gradient-based learning algorithms, namely backpropagation. Evolution strategies (ES) can rival backprop-based algorithms such as Q-learning and policy gradients on challenging deep reinforcement learning (RL) problems. However, ES can be considered a gradient-based algorithm because it performs stochastic gradient descent via an operation similar to a finite-difference approximation of the gradient. That raises the question of whether non-gradient-based evolutionary algorithms can work at DNN scales. Here we demonstrate they can: we evolve the weights of a DNN with a simple, gradient-free, population-based genetic algorithm (GA) and it performs well on hard deep RL problems, including Atari and humanoid locomotion. The Deep GA successfully evolves networks with over four million free parameters, the largest neural networks ever evolved with a traditional evolutionary algorithm. These results (1) expand our sense of the scale at which GAs can operate, (2) suggest intriguingly that in some cases following the gradient is not the best choice for optimizing performance, and (3) make immediately available the multitude of neuroevolution techniques that improve performance. We demonstrate the latter by showing that combining DNNs with novelty search, which encourages exploration on tasks with deceptive or sparse reward functions, can solve a high-dimensional problem on which reward-maximizing algorithms (e.g.$\backslash$ DQN, A3C, ES, and the GA) fail. Additionally, the Deep GA is faster than ES, A3C, and DQN (it can train Atari in {\$}{\{}\backslashraise.17ex\backslashhbox{\{}{\$}$\backslash$scriptstyle$\backslash$sim{\$}{\}}{\}}{\$}4 hours on one desktop or {\$}{\{}\backslashraise.17ex\backslashhbox{\{}{\$}$\backslash$scriptstyle$\backslash$sim{\$}{\}}{\}}{\$}1 hour distributed on 720 cores), and enables a state-of-the-art, up to 10,000-fold compact encoding technique.},
archivePrefix = {arXiv},
arxivId = {1712.06567},
author = {Such, Felipe Petroski and Madhavan, Vashisht and Conti, Edoardo and Lehman, Joel and Stanley, Kenneth O. and Clune, Jeff},
doi = {1712.06567},
eprint = {1712.06567},
file = {:home/haakonrr/OneDrive/articles/ml/deep-neuroevolution.pdf:pdf},
title = {{Deep Neuroevolution: Genetic Algorithms Are a Competitive Alternative for Training Deep Neural Networks for Reinforcement Learning}},
url = {http://arxiv.org/abs/1712.06567},
year = {2017}
}
@article{Wiig2015,
abstract = {This paper proves that an integral line-of-sight guidance law for path following control of underactuated marine vessels provides uniform semiglobal exponential stability. The stability result is stronger than what has been proved in previous literature, with stronger convergence properties and more robustness. The analysis is based on the 3-dimensional maneuvering control model of marine vessels, which describes both surface vessels and underwater vehicles moving in a horizontal plane. Both the kinematics and dynamics of the system is taken into account, as well as disturbances from constant and irrotational ocean currents. Simulation results are presented to validate the theoretical analysis.},
author = {Wiig, Martin S. and Pettersen, Kristin Y. and Krogstad, Thomas R.},
doi = {10.1016/j.ifacol.2015.10.259},
file = {:home/haakonrr/OneDrive/articles/autonomous/LOS-unifrom-semiglobal-stability-Fossen.pdf:pdf},
issn = {24058963},
journal = {IFAC-PapersOnLine},
keywords = {Cascade control,Disturbance compensation,Exponentially stable,Guidance systems,LOS guidance,Path following,Underactuated vessel},
number = {16},
pages = {61--68},
publisher = {Elsevier Ltd},
title = {{Uniform semiglobal exponential stability of integral line-of-sight guidance laws}},
url = {http://dx.doi.org/10.1016/j.automatica.2014.10.018},
volume = {28},
year = {2015}
}
@article{Phillips2013,
abstract = {In the last few lectures we saw how to convert from a document full of words or characters to a set, and then to a matrix, and then to a k-dimensional vector. And from the final vector we could approximate the Jaccard distance between two documents. However, now we face a new challenge. We have many many documents (say n documents) and we want to find the ones that are close. But we don't want to calculate n 2 ≈ n 2 /2 distances to determine which ones are really similar. In particular, consider we have n = 1,000,000 items and we want to ask two questions: (Q1): Which items are similar? (Q2): Given a query item, which others are similar to the query? For (Q1) we don't want to check all roughly n 2 distance (no matter how fast each computation is), and for (Q2) we don't want to check all n items. In both cases we somehow want to figure out which ones might be close and the check those. Consider n points in the plane R 2 . How can we quickly answer these questions: • Hierarchical models (range trees, kd-trees, B-trees) don't work well in high dimensions. We will return to these in L8. • Lay down a Grid: Close points should be in same grid cell. But some can always lay across the boundary (no matter how close). Some may be further than 1 grid cell, but still close. And in high dimensions, the number of neighboring grid cells grows exponentially. One option is to randomly shift (and rotate) and try again.},
author = {Phillips, Jeff M},
file = {:home/haakonrr/OneDrive/articles/hashing/locality-sensitive-hashing.pdf:pdf},
journal = {Data Mining},
title = {{6 Locality Sensitive Hashing}},
year = {2013}
}
@article{Zhan2016,
abstract = {Policy advice is a transfer learning method where a student agent is able to learn faster via advice from a teacher. However, both this and other reinforcement learning transfer methods have little theoretical analysis. This paper formally defines a setting where multiple teacher agents can provide advice to a student and introduces an algorithm to leverage both autonomous exploration and teacher's advice. Our regret bounds justify the intuition that good teachers help while bad teachers hurt. Using our formalization, we are also able to quantify, for the first time, when negative transfer can occur within such a reinforcement learning setting.},
archivePrefix = {arXiv},
arxivId = {1604.03986},
author = {Zhan, Yusen and Ammar, Haitham Bou and Taylor, Matthew E.},
doi = {10.1038/nature14236},
eprint = {1604.03986},
file = {:home/haakonrr/OneDrive/articles/rl/human-level-control-through-deep-RL.pdf:pdf},
isbn = {1476-4687 (Electronic) 0028-0836 (Linking)},
issn = {10450823},
journal = {IJCAI International Joint Conference on Artificial Intelligence},
number = {7540},
pages = {2315--2321},
pmid = {25719670},
publisher = {Nature Publishing Group},
title = {{Theoretically-grounded policy advice from multiple teachers in reinforcement learning settings with applications to negative transfer}},
url = {http://dx.doi.org/10.1038/nature14236},
volume = {2016-January},
year = {2016}
}
@article{,
file = {:home/haakonrr/OneDrive/articles/hashing/learning-to-hash.pdf:pdf},
pages = {2663--2669},
title = {{L e ar ning “ Fo r giving ” H as h Fun ct ion s : A l go r i t hm s a nd L ar g e S ca le T e sts}},
year = {2005}
}
@article{Wang2014,
abstract = {Similarity search (nearest neighbor search) is a problem of pursuing the data items whose distances to a query item are the smallest from a large database. Various methods have been developed to address this problem, and recently a lot of efforts have been devoted to approximate search. In this paper, we present a survey on one of the main solutions, hashing, which has been widely studied since the pioneering work locality sensitive hashing. We divide the hashing algorithms two main categories: locality sensitive hashing, which designs hash functions without exploring the data distribution and learning to hash, which learns hash functions according the data distribution, and review them from various aspects, including hash function design and distance measure and search scheme in the hash coding space.},
archivePrefix = {arXiv},
arxivId = {1408.2927},
author = {Wang, Jingdong and Shen, Heng Tao and Song, Jingkuan and Ji, Jianqiu},
doi = {10.1561/2200000016},
eprint = {1408.2927},
file = {:home/haakonrr/OneDrive/articles/hashing/hashing-for-similarity-search.pdf:pdf},
isbn = {9781627480031},
issn = {10495258},
pages = {1--29},
pmid = {17504609},
title = {{Hashing for Similarity Search: A Survey}},
url = {http://arxiv.org/abs/1408.2927},
year = {2014}
}
@article{Andoni,
archivePrefix = {arXiv},
arxivId = {arXiv:1306.1547v3},
author = {Andoni, Alexandr and Indyk, Piotr},
doi = {10.1137/1.9781611973402.76},
eprint = {arXiv:1306.1547v3},
file = {:home/haakonrr/OneDrive/articles/hashing/beyond-locality-sensitive-hashing.pdf:pdf},
isbn = {9781611973389},
issn = {9781611973389},
number = {1},
title = {{Beyond Locality-Sensitive Hashing}}
}
@book{Moe2016a,
author = {Moe, Signe},
file = {:home/haakonrr/OneDrive/articles/autonomous/PhD thesis Signe Moe.pdf:pdf},
isbn = {9788232619825},
keywords = {thesis template, NTNU},
number = {November},
title = {{Signe Moe Guidance and Control of Robot Manipulators and Autonomous Marine Robots Signe Moe Guidance and Control of Robot Manipulators and Autonomous Marine Robots}},
year = {2016}
}
@article{Moe2014,
author = {Moe, Signe and Caharija, Walter and Pettersen, Kristin Y and Schjolberg, Ingrid},
file = {:home/haakonrr/OneDrive/articles/autonomous/Master Signe Moe.pdf:pdf},
isbn = {1479932728},
journal = {American Control Conference (ACC), 2014},
pages = {3856--3861},
title = {{Path following of underactuated marine surface vessels in the presence of unknown ocean currents}},
year = {2014}
}
@article{Tokic2010,
abstract = {This paper presents “Value-Difference Based Exploration” (VDBE), a method for balancing the exploration/exploitation dilemma inherent to reinforcement learning. The proposed method adapts the ex- ploration parameter of $\epsilon$-greedy in dependence of the temporal-difference error observed from value-function backups, which is considered as a measure of the agent's uncertainty about the environment. VDBE is evaluated on a multi-armed bandit task, which allows for insight into the behavior of the method. Preliminary results indicate that VDBE seems to be more parameter robust than commonly used ad hoc approaches such as $\epsilon$-greedy or softmax.},
author = {Tokic, Michel},
doi = {10.1007/978-3-642-16111-7_23},
file = {:home/haakonrr/OneDrive/articles/rl/adaptive-epsilon-greedy-exploration.pdf:pdf},
isbn = {3642161103},
issn = {03029743},
journal = {Lecture Notes in Computer Science (including subseries Lecture Notes in Artificial Intelligence and Lecture Notes in Bioinformatics)},
pages = {203--210},
title = {{Adaptive $\epsilon$-greedy exploration in reinforcement learning based on value differences}},
volume = {6359 LNAI},
year = {2010}
}
@article{Schaul2015,
abstract = {Experience replay lets online reinforcement learning agents remember and reuse experiences from the past. In prior work, experience transitions were uniformly sampled from a replay memory. However, this approach simply replays transitions at the same frequency that they were originally experienced, regardless of their significance. In this paper we develop a framework for prioritizing experience, so as to replay important transitions more frequently, and therefore learn more efficiently. We use prioritized experience replay in Deep Q-Networks (DQN), a reinforcement learning algorithm that achieved human-level performance across many Atari games. DQN with prioritized experience replay achieves a new state-of-the-art, outperforming DQN with uniform replay on 41 out of 49 games.},
archivePrefix = {arXiv},
arxivId = {1511.05952},
author = {Schaul, Tom and Quan, John and Antonoglou, Ioannis and Silver, David},
doi = {10.1038/nature14236},
eprint = {1511.05952},
file = {:home/haakonrr/OneDrive/articles/rl/prioritised-experience-replay.pdf:pdf},
isbn = {978-1-4799-0356-6},
issn = {0028-0836},
pages = {1--21},
pmid = {25719670},
title = {{Prioritized Experience Replay}},
url = {http://arxiv.org/abs/1511.05952},
year = {2015}
}
@article{Tsai2014,
abstract = {Hashing has recently attracted considerable attention for large scale similarity search. However, learning compact codes with good performance is still a challenge. In many cases, the real-world data lies on a low-dimensional manifold embedded in high-dimensional ambient space. To capture meaningful neighbors, a compact hashing representation should be able to uncover the intrinsic geometric structure of the manifold, e.g., the neighborhood relationships between subregions. Most existing hashing methods only consider this issue during mapping data points into certain projected dimensions. When getting the binary codes, they either directly quantize the projected values with a threshold, or use an orthogonal matrix to refine the initial projection matrix, which both consider projection and quantization separately, and will not well preserve the locality structure in the whole learning process. In this paper, we propose a novel hashing algorithm called Locality Preserving Hashing to effectively solve the above problems. Specifically, we learn a set of locality preserving projections with a joint optimization framework, which minimizes the average projection distance and quantization loss simultaneously. Experimental comparisons with other state-of-the-art methods on two large scale datasets demonstrate the effectiveness and efficiency of our method.},
author = {Tsai, Yi Hsuan and Yang, Ming Hsuan},
doi = {10.1109/ICIP.2014.7025604},
file = {:home/haakonrr/Documents/articles/locality-preserving-hashing.html:html},
isbn = {9781479957514},
issn = {9781479957514},
journal = {2014 IEEE International Conference on Image Processing, ICIP 2014},
keywords = {Hashing,image retrieval,visual search},
pages = {2988--2992},
title = {{Locality preserving hashing}},
url = {https://ieeexplore.ieee.org/document/7025604/},
year = {2014}
}
@article{Johansen2016,
abstract = {—This paper describes a concept for a collision avoid-ance system for ships, based on model predictive control. A finite set of alternative control behaviors are generated by varying two parameters: offsets to the guidance course angle commanded to the autopilot, and changes to the propulsion command ranging from nominal speed to full reverse. Using simulated predictions of the trajectories of the obstacles and ship, the compliance with COLREGS and collision hazards associated with each of the alternative control behaviors are evaluated on a finite prediction horizon, and the optimal control behavior is selected. Robustness to sensing error, predicted obstacle behavior, and environmental conditions can be ensured by evaluating multiple scenarios for each control behavior. The method is conceptually and computationally simple and yet quite versatile as it can account for the dynamics of the ship, the dynamics of the steering and propulsion system, forces due to wind and ocean current, and any number of obstacles. Simulations show that the method is effective and can manage complex scenarios with multiple dynamic obstacles and uncertainty associated with sensors and predictions.},
author = {Johansen, Tor Arne and Perez, Tristan and Cristofaro, Andrea},
doi = {10.1109/TITS.2016.2551780},
file = {:home/haakonrr/Documents/articles/autonomous/jpc - ship collision avoidance using mpc.pdf:pdf},
isbn = {1524-9050},
issn = {15249050},
journal = {IEEE Transactions on Intelligent Transportation Systems},
keywords = {Autonomous ships,collision avoidance,control systems,hazard,safety,trajectory optimization},
number = {12},
pages = {3407--3422},
title = {{Ship collision avoidance and COLREGS compliance using simulation-based control behavior selection with predictive hazard assessment}},
volume = {17},
year = {2016}
}