]> git.proxmox.com Git - pve-ha-manager.git/blame - src/PVE/HA/LRM.pm
add README for regresstion test
[pve-ha-manager.git] / src / PVE / HA / LRM.pm
CommitLineData
5f095798
DM
1package PVE::HA::LRM;
2
3# Local Resource Manager
4
5use strict;
6use warnings;
c4a221bc
DM
7use Data::Dumper;
8use POSIX qw(:sys_wait_h);
5f095798
DM
9
10use PVE::SafeSyslog;
11use PVE::Tools;
12use PVE::HA::Tools;
13
14# Server can have several states:
15
16my $valid_states = {
ec911edd 17 wait_for_agent_lock => "waiting for agent lock",
0bba8f60 18 active => "got agent_lock",
5f095798
DM
19 lost_agent_lock => "lost agent_lock",
20};
21
22sub new {
23 my ($this, $haenv) = @_;
24
25 my $class = ref($this) || $this;
26
27 my $self = bless {
28 haenv => $haenv,
29 status => { state => 'startup' },
c4a221bc
DM
30 workers => {},
31 results => {},
067cdf33 32 shutdown_request => 0,
5f095798
DM
33 }, $class;
34
b0bf08a9 35 $self->set_local_status({ state => 'wait_for_agent_lock' });
5f095798
DM
36
37 return $self;
38}
39
40sub shutdown_request {
41 my ($self) = @_;
42
43 $self->{shutdown_request} = 1;
44}
45
46sub get_local_status {
47 my ($self) = @_;
48
49 return $self->{status};
50}
51
52sub set_local_status {
53 my ($self, $new) = @_;
54
55 die "invalid state '$new->{state}'" if !$valid_states->{$new->{state}};
56
57 my $haenv = $self->{haenv};
58
59 my $old = $self->{status};
60
61 # important: only update if if really changed
62 return if $old->{state} eq $new->{state};
63
0bba8f60 64 $haenv->log('info', "status change $old->{state} => $new->{state}");
5f095798
DM
65
66 $new->{state_change_time} = $haenv->get_time();
67
68 $self->{status} = $new;
69}
70
71sub get_protected_ha_agent_lock {
72 my ($self) = @_;
73
74 my $haenv = $self->{haenv};
75
76 my $count = 0;
77 my $starttime = $haenv->get_time();
78
79 for (;;) {
80
81 if ($haenv->get_ha_agent_lock()) {
82 if ($self->{ha_agent_wd}) {
83 $haenv->watchdog_update($self->{ha_agent_wd});
84 } else {
85 my $wfh = $haenv->watchdog_open();
86 $self->{ha_agent_wd} = $wfh;
87 }
88 return 1;
89 }
90
91 last if ++$count > 5; # try max 5 time
92
93 my $delay = $haenv->get_time() - $starttime;
94 last if $delay > 5; # for max 5 seconds
95
96 $haenv->sleep(1);
97 }
98
99 return 0;
100}
101
067cdf33
DM
102sub fenced_service_count {
103 my ($self) = @_;
104
105 my $haenv = $self->{haenv};
106
107 my $nodename = $haenv->nodename();
108
109 my $ss = $self->{service_status};
110
111 my $count = 0;
112
113 foreach my $sid (keys %$ss) {
114 my $sd = $ss->{$sid};
115 next if !$sd->{node};
116 next if $sd->{node} ne $nodename;
117 my $req_state = $sd->{state};
118 next if !defined($req_state);
119 if ($req_state eq 'fence') {
120 $count++;
121 next;
122 }
123 }
124
125 return $count;
126}
546e2f1f
DM
127
128sub active_service_count {
129 my ($self) = @_;
130
131 my $haenv = $self->{haenv};
132
133 my $nodename = $haenv->nodename();
134
135 my $ss = $self->{service_status};
136
137 my $count = 0;
138
139 foreach my $sid (keys %$ss) {
140 my $sd = $ss->{$sid};
141 next if !$sd->{node};
142 next if $sd->{node} ne $nodename;
143 my $req_state = $sd->{state};
144 next if !defined($req_state);
145 next if $req_state eq 'stopped';
146
147 $count++;
148 }
149
150 return $count;
151}
067cdf33 152
5f095798
DM
153sub do_one_iteration {
154 my ($self) = @_;
155
156 my $haenv = $self->{haenv};
157
158 my $status = $self->get_local_status();
159 my $state = $status->{state};
160
067cdf33
DM
161 my $ms = $haenv->read_manager_status();
162 $self->{service_status} = $ms->{service_status} || {};
163
164 my $fence_request = $self->fenced_service_count();
165
5f095798
DM
166 # do state changes first
167
168 my $ctime = $haenv->get_time();
169
b0bf08a9 170 if ($state eq 'wait_for_agent_lock') {
5f095798 171
546e2f1f 172 my $service_count = $self->active_service_count();
5f095798 173
067cdf33 174 if (!$fence_request && $service_count && $haenv->quorate()) {
0bba8f60
DM
175 if ($self->get_protected_ha_agent_lock()) {
176 $self->set_local_status({ state => 'active' });
5f095798
DM
177 }
178 }
179
180 } elsif ($state eq 'lost_agent_lock') {
181
067cdf33 182 if (!$fence_request && $haenv->quorate()) {
0bba8f60
DM
183 if ($self->get_protected_ha_agent_lock()) {
184 $self->set_local_status({ state => 'active' });
5f095798
DM
185 }
186 }
187
0bba8f60 188 } elsif ($state eq 'active') {
5f095798 189
067cdf33
DM
190 if ($fence_request) {
191 $haenv->log('err', "node need to be fenced - releasing agent_lock\n");
192 $self->set_local_status({ state => 'lost_agent_lock'});
193 } elsif (!$self->get_protected_ha_agent_lock()) {
5f095798
DM
194 $self->set_local_status({ state => 'lost_agent_lock'});
195 }
196 }
197
198 $status = $self->get_local_status();
199 $state = $status->{state};
200
201 # do work
202
203 if ($state eq 'wait_for_agent_lock') {
204
205 return 0 if $self->{shutdown_request};
206
207 $haenv->sleep(5);
208
0bba8f60 209 } elsif ($state eq 'active') {
5f095798
DM
210
211 my $startime = $haenv->get_time();
212
213 my $max_time = 10;
214
215 my $shutdown = 0;
216
217 # do work (max_time seconds)
218 eval {
219 # fixme: set alert timer
220
221 if ($self->{shutdown_request}) {
222
223 # fixme: request service stop or relocate ?
224
546e2f1f 225 my $service_count = $self->active_service_count();
5f095798
DM
226
227 if ($service_count == 0) {
228
229 if ($self->{ha_agent_wd}) {
230 $haenv->watchdog_close($self->{ha_agent_wd});
231 delete $self->{ha_agent_wd};
232 }
233
234 $shutdown = 1;
235 }
c4a221bc 236 } else {
c4a221bc
DM
237
238 $self->manage_resources();
067cdf33 239
5f095798
DM
240 }
241 };
242 if (my $err = $@) {
243 $haenv->log('err', "got unexpected error - $err");
244 }
245
246 return 0 if $shutdown;
247
248 $haenv->sleep_until($startime + $max_time);
249
250 } elsif ($state eq 'lost_agent_lock') {
251
252 # Note: watchdog is active an will triger soon!
253
254 # so we hope to get the lock back soon!
255
256 if ($self->{shutdown_request}) {
257
546e2f1f 258 my $service_count = $self->active_service_count();
5f095798 259
546e2f1f 260 if ($service_count > 0) {
5f095798 261 $haenv->log('err', "get shutdown request in state 'lost_agent_lock' - " .
546e2f1f 262 "detected $service_count running services");
5f095798 263
546e2f1f 264 } else {
5f095798 265
546e2f1f 266 # all services are stopped, so we can close the watchdog
5f095798 267
546e2f1f
DM
268 if ($self->{ha_agent_wd}) {
269 $haenv->watchdog_close($self->{ha_agent_wd});
270 delete $self->{ha_agent_wd};
271 }
272
273 return 0;
5f095798 274 }
5f095798
DM
275 }
276
b0bf08a9
DM
277 $haenv->sleep(5);
278
5f095798
DM
279 } else {
280
281 die "got unexpected status '$state'\n";
282
283 }
284
285 return 1;
286}
287
c4a221bc
DM
288sub manage_resources {
289 my ($self) = @_;
290
291 my $haenv = $self->{haenv};
292
293 my $nodename = $haenv->nodename();
294
c4a221bc
DM
295 my $ss = $self->{service_status};
296
297 foreach my $sid (keys %$ss) {
298 my $sd = $ss->{$sid};
299 next if !$sd->{node};
300 next if !$sd->{uid};
301 next if $sd->{node} ne $nodename;
302 my $req_state = $sd->{state};
303 next if !defined($req_state);
c4a221bc 304 eval {
e88469ba 305 $self->queue_resource_command($sid, $sd->{uid}, $req_state, $sd->{target});
c4a221bc
DM
306 };
307 if (my $err = $@) {
f31b7e94 308 $haenv->log('err', "unable to run resource agent for '$sid' - $err"); # fixme
c4a221bc
DM
309 }
310 }
311
f31b7e94 312 my $starttime = $haenv->get_time();
c4a221bc
DM
313
314 # start workers
315 my $max_workers = 4;
316
6dbf93a0 317 my $sc = $haenv->read_service_config();
f31b7e94
DM
318
319 while (($haenv->get_time() - $starttime) < 5) {
c4a221bc
DM
320 my $count = $self->check_active_workers();
321
322 foreach my $sid (keys %{$self->{workers}}) {
323 last if $count >= $max_workers;
324 my $w = $self->{workers}->{$sid};
6dbf93a0
DM
325 my $cd = $sc->{$sid};
326 if (!$cd) {
f31b7e94 327 $haenv->log('err', "missing resource configuration for '$sid'");
6dbf93a0
DM
328 next;
329 }
c4a221bc 330 if (!$w->{pid}) {
f31b7e94
DM
331 if ($haenv->can_fork()) {
332 my $pid = fork();
333 if (!defined($pid)) {
334 $haenv->log('err', "fork worker failed");
335 $count = 0; last; # abort, try later
336 } elsif ($pid == 0) {
337 # do work
338 my $res = -1;
339 eval {
340 $res = $haenv->exec_resource_agent($sid, $cd, $w->{state}, $w->{target});
341 };
342 if (my $err = $@) {
343 $haenv->log('err', $err);
344 POSIX::_exit(-1);
345 }
346 POSIX::_exit($res);
347 } else {
348 $count++;
349 $w->{pid} = $pid;
350 }
351 } else {
c4a221bc
DM
352 my $res = -1;
353 eval {
6dbf93a0 354 $res = $haenv->exec_resource_agent($sid, $cd, $w->{state}, $w->{target});
c4a221bc
DM
355 };
356 if (my $err = $@) {
f31b7e94
DM
357 $haenv->log('err', $err);
358 }
359 $self->resource_command_finished($sid, $w->{uid}, $res);
c4a221bc
DM
360 }
361 }
362 }
363
364 last if !$count;
365
f31b7e94 366 $haenv->sleep(1);
c4a221bc
DM
367 }
368}
369
370# fixme: use a queue an limit number of parallel workers?
371sub queue_resource_command {
e88469ba 372 my ($self, $sid, $uid, $state, $target) = @_;
c4a221bc
DM
373
374 if (my $w = $self->{workers}->{$sid}) {
375 return if $w->{pid}; # already started
376 # else, delete and overwrite queue entry with new command
377 delete $self->{workers}->{$sid};
378 }
379
380 $self->{workers}->{$sid} = {
381 sid => $sid,
382 uid => $uid,
383 state => $state,
384 };
e88469ba
DM
385
386 $self->{workers}->{$sid}->{target} = $target if $target;
c4a221bc
DM
387}
388
389sub check_active_workers {
390 my ($self) = @_;
391
392 # finish/count workers
393 my $count = 0;
394 foreach my $sid (keys %{$self->{workers}}) {
395 my $w = $self->{workers}->{$sid};
396 if (my $pid = $w->{pid}) {
397 # check status
398 my $waitpid = waitpid($pid, WNOHANG);
399 if (defined($waitpid) && ($waitpid == $pid)) {
400 $self->resource_command_finished($sid, $w->{uid}, $?);
401 } else {
402 $count++;
403 }
404 }
405 }
406
407 return $count;
408}
409
410sub resource_command_finished {
411 my ($self, $sid, $uid, $status) = @_;
412
413 my $haenv = $self->{haenv};
414
415 my $w = delete $self->{workers}->{$sid};
416 return if !$w; # should not happen
417
418 my $exit_code = -1;
419
420 if ($status == -1) {
0f70400d 421 $haenv->log('err', "resource agent $sid finished - failed to execute");
c4a221bc 422 } elsif (my $sig = ($status & 127)) {
0f70400d 423 $haenv->log('err', "resource agent $sid finished - got signal $sig");
c4a221bc
DM
424 } else {
425 $exit_code = ($status >> 8);
c4a221bc
DM
426 }
427
428 $self->{results}->{$uid} = {
429 sid => $w->{sid},
430 state => $w->{state},
431 exit_code => $exit_code,
432 };
433
434 my $ss = $self->{service_status};
435
436 # compute hash of valid/existing uids
437 my $valid_uids = {};
438 foreach my $sid (keys %$ss) {
439 my $sd = $ss->{$sid};
440 next if !$sd->{uid};
441 $valid_uids->{$sd->{uid}} = 1;
442 }
443
444 my $results = {};
445 foreach my $id (keys %{$self->{results}}) {
446 next if !$valid_uids->{$id};
447 $results->{$id} = $self->{results}->{$id};
448 }
449 $self->{results} = $results;
450
451 $haenv->write_lrm_status($results);
452}
453
5f095798 4541;