version 1.18, 2003/08/06 09:52:29
|
version 1.22, 2003/09/02 10:34:47
|
Line 45
|
Line 45
|
|
|
# Change log: |
# Change log: |
# $Log$ |
# $Log$ |
|
# Revision 1.22 2003/09/02 10:34:47 foxr |
|
# - Fix errors in host dead detection logic (too many cases where the |
|
# retries left were not getting incremented or just not checked). |
|
# - Added some additional status to the ps axuww display: |
|
# o Remaining retries on a host. |
|
# o >>> DEAD <<< indicator if I've given up on a host. |
|
# - Tested the SIGHUP will reset the retries remaining count (thanks to |
|
# the above status stuff, and get allow the loncnew to re-try again |
|
# on the host (thanks to the log). |
|
# |
|
# Revision 1.21 2003/08/26 09:19:51 foxr |
|
# How embarrassing... put in the SocketTimeout function in loncnew and forgot |
|
# to actually hook it into the LondTransaction. Added this to MakeLondConnection |
|
# where it belongs... hopefully transactions (not just connection attempts) will |
|
# timeout more speedily than the socket errors will catch it. |
|
# |
|
# Revision 1.20 2003/08/25 18:48:11 albertel |
|
# - fixing a forgotten ; |
|
# |
|
# Revision 1.19 2003/08/19 09:31:46 foxr |
|
# Get socket directory from configuration rather than the old hard coded test |
|
# way that I forgot to un-hard code. |
|
# |
# Revision 1.18 2003/08/06 09:52:29 foxr |
# Revision 1.18 2003/08/06 09:52:29 foxr |
# Also needed to remember to fail in-flight transactions if their sends fail. |
# Also needed to remember to fail in-flight transactions if their sends fail. |
# |
# |
Line 77
|
Line 100
|
# Revision 1.10 2003/06/24 02:46:04 foxr |
# Revision 1.10 2003/06/24 02:46:04 foxr |
# Put a limit on the number of times we'll retry a connection. |
# Put a limit on the number of times we'll retry a connection. |
# Start getting the signal stuff put in as well...note that need to get signals |
# Start getting the signal stuff put in as well...note that need to get signals |
# going or else 6the client will permanently give up on dead servers. |
# going or else the client will permanently give up on dead servers. |
# |
# |
# Revision 1.9 2003/06/13 02:38:43 foxr |
# Revision 1.9 2003/06/13 02:38:43 foxr |
# Add logging in 'expected format' |
# Add logging in 'expected format' |
Line 146 my $IdleTimeout= 3600; # Wait an hour b
|
Line 169 my $IdleTimeout= 3600; # Wait an hour b
|
# The variables below are only used by the child processes. |
# The variables below are only used by the child processes. |
# |
# |
my $RemoteHost; # Name of host child is talking to. |
my $RemoteHost; # Name of host child is talking to. |
my $UnixSocketDir= "/home/httpd/sockets"; |
my $UnixSocketDir= $perlvar{'lonSockDir'}; |
my $IdleConnections = Stack->new(); # Set of idle connections |
my $IdleConnections = Stack->new(); # Set of idle connections |
my %ActiveConnections; # Connections to the remote lond. |
my %ActiveConnections; # Connections to the remote lond. |
my %ActiveTransactions; # LondTransactions in flight. |
my %ActiveTransactions; # LondTransactions in flight. |
Line 316 sub ShowStatus {
|
Line 339 sub ShowStatus {
|
sub SocketTimeout { |
sub SocketTimeout { |
my $Socket = shift; |
my $Socket = shift; |
|
|
KillSocket($Socket); |
KillSocket($Socket); # A transaction timeout also counts as |
|
# a connection failure: |
|
$ConnectionRetriesLeft--; |
} |
} |
|
|
=pod |
=pod |
Line 330 Invoked each timer tick.
|
Line 355 Invoked each timer tick.
|
|
|
sub Tick { |
sub Tick { |
my $client; |
my $client; |
ShowStatus(GetServerHost()." Connection count: ".$ConnectionCount); |
if($ConnectionRetriesLeft > 0) { |
|
ShowStatus(GetServerHost()." Connection count: ".$ConnectionCount |
|
." Retries remaining: ".$ConnectionRetriesLeft); |
|
} else { |
|
ShowStatus(GetServerHost()." >> DEAD <<"); |
|
} |
# Is it time to prune connection count: |
# Is it time to prune connection count: |
|
|
|
|
Line 362 sub Tick {
|
Line 391 sub Tick {
|
my $Connections = ($Requests <= $MaxConnectionCount) ? |
my $Connections = ($Requests <= $MaxConnectionCount) ? |
$Requests : $MaxConnectionCount; |
$Requests : $MaxConnectionCount; |
Debug(1,"Work but no connections, start ".$Connections." of them"); |
Debug(1,"Work but no connections, start ".$Connections." of them"); |
|
my $successCount = 0; |
for ($i =0; $i < $Connections; $i++) { |
for ($i =0; $i < $Connections; $i++) { |
MakeLondConnection(); |
$successCount += MakeLondConnection(); |
|
} |
|
if($successCount == 0) { # All connections failed: |
|
Debug(1,"Work in queue failed to make any connectiouns\n"); |
|
EmptyQueue(); # Fail pending transactions with con_lost. |
} |
} |
} else { |
} else { |
|
ShowStatus(GetServerHost()." >>> DEAD!!! <<<"); |
Debug(1,"Work in queue, but gave up on connections..flushing\n"); |
Debug(1,"Work in queue, but gave up on connections..flushing\n"); |
EmptyQueue(); # Connections can't be established. |
EmptyQueue(); # Connections can't be established. |
} |
} |
Line 619 sub FailTransaction {
|
Line 654 sub FailTransaction {
|
Debug(1," Replying con_lost to ".$transaction->getRequest()); |
Debug(1," Replying con_lost to ".$transaction->getRequest()); |
StartClientReply($transaction, "con_lost\n"); |
StartClientReply($transaction, "con_lost\n"); |
} |
} |
|
if($ConnectionRetriesLeft <= 0) { |
|
Log("CRITICAL", "Host marked dead: ".GetServerHost()); |
|
} |
|
|
} |
} |
|
|
Line 630 sub FailTransaction {
|
Line 668 sub FailTransaction {
|
|
|
=cut |
=cut |
sub EmptyQueue { |
sub EmptyQueue { |
|
$ConnectionRetriesLeft--; # Counts as connection failure too. |
while($WorkQueue->Count()) { |
while($WorkQueue->Count()) { |
my $request = $WorkQueue->dequeue(); # This is a transaction |
my $request = $WorkQueue->dequeue(); # This is a transaction |
FailTransaction($request); |
FailTransaction($request); |
Line 696 sub KillSocket {
|
Line 735 sub KillSocket {
|
# work queue, the work all gets failed with con_lost. |
# work queue, the work all gets failed with con_lost. |
# |
# |
if($ConnectionCount == 0) { |
if($ConnectionCount == 0) { |
EmptyQueue; |
EmptyQueue(); |
} |
} |
} |
} |
|
|
Line 786 sub LondReadable {
|
Line 825 sub LondReadable {
|
} |
} |
$Watcher->cancel(); |
$Watcher->cancel(); |
KillSocket($Socket); |
KillSocket($Socket); |
|
$ConnectionRetriesLeft--; # Counts as connection failure |
return; |
return; |
} |
} |
SocketDump(6,$Socket); |
SocketDump(6,$Socket); |
Line 819 sub LondReadable {
|
Line 859 sub LondReadable {
|
} elsif ($State eq "Idle") { |
} elsif ($State eq "Idle") { |
# If necessary, complete a transaction and then go into the |
# If necessary, complete a transaction and then go into the |
# idle queue. |
# idle queue. |
|
# Note that a trasition to idle indicates a live lond |
|
# on the other end so reset the connection retries. |
|
# |
|
$ConnectionRetriesLeft = $ConnectionRetries; # success resets the count |
$Watcher->cancel(); |
$Watcher->cancel(); |
if(exists($ActiveTransactions{$Socket})) { |
if(exists($ActiveTransactions{$Socket})) { |
Debug(8,"Completing transaction!!"); |
Debug(8,"Completing transaction!!"); |
Line 1074 sub MakeLondConnection {
|
Line 1118 sub MakeLondConnection {
|
$ConnectionRetriesLeft--; |
$ConnectionRetriesLeft--; |
return 0; # Failure. |
return 0; # Failure. |
} else { |
} else { |
$ConnectionRetriesLeft = $ConnectionRetries; # success resets the count |
|
# The connection needs to have writability |
# The connection needs to have writability |
# monitored in order to send the init sequence |
# monitored in order to send the init sequence |
# that starts the whole authentication/key |
# that starts the whole authentication/key |
Line 1087 sub MakeLondConnection {
|
Line 1131 sub MakeLondConnection {
|
&Debug(9,"MakeLondConnection got socket: ".$Socket); |
&Debug(9,"MakeLondConnection got socket: ".$Socket); |
} |
} |
|
|
|
$Connection->SetTimeoutCallback(\&SocketTimeout); |
|
|
$event = Event->io(fd => $Socket, |
$event = Event->io(fd => $Socket, |
poll => 'w', |
poll => 'w', |
cb => \&LondWritable, |
cb => \&LondWritable, |
Line 1186 sub QueueTransaction {
|
Line 1231 sub QueueTransaction {
|
Debug(8,"Must queue..."); |
Debug(8,"Must queue..."); |
$WorkQueue->enqueue($requestData); |
$WorkQueue->enqueue($requestData); |
if($ConnectionCount < $MaxConnectionCount) { |
if($ConnectionCount < $MaxConnectionCount) { |
Debug(4,"Starting additional lond connection"); |
if($ConnectionRetriesLeft > 0) { |
if(MakeLondConnection() == 0) { |
Debug(4,"Starting additional lond connection"); |
EmptyQueue(); # Fail transactions, can't make connection. |
if(MakeLondConnection() == 0) { |
|
EmptyQueue(); # Fail transactions, can't make connection. |
|
} |
|
} else { |
|
ShowStatus(GetServerHost()." >>> DEAD !!!! <<<"); |
|
EmptyQueue(); # It's worse than that ... he's dead Jim. |
} |
} |
} |
} |
} else { # Can start the request: |
} else { # Can start the request: |
Line 1354 sub SetupLoncListener {
|
Line 1404 sub SetupLoncListener {
|
Child USR1 signal handler to report the most recent status |
Child USR1 signal handler to report the most recent status |
into the status file. |
into the status file. |
|
|
|
We also use this to reset the retries count in order to allow the |
|
client to retry connections with a previously dead server. |
=cut |
=cut |
sub ChildStatus { |
sub ChildStatus { |
my $event = shift; |
my $event = shift; |
Line 1364 sub ChildStatus {
|
Line 1416 sub ChildStatus {
|
my $fh = IO::File->new(">>$docdir/lon-status/loncstatus.txt"); |
my $fh = IO::File->new(">>$docdir/lon-status/loncstatus.txt"); |
print $fh $$."\t".$RemoteHost."\t".$Status."\t". |
print $fh $$."\t".$RemoteHost."\t".$Status."\t". |
$RecentLogEntry."\n"; |
$RecentLogEntry."\n"; |
|
$ConnectionRetriesLeft = $ConnectionRetries; |
} |
} |
|
|
=pod |
=pod |