API-Docker

 view release on metacpan or  search on metacpan

lib/API/Docker/Role/HTTP.pm  view on Meta::CPAN

package API::Docker::Role::HTTP;
# ABSTRACT: HTTP transport role for Docker Engine API
our $VERSION = '0.004';
use Moo::Role;
use IO::Socket::UNIX;
use IO::Socket::INET;
# For the sysread method on a plain filehandle: _pull calls it as a method so
# that IO::Socket::SSL's own gets picked up rather than the builtin. See _pull.
use IO::Handle;
use Socket qw( SOL_SOCKET SO_RCVTIMEO );
# How a read that delivered nothing says it ran out of time rather than out of
# stream (EAGAIN/EWOULDBLOCK), and how it says it was interrupted rather than
# either (EINTR). See _pull and _timed_out.
use Errno qw( EAGAIN EWOULDBLOCK EINTR );
use JSON::MaybeXS qw( encode_json decode_json );
use Scalar::Util qw( looks_like_number );
use Path::Tiny;
use Carp qw( croak shortmess );
use Log::Any qw( $log );
use API::Docker::Error::HTTP;
use API::Docker::Error::Stream;
use API::Docker::Error::Timeout;
use API::Docker::Error::Truncated;
use namespace::clean;


requires 'host';
requires 'api_version';
requires 'tls';
requires 'cert_path';
requires 'tls_insecure';

# Docker stream frame types, indexed by the first byte of the frame header.
my @STREAM_TYPE = qw( stdin stdout stderr );

# A field name is an RFC 9110 token and nothing else. Anything outside this
# set -- CR, LF, a space, a colon -- is rejected rather than stripped; see
# _assert_header_name.
my $HEADER_NAME = qr/\A[0-9A-Za-z!#\$%&'*+.^_`|~-]+\z/;

# The request-target path is caller data -- a container name, an image
# reference -- spliced straight into the request line as /v$version$path, so a
# byte the line's own grammar reads rewrites the request rather than naming a
# resource: CR or LF ends the line, a space opens the HTTP-version field, and
# a ? or # opens the query or fragment. It is held to the RFC 3986 origin-form
# path character set -- unreserved, the sub-delims, and : @ % / -- and
# rejected, not sanitised, for the reason a header name is (see
# _assert_request_path). Query parameters carry the ? and everything after it
# and are assembled separately below, each element run through _uri_encode.
my $REQUEST_PATH = qr{\A[A-Za-z0-9\-._~:/\@!\$&'()*+,;=%]*\z};

# The three units a response can be cut into, one option each. A request picks
# one of them, or none and gets the buffered path; see _stream_handler.
my @STREAM_OPTION = qw( on_event on_frame on_chunk );

# What a response body has to start with to be worth handing to decode_json.
# An object or an array is not the whole of JSON: the engine answers several
# endpoints with a bare JSON scalar, and a `null` used to come back as the
# four-character string 'null'. See _request.
my $JSON_BODY = qr/\A\s*(?:[\[\{"]|-?[0-9]|true|false|null)/;

# How much is asked for per sysread. Strictly an upper bound -- sysread
# returns what has arrived rather than filling to it (see _pull), so on a live
# feed a call typically comes back with one burst, and asking for 64K costs
# nothing but the size of the buffer it lands in.
my $READ_SIZE = 64 * 1024;

has read_timeout => (
  is => 'ro',
);


has connect_timeout => (
  is => 'ro',
);


has _socket => (
  is      => 'lazy',
  clearer => '_clear_socket',
);

# The connect timeout and the endpoint it belongs to, on their way to
# _build__socket. It is a lazy builder and so cannot be handed an argument,
# and the value is per request rather than per client -- hence rw, with
# _reconnect as the only writer, setting it immediately before the build and
# clearing it immediately after. Unset is the whole of the old behaviour: no
# Timeout on any constructor, and the plain croak on a failure.
has _pending_connect => (
  is       => 'rw',
  init_arg => undef,
);

sub _build__socket {
  my ($self) = @_;
  my $host    = $self->host;
  my $pending = $self->_pending_connect;
  my $timeout = $pending ? $pending->{timeout} : undef;

  if ($host =~ m{^unix://(.+)$}) {
    my $path = $1;
    $log->debugf("Connecting to Unix socket: %s", $path);
    my $sock = IO::Socket::UNIX->new(
      Peer => $path,
      Type => SOCK_STREAM,
      $timeout ? (Timeout => $timeout) : (),
    );
    unless ($sock) {
      # Asked before anything else can touch $@ or $!, which is the whole of
      # the evidence; see _connect_expired.
      $self->_croak_connect_timeout($pending, 'unix://' . $path)
        if $self->_connect_expired($timeout);
      croak "Cannot connect to Unix socket $path: $!";
    }
    return $sock;
  }
  elsif ($host =~ m{^tcp://([^:]+):(\d+)$}) {
    my ($addr, $port) = ($1, $2);

    unless ($self->tls) {
      $log->debugf("Connecting to TCP %s:%s", $addr, $port);
      my $sock = IO::Socket::INET->new(
        PeerAddr => $addr,
        PeerPort => $port,
        Proto    => 'tcp',
        $timeout ? (Timeout => $timeout) : (),
      );
      unless ($sock) {
        $self->_croak_connect_timeout($pending, $addr . ':' . $port)
          if $self->_connect_expired($timeout);
        croak "Cannot connect to $addr:$port: $!";
      }
      return $sock;
    }

    # Built before the connection is opened: a cert_path that names nothing,
    # or half a client certificate, is a configuration mistake and the caller
    # should hear about it as one rather than as a handshake failure.
    my %ssl = $self->_ssl_options($addr);

    $log->debugf("Connecting to TCP %s:%s over TLS (verification %s)",
      $addr, $port, $self->tls_insecure ? 'off' : 'on');
    my $sock = IO::Socket::SSL->new(
      PeerAddr => $addr,
      PeerPort => $port,
      Proto    => 'tcp',
      $timeout ? (Timeout => $timeout) : (),

lib/API/Docker/Role/HTTP.pm  view on Meta::CPAN

  # worth stopping on rather than quietly connecting without the certificates
  # the caller believes are in use.
  my $path = path($dir);
  croak __PACKAGE__ . ": cert_path $dir is not a directory. TLS expects the "
    . 'layout the docker CLI writes -- ca.pem, cert.pem and key.pem in one '
    . 'directory -- and this names nothing that could hold it'
    unless $path->is_dir;

  my %ssl;

  # No ca.pem is not an error: verifying a daemon behind a terminator with a
  # publicly trusted certificate needs no private trust anchor, and the
  # default store is then the right one. See L</"TLS on a tcp:// connection">.
  my $ca = $path->child('ca.pem');
  $ssl{SSL_ca_file} = "$ca" if $ca->exists;

  my $cert = $path->child('cert.pem');
  my $key  = $path->child('key.pem');
  my @half = grep { !$_->[1]->exists }
    ( [ 'cert.pem', $cert ], [ 'key.pem', $key ] );

  # One of the two is never a mode, only ever an accident: a key with no
  # certificate proves nothing and a certificate with no key cannot be used.
  croak __PACKAGE__ . ': cert_path ' . $dir . ' has ' . $half[0][0]
    . ' missing while the other half of the client certificate is there. '
    . 'Both cert.pem and key.pem are needed, or neither'
    if @half == 1;

  if (!@half) {
    $ssl{SSL_cert_file} = "$cert";
    $ssl{SSL_key_file}  = "$key";
  }

  return %ssl;
}

sub _reconnect {
  my ($self, $pending) = @_;
  $self->_clear_socket;

  # Cleared on the way out whichever way the build went, so a later _socket
  # built by anything but a request -- a test subclass, a caller reaching for
  # it directly -- never picks up the last request's bound.
  $self->_pending_connect($pending);
  my $sock;
  my $ok  = eval { $sock = $self->_socket; 1 };
  my $err = $@;
  $self->_pending_connect(undef);
  die $err unless $ok;

  return $sock;
}

# Whether the connect that has just failed failed because the bound fired.
# Asked with nothing in between, because $@ and $! are the whole of the
# evidence and both are global.
#
# $@ rather than errno: IO::Socket writes 'connect: timeout' there, and only
# there, when its own select() ran out -- measured, against a host that drops
# SYNs, where $! is ETIMEDOUT, which the kernel also produces on its own after
# two minutes with no Timeout set at all.
#
# EAGAIN is the second shape and belongs to unix:// alone. Measured against a
# listener whose backlog is full: with no Timeout the connect blocks
# indefinitely (still blocked after 8s), and with one it fails at once with
# EAGAIN, because IO::Socket does the timed connect non-blocking and an
# AF_UNIX connect has no in-progress state to wait on. So on that transport
# the option does not wait, it refuses -- but a connect that failed with
# EAGAIN is still one the bound ended, and reporting it as anything else would
# name a cause the caller cannot act on.
sub _connect_expired {
  my ($self, $timeout) = @_;

  return 0 unless $timeout;
  return 1 if defined $@ && $@ =~ /connect: timeout\z/;
  return 1 if $! == EAGAIN || $! == EWOULDBLOCK;
  return 0;
}

sub _croak_connect_timeout {
  my ($self, $pending, $where) = @_;

  my $endpoint = $pending->{endpoint};
  # See _croak_timeout for why the object goes into a variable first and why
  # the location is captured by hand.
  my $error = API::Docker::Error::Timeout->new(
    message  => 'Docker API connect timeout'
      . (defined $endpoint && length $endpoint ? ' (' . $endpoint . ')' : '')
      . ': ' . $where . ' did not accept within ' . $pending->{timeout} . 's',
    location => shortmess(''),
    endpoint => defined $endpoint ? $endpoint : '',
    timeout  => $pending->{timeout},
    phase    => 'connect',
  );
  croak $error;
}

# undef for "no timeout", which is both the default and the explicit 0, so a
# client carrying a default can be opted out of for one request. Anything that
# is not a non-negative number is a caller mistake and is refused rather than
# rounded to something: silently reading a typo as "off" would hand back the
# hang the caller was asking to be protected from.
sub _timeout_value {
  my ($self, $name, $timeout) = @_;

  return undef unless defined $timeout;
  croak __PACKAGE__ . '->_request ' . $name . ' must be a non-negative number '
    . 'of seconds (0 or undef for none), not "' . $timeout . '"'
    unless !ref $timeout && looks_like_number($timeout) && $timeout >= 0;

  return $timeout > 0 ? $timeout : undef;
}

sub _read_timeout_value {
  my ($self, $timeout) = @_;
  return $self->_timeout_value('read_timeout', $timeout);
}

sub _connect_timeout_value {
  my ($self, $timeout) = @_;
  return $self->_timeout_value('connect_timeout', $timeout);
}

# Why SO_RCVTIMEO and not select(): a bound that reads the socket cannot see
# what is already buffered above it, and would fire while the data it was
# waiting for was in hand. That was true of PerlIO's read-ahead when this was
# written (measured: after one readline of a socket holding
# "one\ntwo\nthree\n", two whole lines sit in the PerlIO buffer and select()
# says the handle is not ready), and it is true of _read_buffer now. A
# select-based bound would have to be asked only when that buffer is empty,
# which is one more invariant to keep for no gain: SO_RCVTIMEO bounds the one
# syscall in _pull for one setsockopt, and gets idle-since-the-last-byte
# semantics for free, which is the semantics these endpoints need (karr k52:
# the buffered frames arrive, and *then* the socket stalls -- a
# time-to-first-byte bound would never fire).
sub _apply_read_timeout {
  my ($self, $sock, $timeout) = @_;

  return unless $timeout;

  my $packed;
  if ($^O eq 'MSWin32' || $^O eq 'cygwin') {
    # Winsock takes a DWORD of milliseconds here rather than a struct timeval,
    # and reads a zero as "wait forever" -- so a sub-millisecond request is
    # rounded up instead of becoming the hang it asked to avoid. Reasoned from
    # the Winsock documentation and NOT measured: there is no Windows here.
    # What makes that safe to ship is the croak below -- a shape the platform
    # rejects is reported rather than ignored.
    my $ms = int($timeout * 1000 + 0.5);
    $ms = 1 if $ms < 1;
    $packed = pack('L', $ms);
  }
  else {
    # struct timeval: two native longs, seconds then microseconds. Measured on
    # Linux x86_64 against unix://, plain tcp:// and TLS.
    my $sec  = int($timeout);
    my $usec = int(($timeout - $sec) * 1_000_000 + 0.5);
    if ($usec >= 1_000_000) { $sec++; $usec -= 1_000_000 }
    $packed = pack('l!l!', $sec, $usec);
  }

  # Never a warning and never a silent pass. A caller that asked for a bound
  # and did not get one is left waiting on exactly the hang the option exists
  # to end, and would have no way to tell that from a daemon being slow.
  setsockopt($sock, SOL_SOCKET, SO_RCVTIMEO, $packed)
    or croak __PACKAGE__ . ': cannot set a read timeout of ' . $timeout
      . 's on this socket: ' . $! . '. Refusing to continue without it -- a '
      . 'bound that is not in force is worse than no bound at all, because '
      . 'the caller is relying on it';

  return;
}

# A read that did not deliver did not deliver for one of two reasons, and they
# are not the same thing: the stream ended, or the clock ran out. errno is the
# only thing that separates them -- measured, both eof() and $fh->error are
# true after a timeout just as they are at a clean end, and asking eof() costs
# a second full timeout. So $! is zeroed immediately before the read in _pull
# and captured immediately after it, with nothing in between: it is only
# meaningful after a failure, and any operation in between would overwrite it.
#
# Without this the readers would take a timeout for the end of the response and
# return a truncated body as a whole one. That is the reason karr k59 is not
# just the setsockopt: switching the option on alone would turn a hang into
# silent data loss, which is the worse of the two.
sub _timed_out {
  my ($self, $ctx, $errno) = @_;

  return 0 unless $ctx->{timeout};
  return ($errno == EAGAIN || $errno == EWOULDBLOCK) ? 1 : 0;
}

sub _croak_timeout {
  my ($self, $ctx, $partial) = @_;

  $partial = '' unless defined $partial;
  # Only ever set once a stream is past its status line, so an error body read
  # whole on the way to a >= 400 croak is still reported in bytes.
  my $summary = $ctx->{summary} ? $ctx->{summary}->() : undef;

  my $after = $summary
    ? ' after ' . $summary->{delivered} . ' unit'
      . ($summary->{delivered} == 1 ? '' : 's')
    : length($partial)
      ? ' after ' . length($partial) . ' byte'
        . (length($partial) == 1 ? '' : 's')
      : ', nothing arrived at all';

  # Carp hands a reference straight back rather than decorating it, so this
  # croak is a die with an object -- hence the location captured by hand,
  # which names the same frame a croak of a plain string would have named.
  # The object goes into a variable first: `croak CLASS->new(...)` is indirect
  # object syntax and parses as CLASS->croak(new(...)).
  my $error = API::Docker::Error::Timeout->new(
    message  => 'Docker API read timeout (' . $ctx->{endpoint} . '): '
      . $ctx->{timeout} . 's of silence' . $after,
    location => shortmess(''),
    endpoint => $ctx->{endpoint},
    timeout  => $ctx->{timeout},
    partial  => $partial,
    summary  => $summary,
  );
  croak $error;
}

# The other way a response ends before it is finished, and the one that needs
# no option to be armed: the daemon closed mid-sentence (karr k64).
#
# It is deliberately not folded into _croak_timeout. A timeout is the absence
# of an answer inside a bound the caller asked for, and it can fire on a
# response that would have completed; this is a statement about the response
# itself, made by comparing the body against what the response announced, and
# it fires whether or not anything was bounded. The two carry the same
# "here is what did arrive" contract and nothing else.
#
# $ctx->{partial} and $ctx->{summary} are read exactly as _croak_timeout reads
# them, so a buffered read reports bytes and a streamed one reports units,
# with no site having to know which it is.
sub _croak_truncated {
  my ($self, $ctx, %what) = @_;

  my $summary = $ctx->{summary} ? $ctx->{summary}->() : undef;
  my $partial = $ctx->{partial} ? ${ $ctx->{partial} } : '';

  my $arrived = $summary
    ? $summary->{delivered} . ' unit'
      . ($summary->{delivered} == 1 ? '' : 's') . ' delivered'
    : length($partial)
      ? length($partial) . ' byte'
        . (length($partial) == 1 ? '' : 's') . ' arrived'
      : 'nothing arrived at all';

  # Empty when a reader is driven directly rather than through _request, which
  # is how t/role_http.t drives them: an endpoint nobody named is left out of
  # the message rather than interpolated as the empty string.
  my $endpoint = defined $ctx->{endpoint} ? $ctx->{endpoint} : '';

lib/API/Docker/Role/HTTP.pm  view on Meta::CPAN

  # detail, there being no count to put in one.
  my $detail = $what{detail};
  unless (defined $detail) {
    my $short = $what{expected} - $what{received};
    $detail = $what{piece} . ' stopped ' . $short . ' byte'
      . ($short == 1 ? '' : 's') . ' short of the ' . $what{expected}
      . ' it announced';
  }

  # Carp hands a reference straight back rather than decorating it, so this
  # croak is a die with an object -- hence the location captured by hand,
  # which names the same frame a croak of a plain string would have named.
  # The object goes into a variable first: `croak CLASS->new(...)` is indirect
  # object syntax and parses as CLASS->croak(new(...)).
  my $error = API::Docker::Error::Truncated->new(
    message => 'Docker API response truncated'
      . (length $endpoint ? ' (' . $endpoint . ')' : '') . ': '
      . $detail . '; ' . $arrived,
    location => shortmess(''),
    endpoint => $endpoint,
    phase    => $what{phase},
    expected => $what{expected},
    received => $what{received},
    partial  => $partial,
    summary  => $summary,
  );
  croak $error;
}

# ---------------------------------------------------------------------------
# Reading, in one buffer regime
#
# Every byte of a response is taken off the handle by _pull and by nothing
# else, and every reader below is served out of the buffer _pull fills. That
# is not an optimisation, it is the only shape that works (karr k60).
#
# What forced it: perl's read() is fread-shaped. It loops until it has the
# LENGTH it was asked for or the stream ends -- it does not return what has
# arrived. Measured on an AF_UNIX socketpair whose peer writes 6 bytes, waits
# half a second, writes 6 more and closes: read($sock, $buf, 65536) came back
# with 12 after 0.90s, having waited for the close, while sysread came back
# with 6 in 0.00s. On the endpoints with neither a Content-Length nor chunked
# encoding -- attach, logs(follow), exec/start, all
# application/vnd.docker.raw-stream -- the reader asks for $READ_SIZE, so
# read() delivered nothing to an on_frame/on_chunk callback until 64K had
# piled up or the daemon hung up. On a stream that never ends it would deliver
# nothing at all. The POD promised those callbacks the bytes as they arrive,
# and that promise was not kept.
#
# Why it could not be fixed at the one site that had the bug: _read_head read
# the status line and the headers with <$sock>, and PerlIO reads ahead. The
# bytes past the header block were sitting in a buffer this code cannot reach
# -- there is no supported way to take them back out; ungetc is layer-
# dependent, seek does not work on a socket, and select/MSG_PEEK see the
# kernel's buffer rather than PerlIO's. So switching only the body reads to
# sysread would have silently dropped the start of every body. Either all read
# sites move together or none do.
#
# The buffer lives on the handle rather than on the client or in the context:
# it is the unconsumed bytes of *that* handle, its lifetime is the handle's,
# and a client that opens a socket per request therefore has nothing to reset.
# ${*$sock}{...} is the IO::Socket idiom for exactly this and was measured to
# work on a real socket, a lexical filehandle, a bareword glob and a tied
# handle alike.
my $RBUF = __PACKAGE__ . '/rbuf';

sub _read_buffer {
  my ($self, $sock) = @_;

  ${*$sock}{$RBUF} = '' unless defined ${*$sock}{$RBUF};
  return \${*$sock}{$RBUF};
}

# The one physical read. It answers with what happened rather than with a
# count, because a count makes every reader re-derive the same distinction and
# an undef quietly becoming an end of stream at any one of them is the silent
# truncation this transport must not have:
#
#   'data'     something was appended to the buffer -- however little
#   'eof'      the stream ended
#   'timeout'  the read_timeout expired with nothing to show for it
#
# What TLS does here, since it is not obvious from the call. IO::Socket::SSL
# ties the glob, so the builtin sysread reaches the same place -- SSL_HANDLE's
# READ delegates to the object's own sysread -- and the method form is written
# out only so that the dispatch is visible rather than accidental. What does
# matter is sysread rather than read: IO::Socket::SSL's sysread is a single
# Net::SSLeay::read, one record, while its read is ssl_read_all on a blocking
# socket, which is the same fill semantics this is here to get away from.
#
# A short positive read is never an end of stream and never an expiry. Over
# TLS it is the normal case, one plaintext record at a time; over a plain
# socket it is whatever the kernel had. Both are data.
#
# SSL_WANT_READ and SSL_WANT_WRITE are deliberately not retried. On a blocking
# socket they arrive as EWOULDBLOCK (IO::Socket::SSL's _skip_rw_error does
# `$! ||= EWOULDBLOCK`) and mean the underlying receive would have blocked --
# which, with SO_RCVTIMEO in force, is the bound firing and nothing else.
# Retrying would be a busy loop on WANT_READ and could not make progress on
# WANT_WRITE in any case, so they are reported as the timeout they are.
sub _pull {
  my ($self, $sock, $ctx) = @_;

  my $buf = $self->_read_buffer($sock);

  while (1) {
    # errno immediately before, errno immediately after, nothing in between:
    # it is only meaningful after a failure, and any operation at all would
    # overwrite it. See _timed_out.
    $! = 0;
    my $n = $sock->sysread(my $got, $READ_SIZE);
    my $errno = 0 + $!;

    if (defined $n) {
      return 'eof' unless $n;
      $$buf .= $got;
      return 'data';
    }

    # A signal is not an answer. perl's read() retried here of its own accord
    # (PerlIOUnix_read loops while errno is EINTR), so retrying keeps the
    # behaviour this replaces rather than introducing one.
    next if $errno == EINTR;

    return 'timeout' if $self->_timed_out($ctx, $errno);

    # Anything else -- a reset connection, a handle that cannot be read at
    # all -- ends the response, which is what it did before this too: every
    # reader answered a failed read with `last unless $n`. Whether the
    # response was complete when it ended is a question about its structure,
    # and is asked by the readers that know the structure.
    return 'eof';
  }
}

# The two reads every reader below is built out of. Both serve from the buffer
# and pull only when it is empty, so both hand back what has arrived rather
# than waiting for what was asked for.
sub _read_line {
  my ($self, $sock, $ctx) = @_;
  $ctx ||= {};

  my $buf = $self->_read_buffer($sock);
  my $idx = index($$buf, "\n");

  while ($idx < 0) {
    my $kind = $self->_pull($sock, $ctx);
    # The part of a line already in the buffer is dropped, exactly as the
    # readline this replaces dropped it: every line read here is protocol --
    # a status line, a header, a chunk header -- never payload, so there is no
    # callback it could belong to. What a buffered body had collected is
    # reported instead, from $ctx->{partial}.
    $self->_croak_timeout($ctx, $ctx->{partial} ? ${ $ctx->{partial} } : '')
      if $kind eq 'timeout';
    last if $kind eq 'eof';
    $idx = index($$buf, "\n");
  }

  return substr($$buf, 0, $idx + 1, '') if $idx >= 0;

  # The stream ended. Whatever is left is a final line with no terminator,
  # which is what readline hands back there as well; nothing left is undef.
  return undef unless length $$buf;
  return substr($$buf, 0, length($$buf), '');
}

# Returns (count, bytes) like the read() it replaces, so `last unless $n`
# still ends a loop at the end of the response. What changed is the count: it
# is now what had arrived, never more and no longer padded out by waiting.
# Every caller loops until it has what it needs, so a short count is a
# delivery rather than a truncation.
sub _read_bytes {
  my ($self, $sock, $want, $ctx) = @_;
  $ctx ||= {};

  my $buf = $self->_read_buffer($sock);

  if (!length $$buf) {
    my $kind = $self->_pull($sock, $ctx);
    # Nothing is lost with this exception, and unlike under read() nothing has
    # to be rescued for it either. sysread does not hand back data and EAGAIN
    # together the way PerlIO's read() did: the bytes of every successful pull
    # are already in the accumulator or already through the callback by the
    # time a later pull expires, and the pull that expires carries none.
    $self->_croak_timeout($ctx, $ctx->{partial} ? ${ $ctx->{partial} } : '')
      if $kind eq 'timeout';
    return (0, '') if $kind eq 'eof';

lib/API/Docker/Role/HTTP.pm  view on Meta::CPAN

=item * Demultiplexing of the Docker stream format (L</stream_frames>)

=item * Incremental delivery of a response through a per-request callback, so
the endpoints that never close are usable at all (L</"Streaming a response as
it arrives">)

=item * Request/response logging via L<Log::Any>

=item * Automatic connection management

=back

Consuming classes must provide C<host>, C<api_version>, C<tls>, C<cert_path>
and C<tls_insecure> attributes. The last three are read only by the C<tcp://>
branch of the socket builder, and only when TLS is asked for, but the contract
is stated once rather than probed for at connect time.

A C<unix://> connection is a local socket with no wire to protect and is never
encrypted; it ignores all three attributes, and L<API::Docker> refuses the
combination at construction rather than letting a request for an encrypted
transport be answered with an unencrypted one. A C<tcp://> connection is
B<plaintext unless C<< tls => 1 >>>, which is the whole of the difference --
see L</"TLS on a tcp:// connection">.

=head2 TLS on a tcp:// connection

C<< tls => 1 >> replaces the L<IO::Socket::INET> connection with an
L<IO::Socket::SSL> one and changes nothing else: the same request writer, the
same reader, the same everything above the socket.

    my $docker = API::Docker->new(
      host      => 'tcp://dockerhost:2376',
      tls       => 1,
      cert_path => '/home/me/.docker',
    );

=head3 What the certificates are, and where

C<cert_path> names a directory in the layout the C<docker> CLI writes, and
each of the three files is used if it is there:

=over

=item * F<ca.pem> - the trust anchor the daemon's certificate is checked
against

=item * F<cert.pem> and F<key.pem> - this client's certificate and private
key, sent when the daemon asks the client to identify itself

=back

The two halves of the client certificate go together: one of them present
without the other is a croak, because a key with no certificate proves nothing
and a certificate with no key cannot be used. A directory holding only
F<ca.pem> is fine -- that is a daemon this client verifies but does not
authenticate to. A C<cert_path> that names nothing is a croak: it is read only
once TLS was asked for, and at that point a path pointing nowhere means the
caller believes certificates are in use that are not.

C<cert_path> defaults from C<DOCKER_CERT_PATH>, so on a machine that also runs
the C<docker> CLI it arrives set. Without C<< tls => 1 >> nothing reads it, so
that costs nothing; with it, pass C<< cert_path => undef >> to use the system
trust store instead of the CLI's private one.

=head3 TLS with no certificates at all

It means B<encrypt and verify against the system trust store>, not an error.

C<tls> asks for a connection that is encrypted and whose far end is
authenticated. It does not ask to authenticate this client, which is what the
files on disk are for, and treating the absence of a client certificate as a
missing precondition would conflate the two. The deployment with no
certificate files is real, and is the one this role's documentation used to
recommend before there was any TLS here: a terminator -- nginx, stunnel,
Traefik -- in front of the daemon, holding a publicly trusted certificate.
There is nothing for a C<cert_path> to point at in that setup.

It is also the safe reading rather than the lax one: verification stays on
either way, so the mode reached by configuring nothing is the verifying mode.
A stock C<dockerd --tlsverify> uses a private CA that the system store does
not have, and such a connection fails with a verification error naming exactly
that -- which is the intended outcome, not a silent downgrade. Point
C<cert_path> at the directory holding its F<ca.pem> and it verifies.

=head3 Turning verification off

C<< tls_insecure => 1 >>, and the name is the whole of the warning. It sets
C<SSL_VERIFY_NONE> and switches the hostname check off, which leaves a
connection that is encrypted against a passive listener and against nothing
else: whoever answers chooses the certificate, so anyone able to redirect the
connection reads and rewrites everything on it -- registry credentials,
image contents, the commands containers are started with.

It exists for a self-signed daemon certificate whose CA is genuinely not to
hand. The better answer to that is nearly always F<ca.pem>: a self-signed
certificate is its own CA and can be used as the anchor directly.

=head3 The dependency

L<IO::Socket::SSL> is a B<recommended>, not a required, dependency, and it is
loaded at the moment the first TLS connection is opened. It brings in
L<Net::SSLeay>, which is XS compiled against libssl, and the C<unix://>
transport -- local Docker, rootless Podman, the default -- never needs any of
it; requiring it would make this client unbuildable on a machine with no
OpenSSL headers for the sake of a transport it is not using. Without it,
C<< tls => 1 >> croaks naming the module and how to install it, at the same
point every other connection failure is reported.

=head2 read_timeout

Seconds of silence after which a request gives up and croaks with an
L<API::Docker::Error::Timeout>. C<undef> -- the default, and what every
existing caller gets -- means no timeout at all and is the behaviour this
distribution has always had. C<0> means the same and is the way to say it
explicitly, so a client carrying a default can be opted out of per request.

    my $docker = API::Docker->new(read_timeout => 30);
    $docker->system->using(read_timeout => 0)->events;   # this one may wait

Per request it is an option of L</get>, L</post>, L</put>, L</delete_request>
and L</head>. A resource class carries it through
L<API::Docker::Role::Using/using>, which clones the class rather than taking
it per method -- up for a slow endpoint, down for a stream that should not
stall, off with C<0>.

See L</"Bounding a request that never ends"> for what it does and does not
cover, and L<API::Docker/"What a timeout covers"> for the same question
asked of both bounds at once.

=head2 connect_timeout

Seconds after which opening the connection gives up and croaks with an
L<API::Docker::Error::Timeout> whose C<< ->phase >> is C<'connect'>. C<undef>
-- the default, and what every existing caller gets -- means no bound and is
the behaviour this distribution has always had; C<0> means the same and is the
way to say it explicitly.

    my $docker = API::Docker->new(connect_timeout => 5);
    $docker->system->using(connect_timeout => 0)->version;  # may wait

Separate from L</read_timeout> rather than folded into it, because the two
bound different things and want different numbers: a connect is either
immediate or broken, while a read is waiting on work the daemon has to do.

Per request it is an option of L</get>, L</post>, L</put>, L</delete_request>
and L</head>. A resource class carries it through
L<API::Docker::Role::Using/using>. See L</"Bounding the connection itself">

lib/API/Docker/Role/HTTP.pm  view on Meta::CPAN


=item * C<ndjson> - Parse the body as newline-delimited JSON and always
return an ArrayRef of events, even for a stream carrying a single object.
Named for the format rather than C<stream>, which is already a query
parameter of C</events> and C</containers/{id}/stats>. An C<errorDetail>
event in such a stream croaks; see L</"Failure inside a 200 response">

=item * C<croak_on_error> - Default true, and only consulted with
C<< ndjson => 1 >>. Set it false for a stream whose objects are engine data
rather than the outcome of one operation -- C</events> is the only such
endpoint here

=item * C<raw> - Never decode the body; return the response bytes verbatim

=item * C<response> - HashRef the status line and the response headers are
written into; see L</"Reading the status line and the response headers">

=item * C<on_event>, C<on_frame>, C<on_chunk> - CodeRef called with each unit
of the response as it arrives, instead of the body being buffered and
returned. At most one of the three; see L</"Streaming a response as it
arrives">

=item * C<read_timeout> - Seconds of silence after which this request gives up
and croaks with an L<API::Docker::Error::Timeout>. Overrides the
L</read_timeout> attribute; C<0> means no timeout. See L</"Bounding a request
that never ends">

=item * C<connect_timeout> - Seconds after which opening the connection gives
up and croaks with an L<API::Docker::Error::Timeout> whose C<< ->phase >> is
C<'connect'>. Overrides the L</connect_timeout> attribute; C<0> means no
bound. See L</"Bounding the connection itself">

=item * C<headers> names are validated, not sanitised; see
L</"Header names are rejected, header values are stripped">

=back

=head2 Bounding a request that never ends

Nothing above stops a request waiting forever. C<Connection: close> asks the
daemon to hang up when it is done, and the readers wait for that -- so a
daemon that has nothing more to send and does not hang up leaves the client
blocked with no way out. That is not hypothetical: attaching to a container
that has B<already exited> answers, delivers the buffered frames and then
holds the connection open indefinitely on rootless Podman (karr k52), and
C</containers/{id}/stats> opened on a running container does not end when that
container exits on Docker -- it degrades into zero-filled readings and keeps
going (karr k59).

L</read_timeout> bounds that:

    # Give up after two seconds of silence rather than waiting forever.
    my $frames = $docker->containers->using(read_timeout => 2)->attach($id);

=head3 It is an idle timeout, not a deadline

The clock measures the time since the last byte arrived, not the time since
the request started. A stream that keeps producing runs as long as it likes;
one that stops producing is cut off. That distinction is the whole point --
both hangs above deliver data first and stall afterwards, so a bound on the
total time would have to be set longer than any legitimate stream, and a bound
on the time to the first byte would never fire at all.

=head3 There is no default, and no per-endpoint default either

Off unless asked for, everywhere. Whether a silence is a stall or normal is a
property of the workload rather than of the endpoint: C</build> with a large
context is legitimately quiet for as long as C</events> is, and a built-in
default on C<attach> would kill a perfectly healthy session at an idle shell
prompt. So no existing call changes behaviour, and picking the number is the
caller's -- who is the only one who knows what the request is for.

For the two endpoints above, if you want a figure to start from: a couple of
seconds is right for C<attach> or C<logs> used to collect what is already
there, and something above the daemon's own emit interval -- Docker sends a
stats reading about once a second -- for C<stats>.

=head3 What happens when it expires

The request croaks, on every path, with an L<API::Docker::Error::Timeout>. It
never returns a truncated response: a short body satisfies every return shape
this role promises and would be indistinguishable from a complete one. The
exception carries what did arrive -- C<< ->partial >> for a buffered request,
C<< ->summary >> for a streamed one -- so collecting what there is and then
stopping is an C<eval>:

    my $out = '';
    eval {
        $docker->containers->using(read_timeout => 2)->attach($id,
            on_frame => sub { $out .= $_[0]{data} });
    };
    die $@ if $@ && !(ref $@
        && $@->isa('API::Docker::Error::Timeout'));

That class's own documentation has the reasoning for why this is fatal even
where the caller already holds every unit.

=head3 What it does not cover

Only reading. Connecting is bounded separately by L</connect_timeout>, and
writing the request is not bounded at all -- which matters only for a large
C</build> context sent to a daemon that has stopped reading.

It is implemented with C<SO_RCVTIMEO> on the socket, which was measured to
behave the same over C<unix://>, plain C<tcp://> and TLS: the timeout fires,
the handle is not left unusable, and reading afterwards works. The C<struct
timeval> it is set with was measured on Linux; on Windows the millisecond
C<DWORD> Winsock documents is sent instead, which is reasoned rather than
measured. A platform that rejects either croaks rather than continuing without
the bound.

Over TLS it is not quite an idle timer on the plaintext. C<SO_RCVTIMEO> bounds
each blocking receive on the underlying socket, and one plaintext read can
consume several of those while a TLS record arrives in pieces -- so a record
dribbling in slowly enough resets the clock without a byte reaching the
caller. It still bounds the hang, which is what it is for.

=head2 Bounding the connection itself

L</connect_timeout> is the other half, and it is off by default for the same
reason: nothing here changes behaviour unless it is asked for.

    my $docker = API::Docker->new(connect_timeout => 5, read_timeout => 30);

What it does is not the same on all three transports, and the difference was
measured rather than assumed:

=over

=item * C<tcp://> -- a real bound. Against a host that drops SYNs, an unbounded
connect waits for the kernel's own timeout, which on Linux is over two
minutes; C<< connect_timeout => 2 >> gave up after 2.00s. This is the case the
option exists for.

=item * C<unix://> -- a bound, but it does not wait. A connect to a Unix socket
whose listen backlog is full blocks: measured against a listener with
C<< Listen => 1 >> and nobody accepting, still blocked after 8 seconds. With a
C<connect_timeout> set it fails at once instead, with C<EAGAIN> -- because
C<IO::Socket> performs a timed connect non-blocking, and an C<AF_UNIX> connect
has no in-progress state to wait on. So the hang is gone, at the price of not
tolerating even a momentary backlog. A socket path that does not exist is
C<ENOENT> either way and is not affected.

=item * TLS -- bounds the TCP connect only. The handshake that follows it runs
on the connected socket, before L</read_timeout>'s C<SO_RCVTIMEO> is applied,
and is not covered by either.

=back

An expiry croaks with an L<API::Docker::Error::Timeout> carrying
C<< ->phase >> C<'connect'>, C<< ->timeout >> the value that expired and an
empty C<< ->partial >> -- there is no response to have part of. Every other
connect failure croaks with the plain string it always did: a refused
connection, a missing socket path and a rejected certificate are diagnoses,
not timeouts, and rewriting them as one would name a cause the caller cannot
act on.

=head2 Streaming a response as it arrives

Without one of these options a request is read whole, then parsed. That is
right for a request/response endpoint and wrong for every endpoint whose point
is that it keeps going: C<< logs(follow => 1) >>, C</events> with no C<until>
and C</containers/{id}/stats> with no C<< stream => 0 >> never return, because
the daemon never closes and there is nothing else to wait for.

A callback is half the answer -- it decides what to do with each unit, and it
can stop. L</"Bounding a request that never ends"> is the other half, for the
stream that stops arriving without ever ending.

Pass a callback and the body is handed over piece by piece instead:

    my $summary = $client->get('/events',
      croak_on_error => 0,
      on_event       => sub {
        my ($event, $stop) = @_;
        print $event->{status}, "\n";
        $stop->() if $event->{status} eq 'destroy';
      },
    );

    $summary;   # { delivered => 7, stopped => 1 }

=head3 One unit per call, and three units to choose from

The engine's streaming endpoints do not share a natural unit, so there is an
option per unit and a request picks one:

=over

=item * C<on_event> - one decoded HashRef per newline-delimited JSON object.
For C</events> and the C</build>, C</images/create>, C</images/*/push>
progress streams

=item * C<on_frame> - one C<< { stream => ..., data => ... } >> HashRef per
demultiplexed frame of the Docker stream format. For
C<< /containers/{id}/logs >> and C<< /exec/{id}/start >>; normally reached
through L</stream_frames> rather than directly

lib/API/Docker/Role/HTTP.pm  view on Meta::CPAN

    $res{status};             # 204
    $res{reason};             # 'No Content'
    $res{headers}{'api-version'};   # header names are lowercased

The hash is overwritten on every call and filled B<before> the C<< >= 400 >>
croak, so a caller that wraps the request in C<eval> can still read the status
of a failed one. The return value is unaffected, so passing C<response> never
changes what a method hands back.

Two things need it. The engine answers a state change that did nothing with
B<304 Not Modified> -- starting a running container, stopping a stopped one --
which carries no body, exactly like the 204 of a change that did happen; see
L<API::Docker::API::Containers/start>. And C<< HEAD /containers/{id}/archive >>
carries its whole payload in the C<X-Docker-Container-Path-Stat> header, with
no body to return at all.

=head2 Failure on the status line

A status of 400 or above croaks with an L<API::Docker::Error::HTTP>. The
message is the engine's C<message> field, its C<errorDetail.message>, its flat
C<error> key or the raw body, in that order of preference, wrapped as
C<Docker API error (STATUS): REASON> -- the same text this croak has always
carried, and the object stringifies to it byte for byte, Carp's location
suffix included. Code that catches C<$@> as a string cannot tell the
difference and needs no change.

What the object adds is C<< $err->status >>. The message is engine-specific
prose: killing a stopped container answers 409 with C<can only kill running
containers ... container state improper> on rootless Podman 5.4.2, while
Docker's own example for that case reads C<Container E<lt>idE<gt> is not
running>. Anything that had to tell "no such container" from "wrong state"
apart was matching on that prose; the status code is the same distinction
without it. C<< ->reason >>, C<< ->body >> and C<< ->data >> carry the rest of
what the engine said.

This is B<not> a replacement for the C<response> option above, which stays the
only way to the status of a request that did not fail -- a 304, or a header
carrying the whole payload of a successful C<HEAD>.

=head2 Failure inside a 200 response

C</build>, C</images/create> (pull) and C</images/{name}/push> report a failed
operation as an C<errorDetail> object B<inside> a stream the daemon already
answered with HTTP 200. The status line is committed before the operation is
attempted, so the C<< >= 400 >> check above cannot see it, and a client that
trusts the status hands a broken build back as a success.

So an C<< ndjson => 1 >> request scans the decoded events and croaks with an
L<API::Docker::Error::Stream> the moment one carries C<errorDetail>. That
object stringifies to the reason plus Carp's usual location suffix, so
C<eval>-and-inspect-C<$@> code cannot tell it from the plain croak it
replaces; C<< $err->events >> carries the complete event list, so the progress
output that led up to the failure is not lost with the return value.

The trigger is the C<errorDetail> key alone. The flat C<error> key the engine
sends beside it holds the same text and is used only as a fallback message,
never as the trigger on its own.

C<< croak_on_error => 0 >> turns the scan off for a stream that is a feed
rather than an operation. The check is on by default, and opting out is per
endpoint, because the set of operation-shaped streaming endpoints is
open-ended while the feed-shaped ones are C</events> and nothing else: a new
endpoint added without a thought about this gets the loud behaviour, not the
silent one.

=head2 Failure in the middle of a response

The daemon can also stop saying anything in the middle of saying it. A status
line with no terminator, a header block with no blank line to close it, a body
shorter than its C<Content-Length>, a chunk shorter than its own header, a
chunk header cut in half, a chunked body with no terminating zero chunk: each
of those is a response that ended before it was finished, and each croaks with
an L<API::Docker::Error::Truncated>.

    my $tar = eval { $docker->images->get_tar('busybox') };
    die $@ if $@ && !(ref $@
        && $@->isa('API::Docker::Error::Truncated'));

It is a structural check, so it needs no option, applies to every request, and
cannot fire on a response that is complete. Which question it asks depends on
how the piece is framed: where the response announced a length, what arrived
is compared against it; where the framing is by terminator instead -- the head
and the chunk headers -- it asks whether the terminator came before the stream
ended, which is decidable without anything to compare. The exception carries
what did arrive: C<< ->partial >> for a buffered request, C<< ->summary >> for
a streamed one, and C<< ->phase >> for which piece of the framing ran out.

This B<is> a behaviour change and not a bug fix in passing. Until it existed
every shape above was returned rather than raised, and none of them was
distinguishable from a complete response: C<ndjson> gave a shorter ArrayRef,
C<raw> gave fewer bytes, the default gave whatever the truncated bytes
happened to parse as. A cut head was quieter still -- the response was read on
with whichever headers had arrived, and one cut before C<Content-Length> and
C<Transfer-Encoding> left neither, which is the close-delimited path below,
where an EOF is the legitimate end and nothing looks wrong. Code that was
silently receiving half a response now gets an exception where it used to get
a value.

The one thing here that is B<not> raised as an object: a connection that
closed without a single byte of a status line still croaks with the plain
C<No response from Docker daemon> string it always has. Nothing about it was
ever silent, and it is a message callers may be matching on.

=head3 Where an end of stream is still the end

A body delimited by nothing but the close. C<attach>,
C<< logs(follow => 1) >>, C</exec/{id}/start> -- the whole
C<application/vnd.docker.raw-stream> family -- carry neither a
C<Content-Length> nor chunked encoding, so the response announces no end and
there is nothing for a short one to be short of. That is how every one of them
finishes, and treating it as truncation would break all of them.

Their B<heads> are another matter and are checked like every other head. An
engine writes those two by hand rather than through its HTTP server, so it is
worth saying that they are well-formed: both answer with C<HTTP/1.1 200 OK>, a
single C<Content-Type> line and the blank line, measured on Docker 29.7.2 and
on rootless Podman 5.8.4. So does every other shape either of them produces --
200, 204, 304, C<HEAD>, chunked. Nothing legitimate ends a head without its
blank line.

The same goes for a stream a callback ended with C<< $stop->() >>: the rest of
the response is unread because the caller said so, and every check on the
streaming path is skipped once it has.

=head3 Against a timeout, and against a status

L<API::Docker::Error::Timeout> is the daemon going B<quiet> for longer than a
bound the caller asked for; this is the daemon B<closing> mid-response, and
needs no bound to be noticed. The two share a contract -- neither ever returns
a short body, and both hand over what arrived -- and are separate classes
because only one of them is about an option, and only one of them can fire on
a response that would have completed.

A response whose status is 400 or above raises this rather than an
L<API::Docker::Error::HTTP> when it is B<its> body that was cut short, which
is the rule the timeout already follows in the same place: the transport
cannot tell a caller what the engine said when it did not finish saying it.
C<< ->partial >> holds the part of the error body that did arrive.

=head2 Header names are rejected, header values are stripped

A CR or LF in a header B<value> is stripped and the value is flattened onto
its own line. A header B<name> that is not an RFC 9110 token is refused with
a croak instead.

The asymmetry is deliberate. A value can pick up a stray newline honestly --
C<MIME::Base64::encode_base64> wraps its output by default, and a token pasted
out of a file brings its line ending along -- and flattening it preserves what
the caller meant. A name is a literal the programmer wrote; there is no benign
way for one to contain CR, LF, a space or a colon, and quietly rewriting
C<< "X-Foo\r\nX-Bar" >> into C<X-FooX-Bar> would put a header on the wire
under a name nobody asked for. Validating against the token grammar also
catches the separators that would corrupt the request without injecting
anything.

=head2 A request path is rejected, not sanitised

The C<$path> given to L</get>, L</post>, L</put>, L</delete_request>, L</head>
and C<_request> is spliced straight into the request line as
C<< $method /v$version$path HTTP/1.1 >>, and it carries caller data: the
resource methods build it by interpolation -- C<< "/containers/$id/json" >>,
C<< "/images/$name/push" >> -- so a container name or an image reference the
user typed ends up in the request line unescaped. A byte the line's own
grammar reads therefore rewrites the request rather than naming a resource: a
CR or LF ends the line and opens a header of its own, a space starts the
HTTP-version field, and a C<?> or C<#> opens the query string or fragment.

So the path is checked against the RFC 3986 origin-form character set --
unreserved, the sub-delims, and C<:> C<@> C<%> C<< / >>, which is the set an
image reference lives in -- and a path outside it is refused with a croak
before anything reaches the wire, the same treatment and for the same reason a
header name gets. Sanitising is not on the table here: percent-encoding the
path at this layer cannot tell a separator from data, so it would either
mangle every C<< / >> and C<:> or leave the injection open. Query parameters
belong in C<params>, which is assembled separately and runs each element
through C<_uri_encode>.

=head2 post

    my $data = $client->post($path, $body, %opts);

Perform HTTP POST request. C<$body> is automatically JSON-encoded if provided.

Options: C<params>, C<headers>, C<ndjson>, C<croak_on_error>, C<raw>,
C<response> and the C<on_event>/C<on_frame>/C<on_chunk> callbacks as for
L</get>, plus C<raw_body> and C<content_type> for sending a non-JSON payload
such as a build context tarball.

=head2 put

    my $data = $client->put($path, $body, %opts);

Perform HTTP PUT request. C<$body> is automatically JSON-encoded if provided.

Options: C<params>, C<headers>, C<ndjson>, C<croak_on_error>, C<raw>,
C<response> and the C<on_event>/C<on_frame>/C<on_chunk> callbacks as for
L</get>, plus C<raw_body> and C<content_type> for sending a non-JSON payload
-- C<< containers->put_archive >> uses both to send a tar stream.

=head2 delete_request

    my $data = $client->delete_request($path, %opts);

Perform HTTP DELETE request.

Options: C<params> (hashref of query parameters).

=head2 head

    my %res;
    $client->head("/containers/$id/archive",
      params   => { path => '/etc/hostname' },
      response => \%res,
    );
    my $stat = decode_json(decode_base64($res{headers}{'x-docker-container-path-stat'}));

Perform HTTP HEAD request. Always returns C<undef>: a HEAD response has no
body by definition, so everything it says is in the status line and the
headers, and C<response> is the only way to reach them.

The body is not read even when the response announces one. A HEAD response
repeats the header fields the equivalent GET would send, C<Content-Length>
among them, and then sends nothing -- reading it would block on bytes that
never arrive. Measured against Podman 5.4.2 (API 1.41),
C<< HEAD /containers/{id}/archive >> in fact announces no length at all, only
C<X-Docker-Container-Path-Stat> -- but an engine that does announce one is not
waited on either.

Options: C<params>, C<headers> and C<response> as for L</get>.



( run in 0.994 second using v1.01-cache-2.11-cpan-5c0b1e786e0 )