Skip to content
Open
Show file tree
Hide file tree
Changes from 2 commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
85 changes: 48 additions & 37 deletions Dockerfile
Original file line number Diff line number Diff line change
@@ -1,54 +1,65 @@
FROM debian:jessie
FROM docker.elastic.co/elasticsearch/elasticsearch:5.3.0

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Why swap out for the official image from Elastic? We were using the debian image because it's lean and gives us a good starting point.

There are a couple problems with the official image for us here:

  • according to the Elastic docs this image includes X-Pack, which has only a trial license. I don't even want to start down the road of reconciling this repo's MPL license with that. 😀 So I don't think this is appropriate for us to distribute.
  • I can't find the Dockerfile for this image anywhere, or at least not easily, so even if we did use this image we should include a link on where to find the Dockerfile.

@sberryman sberryman Apr 7, 2017

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Yeah, I had a feeling this would be an issue. This is due to the "official" elasticsearch image being depreciated.

This image will receive no further updates after 2017-06-20 (June 20, 2017). Please adjust your usage accordingly. - Docker Hub

X-Pack is definitely part of the official image and has been turned off via env vars. I can remove it if that will help? Or obviously drop back to debian and build it.

Funny you mentioned not being able to find the Dockerfiles, I remember coming across a github issue mentioning the same thing. I can't seem to find that issue right now though.

Dockerfile
Base Dockerfile

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

X-Pack is definitely part of the official image and has been turned off via env vars. I can remove it if that will help?

Removing it doesn't remove it from the shipped binary blobs because of the layered file system. If we going to try to use their current build I'd say we should just fork from what they have in their Dockerfiles.

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Okay, would you like to fork based on their dockerfile? I would think copying the "official" repository from Elastic would be the best option for long term support.

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

We could use FROM docker.elastic.co/elasticsearch/elasticsearch-alpine-base:latest but after that I imagine we could just copy in the contents of https://github.com/elastic/elasticsearch-docker/blob/master/build/elasticsearch/Dockerfile into this directory and remove the bits we don't want. We'd need to put the LICENSE notice for their repo at the top of that file.

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Sounds good, I'll work on that tomorrow morning


RUN apt-get update && \
apt-get install -y \
openjdk-7-jre-headless \
curl \
jq \
&& rm -rf /var/lib/apt/lists/*
# need to drop back into root!
USER root

ENV JAVA_HOME /usr/lib/jvm/java-7-openjdk-amd64

# If we wanted the development version we could pull that instead but we want to
# run a production environment here.
RUN export ES_PKG=elasticsearch-2.2.0.deb && \
curl -LO https://download.elasticsearch.org/elasticsearch/release/org/elasticsearch/distribution/deb/elasticsearch/2.2.0/${ES_PKG} && \
dpkg -i ${ES_PKG} && \
rm ${ES_PKG} && \
rm /etc/elasticsearch/elasticsearch.yml
RUN apk update && \
apk add jq curl unzip tar && \
rm -rf /var/cache/apk/*

# Add Containerpilot and set its configuration
ENV CONTAINERPILOT_VER 2.1.0
ENV CONTAINERPILOT file:///etc/containerpilot.json
ENV CONSUL_VERSION=0.7.5 \
CONSUL_CLI_VER=0.3.1 \
CONTAINERPILOT_VER=2.7.2 \
CONTAINERPILOT=file:///etc/containerpilot.json

# Add consul agent
RUN export CONSUL_CHECKSUM=40ce7175535551882ecdff21fdd276cef6eaab96be8a8260e0599fadb6f1f5b8 \
&& curl --retry 7 --fail -vo /tmp/consul.zip "https://releases.hashicorp.com/consul/${CONSUL_VERSION}/consul_${CONSUL_VERSION}_linux_amd64.zip" \
&& echo "${CONSUL_CHECKSUM} /tmp/consul.zip" | sha256sum -c \
&& unzip /tmp/consul -d /usr/local/bin \
&& rm /tmp/consul.zip

# Consul client
RUN export CONSUL_CLIENT_CHECKSUM=037150d3d689a0babf4ba64c898b4497546e2fffeb16354e25cef19867e763f1 \
&& curl -Lso /tmp/consul-cli.tgz "https://github.com/CiscoCloud/consul-cli/releases/download/v${CONSUL_CLI_VER}/consul-cli_${CONSUL_CLI_VER}_linux_amd64.tar.gz" \
&& echo "${CONSUL_CLIENT_CHECKSUM} /tmp/consul-cli.tgz" | sha256sum -c \
&& tar zxf /tmp/consul-cli.tgz -C /usr/local/bin --strip-components 1 \
&& rm /tmp/consul-cli.tgz

RUN export CONTAINERPILOT_CHECKSUM=e7973bf036690b520b450c3a3e121fc7cd26f1a2 \
# Add ContainerPilot and set its configuration file path
RUN export CONTAINERPILOT_CHECKSUM=e886899467ced6d7c76027d58c7f7554c2fb2bcc \
&& curl -Lso /tmp/containerpilot.tar.gz \
"https://github.com/joyent/containerpilot/releases/download/${CONTAINERPILOT_VER}/containerpilot-${CONTAINERPILOT_VER}.tar.gz" \
"https://github.com/joyent/containerpilot/releases/download/${CONTAINERPILOT_VER}/containerpilot-${CONTAINERPILOT_VER}.tar.gz" \
&& echo "${CONTAINERPILOT_CHECKSUM} /tmp/containerpilot.tar.gz" | sha1sum -c \
&& tar zxf /tmp/containerpilot.tar.gz -C /usr/local/bin \
&& rm /tmp/containerpilot.tar.gz

# Add our configuration files and scripts
COPY /etc/containerpilot.json /etc/containerpilot.json
COPY /etc/elasticsearch.yml /usr/share/elasticsearch/config/elasticsearch.yml
COPY /bin/* /usr/local/bin/

# Should we remove unzip?
# RUN apk del unzip tar

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Removing anything in a different layer doesn't reduce total image size.

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Interesting, does it if you squash it?


# Create and take ownership over required directories
RUN mkdir -p /var/lib/elasticsearch/data && \
chown -R elasticsearch:elasticsearch /var/lib/elasticsearch/data && \
chown -R root:elasticsearch /etc/elasticsearch && \
chmod g+w /etc/elasticsearch
RUN mkdir -p /opt/consul/config \
&& mkdir -p /opt/consul/data \
&& chmod 770 /opt/consul/data \
&& chown -R elasticsearch:elasticsearch /opt/consul \
&& mkdir -p /etc/containerpilot \
&& chmod -R g+w /etc/containerpilot \
&& chmod +x /usr/local/bin/elastic-server.sh \
&& chown -R elasticsearch:elasticsearch /etc/containerpilot

# back to elastic USER
USER elasticsearch

# Add our configuration files and scripts
COPY /etc/containerpilot.json /etc
COPY /etc/elasticsearch.yml /etc/elasticsearch/elasticsearch.yml
COPY /bin/manage.sh /usr/local/bin

# Expose the data directory as a volume in case we want to mount these
# as a --volumes-from target; it's important that this VOLUME comes
# after the creation of the directory so that we preserve ownership.
VOLUME /var/lib/elasticsearch/data

# We don't need to expose these ports in order for other containers on Triton
# to reach this container in the default networking environment, but if we
# leave this here then we get the ports as well-known environment variables
# for purposes of linking.
EXPOSE 9200
EXPOSE 9300

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Were these removed just because the official image has them already? (Again, not sure because I can't find the Dockerfile.)

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

VOLUME ["/usr/share/elasticsearch/data"]

# Start with containerpilot then to our wrapper
CMD ["containerpilot", "/usr/local/bin/elastic-server.sh"]
6 changes: 6 additions & 0 deletions bin/elastic-server.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,6 @@
#!/bin/sh -xe
manage.sh onStart #|| exit $?

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Anything like this should be in a preStart, not wrapping the main app. That would also eliminate the need for this shell script.

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I copied this from the redis autopilot example.
https://github.com/autopilotpattern/redis/blob/master/bin/redis-server-sentinel.sh Where it obviously does a bit more.

Does preStart fire BEFORE you start the co-processes? I'm asking because we need the co-process consul agent to start before we start the onStart function.

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Did some research and unfortunately this has to be a separate wrapper as the coprocess doesn't start until immediately after preStart. We need the coprocess (consul agent) to start prior to onStart. onStart's job is there to check whether it has joined the consul cluster at which point we can verify if other services exist. If we did this in preStart, the consul agent won't be running and will assume every container should become the master.

Coprocesses are started immediately following the exit of the preStart hook, before polling for health or backend hooks, and before the main application starts. - Containerpilot Docs

I think this could be confusing for developers like me who are just getting started with containerpilot and consul agent running as a coprocess. Is there a reason you guys use CONSUL_AGENT=1 in most of the examples. What is the purpose of making that optional? Meaning, why wouldn't you always use consul agent on each container?

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Bah, no you're right. This will be fixed in CPv3 because we're forcing people to admit they need a real init system. But in the meantime we're stuck with it.

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

How would a "real" init system change this? I'm not sure what you mean by real init though.

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

In CPv3 we're supporting multiple processes as first-class, rather than having a "main" process and coprocesses. The end user will create the chain of dependencies explicitly, so in this case we'll be able to have everything wait on the consul agent. This is pretty much how every modern init system works, but was something we were avoiding in ContainerPilot originally because of community... obstinance(?) around the notion of there being more than one process in a container. Turns out that's exactly what you want a lot of the time. 😀

if [[ $? != 0 ]]; then
exit $?
fi
exec /usr/share/elasticsearch/bin/es-docker $*
151 changes: 107 additions & 44 deletions bin/manage.sh
Original file line number Diff line number Diff line change
@@ -1,67 +1,130 @@
#!/bin/bash

MASTER=null
CONSUL_HOST=${CONSUL}
CONSUL_AGENT=${CONSUL_AGENT:=false}

if [[ -z ${CONSUL} ]]; then
readonly lockPath=service/elasticsearch-master/locks/master

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

What is this for? It's unused anywhere else I think.

@sberryman sberryman Apr 7, 2017

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I'm going to pull it, it was left in there when I was testing using a lock


if [ $CONSUL_AGENT != false ]; then
CONSUL_HOST='localhost'
fi

if [[ -z $CONSUL_HOST ]]; then
echo "Missing CONSUL environment variable"
exit 1
fi

consulCommand() {
consul-cli --quiet --consul="${CONSUL_HOST}:8500" $*

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

There's already a consul binary in the container. We don't need this wrapper.

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

So you want me to pull this out and make the following changes?

Service addresses

Use curl and jq to get service addresses
https://github.com/sberryman/elasticsearch/blob/wip-v5/bin/manage.sh#L28

waitForLeader

waitForLeader using consul members -status alive | grep server

Replace: https://github.com/sberryman/elasticsearch/blob/wip-v5/bin/manage.sh#L48-L64
With: https://github.com/sberryman/autopilotpattern-nats/blob/master/bin/manage.sh#L28-L44

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Yeah, that should do it.

}

preStart() {

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Unless I'm missing something this preStart doesn't do anything, right? Almost everything you have in the onStart should really be in the preStart I think.

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I'm guessing you mean preStart not the highlighted consulCommand... But correct, I just left it in there as a placeholder, will remove it now.

# happy path is that there's a master available and we can cluster
configureMaster

# data-only nodes can only loop until there's a master available
if [ ${ES_NODE_MASTER} == false ]; then
while true
do
sleep 1.7
configureMaster
logDebug "preStart"
}

onStart() {
logDebug "onStart"

waitForLeader

getRegisteredServiceName
if [[ "${registeredServiceName}" == "elasticsearch-data" ]]; then

# wait for a healthy master
local i
for (( i = 0; i < ${MASTER_WAIT_TIMEOUT-60}; i++ )); do
getServiceAddresses "elasticsearch-master"
if [[ ${serviceAddresses} ]]; then
MASTER=$serviceAddresses
break
fi
sleep 1
done
exit 0
fi

# for a master+data node, we'll retry to see if there's another
# master in the cluster in the process of starting up. But we
# bail out if we exceed the retries and just bootstrap the cluster
if [ ${ES_NODE_DATA} == true ]; then
local n=0
until [ $n -ge 2 ]
do
sleep 1.7
configureMaster
n=$((n+1))
else

# wait for a healthy master

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

It is ugly how it just keeps hitting consul every second to see if there are any other services registered as 'elasticsearch-master'. Maybe creating a lock is a good idea?

You mean here? That's only during startup, so while I suppose it's ugly it's also rather simple and it goes away after the initial startup. How does creating a lock help us out here?

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

It doesn't now that I've had a chance to sleep on it.

local i
for (( i = 0; i < ${MASTER_WAIT_TIMEOUT-60}; i++ )); do
getServiceAddresses "elasticsearch-master"
if [[ ${serviceAddresses} ]]; then
MASTER=$serviceAddresses
break
fi
sleep 1
done

fi

# for a master-only node or master+data node that's exceeded the
# retry attempts, we'll assume this is the first master and bootstrap
# the cluster
MASTER=127.0.0.1
replace
# replace zen hosts
replaceZenHosts
}

# get the list of ES master nodes from Consul
configureMaster() {
MASTER=$(curl -Ls --fail http://${CONSUL}:8500/v1/catalog/service/elasticsearch-master | jq -r '.[0].ServiceAddress')
if [[ $MASTER != "null" ]] && [[ -n $MASTER ]]; then
replace
exit 0
fi
# if there's no master we fall thru and let the caller figure
# out what to do next
health() {
local privateIp=$(ip addr show eth0 | grep -o '[0-9]\{1,3\}\.[0-9]\{1,3\}\.[0-9]\{1,3\}\.[0-9]\{1,3\}')
/usr/bin/curl --fail -s -o /dev/null http://${privateIp}:9200
}

waitForLeader() {
logDebug "Waiting for consul leader"
local tries=0
while true
do
logDebug "Waiting for consul leader"
tries=$((tries + 1))
local leader=$(consulCommand --template="{{.}}" status leader)
if [[ -n "$leader" ]]; then
break
elif [[ $tries -eq 60 ]]; then
echo "No consul leader"
exit 1
fi
sleep 1
done
}

getServiceAddresses() {
local serviceInfo=$(consulCommand health service --passing "$1")
serviceAddresses=($(echo $serviceInfo | jq -r '.[].Service.Address'))
logDebug "serviceAddresses $1 ${serviceAddresses[*]}"
}

getRegisteredServiceName() {
registeredServiceName=$(jq -r '.services[0].name' /etc/containerpilot.json)
}

getNodeAddress() {
nodeAddress=$(ifconfig eth0 | awk '/inet addr/ {gsub("addr:", "", $2); print $2}')
}

# update discovery.zen.ping.unicast.hosts
replace() {
replaceZenHosts() {
REPLACEMENT=$(printf 's/^discovery\.zen\.ping\.unicast\.hosts.*$/discovery.zen.ping.unicast.hosts: ["%s"]/' ${MASTER})
sed -i "${REPLACEMENT}" /etc/elasticsearch/elasticsearch.yml
sed -i "${REPLACEMENT}" /usr/share/elasticsearch/config/elasticsearch.yml
}

health() {
local privateIp=$(ip addr show eth0 | grep -o '[0-9]\{1,3\}\.[0-9]\{1,3\}\.[0-9]\{1,3\}\.[0-9]\{1,3\}')
/usr/bin/curl --fail -s -o /dev/null http://${privateIp}:9200
logDebug() {
if [[ "${LOG_LEVEL}" == "DEBUG" ]]; then
echo "manage: $*"
fi
}

help() {
echo "Usage: ./manage.sh preStart => configure Consul agent"
echo " ./manage.sh onStart => first-run configuration"
echo " ./manage.sh health => health check Elastic"
echo " ./manage.sh preStop => prepare for stop"
}

# do whatever the arg is
$1
until
cmd=$1
if [[ -z "$cmd" ]]; then
help
fi
shift 1
$cmd "$@"
[ "$?" -ne 127 ]
do
help
exit
done
38 changes: 27 additions & 11 deletions etc/containerpilot.json
Original file line number Diff line number Diff line change
@@ -1,13 +1,29 @@
{
"consul": "{{ .CONSUL }}:8500",
"preStart": "/usr/local/bin/manage.sh preStart",
"services": [
{
"name": "{{ .ES_SERVICE_NAME }}",
"port": 9300,
"health": "/usr/local/bin/manage.sh health",
"poll": 10,
"ttl": 25
}
]
"consul": "{{ if .CONSUL_AGENT }}localhost{{ else }}{{ if .CONSUL }}{{ .CONSUL }}{{ else }}consul{{ end }}{{ end }}:8500",
"logging": {
"level": "{{ if .LOG_LEVEL }}{{ .LOG_LEVEL }}{{ else }}INFO{{ end }}",
"format": "text",
"output": "stdout"
},
"preStart": "/usr/local/bin/manage.sh preStart",
"services": [{
"name": "{{ .ES_SERVICE_NAME }}",
"port": 9300,
"health": "/usr/local/bin/manage.sh health",
"poll": 10,
"ttl": 25
}],
"coprocesses": [{{ if .CONSUL_AGENT }}
{
"command": ["/usr/local/bin/consul", "agent",
"-data-dir=/opt/consul/data",
"-config-dir=/opt/consul/config",
"-rejoin",
"-retry-join", "{{ if .CONSUL }}{{ .CONSUL }}{{ else }}consul{{ end }}",
"-retry-max", "10",
"-retry-interval", "10s"
],
"restarts": "unlimited"
}
{{ end }}]
}
Loading