Stop sharing state between processes in test farm tests (#7057)

* Set LOGDIR at top of script.

* Set sentinel at top of script.

* Don't use EC2 global to block on instance start.

* Remove global boto3 state.

* Pass in boulder_url.

* Create main function.

* Add link to reload docs.
This commit is contained in:
Brad Warren
2019-05-17 20:36:58 +02:00
committed by Adrien Ferrand
parent 11c3e7107c
commit 16834a0d78
+89 -82
View File
@@ -99,11 +99,9 @@ PROFILE = cl_args.aws_profile
# Globals # Globals
#------------------------------------------------------------------------------- #-------------------------------------------------------------------------------
BOULDER_AMI = 'ami-072a9534772bec854' # premade shared boulder AMI 18.04LTS us-east-1 BOULDER_AMI = 'ami-072a9534772bec854' # premade shared boulder AMI 18.04LTS us-east-1
LOGDIR = "" #points to logging / working directory LOGDIR = "letest-%d"%int(time.time()) #points to logging / working directory
# boto3/AWS api globals
AWS_SESSION = None
EC2 = None
SECURITY_GROUP_NAME = 'certbot-security-group' SECURITY_GROUP_NAME = 'certbot-security-group'
SENTINEL = None #queue kill signal
SUBNET_NAME = 'certbot-subnet' SUBNET_NAME = 'certbot-subnet'
class Status(object): class Status(object):
@@ -144,17 +142,19 @@ def make_security_group(vpc):
mysg.authorize_ingress(IpProtocol="udp", CidrIp="0.0.0.0/0", FromPort=60000, ToPort=61000) mysg.authorize_ingress(IpProtocol="udp", CidrIp="0.0.0.0/0", FromPort=60000, ToPort=61000)
return mysg return mysg
def make_instance(instance_name, def make_instance(ec2_client,
instance_name,
ami_id, ami_id,
keyname, keyname,
security_group_id, security_group_id,
subnet_id, subnet_id,
machine_type='t2.micro', machine_type='t2.micro',
userdata=""): #userdata contains bash or cloud-init script userdata=""): #userdata contains bash or cloud-init script
block_device_mappings = _get_block_device_mappings(ami_id) block_device_mappings = _get_block_device_mappings(ec2_client, ami_id)
tags = [{'Key': 'Name', 'Value': instance_name}] tags = [{'Key': 'Name', 'Value': instance_name}]
tag_spec = [{'ResourceType': 'instance', 'Tags': tags}] tag_spec = [{'ResourceType': 'instance', 'Tags': tags}]
return EC2.create_instances(BlockDeviceMappings=block_device_mappings, return ec2_client.create_instances(
BlockDeviceMappings=block_device_mappings,
ImageId=ami_id, ImageId=ami_id,
SecurityGroupIds=[security_group_id], SecurityGroupIds=[security_group_id],
SubnetId=subnet_id, SubnetId=subnet_id,
@@ -165,7 +165,7 @@ def make_instance(instance_name,
InstanceType=machine_type, InstanceType=machine_type,
TagSpecifications=tag_spec)[0] TagSpecifications=tag_spec)[0]
def _get_block_device_mappings(ami_id): def _get_block_device_mappings(ec2_client, ami_id):
"""Returns the list of block device mappings to ensure cleanup. """Returns the list of block device mappings to ensure cleanup.
This list sets connected EBS volumes to be deleted when the EC2 This list sets connected EBS volumes to be deleted when the EC2
@@ -178,7 +178,7 @@ def _get_block_device_mappings(ami_id):
# * https://docs.aws.amazon.com/AWSCloudFormation/latest/UserGuide/aws-properties-ec2-blockdev-template.html # * https://docs.aws.amazon.com/AWSCloudFormation/latest/UserGuide/aws-properties-ec2-blockdev-template.html
return [{'DeviceName': mapping['DeviceName'], return [{'DeviceName': mapping['DeviceName'],
'Ebs': {'DeleteOnTermination': True}} 'Ebs': {'DeleteOnTermination': True}}
for mapping in EC2.Image(ami_id).block_device_mappings for mapping in ec2_client.Image(ami_id).block_device_mappings
if not mapping.get('Ebs', {}).get('DeleteOnTermination', True)] if not mapping.get('Ebs', {}).get('DeleteOnTermination', True)]
@@ -217,20 +217,18 @@ def block_until_ssh_open(ipstring, wait_time=10, timeout=120):
def block_until_instance_ready(booting_instance, wait_time=5, extra_wait_time=20): def block_until_instance_ready(booting_instance, wait_time=5, extra_wait_time=20):
"Blocks booting_instance until AWS EC2 instance is ready to accept SSH connections" "Blocks booting_instance until AWS EC2 instance is ready to accept SSH connections"
# the reinstantiation from id is necessary to force boto3 state = booting_instance.state['Name']
# to correctly update the 'state' variable during init ip = booting_instance.public_ip_address
_id = booting_instance.id while state != 'running' or ip is None:
_instance = EC2.Instance(id=_id)
_state = _instance.state['Name']
_ip = _instance.public_ip_address
while _state != 'running' or _ip is None:
time.sleep(wait_time) time.sleep(wait_time)
_instance = EC2.Instance(id=_id) # The instance needs to be reloaded to update its local attributes. See
_state = _instance.state['Name'] # https://boto3.amazonaws.com/v1/documentation/api/latest/reference/services/ec2.html#EC2.Instance.reload.
_ip = _instance.public_ip_address booting_instance.reload()
block_until_ssh_open(_ip) state = booting_instance.state['Name']
ip = booting_instance.public_ip_address
block_until_ssh_open(ip)
time.sleep(extra_wait_time) time.sleep(extra_wait_time)
return _instance return booting_instance
# Fabric Routines # Fabric Routines
@@ -307,7 +305,7 @@ def grab_certbot_log():
cat ./certbot.log; else echo "[nolocallog]"; fi') cat ./certbot.log; else echo "[nolocallog]"; fi')
def create_client_instance(target, security_group_id, subnet_id): def create_client_instance(ec2_client, target, security_group_id, subnet_id):
"""Create a single client instance for running tests.""" """Create a single client instance for running tests."""
if target['virt'] == 'hvm': if target['virt'] == 'hvm':
machine_type = 't2.medium' if cl_args.fast else 't2.micro' machine_type = 't2.medium' if cl_args.fast else 't2.micro'
@@ -320,7 +318,8 @@ def create_client_instance(target, security_group_id, subnet_id):
userdata = '' userdata = ''
name = 'le-%s'%target['name'] name = 'le-%s'%target['name']
print(name, end=" ") print(name, end=" ")
return make_instance(name, return make_instance(ec2_client,
name,
target['ami'], target['ami'],
KEYNAME, KEYNAME,
machine_type=machine_type, machine_type=machine_type,
@@ -329,22 +328,28 @@ def create_client_instance(target, security_group_id, subnet_id):
userdata=userdata) userdata=userdata)
def test_client_process(inqueue, outqueue): def test_client_process(inqueue, outqueue, boulder_url):
cur_proc = mp.current_process() cur_proc = mp.current_process()
for inreq in iter(inqueue.get, SENTINEL): for inreq in iter(inqueue.get, SENTINEL):
ii, target = inreq ii, instance_id, target = inreq
# Each client process is given its own session due to the suggestion at
# https://boto3.amazonaws.com/v1/documentation/api/latest/guide/resources.html?highlight=multithreading#multithreading-multiprocessing.
aws_session = boto3.session.Session(profile_name=PROFILE)
ec2_client = aws_session.resource('ec2')
instance = ec2_client.Instance(id=instance_id)
#save all stdout to log file #save all stdout to log file
sys.stdout = open(LOGDIR+'/'+'%d_%s.log'%(ii,target['name']), 'w') sys.stdout = open(LOGDIR+'/'+'%d_%s.log'%(ii,target['name']), 'w')
print("[%s : client %d %s %s]" % (cur_proc.name, ii, target['ami'], target['name'])) print("[%s : client %d %s %s]" % (cur_proc.name, ii, target['ami'], target['name']))
instances[ii] = block_until_instance_ready(instances[ii]) instance = block_until_instance_ready(instance)
print("server %s at %s"%(instances[ii], instances[ii].public_ip_address)) print("server %s at %s"%(instance, instance.public_ip_address))
env.host_string = "%s@%s"%(target['user'], instances[ii].public_ip_address) env.host_string = "%s@%s"%(target['user'], instance.public_ip_address)
print(env.host_string) print(env.host_string)
try: try:
install_and_launch_certbot(instances[ii], boulder_url, target) install_and_launch_certbot(instance, boulder_url, target)
outqueue.put((ii, target, Status.PASS)) outqueue.put((ii, target, Status.PASS))
print("%s - %s SUCCESS"%(target['ami'], target['name'])) print("%s - %s SUCCESS"%(target['ami'], target['name']))
except: except:
@@ -382,30 +387,25 @@ def cleanup(cl_args, instances, targetlist):
"%s@%s"%(target['user'], instances[ii].public_ip_address)) "%s@%s"%(target['user'], instances[ii].public_ip_address))
def main():
#------------------------------------------------------------------------------- # Fabric library controlled through global env parameters
# SCRIPT BEGINS env.key_filename = KEYFILE
#------------------------------------------------------------------------------- env.shell = '/bin/bash -l -i -c'
env.connection_attempts = 5
# Fabric library controlled through global env parameters env.timeout = 10
env.key_filename = KEYFILE # replace default SystemExit thrown by fabric during trouble
env.shell = '/bin/bash -l -i -c' class FabricException(Exception):
env.connection_attempts = 5
env.timeout = 10
# replace default SystemExit thrown by fabric during trouble
class FabricException(Exception):
pass pass
env['abort_exception'] = FabricException env['abort_exception'] = FabricException
# Set up local copy of git repo # Set up local copy of git repo
#------------------------------------------------------------------------------- #-------------------------------------------------------------------------------
LOGDIR = "letest-%d"%int(time.time()) print("Making local dir for test repo and logs: %s"%LOGDIR)
print("Making local dir for test repo and logs: %s"%LOGDIR) local('mkdir %s'%LOGDIR)
local('mkdir %s'%LOGDIR)
# figure out what git object to test and locally create it in LOGDIR # figure out what git object to test and locally create it in LOGDIR
print("Making local git repo") print("Making local git repo")
try: try:
if cl_args.pull_request != '~': if cl_args.pull_request != '~':
print('Testing PR %s '%cl_args.pull_request, print('Testing PR %s '%cl_args.pull_request,
"MERGING into master" if cl_args.merge_master else "") "MERGING into master" if cl_args.merge_master else "")
@@ -416,62 +416,63 @@ try:
else: else:
print('Testing master of %s'%cl_args.repo) print('Testing master of %s'%cl_args.repo)
execute(local_git_clone, cl_args.repo) execute(local_git_clone, cl_args.repo)
except FabricException: except FabricException:
print("FAIL: trouble with git repo") print("FAIL: trouble with git repo")
traceback.print_exc() traceback.print_exc()
exit() exit()
# Set up EC2 instances # Set up EC2 instances
#------------------------------------------------------------------------------- #-------------------------------------------------------------------------------
configdata = yaml.load(open(cl_args.config_file, 'r')) configdata = yaml.load(open(cl_args.config_file, 'r'))
targetlist = configdata['targets'] targetlist = configdata['targets']
print('Testing against these images: [%d total]'%len(targetlist)) print('Testing against these images: [%d total]'%len(targetlist))
for target in targetlist: for target in targetlist:
print(target['ami'], target['name']) print(target['ami'], target['name'])
print("Connecting to EC2 using\n profile %s\n keyname %s\n keyfile %s"%(PROFILE, KEYNAME, KEYFILE)) print("Connecting to EC2 using\n profile %s\n keyname %s\n keyfile %s"%(PROFILE, KEYNAME, KEYFILE))
AWS_SESSION = boto3.session.Session(profile_name=PROFILE) aws_session = boto3.session.Session(profile_name=PROFILE)
EC2 = AWS_SESSION.resource('ec2') ec2_client = aws_session.resource('ec2')
print("Determining Subnet") print("Determining Subnet")
for subnet in EC2.subnets.all(): for subnet in ec2_client.subnets.all():
if should_use_subnet(subnet): if should_use_subnet(subnet):
subnet_id = subnet.id subnet_id = subnet.id
vpc_id = subnet.vpc.id vpc_id = subnet.vpc.id
break break
else: else:
print("No usable subnet exists!") print("No usable subnet exists!")
print("Please create a VPC with a subnet named {0}".format(SUBNET_NAME)) print("Please create a VPC with a subnet named {0}".format(SUBNET_NAME))
print("that maps public IPv4 addresses to instances launched in the subnet.") print("that maps public IPv4 addresses to instances launched in the subnet.")
sys.exit(1) sys.exit(1)
print("Making Security Group") print("Making Security Group")
vpc = EC2.Vpc(vpc_id) vpc = ec2_client.Vpc(vpc_id)
sg_exists = False sg_exists = False
for sg in vpc.security_groups.all(): for sg in vpc.security_groups.all():
if sg.group_name == SECURITY_GROUP_NAME: if sg.group_name == SECURITY_GROUP_NAME:
security_group_id = sg.id security_group_id = sg.id
sg_exists = True sg_exists = True
print(" %s already exists"%SECURITY_GROUP_NAME) print(" %s already exists"%SECURITY_GROUP_NAME)
if not sg_exists: if not sg_exists:
security_group_id = make_security_group(vpc).id security_group_id = make_security_group(vpc).id
time.sleep(30) time.sleep(30)
boulder_preexists = False boulder_preexists = False
boulder_servers = EC2.instances.filter(Filters=[ boulder_servers = ec2_client.instances.filter(Filters=[
{'Name': 'tag:Name', 'Values': ['le-boulderserver']}, {'Name': 'tag:Name', 'Values': ['le-boulderserver']},
{'Name': 'instance-state-name', 'Values': ['running']}]) {'Name': 'instance-state-name', 'Values': ['running']}])
boulder_server = next(iter(boulder_servers), None) boulder_server = next(iter(boulder_servers), None)
print("Requesting Instances...") print("Requesting Instances...")
if boulder_server: if boulder_server:
print("Found existing boulder server:", boulder_server) print("Found existing boulder server:", boulder_server)
boulder_preexists = True boulder_preexists = True
else: else:
print("Can't find a boulder server, starting one...") print("Can't find a boulder server, starting one...")
boulder_server = make_instance('le-boulderserver', boulder_server = make_instance(ec2_client,
'le-boulderserver',
BOULDER_AMI, BOULDER_AMI,
KEYNAME, KEYNAME,
machine_type='t2.micro', machine_type='t2.micro',
@@ -479,12 +480,15 @@ else:
security_group_id=security_group_id, security_group_id=security_group_id,
subnet_id=subnet_id) subnet_id=subnet_id)
instances = [] instances = []
try: try:
if not cl_args.boulderonly: if not cl_args.boulderonly:
print("Creating instances: ", end="") print("Creating instances: ", end="")
for target in targetlist: for target in targetlist:
instances.append(create_client_instance(target, security_group_id, subnet_id)) instances.append(
create_client_instance(ec2_client, target,
security_group_id, subnet_id)
)
print() print()
# Configure and launch boulder server # Configure and launch boulder server
@@ -520,7 +524,6 @@ try:
manager = Manager() manager = Manager()
outqueue = manager.Queue() outqueue = manager.Queue()
inqueue = manager.Queue() inqueue = manager.Queue()
SENTINEL = None #queue kill signal
# launch as many processes as clients to test # launch as many processes as clients to test
num_processes = len(targetlist) num_processes = len(targetlist)
@@ -529,14 +532,14 @@ try:
# initiate process execution # initiate process execution
for i in range(num_processes): for i in range(num_processes):
p = mp.Process(target=test_client_process, args=(inqueue, outqueue)) p = mp.Process(target=test_client_process, args=(inqueue, outqueue, boulder_url))
jobs.append(p) jobs.append(p)
p.daemon = True # kills subprocesses if parent is killed p.daemon = True # kills subprocesses if parent is killed
p.start() p.start()
# fill up work queue # fill up work queue
for ii, target in enumerate(targetlist): for ii, target in enumerate(targetlist):
inqueue.put((ii, target)) inqueue.put((ii, instances[ii].id, target))
# add SENTINELs to end client processes # add SENTINELs to end client processes
for i in range(num_processes): for i in range(num_processes):
@@ -577,8 +580,12 @@ try:
if failed: if failed:
sys.exit(1) sys.exit(1)
finally: finally:
cleanup(cl_args, instances, targetlist) cleanup(cl_args, instances, targetlist)
# kill any connections # kill any connections
fabric.network.disconnect_all() fabric.network.disconnect_all()
if __name__ == '__main__':
main()