2008-02-28 01:57:57 +00:00
|
|
|
/* -*- C -*-
|
|
|
|
*
|
|
|
|
* Copyright (c) 2004-2005 The Trustees of Indiana University and Indiana
|
|
|
|
* University Research and Technology
|
|
|
|
* Corporation. All rights reserved.
|
|
|
|
* Copyright (c) 2004-2005 The University of Tennessee and The University
|
|
|
|
* of Tennessee Research Foundation. All rights
|
|
|
|
* reserved.
|
|
|
|
* Copyright (c) 2004-2005 High Performance Computing Center Stuttgart,
|
|
|
|
* University of Stuttgart. All rights reserved.
|
|
|
|
* Copyright (c) 2004-2005 The Regents of the University of California.
|
|
|
|
* All rights reserved.
|
|
|
|
* $COPYRIGHT$
|
|
|
|
*
|
|
|
|
* Additional copyrights may follow
|
|
|
|
*
|
|
|
|
* $HEADER$
|
|
|
|
*/
|
|
|
|
/** @file:
|
|
|
|
*
|
|
|
|
*/
|
|
|
|
|
|
|
|
/*
|
|
|
|
* includes
|
|
|
|
*/
|
|
|
|
#include "orte_config.h"
|
|
|
|
|
2009-07-16 18:27:33 +00:00
|
|
|
#ifdef HAVE_STRING_H
|
2009-03-13 02:10:32 +00:00
|
|
|
#include <string.h>
|
|
|
|
#endif
|
2008-06-02 21:46:34 +00:00
|
|
|
#ifdef HAVE_SYS_TIME_H
|
|
|
|
#include <sys/time.h>
|
|
|
|
#endif
|
|
|
|
|
2008-02-28 01:57:57 +00:00
|
|
|
#include "opal/mca/mca.h"
|
|
|
|
#include "opal/mca/base/mca_base_param.h"
|
|
|
|
#include "opal/dss/dss.h"
|
2009-12-17 19:39:53 +00:00
|
|
|
|
2009-03-13 02:10:32 +00:00
|
|
|
#include "orte/constants.h"
|
|
|
|
#include "orte/types.h"
|
2008-02-28 01:57:57 +00:00
|
|
|
#include "orte/util/proc_info.h"
|
|
|
|
#include "orte/mca/errmgr/errmgr.h"
|
|
|
|
#include "orte/mca/rml/rml.h"
|
2009-02-14 02:26:12 +00:00
|
|
|
#include "orte/mca/rml/rml_types.h"
|
2008-02-28 01:57:57 +00:00
|
|
|
#include "orte/mca/routed/routed.h"
|
2009-07-14 14:34:11 +00:00
|
|
|
#include "orte/mca/ras/base/base.h"
|
2008-02-28 01:57:57 +00:00
|
|
|
#include "orte/util/name_fns.h"
|
|
|
|
#include "orte/runtime/orte_globals.h"
|
2008-02-28 19:58:32 +00:00
|
|
|
#include "orte/runtime/orte_wait.h"
|
2008-02-28 01:57:57 +00:00
|
|
|
|
|
|
|
#include "orte/mca/plm/plm_types.h"
|
|
|
|
#include "orte/mca/plm/plm.h"
|
|
|
|
#include "orte/mca/plm/base/plm_private.h"
|
2008-02-28 19:58:32 +00:00
|
|
|
#include "orte/mca/plm/base/base.h"
|
2008-02-28 01:57:57 +00:00
|
|
|
|
|
|
|
static bool recv_issued=false;
|
2009-07-19 18:07:04 +00:00
|
|
|
static opal_mutex_t lock;
|
|
|
|
static opal_list_t recvs;
|
|
|
|
static opal_event_t ready;
|
|
|
|
static int ready_fd[2];
|
|
|
|
static bool processing;
|
|
|
|
|
|
|
|
static void process_msg(int fd, short event, void *data);
|
2008-02-28 01:57:57 +00:00
|
|
|
|
|
|
|
int orte_plm_base_comm_start(void)
|
|
|
|
{
|
|
|
|
int rc;
|
|
|
|
|
|
|
|
if (recv_issued) {
|
|
|
|
return ORTE_SUCCESS;
|
|
|
|
}
|
|
|
|
|
2008-06-09 14:53:58 +00:00
|
|
|
OPAL_OUTPUT_VERBOSE((5, orte_plm_globals.output,
|
2008-02-28 01:57:57 +00:00
|
|
|
"%s plm:base:receive start comm",
|
2009-03-05 21:50:47 +00:00
|
|
|
ORTE_NAME_PRINT(ORTE_PROC_MY_NAME)));
|
2008-02-28 01:57:57 +00:00
|
|
|
|
2009-07-19 18:07:04 +00:00
|
|
|
processing = false;
|
|
|
|
OBJ_CONSTRUCT(&lock, opal_mutex_t);
|
|
|
|
OBJ_CONSTRUCT(&recvs, opal_list_t);
|
2009-07-22 07:39:52 +00:00
|
|
|
#ifndef __WINDOWS__
|
2009-07-19 18:07:04 +00:00
|
|
|
pipe(ready_fd);
|
2009-07-22 07:39:52 +00:00
|
|
|
#else
|
|
|
|
if (evutil_socketpair(AF_UNIX, SOCK_STREAM, 0, ready_fd) == -1) {
|
|
|
|
return ORTE_ERROR;
|
|
|
|
}
|
|
|
|
#endif
|
|
|
|
|
2009-07-19 18:07:04 +00:00
|
|
|
opal_event_set(&ready, ready_fd[0], OPAL_EV_READ, process_msg, NULL);
|
|
|
|
opal_event_add(&ready, 0);
|
|
|
|
|
2008-02-28 01:57:57 +00:00
|
|
|
if (ORTE_SUCCESS != (rc = orte_rml.recv_buffer_nb(ORTE_NAME_WILDCARD,
|
|
|
|
ORTE_RML_TAG_PLM,
|
2008-02-28 19:58:32 +00:00
|
|
|
ORTE_RML_NON_PERSISTENT,
|
2008-02-28 01:57:57 +00:00
|
|
|
orte_plm_base_recv,
|
|
|
|
NULL))) {
|
|
|
|
ORTE_ERROR_LOG(rc);
|
|
|
|
}
|
|
|
|
recv_issued = true;
|
|
|
|
|
|
|
|
return rc;
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
int orte_plm_base_comm_stop(void)
|
|
|
|
{
|
|
|
|
if (!recv_issued) {
|
|
|
|
return ORTE_SUCCESS;
|
|
|
|
}
|
|
|
|
|
2009-07-19 18:07:04 +00:00
|
|
|
OBJ_DESTRUCT(&recvs);
|
|
|
|
opal_event_del(&ready);
|
2009-07-22 07:39:52 +00:00
|
|
|
#ifndef __WINDOWS__
|
2009-07-19 18:07:04 +00:00
|
|
|
close(ready_fd[0]);
|
2009-07-22 07:39:52 +00:00
|
|
|
#else
|
|
|
|
closesocket(ready_fd[0]);
|
|
|
|
#endif
|
2009-07-19 18:07:04 +00:00
|
|
|
processing = false;
|
|
|
|
OBJ_DESTRUCT(&lock);
|
|
|
|
|
2008-06-09 14:53:58 +00:00
|
|
|
OPAL_OUTPUT_VERBOSE((5, orte_plm_globals.output,
|
2008-02-28 01:57:57 +00:00
|
|
|
"%s plm:base:receive stop comm",
|
2009-03-05 21:50:47 +00:00
|
|
|
ORTE_NAME_PRINT(ORTE_PROC_MY_NAME)));
|
2008-02-28 01:57:57 +00:00
|
|
|
|
2008-04-14 18:26:08 +00:00
|
|
|
orte_rml.recv_cancel(ORTE_NAME_WILDCARD, ORTE_RML_TAG_PLM);
|
2008-02-28 01:57:57 +00:00
|
|
|
recv_issued = false;
|
|
|
|
|
2008-04-14 18:26:08 +00:00
|
|
|
return ORTE_SUCCESS;
|
2008-02-28 01:57:57 +00:00
|
|
|
}
|
|
|
|
|
|
|
|
|
2008-02-28 19:58:32 +00:00
|
|
|
/* process incoming messages in order of receipt */
|
2009-12-17 19:39:53 +00:00
|
|
|
static void process_msg(int fd, short event, void *data)
|
2008-02-28 01:57:57 +00:00
|
|
|
{
|
2009-07-19 18:07:04 +00:00
|
|
|
orte_msg_packet_t *msgpkt;
|
2008-02-28 01:57:57 +00:00
|
|
|
orte_plm_cmd_flag_t command;
|
|
|
|
orte_std_cntr_t count;
|
|
|
|
orte_jobid_t job;
|
2008-03-26 21:18:16 +00:00
|
|
|
orte_job_t *jdata, *parent;
|
2008-03-25 14:57:34 +00:00
|
|
|
opal_buffer_t answer;
|
2008-02-28 01:57:57 +00:00
|
|
|
orte_vpid_t vpid;
|
2009-06-26 20:54:58 +00:00
|
|
|
orte_proc_t *proc;
|
2008-02-28 01:57:57 +00:00
|
|
|
orte_proc_state_t state;
|
|
|
|
orte_exit_code_t exit_code;
|
2009-07-19 18:07:04 +00:00
|
|
|
int rc=ORTE_SUCCESS, ret;
|
2008-06-02 21:46:34 +00:00
|
|
|
struct timeval beat;
|
2009-06-26 20:54:58 +00:00
|
|
|
orte_app_context_t *app, *child_app;
|
2009-07-19 18:07:04 +00:00
|
|
|
opal_list_item_t *item;
|
|
|
|
int dump[128];
|
|
|
|
|
|
|
|
OPAL_THREAD_LOCK(&lock);
|
2008-02-28 19:58:32 +00:00
|
|
|
|
2009-09-09 05:31:06 +00:00
|
|
|
OPAL_OUTPUT_VERBOSE((5, orte_plm_globals.output,
|
|
|
|
"%s plm:base:receive processing msg",
|
|
|
|
ORTE_NAME_PRINT(ORTE_PROC_MY_NAME)));
|
|
|
|
|
2009-07-19 18:07:04 +00:00
|
|
|
/* tag that we are processing the list */
|
|
|
|
processing = true;
|
2009-02-09 20:44:44 +00:00
|
|
|
|
2009-07-19 18:07:04 +00:00
|
|
|
/* clear the file descriptor to stop the event from refiring */
|
2009-07-22 07:39:52 +00:00
|
|
|
#ifndef __WINDOWS__
|
2009-07-19 18:07:04 +00:00
|
|
|
read(fd, &dump, sizeof(dump));
|
2009-07-22 07:39:52 +00:00
|
|
|
#else
|
|
|
|
recv(fd, (char *) &dump, sizeof(dump), 0);
|
|
|
|
#endif
|
2008-02-28 19:58:32 +00:00
|
|
|
|
2009-07-19 18:07:04 +00:00
|
|
|
while (NULL != (item = opal_list_remove_first(&recvs))) {
|
|
|
|
msgpkt = (orte_msg_packet_t*)item;
|
|
|
|
|
|
|
|
/* setup a default response */
|
|
|
|
OBJ_CONSTRUCT(&answer, opal_buffer_t);
|
|
|
|
job = ORTE_JOBID_INVALID;
|
|
|
|
|
|
|
|
count = 1;
|
|
|
|
if (ORTE_SUCCESS != (rc = opal_dss.unpack(msgpkt->buffer, &command, &count, ORTE_PLM_CMD))) {
|
|
|
|
ORTE_ERROR_LOG(rc);
|
|
|
|
goto CLEANUP;
|
|
|
|
}
|
|
|
|
|
|
|
|
switch (command) {
|
|
|
|
case ORTE_PLM_LAUNCH_JOB_CMD:
|
|
|
|
OPAL_OUTPUT_VERBOSE((5, orte_plm_globals.output,
|
|
|
|
"%s plm:base:receive job launch command",
|
|
|
|
ORTE_NAME_PRINT(ORTE_PROC_MY_NAME)));
|
2009-02-06 15:31:33 +00:00
|
|
|
|
2009-07-19 18:07:04 +00:00
|
|
|
/* unpack the job object */
|
|
|
|
count = 1;
|
|
|
|
if (ORTE_SUCCESS != (rc = opal_dss.unpack(msgpkt->buffer, &jdata, &count, ORTE_JOB))) {
|
2009-07-14 14:34:11 +00:00
|
|
|
ORTE_ERROR_LOG(rc);
|
|
|
|
goto ANSWER_LAUNCH;
|
|
|
|
}
|
|
|
|
|
2009-07-19 18:07:04 +00:00
|
|
|
/* if is a LOCAL slave cmd */
|
|
|
|
if (jdata->controls & ORTE_JOB_CONTROL_LOCAL_SLAVE) {
|
|
|
|
/* In this case, I cannot lookup job info. All I do is pass
|
|
|
|
* this along to the local launcher, IF it is available
|
2009-06-26 20:54:58 +00:00
|
|
|
*/
|
2009-07-19 18:07:04 +00:00
|
|
|
if (NULL == orte_plm.spawn) {
|
|
|
|
/* can't do this operation */
|
|
|
|
ORTE_ERROR_LOG(ORTE_ERR_NOT_SUPPORTED);
|
|
|
|
rc = ORTE_ERR_NOT_SUPPORTED;
|
|
|
|
goto ANSWER_LAUNCH;
|
|
|
|
}
|
|
|
|
if (ORTE_SUCCESS != (rc = orte_plm.spawn(jdata))) {
|
|
|
|
ORTE_ERROR_LOG(rc);
|
|
|
|
goto ANSWER_LAUNCH;
|
|
|
|
}
|
|
|
|
job = jdata->jobid;
|
|
|
|
} else { /* this is a GLOBAL launch cmd */
|
|
|
|
/* get the parent's job object */
|
|
|
|
if (NULL == (parent = orte_get_job_data_object(msgpkt->sender.jobid))) {
|
|
|
|
ORTE_ERROR_LOG(ORTE_ERR_NOT_FOUND);
|
|
|
|
goto ANSWER_LAUNCH;
|
|
|
|
}
|
|
|
|
|
|
|
|
/* if the prefix was set in the parent's job, we need to transfer
|
|
|
|
* that prefix to the child's app_context so any further launch of
|
|
|
|
* orteds can find the correct binary. There always has to be at
|
|
|
|
* least one app_context in both parent and child, so we don't
|
|
|
|
* need to check that here. However, be sure not to overwrite
|
|
|
|
* the prefix if the user already provided it!
|
|
|
|
*/
|
|
|
|
app = (orte_app_context_t*)opal_pointer_array_get_item(parent->apps, 0);
|
|
|
|
child_app = (orte_app_context_t*)opal_pointer_array_get_item(jdata->apps, 0);
|
|
|
|
if (NULL != app->prefix_dir &&
|
|
|
|
NULL == child_app->prefix_dir) {
|
|
|
|
child_app->prefix_dir = strdup(app->prefix_dir);
|
|
|
|
}
|
|
|
|
|
|
|
|
/* process any add-hostfile and add-host options that were provided */
|
|
|
|
if (ORTE_SUCCESS != (rc = orte_ras_base_add_hosts(jdata))) {
|
|
|
|
ORTE_ERROR_LOG(rc);
|
|
|
|
goto ANSWER_LAUNCH;
|
|
|
|
}
|
|
|
|
|
|
|
|
/* find the sender's node in the job map */
|
|
|
|
if (NULL != (proc = (orte_proc_t*)opal_pointer_array_get_item(parent->procs, msgpkt->sender.vpid))) {
|
|
|
|
/* set the bookmark so the child starts from that place - this means
|
|
|
|
* that the first child process could be co-located with the proc
|
|
|
|
* that called comm_spawn, assuming slots remain on that node. Otherwise,
|
|
|
|
* the procs will start on the next available node
|
|
|
|
*/
|
|
|
|
jdata->bookmark = proc->node;
|
|
|
|
}
|
|
|
|
|
|
|
|
/* launch it */
|
|
|
|
if (ORTE_SUCCESS != (rc = orte_plm.spawn(jdata))) {
|
|
|
|
ORTE_ERROR_LOG(rc);
|
|
|
|
goto ANSWER_LAUNCH;
|
|
|
|
}
|
|
|
|
job = jdata->jobid;
|
|
|
|
|
|
|
|
/* return the favor so that any repetitive comm_spawns track each other */
|
|
|
|
parent->bookmark = jdata->bookmark;
|
2009-06-26 20:54:58 +00:00
|
|
|
}
|
2009-02-06 15:31:33 +00:00
|
|
|
|
2009-07-19 18:07:04 +00:00
|
|
|
/* if the child is an ORTE job, wait for the procs to report they are alive */
|
|
|
|
if (!(jdata->controls & ORTE_JOB_CONTROL_NON_ORTE_JOB)) {
|
|
|
|
ORTE_PROGRESSED_WAIT(false, jdata->num_reported, jdata->num_procs);
|
2009-02-06 15:31:33 +00:00
|
|
|
}
|
2008-02-28 19:58:32 +00:00
|
|
|
|
2009-07-19 18:07:04 +00:00
|
|
|
ANSWER_LAUNCH:
|
2008-06-09 14:53:58 +00:00
|
|
|
OPAL_OUTPUT_VERBOSE((5, orte_plm_globals.output,
|
2009-07-19 18:07:04 +00:00
|
|
|
"%s plm:base:receive job %s launched",
|
2009-03-05 21:50:47 +00:00
|
|
|
ORTE_NAME_PRINT(ORTE_PROC_MY_NAME),
|
|
|
|
ORTE_JOBID_PRINT(job)));
|
2008-02-28 01:57:57 +00:00
|
|
|
|
2009-07-19 18:07:04 +00:00
|
|
|
/* pack the jobid to be returned */
|
|
|
|
if (ORTE_SUCCESS != (ret = opal_dss.pack(&answer, &job, 1, ORTE_JOBID))) {
|
|
|
|
ORTE_ERROR_LOG(ret);
|
|
|
|
}
|
|
|
|
|
|
|
|
/* send the response back to the sender */
|
|
|
|
if (0 > (ret = orte_rml.send_buffer(&msgpkt->sender, &answer, ORTE_RML_TAG_PLM_PROXY, 0))) {
|
|
|
|
ORTE_ERROR_LOG(ret);
|
2008-02-28 01:57:57 +00:00
|
|
|
}
|
2009-07-19 18:07:04 +00:00
|
|
|
break;
|
|
|
|
|
|
|
|
case ORTE_PLM_UPDATE_PROC_STATE:
|
|
|
|
OPAL_OUTPUT_VERBOSE((5, orte_plm_globals.output,
|
|
|
|
"%s plm:base:receive update proc state command",
|
|
|
|
ORTE_NAME_PRINT(ORTE_PROC_MY_NAME)));
|
2008-02-28 01:57:57 +00:00
|
|
|
count = 1;
|
2009-07-19 18:07:04 +00:00
|
|
|
jdata = NULL;
|
|
|
|
while (ORTE_SUCCESS == (rc = opal_dss.unpack(msgpkt->buffer, &job, &count, ORTE_JOBID))) {
|
2008-02-28 01:57:57 +00:00
|
|
|
|
2008-06-09 14:53:58 +00:00
|
|
|
OPAL_OUTPUT_VERBOSE((5, orte_plm_globals.output,
|
2009-07-19 18:07:04 +00:00
|
|
|
"%s plm:base:receive got update_proc_state for job %s",
|
2009-03-05 21:50:47 +00:00
|
|
|
ORTE_NAME_PRINT(ORTE_PROC_MY_NAME),
|
2009-07-19 18:07:04 +00:00
|
|
|
ORTE_JOBID_PRINT(job)));
|
2008-02-28 01:57:57 +00:00
|
|
|
|
2009-07-19 18:07:04 +00:00
|
|
|
/* lookup the job object */
|
|
|
|
if (NULL == (jdata = orte_get_job_data_object(job))) {
|
|
|
|
/* this job may already have been removed from the array, so just cleanly
|
|
|
|
* ignore this request
|
|
|
|
*/
|
|
|
|
goto CLEANUP;
|
|
|
|
}
|
|
|
|
count = 1;
|
|
|
|
while (ORTE_SUCCESS == (rc = opal_dss.unpack(msgpkt->buffer, &vpid, &count, ORTE_VPID))) {
|
|
|
|
if (ORTE_VPID_INVALID == vpid) {
|
|
|
|
/* flag indicates that this job is complete - move on */
|
|
|
|
break;
|
|
|
|
}
|
|
|
|
/* unpack the state */
|
|
|
|
count = 1;
|
|
|
|
if (ORTE_SUCCESS != (rc = opal_dss.unpack(msgpkt->buffer, &state, &count, ORTE_PROC_STATE))) {
|
|
|
|
ORTE_ERROR_LOG(rc);
|
|
|
|
goto CLEANUP;
|
|
|
|
}
|
|
|
|
/* unpack the exit code */
|
|
|
|
count = 1;
|
|
|
|
if (ORTE_SUCCESS != (rc = opal_dss.unpack(msgpkt->buffer, &exit_code, &count, ORTE_EXIT_CODE))) {
|
|
|
|
ORTE_ERROR_LOG(rc);
|
|
|
|
goto CLEANUP;
|
|
|
|
}
|
|
|
|
|
2009-06-26 20:54:58 +00:00
|
|
|
OPAL_OUTPUT_VERBOSE((5, orte_plm_globals.output,
|
2009-07-19 18:07:04 +00:00
|
|
|
"%s plm:base:receive got update_proc_state for vpid %lu state %x exit_code %d",
|
2009-06-26 20:54:58 +00:00
|
|
|
ORTE_NAME_PRINT(ORTE_PROC_MY_NAME),
|
2009-07-19 18:07:04 +00:00
|
|
|
(unsigned long)vpid, (unsigned int)state, (int)exit_code));
|
|
|
|
|
|
|
|
/* retrieve the proc object */
|
|
|
|
if (NULL == (proc = (orte_proc_t*)opal_pointer_array_get_item(jdata->procs, vpid))) {
|
|
|
|
/* this proc is no longer in table - skip it */
|
|
|
|
OPAL_OUTPUT_VERBOSE((5, orte_plm_globals.output,
|
|
|
|
"%s plm:base:receive proc %s is not in proc table",
|
|
|
|
ORTE_NAME_PRINT(ORTE_PROC_MY_NAME),
|
|
|
|
ORTE_VPID_PRINT(vpid)));
|
|
|
|
continue;
|
|
|
|
}
|
|
|
|
|
2009-10-01 13:44:34 +00:00
|
|
|
OPAL_OUTPUT_VERBOSE((5, orte_plm_globals.output,
|
|
|
|
"%s plm:base:receive updating state for proc %s current state %x new state %x",
|
|
|
|
ORTE_NAME_PRINT(ORTE_PROC_MY_NAME),
|
|
|
|
ORTE_NAME_PRINT(&proc->name),
|
|
|
|
(unsigned int)proc->state, (unsigned int)state));
|
|
|
|
|
2009-07-19 18:07:04 +00:00
|
|
|
/* update the termination counter IFF the state is changing to something
|
|
|
|
* indicating terminated
|
|
|
|
*/
|
|
|
|
if (ORTE_PROC_STATE_UNTERMINATED < state &&
|
|
|
|
ORTE_PROC_STATE_UNTERMINATED > proc->state) {
|
|
|
|
++jdata->num_terminated;
|
|
|
|
}
|
|
|
|
/* update the data */
|
|
|
|
proc->state = state;
|
|
|
|
proc->exit_code = exit_code;
|
|
|
|
|
|
|
|
/* update orte's exit status if it is non-zero */
|
|
|
|
ORTE_UPDATE_EXIT_STATUS(exit_code);
|
|
|
|
|
2008-02-28 01:57:57 +00:00
|
|
|
}
|
2009-07-19 18:07:04 +00:00
|
|
|
count = 1;
|
2008-02-28 01:57:57 +00:00
|
|
|
}
|
2009-07-19 18:07:04 +00:00
|
|
|
if (ORTE_ERR_UNPACK_READ_PAST_END_OF_BUFFER != rc) {
|
|
|
|
ORTE_ERROR_LOG(rc);
|
|
|
|
} else {
|
|
|
|
rc = ORTE_SUCCESS;
|
|
|
|
}
|
|
|
|
/* NOTE: jdata CAN BE NULL. This is caused by an orted
|
|
|
|
* being ordered to kill all its procs, but there are no
|
|
|
|
* procs left alive on that node. This can happen, for example,
|
|
|
|
* when a proc aborts somewhere, but the procs on this node
|
|
|
|
* have completed.
|
|
|
|
* So check job has to know how to handle a NULL pointer
|
|
|
|
*/
|
|
|
|
orte_plm_base_check_job_completed(jdata);
|
|
|
|
break;
|
|
|
|
|
|
|
|
case ORTE_PLM_HEARTBEAT_CMD:
|
2009-06-26 20:54:58 +00:00
|
|
|
OPAL_OUTPUT_VERBOSE((5, orte_plm_globals.output,
|
2009-07-19 18:07:04 +00:00
|
|
|
"%s plm:base:receive got heartbeat from %s",
|
2009-06-26 20:54:58 +00:00
|
|
|
ORTE_NAME_PRINT(ORTE_PROC_MY_NAME),
|
2009-07-19 18:07:04 +00:00
|
|
|
ORTE_NAME_PRINT(&msgpkt->sender)));
|
|
|
|
/* lookup the daemon object */
|
|
|
|
if (NULL == (jdata = orte_get_job_data_object(ORTE_PROC_MY_NAME->jobid))) {
|
|
|
|
/* this job can not possibly have been removed, so this is an error */
|
|
|
|
ORTE_ERROR_LOG(ORTE_ERR_NOT_FOUND);
|
|
|
|
goto CLEANUP;
|
|
|
|
}
|
|
|
|
gettimeofday(&beat, NULL);
|
|
|
|
if (NULL == (proc = (orte_proc_t*)opal_pointer_array_get_item(jdata->procs, msgpkt->sender.vpid))) {
|
|
|
|
/* this proc is no longer in table - skip it */
|
|
|
|
OPAL_OUTPUT_VERBOSE((5, orte_plm_globals.output,
|
|
|
|
"%s plm:base:receive daemon %s is not in proc table",
|
|
|
|
ORTE_NAME_PRINT(ORTE_PROC_MY_NAME),
|
|
|
|
ORTE_VPID_PRINT(msgpkt->sender.vpid)));
|
|
|
|
break;
|
|
|
|
}
|
|
|
|
proc->beat = beat.tv_sec;
|
2009-06-26 20:54:58 +00:00
|
|
|
break;
|
2009-07-19 18:07:04 +00:00
|
|
|
|
|
|
|
default:
|
|
|
|
ORTE_ERROR_LOG(ORTE_ERR_VALUE_OUT_OF_BOUNDS);
|
|
|
|
rc = ORTE_ERR_VALUE_OUT_OF_BOUNDS;
|
|
|
|
break;
|
|
|
|
}
|
|
|
|
|
|
|
|
CLEANUP:
|
|
|
|
/* release the message */
|
|
|
|
OBJ_RELEASE(msgpkt);
|
|
|
|
OBJ_DESTRUCT(&answer);
|
|
|
|
if (ORTE_SUCCESS != rc) {
|
|
|
|
goto DEPART;
|
|
|
|
}
|
2008-02-28 01:57:57 +00:00
|
|
|
}
|
2009-07-19 18:07:04 +00:00
|
|
|
|
|
|
|
/* reset the event */
|
|
|
|
processing = false;
|
|
|
|
opal_event_add(&ready, 0);
|
|
|
|
|
|
|
|
DEPART:
|
|
|
|
/* release the thread */
|
|
|
|
OPAL_THREAD_UNLOCK(&lock);
|
2008-02-28 01:57:57 +00:00
|
|
|
|
2009-02-09 20:44:44 +00:00
|
|
|
/* see if an error occurred - if so, wakeup the HNP so we can exit */
|
2009-05-04 11:07:40 +00:00
|
|
|
if (ORTE_PROC_IS_HNP && ORTE_SUCCESS != rc) {
|
2008-08-05 15:09:29 +00:00
|
|
|
orte_trigger_event(&orte_exit);
|
2008-02-28 01:57:57 +00:00
|
|
|
}
|
2009-07-19 18:07:04 +00:00
|
|
|
|
2008-02-28 19:58:32 +00:00
|
|
|
}
|
|
|
|
|
|
|
|
/*
|
|
|
|
* handle message from proxies
|
|
|
|
* NOTE: The incoming buffer "buffer" is OBJ_RELEASED by the calling program.
|
|
|
|
* DO NOT RELEASE THIS BUFFER IN THIS CODE
|
|
|
|
*/
|
|
|
|
|
|
|
|
void orte_plm_base_recv(int status, orte_process_name_t* sender,
|
|
|
|
opal_buffer_t* buffer, orte_rml_tag_t tag,
|
|
|
|
void* cbdata)
|
|
|
|
{
|
|
|
|
int rc;
|
2008-02-28 01:57:57 +00:00
|
|
|
|
2008-06-09 14:53:58 +00:00
|
|
|
OPAL_OUTPUT_VERBOSE((5, orte_plm_globals.output,
|
2008-02-28 19:58:32 +00:00
|
|
|
"%s plm:base:receive got message from %s",
|
2009-03-05 21:50:47 +00:00
|
|
|
ORTE_NAME_PRINT(ORTE_PROC_MY_NAME),
|
|
|
|
ORTE_NAME_PRINT(sender)));
|
2008-02-28 19:58:32 +00:00
|
|
|
|
|
|
|
/* don't process this right away - we need to get out of the recv before
|
|
|
|
* we process the message as it may ask us to do something that involves
|
|
|
|
* more messaging! Instead, setup an event so that the message gets processed
|
|
|
|
* as soon as we leave the recv.
|
|
|
|
*
|
|
|
|
* The macro makes a copy of the buffer, which we release above - the incoming
|
|
|
|
* buffer, however, is NOT released here, although its payload IS transferred
|
|
|
|
* to the message buffer for later processing
|
|
|
|
*/
|
2009-07-19 18:07:04 +00:00
|
|
|
ORTE_PROCESS_MESSAGE(&recvs, &lock, processing, ready_fd[1], true, sender, &buffer);
|
|
|
|
|
2008-02-28 19:58:32 +00:00
|
|
|
/* reissue the recv */
|
|
|
|
if (ORTE_SUCCESS != (rc = orte_rml.recv_buffer_nb(ORTE_NAME_WILDCARD,
|
|
|
|
ORTE_RML_TAG_PLM,
|
|
|
|
ORTE_RML_NON_PERSISTENT,
|
|
|
|
orte_plm_base_recv,
|
|
|
|
NULL))) {
|
|
|
|
ORTE_ERROR_LOG(rc);
|
|
|
|
}
|
2009-07-17 02:28:47 +00:00
|
|
|
|
2008-02-28 01:57:57 +00:00
|
|
|
return;
|
|
|
|
}
|
|
|
|
|
2009-07-19 18:07:04 +00:00
|
|
|
/* where HNP messages come */
|
|
|
|
void orte_plm_base_receive_process_msg(int fd, short event, void *data)
|
|
|
|
{
|
|
|
|
orte_message_event_t *mev = (orte_message_event_t*)data;
|
|
|
|
|
|
|
|
ORTE_PROCESS_MESSAGE(&recvs, &lock, processing, ready_fd[1], false, &mev->sender, &mev->buffer);
|
|
|
|
OBJ_RELEASE(mev);
|
|
|
|
}
|