open-nomad/command/agent/alloc_endpoint.go

// Copyright (c) HashiCorp, Inc.
// SPDX-License-Identifier: MPL-2.0

package agent

import (
	"context"
	"encoding/json"
	"fmt"
	"io"
	"net"
	"net/http"
	"slices"
	"strconv"
	"strings"

	"github.com/golang/snappy"
	"github.com/gorilla/websocket"
	"github.com/hashicorp/go-msgpack/codec"
	cstructs "github.com/hashicorp/nomad/client/structs"
	"github.com/hashicorp/nomad/nomad/structs"
	"github.com/hashicorp/nomad/plugins/drivers"
)

const (
	allocNotFoundErr    = "allocation not found"
	resourceNotFoundErr = "resource not found"
)

func (s *HTTPServer) AllocsRequest(resp http.ResponseWriter, req *http.Request) (interface{}, error) {
	if req.Method != http.MethodGet {
		return nil, CodedError(405, ErrInvalidMethod)
	}

	args := structs.AllocListRequest{}
	if s.parse(resp, req, &args.Region, &args.QueryOptions) {
		return nil, nil
	}

	// Parse resources and task_states field selection
	resources, err := parseBool(req, "resources")
	if err != nil {
		return nil, err
	}
	taskStates, err := parseBool(req, "task_states")
	if err != nil {
		return nil, err
	}

	if resources != nil || taskStates != nil {
		args.Fields = structs.NewAllocStubFields()
		if resources != nil {
			args.Fields.Resources = *resources
		}
		if taskStates != nil {
			args.Fields.TaskStates = *taskStates
		}
	}

	var out structs.AllocListResponse
	if err := s.agent.RPC("Alloc.List", &args, &out); err != nil {
		return nil, err
	}

	setMeta(resp, &out.QueryMeta)
	if out.Allocations == nil {
		out.Allocations = make([]*structs.AllocListStub, 0)
	}
	for _, alloc := range out.Allocations {
		alloc.SetEventDisplayMessages()
	}
	return out.Allocations, nil
}

func (s *HTTPServer) AllocSpecificRequest(resp http.ResponseWriter, req *http.Request) (interface{}, error) {
	reqSuffix := strings.TrimPrefix(req.URL.Path, "/v1/allocation/")

	// tokenize the suffix of the path to get the alloc id and find the action
	// invoked on the alloc id
	tokens := strings.Split(reqSuffix, "/")
	if len(tokens) > 2 || len(tokens) < 1 {
		return nil, CodedError(404, resourceNotFoundErr)
	}
	allocID := tokens[0]

	if len(tokens) == 1 {
		return s.allocGet(allocID, resp, req)
	}

	switch tokens[1] {
	case "checks":
		return s.allocChecks(allocID, resp, req)
	case "stop":
		return s.allocStop(allocID, resp, req)
	case "services":
		return s.allocServiceRegistrations(resp, req, allocID)
	}

	return nil, CodedError(404, resourceNotFoundErr)
}

func (s *HTTPServer) allocGet(allocID string, resp http.ResponseWriter, req *http.Request) (interface{}, error) {
	if req.Method != http.MethodGet {
		return nil, CodedError(405, ErrInvalidMethod)
	}

	args := structs.AllocSpecificRequest{
		AllocID: allocID,
	}
	if s.parse(resp, req, &args.Region, &args.QueryOptions) {
		return nil, nil
	}

	var out structs.SingleAllocResponse
	if err := s.agent.RPC("Alloc.GetAlloc", &args, &out); err != nil {
		return nil, err
	}

	setMeta(resp, &out.QueryMeta)
	if out.Alloc == nil {
		return nil, CodedError(404, "alloc not found")
	}

	// Decode the payload if there is any

	alloc := out.Alloc
	if alloc.Job != nil && len(alloc.Job.Payload) != 0 {
		decoded, err := snappy.Decode(nil, alloc.Job.Payload)
		if err != nil {
			return nil, err
		}
		alloc = alloc.Copy()
		alloc.Job.Payload = decoded
	}
	alloc.SetEventDisplayMessages()

	// Handle 0.12 ports upgrade path
	alloc = alloc.Copy()
	alloc.AllocatedResources.Canonicalize()

	return alloc, nil
}

func (s *HTTPServer) allocStop(allocID string, resp http.ResponseWriter, req *http.Request) (interface{}, error) {
	if !(req.Method == "POST" || req.Method == "PUT") {
		return nil, CodedError(405, ErrInvalidMethod)
	}

	noShutdownDelay := false
	if noShutdownDelayQS := req.URL.Query().Get("no_shutdown_delay"); noShutdownDelayQS != "" {
		var err error
		noShutdownDelay, err = strconv.ParseBool(noShutdownDelayQS)
		if err != nil {
			return nil, fmt.Errorf("no_shutdown_delay value is not a boolean: %v", err)
		}
	}

	sr := &structs.AllocStopRequest{
		AllocID:         allocID,
		NoShutdownDelay: noShutdownDelay,
	}
	s.parseWriteRequest(req, &sr.WriteRequest)

	var out structs.AllocStopResponse
	rpcErr := s.agent.RPC("Alloc.Stop", &sr, &out)

	if rpcErr != nil {
		if structs.IsErrUnknownAllocation(rpcErr) {
			rpcErr = CodedError(404, allocNotFoundErr)
		}
		return nil, rpcErr
	}

	setIndex(resp, out.Index)
	return &out, nil
}

// allocServiceRegistrations returns a list of all service registrations
// assigned to the job identifier. It is callable via the
// /v1/allocation/:alloc_id/services HTTP API and uses the
// structs.AllocServiceRegistrationsRPCMethod RPC method.
func (s *HTTPServer) allocServiceRegistrations(
	resp http.ResponseWriter, req *http.Request, allocID string) (interface{}, error) {

	// The endpoint only supports GET requests.
	if req.Method != http.MethodGet {
		return nil, CodedError(http.StatusMethodNotAllowed, ErrInvalidMethod)
	}

	// Set up the request args and parse this to ensure the query options are
	// set.
	args := structs.AllocServiceRegistrationsRequest{AllocID: allocID}
	if s.parse(resp, req, &args.Region, &args.QueryOptions) {
		return nil, nil
	}

	// Perform the RPC request.
	var reply structs.AllocServiceRegistrationsResponse
	if err := s.agent.RPC(structs.AllocServiceRegistrationsRPCMethod, &args, &reply); err != nil {
		return nil, err
	}

	setMeta(resp, &reply.QueryMeta)

	if reply.Services == nil {
		return nil, CodedError(http.StatusNotFound, allocNotFoundErr)
	}
	return reply.Services, nil
}

func (s *HTTPServer) ClientAllocRequest(resp http.ResponseWriter, req *http.Request) (interface{}, error) {
	reqSuffix := strings.TrimPrefix(req.URL.Path, "/v1/client/allocation/")

	// tokenize the suffix of the path to get the alloc id and find the action
	// invoked on the alloc id
	tokens := strings.Split(reqSuffix, "/")
	if len(tokens) != 2 {
		return nil, CodedError(404, resourceNotFoundErr)
	}
	allocID := tokens[0]
	switch tokens[1] {
	case "checks":
		return s.allocChecks(allocID, resp, req)
	case "stats":
		return s.allocStats(allocID, resp, req)
	case "exec":
		return s.allocExec(allocID, resp, req)
	case "snapshot":
		if s.agent.Client() == nil {
			return nil, clientNotRunning
		}
		return s.allocSnapshot(allocID, resp, req)
	case "restart":
		return s.allocRestart(allocID, resp, req)
	case "gc":
		return s.allocGC(allocID, resp, req)
	case "signal":
		return s.allocSignal(allocID, resp, req)
	}

	return nil, CodedError(404, resourceNotFoundErr)
}

func (s *HTTPServer) ClientGCRequest(resp http.ResponseWriter, req *http.Request) (interface{}, error) {

	// Build the request and get the requested Node ID
	args := structs.NodeSpecificRequest{}
	s.parse(resp, req, &args.QueryOptions.Region, &args.QueryOptions)
	parseNode(req, &args.NodeID)

	// Determine the handler to use
	useLocalClient, useClientRPC, useServerRPC := s.rpcHandlerForNode(args.NodeID)

	// Make the RPC
	var reply structs.GenericResponse
	var rpcErr error
	if useLocalClient {
		rpcErr = s.agent.Client().ClientRPC("Allocations.GarbageCollectAll", &args, &reply)
	} else if useClientRPC {
		rpcErr = s.agent.Client().RPC("ClientAllocations.GarbageCollectAll", &args, &reply)
	} else if useServerRPC {
		rpcErr = s.agent.Server().RPC("ClientAllocations.GarbageCollectAll", &args, &reply)
	} else {
		rpcErr = CodedError(400, "No local Node and node_id not provided")
	}

	if rpcErr != nil {
		if structs.IsErrNoNodeConn(rpcErr) {
			rpcErr = CodedError(404, rpcErr.Error())
		}
	}

	return nil, rpcErr
}

func (s *HTTPServer) allocRestart(allocID string, resp http.ResponseWriter, req *http.Request) (interface{}, error) {
	// Build the request and parse the ACL token
	args := structs.AllocRestartRequest{
		AllocID:  allocID,
		TaskName: "",
	}
	s.parse(resp, req, &args.QueryOptions.Region, &args.QueryOptions)

	// Explicitly parse the body separately to disallow overriding AllocID in req Body.
	var reqBody struct {
		TaskName string
		AllTasks bool
	}
	err := json.NewDecoder(req.Body).Decode(&reqBody)
	if err != nil && err != io.EOF {
		return nil, err
	}
	if reqBody.TaskName != "" {
		args.TaskName = reqBody.TaskName
	}
	if reqBody.AllTasks {
		args.AllTasks = reqBody.AllTasks
	}

	// Determine the handler to use
	useLocalClient, useClientRPC, useServerRPC := s.rpcHandlerForAlloc(allocID)

	// Make the RPC
	var reply structs.GenericResponse
	var rpcErr error
	if useLocalClient {
		rpcErr = s.agent.Client().ClientRPC("Allocations.Restart", &args, &reply)
	} else if useClientRPC {
		rpcErr = s.agent.Client().RPC("ClientAllocations.Restart", &args, &reply)
	} else if useServerRPC {
		rpcErr = s.agent.Server().RPC("ClientAllocations.Restart", &args, &reply)
	} else {
		rpcErr = CodedError(400, "No local Node and node_id not provided")
	}

	if rpcErr != nil {
		if structs.IsErrNoNodeConn(rpcErr) || structs.IsErrUnknownAllocation(rpcErr) {
			rpcErr = CodedError(404, rpcErr.Error())
		}
	}

	return reply, rpcErr
}

func (s *HTTPServer) allocGC(allocID string, resp http.ResponseWriter, req *http.Request) (interface{}, error) {
	// Build the request and parse the ACL token
	args := structs.AllocSpecificRequest{
		AllocID: allocID,
	}
	s.parse(resp, req, &args.QueryOptions.Region, &args.QueryOptions)

	// Determine the handler to use
	useLocalClient, useClientRPC, useServerRPC := s.rpcHandlerForAlloc(allocID)

	// Make the RPC
	var reply structs.GenericResponse
	var rpcErr error
	if useLocalClient {
		rpcErr = s.agent.Client().ClientRPC("Allocations.GarbageCollect", &args, &reply)
	} else if useClientRPC {
		rpcErr = s.agent.Client().RPC("ClientAllocations.GarbageCollect", &args, &reply)
	} else if useServerRPC {
		rpcErr = s.agent.Server().RPC("ClientAllocations.GarbageCollect", &args, &reply)
	} else {
		rpcErr = CodedError(400, "No local Node and node_id not provided")
	}

	if rpcErr != nil {
		if structs.IsErrNoNodeConn(rpcErr) || structs.IsErrUnknownAllocation(rpcErr) {
			rpcErr = CodedError(404, rpcErr.Error())
		}
	}

	return nil, rpcErr
}

func (s *HTTPServer) allocSignal(allocID string, resp http.ResponseWriter, req *http.Request) (interface{}, error) {
	if !(req.Method == "POST" || req.Method == "PUT") {
		return nil, CodedError(405, ErrInvalidMethod)
	}

	// Build the request and parse the ACL token
	args := structs.AllocSignalRequest{}
	err := decodeBody(req, &args)
	if err != nil {
		return nil, CodedError(400, fmt.Sprintf("Failed to decode body: %v", err))
	}
	s.parse(resp, req, &args.QueryOptions.Region, &args.QueryOptions)
	args.AllocID = allocID

	// Determine the handler to use
	useLocalClient, useClientRPC, useServerRPC := s.rpcHandlerForAlloc(allocID)

	// Make the RPC
	var reply structs.GenericResponse
	var rpcErr error
	if useLocalClient {
		rpcErr = s.agent.Client().ClientRPC("Allocations.Signal", &args, &reply)
	} else if useClientRPC {
		rpcErr = s.agent.Client().RPC("ClientAllocations.Signal", &args, &reply)
	} else if useServerRPC {
		rpcErr = s.agent.Server().RPC("ClientAllocations.Signal", &args, &reply)
	} else {
		rpcErr = CodedError(400, "No local Node and node_id not provided")
	}

	if rpcErr != nil {
		if structs.IsErrNoNodeConn(rpcErr) || structs.IsErrUnknownAllocation(rpcErr) {
			rpcErr = CodedError(404, rpcErr.Error())
		}
	}

	return reply, rpcErr
}

func (s *HTTPServer) allocSnapshot(allocID string, resp http.ResponseWriter, req *http.Request) (interface{}, error) {
	var secret string
	s.parseToken(req, &secret)
	if !s.agent.Client().ValidateMigrateToken(allocID, secret) {
		return nil, structs.ErrPermissionDenied
	}

	allocFS, err := s.agent.Client().GetAllocFS(allocID)
	if err != nil {
		return nil, fmt.Errorf(allocNotFoundErr)
	}
	if err := allocFS.Snapshot(resp); err != nil {
		return nil, fmt.Errorf("error making snapshot: %v", err)
	}
	return nil, nil
}

func (s *HTTPServer) allocStats(allocID string, resp http.ResponseWriter, req *http.Request) (interface{}, error) {

	// Build the request and parse the ACL token
	task := req.URL.Query().Get("task")
	args := cstructs.AllocStatsRequest{
		AllocID: allocID,
		Task:    task,
	}
	s.parse(resp, req, &args.QueryOptions.Region, &args.QueryOptions)

	// Determine the handler to use
	useLocalClient, useClientRPC, useServerRPC := s.rpcHandlerForAlloc(allocID)

	// Make the RPC
	var reply cstructs.AllocStatsResponse
	var rpcErr error
	if useLocalClient {
		rpcErr = s.agent.Client().ClientRPC("Allocations.Stats", &args, &reply)
	} else if useClientRPC {
		rpcErr = s.agent.Client().RPC("ClientAllocations.Stats", &args, &reply)
	} else if useServerRPC {
		rpcErr = s.agent.Server().RPC("ClientAllocations.Stats", &args, &reply)
	} else {
		rpcErr = CodedError(400, "No local Node and node_id not provided")
	}

	if rpcErr != nil {
		if structs.IsErrNoNodeConn(rpcErr) || structs.IsErrUnknownAllocation(rpcErr) {
			rpcErr = CodedError(404, rpcErr.Error())
		}
	}

	return reply.Stats, rpcErr
}

func (s *HTTPServer) allocChecks(allocID string, resp http.ResponseWriter, req *http.Request) (any, error) {
	// Build the request and parse the ACL token
	args := cstructs.AllocChecksRequest{
		AllocID: allocID,
	}
	s.parse(resp, req, &args.QueryOptions.Region, &args.QueryOptions)

	// Determine the handler to use
	useLocalClient, useClientRPC, useServerRPC := s.rpcHandlerForAlloc(allocID)

	// Make the RPC
	var reply cstructs.AllocChecksResponse
	var rpcErr error
	switch {
	case useLocalClient:
		rpcErr = s.agent.Client().ClientRPC("Allocations.Checks", &args, &reply)
	case useClientRPC:
		rpcErr = s.agent.Client().RPC("ClientAllocations.Checks", &args, &reply)
	case useServerRPC:
		rpcErr = s.agent.Server().RPC("ClientAllocations.Checks", &args, &reply)
	default:
		rpcErr = CodedError(400, "No local Node and node_id not provided")
	}

	if rpcErr != nil {
		if structs.IsErrNoNodeConn(rpcErr) || structs.IsErrUnknownAllocation(rpcErr) {
			rpcErr = CodedError(404, rpcErr.Error())
		}
	}

	return reply.Results, rpcErr
}

func (s *HTTPServer) allocExec(allocID string, resp http.ResponseWriter, req *http.Request) (interface{}, error) {
	// Build the request and parse the ACL token
	task := req.URL.Query().Get("task")
	cmdJsonStr := req.URL.Query().Get("command")
	var command []string
	err := json.Unmarshal([]byte(cmdJsonStr), &command)
	if err != nil {
		// this shouldn't happen, []string is always be serializable to json
		return nil, fmt.Errorf("failed to marshal command into json: %v", err)
	}

	ttyB := false
	if tty := req.URL.Query().Get("tty"); tty != "" {
		ttyB, err = strconv.ParseBool(tty)
		if err != nil {
			return nil, fmt.Errorf("tty value is not a boolean: %v", err)
		}
	}

	args := cstructs.AllocExecRequest{
		AllocID: allocID,
		Task:    task,
		Cmd:     command,
		Tty:     ttyB,
	}
	s.parse(resp, req, &args.QueryOptions.Region, &args.QueryOptions)

	conn, err := s.wsUpgrader.Upgrade(resp, req, nil)
	if err != nil {
		return nil, fmt.Errorf("failed to upgrade connection: %v", err)
	}

	if err := readWsHandshake(conn.ReadJSON, req, &args.QueryOptions); err != nil {
		conn.WriteMessage(websocket.CloseMessage,
			websocket.FormatCloseMessage(toWsCode(400), err.Error()))
		return nil, err
	}

	return s.execStreamImpl(conn, &args)
}

// readWsHandshake reads the websocket handshake message and sets
// query authentication token, if request requires a handshake
func readWsHandshake(readFn func(interface{}) error, req *http.Request, q *structs.QueryOptions) error {

	// Avoid handshake if request doesn't require one
	if hv := req.URL.Query().Get("ws_handshake"); hv == "" {
		return nil
	} else if h, err := strconv.ParseBool(hv); err != nil {
		return fmt.Errorf("ws_handshake value is not a boolean: %v", err)
	} else if !h {
		return nil
	}

	var h wsHandshakeMessage
	err := readFn(&h)
	if err != nil {
		return err
	}

	supportedWSHandshakeVersion := 1
	if h.Version != supportedWSHandshakeVersion {
		return fmt.Errorf("unexpected handshake value: %v", h.Version)
	}

	q.AuthToken = h.AuthToken
	return nil
}

type wsHandshakeMessage struct {
	Version   int    `json:"version"`
	AuthToken string `json:"auth_token"`
}

func (s *HTTPServer) execStreamImpl(ws *websocket.Conn, args *cstructs.AllocExecRequest) (interface{}, error) {
	allocID := args.AllocID
	method := "Allocations.Exec"

	// Get the correct handler
	localClient, remoteClient, localServer := s.rpcHandlerForAlloc(allocID)
	var handler structs.StreamingRpcHandler
	var handlerErr error
	if localClient {
		handler, handlerErr = s.agent.Client().StreamingRpcHandler(method)
	} else if remoteClient {
		handler, handlerErr = s.agent.Client().RemoteStreamingRpcHandler(method)
	} else if localServer {
		handler, handlerErr = s.agent.Server().StreamingRpcHandler(method)
	}

	if handlerErr != nil {
		return nil, CodedError(500, handlerErr.Error())
	}

	// Create a pipe connecting the (possibly remote) handler to the http response
	httpPipe, handlerPipe := net.Pipe()
	decoder := codec.NewDecoder(httpPipe, structs.MsgpackHandle)
	encoder := codec.NewEncoder(httpPipe, structs.MsgpackHandle)

	// Create a goroutine that closes the pipe if the connection closes.
	ctx, cancel := context.WithCancel(context.Background())
	go func() {
		<-ctx.Done()
		httpPipe.Close()

		// don't close ws - wait to drain messages
	}()

	// Create a channel that decodes the results
	errCh := make(chan HTTPCodedError, 2)

	// stream response
	go func() {
		defer cancel()

		// Send the request
		if err := encoder.Encode(args); err != nil {
			errCh <- CodedError(500, err.Error())
			return
		}

		go forwardExecInput(encoder, ws, errCh)

		for {
			var res cstructs.StreamErrWrapper
			err := decoder.Decode(&res)
			if isClosedError(err) {
				ws.WriteMessage(websocket.CloseMessage, websocket.FormatCloseMessage(websocket.CloseNormalClosure, ""))
				errCh <- nil
				return
			}

			if err != nil {
				errCh <- CodedError(500, err.Error())
				return
			}
			decoder.Reset(httpPipe)

			if err := res.Error; err != nil {
				code := 500
				if err.Code != nil {
					code = int(*err.Code)
				}
				errCh <- CodedError(code, err.Error())
				return
			}

			if err := ws.WriteMessage(websocket.TextMessage, res.Payload); err != nil {
				errCh <- CodedError(500, err.Error())
				return
			}
		}
	}()

	// start streaming request to streaming RPC - returns when streaming completes or errors
	handler(handlerPipe)
	// stop streaming background goroutines for streaming - but not websocket activity
	cancel()
	// retrieve any error and/or wait until goroutine stop and close errCh connection before
	// closing websocket connection
	codedErr := <-errCh

	// we won't return an error on ws close, but at least make it available in
	// the logs so we can trace spurious disconnects
	if codedErr != nil {
		s.logger.Debug("alloc exec channel closed with error", "error", codedErr)
	}

	if isClosedError(codedErr) {
		codedErr = nil
	} else if codedErr != nil {
		ws.WriteMessage(websocket.CloseMessage,
			websocket.FormatCloseMessage(toWsCode(codedErr.Code()), codedErr.Error()))
	}
	ws.Close()

	return nil, codedErr
}

func toWsCode(httpCode int) int {
	switch httpCode {
	case 500:
		return websocket.CloseInternalServerErr
	default:
		// placeholder error code
		return websocket.ClosePolicyViolation
	}
}

func isClosedError(err error) bool {
	if err == nil {
		return false
	}

	// check if the websocket "error" is one of the benign "close" status codes
	if codedErr, ok := err.(HTTPCodedError); ok {
		return slices.ContainsFunc([]string{
			"close 1000", // CLOSE_NORMAL
			"close 1001", // CLOSE_GOING_AWAY
			"close 1005", // CLOSED_NO_STATUS
		}, func(s string) bool { return strings.Contains(codedErr.Error(), s) })
	}

	return err == io.EOF ||
		err == io.ErrClosedPipe ||
		strings.Contains(err.Error(), "closed") ||
		strings.Contains(err.Error(), "EOF")
}

// forwardExecInput forwards exec input (e.g. stdin) from websocket connection
// to the streaming RPC connection to client
func forwardExecInput(encoder *codec.Encoder, ws *websocket.Conn, errCh chan<- HTTPCodedError) {
	for {
		sf := &drivers.ExecTaskStreamingRequestMsg{}
		err := ws.ReadJSON(sf)
		if err == io.EOF {
			return
		}

		if err != nil {
			errCh <- CodedError(500, err.Error())
			return
		}

		err = encoder.Encode(sf)
		if err != nil {
			errCh <- CodedError(500, err.Error())
		}
	}
}
-												[COMPLIANCE] Add Copyright and License Headers

											
										
										
											2023-04-10 15:36:59 +00:00
+								// Copyright (c) HashiCorp, Inc.
 								// SPDX-License-Identifier: MPL-2.0
-												http: adding allocs list endpoint

											
										
										
											2015-09-06 22:37:21 +00:00
+								package agent
 								import (
-												agent: add websocket handler for nomad exec

This adds a websocket endpoint for handling `nomad exec`.

The endpoint is a websocket interface, as we require a bi-directional
streaming (to handle both input and output), which is not very appropriate for
plain HTTP 1.0. Using websocket makes implementing the web ui a bit simpler. I
considered using golang http hijack capability to treat http request as a plain
connection, but the web interface would be too complicated potentially.

Furthermore, the API endpoint operates against the raw core nomad exec streaming
datastructures, defined in protobuf, with json serializer.  Our APIs use json
interfaces in general, and protobuf generates json friendly golang structs.
Reusing the structs here simplify interface and reduce conversion overhead.

											
										
										
											2019-04-28 21:33:25 +00:00
+									"context"
-												allocs: Add nomad alloc restart

This adds a `nomad alloc restart` command and api that allows a job operator
with the alloc-lifecycle acl to perform an in-place restart of a Nomad
allocation, or a given subtask.

											
										
										
											2019-04-01 12:56:02 +00:00
+									"encoding/json"
-												Adding a snapshot endpoint on the client (#1730)


											
										
										
											2016-09-22 04:28:12 +00:00
+									"fmt"
-												agent: add websocket handler for nomad exec

This adds a websocket endpoint for handling `nomad exec`.

The endpoint is a websocket interface, as we require a bi-directional
streaming (to handle both input and output), which is not very appropriate for
plain HTTP 1.0. Using websocket makes implementing the web ui a bit simpler. I
considered using golang http hijack capability to treat http request as a plain
connection, but the web interface would be too complicated potentially.

Furthermore, the API endpoint operates against the raw core nomad exec streaming
datastructures, defined in protobuf, with json serializer.  Our APIs use json
interfaces in general, and protobuf generates json friendly golang structs.
Reusing the structs here simplify interface and reduce conversion overhead.

											
										
										
											2019-04-28 21:33:25 +00:00
+									"io"
 									"net"
-												http: adding allocs list endpoint

											
										
										
											2015-09-06 22:37:21 +00:00
+									"net/http"
-												backport of commit 742651f2f715af69dda77b8ffb3af3d114e25ac2

											
										
										
											2023-11-27 08:33:08 +00:00
+									"slices"
-												agent: add websocket handler for nomad exec

This adds a websocket endpoint for handling `nomad exec`.

The endpoint is a websocket interface, as we require a bi-directional
streaming (to handle both input and output), which is not very appropriate for
plain HTTP 1.0. Using websocket makes implementing the web ui a bit simpler. I
considered using golang http hijack capability to treat http request as a plain
connection, but the web interface would be too complicated potentially.

Furthermore, the API endpoint operates against the raw core nomad exec streaming
datastructures, defined in protobuf, with json serializer.  Our APIs use json
interfaces in general, and protobuf generates json friendly golang structs.
Reusing the structs here simplify interface and reduce conversion overhead.

											
										
										
											2019-04-28 21:33:25 +00:00
+									"strconv"
-												http: adding allocs list endpoint

											
										
										
											2015-09-06 22:37:21 +00:00
+									"strings"
-												Decompress

											
										
										
											2016-11-29 00:05:56 +00:00
+									"github.com/golang/snappy"
-												agent: add websocket handler for nomad exec

This adds a websocket endpoint for handling `nomad exec`.

The endpoint is a websocket interface, as we require a bi-directional
streaming (to handle both input and output), which is not very appropriate for
plain HTTP 1.0. Using websocket makes implementing the web ui a bit simpler. I
considered using golang http hijack capability to treat http request as a plain
connection, but the web interface would be too complicated potentially.

Furthermore, the API endpoint operates against the raw core nomad exec streaming
datastructures, defined in protobuf, with json serializer.  Our APIs use json
interfaces in general, and protobuf generates json friendly golang structs.
Reusing the structs here simplify interface and reduce conversion overhead.

											
										
										
											2019-04-28 21:33:25 +00:00
+									"github.com/gorilla/websocket"
-												Revert "hashicorp/go-msgpack v2 (#16810)" (#17047)

This reverts commit 8a98520d56eed3848096734487d8bd3eb9162a65.

											
										
										
											2023-05-01 21:18:34 +00:00
+									"github.com/hashicorp/go-msgpack/codec"
-												HTTP agent

											
										
										
											2018-02-06 18:53:00 +00:00
+									cstructs "github.com/hashicorp/nomad/client/structs"
-												http: adding allocs list endpoint

											
										
										
											2015-09-06 22:37:21 +00:00
+									"github.com/hashicorp/nomad/nomad/structs"
-												agent: add websocket handler for nomad exec

This adds a websocket endpoint for handling `nomad exec`.

The endpoint is a websocket interface, as we require a bi-directional
streaming (to handle both input and output), which is not very appropriate for
plain HTTP 1.0. Using websocket makes implementing the web ui a bit simpler. I
considered using golang http hijack capability to treat http request as a plain
connection, but the web interface would be too complicated potentially.

Furthermore, the API endpoint operates against the raw core nomad exec streaming
datastructures, defined in protobuf, with json serializer.  Our APIs use json
interfaces in general, and protobuf generates json friendly golang structs.
Reusing the structs here simplify interface and reduce conversion overhead.

											
										
										
											2019-04-28 21:33:25 +00:00
+									"github.com/hashicorp/nomad/plugins/drivers"
-												http: adding allocs list endpoint

											
										
										
											2015-09-06 22:37:21 +00:00
+								)
-												Added a test for alloc stats api endpoint

											
										
										
											2016-05-28 00:28:17 +00:00
+								const (
-												Adding a snapshot endpoint on the client (#1730)


											
										
										
											2016-09-22 04:28:12 +00:00
+									allocNotFoundErr    = "allocation not found"
 									resourceNotFoundErr = "resource not found"
-												Added a test for alloc stats api endpoint

											
										
										
											2016-05-28 00:28:17 +00:00
+								)
-												http: adding allocs list endpoint

											
										
										
											2015-09-06 22:37:21 +00:00
+								func (s *HTTPServer) AllocsRequest(resp http.ResponseWriter, req *http.Request) (interface{}, error) {
-												chore(lint): use Go stdlib variables for HTTP methods and status codes (#17968) (#18074)

Co-authored-by: Ville Vesilehto <ville@vesilehto.fi>

											
										
										
											2023-07-26 15:38:39 +00:00
+									if req.Method != http.MethodGet {
-												http: adding allocs list endpoint

											
										
										
											2015-09-06 22:37:21 +00:00
+										return nil, CodedError(405, ErrInvalidMethod)
 									}
 									args := structs.AllocListRequest{}
 									if s.parse(resp, req, &args.Region, &args.QueryOptions) {
 										return nil, nil
 									}
-												api: add field filters to /v1/{allocations,nodes}

Fixes #9017

The ?resources=true query parameter includes resources in the object
stub listings. Specifically:

- For `/v1/nodes?resources=true` both the `NodeResources` and
  `ReservedResources` field are included.
- For `/v1/allocations?resources=true` the `AllocatedResources` field is
  included.

The ?task_states=false query parameter removes TaskStates from
/v1/allocations responses. (By default TaskStates are included.)

											
										
										
											2020-10-09 05:21:41 +00:00
+									// Parse resources and task_states field selection
-												unify boolean parameter parsing

											
										
										
											2020-10-14 19:23:25 +00:00
+									resources, err := parseBool(req, "resources")
 									if err != nil {
-												api: add field filters to /v1/{allocations,nodes}

Fixes #9017

The ?resources=true query parameter includes resources in the object
stub listings. Specifically:

- For `/v1/nodes?resources=true` both the `NodeResources` and
  `ReservedResources` field are included.
- For `/v1/allocations?resources=true` the `AllocatedResources` field is
  included.

The ?task_states=false query parameter removes TaskStates from
/v1/allocations responses. (By default TaskStates are included.)

											
										
										
											2020-10-09 05:21:41 +00:00
+										return nil, err
 									}
-												unify boolean parameter parsing

											
										
										
											2020-10-14 19:23:25 +00:00
+									taskStates, err := parseBool(req, "task_states")
 									if err != nil {
-												api: add field filters to /v1/{allocations,nodes}

Fixes #9017

The ?resources=true query parameter includes resources in the object
stub listings. Specifically:

- For `/v1/nodes?resources=true` both the `NodeResources` and
  `ReservedResources` field are included.
- For `/v1/allocations?resources=true` the `AllocatedResources` field is
  included.

The ?task_states=false query parameter removes TaskStates from
/v1/allocations responses. (By default TaskStates are included.)

											
										
										
											2020-10-09 05:21:41 +00:00
+										return nil, err
 									}
-												unify boolean parameter parsing

											
										
										
											2020-10-14 19:23:25 +00:00
+									if resources != nil || taskStates != nil {
 										args.Fields = structs.NewAllocStubFields()
 										if resources != nil {
 											args.Fields.Resources = *resources
 										}
 										if taskStates != nil {
 											args.Fields.TaskStates = *taskStates
 										}
 									}
-												http: adding allocs list endpoint

											
										
										
											2015-09-06 22:37:21 +00:00
+									var out structs.AllocListResponse
 									if err := s.agent.RPC("Alloc.List", &args, &out); err != nil {
 										return nil, err
 									}
 									setMeta(resp, &out.QueryMeta)
-												http: list results are never null

											
										
										
											2015-09-07 17:03:10 +00:00
+									if out.Allocations == nil {
 										out.Allocations = make([]*structs.AllocListStub, 0)
 									}
-												Populate DisplayMessage in various http endpoints that return allocations, plus unit tests.

											
										
										
											2017-11-17 20:53:26 +00:00
+									for _, alloc := range out.Allocations {
 										alloc.SetEventDisplayMessages()
 									}
-												http: adding allocs list endpoint

											
										
										
											2015-09-06 22:37:21 +00:00
+									return out.Allocations, nil
 								}
 								func (s *HTTPServer) AllocSpecificRequest(resp http.ResponseWriter, req *http.Request) (interface{}, error) {
-												allocs: Add nomad alloc stop

This adds a `nomad alloc stop` command that can be used to stop and
force migrate an allocation to a different node.

This is built on top of the AllocUpdateDesiredTransitionRequest and
explicitly limits the scope of access to that transition to expose it
under the alloc-lifecycle ACL.

The API returns the follow up eval that can be used as part of
monitoring in the CLI or parsed and used in an external tool.

											
										
										
											2019-04-01 14:21:03 +00:00
+									reqSuffix := strings.TrimPrefix(req.URL.Path, "/v1/allocation/")
 									// tokenize the suffix of the path to get the alloc id and find the action
 									// invoked on the alloc id
 									tokens := strings.Split(reqSuffix, "/")
 									if len(tokens) > 2 || len(tokens) < 1 {
 										return nil, CodedError(404, resourceNotFoundErr)
 									}
 									allocID := tokens[0]
 									if len(tokens) == 1 {
 										return s.allocGet(allocID, resp, req)
 									}
 									switch tokens[1] {
-												client: add support for checks in nomad services

This PR adds support for specifying checks in services registered to
the built-in nomad service provider.

Currently only HTTP and TCP checks are supported, though more types
could be added later.

											
										
										
											2022-06-07 14:18:19 +00:00
+									case "checks":
 										return s.allocChecks(allocID, resp, req)
-												allocs: Add nomad alloc stop

This adds a `nomad alloc stop` command that can be used to stop and
force migrate an allocation to a different node.

This is built on top of the AllocUpdateDesiredTransitionRequest and
explicitly limits the scope of access to that transition to expose it
under the alloc-lifecycle ACL.

The API returns the follow up eval that can be used as part of
monitoring in the CLI or parsed and used in an external tool.

											
										
										
											2019-04-01 14:21:03 +00:00
+									case "stop":
 										return s.allocStop(allocID, resp, req)
-												http: add alloc service registration agent HTTP endpoint.

											
										
										
											2022-03-03 11:13:32 +00:00
+									case "services":
 										return s.allocServiceRegistrations(resp, req, allocID)
-												allocs: Add nomad alloc stop

This adds a `nomad alloc stop` command that can be used to stop and
force migrate an allocation to a different node.

This is built on top of the AllocUpdateDesiredTransitionRequest and
explicitly limits the scope of access to that transition to expose it
under the alloc-lifecycle ACL.

The API returns the follow up eval that can be used as part of
monitoring in the CLI or parsed and used in an external tool.

											
										
										
											2019-04-01 14:21:03 +00:00
+									}
 									return nil, CodedError(404, resourceNotFoundErr)
 								}
 								func (s *HTTPServer) allocGet(allocID string, resp http.ResponseWriter, req *http.Request) (interface{}, error) {
-												chore(lint): use Go stdlib variables for HTTP methods and status codes (#17968) (#18074)

Co-authored-by: Ville Vesilehto <ville@vesilehto.fi>

											
										
										
											2023-07-26 15:38:39 +00:00
+									if req.Method != http.MethodGet {
-												http: adding allocs list endpoint

											
										
										
											2015-09-06 22:37:21 +00:00
+										return nil, CodedError(405, ErrInvalidMethod)
 									}
-												http: adding alloc lookup endpoint

											
										
										
											2015-09-06 22:49:44 +00:00
+									args := structs.AllocSpecificRequest{
 										AllocID: allocID,
 									}
 									if s.parse(resp, req, &args.Region, &args.QueryOptions) {
 										return nil, nil
 									}
 									var out structs.SingleAllocResponse
 									if err := s.agent.RPC("Alloc.GetAlloc", &args, &out); err != nil {
 										return nil, err
 									}
-												http: adding allocs list endpoint

											
										
										
											2015-09-06 22:37:21 +00:00
-												http: adding alloc lookup endpoint

											
										
										
											2015-09-06 22:49:44 +00:00
+									setMeta(resp, &out.QueryMeta)
 									if out.Alloc == nil {
 										return nil, CodedError(404, "alloc not found")
 									}
-												Decompress

											
										
										
											2016-11-29 00:05:56 +00:00
-												Rename structs

											
										
										
											2016-12-14 20:50:08 +00:00
+									// Decode the payload if there is any
-												Update UI to use new allocated ports fields (#8631)

* nomad: canonicalize alloc shared resources to populate ports

* ui: network ports

* ui: remove unused task network references and update tests with new shared ports model

* ui: lint

* ui: revert auto formatting

* ui: remove unused page objects

* structs: remove unrelated test from bad conflict resolution

* ui: formatting

											
										
										
											2020-08-20 15:07:13 +00:00
-												Decompress

											
										
										
											2016-11-29 00:05:56 +00:00
+									alloc := out.Alloc
-												Rename structs

											
										
										
											2016-12-14 20:50:08 +00:00
+									if alloc.Job != nil && len(alloc.Job.Payload) != 0 {
 										decoded, err := snappy.Decode(nil, alloc.Job.Payload)
-												Decompress

											
										
										
											2016-11-29 00:05:56 +00:00
+										if err != nil {
 											return nil, err
 										}
 										alloc = alloc.Copy()
-												Rename structs

											
										
										
											2016-12-14 20:50:08 +00:00
+										alloc.Job.Payload = decoded
-												Decompress

											
										
										
											2016-11-29 00:05:56 +00:00
+									}
-												Populate DisplayMessage in various http endpoints that return allocations, plus unit tests.

											
										
										
											2017-11-17 20:53:26 +00:00
+									alloc.SetEventDisplayMessages()
-												Decompress

											
										
										
											2016-11-29 00:05:56 +00:00
-												Update UI to use new allocated ports fields (#8631)

* nomad: canonicalize alloc shared resources to populate ports

* ui: network ports

* ui: remove unused task network references and update tests with new shared ports model

* ui: lint

* ui: revert auto formatting

* ui: remove unused page objects

* structs: remove unrelated test from bad conflict resolution

* ui: formatting

											
										
										
											2020-08-20 15:07:13 +00:00
+									// Handle 0.12 ports upgrade path
-												updated alloc_endpoint to mutate a copy of the returned allocation, instead of the instance in the state store

											
										
										
											2020-11-15 17:52:50 +00:00
+									alloc = alloc.Copy()
-												structs: canonicalize allocatedtaskresources to populate shared ports (#9309)


											
										
										
											2020-11-11 21:21:47 +00:00
+									alloc.AllocatedResources.Canonicalize()
-												Update UI to use new allocated ports fields (#8631)

* nomad: canonicalize alloc shared resources to populate ports

* ui: network ports

* ui: remove unused task network references and update tests with new shared ports model

* ui: lint

* ui: revert auto formatting

* ui: remove unused page objects

* structs: remove unrelated test from bad conflict resolution

* ui: formatting

											
										
										
											2020-08-20 15:07:13 +00:00
-												Decompress

											
										
										
											2016-11-29 00:05:56 +00:00
+									return alloc, nil
-												http: adding allocs list endpoint

											
										
										
											2015-09-06 22:37:21 +00:00
+								}
-												Changed the stats endpoints

											
										
										
											2016-05-24 23:41:35 +00:00
-												allocs: Add nomad alloc stop

This adds a `nomad alloc stop` command that can be used to stop and
force migrate an allocation to a different node.

This is built on top of the AllocUpdateDesiredTransitionRequest and
explicitly limits the scope of access to that transition to expose it
under the alloc-lifecycle ACL.

The API returns the follow up eval that can be used as part of
monitoring in the CLI or parsed and used in an external tool.

											
										
										
											2019-04-01 14:21:03 +00:00
+								func (s *HTTPServer) allocStop(allocID string, resp http.ResponseWriter, req *http.Request) (interface{}, error) {
 									if !(req.Method == "POST" || req.Method == "PUT") {
 										return nil, CodedError(405, ErrInvalidMethod)
 									}
-												Changed the stats endpoints

											
										
										
											2016-05-24 23:41:35 +00:00
-												provide `-no-shutdown-delay` flag for job/alloc stop (#11596)

Some operators use very long group/task `shutdown_delay` settings to
safely drain network connections to their workloads after service
deregistration. But during incident response, they may want to cause
that drain to be skipped so they can quickly shed load.

Provide a `-no-shutdown-delay` flag on the `nomad alloc stop` and
`nomad job stop` commands that bypasses the delay. This sets a new
desired transition state on the affected allocations that the
allocation/task runner will identify during pre-kill on the client.

Note (as documented here) that using this flag will almost always
result in failed inbound network connections for workloads as the
tasks will exit before clients receive updated service discovery
information and won't be gracefully drained.

											
										
										
											2021-12-13 19:54:53 +00:00
+									noShutdownDelay := false
 									if noShutdownDelayQS := req.URL.Query().Get("no_shutdown_delay"); noShutdownDelayQS != "" {
 										var err error
 										noShutdownDelay, err = strconv.ParseBool(noShutdownDelayQS)
 										if err != nil {
 											return nil, fmt.Errorf("no_shutdown_delay value is not a boolean: %v", err)
 										}
 									}
-												allocs: Add nomad alloc stop

This adds a `nomad alloc stop` command that can be used to stop and
force migrate an allocation to a different node.

This is built on top of the AllocUpdateDesiredTransitionRequest and
explicitly limits the scope of access to that transition to expose it
under the alloc-lifecycle ACL.

The API returns the follow up eval that can be used as part of
monitoring in the CLI or parsed and used in an external tool.

											
										
										
											2019-04-01 14:21:03 +00:00
+									sr := &structs.AllocStopRequest{
-												provide `-no-shutdown-delay` flag for job/alloc stop (#11596)

Some operators use very long group/task `shutdown_delay` settings to
safely drain network connections to their workloads after service
deregistration. But during incident response, they may want to cause
that drain to be skipped so they can quickly shed load.

Provide a `-no-shutdown-delay` flag on the `nomad alloc stop` and
`nomad job stop` commands that bypasses the delay. This sets a new
desired transition state on the affected allocations that the
allocation/task runner will identify during pre-kill on the client.

Note (as documented here) that using this flag will almost always
result in failed inbound network connections for workloads as the
tasks will exit before clients receive updated service discovery
information and won't be gracefully drained.

											
										
										
											2021-12-13 19:54:53 +00:00
+										AllocID:         allocID,
 										NoShutdownDelay: noShutdownDelay,
-												allocs: Add nomad alloc stop

This adds a `nomad alloc stop` command that can be used to stop and
force migrate an allocation to a different node.

This is built on top of the AllocUpdateDesiredTransitionRequest and
explicitly limits the scope of access to that transition to expose it
under the alloc-lifecycle ACL.

The API returns the follow up eval that can be used as part of
monitoring in the CLI or parsed and used in an external tool.

											
										
										
											2019-04-01 14:21:03 +00:00
+									}
 									s.parseWriteRequest(req, &sr.WriteRequest)
 									var out structs.AllocStopResponse
-												alloc lifecycle: 404 when attempting to stop non-existent allocation

											
										
										
											2019-06-20 20:52:40 +00:00
+									rpcErr := s.agent.RPC("Alloc.Stop", &sr, &out)
 									if rpcErr != nil {
 										if structs.IsErrUnknownAllocation(rpcErr) {
 											rpcErr = CodedError(404, allocNotFoundErr)
 										}
-												api: return X-Nomad-Index header on allocation stop

											
										
										
											2019-06-21 16:20:06 +00:00
+										return nil, rpcErr
-												alloc lifecycle: 404 when attempting to stop non-existent allocation

											
										
										
											2019-06-20 20:52:40 +00:00
+									}
-												api: return X-Nomad-Index header on allocation stop

											
										
										
											2019-06-21 16:20:06 +00:00
+									setIndex(resp, out.Index)
 									return &out, nil
-												allocs: Add nomad alloc stop

This adds a `nomad alloc stop` command that can be used to stop and
force migrate an allocation to a different node.

This is built on top of the AllocUpdateDesiredTransitionRequest and
explicitly limits the scope of access to that transition to expose it
under the alloc-lifecycle ACL.

The API returns the follow up eval that can be used as part of
monitoring in the CLI or parsed and used in an external tool.

											
										
										
											2019-04-01 14:21:03 +00:00
+								}
-												http: add alloc service registration agent HTTP endpoint.

											
										
										
											2022-03-03 11:13:32 +00:00
+								// allocServiceRegistrations returns a list of all service registrations
 								// assigned to the job identifier. It is callable via the
 								// /v1/allocation/:alloc_id/services HTTP API and uses the
 								// structs.AllocServiceRegistrationsRPCMethod RPC method.
 								func (s *HTTPServer) allocServiceRegistrations(
 									resp http.ResponseWriter, req *http.Request, allocID string) (interface{}, error) {
 									// The endpoint only supports GET requests.
 									if req.Method != http.MethodGet {
 										return nil, CodedError(http.StatusMethodNotAllowed, ErrInvalidMethod)
 									}
 									// Set up the request args and parse this to ensure the query options are
 									// set.
 									args := structs.AllocServiceRegistrationsRequest{AllocID: allocID}
 									if s.parse(resp, req, &args.Region, &args.QueryOptions) {
 										return nil, nil
 									}
 									// Perform the RPC request.
 									var reply structs.AllocServiceRegistrationsResponse
 									if err := s.agent.RPC(structs.AllocServiceRegistrationsRPCMethod, &args, &reply); err != nil {
 										return nil, err
 									}
 									setMeta(resp, &reply.QueryMeta)
 									if reply.Services == nil {
 										return nil, CodedError(http.StatusNotFound, allocNotFoundErr)
 									}
 									return reply.Services, nil
 								}
-												allocs: Add nomad alloc stop

This adds a `nomad alloc stop` command that can be used to stop and
force migrate an allocation to a different node.

This is built on top of the AllocUpdateDesiredTransitionRequest and
explicitly limits the scope of access to that transition to expose it
under the alloc-lifecycle ACL.

The API returns the follow up eval that can be used as part of
monitoring in the CLI or parsed and used in an external tool.

											
										
										
											2019-04-01 14:21:03 +00:00
+								func (s *HTTPServer) ClientAllocRequest(resp http.ResponseWriter, req *http.Request) (interface{}, error) {
-												Changed the stats endpoints

											
										
										
											2016-05-24 23:41:35 +00:00
+									reqSuffix := strings.TrimPrefix(req.URL.Path, "/v1/client/allocation/")
 									// tokenize the suffix of the path to get the alloc id and find the action
 									// invoked on the alloc id
 									tokens := strings.Split(reqSuffix, "/")
-												Adding a snapshot endpoint on the client (#1730)


											
										
										
											2016-09-22 04:28:12 +00:00
+									if len(tokens) != 2 {
 										return nil, CodedError(404, resourceNotFoundErr)
-												Changed the stats endpoints

											
										
										
											2016-05-24 23:41:35 +00:00
+									}
 									allocID := tokens[0]
-												Adding a snapshot endpoint on the client (#1730)


											
										
										
											2016-09-22 04:28:12 +00:00
+									switch tokens[1] {
-												client: add support for checks in nomad services

This PR adds support for specifying checks in services registered to
the built-in nomad service provider.

Currently only HTTP and TCP checks are supported, though more types
could be added later.

											
										
										
											2022-06-07 14:18:19 +00:00
+									case "checks":
 										return s.allocChecks(allocID, resp, req)
-												Adding a snapshot endpoint on the client (#1730)


											
										
										
											2016-09-22 04:28:12 +00:00
+									case "stats":
 										return s.allocStats(allocID, resp, req)
-												agent: add websocket handler for nomad exec

This adds a websocket endpoint for handling `nomad exec`.

The endpoint is a websocket interface, as we require a bi-directional
streaming (to handle both input and output), which is not very appropriate for
plain HTTP 1.0. Using websocket makes implementing the web ui a bit simpler. I
considered using golang http hijack capability to treat http request as a plain
connection, but the web interface would be too complicated potentially.

Furthermore, the API endpoint operates against the raw core nomad exec streaming
datastructures, defined in protobuf, with json serializer.  Our APIs use json
interfaces in general, and protobuf generates json friendly golang structs.
Reusing the structs here simplify interface and reduce conversion overhead.

											
										
										
											2019-04-28 21:33:25 +00:00
+									case "exec":
 										return s.allocExec(allocID, resp, req)
-												Adding a snapshot endpoint on the client (#1730)


											
										
										
											2016-09-22 04:28:12 +00:00
+									case "snapshot":
-												Task API via Unix Domain Socket (#15864)

This change introduces the Task API: a portable way for tasks to access Nomad's HTTP API. This particular implementation uses a Unix Domain Socket and, unlike the agent's HTTP API, always requires authentication even if ACLs are disabled.

This PR contains the core feature and tests but followup work is required for the following TODO items:

- Docs - might do in a followup since dynamic node metadata / task api / workload id all need to interlink
- Unit tests for auth middleware
- Caching for auth middleware
- Rate limiting on negative lookups for auth middleware

---------

Co-authored-by: Seth Hoenig <shoenig@duck.com>

											
										
										
											2023-02-06 19:31:22 +00:00
+										if s.agent.Client() == nil {
-												HTTP agent

											
										
										
											2018-02-06 18:53:00 +00:00
+											return nil, clientNotRunning
 										}
-												fixing up code review comments

											
										
										
											2017-10-07 00:54:09 +00:00
+										return s.allocSnapshot(allocID, resp, req)
-												allocs: Add nomad alloc restart

This adds a `nomad alloc restart` command and api that allows a job operator
with the alloc-lifecycle acl to perform an in-place restart of a Nomad
allocation, or a given subtask.

											
										
										
											2019-04-01 12:56:02 +00:00
+									case "restart":
 										return s.allocRestart(allocID, resp, req)
-												Added a garbage collector for allocations

											
										
										
											2016-12-12 06:33:12 +00:00
+									case "gc":
 										return s.allocGC(allocID, resp, req)
-												allocs: Add nomad alloc signal command

This command will be used to send a signal to either a single task within an
allocation, or all of the tasks if <task-name> is omitted. If the sent signal
terminates the allocation, it will be treated as if the allocation has crashed,
rather than as if it was operator-terminated.

Signal validation is currently handled by the driver itself and nomad
does not attempt to restrict or validate them.

											
										
										
											2019-04-03 10:46:15 +00:00
+									case "signal":
 										return s.allocSignal(allocID, resp, req)
-												Adding a snapshot endpoint on the client (#1730)


											
										
										
											2016-09-22 04:28:12 +00:00
+									}
 									return nil, CodedError(404, resourceNotFoundErr)
 								}
-												Added a garbage collector for allocations

											
										
										
											2016-12-12 06:33:12 +00:00
+								func (s *HTTPServer) ClientGCRequest(resp http.ResponseWriter, req *http.Request) (interface{}, error) {
-												HTTP agent

											
										
										
											2018-02-06 18:53:00 +00:00
-												Dynamic Node Metadata (#15844)

Fixes #14617
Dynamic Node Metadata allows Nomad users, and their jobs, to update Node metadata through an API. Currently Node metadata is only reloaded when a Client agent is restarted.

Includes new UI for editing metadata as well.

---------

Co-authored-by: Phil Renaud <phil.renaud@hashicorp.com>

											
										
										
											2023-02-07 22:42:25 +00:00
+									// Build the request and get the requested Node ID
 									args := structs.NodeSpecificRequest{}
-												HTTP agent

											
										
										
											2018-02-06 18:53:00 +00:00
+									s.parse(resp, req, &args.QueryOptions.Region, &args.QueryOptions)
-												Dynamic Node Metadata (#15844)

Fixes #14617
Dynamic Node Metadata allows Nomad users, and their jobs, to update Node metadata through an API. Currently Node metadata is only reloaded when a Client agent is restarted.

Includes new UI for editing metadata as well.

---------

Co-authored-by: Phil Renaud <phil.renaud@hashicorp.com>

											
										
										
											2023-02-07 22:42:25 +00:00
+									parseNode(req, &args.NodeID)
-												HTTP agent

											
										
										
											2018-02-06 18:53:00 +00:00
 									// Determine the handler to use
-												Dynamic Node Metadata (#15844)

Fixes #14617
Dynamic Node Metadata allows Nomad users, and their jobs, to update Node metadata through an API. Currently Node metadata is only reloaded when a Client agent is restarted.

Includes new UI for editing metadata as well.

---------

Co-authored-by: Phil Renaud <phil.renaud@hashicorp.com>

											
										
										
											2023-02-07 22:42:25 +00:00
+									useLocalClient, useClientRPC, useServerRPC := s.rpcHandlerForNode(args.NodeID)
-												HTTP agent

											
										
										
											2018-02-06 18:53:00 +00:00
 									// Make the RPC
 									var reply structs.GenericResponse
 									var rpcErr error
 									if useLocalClient {
 										rpcErr = s.agent.Client().ClientRPC("Allocations.GarbageCollectAll", &args, &reply)
 									} else if useClientRPC {
 										rpcErr = s.agent.Client().RPC("ClientAllocations.GarbageCollectAll", &args, &reply)
 									} else if useServerRPC {
 										rpcErr = s.agent.Server().RPC("ClientAllocations.GarbageCollectAll", &args, &reply)
 									} else {
 										rpcErr = CodedError(400, "No local Node and node_id not provided")
 									}
 									if rpcErr != nil {
 										if structs.IsErrNoNodeConn(rpcErr) {
 											rpcErr = CodedError(404, rpcErr.Error())
 										}
-												/v1/client/gc ACL enforcement

											
										
										
											2017-10-04 22:08:58 +00:00
+									}
-												HTTP agent

											
										
										
											2018-02-06 18:53:00 +00:00
+									return nil, rpcErr
-												Added a garbage collector for allocations

											
										
										
											2016-12-12 06:33:12 +00:00
+								}
-												allocs: Add nomad alloc restart

This adds a `nomad alloc restart` command and api that allows a job operator
with the alloc-lifecycle acl to perform an in-place restart of a Nomad
allocation, or a given subtask.

											
										
										
											2019-04-01 12:56:02 +00:00
+								func (s *HTTPServer) allocRestart(allocID string, resp http.ResponseWriter, req *http.Request) (interface{}, error) {
 									// Build the request and parse the ACL token
 									args := structs.AllocRestartRequest{
 										AllocID:  allocID,
 										TaskName: "",
 									}
 									s.parse(resp, req, &args.QueryOptions.Region, &args.QueryOptions)
 									// Explicitly parse the body separately to disallow overriding AllocID in req Body.
 									var reqBody struct {
 										TaskName string
-												Task lifecycle restart (#14127)

* allocrunner: handle lifecycle when all tasks die

When all tasks die the Coordinator must transition to its terminal
state, coordinatorStatePoststop, to unblock poststop tasks. Since this
could happen at any time (for example, a prestart task dies), all states
must be able to transition to this terminal state.

* allocrunner: implement different alloc restarts

Add a new alloc restart mode where all tasks are restarted, even if they
have already exited. Also unifies the alloc restart logic to use the
implementation that restarts tasks concurrently and ignores
ErrTaskNotRunning errors since those are expected when restarting the
allocation.

* allocrunner: allow tasks to run again

Prevent the task runner Run() method from exiting to allow a dead task
to run again. When the task runner is signaled to restart, the function
will jump back to the MAIN loop and run it again.

The task runner determines if a task needs to run again based on two new
task events that were added to differentiate between a request to
restart a specific task, the tasks that are currently running, or all
tasks that have already run.

* api/cli: add support for all tasks alloc restart

Implement the new -all-tasks alloc restart CLI flag and its API
counterpar, AllTasks. The client endpoint calls the appropriate restart
method from the allocrunner depending on the restart parameters used.

* test: fix tasklifecycle Coordinator test

* allocrunner: kill taskrunners if all tasks are dead

When all non-poststop tasks are dead we need to kill the taskrunners so
we don't leak their goroutines, which are blocked in the alloc restart
loop. This also ensures the allocrunner exits on its own.

* taskrunner: fix tests that waited on WaitCh

Now that "dead" tasks may run again, the taskrunner Run() method will
not return when the task finishes running, so tests must wait for the
task state to be "dead" instead of using the WaitCh, since it won't be
closed until the taskrunner is killed.

* tests: add tests for all tasks alloc restart

* changelog: add entry for #14127

* taskrunner: fix restore logic.

The first implementation of the task runner restore process relied on
server data (`tr.Alloc().TerminalStatus()`) which may not be available
to the client at the time of restore.

It also had the incorrect code path. When restoring a dead task the
driver handle always needs to be clear cleanly using `clearDriverHandle`
otherwise, after exiting the MAIN loop, the task may be killed by
`tr.handleKill`.

The fix is to store the state of the Run() loop in the task runner local
client state: if the task runner ever exits this loop cleanly (not with
a shutdown) it will never be able to run again. So if the Run() loops
starts with this local state flag set, it must exit early.

This local state flag is also being checked on task restart requests. If
the task is "dead" and its Run() loop is not active it will never be
able to run again.

* address code review requests

* apply more code review changes

* taskrunner: add different Restart modes

Using the task event to differentiate between the allocrunner restart
methods proved to be confusing for developers to understand how it all
worked.

So instead of relying on the event type, this commit separated the logic
of restarting an taskRunner into two methods:
- `Restart` will retain the current behaviour and only will only restart
  the task if it's currently running.
- `ForceRestart` is the new method where a `dead` task is allowed to
  restart if its `Run()` method is still active. Callers will need to
  restart the allocRunner taskCoordinator to make sure it will allow the
  task to run again.

* minor fixes

											
										
										
											2022-08-24 21:43:07 +00:00
+										AllTasks bool
-												allocs: Add nomad alloc restart

This adds a `nomad alloc restart` command and api that allows a job operator
with the alloc-lifecycle acl to perform an in-place restart of a Nomad
allocation, or a given subtask.

											
										
										
											2019-04-01 12:56:02 +00:00
+									}
 									err := json.NewDecoder(req.Body).Decode(&reqBody)
-												alloc-lifecycle: Fix restart with empty body

Currently when you submit a manual request to the alloc lifecycle API
with a version of Curl that will submit empty bodies, the alloc restart
api will fail with an EOF error.

This behaviour is undesired, as it is reasonable to not submit a body at
all when restarting an entire allocation rather than an individual task.

This fixes it by ignoring EOF (not unexpected EOF) errors and treating
them as entire task restarts.

											
										
										
											2019-06-12 13:19:37 +00:00
+									if err != nil && err != io.EOF {
-												allocs: Add nomad alloc restart

This adds a `nomad alloc restart` command and api that allows a job operator
with the alloc-lifecycle acl to perform an in-place restart of a Nomad
allocation, or a given subtask.

											
										
										
											2019-04-01 12:56:02 +00:00
+										return nil, err
 									}
 									if reqBody.TaskName != "" {
 										args.TaskName = reqBody.TaskName
 									}
-												Task lifecycle restart (#14127)

* allocrunner: handle lifecycle when all tasks die

When all tasks die the Coordinator must transition to its terminal
state, coordinatorStatePoststop, to unblock poststop tasks. Since this
could happen at any time (for example, a prestart task dies), all states
must be able to transition to this terminal state.

* allocrunner: implement different alloc restarts

Add a new alloc restart mode where all tasks are restarted, even if they
have already exited. Also unifies the alloc restart logic to use the
implementation that restarts tasks concurrently and ignores
ErrTaskNotRunning errors since those are expected when restarting the
allocation.

* allocrunner: allow tasks to run again

Prevent the task runner Run() method from exiting to allow a dead task
to run again. When the task runner is signaled to restart, the function
will jump back to the MAIN loop and run it again.

The task runner determines if a task needs to run again based on two new
task events that were added to differentiate between a request to
restart a specific task, the tasks that are currently running, or all
tasks that have already run.

* api/cli: add support for all tasks alloc restart

Implement the new -all-tasks alloc restart CLI flag and its API
counterpar, AllTasks. The client endpoint calls the appropriate restart
method from the allocrunner depending on the restart parameters used.

* test: fix tasklifecycle Coordinator test

* allocrunner: kill taskrunners if all tasks are dead

When all non-poststop tasks are dead we need to kill the taskrunners so
we don't leak their goroutines, which are blocked in the alloc restart
loop. This also ensures the allocrunner exits on its own.

* taskrunner: fix tests that waited on WaitCh

Now that "dead" tasks may run again, the taskrunner Run() method will
not return when the task finishes running, so tests must wait for the
task state to be "dead" instead of using the WaitCh, since it won't be
closed until the taskrunner is killed.

* tests: add tests for all tasks alloc restart

* changelog: add entry for #14127

* taskrunner: fix restore logic.

The first implementation of the task runner restore process relied on
server data (`tr.Alloc().TerminalStatus()`) which may not be available
to the client at the time of restore.

It also had the incorrect code path. When restoring a dead task the
driver handle always needs to be clear cleanly using `clearDriverHandle`
otherwise, after exiting the MAIN loop, the task may be killed by
`tr.handleKill`.

The fix is to store the state of the Run() loop in the task runner local
client state: if the task runner ever exits this loop cleanly (not with
a shutdown) it will never be able to run again. So if the Run() loops
starts with this local state flag set, it must exit early.

This local state flag is also being checked on task restart requests. If
the task is "dead" and its Run() loop is not active it will never be
able to run again.

* address code review requests

* apply more code review changes

* taskrunner: add different Restart modes

Using the task event to differentiate between the allocrunner restart
methods proved to be confusing for developers to understand how it all
worked.

So instead of relying on the event type, this commit separated the logic
of restarting an taskRunner into two methods:
- `Restart` will retain the current behaviour and only will only restart
  the task if it's currently running.
- `ForceRestart` is the new method where a `dead` task is allowed to
  restart if its `Run()` method is still active. Callers will need to
  restart the allocRunner taskCoordinator to make sure it will allow the
  task to run again.

* minor fixes

											
										
										
											2022-08-24 21:43:07 +00:00
+									if reqBody.AllTasks {
 										args.AllTasks = reqBody.AllTasks
 									}
-												allocs: Add nomad alloc restart

This adds a `nomad alloc restart` command and api that allows a job operator
with the alloc-lifecycle acl to perform an in-place restart of a Nomad
allocation, or a given subtask.

											
										
										
											2019-04-01 12:56:02 +00:00
 									// Determine the handler to use
 									useLocalClient, useClientRPC, useServerRPC := s.rpcHandlerForAlloc(allocID)
 									// Make the RPC
 									var reply structs.GenericResponse
 									var rpcErr error
 									if useLocalClient {
 										rpcErr = s.agent.Client().ClientRPC("Allocations.Restart", &args, &reply)
 									} else if useClientRPC {
 										rpcErr = s.agent.Client().RPC("ClientAllocations.Restart", &args, &reply)
 									} else if useServerRPC {
 										rpcErr = s.agent.Server().RPC("ClientAllocations.Restart", &args, &reply)
 									} else {
 										rpcErr = CodedError(400, "No local Node and node_id not provided")
 									}
 									if rpcErr != nil {
 										if structs.IsErrNoNodeConn(rpcErr) || structs.IsErrUnknownAllocation(rpcErr) {
 											rpcErr = CodedError(404, rpcErr.Error())
 										}
 									}
 									return reply, rpcErr
 								}
-												Added a garbage collector for allocations

											
										
										
											2016-12-12 06:33:12 +00:00
+								func (s *HTTPServer) allocGC(allocID string, resp http.ResponseWriter, req *http.Request) (interface{}, error) {
-												HTTP agent

											
										
										
											2018-02-06 18:53:00 +00:00
+									// Build the request and parse the ACL token
 									args := structs.AllocSpecificRequest{
 										AllocID: allocID,
 									}
 									s.parse(resp, req, &args.QueryOptions.Region, &args.QueryOptions)
-												/v1/client/allocation/./{stats,gc} ACL enforcement

											
										
										
											2017-10-06 00:33:05 +00:00
-												HTTP agent

											
										
										
											2018-02-06 18:53:00 +00:00
+									// Determine the handler to use
 									useLocalClient, useClientRPC, useServerRPC := s.rpcHandlerForAlloc(allocID)
-												/v1/client/allocation/./{stats,gc} ACL enforcement

											
										
										
											2017-10-06 00:33:05 +00:00
-												HTTP agent

											
										
										
											2018-02-06 18:53:00 +00:00
+									// Make the RPC
 									var reply structs.GenericResponse
 									var rpcErr error
 									if useLocalClient {
 										rpcErr = s.agent.Client().ClientRPC("Allocations.GarbageCollect", &args, &reply)
 									} else if useClientRPC {
 										rpcErr = s.agent.Client().RPC("ClientAllocations.GarbageCollect", &args, &reply)
 									} else if useServerRPC {
 										rpcErr = s.agent.Server().RPC("ClientAllocations.GarbageCollect", &args, &reply)
 									} else {
 										rpcErr = CodedError(400, "No local Node and node_id not provided")
-												/v1/client/allocation/./{stats,gc} ACL enforcement

											
										
										
											2017-10-06 00:33:05 +00:00
+									}
-												Fix GC'd alloc tracking

The Client.allocs map now contains all AllocRunners again, not just
un-GC'd AllocRunners. Client.allocs is only pruned when the server GCs
allocs.

Also stops logging "marked for GC" twice.

											
										
										
											2017-10-19 00:06:46 +00:00
-												HTTP agent

											
										
										
											2018-02-06 18:53:00 +00:00
+									if rpcErr != nil {
-												feedback and rebasing

											
										
										
											2018-02-13 23:50:51 +00:00
+										if structs.IsErrNoNodeConn(rpcErr) || structs.IsErrUnknownAllocation(rpcErr) {
-												HTTP agent

											
										
										
											2018-02-06 18:53:00 +00:00
+											rpcErr = CodedError(404, rpcErr.Error())
 										}
-												Fix regression by returning error on unknown alloc

											
										
										
											2017-10-28 00:00:11 +00:00
+									}
-												HTTP agent

											
										
										
											2018-02-06 18:53:00 +00:00
 									return nil, rpcErr
-												Added a garbage collector for allocations

											
										
										
											2016-12-12 06:33:12 +00:00
+								}
-												allocs: Add nomad alloc signal command

This command will be used to send a signal to either a single task within an
allocation, or all of the tasks if <task-name> is omitted. If the sent signal
terminates the allocation, it will be treated as if the allocation has crashed,
rather than as if it was operator-terminated.

Signal validation is currently handled by the driver itself and nomad
does not attempt to restrict or validate them.

											
										
										
											2019-04-03 10:46:15 +00:00
+								func (s *HTTPServer) allocSignal(allocID string, resp http.ResponseWriter, req *http.Request) (interface{}, error) {
 									if !(req.Method == "POST" || req.Method == "PUT") {
 										return nil, CodedError(405, ErrInvalidMethod)
 									}
 									// Build the request and parse the ACL token
 									args := structs.AllocSignalRequest{}
 									err := decodeBody(req, &args)
 									if err != nil {
 										return nil, CodedError(400, fmt.Sprintf("Failed to decode body: %v", err))
 									}
 									s.parse(resp, req, &args.QueryOptions.Region, &args.QueryOptions)
 									args.AllocID = allocID
 									// Determine the handler to use
 									useLocalClient, useClientRPC, useServerRPC := s.rpcHandlerForAlloc(allocID)
 									// Make the RPC
 									var reply structs.GenericResponse
 									var rpcErr error
 									if useLocalClient {
 										rpcErr = s.agent.Client().ClientRPC("Allocations.Signal", &args, &reply)
 									} else if useClientRPC {
 										rpcErr = s.agent.Client().RPC("ClientAllocations.Signal", &args, &reply)
 									} else if useServerRPC {
 										rpcErr = s.agent.Server().RPC("ClientAllocations.Signal", &args, &reply)
 									} else {
 										rpcErr = CodedError(400, "No local Node and node_id not provided")
 									}
 									if rpcErr != nil {
 										if structs.IsErrNoNodeConn(rpcErr) || structs.IsErrUnknownAllocation(rpcErr) {
 											rpcErr = CodedError(404, rpcErr.Error())
 										}
 									}
 									return reply, rpcErr
 								}
-												fixing up code review comments

											
										
										
											2017-10-07 00:54:09 +00:00
+								func (s *HTTPServer) allocSnapshot(allocID string, resp http.ResponseWriter, req *http.Request) (interface{}, error) {
 									var secret string
 									s.parseToken(req, &secret)
 									if !s.agent.Client().ValidateMigrateToken(allocID, secret) {
-												fixups from code review

change creation of a migrate token to be for a previous allocation

											
										
										
											2017-10-10 00:23:26 +00:00
+										return nil, structs.ErrPermissionDenied
-												Add functionality for authenticated volumes

											
										
										
											2017-10-03 17:53:32 +00:00
+									}
-												Adding a snapshot endpoint on the client (#1730)


											
										
										
											2016-09-22 04:28:12 +00:00
+									allocFS, err := s.agent.Client().GetAllocFS(allocID)
 									if err != nil {
 										return nil, fmt.Errorf(allocNotFoundErr)
 									}
 									if err := allocFS.Snapshot(resp); err != nil {
 										return nil, fmt.Errorf("error making snapshot: %v", err)
 									}
 									return nil, nil
 								}
-												Changed the stats endpoints

											
										
										
											2016-05-24 23:41:35 +00:00
-												Adding a snapshot endpoint on the client (#1730)


											
										
										
											2016-09-22 04:28:12 +00:00
+								func (s *HTTPServer) allocStats(allocID string, resp http.ResponseWriter, req *http.Request) (interface{}, error) {
-												/v1/client/allocation/./{stats,gc} ACL enforcement

											
										
										
											2017-10-06 00:33:05 +00:00
-												HTTP agent

											
										
										
											2018-02-06 18:53:00 +00:00
+									// Build the request and parse the ACL token
 									task := req.URL.Query().Get("task")
 									args := cstructs.AllocStatsRequest{
 										AllocID: allocID,
 										Task:    task,
 									}
 									s.parse(resp, req, &args.QueryOptions.Region, &args.QueryOptions)
-												/v1/client/allocation/./{stats,gc} ACL enforcement

											
										
										
											2017-10-06 00:33:05 +00:00
-												HTTP agent

											
										
										
											2018-02-06 18:53:00 +00:00
+									// Determine the handler to use
 									useLocalClient, useClientRPC, useServerRPC := s.rpcHandlerForAlloc(allocID)
 									// Make the RPC
 									var reply cstructs.AllocStatsResponse
 									var rpcErr error
 									if useLocalClient {
 										rpcErr = s.agent.Client().ClientRPC("Allocations.Stats", &args, &reply)
 									} else if useClientRPC {
 										rpcErr = s.agent.Client().RPC("ClientAllocations.Stats", &args, &reply)
 									} else if useServerRPC {
 										rpcErr = s.agent.Server().RPC("ClientAllocations.Stats", &args, &reply)
 									} else {
 										rpcErr = CodedError(400, "No local Node and node_id not provided")
-												/v1/client/allocation/./{stats,gc} ACL enforcement

											
										
										
											2017-10-06 00:33:05 +00:00
+									}
-												HTTP agent

											
										
										
											2018-02-06 18:53:00 +00:00
+									if rpcErr != nil {
-												feedback and rebasing

											
										
										
											2018-02-13 23:50:51 +00:00
+										if structs.IsErrNoNodeConn(rpcErr) || structs.IsErrUnknownAllocation(rpcErr) {
-												HTTP agent

											
										
										
											2018-02-06 18:53:00 +00:00
+											rpcErr = CodedError(404, rpcErr.Error())
 										}
-												Changing the api of the stats endpoints

											
										
										
											2016-05-27 21:15:51 +00:00
+									}
-												HTTP agent

											
										
										
											2018-02-06 18:53:00 +00:00
+									return reply.Stats, rpcErr
-												Changed the stats endpoints

											
										
										
											2016-05-24 23:41:35 +00:00
+								}
-												agent: add websocket handler for nomad exec

This adds a websocket endpoint for handling `nomad exec`.

The endpoint is a websocket interface, as we require a bi-directional
streaming (to handle both input and output), which is not very appropriate for
plain HTTP 1.0. Using websocket makes implementing the web ui a bit simpler. I
considered using golang http hijack capability to treat http request as a plain
connection, but the web interface would be too complicated potentially.

Furthermore, the API endpoint operates against the raw core nomad exec streaming
datastructures, defined in protobuf, with json serializer.  Our APIs use json
interfaces in general, and protobuf generates json friendly golang structs.
Reusing the structs here simplify interface and reduce conversion overhead.

											
										
										
											2019-04-28 21:33:25 +00:00
-												client: add support for checks in nomad services

This PR adds support for specifying checks in services registered to
the built-in nomad service provider.

Currently only HTTP and TCP checks are supported, though more types
could be added later.

											
										
										
											2022-06-07 14:18:19 +00:00
+								func (s *HTTPServer) allocChecks(allocID string, resp http.ResponseWriter, req *http.Request) (any, error) {
 									// Build the request and parse the ACL token
 									args := cstructs.AllocChecksRequest{
 										AllocID: allocID,
 									}
 									s.parse(resp, req, &args.QueryOptions.Region, &args.QueryOptions)
 									// Determine the handler to use
 									useLocalClient, useClientRPC, useServerRPC := s.rpcHandlerForAlloc(allocID)
 									// Make the RPC
 									var reply cstructs.AllocChecksResponse
 									var rpcErr error
 									switch {
 									case useLocalClient:
 										rpcErr = s.agent.Client().ClientRPC("Allocations.Checks", &args, &reply)
 									case useClientRPC:
-												client: fix RPC forwarding when querying checks for alloc. (#14498)

When querying the checks for an allocation, the request must be
forwarded to the agent that is running the allocation. If the
initial request is made to a server agent, the request can be made
directly to the client agent running the allocation. If the
request is made to a client agent not running the alloc, the
request needs to be forwarded to a server and then the correct
client.

											
										
										
											2022-09-08 14:55:23 +00:00
+										rpcErr = s.agent.Client().RPC("ClientAllocations.Checks", &args, &reply)
-												client: add support for checks in nomad services

This PR adds support for specifying checks in services registered to
the built-in nomad service provider.

Currently only HTTP and TCP checks are supported, though more types
could be added later.

											
										
										
											2022-06-07 14:18:19 +00:00
+									case useServerRPC:
-												client: fix RPC forwarding when querying checks for alloc. (#14498)

When querying the checks for an allocation, the request must be
forwarded to the agent that is running the allocation. If the
initial request is made to a server agent, the request can be made
directly to the client agent running the allocation. If the
request is made to a client agent not running the alloc, the
request needs to be forwarded to a server and then the correct
client.

											
										
										
											2022-09-08 14:55:23 +00:00
+										rpcErr = s.agent.Server().RPC("ClientAllocations.Checks", &args, &reply)
-												client: add support for checks in nomad services

This PR adds support for specifying checks in services registered to
the built-in nomad service provider.

Currently only HTTP and TCP checks are supported, though more types
could be added later.

											
										
										
											2022-06-07 14:18:19 +00:00
+									default:
 										rpcErr = CodedError(400, "No local Node and node_id not provided")
 									}
 									if rpcErr != nil {
 										if structs.IsErrNoNodeConn(rpcErr) || structs.IsErrUnknownAllocation(rpcErr) {
 											rpcErr = CodedError(404, rpcErr.Error())
 										}
 									}
 									return reply.Results, rpcErr
 								}
-												agent: add websocket handler for nomad exec

This adds a websocket endpoint for handling `nomad exec`.

The endpoint is a websocket interface, as we require a bi-directional
streaming (to handle both input and output), which is not very appropriate for
plain HTTP 1.0. Using websocket makes implementing the web ui a bit simpler. I
considered using golang http hijack capability to treat http request as a plain
connection, but the web interface would be too complicated potentially.

Furthermore, the API endpoint operates against the raw core nomad exec streaming
datastructures, defined in protobuf, with json serializer.  Our APIs use json
interfaces in general, and protobuf generates json friendly golang structs.
Reusing the structs here simplify interface and reduce conversion overhead.

											
										
										
											2019-04-28 21:33:25 +00:00
+								func (s *HTTPServer) allocExec(allocID string, resp http.ResponseWriter, req *http.Request) (interface{}, error) {
 									// Build the request and parse the ACL token
 									task := req.URL.Query().Get("task")
 									cmdJsonStr := req.URL.Query().Get("command")
 									var command []string
 									err := json.Unmarshal([]byte(cmdJsonStr), &command)
 									if err != nil {
 										// this shouldn't happen, []string is always be serializable to json
 										return nil, fmt.Errorf("failed to marshal command into json: %v", err)
 									}
 									ttyB := false
 									if tty := req.URL.Query().Get("tty"); tty != "" {
 										ttyB, err = strconv.ParseBool(tty)
 										if err != nil {
 											return nil, fmt.Errorf("tty value is not a boolean: %v", err)
 										}
 									}
 									args := cstructs.AllocExecRequest{
 										AllocID: allocID,
 										Task:    task,
 										Cmd:     command,
 										Tty:     ttyB,
 									}
 									s.parse(resp, req, &args.QueryOptions.Region, &args.QueryOptions)
 									conn, err := s.wsUpgrader.Upgrade(resp, req, nil)
 									if err != nil {
 										return nil, fmt.Errorf("failed to upgrade connection: %v", err)
 									}
-												backend: support WS authentication handshake in alloc/exec

The javascript Websocket API doesn't support setting custom headers
(e.g. `X-Nomad-Token`).  This change adds support for having an
authentication handshake message: clients can set `ws_handshake` URL
query parameter to true and send a single handshake message with auth
token first before any other mssage.

This is a backward compatible change: it does not affect nomad CLI path, as it
doesn't set `ws_handshake` parameter.

											
										
										
											2020-04-03 15:18:54 +00:00
+									if err := readWsHandshake(conn.ReadJSON, req, &args.QueryOptions); err != nil {
 										conn.WriteMessage(websocket.CloseMessage,
 											websocket.FormatCloseMessage(toWsCode(400), err.Error()))
 										return nil, err
 									}
-												agent: add websocket handler for nomad exec

This adds a websocket endpoint for handling `nomad exec`.

The endpoint is a websocket interface, as we require a bi-directional
streaming (to handle both input and output), which is not very appropriate for
plain HTTP 1.0. Using websocket makes implementing the web ui a bit simpler. I
considered using golang http hijack capability to treat http request as a plain
connection, but the web interface would be too complicated potentially.

Furthermore, the API endpoint operates against the raw core nomad exec streaming
datastructures, defined in protobuf, with json serializer.  Our APIs use json
interfaces in general, and protobuf generates json friendly golang structs.
Reusing the structs here simplify interface and reduce conversion overhead.

											
										
										
											2019-04-28 21:33:25 +00:00
+									return s.execStreamImpl(conn, &args)
 								}
-												fixup! backend: support WS authentication handshake in alloc/exec

											
										
										
											2020-04-03 18:20:31 +00:00
+								// readWsHandshake reads the websocket handshake message and sets
 								// query authentication token, if request requires a handshake
-												backend: support WS authentication handshake in alloc/exec

The javascript Websocket API doesn't support setting custom headers
(e.g. `X-Nomad-Token`).  This change adds support for having an
authentication handshake message: clients can set `ws_handshake` URL
query parameter to true and send a single handshake message with auth
token first before any other mssage.

This is a backward compatible change: it does not affect nomad CLI path, as it
doesn't set `ws_handshake` parameter.

											
										
										
											2020-04-03 15:18:54 +00:00
+								func readWsHandshake(readFn func(interface{}) error, req *http.Request, q *structs.QueryOptions) error {
-												fixup! backend: support WS authentication handshake in alloc/exec

											
										
										
											2020-04-03 18:20:31 +00:00
+									// Avoid handshake if request doesn't require one
-												backend: support WS authentication handshake in alloc/exec

The javascript Websocket API doesn't support setting custom headers
(e.g. `X-Nomad-Token`).  This change adds support for having an
authentication handshake message: clients can set `ws_handshake` URL
query parameter to true and send a single handshake message with auth
token first before any other mssage.

This is a backward compatible change: it does not affect nomad CLI path, as it
doesn't set `ws_handshake` parameter.

											
										
										
											2020-04-03 15:18:54 +00:00
+									if hv := req.URL.Query().Get("ws_handshake"); hv == "" {
 										return nil
 									} else if h, err := strconv.ParseBool(hv); err != nil {
 										return fmt.Errorf("ws_handshake value is not a boolean: %v", err)
 									} else if !h {
 										return nil
 									}
 									var h wsHandshakeMessage
 									err := readFn(&h)
 									if err != nil {
 										return err
 									}
-												fixup! backend: support WS authentication handshake in alloc/exec

											
										
										
											2020-04-03 18:20:31 +00:00
+									supportedWSHandshakeVersion := 1
 									if h.Version != supportedWSHandshakeVersion {
-												backend: support WS authentication handshake in alloc/exec

The javascript Websocket API doesn't support setting custom headers
(e.g. `X-Nomad-Token`).  This change adds support for having an
authentication handshake message: clients can set `ws_handshake` URL
query parameter to true and send a single handshake message with auth
token first before any other mssage.

This is a backward compatible change: it does not affect nomad CLI path, as it
doesn't set `ws_handshake` parameter.

											
										
										
											2020-04-03 15:18:54 +00:00
+										return fmt.Errorf("unexpected handshake value: %v", h.Version)
 									}
 									q.AuthToken = h.AuthToken
 									return nil
 								}
 								type wsHandshakeMessage struct {
 									Version   int    `json:"version"`
 									AuthToken string `json:"auth_token"`
 								}
-												agent: add websocket handler for nomad exec

This adds a websocket endpoint for handling `nomad exec`.

The endpoint is a websocket interface, as we require a bi-directional
streaming (to handle both input and output), which is not very appropriate for
plain HTTP 1.0. Using websocket makes implementing the web ui a bit simpler. I
considered using golang http hijack capability to treat http request as a plain
connection, but the web interface would be too complicated potentially.

Furthermore, the API endpoint operates against the raw core nomad exec streaming
datastructures, defined in protobuf, with json serializer.  Our APIs use json
interfaces in general, and protobuf generates json friendly golang structs.
Reusing the structs here simplify interface and reduce conversion overhead.

											
										
										
											2019-04-28 21:33:25 +00:00
+								func (s *HTTPServer) execStreamImpl(ws *websocket.Conn, args *cstructs.AllocExecRequest) (interface{}, error) {
 									allocID := args.AllocID
 									method := "Allocations.Exec"
 									// Get the correct handler
 									localClient, remoteClient, localServer := s.rpcHandlerForAlloc(allocID)
 									var handler structs.StreamingRpcHandler
 									var handlerErr error
 									if localClient {
 										handler, handlerErr = s.agent.Client().StreamingRpcHandler(method)
 									} else if remoteClient {
 										handler, handlerErr = s.agent.Client().RemoteStreamingRpcHandler(method)
 									} else if localServer {
 										handler, handlerErr = s.agent.Server().StreamingRpcHandler(method)
 									}
 									if handlerErr != nil {
 										return nil, CodedError(500, handlerErr.Error())
 									}
 									// Create a pipe connecting the (possibly remote) handler to the http response
 									httpPipe, handlerPipe := net.Pipe()
 									decoder := codec.NewDecoder(httpPipe, structs.MsgpackHandle)
 									encoder := codec.NewEncoder(httpPipe, structs.MsgpackHandle)
 									// Create a goroutine that closes the pipe if the connection closes.
 									ctx, cancel := context.WithCancel(context.Background())
 									go func() {
 										<-ctx.Done()
 										httpPipe.Close()
 										// don't close ws - wait to drain messages
 									}()
 									// Create a channel that decodes the results
 									errCh := make(chan HTTPCodedError, 2)
 									// stream response
 									go func() {
 										defer cancel()
 										// Send the request
 										if err := encoder.Encode(args); err != nil {
 											errCh <- CodedError(500, err.Error())
 											return
 										}
 										go forwardExecInput(encoder, ws, errCh)
 										for {
 											var res cstructs.StreamErrWrapper
 											err := decoder.Decode(&res)
 											if isClosedError(err) {
 												ws.WriteMessage(websocket.CloseMessage, websocket.FormatCloseMessage(websocket.CloseNormalClosure, ""))
 												errCh <- nil
 												return
 											}
 											if err != nil {
 												errCh <- CodedError(500, err.Error())
 												return
 											}
 											decoder.Reset(httpPipe)
 											if err := res.Error; err != nil {
 												code := 500
 												if err.Code != nil {
 													code = int(*err.Code)
 												}
 												errCh <- CodedError(code, err.Error())
 												return
 											}
 											if err := ws.WriteMessage(websocket.TextMessage, res.Payload); err != nil {
 												errCh <- CodedError(500, err.Error())
 												return
 											}
 										}
 									}()
 									// start streaming request to streaming RPC - returns when streaming completes or errors
 									handler(handlerPipe)
 									// stop streaming background goroutines for streaming - but not websocket activity
 									cancel()
-												agent: surface websocket errors in logs

The websocket interface used for `alloc exec` has to silently drop client send
errors because otherwise those errors would interleave with the streamed
output. But we may be able to surface errors that cause terminated websockets
a little better in the HTTP server logs.

											
										
										
											2021-05-24 13:46:45 +00:00
+									// retrieve any error and/or wait until goroutine stop and close errCh connection before
-												agent: add websocket handler for nomad exec

This adds a websocket endpoint for handling `nomad exec`.

The endpoint is a websocket interface, as we require a bi-directional
streaming (to handle both input and output), which is not very appropriate for
plain HTTP 1.0. Using websocket makes implementing the web ui a bit simpler. I
considered using golang http hijack capability to treat http request as a plain
connection, but the web interface would be too complicated potentially.

Furthermore, the API endpoint operates against the raw core nomad exec streaming
datastructures, defined in protobuf, with json serializer.  Our APIs use json
interfaces in general, and protobuf generates json friendly golang structs.
Reusing the structs here simplify interface and reduce conversion overhead.

											
										
										
											2019-04-28 21:33:25 +00:00
+									// closing websocket connection
 									codedErr := <-errCh
-												agent: surface websocket errors in logs

The websocket interface used for `alloc exec` has to silently drop client send
errors because otherwise those errors would interleave with the streamed
output. But we may be able to surface errors that cause terminated websockets
a little better in the HTTP server logs.

											
										
										
											2021-05-24 13:46:45 +00:00
+									// we won't return an error on ws close, but at least make it available in
 									// the logs so we can trace spurious disconnects
-												http: only log alloc/exec errors when non-nil (#13730)


											
										
										
											2022-07-13 16:44:51 +00:00
+									if codedErr != nil {
 										s.logger.Debug("alloc exec channel closed with error", "error", codedErr)
 									}
-												agent: surface websocket errors in logs

The websocket interface used for `alloc exec` has to silently drop client send
errors because otherwise those errors would interleave with the streamed
output. But we may be able to surface errors that cause terminated websockets
a little better in the HTTP server logs.

											
										
										
											2021-05-24 13:46:45 +00:00
-												agent: add websocket handler for nomad exec

This adds a websocket endpoint for handling `nomad exec`.

The endpoint is a websocket interface, as we require a bi-directional
streaming (to handle both input and output), which is not very appropriate for
plain HTTP 1.0. Using websocket makes implementing the web ui a bit simpler. I
considered using golang http hijack capability to treat http request as a plain
connection, but the web interface would be too complicated potentially.

Furthermore, the API endpoint operates against the raw core nomad exec streaming
datastructures, defined in protobuf, with json serializer.  Our APIs use json
interfaces in general, and protobuf generates json friendly golang structs.
Reusing the structs here simplify interface and reduce conversion overhead.

											
										
										
											2019-04-28 21:33:25 +00:00
+									if isClosedError(codedErr) {
 										codedErr = nil
 									} else if codedErr != nil {
 										ws.WriteMessage(websocket.CloseMessage,
 											websocket.FormatCloseMessage(toWsCode(codedErr.Code()), codedErr.Error()))
 									}
 									ws.Close()
 									return nil, codedErr
 								}
 								func toWsCode(httpCode int) int {
 									switch httpCode {
 									case 500:
 										return websocket.CloseInternalServerErr
 									default:
 										// placeholder error code
 										return websocket.ClosePolicyViolation
 									}
 								}
 								func isClosedError(err error) bool {
 									if err == nil {
 										return false
 									}
-												backport of commit 742651f2f715af69dda77b8ffb3af3d114e25ac2

											
										
										
											2023-11-27 08:33:08 +00:00
+									// check if the websocket "error" is one of the benign "close" status codes
 									if codedErr, ok := err.(HTTPCodedError); ok {
 										return slices.ContainsFunc([]string{
 											"close 1000", // CLOSE_NORMAL
 											"close 1001", // CLOSE_GOING_AWAY
 											"close 1005", // CLOSED_NO_STATUS
 										}, func(s string) bool { return strings.Contains(codedErr.Error(), s) })
 									}
-												agent: add websocket handler for nomad exec

This adds a websocket endpoint for handling `nomad exec`.

The endpoint is a websocket interface, as we require a bi-directional
streaming (to handle both input and output), which is not very appropriate for
plain HTTP 1.0. Using websocket makes implementing the web ui a bit simpler. I
considered using golang http hijack capability to treat http request as a plain
connection, but the web interface would be too complicated potentially.

Furthermore, the API endpoint operates against the raw core nomad exec streaming
datastructures, defined in protobuf, with json serializer.  Our APIs use json
interfaces in general, and protobuf generates json friendly golang structs.
Reusing the structs here simplify interface and reduce conversion overhead.

											
										
										
											2019-04-28 21:33:25 +00:00
+									return err == io.EOF ||
 										err == io.ErrClosedPipe ||
 										strings.Contains(err.Error(), "closed") ||
 										strings.Contains(err.Error(), "EOF")
 								}
 								// forwardExecInput forwards exec input (e.g. stdin) from websocket connection
 								// to the streaming RPC connection to client
 								func forwardExecInput(encoder *codec.Encoder, ws *websocket.Conn, errCh chan<- HTTPCodedError) {
 									for {
 										sf := &drivers.ExecTaskStreamingRequestMsg{}
 										err := ws.ReadJSON(sf)
 										if err == io.EOF {
 											return
 										}
 										if err != nil {
 											errCh <- CodedError(500, err.Error())
 											return
 										}
 										err = encoder.Encode(sf)
 										if err != nil {
 											errCh <- CodedError(500, err.Error())
 										}
 									}
 								}