mirror of
https://github.com/ClusterCockpit/cc-backend
synced 2026-08-31 08:57:14 +02:00
Only archived job data and internal-store node-list data honoured the selected resample algorithm. For running jobs the algorithm was dropped: MetricDataRepository.LoadData had no such parameter, so the memory store fell back to an empty string, which cc-lib's GetResampler maps to LTTB. The external store client was worse - its APIQueryRequest had no ResampleAlgo field at all, and LoadNodeListData accepted the parameter without using it. Add resampleAlgo to the LoadData interface (mirroring LoadNodeListData), forward it from metricdispatch, and set it on both stores' requests. The field is tagged omitempty, so the wire format is unchanged when empty - verify the deployed cc-metric-store accepts it before relying on it there. The REST job endpoints pass a non-zero resolution and therefore do resample, so they now request the configured default instead of an empty string. Add config.ResampleAlgo() for that, since "" is not a neutral value at this layer. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
651 lines
18 KiB
Go
651 lines
18 KiB
Go
// Copyright (C) NHR@FAU, University Erlangen-Nuremberg.
|
|
// All rights reserved. This file is part of cc-backend.
|
|
// Use of this source code is governed by a MIT-style
|
|
// license that can be found in the LICENSE file.
|
|
package api_test
|
|
|
|
import (
|
|
"bytes"
|
|
"context"
|
|
"encoding/json"
|
|
"fmt"
|
|
"net/http"
|
|
"net/http/httptest"
|
|
"os"
|
|
"path/filepath"
|
|
"reflect"
|
|
"strings"
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/ClusterCockpit/cc-backend/internal/api"
|
|
"github.com/ClusterCockpit/cc-backend/internal/archiver"
|
|
"github.com/ClusterCockpit/cc-backend/internal/auth"
|
|
"github.com/ClusterCockpit/cc-backend/internal/config"
|
|
"github.com/ClusterCockpit/cc-backend/internal/graph"
|
|
"github.com/ClusterCockpit/cc-backend/internal/metricdispatch"
|
|
"github.com/ClusterCockpit/cc-backend/internal/repository"
|
|
"github.com/ClusterCockpit/cc-backend/pkg/archive"
|
|
"github.com/ClusterCockpit/cc-backend/pkg/metricstore"
|
|
ccconf "github.com/ClusterCockpit/cc-lib/v2/ccConfig"
|
|
cclog "github.com/ClusterCockpit/cc-lib/v2/ccLogger"
|
|
"github.com/ClusterCockpit/cc-lib/v2/schema"
|
|
"github.com/go-chi/chi/v5"
|
|
|
|
_ "github.com/mattn/go-sqlite3"
|
|
)
|
|
|
|
func setup(t *testing.T) *api.RestAPI {
|
|
repository.ResetConnection()
|
|
|
|
const testconfig = `{
|
|
"main": {
|
|
"addr": "0.0.0.0:8080",
|
|
"validate": false,
|
|
"api-allowed-ips": [
|
|
"*"
|
|
]
|
|
},
|
|
"metric-store": {
|
|
"checkpoints": {
|
|
"interval": "12h"
|
|
},
|
|
"retention-in-memory": "48h",
|
|
"memory-cap": 100
|
|
},
|
|
"archive": {
|
|
"kind": "file",
|
|
"path": "./var/job-archive"
|
|
},
|
|
"auth": {
|
|
"jwts": {
|
|
"max-age": "2m"
|
|
}
|
|
}
|
|
}`
|
|
const testclusterJSON = `{
|
|
"name": "testcluster",
|
|
"subClusters": [
|
|
{
|
|
"name": "sc1",
|
|
"nodes": "host123,host124,host125",
|
|
"processorType": "Intel Core i7-4770",
|
|
"socketsPerNode": 1,
|
|
"coresPerSocket": 4,
|
|
"threadsPerCore": 2,
|
|
"flopRateScalar": {
|
|
"unit": {
|
|
"prefix": "G",
|
|
"base": "F/s"
|
|
},
|
|
"value": 14
|
|
},
|
|
"flopRateSimd": {
|
|
"unit": {
|
|
"prefix": "G",
|
|
"base": "F/s"
|
|
},
|
|
"value": 112
|
|
},
|
|
"memoryBandwidth": {
|
|
"unit": {
|
|
"prefix": "G",
|
|
"base": "B/s"
|
|
},
|
|
"value": 24
|
|
},
|
|
"numberOfNodes": 70,
|
|
"topology": {
|
|
"node": [0, 1, 2, 3, 4, 5, 6, 7],
|
|
"socket": [[0, 1, 2, 3, 4, 5, 6, 7]],
|
|
"memoryDomain": [[0, 1, 2, 3, 4, 5, 6, 7]],
|
|
"die": [[0, 1, 2, 3, 4, 5, 6, 7]],
|
|
"core": [[0], [1], [2], [3], [4], [5], [6], [7]]
|
|
}
|
|
}
|
|
],
|
|
"metricConfig": [
|
|
{
|
|
"name": "load_one",
|
|
"unit": { "base": ""},
|
|
"scope": "node",
|
|
"timestep": 60,
|
|
"aggregation": "avg",
|
|
"peak": 8,
|
|
"normal": 0,
|
|
"caution": 0,
|
|
"alert": 0
|
|
}
|
|
]
|
|
}`
|
|
|
|
cclog.Init("info", true)
|
|
tmpdir := t.TempDir()
|
|
jobarchive := filepath.Join(tmpdir, "job-archive")
|
|
if err := os.Mkdir(jobarchive, 0o777); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
if err := os.WriteFile(filepath.Join(jobarchive, "version.txt"), fmt.Appendf(nil, "%d", 3), 0o666); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
if err := os.Mkdir(filepath.Join(jobarchive, "testcluster"), 0o777); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
if err := os.WriteFile(filepath.Join(jobarchive, "testcluster", "cluster.json"), []byte(testclusterJSON), 0o666); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
dbfilepath := filepath.Join(tmpdir, "test.db")
|
|
err := repository.MigrateDB(dbfilepath)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
cfgFilePath := filepath.Join(tmpdir, "config.json")
|
|
if err := os.WriteFile(cfgFilePath, []byte(testconfig), 0o666); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
ccconf.Init(cfgFilePath)
|
|
metricstore.MetricStoreHandle = &metricstore.InternalMetricStore{}
|
|
|
|
// Load and check main configuration
|
|
if cfg := ccconf.GetPackageConfig("main"); cfg != nil {
|
|
config.Init(cfg)
|
|
} else {
|
|
cclog.Abort("Main configuration must be present")
|
|
}
|
|
archiveCfg := fmt.Sprintf("{\"kind\": \"file\",\"path\": \"%s\"}", jobarchive)
|
|
|
|
repository.Connect(dbfilepath)
|
|
|
|
if err := archive.Init(json.RawMessage(archiveCfg)); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
// metricstore initialization removed - it's initialized via callback in tests
|
|
|
|
archiver.Start(repository.GetJobRepository(), context.Background())
|
|
|
|
if cfg := ccconf.GetPackageConfig("auth"); cfg != nil {
|
|
auth.Init(&cfg)
|
|
} else {
|
|
cclog.Warn("Authentication disabled due to missing configuration")
|
|
auth.Init(nil)
|
|
}
|
|
|
|
graph.Init()
|
|
|
|
return api.New()
|
|
}
|
|
|
|
func cleanup() {
|
|
if err := archiver.Shutdown(5 * time.Second); err != nil {
|
|
cclog.Warnf("Archiver shutdown timeout in tests: %v", err)
|
|
}
|
|
}
|
|
|
|
/*
|
|
* This function starts a job, stops it, and then reads its data from the job-archive.
|
|
* Do not run sub-tests in parallel! Tests should not be run in parallel at all, because
|
|
* at least `setup` modifies global state.
|
|
*/
|
|
func TestRestApi(t *testing.T) {
|
|
restapi := setup(t)
|
|
t.Cleanup(cleanup)
|
|
testData := schema.JobData{Metrics: map[string]schema.ScopedMetrics{
|
|
"load_one": {
|
|
schema.MetricScopeNode: {
|
|
Unit: schema.Unit{Base: "load"},
|
|
Timestep: 60,
|
|
Series: []schema.Series{
|
|
{
|
|
Hostname: "host123",
|
|
Statistics: schema.MetricStatistics{Min: 0.1, Avg: 0.2, Max: 0.3},
|
|
Data: []schema.Float{0.1, 0.1, 0.1, 0.2, 0.2, 0.2, 0.3, 0.3, 0.3},
|
|
},
|
|
},
|
|
},
|
|
},
|
|
}}
|
|
|
|
metricstore.TestLoadDataCallback = func(job *schema.Job, metrics []string, scopes []schema.MetricScope, ctx context.Context, resolution int, resampleAlgo string) (schema.JobData, error) {
|
|
return testData, nil
|
|
}
|
|
|
|
r := chi.NewRouter()
|
|
restapi.MountAPIRoutes(r)
|
|
|
|
var TestJobID int64 = 123
|
|
TestClusterName := "testcluster"
|
|
var TestStartTime int64 = 123456789
|
|
|
|
const startJobBody string = `{
|
|
"jobId": 123,
|
|
"user": "testuser",
|
|
"project": "testproj",
|
|
"cluster": "testcluster",
|
|
"partition": "default",
|
|
"walltime": 3600,
|
|
"arrayJobId": 0,
|
|
"numNodes": 1,
|
|
"numHwthreads": 8,
|
|
"numAcc": 0,
|
|
"shared": "none",
|
|
"monitoringStatus": 1,
|
|
"smt": 1,
|
|
"resources": [
|
|
{
|
|
"hostname": "host123",
|
|
"hwthreads": [0, 1, 2, 3, 4, 5, 6, 7]
|
|
}
|
|
],
|
|
"metaData": { "jobScript": "blablabla..." },
|
|
"startTime": 123456789
|
|
}`
|
|
|
|
const contextUserKey repository.ContextKey = "user"
|
|
contextUserValue := &schema.User{
|
|
Username: "testuser",
|
|
Projects: make([]string, 0),
|
|
Roles: []string{"user"},
|
|
AuthType: 0,
|
|
AuthSource: 2,
|
|
}
|
|
|
|
if ok := t.Run("StartJob", func(t *testing.T) {
|
|
req := httptest.NewRequest(http.MethodPost, "/jobs/start_job/", bytes.NewBuffer([]byte(startJobBody)))
|
|
recorder := httptest.NewRecorder()
|
|
|
|
ctx := context.WithValue(req.Context(), contextUserKey, contextUserValue)
|
|
|
|
r.ServeHTTP(recorder, req.WithContext(ctx))
|
|
response := recorder.Result()
|
|
if response.StatusCode != http.StatusCreated {
|
|
t.Fatal(response.Status, recorder.Body.String())
|
|
}
|
|
// resolver := graph.GetResolverInstance()
|
|
restapi.JobRepository.SyncJobs()
|
|
job, err := restapi.JobRepository.Find(&TestJobID, &TestClusterName, &TestStartTime)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
// job.Tags, err = resolver.Job().Tags(ctx, job)
|
|
// if err != nil {
|
|
// t.Fatal(err)
|
|
// }
|
|
|
|
if job.JobID != 123 ||
|
|
job.User != "testuser" ||
|
|
job.Project != "testproj" ||
|
|
job.Cluster != "testcluster" ||
|
|
job.SubCluster != "sc1" ||
|
|
job.Partition != "default" ||
|
|
job.Walltime != 3600 ||
|
|
job.ArrayJobID != 0 ||
|
|
job.NumNodes != 1 ||
|
|
job.NumHWThreads != 8 ||
|
|
job.NumAcc != 0 ||
|
|
job.MonitoringStatus != 1 ||
|
|
job.SMT != 1 ||
|
|
!reflect.DeepEqual(job.Resources, []*schema.Resource{{Hostname: "host123", HWThreads: []int{0, 1, 2, 3, 4, 5, 6, 7}}}) ||
|
|
job.StartTime != 123456789 {
|
|
t.Fatalf("unexpected job properties: %#v", job)
|
|
}
|
|
|
|
// if len(job.Tags) != 1 || job.Tags[0].Type != "testTagType" || job.Tags[0].Name != "testTagName" || job.Tags[0].Scope != "testuser" {
|
|
// t.Fatalf("unexpected tags: %#v", job.Tags)
|
|
// }
|
|
}); !ok {
|
|
return
|
|
}
|
|
|
|
const stopJobBody string = `{
|
|
"jobId": 123,
|
|
"startTime": 123456789,
|
|
"cluster": "testcluster",
|
|
|
|
"jobState": "completed",
|
|
"stopTime": 123457789
|
|
}`
|
|
|
|
var stoppedJob *schema.Job
|
|
if ok := t.Run("StopJob", func(t *testing.T) {
|
|
req := httptest.NewRequest(http.MethodPost, "/jobs/stop_job/", bytes.NewBuffer([]byte(stopJobBody)))
|
|
recorder := httptest.NewRecorder()
|
|
|
|
ctx := context.WithValue(req.Context(), contextUserKey, contextUserValue)
|
|
|
|
r.ServeHTTP(recorder, req.WithContext(ctx))
|
|
response := recorder.Result()
|
|
if response.StatusCode != http.StatusOK {
|
|
t.Fatal(response.Status, recorder.Body.String())
|
|
}
|
|
|
|
// Archiving happens asynchronously, will be completed in cleanup
|
|
job, err := restapi.JobRepository.Find(&TestJobID, &TestClusterName, &TestStartTime)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
if job.State != schema.JobStateCompleted {
|
|
t.Fatal("expected job to be completed")
|
|
}
|
|
|
|
if job.Duration != (123457789 - 123456789) {
|
|
t.Fatalf("unexpected job properties: %#v", job)
|
|
}
|
|
|
|
job.MetaData, err = restapi.JobRepository.FetchMetadata(job)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
if !reflect.DeepEqual(job.MetaData, map[string]string{"jobScript": "blablabla..."}) {
|
|
t.Fatalf("unexpected job.metaData: %#v", job.MetaData)
|
|
}
|
|
|
|
stoppedJob = job
|
|
}); !ok {
|
|
return
|
|
}
|
|
|
|
t.Run("CheckArchive", func(t *testing.T) {
|
|
data, err := metricdispatch.LoadData(stoppedJob, []string{"load_one"}, []schema.MetricScope{schema.MetricScopeNode}, context.Background(), 60, "")
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
if !reflect.DeepEqual(data, testData) {
|
|
t.Fatal("unexpected data fetched from archive")
|
|
}
|
|
})
|
|
|
|
t.Run("CheckDoubleStart", func(t *testing.T) {
|
|
// Starting a job with the same jobId and cluster should only be allowed if the startTime is far appart!
|
|
body := strings.ReplaceAll(startJobBody, `"startTime": 123456789`, `"startTime": 123456790`)
|
|
|
|
req := httptest.NewRequest(http.MethodPost, "/jobs/start_job/", bytes.NewBuffer([]byte(body)))
|
|
recorder := httptest.NewRecorder()
|
|
|
|
ctx := context.WithValue(req.Context(), contextUserKey, contextUserValue)
|
|
|
|
r.ServeHTTP(recorder, req.WithContext(ctx))
|
|
response := recorder.Result()
|
|
if response.StatusCode != http.StatusUnprocessableEntity {
|
|
t.Fatal(response.Status, recorder.Body.String())
|
|
}
|
|
})
|
|
|
|
const startJobBodyFailed string = `{
|
|
"jobId": 12345,
|
|
"user": "testuser",
|
|
"project": "testproj",
|
|
"cluster": "testcluster",
|
|
"partition": "default",
|
|
"walltime": 3600,
|
|
"numNodes": 1,
|
|
"shared": "none",
|
|
"monitoringStatus": 1,
|
|
"smt": 1,
|
|
"resources": [
|
|
{
|
|
"hostname": "host123"
|
|
}
|
|
],
|
|
"startTime": 12345678
|
|
}`
|
|
|
|
ok := t.Run("StartJobFailed", func(t *testing.T) {
|
|
req := httptest.NewRequest(http.MethodPost, "/jobs/start_job/", bytes.NewBuffer([]byte(startJobBodyFailed)))
|
|
recorder := httptest.NewRecorder()
|
|
|
|
ctx := context.WithValue(req.Context(), contextUserKey, contextUserValue)
|
|
|
|
r.ServeHTTP(recorder, req.WithContext(ctx))
|
|
response := recorder.Result()
|
|
if response.StatusCode != http.StatusCreated {
|
|
t.Fatal(response.Status, recorder.Body.String())
|
|
}
|
|
})
|
|
if !ok {
|
|
t.Fatal("subtest failed")
|
|
}
|
|
|
|
time.Sleep(1 * time.Second)
|
|
restapi.JobRepository.SyncJobs()
|
|
|
|
const stopJobBodyFailed string = `{
|
|
"jobId": 12345,
|
|
"cluster": "testcluster",
|
|
|
|
"jobState": "failed",
|
|
"stopTime": 12355678
|
|
}`
|
|
|
|
ok = t.Run("StopJobFailed", func(t *testing.T) {
|
|
req := httptest.NewRequest(http.MethodPost, "/jobs/stop_job/", bytes.NewBuffer([]byte(stopJobBodyFailed)))
|
|
recorder := httptest.NewRecorder()
|
|
|
|
ctx := context.WithValue(req.Context(), contextUserKey, contextUserValue)
|
|
|
|
r.ServeHTTP(recorder, req.WithContext(ctx))
|
|
response := recorder.Result()
|
|
if response.StatusCode != http.StatusOK {
|
|
t.Fatal(response.Status, recorder.Body.String())
|
|
}
|
|
|
|
// Archiving happens asynchronously, will be completed in cleanup
|
|
jobid, cluster := int64(12345), "testcluster"
|
|
job, err := restapi.JobRepository.Find(&jobid, &cluster, nil)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
if job.State != schema.JobStateFailed {
|
|
t.Fatal("expected job to be failed")
|
|
}
|
|
})
|
|
if !ok {
|
|
t.Fatal("subtest failed")
|
|
}
|
|
|
|
t.Run("GetUsedNodesNoRunning", func(t *testing.T) {
|
|
contextUserValue := &schema.User{
|
|
Username: "testuser",
|
|
Projects: make([]string, 0),
|
|
Roles: []string{"api"},
|
|
AuthType: 0,
|
|
AuthSource: 2,
|
|
}
|
|
|
|
req := httptest.NewRequest(http.MethodGet, "/jobs/used_nodes?ts=123456790", nil)
|
|
recorder := httptest.NewRecorder()
|
|
|
|
ctx := context.WithValue(req.Context(), contextUserKey, contextUserValue)
|
|
|
|
r.ServeHTTP(recorder, req.WithContext(ctx))
|
|
response := recorder.Result()
|
|
if response.StatusCode != http.StatusOK {
|
|
t.Fatal(response.Status, recorder.Body.String())
|
|
}
|
|
|
|
var result api.GetUsedNodesAPIResponse
|
|
if err := json.NewDecoder(response.Body).Decode(&result); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
if result.UsedNodes == nil {
|
|
t.Fatal("expected usedNodes to be non-nil")
|
|
}
|
|
|
|
if len(result.UsedNodes) != 0 {
|
|
t.Fatalf("expected no used nodes for stopped jobs, got: %v", result.UsedNodes)
|
|
}
|
|
})
|
|
}
|
|
|
|
// TestStopJobWithReusedJobId verifies that stopping a recently started job works
|
|
// even when an older job with the same jobId exists in the job table (e.g. with
|
|
// state "failed"). This is a regression test for the bug where Find() on the job
|
|
// table would match the old job instead of the new one still in job_cache.
|
|
func TestStopJobWithReusedJobId(t *testing.T) {
|
|
restapi := setup(t)
|
|
t.Cleanup(cleanup)
|
|
|
|
testData := schema.JobData{Metrics: map[string]schema.ScopedMetrics{
|
|
"load_one": {
|
|
schema.MetricScopeNode: {
|
|
Unit: schema.Unit{Base: "load"},
|
|
Timestep: 60,
|
|
Series: []schema.Series{
|
|
{
|
|
Hostname: "host123",
|
|
Statistics: schema.MetricStatistics{Min: 0.1, Avg: 0.2, Max: 0.3},
|
|
Data: []schema.Float{0.1, 0.1, 0.1, 0.2, 0.2, 0.2, 0.3, 0.3, 0.3},
|
|
},
|
|
},
|
|
},
|
|
},
|
|
}}
|
|
|
|
metricstore.TestLoadDataCallback = func(job *schema.Job, metrics []string, scopes []schema.MetricScope, ctx context.Context, resolution int, resampleAlgo string) (schema.JobData, error) {
|
|
return testData, nil
|
|
}
|
|
|
|
r := chi.NewRouter()
|
|
restapi.MountAPIRoutes(r)
|
|
|
|
const contextUserKey repository.ContextKey = "user"
|
|
contextUserValue := &schema.User{
|
|
Username: "testuser",
|
|
Projects: make([]string, 0),
|
|
Roles: []string{"user"},
|
|
AuthType: 0,
|
|
AuthSource: 2,
|
|
}
|
|
|
|
// Step 1: Start the first job (jobId=999)
|
|
const startJobBody1 string = `{
|
|
"jobId": 999,
|
|
"user": "testuser",
|
|
"project": "testproj",
|
|
"cluster": "testcluster",
|
|
"partition": "default",
|
|
"walltime": 3600,
|
|
"numNodes": 1,
|
|
"numHwthreads": 8,
|
|
"numAcc": 0,
|
|
"shared": "none",
|
|
"monitoringStatus": 1,
|
|
"smt": 1,
|
|
"resources": [{"hostname": "host123", "hwthreads": [0, 1, 2, 3, 4, 5, 6, 7]}],
|
|
"startTime": 200000000
|
|
}`
|
|
|
|
if ok := t.Run("StartFirstJob", func(t *testing.T) {
|
|
req := httptest.NewRequest(http.MethodPost, "/jobs/start_job/", bytes.NewBuffer([]byte(startJobBody1)))
|
|
recorder := httptest.NewRecorder()
|
|
ctx := context.WithValue(req.Context(), contextUserKey, contextUserValue)
|
|
r.ServeHTTP(recorder, req.WithContext(ctx))
|
|
if recorder.Result().StatusCode != http.StatusCreated {
|
|
t.Fatal(recorder.Result().Status, recorder.Body.String())
|
|
}
|
|
}); !ok {
|
|
return
|
|
}
|
|
|
|
// Step 2: Sync to move job from cache to job table, then stop it as "failed"
|
|
time.Sleep(1 * time.Second)
|
|
restapi.JobRepository.SyncJobs()
|
|
|
|
const stopJobBody1 string = `{
|
|
"jobId": 999,
|
|
"startTime": 200000000,
|
|
"cluster": "testcluster",
|
|
"jobState": "failed",
|
|
"stopTime": 200001000
|
|
}`
|
|
|
|
if ok := t.Run("StopFirstJobAsFailed", func(t *testing.T) {
|
|
req := httptest.NewRequest(http.MethodPost, "/jobs/stop_job/", bytes.NewBuffer([]byte(stopJobBody1)))
|
|
recorder := httptest.NewRecorder()
|
|
ctx := context.WithValue(req.Context(), contextUserKey, contextUserValue)
|
|
r.ServeHTTP(recorder, req.WithContext(ctx))
|
|
if recorder.Result().StatusCode != http.StatusOK {
|
|
t.Fatal(recorder.Result().Status, recorder.Body.String())
|
|
}
|
|
|
|
jobid, cluster := int64(999), "testcluster"
|
|
job, err := restapi.JobRepository.Find(&jobid, &cluster, nil)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if job.State != schema.JobStateFailed {
|
|
t.Fatalf("expected first job to be failed, got: %s", job.State)
|
|
}
|
|
}); !ok {
|
|
return
|
|
}
|
|
|
|
// Wait for archiving to complete
|
|
time.Sleep(1 * time.Second)
|
|
|
|
// Step 3: Start a NEW job with the same jobId=999 but different startTime.
|
|
// This job will sit in job_cache (not yet synced).
|
|
const startJobBody2 string = `{
|
|
"jobId": 999,
|
|
"user": "testuser",
|
|
"project": "testproj",
|
|
"cluster": "testcluster",
|
|
"partition": "default",
|
|
"walltime": 3600,
|
|
"numNodes": 1,
|
|
"numHwthreads": 8,
|
|
"numAcc": 0,
|
|
"shared": "none",
|
|
"monitoringStatus": 1,
|
|
"smt": 1,
|
|
"resources": [{"hostname": "host123", "hwthreads": [0, 1, 2, 3, 4, 5, 6, 7]}],
|
|
"startTime": 300000000
|
|
}`
|
|
|
|
if ok := t.Run("StartSecondJob", func(t *testing.T) {
|
|
req := httptest.NewRequest(http.MethodPost, "/jobs/start_job/", bytes.NewBuffer([]byte(startJobBody2)))
|
|
recorder := httptest.NewRecorder()
|
|
ctx := context.WithValue(req.Context(), contextUserKey, contextUserValue)
|
|
r.ServeHTTP(recorder, req.WithContext(ctx))
|
|
if recorder.Result().StatusCode != http.StatusCreated {
|
|
t.Fatal(recorder.Result().Status, recorder.Body.String())
|
|
}
|
|
}); !ok {
|
|
return
|
|
}
|
|
|
|
// Step 4: Stop the second job WITHOUT syncing first.
|
|
// Before the fix, this would fail because Find() on the job table would
|
|
// match the old failed job (jobId=999) and reject with "already stopped".
|
|
const stopJobBody2 string = `{
|
|
"jobId": 999,
|
|
"startTime": 300000000,
|
|
"cluster": "testcluster",
|
|
"jobState": "completed",
|
|
"stopTime": 300001000
|
|
}`
|
|
|
|
t.Run("StopSecondJobBeforeSync", func(t *testing.T) {
|
|
req := httptest.NewRequest(http.MethodPost, "/jobs/stop_job/", bytes.NewBuffer([]byte(stopJobBody2)))
|
|
recorder := httptest.NewRecorder()
|
|
ctx := context.WithValue(req.Context(), contextUserKey, contextUserValue)
|
|
r.ServeHTTP(recorder, req.WithContext(ctx))
|
|
if recorder.Result().StatusCode != http.StatusOK {
|
|
t.Fatalf("expected stop to succeed for cached job, got: %s %s",
|
|
recorder.Result().Status, recorder.Body.String())
|
|
}
|
|
})
|
|
}
|