mirror of
https://github.com/ClusterCockpit/cc-backend
synced 2026-07-27 00:37:14 +02:00
cc-lib changed JobData, ScopedJobStats and Job.Statistics from flat maps to structs (a .Metrics map plus array-valued Groups) to represent filesystem (and future interconnect) metric groups. Migrate all construction, indexing and iteration to the new API across the archive backends, metricstore query path, metric dispatcher, archiver, importer, tagger, taskmanager, repository and API layers. Semantics: DecodeJobStats now projects JobData.Groups into ScopedJobStats.Groups, and the archiver derives per-filesystem node-scope statistics into Job.Statistics.Groups. deepCopy, resampling and the metric/scope filter are group-aware. The metricstore internal storage (buffers, selector tree, checkpoint/parquet) is unchanged; all conversion stays at the LoadData/archive-codec seam. Note: requires the corresponding cc-lib release; bump the cc-lib dependency version once tagged (a local go.mod replace was used during development and is intentionally not committed). Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
651 lines
18 KiB
Go
651 lines
18 KiB
Go
// Copyright (C) NHR@FAU, University Erlangen-Nuremberg.
|
|
// All rights reserved. This file is part of cc-backend.
|
|
// Use of this source code is governed by a MIT-style
|
|
// license that can be found in the LICENSE file.
|
|
package api_test
|
|
|
|
import (
|
|
"bytes"
|
|
"context"
|
|
"encoding/json"
|
|
"fmt"
|
|
"net/http"
|
|
"net/http/httptest"
|
|
"os"
|
|
"path/filepath"
|
|
"reflect"
|
|
"strings"
|
|
"testing"
|
|
"time"
|
|
|
|
"github.com/ClusterCockpit/cc-backend/internal/api"
|
|
"github.com/ClusterCockpit/cc-backend/internal/archiver"
|
|
"github.com/ClusterCockpit/cc-backend/internal/auth"
|
|
"github.com/ClusterCockpit/cc-backend/internal/config"
|
|
"github.com/ClusterCockpit/cc-backend/internal/graph"
|
|
"github.com/ClusterCockpit/cc-backend/internal/metricdispatch"
|
|
"github.com/ClusterCockpit/cc-backend/internal/repository"
|
|
"github.com/ClusterCockpit/cc-backend/pkg/archive"
|
|
"github.com/ClusterCockpit/cc-backend/pkg/metricstore"
|
|
ccconf "github.com/ClusterCockpit/cc-lib/v2/ccConfig"
|
|
cclog "github.com/ClusterCockpit/cc-lib/v2/ccLogger"
|
|
"github.com/ClusterCockpit/cc-lib/v2/schema"
|
|
"github.com/go-chi/chi/v5"
|
|
|
|
_ "github.com/mattn/go-sqlite3"
|
|
)
|
|
|
|
func setup(t *testing.T) *api.RestAPI {
|
|
repository.ResetConnection()
|
|
|
|
const testconfig = `{
|
|
"main": {
|
|
"addr": "0.0.0.0:8080",
|
|
"validate": false,
|
|
"api-allowed-ips": [
|
|
"*"
|
|
]
|
|
},
|
|
"metric-store": {
|
|
"checkpoints": {
|
|
"interval": "12h"
|
|
},
|
|
"retention-in-memory": "48h",
|
|
"memory-cap": 100
|
|
},
|
|
"archive": {
|
|
"kind": "file",
|
|
"path": "./var/job-archive"
|
|
},
|
|
"auth": {
|
|
"jwts": {
|
|
"max-age": "2m"
|
|
}
|
|
}
|
|
}`
|
|
const testclusterJSON = `{
|
|
"name": "testcluster",
|
|
"subClusters": [
|
|
{
|
|
"name": "sc1",
|
|
"nodes": "host123,host124,host125",
|
|
"processorType": "Intel Core i7-4770",
|
|
"socketsPerNode": 1,
|
|
"coresPerSocket": 4,
|
|
"threadsPerCore": 2,
|
|
"flopRateScalar": {
|
|
"unit": {
|
|
"prefix": "G",
|
|
"base": "F/s"
|
|
},
|
|
"value": 14
|
|
},
|
|
"flopRateSimd": {
|
|
"unit": {
|
|
"prefix": "G",
|
|
"base": "F/s"
|
|
},
|
|
"value": 112
|
|
},
|
|
"memoryBandwidth": {
|
|
"unit": {
|
|
"prefix": "G",
|
|
"base": "B/s"
|
|
},
|
|
"value": 24
|
|
},
|
|
"numberOfNodes": 70,
|
|
"topology": {
|
|
"node": [0, 1, 2, 3, 4, 5, 6, 7],
|
|
"socket": [[0, 1, 2, 3, 4, 5, 6, 7]],
|
|
"memoryDomain": [[0, 1, 2, 3, 4, 5, 6, 7]],
|
|
"die": [[0, 1, 2, 3, 4, 5, 6, 7]],
|
|
"core": [[0], [1], [2], [3], [4], [5], [6], [7]]
|
|
}
|
|
}
|
|
],
|
|
"metricConfig": [
|
|
{
|
|
"name": "load_one",
|
|
"unit": { "base": ""},
|
|
"scope": "node",
|
|
"timestep": 60,
|
|
"aggregation": "avg",
|
|
"peak": 8,
|
|
"normal": 0,
|
|
"caution": 0,
|
|
"alert": 0
|
|
}
|
|
]
|
|
}`
|
|
|
|
cclog.Init("info", true)
|
|
tmpdir := t.TempDir()
|
|
jobarchive := filepath.Join(tmpdir, "job-archive")
|
|
if err := os.Mkdir(jobarchive, 0o777); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
if err := os.WriteFile(filepath.Join(jobarchive, "version.txt"), fmt.Appendf(nil, "%d", 3), 0o666); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
if err := os.Mkdir(filepath.Join(jobarchive, "testcluster"), 0o777); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
if err := os.WriteFile(filepath.Join(jobarchive, "testcluster", "cluster.json"), []byte(testclusterJSON), 0o666); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
dbfilepath := filepath.Join(tmpdir, "test.db")
|
|
err := repository.MigrateDB(dbfilepath)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
cfgFilePath := filepath.Join(tmpdir, "config.json")
|
|
if err := os.WriteFile(cfgFilePath, []byte(testconfig), 0o666); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
ccconf.Init(cfgFilePath)
|
|
metricstore.MetricStoreHandle = &metricstore.InternalMetricStore{}
|
|
|
|
// Load and check main configuration
|
|
if cfg := ccconf.GetPackageConfig("main"); cfg != nil {
|
|
config.Init(cfg)
|
|
} else {
|
|
cclog.Abort("Main configuration must be present")
|
|
}
|
|
archiveCfg := fmt.Sprintf("{\"kind\": \"file\",\"path\": \"%s\"}", jobarchive)
|
|
|
|
repository.Connect(dbfilepath)
|
|
|
|
if err := archive.Init(json.RawMessage(archiveCfg)); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
// metricstore initialization removed - it's initialized via callback in tests
|
|
|
|
archiver.Start(repository.GetJobRepository(), context.Background())
|
|
|
|
if cfg := ccconf.GetPackageConfig("auth"); cfg != nil {
|
|
auth.Init(&cfg)
|
|
} else {
|
|
cclog.Warn("Authentication disabled due to missing configuration")
|
|
auth.Init(nil)
|
|
}
|
|
|
|
graph.Init()
|
|
|
|
return api.New()
|
|
}
|
|
|
|
func cleanup() {
|
|
if err := archiver.Shutdown(5 * time.Second); err != nil {
|
|
cclog.Warnf("Archiver shutdown timeout in tests: %v", err)
|
|
}
|
|
}
|
|
|
|
/*
|
|
* This function starts a job, stops it, and then reads its data from the job-archive.
|
|
* Do not run sub-tests in parallel! Tests should not be run in parallel at all, because
|
|
* at least `setup` modifies global state.
|
|
*/
|
|
func TestRestApi(t *testing.T) {
|
|
restapi := setup(t)
|
|
t.Cleanup(cleanup)
|
|
testData := schema.JobData{Metrics: map[string]schema.ScopedMetrics{
|
|
"load_one": map[schema.MetricScope]*schema.JobMetric{
|
|
schema.MetricScopeNode: {
|
|
Unit: schema.Unit{Base: "load"},
|
|
Timestep: 60,
|
|
Series: []schema.Series{
|
|
{
|
|
Hostname: "host123",
|
|
Statistics: schema.MetricStatistics{Min: 0.1, Avg: 0.2, Max: 0.3},
|
|
Data: []schema.Float{0.1, 0.1, 0.1, 0.2, 0.2, 0.2, 0.3, 0.3, 0.3},
|
|
},
|
|
},
|
|
},
|
|
},
|
|
}}
|
|
|
|
metricstore.TestLoadDataCallback = func(job *schema.Job, metrics []string, scopes []schema.MetricScope, ctx context.Context, resolution int) (schema.JobData, error) {
|
|
return testData, nil
|
|
}
|
|
|
|
r := chi.NewRouter()
|
|
restapi.MountAPIRoutes(r)
|
|
|
|
var TestJobID int64 = 123
|
|
TestClusterName := "testcluster"
|
|
var TestStartTime int64 = 123456789
|
|
|
|
const startJobBody string = `{
|
|
"jobId": 123,
|
|
"user": "testuser",
|
|
"project": "testproj",
|
|
"cluster": "testcluster",
|
|
"partition": "default",
|
|
"walltime": 3600,
|
|
"arrayJobId": 0,
|
|
"numNodes": 1,
|
|
"numHwthreads": 8,
|
|
"numAcc": 0,
|
|
"shared": "none",
|
|
"monitoringStatus": 1,
|
|
"smt": 1,
|
|
"resources": [
|
|
{
|
|
"hostname": "host123",
|
|
"hwthreads": [0, 1, 2, 3, 4, 5, 6, 7]
|
|
}
|
|
],
|
|
"metaData": { "jobScript": "blablabla..." },
|
|
"startTime": 123456789
|
|
}`
|
|
|
|
const contextUserKey repository.ContextKey = "user"
|
|
contextUserValue := &schema.User{
|
|
Username: "testuser",
|
|
Projects: make([]string, 0),
|
|
Roles: []string{"user"},
|
|
AuthType: 0,
|
|
AuthSource: 2,
|
|
}
|
|
|
|
if ok := t.Run("StartJob", func(t *testing.T) {
|
|
req := httptest.NewRequest(http.MethodPost, "/jobs/start_job/", bytes.NewBuffer([]byte(startJobBody)))
|
|
recorder := httptest.NewRecorder()
|
|
|
|
ctx := context.WithValue(req.Context(), contextUserKey, contextUserValue)
|
|
|
|
r.ServeHTTP(recorder, req.WithContext(ctx))
|
|
response := recorder.Result()
|
|
if response.StatusCode != http.StatusCreated {
|
|
t.Fatal(response.Status, recorder.Body.String())
|
|
}
|
|
// resolver := graph.GetResolverInstance()
|
|
restapi.JobRepository.SyncJobs()
|
|
job, err := restapi.JobRepository.Find(&TestJobID, &TestClusterName, &TestStartTime)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
// job.Tags, err = resolver.Job().Tags(ctx, job)
|
|
// if err != nil {
|
|
// t.Fatal(err)
|
|
// }
|
|
|
|
if job.JobID != 123 ||
|
|
job.User != "testuser" ||
|
|
job.Project != "testproj" ||
|
|
job.Cluster != "testcluster" ||
|
|
job.SubCluster != "sc1" ||
|
|
job.Partition != "default" ||
|
|
job.Walltime != 3600 ||
|
|
job.ArrayJobID != 0 ||
|
|
job.NumNodes != 1 ||
|
|
job.NumHWThreads != 8 ||
|
|
job.NumAcc != 0 ||
|
|
job.MonitoringStatus != 1 ||
|
|
job.SMT != 1 ||
|
|
!reflect.DeepEqual(job.Resources, []*schema.Resource{{Hostname: "host123", HWThreads: []int{0, 1, 2, 3, 4, 5, 6, 7}}}) ||
|
|
job.StartTime != 123456789 {
|
|
t.Fatalf("unexpected job properties: %#v", job)
|
|
}
|
|
|
|
// if len(job.Tags) != 1 || job.Tags[0].Type != "testTagType" || job.Tags[0].Name != "testTagName" || job.Tags[0].Scope != "testuser" {
|
|
// t.Fatalf("unexpected tags: %#v", job.Tags)
|
|
// }
|
|
}); !ok {
|
|
return
|
|
}
|
|
|
|
const stopJobBody string = `{
|
|
"jobId": 123,
|
|
"startTime": 123456789,
|
|
"cluster": "testcluster",
|
|
|
|
"jobState": "completed",
|
|
"stopTime": 123457789
|
|
}`
|
|
|
|
var stoppedJob *schema.Job
|
|
if ok := t.Run("StopJob", func(t *testing.T) {
|
|
req := httptest.NewRequest(http.MethodPost, "/jobs/stop_job/", bytes.NewBuffer([]byte(stopJobBody)))
|
|
recorder := httptest.NewRecorder()
|
|
|
|
ctx := context.WithValue(req.Context(), contextUserKey, contextUserValue)
|
|
|
|
r.ServeHTTP(recorder, req.WithContext(ctx))
|
|
response := recorder.Result()
|
|
if response.StatusCode != http.StatusOK {
|
|
t.Fatal(response.Status, recorder.Body.String())
|
|
}
|
|
|
|
// Archiving happens asynchronously, will be completed in cleanup
|
|
job, err := restapi.JobRepository.Find(&TestJobID, &TestClusterName, &TestStartTime)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
if job.State != schema.JobStateCompleted {
|
|
t.Fatal("expected job to be completed")
|
|
}
|
|
|
|
if job.Duration != (123457789 - 123456789) {
|
|
t.Fatalf("unexpected job properties: %#v", job)
|
|
}
|
|
|
|
job.MetaData, err = restapi.JobRepository.FetchMetadata(job)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
if !reflect.DeepEqual(job.MetaData, map[string]string{"jobScript": "blablabla..."}) {
|
|
t.Fatalf("unexpected job.metaData: %#v", job.MetaData)
|
|
}
|
|
|
|
stoppedJob = job
|
|
}); !ok {
|
|
return
|
|
}
|
|
|
|
t.Run("CheckArchive", func(t *testing.T) {
|
|
data, err := metricdispatch.LoadData(stoppedJob, []string{"load_one"}, []schema.MetricScope{schema.MetricScopeNode}, context.Background(), 60)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
if !reflect.DeepEqual(data, testData) {
|
|
t.Fatal("unexpected data fetched from archive")
|
|
}
|
|
})
|
|
|
|
t.Run("CheckDoubleStart", func(t *testing.T) {
|
|
// Starting a job with the same jobId and cluster should only be allowed if the startTime is far appart!
|
|
body := strings.ReplaceAll(startJobBody, `"startTime": 123456789`, `"startTime": 123456790`)
|
|
|
|
req := httptest.NewRequest(http.MethodPost, "/jobs/start_job/", bytes.NewBuffer([]byte(body)))
|
|
recorder := httptest.NewRecorder()
|
|
|
|
ctx := context.WithValue(req.Context(), contextUserKey, contextUserValue)
|
|
|
|
r.ServeHTTP(recorder, req.WithContext(ctx))
|
|
response := recorder.Result()
|
|
if response.StatusCode != http.StatusUnprocessableEntity {
|
|
t.Fatal(response.Status, recorder.Body.String())
|
|
}
|
|
})
|
|
|
|
const startJobBodyFailed string = `{
|
|
"jobId": 12345,
|
|
"user": "testuser",
|
|
"project": "testproj",
|
|
"cluster": "testcluster",
|
|
"partition": "default",
|
|
"walltime": 3600,
|
|
"numNodes": 1,
|
|
"shared": "none",
|
|
"monitoringStatus": 1,
|
|
"smt": 1,
|
|
"resources": [
|
|
{
|
|
"hostname": "host123"
|
|
}
|
|
],
|
|
"startTime": 12345678
|
|
}`
|
|
|
|
ok := t.Run("StartJobFailed", func(t *testing.T) {
|
|
req := httptest.NewRequest(http.MethodPost, "/jobs/start_job/", bytes.NewBuffer([]byte(startJobBodyFailed)))
|
|
recorder := httptest.NewRecorder()
|
|
|
|
ctx := context.WithValue(req.Context(), contextUserKey, contextUserValue)
|
|
|
|
r.ServeHTTP(recorder, req.WithContext(ctx))
|
|
response := recorder.Result()
|
|
if response.StatusCode != http.StatusCreated {
|
|
t.Fatal(response.Status, recorder.Body.String())
|
|
}
|
|
})
|
|
if !ok {
|
|
t.Fatal("subtest failed")
|
|
}
|
|
|
|
time.Sleep(1 * time.Second)
|
|
restapi.JobRepository.SyncJobs()
|
|
|
|
const stopJobBodyFailed string = `{
|
|
"jobId": 12345,
|
|
"cluster": "testcluster",
|
|
|
|
"jobState": "failed",
|
|
"stopTime": 12355678
|
|
}`
|
|
|
|
ok = t.Run("StopJobFailed", func(t *testing.T) {
|
|
req := httptest.NewRequest(http.MethodPost, "/jobs/stop_job/", bytes.NewBuffer([]byte(stopJobBodyFailed)))
|
|
recorder := httptest.NewRecorder()
|
|
|
|
ctx := context.WithValue(req.Context(), contextUserKey, contextUserValue)
|
|
|
|
r.ServeHTTP(recorder, req.WithContext(ctx))
|
|
response := recorder.Result()
|
|
if response.StatusCode != http.StatusOK {
|
|
t.Fatal(response.Status, recorder.Body.String())
|
|
}
|
|
|
|
// Archiving happens asynchronously, will be completed in cleanup
|
|
jobid, cluster := int64(12345), "testcluster"
|
|
job, err := restapi.JobRepository.Find(&jobid, &cluster, nil)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
if job.State != schema.JobStateFailed {
|
|
t.Fatal("expected job to be failed")
|
|
}
|
|
})
|
|
if !ok {
|
|
t.Fatal("subtest failed")
|
|
}
|
|
|
|
t.Run("GetUsedNodesNoRunning", func(t *testing.T) {
|
|
contextUserValue := &schema.User{
|
|
Username: "testuser",
|
|
Projects: make([]string, 0),
|
|
Roles: []string{"api"},
|
|
AuthType: 0,
|
|
AuthSource: 2,
|
|
}
|
|
|
|
req := httptest.NewRequest(http.MethodGet, "/jobs/used_nodes?ts=123456790", nil)
|
|
recorder := httptest.NewRecorder()
|
|
|
|
ctx := context.WithValue(req.Context(), contextUserKey, contextUserValue)
|
|
|
|
r.ServeHTTP(recorder, req.WithContext(ctx))
|
|
response := recorder.Result()
|
|
if response.StatusCode != http.StatusOK {
|
|
t.Fatal(response.Status, recorder.Body.String())
|
|
}
|
|
|
|
var result api.GetUsedNodesAPIResponse
|
|
if err := json.NewDecoder(response.Body).Decode(&result); err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
|
|
if result.UsedNodes == nil {
|
|
t.Fatal("expected usedNodes to be non-nil")
|
|
}
|
|
|
|
if len(result.UsedNodes) != 0 {
|
|
t.Fatalf("expected no used nodes for stopped jobs, got: %v", result.UsedNodes)
|
|
}
|
|
})
|
|
}
|
|
|
|
// TestStopJobWithReusedJobId verifies that stopping a recently started job works
|
|
// even when an older job with the same jobId exists in the job table (e.g. with
|
|
// state "failed"). This is a regression test for the bug where Find() on the job
|
|
// table would match the old job instead of the new one still in job_cache.
|
|
func TestStopJobWithReusedJobId(t *testing.T) {
|
|
restapi := setup(t)
|
|
t.Cleanup(cleanup)
|
|
|
|
testData := schema.JobData{Metrics: map[string]schema.ScopedMetrics{
|
|
"load_one": map[schema.MetricScope]*schema.JobMetric{
|
|
schema.MetricScopeNode: {
|
|
Unit: schema.Unit{Base: "load"},
|
|
Timestep: 60,
|
|
Series: []schema.Series{
|
|
{
|
|
Hostname: "host123",
|
|
Statistics: schema.MetricStatistics{Min: 0.1, Avg: 0.2, Max: 0.3},
|
|
Data: []schema.Float{0.1, 0.1, 0.1, 0.2, 0.2, 0.2, 0.3, 0.3, 0.3},
|
|
},
|
|
},
|
|
},
|
|
},
|
|
}}
|
|
|
|
metricstore.TestLoadDataCallback = func(job *schema.Job, metrics []string, scopes []schema.MetricScope, ctx context.Context, resolution int) (schema.JobData, error) {
|
|
return testData, nil
|
|
}
|
|
|
|
r := chi.NewRouter()
|
|
restapi.MountAPIRoutes(r)
|
|
|
|
const contextUserKey repository.ContextKey = "user"
|
|
contextUserValue := &schema.User{
|
|
Username: "testuser",
|
|
Projects: make([]string, 0),
|
|
Roles: []string{"user"},
|
|
AuthType: 0,
|
|
AuthSource: 2,
|
|
}
|
|
|
|
// Step 1: Start the first job (jobId=999)
|
|
const startJobBody1 string = `{
|
|
"jobId": 999,
|
|
"user": "testuser",
|
|
"project": "testproj",
|
|
"cluster": "testcluster",
|
|
"partition": "default",
|
|
"walltime": 3600,
|
|
"numNodes": 1,
|
|
"numHwthreads": 8,
|
|
"numAcc": 0,
|
|
"shared": "none",
|
|
"monitoringStatus": 1,
|
|
"smt": 1,
|
|
"resources": [{"hostname": "host123", "hwthreads": [0, 1, 2, 3, 4, 5, 6, 7]}],
|
|
"startTime": 200000000
|
|
}`
|
|
|
|
if ok := t.Run("StartFirstJob", func(t *testing.T) {
|
|
req := httptest.NewRequest(http.MethodPost, "/jobs/start_job/", bytes.NewBuffer([]byte(startJobBody1)))
|
|
recorder := httptest.NewRecorder()
|
|
ctx := context.WithValue(req.Context(), contextUserKey, contextUserValue)
|
|
r.ServeHTTP(recorder, req.WithContext(ctx))
|
|
if recorder.Result().StatusCode != http.StatusCreated {
|
|
t.Fatal(recorder.Result().Status, recorder.Body.String())
|
|
}
|
|
}); !ok {
|
|
return
|
|
}
|
|
|
|
// Step 2: Sync to move job from cache to job table, then stop it as "failed"
|
|
time.Sleep(1 * time.Second)
|
|
restapi.JobRepository.SyncJobs()
|
|
|
|
const stopJobBody1 string = `{
|
|
"jobId": 999,
|
|
"startTime": 200000000,
|
|
"cluster": "testcluster",
|
|
"jobState": "failed",
|
|
"stopTime": 200001000
|
|
}`
|
|
|
|
if ok := t.Run("StopFirstJobAsFailed", func(t *testing.T) {
|
|
req := httptest.NewRequest(http.MethodPost, "/jobs/stop_job/", bytes.NewBuffer([]byte(stopJobBody1)))
|
|
recorder := httptest.NewRecorder()
|
|
ctx := context.WithValue(req.Context(), contextUserKey, contextUserValue)
|
|
r.ServeHTTP(recorder, req.WithContext(ctx))
|
|
if recorder.Result().StatusCode != http.StatusOK {
|
|
t.Fatal(recorder.Result().Status, recorder.Body.String())
|
|
}
|
|
|
|
jobid, cluster := int64(999), "testcluster"
|
|
job, err := restapi.JobRepository.Find(&jobid, &cluster, nil)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
if job.State != schema.JobStateFailed {
|
|
t.Fatalf("expected first job to be failed, got: %s", job.State)
|
|
}
|
|
}); !ok {
|
|
return
|
|
}
|
|
|
|
// Wait for archiving to complete
|
|
time.Sleep(1 * time.Second)
|
|
|
|
// Step 3: Start a NEW job with the same jobId=999 but different startTime.
|
|
// This job will sit in job_cache (not yet synced).
|
|
const startJobBody2 string = `{
|
|
"jobId": 999,
|
|
"user": "testuser",
|
|
"project": "testproj",
|
|
"cluster": "testcluster",
|
|
"partition": "default",
|
|
"walltime": 3600,
|
|
"numNodes": 1,
|
|
"numHwthreads": 8,
|
|
"numAcc": 0,
|
|
"shared": "none",
|
|
"monitoringStatus": 1,
|
|
"smt": 1,
|
|
"resources": [{"hostname": "host123", "hwthreads": [0, 1, 2, 3, 4, 5, 6, 7]}],
|
|
"startTime": 300000000
|
|
}`
|
|
|
|
if ok := t.Run("StartSecondJob", func(t *testing.T) {
|
|
req := httptest.NewRequest(http.MethodPost, "/jobs/start_job/", bytes.NewBuffer([]byte(startJobBody2)))
|
|
recorder := httptest.NewRecorder()
|
|
ctx := context.WithValue(req.Context(), contextUserKey, contextUserValue)
|
|
r.ServeHTTP(recorder, req.WithContext(ctx))
|
|
if recorder.Result().StatusCode != http.StatusCreated {
|
|
t.Fatal(recorder.Result().Status, recorder.Body.String())
|
|
}
|
|
}); !ok {
|
|
return
|
|
}
|
|
|
|
// Step 4: Stop the second job WITHOUT syncing first.
|
|
// Before the fix, this would fail because Find() on the job table would
|
|
// match the old failed job (jobId=999) and reject with "already stopped".
|
|
const stopJobBody2 string = `{
|
|
"jobId": 999,
|
|
"startTime": 300000000,
|
|
"cluster": "testcluster",
|
|
"jobState": "completed",
|
|
"stopTime": 300001000
|
|
}`
|
|
|
|
t.Run("StopSecondJobBeforeSync", func(t *testing.T) {
|
|
req := httptest.NewRequest(http.MethodPost, "/jobs/stop_job/", bytes.NewBuffer([]byte(stopJobBody2)))
|
|
recorder := httptest.NewRecorder()
|
|
ctx := context.WithValue(req.Context(), contextUserKey, contextUserValue)
|
|
r.ServeHTTP(recorder, req.WithContext(ctx))
|
|
if recorder.Result().StatusCode != http.StatusOK {
|
|
t.Fatalf("expected stop to succeed for cached job, got: %s %s",
|
|
recorder.Result().Status, recorder.Body.String())
|
|
}
|
|
})
|
|
}
|