From d55bfcead8fc5b757096eb4f77f429909853a854 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Martin=20Sj=C3=B6lund?= Date: Tue, 6 Oct 2026 15:04:07 +0200 Subject: [PATCH] Stop createTables from hanging on table locks Jobs got stuck at 0% CPU right after fetching the reference files. On the shared PostgreSQL database every run kept the transaction of its first SELECT open until the results were written, holding a share lock on libversion, omcversion and the branch table for hours. The next run's createTables issued `ALTER TABLE ... ADD COLUMN IF NOT EXISTS` and `CREATE INDEX IF NOT EXISTS`, which take their table lock before checking whether there is anything to do, so it waited for those runs to finish - and every later reader of the table queued behind it. - createTables only alters or indexes what is actually missing, and sets a lock_timeout so a needed lock fails the run instead of hanging it. - test.py commits after choosing the libraries and after querying the expected execution times, so a run no longer locks the tables while testing. - Run test.py with PYTHONUNBUFFERED: stdbuf does not affect Python, so its output lagged behind git's and a hang looked like it was in git. Also print each library before loading it. - NeuralNetwork and URDFModelica reference files live on `main`; the reset to the default `origin/master` failed and left them stale. Assisted-by: Claude Opus 5.5 --- .CI/Jenkinsfile | 2 +- configs/conf.json | 6 ++++-- resultsdb.py | 27 ++++++++++++++++++--------- test.py | 5 +++++ 4 files changed, 28 insertions(+), 12 deletions(-) diff --git a/.CI/Jenkinsfile b/.CI/Jenkinsfile index 8d019b2..2ced32d 100644 --- a/.CI/Jenkinsfile +++ b/.CI/Jenkinsfile @@ -1286,7 +1286,7 @@ def runRegressiontest(branch, name, extraFlags, omsHash, omcompiler, extrasimfla ${cgroupReport} cd OpenModelicaLibraryTesting # Force /usr/bin/omc as being used for generating the mos-files. Ensures consistent behavior among all tested OMC versions - stdbuf -oL -eL time ./test.py --ompython_omhome=/usr ${FMI_TESTING_FLAG}${SOLVER_FLAG} --extraflags='${extraFlags}' --extrasimflags='${extrasimflags}' ${testFlags} --branch="${name}" --output="libraries.openmodelica.org:/var/www/libraries.openmodelica.org/branches/${name}/" --libraries='${libraryPath}/.openmodelica/libraries/' --jobs=${jobs} ${libs_config_file} ${params.OLDLIBS ? "configs/conf-old.json configs/conf-nonstandard.json" : ""} || (killall omc ; false) || exit 1 + PYTHONUNBUFFERED=1 stdbuf -oL -eL time ./test.py --ompython_omhome=/usr ${FMI_TESTING_FLAG}${SOLVER_FLAG} --extraflags='${extraFlags}' --extrasimflags='${extrasimflags}' ${testFlags} --branch="${name}" --output="libraries.openmodelica.org:/var/www/libraries.openmodelica.org/branches/${name}/" --libraries='${libraryPath}/.openmodelica/libraries/' --jobs=${jobs} ${libs_config_file} ${params.OLDLIBS ? "configs/conf-old.json configs/conf-nonstandard.json" : ""} || (killall omc ; false) || exit 1 """) sh 'date' // In the image: the script talks to the results database through psycopg2, diff --git a/configs/conf.json b/configs/conf.json index ca57f44..06d471a 100644 --- a/configs/conf.json +++ b/configs/conf.json @@ -441,7 +441,8 @@ "referenceFileNameDelimiter":".", "referenceFiles":{ "giturl":"https://github.com/AMIT-HSBI/NeuralNetwork", - "destination":"ReferenceFiles/NeuralNetwork" + "destination":"ReferenceFiles/NeuralNetwork", + "git-ref": "main" } }, { @@ -696,7 +697,8 @@ "referenceFileNameExtraName":"$ClassName", "referenceFiles":{ "giturl":"https://github.com/DLR-RM/urdfmodelica-referenceresults", - "destination":"ReferenceFiles/URDFModelica" + "destination":"ReferenceFiles/URDFModelica", + "git-ref": "main" } }, { diff --git a/resultsdb.py b/resultsdb.py index 7023931..ba4c24c 100644 --- a/resultsdb.py +++ b/resultsdb.py @@ -503,8 +503,12 @@ def createTables(self, branch): Unlike sqlite there is no schema migration and the indexes stay: other machines are reading the table while this one writes its few thousand rows. + + ALTER TABLE and CREATE INDEX lock the table even when there is nothing to + do, waiting behind every transaction reading it, so only run them if needed. """ cursor = self.cursor() + cursor.execute("SET LOCAL lock_timeout = '5min'") cursor.execute("""CREATE TABLE IF NOT EXISTS omcversion ( date bigint NOT NULL, branch text NOT NULL, omcversion text)""") cursor.execute("""CREATE TABLE IF NOT EXISTS libversion ( @@ -513,19 +517,24 @@ def createTables(self, branch): host text, sysinfo text)""") # The shared database predates the host columns; add them without touching # the rows already in there, which keep their results and read back NULL. + libversionColumns = self.columns("libversion") for col in ["host", "sysinfo"]: - cursor.execute("ALTER TABLE libversion ADD COLUMN IF NOT EXISTS %s text" % col) + if col not in libversionColumns: + cursor.execute("ALTER TABLE libversion ADD COLUMN IF NOT EXISTS %s text" % col) cols = ", ".join("%s %s%s" % (c, POSTGRES_TYPES[t], " NOT NULL" if c in BRANCH_KEY else "") for c, t in BRANCH_COLUMNS) cursor.execute("CREATE TABLE IF NOT EXISTS %s (%s)" % (self.quote(branch), cols)) - cursor.execute("ALTER TABLE %s ADD COLUMN IF NOT EXISTS maxrss bigint" % self.quote(branch)) - for tbl in ["omcversion", "libversion", branch]: - key = KEYS.get(tbl, BRANCH_KEY) - cursor.execute("CREATE UNIQUE INDEX IF NOT EXISTS %s ON %s (%s)" - % (self.quote(("uq_%s_%s" % (tbl, "_".join(key)))[:63]), - self.quote(tbl), ",".join(key))) - cursor.execute("CREATE INDEX IF NOT EXISTS %s ON %s (libname, date)" - % (self.quote(("idx_%s_libname_date" % branch)[:63]), self.quote(branch))) + if "maxrss" not in self.columns(branch): + cursor.execute("ALTER TABLE %s ADD COLUMN IF NOT EXISTS maxrss bigint" % self.quote(branch)) + indexes = [("UNIQUE INDEX", "uq_%s_%s" % (tbl, "_".join(KEYS.get(tbl, BRANCH_KEY))), tbl, + KEYS.get(tbl, BRANCH_KEY)) for tbl in ["omcversion", "libversion", branch]] + indexes.append(("INDEX", "idx_%s_libname_date" % branch, branch, ["libname", "date"])) + for (kind, name, tbl, key) in indexes: + name = name[:63] + if not self.execute("SELECT 1 FROM pg_indexes WHERE schemaname=current_schema() AND indexname=?", + (name,)).fetchone(): + cursor.execute("CREATE %s IF NOT EXISTS %s ON %s (%s)" + % (kind, self.quote(name), self.quote(tbl), ",".join(key))) self.commit() def tables(self): diff --git a/test.py b/test.py index d290365..3edb93f 100755 --- a/test.py +++ b/test.py @@ -800,6 +800,8 @@ def simulatorKey(libname, runner): conf["wasmfmurunners"] = [n for (n, _) in wasmfmurunners] if solvers: conf["solvers"] = [n for (n, _) in solvers] + print("Loading %s" % library) + sys.stdout.flush() if (not canChangeOptLevel) and "optlevel" in conf: print("Deleting optlevel") del conf["optlevel"] @@ -948,6 +950,8 @@ def simulatorKey(libname, runner): db.release() raise SystemExit("Failed to load: %s" % ", ".join(failedToLoad)) +# Do not keep the tables locked while testing; see createTables. +db.commit() print("Checked which libraries to run") sys.stdout.flush() @@ -1061,6 +1065,7 @@ def expectedExec(c): start=monotonic() tests=sorted(tests, key=lambda c: expectedExec(c), reverse=True) +db.commit() stop=monotonic() print("Querying expected execution time: %s" % friendlyStr(stop-start)) sys.stdout.flush()