diff --git a/configs/heavy-models.json b/configs/heavy-models.json new file mode 100644 index 0000000..fb27443 --- /dev/null +++ b/configs/heavy-models.json @@ -0,0 +1,178 @@ +{ + "BuildingSystems": { + "BuildingSystems.Applications.DistrictSimulation.DistrictBerlinKreuzberg": 7.01, + "BuildingSystems.Applications.DistrictSimulation.HCBC_DHN": 7.01, + "BuildingSystems.HAM.HeatConduction.Examples.HeatConduction1DArray": 6.97 + }, + "Buildings_11": { + "Buildings.Controls.DemandResponse.Examples.ClientLBNL90": 6.51, + "Buildings.DHC.Examples.Combined.SeriesConstantFlow": 4.7, + "Buildings.DHC.Examples.Combined.SeriesVariableFlow": 4.71 + }, + "Buildings_12": { + "Buildings.Controls.DemandResponse.Examples.ClientLBNL90": 6.09, + "Buildings.DHC.Examples.Combined.SeriesConstantFlow": 4.65, + "Buildings.DHC.Examples.Combined.SeriesVariableFlow": 4.66 + }, + "Buildings_latest": { + "Buildings.Controls.DemandResponse.Examples.ClientLBNL90": 6.21, + "Buildings.DHC.Examples.Combined.SeriesConstantFlow": 4.73, + "Buildings.DHC.Examples.Combined.SeriesVariableFlow": 4.75, + "Buildings.DHC.Plants.Combined.Examples.AllElectricCWStorage": 4.03, + "Buildings.Fluid.HeatPumps.ModularReversible.Validation.TableData2DLoadDepSHC": 7.86 + }, + "ClaRa": { + "ClaRa.Basics.ControlVolumes.SolidVolumes.Check.Validation_NTUparallel_DiscrPipes": 4.9, + "ClaRa.Examples.SteamPowerPlant_01": 4.79, + "ClaRa.Examples.SteamPowerPlant_CombinedComponents_01": 5.65 + }, + "ClaRa_dev": { + "ClaRa.Basics.ControlVolumes.SolidVolumes.Check.Validation_NTUparallel_DiscrPipes": 8.23, + "ClaRa.Components.Furnace.Check.Test_burner_adiabatic_fuelDrying": 5.44, + "ClaRa.Components.HeatExchangers.Check.Test_RegenerativeAirPreheater": 6.23, + "ClaRa.Components.Mills.PhysicalMills.Check.TestMillBox_1": 5.28, + "ClaRa.Components.Mills.PhysicalMills.Check.TestMillBox_1_measurementInput": 5.04, + "ClaRa.Components.Mills.PhysicalMills.Check.TestMillBox_2": 4.41, + "ClaRa.Components.VolumesValvesFittings.Valves.Check.Test_GasValves": 4.84, + "ClaRa.Examples.SteamPowerPlant_01": 5.04, + "ClaRa.Examples.SteamPowerPlant_CombinedComponents_01": 5.79 + }, + "Greenhouses": { + "Greenhouses.Examples.GlobalSystem_1": 4.29 + }, + "IDEAS": { + "IDEAS.Buildings.Components.Examples.AirflowBoxModel": 7.01, + "IDEAS.Buildings.Components.Examples.FacadeShadeExample": 4.66, + "IDEAS.Buildings.Components.Examples.LightingControl": 7.01, + "IDEAS.Buildings.Components.Examples.NumberOccupants": 7.01, + "IDEAS.Buildings.Components.Examples.RectangularZone": 4.66, + "IDEAS.Buildings.Components.Examples.RectangularZoneEmbedded": 4.64, + "IDEAS.Buildings.Components.Examples.RectangularZoneExternalSurfaces": 4.64, + "IDEAS.Buildings.Components.Examples.RectangularZoneRedeclarationWindows": 4.51, + "IDEAS.Buildings.Components.Examples.RectangularZoneTemplateFloor": 4.68, + "IDEAS.Buildings.Components.Examples.ScalingWindow": 6.97, + "IDEAS.Buildings.Components.Examples.TwoStoreyBoxes": 4.61, + "IDEAS.Buildings.Components.InterzonalAirFlow.Examples.InterzonalAirFlow": 7.01, + "IDEAS.Buildings.Examples.OpenDoorComparison": 7.01, + "IDEAS.Buildings.Examples.ScreenComparison": 4.59, + "IDEAS.Buildings.Validation.BESTEST": 6.2, + "IDEAS.Examples.DetailedResidentialNoHeating": 7.01, + "IDEAS.Examples.PPD12.Heating": 7.01, + "IDEAS.Examples.PPD12.Structure": 7.01, + "IDEAS.Examples.PPD12.VentilationMPC": 7.01, + "IDEAS.Examples.PPD12.VentilationRBC": 7.01, + "IDEAS.Examples.Tutorial.DetailedHouse.DetailedHouse10": 4.65, + "IDEAS.Examples.Tutorial.DetailedHouse.DetailedHouse5": 4.54, + "IDEAS.Examples.Tutorial.DetailedHouse.DetailedHouse6": 4.74, + "IDEAS.Examples.Tutorial.DetailedHouse.DetailedHouse7": 4.75, + "IDEAS.Examples.Tutorial.DetailedHouse.DetailedHouse8": 4.74, + "IDEAS.Examples.Tutorial.DetailedHouse.DetailedHouse9": 4.73, + "IDEAS.Examples.TwinHouses.BuildingN2_Exp1": 7.01, + "IDEAS.Examples.TwinHouses.BuildingN2_Exp2": 7.01, + "IDEAS.Examples.TwinHouses.BuildingN2_Exp2_Tset": 7.01, + "IDEAS.Examples.TwinHouses.BuildingO5_Exp1_1Port": 7.01, + "IDEAS.Examples.TwinHouses.BuildingO5_Exp1_2Port": 7.01, + "IDEAS.Fluid.HeatExchangers.RadiantSlab.Examples.EmbeddedPipeNDiscr": 6.86 + }, + "IDEAS_dev": { + "IDEAS.Buildings.Components.Examples.AirflowBoxModel": 7.01, + "IDEAS.Buildings.Components.Examples.FacadeShadeExample": 4.66, + "IDEAS.Buildings.Components.Examples.LightingControl": 7.01, + "IDEAS.Buildings.Components.Examples.NumberOccupants": 7.01, + "IDEAS.Buildings.Components.Examples.RectangularZone": 4.66, + "IDEAS.Buildings.Components.Examples.RectangularZoneEmbedded": 4.64, + "IDEAS.Buildings.Components.Examples.RectangularZoneExternalSurfaces": 4.64, + "IDEAS.Buildings.Components.Examples.RectangularZoneRedeclarationWindows": 4.51, + "IDEAS.Buildings.Components.Examples.RectangularZoneTemplateFloor": 4.61, + "IDEAS.Buildings.Components.Examples.ScalingWindow": 6.96, + "IDEAS.Buildings.Components.Examples.TwoStoreyBoxes": 4.61, + "IDEAS.Buildings.Components.InterzonalAirFlow.Examples.InterzonalAirFlow": 7.01, + "IDEAS.Buildings.Examples.OpenDoorComparison": 7.01, + "IDEAS.Buildings.Examples.ScreenComparison": 4.59, + "IDEAS.Buildings.Validation.BESTEST": 6.41, + "IDEAS.Examples.DetailedResidentialNoHeating": 7.01, + "IDEAS.Examples.PPD12.Heating": 7.01, + "IDEAS.Examples.PPD12.Structure": 7.01, + "IDEAS.Examples.PPD12.VentilationMPC": 7.01, + "IDEAS.Examples.PPD12.VentilationRBC": 7.01, + "IDEAS.Examples.Tutorial.DetailedHouse.DetailedHouse10": 4.66, + "IDEAS.Examples.Tutorial.DetailedHouse.DetailedHouse5": 4.69, + "IDEAS.Examples.Tutorial.DetailedHouse.DetailedHouse6": 4.75, + "IDEAS.Examples.Tutorial.DetailedHouse.DetailedHouse7": 4.76, + "IDEAS.Examples.Tutorial.DetailedHouse.DetailedHouse8": 4.74, + "IDEAS.Examples.Tutorial.DetailedHouse.DetailedHouse9": 4.74, + "IDEAS.Examples.TwinHouses.BuildingN2_Exp1": 7.01, + "IDEAS.Examples.TwinHouses.BuildingN2_Exp2": 7.01, + "IDEAS.Examples.TwinHouses.BuildingN2_Exp2_Tset": 7.01, + "IDEAS.Examples.TwinHouses.BuildingO5_Exp1": 7.01, + "IDEAS.Examples.TwinHouses.BuildingO5_Exp1_1Port": 7.01, + "IDEAS.Examples.TwinHouses.BuildingO5_Exp1_2Port": 7.01, + "IDEAS.Fluid.HeatExchangers.RadiantSlab.Examples.EmbeddedPipeNDiscr": 6.86 + }, + "Modelica_3.2.3": { + "Modelica.Blocks.Examples.Rectifier12pulseFFT": 5.31, + "Modelica.Blocks.Examples.Rectifier6pulseFFT": 5.58 + }, + "OpenIPSL_2.0.0": { + "OpenIPSL.Examples.DAEMode.N44_Base_Case_Systems.Nordic44_Base_Case_StateEvents": 4.45, + "OpenIPSL.Examples.DAEMode.N44_Base_Case_Systems.Nordic44_Base_Case_StateEvents2": 4.45, + "OpenIPSL.Examples.DAEMode.N44_Base_Case_Systems.Nordic44_Base_Case_StateEvents3": 4.63, + "OpenIPSL.Examples.DAEMode.N44_Original_Systems.Nordic44_Original_Case_Bus_Fault": 4.0, + "OpenIPSL.Examples.DAEMode.N44_Original_Systems.Nordic44_Original_Case_Line_Opening": 4.1, + "OpenIPSL.Examples.N44.Base_Case.Nordic44_Base_Case": 4.41, + "OpenIPSL.Examples.N44.Original.Nordic44_Original_Case": 4.2 + }, + "ScalableTestGrids_noopt": { + "ScalableTestGrids.Models.Type1.Type1_N_3_M_4": 4.85, + "ScalableTestGrids.Models.Type1.Type1_N_4_M_4": 5.64, + "ScalableTestGrids.Models.Type1.Type1_N_6_M_4": 6.67, + "ScalableTestGrids.Models.Type1.Type1_N_8_M_4": 10.43, + "ScalableTestGrids.Models.Type1.Type1_reduced_N_3_M_4": 4.85, + "ScalableTestGrids.Models.Type1.Type1_reduced_N_4_M_4": 6.4, + "ScalableTestGrids.Models.Type1.Type1_reduced_N_6_M_4": 6.68, + "ScalableTestGrids.Models.Type2.Type2_noTap___N_3_M_4": 4.85, + "ScalableTestGrids.Models.Type2.Type2_noTap___N_4_M_4": 7.79, + "ScalableTestGrids.Models.Type2.Type2_noTap___N_6_M_4": 6.65, + "ScalableTestGrids.Models.Type2.Type2_tapEv___N_3_M_4": 4.77, + "ScalableTestGrids.Models.Type2.Type2_tapEv___N_4_M_4": 8.05, + "ScalableTestGrids.Models.Type2.Type2_tapEv___N_6_M_4": 5.89, + "ScalableTestGrids.Models.Type2.Type2_tapNoEv_N_3_M_4": 4.77, + "ScalableTestGrids.Models.Type2.Type2_tapNoEv_N_4_M_4": 8.06, + "ScalableTestGrids.Models.Type2.Type2_tapNoEv_N_6_M_4": 10.47 + }, + "ScalableTestSuite": { + "ScalableTestSuite.Electrical.BreakerCircuits.ScaledExperiments.BreakerNetworkDelayed_N_1280_M_10": 5.73, + "ScalableTestSuite.Electrical.DistributionSystemAC.ScaledExperiments.DistributionSystemLinearIndividual_N_40_M_40": 6.13, + "ScalableTestSuite.Electrical.DistributionSystemAC.ScaledExperiments.DistributionSystemLinearIndividual_N_56_M_56": 10.46, + "ScalableTestSuite.Electrical.DistributionSystemAC.ScaledExperiments.DistributionSystemLinear_N_20_M_20": 4.29, + "ScalableTestSuite.Electrical.DistributionSystemAC.ScaledExperiments.DistributionSystemLinear_N_40_M_40": 7.39, + "ScalableTestSuite.Electrical.DistributionSystemAC.ScaledExperiments.DistributionSystemLinear_N_56_M_56": 10.37, + "ScalableTestSuite.Electrical.DistributionSystemDC.ScaledExperiments.DistributionSystemModelicaActiveLoads_N_56_M_56": 4.51, + "ScalableTestSuite.Electrical.DistributionSystemDC.ScaledExperiments.DistributionSystemModelicaActiveLoads_N_80_M_80": 7.07, + "ScalableTestSuite.Electrical.DistributionSystemDC.ScaledExperiments.DistributionSystemModelicaIndividual_N_80_M_80": 8.33, + "ScalableTestSuite.Electrical.DistributionSystemDC.ScaledExperiments.DistributionSystemModelica_N_112_M_112": 7.7, + "ScalableTestSuite.Electrical.DistributionSystemDC.ScaledExperiments.DistributionSystemModelica_N_56_M_56": 4.39, + "ScalableTestSuite.Electrical.DistributionSystemDC.ScaledExperiments.DistributionSystemModelica_N_80_M_80": 7.3, + "ScalableTestSuite.Elementary.ParameterArrays.ScaledExperiments.Table_N_140_M_140": 9.54, + "ScalableTestSuite.Elementary.ParameterArrays.ScaledExperiments.Table_N_200_M_200": 11.92, + "ScalableTestSuite.Thermal.Advection.ScaledExperiments.SimpleAdvection_N_12800": 4.97 + }, + "ScalableTestSuite_noopt": { + "ScalableTestSuite.Electrical.BreakerCircuits.ScaledExperiments.BreakerNetworkDelayed_N_1280_M_10": 5.73, + "ScalableTestSuite.Electrical.DistributionSystemAC.ScaledExperiments.DistributionSystemLinearIndividual_N_40_M_40": 6.17, + "ScalableTestSuite.Electrical.DistributionSystemAC.ScaledExperiments.DistributionSystemLinearIndividual_N_56_M_56": 8.33, + "ScalableTestSuite.Electrical.DistributionSystemAC.ScaledExperiments.DistributionSystemLinear_N_20_M_20": 4.29, + "ScalableTestSuite.Electrical.DistributionSystemAC.ScaledExperiments.DistributionSystemLinear_N_28_M_28": 8.15, + "ScalableTestSuite.Electrical.DistributionSystemAC.ScaledExperiments.DistributionSystemLinear_N_40_M_40": 6.34, + "ScalableTestSuite.Electrical.DistributionSystemAC.ScaledExperiments.DistributionSystemLinear_N_56_M_56": 7.07, + "ScalableTestSuite.Electrical.DistributionSystemDC.ScaledExperiments.DistributionSystemModelicaActiveLoads_N_56_M_56": 4.49, + "ScalableTestSuite.Electrical.DistributionSystemDC.ScaledExperiments.DistributionSystemModelicaActiveLoads_N_80_M_80": 8.59, + "ScalableTestSuite.Electrical.DistributionSystemDC.ScaledExperiments.DistributionSystemModelicaIndividual_N_80_M_80": 7.28, + "ScalableTestSuite.Electrical.DistributionSystemDC.ScaledExperiments.DistributionSystemModelica_N_112_M_112": 7.77, + "ScalableTestSuite.Electrical.DistributionSystemDC.ScaledExperiments.DistributionSystemModelica_N_56_M_56": 4.39, + "ScalableTestSuite.Electrical.DistributionSystemDC.ScaledExperiments.DistributionSystemModelica_N_80_M_80": 8.4, + "ScalableTestSuite.Elementary.ParameterArrays.ScaledExperiments.Table_N_140_M_140": 9.54, + "ScalableTestSuite.Elementary.ParameterArrays.ScaledExperiments.Table_N_200_M_200": 11.99, + "ScalableTestSuite.Thermal.Advection.ScaledExperiments.SimpleAdvection_N_12800": 4.96 + } +} diff --git a/heavy-models.py b/heavy-models.py new file mode 100755 index 0000000..0424583 --- /dev/null +++ b/heavy-models.py @@ -0,0 +1,120 @@ +#!/usr/bin/env python3 +""" +Propose the models for configs/heavy-models.json, from the memory the database +has seen them use. + + ./heavy-models.py --db postgresql://omread@localhost/omdb + ./heavy-models.py --db postgresql://omread@localhost/omdb --threshold 2 > configs/heavy-models.json + +It prints the file and never writes it: the list is meant to be read and edited +before it is used. What it proposes is the most a model has been measured at +over the newest runs of every branch given, because a model that dies early on +one branch reports what it had reached rather than what it needs - the several +dozen models that report 7.01 GiB on the C branches are all sitting on the +8 GB `ulimit -v`, not asking for 7 GiB. So the maximum over branches is a lower +bound worth believing, and a single branch is not. + +Runs older than the maxrss column read 0 and are skipped; a model that has never +completed anywhere cannot be proposed at all, and is reported instead. +""" + +import argparse, collections, sys +import simplejson as json +import resultsdb, shared + +GiB = float(1 << 30) + +# Right at a ulimit -v is where a translation died, not what it wanted. +ULIMIT_WALL_GiB = [7.01, 15.01] +ULIMIT_WALL_TOLERANCE = 0.02 + + +def atUlimitWall(gib): + return any(abs(gib - wall) < ULIMIT_WALL_TOLERANCE for wall in ULIMIT_WALL_GiB) + + +def peaks(cursor, db, branches, runs): + """The most every model has been seen to use, and which branch saw it.""" + best = {} + for branch in branches: + dates = [row[0] for row in cursor.execute( + "SELECT DISTINCT date FROM %s ORDER BY date DESC LIMIT %d" % (db.quote(branch), runs))] + if not dates: + print("No results for branch %s" % branch) + continue + holes = ",".join("?" * len(dates)) + for (libname, model, maxrss) in cursor.execute( + "SELECT libname, model, MAX(maxrss) FROM %s WHERE date IN (%s) AND maxrss > 0 " + "GROUP BY libname, model" % (db.quote(branch), holes), tuple(dates)): + key = (libname, model) + if maxrss / GiB > best.get(key, (0.0, None))[0]: + best[key] = (maxrss / GiB, branch) + return best + + +def unmeasured(cursor, db, branches, runs): + """Models no run has ever reported memory for: killed before they could, or + only ever tested before the column existed.""" + seen = set() + measured = set() + for branch in branches: + dates = [row[0] for row in cursor.execute( + "SELECT DISTINCT date FROM %s ORDER BY date DESC LIMIT %d" % (db.quote(branch), runs))] + if not dates: + continue + holes = ",".join("?" * len(dates)) + for (libname, model, maxrss) in cursor.execute( + "SELECT libname, model, MAX(maxrss) FROM %s WHERE date IN (%s) AND finalphase >= 0 " + "GROUP BY libname, model" % (db.quote(branch), holes), tuple(dates)): + seen.add((libname, model)) + if maxrss: + measured.add((libname, model)) + return seen - measured + + +def main(): + parser = argparse.ArgumentParser( + description="Propose the heavy-model list from measured memory use", + formatter_class=argparse.RawDescriptionHelpFormatter, epilog=__doc__) + parser.add_argument("--branch", default="master wasm-jit cpp newInst-newBackend", + help="Branches whose results are read, separated by spaces. Each model is " + "credited with the most any of them used.") + parser.add_argument("--runs", type=int, default=20, + help="How many of the newest runs of each branch to read (default 20)") + parser.add_argument("--threshold", type=float, default=4.0, + help="How many GiB a model has to have used to be proposed (default 4)") + resultsdb.addArgument(parser) + args = parser.parse_args() + + db = resultsdb.connect(args.db) + cursor = db.cursor() + branches = [shared.resultTable(b) for b in args.branch.split(" ") if b] + + best = peaks(cursor, db, branches, args.runs) + out = collections.defaultdict(dict) + walls = [] + for ((libname, model), (gib, branch)) in best.items(): + if gib < args.threshold: + continue + if atUlimitWall(gib): + walls.append((libname, model, gib)) + out[libname][model] = round(gib, 2) + + print(json.dumps(dict((lib, dict(sorted(models.items()))) for (lib, models) in sorted(out.items())), + indent=1, sort_keys=True)) + + # The report goes to stderr, so that stdout is the file. + heavy = sum(len(models) for models in out.values()) + report = ["%d models over %g GiB in %d libraries" % (heavy, args.threshold, len(out))] + for (libname, model, gib) in sorted(walls): + report.append(" ! %s %s stopped at %.2f GiB, which is a ulimit -v rather than what it needs" + % (libname, model, gib)) + never = unmeasured(cursor, db, branches, args.runs) + if never: + report.append(" ? %d models have never reported memory (killed before writing, or only " + "tested before the column existed); they are scheduled as light ones" % len(never)) + sys.stderr.write("\n".join(report) + "\n") + + +if __name__ == "__main__": + main() diff --git a/remove-run.py b/remove-run.py new file mode 100755 index 0000000..db25d5a --- /dev/null +++ b/remove-run.py @@ -0,0 +1,125 @@ +#!/usr/bin/env python3 +""" +Remove one test run from the database, all of its result tables together. + + ./remove-run.py --db postgresql://om@localhost/omdb wasm-jit + ./remove-run.py --db postgresql://om@localhost/omdb wasm-jit --write + +The first says what it would delete, the second deletes it. Without --date or +--omcversion it takes the newest run of that branch. + +A run of test.py --wasmjitrunner or --fmisimulator fills several tables from one +job - wasm-jit, wasm-jit-me and wasm-jit-cs share a date - and removing only the +first would leave the others describing a run that no longer exists. So every +table whose name is the branch or begins with it, and that has rows of that +date, goes at once; --also names any further table, for the --solver runners, +whose tables are named after the solver rather than the branch. + +The rows of a run are in the result table, in omcversion and in libversion. Its +job_claim rows are left alone: they say who tested a library last, the next run +overwrites them, and a finished claim stops nobody. +""" + +import argparse, sys +from datetime import datetime, timezone +import resultsdb, shared + + +def runDate(cursor, branch, date, omcversion): + """The date of the run being removed, and the omc that ran it.""" + if date and omcversion: + raise SystemExit("Give --date or --omcversion, not both") + if omcversion: + where = "branch=? AND omcversion=?" + params = (branch, omcversion) + elif date: + where = "branch=? AND date=?" + params = (branch, date) + else: + where = "branch=?" + params = (branch,) + rows = cursor.execute("SELECT date, omcversion FROM omcversion WHERE %s ORDER BY date DESC" + % where, params).fetchall() + if not rows: + raise SystemExit("No run of %s in omcversion matching that" % branch) + if omcversion and len(rows) > 1: + raise SystemExit("%s ran %s %d times; name one of them with --date %s" + % (branch, omcversion, len(rows), " --date ".join(str(r[0]) for r in rows))) + return rows[0] + + +def resultTables(cursor, db, branch, date, also): + """The tables one job of that branch wrote: itself, whichever of its runners + has rows of that date, and whatever --also names.""" + candidates = [t for t in db.tables() if t not in resultsdb.NON_RESULT_TABLES + and t.startswith(branch + "-")] + ran = [t for t in candidates + if cursor.execute("SELECT COUNT(*) FROM %s WHERE date=?" % db.quote(t), (date,)).fetchone()[0]] + return sorted(set([branch] + ran + list(also))) + + +def counts(cursor, db, tables, date): + """How many rows each table holds for that run, in the order they are deleted.""" + out = [] + for table in tables: + out.append((table, "date=?", (date,))) + for table in ("omcversion", "libversion"): + for name in tables: + out.append((table, "branch=? AND date=?", (name, date))) + return [(table, where, params, + cursor.execute("SELECT COUNT(*) FROM %s WHERE %s" % (db.quote(table), where), + params).fetchone()[0]) + for (table, where, params) in out] + + +def main(): + parser = argparse.ArgumentParser( + description="Remove one test run, all of its result tables together", + formatter_class=argparse.RawDescriptionHelpFormatter, epilog=__doc__) + parser.add_argument("branch", help="The branch whose run to remove, as it is named in the database") + parser.add_argument("--date", type=int, help="The run to remove, as the epoch second in its date column") + parser.add_argument("--omcversion", help="The run to remove, as the omc version that produced it") + parser.add_argument("--also", action="append", default=[], + help="A further table the same job wrote, for --solver runners, whose tables " + "are named after the solver. Repeatable.") + parser.add_argument("--write", action="store_true", help="Delete, instead of only saying what would be deleted") + resultsdb.addArgument(parser) + args = parser.parse_args() + + branch = shared.resultTable(args.branch) + db = resultsdb.connect(args.db) + cursor = db.cursor() + + (date, omcversion) = runDate(cursor, branch, args.date, args.omcversion) + tables = resultTables(cursor, db, branch, date, args.also) + print("%s run of %s, date %d (%s)" + % (branch, omcversion, date, datetime.fromtimestamp(date, tz=timezone.utc).isoformat())) + print("Result tables: %s" % ", ".join(tables)) + + rows = counts(cursor, db, tables, date) + for (table, where, params, n) in rows: + print(" %-24s %6d rows (%s)" % (table, n, " ".join(str(p) for p in params))) + total = sum(n for (_, _, _, n) in rows) + print(" %-24s %6d rows" % ("total", total)) + if not total: + raise SystemExit("Nothing to remove") + + if not args.write: + print("\nNothing deleted. Run it again with --write to delete.") + return + + for (table, where, params, n) in rows: + cursor.execute("DELETE FROM %s WHERE %s" % (db.quote(table), where), params) + db.commit() + + left = [(table, n) for (table, where, params, n) in counts(cursor, db, tables, date) if n] + if left: + print("Still there after deleting: %s" % ", ".join("%s (%d)" % t for t in left)) + sys.exit(1) + print("Removed %d rows." % total) + db.vacuum() + db.close() + + +if __name__ == "__main__": + main() diff --git a/shared.py b/shared.py index 5c3eafc..ee997d3 100644 --- a/shared.py +++ b/shared.py @@ -1,6 +1,6 @@ #!/usr/bin/env python3 -import re, os, signal, string, subprocess +import collections, re, os, signal, string, subprocess, threading, traceback import simplejson as json # Windows has no SIGKILL, and no process group to signal instead; os.kill there @@ -57,6 +57,7 @@ def fixData(data,abortSimulationFlag,alarmFlag,overrideDefaults,defaultCustomCom data["ulimitOmc"] = int(data.get("ulimitOmc") or 660) # 11 minutes to generate the C-code data["ulimitExe"] = int(data.get("ulimitExe") or DEFAULT_ULIMIT_EXE) data["ulimitExeModels"] = dict((k,int(v)) for (k,v) in (data.get("ulimitExeModels") or {}).items()) + data["heavyModels"] = dict((k,float(v)) for (k,v) in (data.get("heavyModels") or {}).items()) data["ulimitLoadModel"] = int(data.get("ulimitLoadModel") or 3*60) # 3 minutes to load the files (could take a while if the ssd is doing backup) simflags = [] if data.get("extraSimFlags"): @@ -98,6 +99,65 @@ def modelUlimitExe(conf, modelName): is one of the few named in ulimitExeModels.""" return conf["ulimitExeModels"].get(modelName) or conf["ulimitExe"] +def runCapped(jobs, isHeavy, run, workers, heavyWorkers, progress=None): + """Run the jobs over that many worker threads, at most heavyWorkers of the + heavy ones at a time. + + Two queues, so that the cap costs memory and not machine time: a worker that + may not start a heavy job takes the next light one rather than wait for a slot, + and waits only when heavy jobs are all that is left. + """ + heavy = collections.deque(job for job in jobs if isHeavy(job)) + light = collections.deque(job for job in jobs if not isHeavy(job)) + total = len(heavy) + len(light) + cond = threading.Condition() + state = {"heavyRunning": 0, "done": 0} + + def take(preferHeavy): + with cond: + while True: + if heavy and state["heavyRunning"] < heavyWorkers and (preferHeavy or not light): + state["heavyRunning"] += 1 + return (heavy.popleft(), True) + if light: + return (light.popleft(), False) + if not heavy: + return (None, False) + cond.wait() + + def worker(preferHeavy): + while True: + (job, wasHeavy) = take(preferHeavy) + if job is None: + return + try: + run(job) + except Exception: + # Losing the worker would leave its share of the queue untested. + traceback.print_exc() + finally: + with cond: + if wasHeavy: + state["heavyRunning"] -= 1 + state["done"] += 1 + done = state["done"] + cond.notify_all() + if progress is not None: + progress(done, total) + + threads = [threading.Thread(target=worker, args=(i < heavyWorkers,), daemon=True) + for i in range(workers)] + for thread in threads: + thread.start() + for thread in threads: + thread.join() + return total + +def isHeavyModel(conf, modelName): + """Whether that model is one of the few needing gigabytes, of which only so + many run at a time.""" + return modelName in conf["heavyModels"] + def simulationFlags(conf, ulimitExe): """The flags a simulation allowed that many seconds is run with.""" if conf["alarmFlag"] == "": diff --git a/test.py b/test.py index a6d020c..97ebfc2 100755 --- a/test.py +++ b/test.py @@ -16,7 +16,7 @@ from monotonic import monotonic from omcommon import friendlyStr, multiple_replace from natsort import natsorted -from shared import readConfig, getReferenceFileName, simulationAcceptsFlag, isFMPy, modelUlimitExe, simulationFlags, alarmGrace +from shared import readConfig, getReferenceFileName, simulationAcceptsFlag, isFMPy, isHeavyModel, modelUlimitExe, simulationFlags, alarmGrace from platform import processor import shared, resultsdb @@ -40,6 +40,8 @@ parser.add_argument('--wasmjitrunner', action='append', default=[], help="Export every model once as a wasm artifact (buildModelFMU with fmuType=me_cs, platforms={wasm,}) and simulate that one artifact each of these ways: 'sim' runs the translated model the way simulate() does, 'me' and 'cs' the artifact's FMI 3.0 interfaces. Comma-separated or repeated; the first fills --branch and each further one -, so --branch=master-wasm-jit with sim,me,cs fills master-wasm-jit, master-wasm-jit-me and master-wasm-jit-cs. See configs/wasm-jit-runners.json. Only for simCodeTarget=wasm-jit.") parser.add_argument('--solver', action='append', default=[], help="Build every model once and simulate it once per solver, so that testing another solver costs a simulation rather than a build. 'default' is the model's own solver and each further name is a -s the simulation is given; comma-separated or repeated. Every solver stores its results in the table its entry names, so --branch=master with default,cvode,gbode fills master, cvode and gbode. See configs/solvers.json.") parser.add_argument('--ulimitvmem', help="Virtual memory limit (in kB) (linux only)", type=int, default=8*1024*1024) +parser.add_argument('--heavyjobs', help="How many of the models listed in --heavymodels may run at the same time. They are the handful that need gigabytes each, and running sixteen of them together is what takes the machine out of memory.", type=int, default=4) +parser.add_argument('--heavymodels', help="JSON file naming the models that need gigabytes, as {library: {model: GiB}}. Hand-curated; heavy-models.py proposes what to put in it.", default="configs/heavy-models.json") parser.add_argument('--default', action='append', help="Add a default value for some configuration key, such as --default=ulimitExe=60. The equals sign is mandatory.", default=[]) parser.add_argument('-j', '--jobs', default=0, help="Ignored and deprecated, use procOMC:0 or procOMC:1 in the config") parser.add_argument('-v', '--verbose', action="store_true", help="Verbose mode.", default=False) @@ -123,6 +125,14 @@ def outputFor(resultBranch): allTestsFmi = args.fmi fmuType = args.fmuType ulimitMemory = args.ulimitvmem +heavyJobs = args.heavyjobs +# {libname: {model: GiB}}, kept in one hand-curated file rather than spread over +# the library entries: the same model is heavy in every configuration listing it. +try: + heavyModelsByLibrary = json.load(open(args.heavymodels)) +except IOError: + heavyModelsByLibrary = {} + print("No %s; every test is scheduled as a light one" % args.heavymodels) docker = args.docker addmsl = args.addmsl @@ -216,6 +226,25 @@ def killTree(pid, sig): except (OSError, psutil.Error): pass +def descendantPids(pid): + """Everything below that process, right now. + + OMPython starts omc in a session of its own, so signalling the process group + does not reach it, and once testmodel.py is dead nothing names it any more: + ask before killing the parent, kill after. + """ + try: + return [p.pid for p in psutil.Process(pid).children(recursive=True)] + except psutil.Error: + return [] + +def killPids(pids, sig): + for pid in pids: + try: + os.kill(pid, sig) + except OSError: + pass + def runCommand(cmd, prefix, timeout): process = [None] def target(): @@ -241,6 +270,7 @@ def target(): if thread.is_alive(): gotTimeout = True + strays = [] if isWin else descendantPids(process[0].pid) if isWin: killTree(process[0].pid, signal.SIGTERM) else: @@ -252,6 +282,8 @@ def target(): else: os.kill(-process[0].pid, shared.SIGKILL) thread.join(10) + # For when testmodel.py did not get far enough into its SIGTERM to do this. + killPids(strays, shared.SIGKILL) if clean: try: @@ -825,6 +857,7 @@ def simulatorKey(libname, runner): prefix = conf["ignoreModelPrefix"] res=list(filter(lambda x: not x.startswith(prefix), res)) libName=shared.libname(library, conf) + conf["heavyModels"].update(heavyModelsByLibrary.get(libName) or {}) todo = runnersToRun(libName, conf) if libName in stats_by_libname or libName in skipped_libs: raise Exception("Duplicate libName found: %s" % libName) @@ -996,21 +1029,27 @@ def expectedExec(c): print("Starting execution of %d tests. Estimated execution time %s (wrong if there are new or few tests).\n" % (numberOfTests, friendlyStr(sum(expectedExec(c) for c in tests)/(1.0*n_jobs)))) sys.stdout.flush() -cmd_res=[0] start=monotonic() start_as_time=time.localtime() testRunStartTimeAsEpoch = int(time.time()) -# Need translateModel + make + exe... -if n_jobs == 1: - verbose = 10 -else: - verbose = 5 -if customTimeout > 0.0: - cmd_res=Parallel(n_jobs=n_jobs, verbose=verbose)(delayed(runScript)(name, customTimeout, data["ulimitMemory"], runverbose) for (model,lib,libName,name,data) in tests) -else: - # Each command that runs out of time keeps running for a grace before it gives - # up; killing testmodel.py during it throws away the phase times. - cmd_res=Parallel(n_jobs=n_jobs, verbose=verbose)(delayed(runScript)(name, 2*(data["ulimitOmc"]+alarmGrace(data["ulimitOmc"]))+modelUlimitExe(data, model)+alarmGrace(modelUlimitExe(data, model))+25, data["ulimitMemory"], runverbose) for (model,lib,libName,name,data) in tests) + +def testTimeout(model, data): + """Need translateModel + make + exe... Each command that runs out of time keeps + running for a grace before it gives up; killing testmodel.py during it throws + away the phase times.""" + if customTimeout > 0.0: + return customTimeout + return 2*(data["ulimitOmc"]+alarmGrace(data["ulimitOmc"]))+modelUlimitExe(data, model)+alarmGrace(modelUlimitExe(data, model))+25 + +def progress(done, total): + if runverbose or n_jobs == 1 or done % 100 == 0: + print("[%d/%d done, %s]" % (done, total, friendlyStr(monotonic()-start))) + sys.stdout.flush() + +shared.runCapped(tests, + lambda test: isHeavyModel(test[4], test[0]), + lambda test: runScript(test[3], testTimeout(test[0], test[4]), test[4]["ulimitMemory"], runverbose), + n_jobs, heavyJobs, progress) stop=monotonic() print("Execution time: %s" % friendlyStr(stop-start)) assert(stop-start >= 0.0) diff --git a/testmodel.py b/testmodel.py index f11006b..00ebc99 100755 --- a/testmodel.py +++ b/testmodel.py @@ -120,10 +120,30 @@ def peakRss(pid): treePeak = 0 processPeak = 0 +preferredVictims = set() + +def preferOomKill(pid): + """Offer a child to the OOM killer ahead of this process, so that omc dies and + python lives to report the phase it had reached. + + Raising a child is allowed unprivileged; lowering this process is not. + """ + if isWin or pid in preferredVictims: + return + preferredVictims.add(pid) + try: + with open("/proc/%d/oom_score_adj" % pid, "w") as fp: + fp.write("1000") + except OSError: + pass + def sampleTree(): global treePeak while True: - treePeak = max(treePeak, sum(currentRss(pid) for pid in descendants())) + pids = descendants() + for pid in pids: + preferOomKill(pid) + treePeak = max(treePeak, sum(currentRss(pid) for pid in pids)) time.sleep(0.2) def noteRss(rss): @@ -340,6 +360,18 @@ def target(res): except OSError: pass +def terminateHandler(signum, frame): + """The outer timeout sends SIGTERM before it SIGKILLs the process group. omc is + not in that group, so unless it is taken down here it outlives the run, and + without the result file the model has no row at all.""" + with open(errFile, 'a+') as fp: + fp.write("[Killed by signal %d after %s]\n" % (signum, monotonic()-startJob)) + writeResult() + killChildren(shared.SIGKILL, "SIGKILL") + os._exit(1) + +signal.signal(signal.SIGTERM, terminateHandler) + with open(errFile, 'a+') as fp: fp.write("Running: %s\n" % " ".join(sys.argv))