diff --git a/examples/myresearch/train.py b/examples/myresearch/train.py index 22499b2..ddcdf7b 100644 --- a/examples/myresearch/train.py +++ b/examples/myresearch/train.py @@ -19,6 +19,7 @@ from fluksio import Port, node device="gpu", device_policy="prefer", timeout=600, + resources={"cpus": 2}, ) def fit(dataset, lr, epochs=25): """Train, reporting the loss as it goes. @@ -28,6 +29,12 @@ def fit(dataset, lr, epochs=25): why there is no `log_metric()` to call. `device="gpu"` with `device_policy="prefer"` sends this to a worker carrying that label when one is attached, and runs it here when none is. + + `resources` is what a real training node needs and the rest of a flow does + not: two cores held for the length of the call, and a worker started with + thread limits saying so — otherwise a numerical library sizes itself to + every core on the machine, and a sweep of these starves the engine. A GPU + one would say `{"gpus": 1}` and hold the card exclusively. """ rows = json.loads(open(fluksio.load_artifact(dataset)).read())["rows"] loss = 1.0