diff --git a/src/common/darktable.c b/src/common/darktable.c index 3a1f0e8189c..03b8bf5dbac 100644 --- a/src/common/darktable.c +++ b/src/common/darktable.c @@ -1845,6 +1845,7 @@ int dt_init(int argc, size_t total_mb = _get_total_memory() / 1024lu; if(total_mb < 8192) total_mb -= 1024; res->total_memory = total_mb * DT_MEGA; + res->cl_uni_memory = 0; char *config_info = calloc(1, DT_PERF_INFOSIZE); if(last_configure_version != DT_CURRENT_PERFORMANCE_CONFIGURE_VERSION @@ -2474,7 +2475,7 @@ size_t dt_get_available_mem() return res->refresource[4*(-level-1)] * DT_MEGA; const int fraction = res->fractions[4*level]; - return MAX(512lu * DT_MEGA, res->total_memory / 1024lu * fraction); + return MAX(512lu * DT_MEGA, (res->total_memory - res->cl_uni_memory) / 1024lu * fraction); } size_t dt_get_singlebuffer_mem() diff --git a/src/common/darktable.h b/src/common/darktable.h index 1979f437b40..cff060d5fd5 100644 --- a/src/common/darktable.h +++ b/src/common/darktable.h @@ -395,6 +395,7 @@ typedef struct dt_sys_resources_t { size_t total_memory; size_t mipmap_memory; + size_t cl_uni_memory; int *fractions; // fractions are calculated as res=input / 1024 * fraction int *refresource; // for the debug resource modes we use fixed settings int level; diff --git a/src/common/opencl.c b/src/common/opencl.c index d93dc43b04a..2d98d590444 100644 --- a/src/common/opencl.c +++ b/src/common/opencl.c @@ -1606,23 +1606,6 @@ void dt_opencl_init(dt_opencl_t *cl, // priorities and pixelpipe synchronization timeout dt_opencl_scheduling_profile_t profile = _opencl_get_scheduling_profile(); _opencl_apply_scheduling_profile(profile); - - // let's keep track on unified memory devices - dt_sys_resources_t *res = &darktable.dtresources; - for(int i = 0; i < cl->num_devs; i++) - { - if(cl->dev[i].unified_memory) - { - const size_t reserved = MIN(cl->dev[i].max_global_mem, res->total_memory * cl->dev[i].unified_fraction); - cl->dev[i].max_global_mem = reserved; - cl->dev[i].max_mem_alloc = MIN(cl->dev[i].max_mem_alloc, reserved); - dt_print_nts(DT_DEBUG_OPENCL, - " UNIFIED MEM SIZE: %.0f MB (%i%%) reserved for '%s' id=%d\n", - (double)reserved / 1024.0 / 1024.0, (int)(100.0f * cl->dev[i].unified_fraction), - cl->dev[i].cname, i); - res->total_memory -= reserved; - } - } } else // initialization failed { @@ -3412,13 +3395,19 @@ void *dt_opencl_alloc_device(const int devid, return dev; } +static cl_ulong _opencl_get_device_memalloc(const int devid) +{ + dt_opencl_t *cl = darktable.opencl; + return MIN(cl->dev[devid].used_available, cl->dev[devid].max_mem_alloc); +} + void *dt_opencl_alloc_device_buffer(const int devid, const size_t size) { if(!_cldev_running(devid)) return NULL; dt_opencl_t *cl = darktable.opencl; - if(cl->dev[devid].max_mem_alloc < size) + if(_opencl_get_device_memalloc(devid) < size) return NULL; cl_int err = CL_SUCCESS; @@ -3443,7 +3432,7 @@ void *dt_opencl_alloc_device_buffer_with_flags(const int devid, if(!_cldev_running(devid)) return NULL; dt_opencl_t *cl = darktable.opencl; - if(cl->dev[devid].max_mem_alloc < size) + if(_opencl_get_device_memalloc(devid) < size) return NULL; cl_int err = CL_SUCCESS; @@ -3610,61 +3599,12 @@ void dt_opencl_memory_statistics(int devid, } } -/* amount of graphics memory declared as available depends on max_global_mem and - "resourcelevel". We garantee - - a headroom of DT_OPENCL_DEFAULT_HEADROOM MB in all cases not using tuned cl - - 256MB to simulate a minimum system - - 2GB to simulate a reference system -*/ -void dt_opencl_check_tuning(const int devid) -{ - dt_sys_resources_t *res = &darktable.dtresources; - dt_opencl_t *cl = darktable.opencl; - if(!_cldev_running(devid)) return; - - const int level = res->level; - const gboolean tunehead = cl->num_devs > 1 - && level >= 0 - && !dt_gimpmode() - && dt_conf_get_bool("opencl_tune_headroom"); - - cl->dev[devid].tunehead = tunehead; - - if(level < 0) - { - cl->dev[devid].used_available = res->refresource[4*(-level-1) + 3] * DT_MEGA; - } - else - { - const size_t allmem = cl->dev[devid].max_global_mem; - const size_t lowmem = 256ul * DT_MEGA; - const size_t dhead = DT_OPENCL_DEFAULT_HEADROOM * DT_MEGA; - if(cl->dev[devid].tunehead) - { - const size_t headroom = (cl->dev[devid].headroom ? DT_MEGA * cl->dev[devid].headroom : dhead) - + (cl->dev[devid].clmem_error ? dhead : 0); - cl->dev[devid].used_available = allmem > headroom ? allmem - headroom : lowmem; - } - else - { - const size_t disposable = allmem > dhead ? allmem - dhead : 0; - const int fraction = MIN(1024, res->fractions[4*res->level + 3]); - cl->dev[devid].used_available = MAX(lowmem, disposable / 1024ul * fraction); - } - } -} - cl_ulong dt_opencl_get_device_available(const int devid) { if(!darktable.opencl->inited || devid <= DT_DEVICE_CPU) return 0; return darktable.opencl->dev[devid].used_available; } -static cl_ulong _opencl_get_device_memalloc(const int devid) -{ - return darktable.opencl->dev[devid].max_mem_alloc; -} - cl_ulong dt_opencl_get_device_memalloc(const int devid) { if(!darktable.opencl->inited || devid <= DT_DEVICE_CPU) return 0; @@ -3756,6 +3696,59 @@ void dt_opencl_update_settings(void) const char *pstr = dt_conf_get_string_const("opencl_scheduling_profile"); dt_print(DT_DEBUG_OPENCL | DT_DEBUG_VERBOSE, "[opencl_update_settings] scheduling profile set to %s", pstr); + + dt_sys_resources_t *res = &darktable.dtresources; + /* If we have cl devices with unified memery we should not use that part + for general dt use. + As that part might change with a different resource level we have to + fix that whenever that changes. + */ + res->cl_uni_memory = 0; + const int level = res->level; + const gboolean tunehead = cl->num_devs > 1 + && level >= 0 + && !dt_gimpmode() + && dt_conf_get_bool("opencl_tune_headroom"); + + for(int i = 0; i < cl->num_devs; i++) + { + cl->dev[i].tunehead = tunehead; + if(level < 0) + { + cl->dev[i].used_available = res->refresource[4*(-level-1) + 3] * DT_MEGA; + } + else + { + const size_t allmem = cl->dev[i].max_global_mem; + const size_t lowmem = 256ul * DT_MEGA; + const size_t dhead = DT_OPENCL_DEFAULT_HEADROOM * DT_MEGA; + if(cl->dev[i].tunehead) + { + const size_t headroom = cl->dev[i].headroom ? DT_MEGA * cl->dev[i].headroom : dhead; + cl->dev[i].used_available = allmem > headroom ? allmem - headroom : lowmem; + } + else + { + const size_t disposable = allmem > dhead ? allmem - dhead : 0; + const int fraction = MIN(1024, res->fractions[4*res->level + 3]); + cl->dev[i].used_available = MAX(lowmem, disposable / 1024ul * fraction); + } + } + + if(cl->dev[i].unified_memory) + { + cl->dev[i].used_available = MIN(cl->dev[i].used_available, res->total_memory * cl->dev[i].unified_fraction); + res->cl_uni_memory += cl->dev[i].used_available; + } + dt_print_nts(DT_DEBUG_OPENCL, + " AVAILABLE CLMEM SIZE: %zu MB%s%s\n", + (size_t)(cl->dev[i].used_available / DT_MEGA), + cl->dev[i].tunehead ? ", tuned" : "", + cl->dev[i].pinned_memory ? ", pinned": ""); + } + if(res->cl_uni_memory) + dt_print_nts(DT_DEBUG_OPENCL, + " UNIFIED SYSMEM SIZE: %zu MB\n", (size_t)(res->cl_uni_memory / DT_MEGA)); } /** read scheduling profile for config variables */ diff --git a/src/common/opencl.h b/src/common/opencl.h index 34a188d8aba..bb814f09847 100644 --- a/src/common/opencl.h +++ b/src/common/opencl.h @@ -572,9 +572,6 @@ gboolean dt_opencl_image_fits_device(const int devid, /** get available memory for the device */ cl_ulong dt_opencl_get_device_available(const int devid); -/** check tuning settings and available memory for the device */ -void dt_opencl_check_tuning(const int devid); - /** get size of allocatable single buffer */ cl_ulong dt_opencl_get_device_memalloc(const int devid); diff --git a/src/develop/pixelpipe_hb.c b/src/develop/pixelpipe_hb.c index beb4f4004ad..dffc36910cb 100644 --- a/src/develop/pixelpipe_hb.c +++ b/src/develop/pixelpipe_hb.c @@ -3113,25 +3113,15 @@ gboolean dt_dev_pixelpipe_process(dt_dev_pixelpipe_t *pipe, dt_iop_buffer_dsc_t _out_format = { 0 }; dt_iop_buffer_dsc_t *out_format = &_out_format; -#ifdef HAVE_OPENCL - dt_opencl_check_tuning(pipe->devid); - if(pipe->devid > DT_DEVICE_CPU) - dt_print_pipe(DT_DEBUG_PIPE, "pipe starting", - pipe, NULL, pipe->devid, &roi, &roi, "'%s' ID=%i, %s using %luMB%s%s", - pipe->image.filename, pipe->image.id, - darktable.opencl->dev[pipe->devid].cname, - darktable.opencl->dev[pipe->devid].used_available / DT_MEGA, - darktable.opencl->dev[pipe->devid].tunehead ? ", tuned" : "", - darktable.opencl->dev[pipe->devid].pinned_memory ? ", pinned": ""); - else - dt_print_pipe(DT_DEBUG_PIPE, "pipe starting", - pipe, NULL, pipe->devid, &roi, &roi, "'%s' ID=%i using %luMB", - pipe->image.filename, pipe->image.id, dt_get_available_mem() / DT_MEGA); -#else + const size_t avail_mem = + #ifdef HAVE_OPENCL + pipe->devid > DT_DEVICE_CPU ? dt_opencl_get_device_available(pipe->devid) : dt_get_available_mem(); + #else + dt_get_available_mem(); + #endif dt_print_pipe(DT_DEBUG_PIPE, "pipe starting", - pipe, NULL, pipe->devid, &roi, &roi, "'%s' ID=%i using %luMB", - pipe->image.filename, pipe->image.id, dt_get_available_mem() / DT_MEGA); -#endif + pipe, NULL, pipe->devid, &roi, &roi, "'%s' ID=%i using %luMB", + pipe->image.filename, pipe->image.id, avail_mem / DT_MEGA); dt_print_mem_usage("before pixelpipe process"); // run pixelpipe recursively and get error status