improve C module parallel

This commit is contained in:
Kinneyzhang 2026-01-25 17:28:04 +08:00
parent 6cf6f75f71
commit 68ca9b191a
3 changed files with 356 additions and 107 deletions

View File

@ -407,6 +407,200 @@ static emacs_value Fekp_c_break_with_arrays(emacs_env *env, ptrdiff_t nargs,
return final; return final;
} }
/*
* Helper to extract paragraph data from Elisp vectors
*/
static bool extract_paragraph_data(
emacs_env *env, emacs_value *args,
int32_t **ideal_prefix, int32_t **min_prefix, int32_t **max_prefix,
int32_t **glue_ideals, int32_t **glue_shrinks, int32_t **glue_stretches,
int32_t **hyph_pos, size_t *n, ptrdiff_t *hyph_count,
int32_t *hyph_width, int32_t *line_width)
{
ptrdiff_t prefix_len = env->vec_size(env, args[0]);
if (prefix_len <= 1)
return false;
*n = prefix_len - 1;
*ideal_prefix = malloc(prefix_len * sizeof(int32_t));
*min_prefix = malloc(prefix_len * sizeof(int32_t));
*max_prefix = malloc(prefix_len * sizeof(int32_t));
*glue_ideals = malloc(*n * sizeof(int32_t));
*glue_shrinks = malloc(*n * sizeof(int32_t));
*glue_stretches = malloc(*n * sizeof(int32_t));
if (!*ideal_prefix || !*min_prefix || !*max_prefix ||
!*glue_ideals || !*glue_shrinks || !*glue_stretches) {
free(*ideal_prefix); free(*min_prefix); free(*max_prefix);
free(*glue_ideals); free(*glue_shrinks); free(*glue_stretches);
return false;
}
for (ptrdiff_t i = 0; i < prefix_len; i++) {
(*ideal_prefix)[i] = env->extract_integer(env, env->vec_get(env, args[0], i));
(*min_prefix)[i] = env->extract_integer(env, env->vec_get(env, args[1], i));
(*max_prefix)[i] = env->extract_integer(env, env->vec_get(env, args[2], i));
}
for (size_t i = 0; i < *n; i++) {
(*glue_ideals)[i] = env->extract_integer(env, env->vec_get(env, args[3], i));
(*glue_shrinks)[i] = env->extract_integer(env, env->vec_get(env, args[4], i));
(*glue_stretches)[i] = env->extract_integer(env, env->vec_get(env, args[5], i));
}
*hyph_count = env->vec_size(env, args[6]);
*hyph_pos = NULL;
if (*hyph_count > 0) {
*hyph_pos = malloc(*hyph_count * sizeof(int32_t));
if (*hyph_pos) {
for (ptrdiff_t i = 0; i < *hyph_count; i++) {
(*hyph_pos)[i] = env->extract_integer(env, env->vec_get(env, args[6], i));
}
}
}
*hyph_width = env->extract_integer(env, args[7]);
*line_width = env->extract_integer(env, args[8]);
return true;
}
/*
* ekp-c-break-batch: Process multiple paragraphs in parallel
*
* Args: vector of (ideal-prefix min-prefix max-prefix glue-ideals glue-shrinks
* glue-stretches hyphen-positions hyphen-width line-width)
*
* Each element is a vector of 9 elements (same as ekp-c-break-with-arrays args).
* Returns vector of (breaks . total-cost) for each paragraph.
*
* This is the high-performance API for processing multi-paragraph text.
*/
static emacs_value Fekp_c_break_batch(emacs_env *env, ptrdiff_t nargs,
emacs_value *args, void *data)
{
(void)data;
if (!ekp_global || nargs < 1)
return env->intern(env, "nil");
ptrdiff_t para_count = env->vec_size(env, args[0]);
if (para_count <= 0)
return env->intern(env, "nil");
/* Allocate batch inputs and temporary storage */
ekp_batch_input_t *inputs = calloc(para_count, sizeof(ekp_batch_input_t));
int32_t **all_ideal = calloc(para_count, sizeof(int32_t *));
int32_t **all_min = calloc(para_count, sizeof(int32_t *));
int32_t **all_max = calloc(para_count, sizeof(int32_t *));
int32_t **all_glue_i = calloc(para_count, sizeof(int32_t *));
int32_t **all_glue_sh = calloc(para_count, sizeof(int32_t *));
int32_t **all_glue_st = calloc(para_count, sizeof(int32_t *));
int32_t **all_hyph = calloc(para_count, sizeof(int32_t *));
if (!inputs || !all_ideal || !all_min || !all_max ||
!all_glue_i || !all_glue_sh || !all_glue_st || !all_hyph) {
free(inputs); free(all_ideal); free(all_min); free(all_max);
free(all_glue_i); free(all_glue_sh); free(all_glue_st); free(all_hyph);
return env->intern(env, "nil");
}
/* Extract all paragraph data */
for (ptrdiff_t p = 0; p < para_count; p++) {
emacs_value para_vec = env->vec_get(env, args[0], p);
/* Extract 9 arguments from this paragraph's vector */
emacs_value para_args[9];
for (int i = 0; i < 9; i++) {
para_args[i] = env->vec_get(env, para_vec, i);
}
size_t n;
ptrdiff_t hyph_count;
int32_t hyph_width, line_width;
if (!extract_paragraph_data(env, para_args,
&all_ideal[p], &all_min[p], &all_max[p],
&all_glue_i[p], &all_glue_sh[p], &all_glue_st[p],
&all_hyph[p], &n, &hyph_count,
&hyph_width, &line_width)) {
/* Cleanup on failure */
for (ptrdiff_t j = 0; j < p; j++) {
free(all_ideal[j]); free(all_min[j]); free(all_max[j]);
free(all_glue_i[j]); free(all_glue_sh[j]); free(all_glue_st[j]);
free(all_hyph[j]);
}
free(inputs); free(all_ideal); free(all_min); free(all_max);
free(all_glue_i); free(all_glue_sh); free(all_glue_st); free(all_hyph);
return env->intern(env, "nil");
}
inputs[p].ideal_prefix = all_ideal[p];
inputs[p].min_prefix = all_min[p];
inputs[p].max_prefix = all_max[p];
inputs[p].glue_ideals = all_glue_i[p];
inputs[p].glue_shrinks = all_glue_sh[p];
inputs[p].glue_stretches = all_glue_st[p];
inputs[p].n = n;
inputs[p].hyphen_positions = all_hyph[p];
inputs[p].hyphen_count = hyph_count > 0 ? (size_t)hyph_count : 0;
inputs[p].hyphen_width = hyph_width;
inputs[p].line_width = line_width;
}
/* Process all paragraphs in parallel */
ekp_result_t **results = ekp_break_batch(inputs, para_count);
/* Cleanup input arrays */
for (ptrdiff_t p = 0; p < para_count; p++) {
free(all_ideal[p]); free(all_min[p]); free(all_max[p]);
free(all_glue_i[p]); free(all_glue_sh[p]); free(all_glue_st[p]);
free(all_hyph[p]);
}
free(inputs); free(all_ideal); free(all_min); free(all_max);
free(all_glue_i); free(all_glue_sh); free(all_glue_st); free(all_hyph);
if (!results)
return env->intern(env, "nil");
/* Build result vector */
emacs_value result_vec = env->funcall(env, env->intern(env, "make-vector"),
2, (emacs_value[]){
env->make_integer(env, para_count),
env->intern(env, "nil")
});
emacs_value cons_sym = env->intern(env, "cons");
for (ptrdiff_t p = 0; p < para_count; p++) {
ekp_result_t *r = results[p];
emacs_value entry;
if (r) {
/* Build (breaks . cost) */
emacs_value breaks_list = env->intern(env, "nil");
for (size_t i = r->break_count; i > 0; i--) {
emacs_value brk = env->make_integer(env, r->breaks[i - 1]);
emacs_value args2[2] = {brk, breaks_list};
breaks_list = env->funcall(env, cons_sym, 2, args2);
}
emacs_value cost = env->make_float(env, r->total_cost);
emacs_value args2[2] = {breaks_list, cost};
entry = env->funcall(env, cons_sym, 2, args2);
ekp_result_destroy(r);
} else {
entry = env->intern(env, "nil");
}
env->vec_set(env, result_vec, p, entry);
}
free(results);
return result_vec;
}
/* /*
* Helper to define functions * Helper to define functions
*/ */
@ -493,6 +687,15 @@ This API ensures C uses Elisp's font-dependent measurements.\n\n\
defun(env, "ekp-c-thread-count", 0, 0, Fekp_c_thread_count, defun(env, "ekp-c-thread-count", 0, 0, Fekp_c_thread_count,
"Return number of worker threads in the thread pool."); "Return number of worker threads in the thread pool.");
defun(env, "ekp-c-break-batch", 1, 1, Fekp_c_break_batch,
"Break multiple paragraphs in parallel.\n\n\
PARAGRAPHS: vector of paragraph data, each element is a vector of 9 items:\n\
[ideal-prefix min-prefix max-prefix glue-ideals glue-shrinks\n\
glue-stretches hyphen-positions hyphen-width line-width]\n\n\
Returns vector of (BREAKS . COST) for each paragraph.\n\
This is the high-performance API for multi-paragraph processing.\n\n\
(fn PARAGRAPHS)");
/* Provide feature */ /* Provide feature */
emacs_value provide_args[1] = {env->intern(env, "ekp-c")}; emacs_value provide_args[1] = {env->intern(env, "ekp-c")};
env->funcall(env, env->intern(env, "provide"), 1, provide_args); env->funcall(env, env->intern(env, "provide"), 1, provide_args);

View File

@ -297,115 +297,46 @@ ekp_result_t *ekp_break_lines(ekp_paragraph_t *p, int32_t line_width)
int fitness_penalty = ekp_global ? ekp_global->fitness_penalty : 100; int fitness_penalty = ekp_global ? ekp_global->fitness_penalty : 100;
double last_ratio = ekp_global ? ekp_global->last_line_ratio : 0.5; double last_ratio = ekp_global ? ekp_global->last_line_ratio : 0.5;
/* For small paragraphs, single-threaded */ /*
if (n < 100 || !ekp_global || !ekp_global->pool) { * Single-threaded DP: simple and correct.
dp_work_t work = { *
.para = p, * Note: Previous "parallel" implementation had data races - multiple
.line_width = line_width, * threads writing to shared demerits[] array without synchronization.
.start = 0, * DP has inherent sequential dependencies (demerits[k] depends on all
.end = n, * demerits[i] where i < k), making intra-paragraph parallelism complex.
.demerits = demerits, *
.backptrs = backptrs, * For real parallelism, use ekp_break_batch() to process multiple
.rest_pixels = rest_pixels, * paragraphs concurrently - that's the correct granularity.
.fitness = fitness, */
.hyphen_counts = hyphen_counts, dp_work_t work = {
.line_counts = line_counts, .para = p,
.prev_demerits = demerits, .line_width = line_width,
.prev_fitness = fitness, .start = 0,
.prev_hyphen_counts = hyphen_counts, .end = n,
.prev_line_counts = line_counts, .demerits = demerits,
.line_penalty = line_penalty, .backptrs = backptrs,
.hyphen_penalty = hyphen_penalty, .rest_pixels = rest_pixels,
.fitness_penalty = fitness_penalty, .fitness = fitness,
.last_line_ratio = last_ratio, .hyphen_counts = hyphen_counts,
}; .line_counts = line_counts,
.prev_demerits = demerits,
.prev_fitness = fitness,
.prev_hyphen_counts = hyphen_counts,
.prev_line_counts = line_counts,
.line_penalty = line_penalty,
.hyphen_penalty = hyphen_penalty,
.fitness_penalty = fitness_penalty,
.last_line_ratio = last_ratio,
};
/* Simple iterative DP */ /* Iterative DP: O(n²) worst case, typically O(n·m) with early termination */
for (size_t i = 0; i < n; i++) { for (size_t i = 0; i < n; i++) {
if (demerits[i] >= EKP_INFINITY) if (demerits[i] >= EKP_INFINITY)
continue; continue;
work.start = i; work.start = i;
work.end = i + 1; work.end = i + 1;
process_dp_range(&work); process_dp_range(&work);
}
} else {
/* Parallel processing for large paragraphs */
/* Split work across threads (wavefront approach) */
size_t chunk_size = n / EKP_THREAD_POOL_SIZE;
if (chunk_size < 10)
chunk_size = 10;
dp_work_t *works = malloc(EKP_THREAD_POOL_SIZE * sizeof(dp_work_t));
if (!works) {
/* Fall back to single-threaded */
for (size_t i = 0; i < n; i++) {
if (demerits[i] >= EKP_INFINITY)
continue;
dp_work_t work = {
.para = p,
.line_width = line_width,
.start = i,
.end = i + 1,
.demerits = demerits,
.backptrs = backptrs,
.rest_pixels = rest_pixels,
.fitness = fitness,
.hyphen_counts = hyphen_counts,
.line_counts = line_counts,
.prev_demerits = demerits,
.prev_fitness = fitness,
.prev_hyphen_counts = hyphen_counts,
.prev_line_counts = line_counts,
.line_penalty = line_penalty,
.hyphen_penalty = hyphen_penalty,
.fitness_penalty = fitness_penalty,
.last_line_ratio = last_ratio,
};
process_dp_range(&work);
}
} else {
/* Wavefront: process in chunks */
for (size_t wave = 0; wave < n; wave += chunk_size) {
size_t wave_end = wave + chunk_size;
if (wave_end > n)
wave_end = n;
size_t work_count = 0;
for (size_t i = wave; i < wave_end; i++) {
if (demerits[i] >= EKP_INFINITY)
continue;
works[work_count] = (dp_work_t){
.para = p,
.line_width = line_width,
.start = i,
.end = i + 1,
.demerits = demerits,
.backptrs = backptrs,
.rest_pixels = rest_pixels,
.fitness = fitness,
.hyphen_counts = hyphen_counts,
.line_counts = line_counts,
.prev_demerits = demerits,
.prev_fitness = fitness,
.prev_hyphen_counts = hyphen_counts,
.prev_line_counts = line_counts,
.line_penalty = line_penalty,
.hyphen_penalty = hyphen_penalty,
.fitness_penalty = fitness_penalty,
.last_line_ratio = last_ratio,
};
ekp_pool_submit(ekp_global->pool, process_dp_range,
&works[work_count]);
work_count++;
}
ekp_pool_wait(ekp_global->pool);
}
free(works);
}
} }
/* Trace back optimal path */ /* Trace back optimal path */
@ -685,6 +616,91 @@ ekp_result_t *ekp_break_with_prefixes(
return result; return result;
} }
/*
* Work item for batch processing
*/
typedef struct {
ekp_batch_input_t *input;
ekp_result_t *result;
} batch_work_t;
static void batch_worker(void *arg)
{
batch_work_t *work = (batch_work_t *)arg;
ekp_batch_input_t *in = work->input;
work->result = ekp_break_with_prefixes(
in->ideal_prefix, in->min_prefix, in->max_prefix,
in->glue_ideals, in->glue_shrinks, in->glue_stretches,
in->n,
in->hyphen_positions, in->hyphen_count,
in->hyphen_width, in->line_width);
}
/*
* Batch line breaking: process multiple paragraphs in parallel
*
* This is the correct parallelization - each paragraph is completely
* independent, so we get linear speedup with zero synchronization overhead.
*/
ekp_result_t **ekp_break_batch(ekp_batch_input_t *inputs, size_t count)
{
if (!inputs || count == 0)
return NULL;
ekp_result_t **results = calloc(count, sizeof(ekp_result_t *));
if (!results)
return NULL;
/* Single paragraph: no point using threads */
if (count == 1 || !ekp_global || !ekp_global->pool) {
for (size_t i = 0; i < count; i++) {
ekp_batch_input_t *in = &inputs[i];
results[i] = ekp_break_with_prefixes(
in->ideal_prefix, in->min_prefix, in->max_prefix,
in->glue_ideals, in->glue_shrinks, in->glue_stretches,
in->n,
in->hyphen_positions, in->hyphen_count,
in->hyphen_width, in->line_width);
}
return results;
}
/* Multiple paragraphs: parallel processing */
batch_work_t *works = malloc(count * sizeof(batch_work_t));
if (!works) {
/* Fallback to sequential */
for (size_t i = 0; i < count; i++) {
ekp_batch_input_t *in = &inputs[i];
results[i] = ekp_break_with_prefixes(
in->ideal_prefix, in->min_prefix, in->max_prefix,
in->glue_ideals, in->glue_shrinks, in->glue_stretches,
in->n,
in->hyphen_positions, in->hyphen_count,
in->hyphen_width, in->line_width);
}
return results;
}
/* Submit all work items */
for (size_t i = 0; i < count; i++) {
works[i].input = &inputs[i];
works[i].result = NULL;
ekp_pool_submit(ekp_global->pool, batch_worker, &works[i]);
}
/* Wait for all to complete */
ekp_pool_wait(ekp_global->pool);
/* Collect results */
for (size_t i = 0; i < count; i++) {
results[i] = works[i].result;
}
free(works);
return results;
}
/* /*
* Initialization and cleanup * Initialization and cleanup
*/ */

View File

@ -244,6 +244,36 @@ ekp_result_t *ekp_break_with_prefixes(
int32_t hyphen_width, int32_t hyphen_width,
int32_t line_width); int32_t line_width);
/*
* Batch input for parallel processing
*/
typedef struct {
const int32_t *ideal_prefix;
const int32_t *min_prefix;
const int32_t *max_prefix;
const int32_t *glue_ideals;
const int32_t *glue_shrinks;
const int32_t *glue_stretches;
size_t n;
const int32_t *hyphen_positions;
size_t hyphen_count;
int32_t hyphen_width;
int32_t line_width;
} ekp_batch_input_t;
/*
* API: Batch line breaking (parallel across paragraphs)
*
* Processes multiple paragraphs concurrently using the thread pool.
* This is the correct parallelization granularity - paragraphs are
* independent, so no synchronization overhead.
*
* Returns array of results (caller must free each result and the array).
*/
ekp_result_t **ekp_break_batch(
ekp_batch_input_t *inputs,
size_t count);
/* /*
* API: Thread pool * API: Thread pool
*/ */