diff options
| author | tslil clingman <tslil@posteo.de> | 2021-01-21 01:18:01 -0500 |
|---|---|---|
| committer | tslil <tslil@posteo.de> | 2026-08-28 19:37:41 +0100 |
| commit | 19359dde885243e2359f7c560ae801efccee7294 (patch) | |
| tree | b27bc6f934e05b6494af8527b6b55bc554dc4c10 /include/negamax_cnn1986.c | |
| parent | 2ebcba8c471678942d0cbf496b79704df9bafff3 (diff) | |
| parent | 9fa3291044ff8d9f0f2b9c9a01cd210bf318ab74 (diff) | |
Merge branch 'negamax'
Diffstat (limited to 'include/negamax_cnn1986.c')
| -rw-r--r-- | include/negamax_cnn1986.c | 389 |
1 files changed, 389 insertions, 0 deletions
diff --git a/include/negamax_cnn1986.c b/include/negamax_cnn1986.c new file mode 100644 index 0000000..fafb38c --- /dev/null +++ b/include/negamax_cnn1986.c @@ -0,0 +1,389 @@ +#include "negamax_cnn1986.h" + +// =================================================================== +// Globals +// =================================================================== + +const float infty = 3.0; +char ct1986_ptn[9]; +uint8_t ct1986_search_depth = 3; + +// =================================================================== +// Implementation of a small convolutional neural network +// =================================================================== + +static float flattened[CONV_NUM+2]; +static float dense1[DENSE1_NUM]; +static float dense2[DENSE2_NUM]; + +#ifndef DETERMINISTIC +union u_f { + uint32_t u; + float f; +}; + +static uint32_t state = 1; +static union u_f fudge; + +#define DOXORSHIFT { \ + state ^= state << 13; \ + state ^= state >> 17; \ + state ^= state << 5; \ + fudge.u = 0x3f800000 | state >> 10; \ + fudge.f = (fudge.f - 1.5) * 0.01; \ + } +#endif + +#define RELU(x) ((x) = ((x)<0)?0:(x)) + +float +ct1986_evaluate_black_win(void) { + /* ------------------ * + * Convolution layer * + * ------------------ */ + // for each kernel + for (uint8_t kern = 0; kern < KERN_NUM; kern++) { + // the stride is 1, march across the board + for (uint8_t bx = 0; bx < KERN_OSIZE; bx++) { + for (uint8_t by = 0; by < KERN_OSIZE; by++) { + flattened[kern+KERN_NUM*(bx+by*KERN_OSIZE)] = + conv2d_biases[kern]; + // Compute the convolution for this position + for (uint8_t ky = 0; ky < KERN_SIZE; ky++) { + for (uint8_t kx = 0; kx < KERN_SIZE; kx++) { + for (uint8_t c = 0; c < KERN_CHAN; c++) { + // Where we are on the board + const uint8_t loc = kx+bx+(ky+by)*5; + // Look up what's on the board at this location, and + // multiply it. For c=0 we have to do some extra work + float lookup = 0; + if (COUNT_AT(loc)>c) { + if (c==0) { + if (STONE_AT(loc) == STONE_STANDING) { + lookup = (colours[loc] & 1) ? +0.25 : -0.25; + } else if (STONE_AT(loc) == STONE_CAPSTONE) { + lookup = (colours[loc] & 1) ? +1.00 : -1.00; + } else { + lookup = (colours[loc] & 1) ? +0.50 : -0.50; + } + } else { + lookup = (colours[loc] & (1<<c)) ? +0.50 : -0.50; + } + } + flattened[kern+KERN_NUM*(bx+by*KERN_OSIZE)] + += lookup*conv2d_weights[kern][ky][kx][c]; + } + } + } + RELU(flattened[kern+KERN_NUM*(bx+by*KERN_OSIZE)]); + } + } + } + // Add input of flat counts + flattened[CONV_NUM] = (float)(white_count & 127)/21.0; + flattened[CONV_NUM+1] = (float)(black_count & 127)/21.0; + /* ------------------ * + * First dense layer * + * ------------------ */ + for (uint8_t d1 = 0; d1 < DENSE1_NUM; d1++) { + dense1[d1] = dense1_biases[d1]; + for (uint8_t fl = 0; fl < CONV_NUM+2; fl++) { + dense1[d1] += flattened[fl]*dense1_weights[d1][fl]; + } + RELU(dense1[d1]); + } + /* ------------------- * + * Second dense layer * + * ------------------- */ + for (uint8_t d2 = 0; d2 < DENSE2_NUM; d2++) { + dense2[d2] = dense2_biases[d2]; + for (uint8_t d1 = 0; d1 < DENSE1_NUM; d1++) { + dense2[d2] += dense1[d1]*dense2_weights[d2][d1]; + } + RELU(dense2[d2]); + } + /* ------------- * + * Output layer * + * ------------- */ + float output = output_bias; + for (uint8_t d2 = 0; d2 < DENSE2_NUM; d2++) { + output += dense2[d2]*output_weights[d2]; + } + // Truncated Pade approximant of logistic function + output = (12.0+output+50.0*output/(output*output+10.0))/24.0; +#ifndef DETERMINISTIC + DOXORSHIFT; + output += fudge.f; +#endif + if (output > 1.0) { + return 1.0; + } + else if (output < 0.0) { + return -1.0; + } + return 2*output-1.0; +} + +// =================================================================== +// α-β negamax using the above evaluator +// =================================================================== + +static void +previous_ply(void) { + if (ply>0) ply--; + if (ply == 1) { + current_colour = C_WHITE; + } else { + if (current_colour == C_BLACK) current_colour = C_WHITE; + else current_colour = C_BLACK; + } +} + +static inline void +push_stones(const int8_t location, const uint8_t count, + const uint8_t new_colours, + const enum STONE_VARIANT top_stone) { + colours[location] = (colours[location] << count) | new_colours; + celldat[location] = top_stone + | ((celldat[location] + ((count << NUM_SHIFT))) & NUM_MASK); +} + +static float val; +static enum WIN_TYPE w; + +// UP DOWN LEFT RIGHT +static const int8_t deltas[4] = { +5, -5, -1, +1}; + +#define WIN_EVALUATE_OR_RECURSE(store,reset) { \ + w = 0xFF; \ + if (ply >= 2*5 - 2) w = check_win(); \ + if (w < 0xFF) { \ + /* Somebody won, assign weights accordingly. Note in particular + that draws are only worth ∞/2 ;) + */ \ + if (w == WIN_ROAD_BLACK || w == WIN_FLAT_BLACK) \ + val = infty; \ + else if (w == WIN_DRAW) val = infty/2.0; \ + else val = -infty; \ + val *= colour; \ + } else if (cur_depth == ct1986_search_depth) { \ + /* We're at the bottom, evaluate */ \ + val = colour * ct1986_evaluate_black_win(); \ + } else { \ + /* We're not at the bottom, recurse first */ \ + next_ply(); \ + val = -ct1986_negamax(cur_depth + 1, -beta, -alpha, -colour); \ + previous_ply(); \ + } \ + { reset }; \ + /* Prune */ \ + if (val >= beta) return beta; \ + /* Update the optimal value, which alpha carries */ \ + if (val > alpha) { \ + alpha = val; \ + if (cur_depth == 0) { store }; \ + } \ + } + +float +ct1986_negamax(const uint8_t cur_depth, float alpha, float beta, + const float colour) { + const uint8_t black = (ply & 1), + material = (black) ? black_count : white_count, + flat = material & 127, + cap = (ply > 2 && (material & 128)), + standing = (ply > 2 && (material & 127)); + + // Step across the board + for (uint8_t row = 0; row < 5; row++) { + for (uint8_t col = 0; col < 5; col++) { + // Try all valid actions for this square. Is it empty? + const uint8_t loc = THE_COORDS(col, row); + const uint8_t count = (COUNT_AT(loc) > 5) ? 5 : COUNT_AT(loc); + // Only try moves after CPS + if (count && ((colours[loc] & 1) == current_colour) && ply>2) { + // There are stones, can we move them in a given direction? + + // Pre-compute end-stops + uint8_t end_stops[4][2]; // (end, not_crush) + // UP DOWN LEFT RIGHT + end_stops[0][0] = (4-row > count) ? count : 4-row; + end_stops[1][0] = (row > count) ? count : row; + end_stops[2][0] = (col > count) ? count : col; + end_stops[3][0] = (4-col > count) ? count : 4-col; + const uint8_t cap_top = STONE_AT(loc) == STONE_CAPSTONE; + for (uint8_t d = 0; d < 4; d++){ + end_stops[d][1] = 1; + const uint8_t stop = end_stops[d][0]; + end_stops[d][0] = 0; + for (uint8_t k = 1; k <= stop; k++) { + const uint8_t stone = STONE_AT(loc+k*deltas[d]); + if (stone == STONE_STANDING) { + if (cap_top) { + end_stops[d][1] = 0; + end_stops[d][0]++; + } + break; + } else if (stone == STONE_CAPSTONE) { + break; + } + end_stops[d][0]++; + } + } + + uint16_t colours_backup[5]; + uint8_t celldat_backup[5], drops[5]; // we only use 4, the + // fifth is to skip a + // bounds check at (*) + // Back up the row of the board + for (uint8_t y = 0; y < 5; y++) { + colours_backup[y] = colours[THE_COORDS(col, y)]; + celldat_backup[y] = celldat[THE_COORDS(col, y)]; + } + + // I'm not a huge fan of looping through enums, but it's + // better than manually unrolling this. Sufficiently smart + // compilers? + for (enum MOVE_DIRECTION dir = M_UP; dir <= M_RIGHT; dir++) { + // Back-up the column once we start looking horizontally + if (dir == M_LEFT) { + for (uint8_t x = 0; x < 5; x++) { + colours_backup[x] = colours[THE_COORDS(x, row)]; + celldat_backup[x] = celldat[THE_COORDS(x, row)]; + } + } + /* + * We don't do anything terribly efficient here just try + * all the ordered partitions of num ∈ {1 … end_stop}, and + * skip the partition if it calls for multiple stones at + * the end with a crush. + */ + uint8_t gaps, t, idx, mask; + for (uint8_t num = 1; num <= count; num++) { + for (uint8_t steps = 1; + steps <= end_stops[dir][0] && steps <= num; + steps++) { + gaps = 0b00000111 >> (4-steps); + do { + // Ensure legal move if we have to crush + const uint8_t last_drop_check = (num > 1) ? (gaps & 1<<(num - 2)) : 1; + if (end_stops[dir][1] || last_drop_check) { + // Translate to a drop sequence + drops[0] = 1; mask = 1; idx = 0; + for (uint8_t d = 0; d + 1 < num; d++) { + if (gaps & mask) { + idx++; + drops[idx] = 1; // (*) we don't need to bounds check + } else { + drops[idx] += 1; + } + mask <<= 1; + } + // Do it, and manually check for win if it's valid + uint8_t j = num; + for (uint8_t k = 0; k < steps; k++) { + j -= drops[k]; + push_stones(loc+(k+1)*deltas[dir], + drops[k], + (colours[loc] >> j) & (0xFFFF >> (0x10 - drops[k])), + (k == steps - 1) ? STONE_AT(loc) : STONE_FLAT); + } + // Then we drop them from the source + colours[loc] >>= num; + const uint8_t dec_count = celldat[loc] - (num << NUM_SHIFT); + celldat[loc] = dec_count & NUM_MASK; + + // First check for wins, if we're at the bottom + // evaluate, otherwise recurse + WIN_EVALUATE_OR_RECURSE({ + // If we did update the optimal value, store + // this move + generate_move(loc, dir, steps, drops, ct1986_ptn); + },{ + // Reset the board data after recursing or + // before returning + if (dir <= M_DOWN) { + for (uint8_t y = 0; y < 5; y++) { + colours[THE_COORDS(col, y)] = colours_backup[y]; + celldat[THE_COORDS(col, y)] = celldat_backup[y]; + } + } else { + for (uint8_t x = 0; x < 5; x++) { + colours[THE_COORDS(x, row)] = colours_backup[x]; + celldat[THE_COORDS(x, row)] = celldat_backup[x]; + } + } + }); + } + /* + * With thanks to + * https://graphics.stanford.edu/~seander/bithacks.html#NextBitPermutation + * we have the following magic to generate the next + * permutation of steps-many set bits + */ + t = (gaps | (gaps - 1)); + gaps = (t + 1) | (((~t & -~t) - 1) >> (__builtin_ctz(gaps) + 1)); + } while (gaps && (gaps + 1 <= (1<<(num-1)))); + } + } + } + } else if (material && count == 0) { + // Empty square, try placements + + if (flat) { + // Generate the placement + if (black) black_count--; + else white_count--; + colours[loc] = current_colour; + celldat[loc] = NUM_INC | STONE_FLAT; + WIN_EVALUATE_OR_RECURSE({ + // If we did update the optimal value, store + generate_place(loc, STONE_FLAT, ct1986_ptn); + },{ + // Reset the state + celldat[loc] = 0; + if (black) black_count++; + else white_count++; + }); + + // Do the same for walls, can't happen without flats + if (standing) { + if (black) black_count--; + else white_count--; + colours[loc] = current_colour; + celldat[loc] = NUM_INC | STONE_STANDING; + WIN_EVALUATE_OR_RECURSE({ + generate_place(loc, STONE_STANDING, ct1986_ptn); + },{ + celldat[loc] = 0; + if (black) black_count++; + else white_count++; + }); + } + } + + // and for caps + if (cap) { + if (black) black_count &= 127; + else white_count &= 127; + colours[loc] = current_colour; + celldat[loc] = NUM_INC | STONE_CAPSTONE; + WIN_EVALUATE_OR_RECURSE({ + generate_place(loc, STONE_CAPSTONE, ct1986_ptn); + },{ + celldat[loc] = 0; + if (black) black_count |= 128; + else white_count |= 128; + }); + } + } + ct1986_display_progress(cur_depth); + } + } + return alpha; +} + +inline float +ct1986_generate(void) { + return ct1986_negamax(0, -infty, infty, (ply&1)?1.0:-1.0); +} |
