aboutsummaryrefslogtreecommitdiff
path: root/include/negamax_cnn1986.c
diff options
context:
space:
mode:
authortslil clingman <tslil@posteo.de>2021-01-21 01:18:01 -0500
committertslil <tslil@posteo.de>2026-08-28 19:37:41 +0100
commit19359dde885243e2359f7c560ae801efccee7294 (patch)
treeb27bc6f934e05b6494af8527b6b55bc554dc4c10 /include/negamax_cnn1986.c
parent2ebcba8c471678942d0cbf496b79704df9bafff3 (diff)
parent9fa3291044ff8d9f0f2b9c9a01cd210bf318ab74 (diff)
Merge branch 'negamax'
Diffstat (limited to 'include/negamax_cnn1986.c')
-rw-r--r--include/negamax_cnn1986.c389
1 files changed, 389 insertions, 0 deletions
diff --git a/include/negamax_cnn1986.c b/include/negamax_cnn1986.c
new file mode 100644
index 0000000..fafb38c
--- /dev/null
+++ b/include/negamax_cnn1986.c
@@ -0,0 +1,389 @@
+#include "negamax_cnn1986.h"
+
+// ===================================================================
+// Globals
+// ===================================================================
+
+const float infty = 3.0;
+char ct1986_ptn[9];
+uint8_t ct1986_search_depth = 3;
+
+// ===================================================================
+// Implementation of a small convolutional neural network
+// ===================================================================
+
+static float flattened[CONV_NUM+2];
+static float dense1[DENSE1_NUM];
+static float dense2[DENSE2_NUM];
+
+#ifndef DETERMINISTIC
+union u_f {
+ uint32_t u;
+ float f;
+};
+
+static uint32_t state = 1;
+static union u_f fudge;
+
+#define DOXORSHIFT { \
+ state ^= state << 13; \
+ state ^= state >> 17; \
+ state ^= state << 5; \
+ fudge.u = 0x3f800000 | state >> 10; \
+ fudge.f = (fudge.f - 1.5) * 0.01; \
+ }
+#endif
+
+#define RELU(x) ((x) = ((x)<0)?0:(x))
+
+float
+ct1986_evaluate_black_win(void) {
+ /* ------------------ *
+ * Convolution layer *
+ * ------------------ */
+ // for each kernel
+ for (uint8_t kern = 0; kern < KERN_NUM; kern++) {
+ // the stride is 1, march across the board
+ for (uint8_t bx = 0; bx < KERN_OSIZE; bx++) {
+ for (uint8_t by = 0; by < KERN_OSIZE; by++) {
+ flattened[kern+KERN_NUM*(bx+by*KERN_OSIZE)] =
+ conv2d_biases[kern];
+ // Compute the convolution for this position
+ for (uint8_t ky = 0; ky < KERN_SIZE; ky++) {
+ for (uint8_t kx = 0; kx < KERN_SIZE; kx++) {
+ for (uint8_t c = 0; c < KERN_CHAN; c++) {
+ // Where we are on the board
+ const uint8_t loc = kx+bx+(ky+by)*5;
+ // Look up what's on the board at this location, and
+ // multiply it. For c=0 we have to do some extra work
+ float lookup = 0;
+ if (COUNT_AT(loc)>c) {
+ if (c==0) {
+ if (STONE_AT(loc) == STONE_STANDING) {
+ lookup = (colours[loc] & 1) ? +0.25 : -0.25;
+ } else if (STONE_AT(loc) == STONE_CAPSTONE) {
+ lookup = (colours[loc] & 1) ? +1.00 : -1.00;
+ } else {
+ lookup = (colours[loc] & 1) ? +0.50 : -0.50;
+ }
+ } else {
+ lookup = (colours[loc] & (1<<c)) ? +0.50 : -0.50;
+ }
+ }
+ flattened[kern+KERN_NUM*(bx+by*KERN_OSIZE)]
+ += lookup*conv2d_weights[kern][ky][kx][c];
+ }
+ }
+ }
+ RELU(flattened[kern+KERN_NUM*(bx+by*KERN_OSIZE)]);
+ }
+ }
+ }
+ // Add input of flat counts
+ flattened[CONV_NUM] = (float)(white_count & 127)/21.0;
+ flattened[CONV_NUM+1] = (float)(black_count & 127)/21.0;
+ /* ------------------ *
+ * First dense layer *
+ * ------------------ */
+ for (uint8_t d1 = 0; d1 < DENSE1_NUM; d1++) {
+ dense1[d1] = dense1_biases[d1];
+ for (uint8_t fl = 0; fl < CONV_NUM+2; fl++) {
+ dense1[d1] += flattened[fl]*dense1_weights[d1][fl];
+ }
+ RELU(dense1[d1]);
+ }
+ /* ------------------- *
+ * Second dense layer *
+ * ------------------- */
+ for (uint8_t d2 = 0; d2 < DENSE2_NUM; d2++) {
+ dense2[d2] = dense2_biases[d2];
+ for (uint8_t d1 = 0; d1 < DENSE1_NUM; d1++) {
+ dense2[d2] += dense1[d1]*dense2_weights[d2][d1];
+ }
+ RELU(dense2[d2]);
+ }
+ /* ------------- *
+ * Output layer *
+ * ------------- */
+ float output = output_bias;
+ for (uint8_t d2 = 0; d2 < DENSE2_NUM; d2++) {
+ output += dense2[d2]*output_weights[d2];
+ }
+ // Truncated Pade approximant of logistic function
+ output = (12.0+output+50.0*output/(output*output+10.0))/24.0;
+#ifndef DETERMINISTIC
+ DOXORSHIFT;
+ output += fudge.f;
+#endif
+ if (output > 1.0) {
+ return 1.0;
+ }
+ else if (output < 0.0) {
+ return -1.0;
+ }
+ return 2*output-1.0;
+}
+
+// ===================================================================
+// α-β negamax using the above evaluator
+// ===================================================================
+
+static void
+previous_ply(void) {
+ if (ply>0) ply--;
+ if (ply == 1) {
+ current_colour = C_WHITE;
+ } else {
+ if (current_colour == C_BLACK) current_colour = C_WHITE;
+ else current_colour = C_BLACK;
+ }
+}
+
+static inline void
+push_stones(const int8_t location, const uint8_t count,
+ const uint8_t new_colours,
+ const enum STONE_VARIANT top_stone) {
+ colours[location] = (colours[location] << count) | new_colours;
+ celldat[location] = top_stone
+ | ((celldat[location] + ((count << NUM_SHIFT))) & NUM_MASK);
+}
+
+static float val;
+static enum WIN_TYPE w;
+
+// UP DOWN LEFT RIGHT
+static const int8_t deltas[4] = { +5, -5, -1, +1};
+
+#define WIN_EVALUATE_OR_RECURSE(store,reset) { \
+ w = 0xFF; \
+ if (ply >= 2*5 - 2) w = check_win(); \
+ if (w < 0xFF) { \
+ /* Somebody won, assign weights accordingly. Note in particular
+ that draws are only worth ∞/2 ;)
+ */ \
+ if (w == WIN_ROAD_BLACK || w == WIN_FLAT_BLACK) \
+ val = infty; \
+ else if (w == WIN_DRAW) val = infty/2.0; \
+ else val = -infty; \
+ val *= colour; \
+ } else if (cur_depth == ct1986_search_depth) { \
+ /* We're at the bottom, evaluate */ \
+ val = colour * ct1986_evaluate_black_win(); \
+ } else { \
+ /* We're not at the bottom, recurse first */ \
+ next_ply(); \
+ val = -ct1986_negamax(cur_depth + 1, -beta, -alpha, -colour); \
+ previous_ply(); \
+ } \
+ { reset }; \
+ /* Prune */ \
+ if (val >= beta) return beta; \
+ /* Update the optimal value, which alpha carries */ \
+ if (val > alpha) { \
+ alpha = val; \
+ if (cur_depth == 0) { store }; \
+ } \
+ }
+
+float
+ct1986_negamax(const uint8_t cur_depth, float alpha, float beta,
+ const float colour) {
+ const uint8_t black = (ply & 1),
+ material = (black) ? black_count : white_count,
+ flat = material & 127,
+ cap = (ply > 2 && (material & 128)),
+ standing = (ply > 2 && (material & 127));
+
+ // Step across the board
+ for (uint8_t row = 0; row < 5; row++) {
+ for (uint8_t col = 0; col < 5; col++) {
+ // Try all valid actions for this square. Is it empty?
+ const uint8_t loc = THE_COORDS(col, row);
+ const uint8_t count = (COUNT_AT(loc) > 5) ? 5 : COUNT_AT(loc);
+ // Only try moves after CPS
+ if (count && ((colours[loc] & 1) == current_colour) && ply>2) {
+ // There are stones, can we move them in a given direction?
+
+ // Pre-compute end-stops
+ uint8_t end_stops[4][2]; // (end, not_crush)
+ // UP DOWN LEFT RIGHT
+ end_stops[0][0] = (4-row > count) ? count : 4-row;
+ end_stops[1][0] = (row > count) ? count : row;
+ end_stops[2][0] = (col > count) ? count : col;
+ end_stops[3][0] = (4-col > count) ? count : 4-col;
+ const uint8_t cap_top = STONE_AT(loc) == STONE_CAPSTONE;
+ for (uint8_t d = 0; d < 4; d++){
+ end_stops[d][1] = 1;
+ const uint8_t stop = end_stops[d][0];
+ end_stops[d][0] = 0;
+ for (uint8_t k = 1; k <= stop; k++) {
+ const uint8_t stone = STONE_AT(loc+k*deltas[d]);
+ if (stone == STONE_STANDING) {
+ if (cap_top) {
+ end_stops[d][1] = 0;
+ end_stops[d][0]++;
+ }
+ break;
+ } else if (stone == STONE_CAPSTONE) {
+ break;
+ }
+ end_stops[d][0]++;
+ }
+ }
+
+ uint16_t colours_backup[5];
+ uint8_t celldat_backup[5], drops[5]; // we only use 4, the
+ // fifth is to skip a
+ // bounds check at (*)
+ // Back up the row of the board
+ for (uint8_t y = 0; y < 5; y++) {
+ colours_backup[y] = colours[THE_COORDS(col, y)];
+ celldat_backup[y] = celldat[THE_COORDS(col, y)];
+ }
+
+ // I'm not a huge fan of looping through enums, but it's
+ // better than manually unrolling this. Sufficiently smart
+ // compilers?
+ for (enum MOVE_DIRECTION dir = M_UP; dir <= M_RIGHT; dir++) {
+ // Back-up the column once we start looking horizontally
+ if (dir == M_LEFT) {
+ for (uint8_t x = 0; x < 5; x++) {
+ colours_backup[x] = colours[THE_COORDS(x, row)];
+ celldat_backup[x] = celldat[THE_COORDS(x, row)];
+ }
+ }
+ /*
+ * We don't do anything terribly efficient here just try
+ * all the ordered partitions of num ∈ {1 … end_stop}, and
+ * skip the partition if it calls for multiple stones at
+ * the end with a crush.
+ */
+ uint8_t gaps, t, idx, mask;
+ for (uint8_t num = 1; num <= count; num++) {
+ for (uint8_t steps = 1;
+ steps <= end_stops[dir][0] && steps <= num;
+ steps++) {
+ gaps = 0b00000111 >> (4-steps);
+ do {
+ // Ensure legal move if we have to crush
+ const uint8_t last_drop_check = (num > 1) ? (gaps & 1<<(num - 2)) : 1;
+ if (end_stops[dir][1] || last_drop_check) {
+ // Translate to a drop sequence
+ drops[0] = 1; mask = 1; idx = 0;
+ for (uint8_t d = 0; d + 1 < num; d++) {
+ if (gaps & mask) {
+ idx++;
+ drops[idx] = 1; // (*) we don't need to bounds check
+ } else {
+ drops[idx] += 1;
+ }
+ mask <<= 1;
+ }
+ // Do it, and manually check for win if it's valid
+ uint8_t j = num;
+ for (uint8_t k = 0; k < steps; k++) {
+ j -= drops[k];
+ push_stones(loc+(k+1)*deltas[dir],
+ drops[k],
+ (colours[loc] >> j) & (0xFFFF >> (0x10 - drops[k])),
+ (k == steps - 1) ? STONE_AT(loc) : STONE_FLAT);
+ }
+ // Then we drop them from the source
+ colours[loc] >>= num;
+ const uint8_t dec_count = celldat[loc] - (num << NUM_SHIFT);
+ celldat[loc] = dec_count & NUM_MASK;
+
+ // First check for wins, if we're at the bottom
+ // evaluate, otherwise recurse
+ WIN_EVALUATE_OR_RECURSE({
+ // If we did update the optimal value, store
+ // this move
+ generate_move(loc, dir, steps, drops, ct1986_ptn);
+ },{
+ // Reset the board data after recursing or
+ // before returning
+ if (dir <= M_DOWN) {
+ for (uint8_t y = 0; y < 5; y++) {
+ colours[THE_COORDS(col, y)] = colours_backup[y];
+ celldat[THE_COORDS(col, y)] = celldat_backup[y];
+ }
+ } else {
+ for (uint8_t x = 0; x < 5; x++) {
+ colours[THE_COORDS(x, row)] = colours_backup[x];
+ celldat[THE_COORDS(x, row)] = celldat_backup[x];
+ }
+ }
+ });
+ }
+ /*
+ * With thanks to
+ * https://graphics.stanford.edu/~seander/bithacks.html#NextBitPermutation
+ * we have the following magic to generate the next
+ * permutation of steps-many set bits
+ */
+ t = (gaps | (gaps - 1));
+ gaps = (t + 1) | (((~t & -~t) - 1) >> (__builtin_ctz(gaps) + 1));
+ } while (gaps && (gaps + 1 <= (1<<(num-1))));
+ }
+ }
+ }
+ } else if (material && count == 0) {
+ // Empty square, try placements
+
+ if (flat) {
+ // Generate the placement
+ if (black) black_count--;
+ else white_count--;
+ colours[loc] = current_colour;
+ celldat[loc] = NUM_INC | STONE_FLAT;
+ WIN_EVALUATE_OR_RECURSE({
+ // If we did update the optimal value, store
+ generate_place(loc, STONE_FLAT, ct1986_ptn);
+ },{
+ // Reset the state
+ celldat[loc] = 0;
+ if (black) black_count++;
+ else white_count++;
+ });
+
+ // Do the same for walls, can't happen without flats
+ if (standing) {
+ if (black) black_count--;
+ else white_count--;
+ colours[loc] = current_colour;
+ celldat[loc] = NUM_INC | STONE_STANDING;
+ WIN_EVALUATE_OR_RECURSE({
+ generate_place(loc, STONE_STANDING, ct1986_ptn);
+ },{
+ celldat[loc] = 0;
+ if (black) black_count++;
+ else white_count++;
+ });
+ }
+ }
+
+ // and for caps
+ if (cap) {
+ if (black) black_count &= 127;
+ else white_count &= 127;
+ colours[loc] = current_colour;
+ celldat[loc] = NUM_INC | STONE_CAPSTONE;
+ WIN_EVALUATE_OR_RECURSE({
+ generate_place(loc, STONE_CAPSTONE, ct1986_ptn);
+ },{
+ celldat[loc] = 0;
+ if (black) black_count |= 128;
+ else white_count |= 128;
+ });
+ }
+ }
+ ct1986_display_progress(cur_depth);
+ }
+ }
+ return alpha;
+}
+
+inline float
+ct1986_generate(void) {
+ return ct1986_negamax(0, -infty, infty, (ply&1)?1.0:-1.0);
+}