1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
|
#include "negamax.h"
// ===================================================================
// Globals
// ===================================================================
const float infty = 3.0;
char negamax_ptn[9];
uint8_t negamax_search_depth = 3;
// ===================================================================
// Helpers
// ===================================================================
static float
negamax(const uint8_t cur_depth, float alpha, float beta,
const float colour);
// ===================================================================
// α-β negamax using the cnn1986 evaluation function and transposition
// tables using Zobrist hasing and a treap
// ===================================================================
void
negamax_init(const uint8_t new_board_size) {
board_size = new_board_size;
action_list_init();
zobrist_init();
tt_init();
}
void
negamax_free(void) {
zobrist_free();
}
float
negamax_generate(void) {
// We need to start with something outside of [-∞,∞] because those
// values are wins
const float safe_infty = infty + 1;
tt_init();
float result = negamax(negamax_search_depth,
-safe_infty, safe_infty,
(ply & 1) ? +1.0 : -1.0);
tt_free();
return result;
}
static enum TT_FLAG flag;
static enum WIN_TYPE w;
static float
negamax(const uint8_t cur_depth, float alpha, float beta,
const float colour) {
uint64_t hash = zobrist_compute();
tt_entry_t *entry = tt_seek(hash);
// CAUTION: >= breaks search stability
if (entry != NULL && entry->depth >= cur_depth) {
if (entry->flag == TT_EXACT) {
return entry->value;
} else if (entry->flag == TT_LOWERBOUND && entry->value > alpha) {
alpha = entry->value;
} else if (entry->flag == TT_UPPERBOUND && entry->value < beta) {
beta = entry->value;
}
if (alpha >= beta) return entry->value;
}
action_list_t *list;
if ((list = action_list_generate()) == NULL)
return alpha; // should never happen!
if (entry != NULL) {
action_move_to_front(entry->action, list);
}
// TODO: what to do if this is never written to?
action_t best_action = list->head->action;
float best_value = -infty;
for (action_node_t *node=list->head; node!=NULL; node=node->next) {
action_take(node->action);
// Compute the value of the node
float node_value;
if (ply >= 2*board_size - 2 && (w = check_win()) < 0xFF) {
node_value = -colour*infty;
// Check win if far enough into the game
if (w == WIN_ROAD_BLACK || w == WIN_FLAT_BLACK) {
node_value = colour*infty;
} else if (w == WIN_DRAW) {
node_value = 0;
}
} else if (cur_depth > 1) {
// If nobody won, or too early and not leaf, recurse
node_value = -negamax(cur_depth - 1, -beta, -alpha, -colour);
} else {
node_value = colour * cnn1986_evaluate_black_win();
}
action_undo(node->action);
negamax_display_progress(cur_depth, list->length);
if (node_value > best_value) {
best_value = node_value;
best_action = node->action;
if (cur_depth == negamax_search_depth)
action_to_ptn(node->action, negamax_ptn);
}
alpha = fmax(best_value, alpha);
if (alpha >= beta) break;
}
action_list_free(list);
flag = TT_EXACT;
if (best_value >= beta) flag = TT_LOWERBOUND;
else if (best_value <= alpha) flag = TT_UPPERBOUND;
if (entry == NULL) {
tt_insert(hash, flag, cur_depth, best_value, best_action);
} else {
entry->flag = flag;
entry->value = best_value;
entry->depth = cur_depth;
entry->action = best_action;
}
return best_value;
}
|