Merge commit 'd942e13398aa5de55224c7d81bfad6b0f5b9e488' into cluster

This is the last commit before there are more conflicts.
This commit is contained in:
Steinar H. Gunderson
2025-12-25 20:02:14 +01:00
32 changed files with 1264 additions and 816 deletions
+1
View File
@@ -58,6 +58,7 @@ Daniel Axtens (daxtens)
Daniel Dugovic (ddugovic)
Daniel Monroe (Ergodice)
Dan Schmidt (dfannius)
DanSamek
Dariusz Orzechowski (dorzechowski)
David (dav1312)
David Zar
+119 -98
View File
@@ -1,118 +1,126 @@
Contributors to Fishtest with >10,000 CPU hours, as of 2024-08-31.
Contributors to Fishtest with >10,000 CPU hours, as of 2025-03-22.
Thank you!
Username CPU Hours Games played
------------------------------------------------------------------
noobpwnftw 40428649 3164740143
technologov 23581394 1076895482
vdv 19425375 718302718
linrock 10034115 643194527
noobpwnftw 41712226 3294628533
vdv 28993864 954145232
technologov 24984442 1115931964
linrock 11463033 741692823
mlang 3026000 200065824
okrout 2572676 237511408
pemo 1836785 62226157
okrout 2726068 248285678
olafm 2420096 161297116
pemo 1838361 62294199
TueRens 1804847 80170868
dew 1689162 100033738
TueRens 1648780 77891164
sebastronomy 1468328 60859092
grandphish2 1466110 91776075
sebastronomy 1655637 67294942
grandphish2 1474752 92156319
JojoM 1130625 73666098
olafm 1067009 74807270
rpngn 973590 59996557
oz 921203 60370346
tvijlbrief 796125 51897690
oz 781847 53910686
rpngn 768460 49812975
gvreuls 751085 52177668
gvreuls 792215 55184194
mibere 703840 46867607
leszek 566598 42024615
cw 519601 34988161
leszek 599745 44681421
cw 519602 34988289
fastgm 503862 30260818
CSU_Dynasty 468784 31385034
maximmasiutin 439192 27893522
ctoks 435148 28541909
CSU_Dynasty 474794 31654170
maximmasiutin 441753 28129452
robal 437950 28869118
ctoks 435150 28542141
crunchy 427414 27371625
bcross 415724 29061187
robal 371112 24642270
mgrabiak 367963 26464704
mgrabiak 380202 27586936
velislav 342588 22140902
ncfish1 329039 20624527
Fisherman 327231 21829379
Sylvain27 317021 11494912
marrco 310446 19587107
Dantist 296386 18031762
tolkki963 262050 22049676
Sylvain27 255595 8864404
Fifis 289595 14969251
tolkki963 286043 23596996
Calis007 272677 17281620
cody 258835 13301710
nordlandia 249322 16420192
Fifis 237657 13065577
marrco 234581 17714473
Calis007 217537 14450582
javran 212141 16507618
glinscott 208125 13277240
drabel 204167 13930674
mhoram 202894 12601997
bking_US 198894 11876016
Wencey 198537 9606420
Thanar 179852 12365359
javran 169679 13481966
armo9494 162863 10937118
sschnee 170521 10891112
armo9494 168141 11177514
DesolatedDodo 160605 10392474
spams 157128 10319326
DesolatedDodo 156683 10211206
Wencey 152308 8375444
maposora 155839 13963260
sqrt2 147963 9724586
vdbergh 140311 9225125
vdbergh 140514 9242985
jcAEie 140086 10603658
CoffeeOne 137100 5024116
malala 136182 8002293
Goatminola 134893 11640524
xoto 133759 9159372
Dubslow 129614 8519312
markkulix 132104 11000548
naclosagc 131472 4660806
Dubslow 129685 8527664
davar 129023 8376525
DMBK 122960 8980062
dsmith 122059 7570238
CypressChess 120784 8672620
sschnee 120526 7547722
maposora 119734 10749710
Wolfgang 120919 8619168
CypressChess 120902 8683904
amicic 119661 7938029
Wolfgang 115713 8159062
cuistot 116864 7828864
sterni1971 113754 6054022
Data 113305 8220352
BrunoBanani 112960 7436849
markkulix 112897 9133168
cuistot 109802 7121030
megaman7de 109139 7360928
skiminki 107583 7218170
sterni1971 104431 5938282
zeryl 104523 6618969
MaZePallas 102823 6633619
sunu 100167 7040199
zeryl 99331 6221261
thirdlife 99156 2245320
thirdlife 99178 2246544
ElbertoOne 99028 7023771
megaman7de 98456 6675076
Goatminola 96765 8257832
TataneSan 97257 4239502
romangol 95662 7784954
bigpen0r 94825 6529241
brabos 92118 6186135
Maxim 90818 3283364
psk 89957 5984901
szupaw 89775 7800606
jromang 87260 5988073
racerschmacer 85805 6122790
Vizvezdenec 83761 5344740
0x3C33 82614 5271253
szupaw 82495 7151686
Spprtr 82103 5663635
BRAVONE 81239 5054681
MarcusTullius 78930 5189659
Mineta 78731 4947996
Torom 77978 2651656
nssy 76497 5259388
cody 76126 4492126
jromang 76106 5236025
MarcusTullius 76103 5061991
woutboat 76072 6022922
Spprtr 75977 5252287
woutboat 76379 6031688
teddybaer 75125 5407666
Pking_cda 73776 5293873
Viren6 73664 1356502
yurikvelo 73611 5046822
Mineta 71130 4711422
Bobo1239 70579 4794999
solarlight 70517 5028306
dv8silencer 70287 3883992
manap 66273 4121774
tinker 64333 4268790
qurashee 61208 3429862
AGI 58195 4329580
DanielMiao1 60181 1317252
AGI 58316 4336328
jojo2357 57435 4944212
robnjr 57262 4053117
Freja 56938 3733019
MaxKlaxxMiner 56879 3423958
ttruscott 56010 3680085
rkl 55132 4164467
jmdana 54697 4012593
jmdana 54988 4041917
notchris 53936 4184018
renouve 53811 3501516
CounterFlow 52536 3203740
finfish 51360 3370515
eva42 51272 3599691
eastorwest 51117 3454811
@@ -122,33 +130,36 @@ GPUex 48686 3684998
OuaisBla 48626 3445134
ronaldjerum 47654 3240695
biffhero 46564 3111352
oryx 45639 3546530
oryx 46141 3583236
jibarbosa 45890 4541218
DeepnessFulled 45734 3944282
abdicj 45577 2631772
VoyagerOne 45476 3452465
mecevdimitar 44240 2584396
speedycpu 43842 3003273
jbwiebe 43305 2805433
gopeto 43046 2821514
YvesKn 42628 2177630
Antihistamine 41788 2761312
mhunt 41735 2691355
jibarbosa 41640 4145702
somethingintheshadows 41502 3330418
homyur 39893 2850481
gri 39871 2515779
DeepnessFulled 39020 3323102
vidar808 39774 1656372
Garf 37741 2999686
SC 37299 2731694
Gaster319 37118 3279678
naclosagc 36562 1279618
Gaster319 37229 3289674
csnodgrass 36207 2688994
ZacHFX 35528 2486328
icewulf 34782 2415146
strelock 34716 2074055
gopeto 33717 2245606
EthanOConnor 33370 2090311
slakovv 32915 2021889
jojo2357 32890 2826662
shawnxu 32019 2802552
shawnxu 32144 2814668
Gelma 31771 1551204
vidar808 31560 1351810
srowen 31181 1732120
kdave 31157 2198362
manapbk 30987 1810399
ZacHFX 30966 2272416
TataneSan 30713 1513402
votoanthuan 30691 2460856
Prcuvu 30377 2170122
anst 30301 2190091
@@ -157,145 +168,155 @@ spcc 29925 1901692
hyperbolic.tom 29840 2017394
chuckstablers 29659 2093438
Pyafue 29650 1902349
WoodMan777 29300 2579864
belzedar94 28846 1811530
mecevdimitar 27610 1721382
chriswk 26902 1868317
xwziegtm 26897 2124586
Jopo12321 26818 1816482
achambord 26582 1767323
somethingintheshadows 26496 2186404
Patrick_G 26276 1801617
yorkman 26193 1992080
srowen 25743 1490684
Ulysses 25413 1702830
Jopo12321 25227 1652482
Ulysses 25517 1711634
SFTUser 25182 1675689
nabildanial 25068 1531665
Sharaf_DG 24765 1786697
rodneyc 24376 1416402
jsys14 24297 1721230
AndreasKrug 24235 1934711
agg177 23890 1395014
AndreasKrug 23754 1890115
Ente 23752 1678188
JanErik 23408 1703875
Isidor 23388 1680691
Norabor 23371 1603244
WoodMan777 23253 2023048
Nullvalue 23155 2022752
fishtester 23115 1581502
wizardassassin 23073 1789536
Skiff84 22984 1053680
cisco2015 22920 1763301
ols 22914 1322047
Hjax 22561 1566151
Zirie 22542 1472937
team-oh 22272 1636708
mkstockfishtester 22253 2029566
Roady 22220 1465606
MazeOfGalious 21978 1629593
sg4032 21950 1643373
tsim67 21747 1330880
tsim67 21939 1343944
ianh2105 21725 1632562
Skiff84 21711 1014212
Serpensin 21704 1809188
xor12 21628 1680365
dex 21612 1467203
nesoneg 21494 1463031
IslandLambda 21468 1239756
user213718 21454 1404128
Serpensin 21452 1790510
sphinx 21211 1384728
qoo_charly_cai 21136 1514927
IslandLambda 21062 1220838
jjoshua2 21001 1423089
Zake9298 20938 1565848
horst.prack 20878 1465656
fishtester 20729 1348888
0xB00B1ES 20590 1208666
ols 20477 1195945
Dinde 20459 1292774
t3hf1sht3ster 20456 670646
j3corre 20405 941444
0x539 20332 1039516
Adrian.Schmidt123 20316 1281436
malfoy 20313 1350694
purpletree 20019 1461026
wei 19973 1745989
teenychess 19819 1762006
rstoesser 19569 1293588
eudhan 19274 1283717
nalanzeyu 19211 396674
vulcan 18871 1729392
wizardassassin 18795 1376884
Karpovbot 18766 1053178
jundery 18445 1115855
mkstockfishtester 18350 1690676
Farseer 18281 1074642
sebv15 18267 1262588
whelanh 17887 347974
ville 17883 1384026
chris 17698 1487385
purplefishies 17595 1092533
dju 17414 981289
iisiraider 17275 1049015
Karby 17177 1030688
DragonLord 17014 1162790
Karby 17008 1013160
pirt 16965 1271519
pirt 16991 1274215
redstone59 16842 1461780
Alb11747 16787 1213990
Naven94 16414 951718
scuzzi 16115 994341
scuzzi 16155 995347
IgorLeMasson 16064 1147232
ako027ako 15671 1173203
xuhdev 15516 1528278
infinigon 15285 965966
Nikolay.IT 15154 1068349
Andrew Grant 15114 895539
OssumOpossum 14857 1007129
LunaticBFF57 14525 1190310
enedene 14476 905279
Hjax 14394 1005013
YELNAMRON 14475 1141330
RickGroszkiewicz 14272 1385984
joendter 14269 982014
bpfliegel 14233 882523
YELNAMRON 14230 1128094
mpx86 14019 759568
jpulman 13982 870599
getraideBFF 13871 1172846
crocogoat 13817 1119086
Nesa92 13806 1116101
crocogoat 13803 1117422
joster 13710 946160
mbeier 13650 1044928
Pablohn26 13552 1088532
wxt9861 13550 1312306
Dark_wizzie 13422 1007152
Rudolphous 13244 883140
Jackfish 13177 894206
MooTheCow 13091 892304
Machariel 13010 863104
nalanzeyu 12996 232590
mabichito 12903 749391
Jackfish 12895 868928
thijsk 12886 722107
AdrianSA 12860 804972
Flopzee 12698 894821
whelanh 12682 266404
szczur90 12684 977536
Kyrega 12661 456438
mschmidt 12644 863193
korposzczur 12606 838168
fatmurphy 12547 853210
Oakwen 12532 855759
icewulf 12447 854878
SapphireBrand 12416 969604
deflectooor 12386 579392
modolief 12386 896470
Farseer 12249 694108
ckaz 12273 754644
Hongildong 12201 648712
pgontarz 12151 848794
dbernier 12103 860824
szczur90 12035 942376
FormazChar 12019 910409
FormazChar 12051 913497
shreven 12044 884734
rensonthemove 11999 971993
stocky 11954 699440
MooTheCow 11923 779432
3cho 11842 1036786
ckaz 11792 732276
ImperiumAeternum 11482 979142
infinity 11470 727027
aga 11412 695127
Def9Infinity 11408 700682
torbjo 11395 729145
Thomas A. Anderson 11372 732094
savage84 11358 670860
Def9Infinity 11345 696552
d64 11263 789184
ali-al-zhrani 11245 779246
ImperiumAeternum 11155 952000
vaskoul 11144 953906
snicolet 11106 869170
dapper 11032 771402
Ethnikoi 10993 945906
Snuuka 10938 435504
Karmatron 10871 678306
gerbil 10871 1005842
OliverClarke 10696 942654
basepi 10637 744851
michaelrpg 10624 748179
Cubox 10621 826448
gerbil 10519 971688
michaelrpg 10509 739239
dragon123118 10421 936506
OIVAS7572 10420 995586
GBx3TV 10388 339952
Garruk 10365 706465
dzjp 10343 732529
RickGroszkiewicz 10263 990798
borinot 10026 902130
+6 -5
View File
@@ -3,11 +3,6 @@
wget_or_curl=$( (command -v wget > /dev/null 2>&1 && echo "wget -qO-") || \
(command -v curl > /dev/null 2>&1 && echo "curl -skL"))
if [ -z "$wget_or_curl" ]; then
>&2 printf "%s\n" "Neither wget or curl is installed." \
"Install one of these tools to download NNUE files automatically."
exit 1
fi
sha256sum=$( (command -v shasum > /dev/null 2>&1 && echo "shasum -a 256") || \
(command -v sha256sum > /dev/null 2>&1 && echo "sha256sum"))
@@ -47,6 +42,12 @@ fetch_network() {
fi
fi
if [ -z "$wget_or_curl" ]; then
>&2 printf "%s\n" "Neither wget or curl is installed." \
"Install one of these tools to download NNUE files automatically."
exit 1
fi
for url in \
"https://tests.stockfishchess.org/api/nn/$_filename" \
"https://github.com/official-stockfish/networks/raw/master/$_filename"; do
+2 -1
View File
@@ -55,7 +55,8 @@ PGOBENCH = $(WINE_PATH) ./$(EXE) bench
SRCS = benchmark.cpp bitboard.cpp evaluate.cpp main.cpp \
misc.cpp movegen.cpp movepick.cpp position.cpp cluster.cpp \
search.cpp thread.cpp timeman.cpp tt.cpp uci.cpp ucioption.cpp tune.cpp syzygy/tbprobe.cpp \
nnue/nnue_misc.cpp nnue/features/half_ka_v2_hm.cpp nnue/network.cpp engine.cpp score.cpp memory.cpp
nnue/nnue_accumulator.cpp nnue/nnue_misc.cpp nnue/features/half_ka_v2_hm.cpp nnue/network.cpp \
engine.cpp score.cpp memory.cpp
HEADERS = benchmark.h bitboard.h evaluate.h misc.h movegen.h movepick.h history.h \
nnue/nnue_misc.h nnue/features/half_ka_v2_hm.h nnue/layers/affine_transform.h \
+2 -3
View File
@@ -32,7 +32,6 @@ uint8_t SquareDistance[SQUARE_NB][SQUARE_NB];
Bitboard LineBB[SQUARE_NB][SQUARE_NB];
Bitboard BetweenBB[SQUARE_NB][SQUARE_NB];
Bitboard PseudoAttacks[PIECE_TYPE_NB][SQUARE_NB];
Bitboard PawnAttacks[COLOR_NB][SQUARE_NB];
alignas(64) Magic Magics[SQUARE_NB][2];
@@ -86,8 +85,8 @@ void Bitboards::init() {
for (Square s1 = SQ_A1; s1 <= SQ_H8; ++s1)
{
PawnAttacks[WHITE][s1] = pawn_attacks_bb<WHITE>(square_bb(s1));
PawnAttacks[BLACK][s1] = pawn_attacks_bb<BLACK>(square_bb(s1));
PseudoAttacks[WHITE][s1] = pawn_attacks_bb<WHITE>(square_bb(s1));
PseudoAttacks[BLACK][s1] = pawn_attacks_bb<BLACK>(square_bb(s1));
for (int step : {-9, -8, -7, -1, 1, 7, 8, 9})
PseudoAttacks[KING][s1] |= safe_destination(s1, step);
+12 -18
View File
@@ -62,7 +62,6 @@ extern uint8_t SquareDistance[SQUARE_NB][SQUARE_NB];
extern Bitboard BetweenBB[SQUARE_NB][SQUARE_NB];
extern Bitboard LineBB[SQUARE_NB][SQUARE_NB];
extern Bitboard PseudoAttacks[PIECE_TYPE_NB][SQUARE_NB];
extern Bitboard PawnAttacks[COLOR_NB][SQUARE_NB];
// Magic holds all magic bitboards relevant data for a single square
@@ -103,17 +102,17 @@ constexpr Bitboard square_bb(Square s) {
// Overloads of bitwise operators between a Bitboard and a Square for testing
// whether a given bit is set in a bitboard, and for setting and clearing bits.
inline Bitboard operator&(Bitboard b, Square s) { return b & square_bb(s); }
inline Bitboard operator|(Bitboard b, Square s) { return b | square_bb(s); }
inline Bitboard operator^(Bitboard b, Square s) { return b ^ square_bb(s); }
inline Bitboard& operator|=(Bitboard& b, Square s) { return b |= square_bb(s); }
inline Bitboard& operator^=(Bitboard& b, Square s) { return b ^= square_bb(s); }
constexpr Bitboard operator&(Bitboard b, Square s) { return b & square_bb(s); }
constexpr Bitboard operator|(Bitboard b, Square s) { return b | square_bb(s); }
constexpr Bitboard operator^(Bitboard b, Square s) { return b ^ square_bb(s); }
constexpr Bitboard& operator|=(Bitboard& b, Square s) { return b |= square_bb(s); }
constexpr Bitboard& operator^=(Bitboard& b, Square s) { return b ^= square_bb(s); }
inline Bitboard operator&(Square s, Bitboard b) { return b & s; }
inline Bitboard operator|(Square s, Bitboard b) { return b | s; }
inline Bitboard operator^(Square s, Bitboard b) { return b ^ s; }
constexpr Bitboard operator&(Square s, Bitboard b) { return b & s; }
constexpr Bitboard operator|(Square s, Bitboard b) { return b | s; }
constexpr Bitboard operator^(Square s, Bitboard b) { return b ^ s; }
inline Bitboard operator|(Square s1, Square s2) { return square_bb(s1) | s2; }
constexpr Bitboard operator|(Square s1, Square s2) { return square_bb(s1) | s2; }
constexpr bool more_than_one(Bitboard b) { return b & (b - 1); }
@@ -155,11 +154,6 @@ constexpr Bitboard pawn_attacks_bb(Bitboard b) {
: shift<SOUTH_WEST>(b) | shift<SOUTH_EAST>(b);
}
inline Bitboard pawn_attacks_bb(Color c, Square s) {
assert(is_ok(s));
return PawnAttacks[c][s];
}
// Returns a bitboard representing an entire line (from board edge
// to board edge) that intersects the two given squares. If the given squares
@@ -216,10 +210,10 @@ inline int edge_distance(File f) { return std::min(f, File(FILE_H - f)); }
// Returns the pseudo attacks of the given piece type
// assuming an empty board.
template<PieceType Pt>
inline Bitboard attacks_bb(Square s) {
inline Bitboard attacks_bb(Square s, Color c = COLOR_NB) {
assert((Pt != PAWN) && (is_ok(s)));
return PseudoAttacks[Pt][s];
assert((Pt != PAWN || c < COLOR_NB) && (is_ok(s)));
return Pt == PAWN ? PseudoAttacks[c][s] : PseudoAttacks[Pt][s];
}
+6 -3
View File
@@ -18,6 +18,7 @@
#include "engine.h"
#include <algorithm>
#include <cassert>
#include <deque>
#include <iosfwd>
@@ -32,6 +33,7 @@
#include "misc.h"
#include "nnue/network.h"
#include "nnue/nnue_common.h"
#include "numa.h"
#include "perft.h"
#include "position.h"
#include "search.h"
@@ -44,8 +46,9 @@ namespace Stockfish {
namespace NN = Eval::NNUE;
constexpr auto StartFEN = "rnbqkbnr/pppppppp/8/8/8/8/PPPPPPPP/RNBQKBNR w KQkq - 0 1";
constexpr int MaxHashMB = Is64Bit ? 33554432 : 2048;
constexpr auto StartFEN = "rnbqkbnr/pppppppp/8/8/8/8/PPPPPPPP/RNBQKBNR w KQkq - 0 1";
constexpr int MaxHashMB = Is64Bit ? 33554432 : 2048;
int MaxThreads = std::max(1024, 4 * int(get_hardware_concurrency()));
Engine::Engine(std::optional<std::string> path) :
binaryDirectory(path ? CommandLine::get_binary_directory(*path) : ""),
@@ -74,7 +77,7 @@ Engine::Engine(std::optional<std::string> path) :
}));
options.add( //
"Threads", Option(1, 1, 1024, [this](const Option&) {
"Threads", Option(1, 1, MaxThreads, [this](const Option&) {
resize_threads();
return thread_allocation_information_as_string();
}));
+14 -12
View File
@@ -38,37 +38,36 @@
namespace Stockfish {
// Returns a static, purely materialistic evaluation of the position from
// the point of view of the given color. It can be divided by PawnValue to get
// the point of view of the side to move. It can be divided by PawnValue to get
// an approximation of the material advantage on the board in terms of pawns.
int Eval::simple_eval(const Position& pos, Color c) {
int Eval::simple_eval(const Position& pos) {
Color c = pos.side_to_move();
return PawnValue * (pos.count<PAWN>(c) - pos.count<PAWN>(~c))
+ (pos.non_pawn_material(c) - pos.non_pawn_material(~c));
}
bool Eval::use_smallnet(const Position& pos) {
int simpleEval = simple_eval(pos, pos.side_to_move());
return std::abs(simpleEval) > 962;
}
bool Eval::use_smallnet(const Position& pos) { return std::abs(simple_eval(pos)) > 962; }
// Evaluate is the evaluator for the outer world. It returns a static evaluation
// of the position from the point of view of the side to move.
Value Eval::evaluate(const Eval::NNUE::Networks& networks,
const Position& pos,
Eval::NNUE::AccumulatorStack& accumulators,
Eval::NNUE::AccumulatorCaches& caches,
int optimism) {
assert(!pos.checkers());
bool smallNet = use_smallnet(pos);
auto [psqt, positional] = smallNet ? networks.small.evaluate(pos, &caches.small)
: networks.big.evaluate(pos, &caches.big);
auto [psqt, positional] = smallNet ? networks.small.evaluate(pos, accumulators, &caches.small)
: networks.big.evaluate(pos, accumulators, &caches.big);
Value nnue = (125 * psqt + 131 * positional) / 128;
// Re-evaluate the position when higher eval accuracy is worth the time spent
if (smallNet && (std::abs(nnue) < 236))
{
std::tie(psqt, positional) = networks.big.evaluate(pos, &caches.big);
std::tie(psqt, positional) = networks.big.evaluate(pos, accumulators, &caches.big);
nnue = (125 * psqt + 131 * positional) / 128;
smallNet = false;
}
@@ -99,7 +98,10 @@ std::string Eval::trace(Position& pos, const Eval::NNUE::Networks& networks) {
if (pos.checkers())
return "Final evaluation: none (in check)";
auto caches = std::make_unique<Eval::NNUE::AccumulatorCaches>(networks);
Eval::NNUE::AccumulatorStack accumulators;
auto caches = std::make_unique<Eval::NNUE::AccumulatorCaches>(networks);
accumulators.reset(pos, networks, *caches);
std::stringstream ss;
ss << std::showpoint << std::noshowpos << std::fixed << std::setprecision(2);
@@ -107,12 +109,12 @@ std::string Eval::trace(Position& pos, const Eval::NNUE::Networks& networks) {
ss << std::showpoint << std::showpos << std::fixed << std::setprecision(2) << std::setw(15);
auto [psqt, positional] = networks.big.evaluate(pos, &caches->big);
auto [psqt, positional] = networks.big.evaluate(pos, accumulators, &caches->big);
Value v = psqt + positional;
v = pos.side_to_move() == WHITE ? v : -v;
ss << "NNUE evaluation " << 0.01 * UCIEngine::to_cp(v, pos) << " (white side)\n";
v = evaluate(networks, pos, *caches, VALUE_ZERO);
v = evaluate(networks, pos, accumulators, *caches, VALUE_ZERO);
v = pos.side_to_move() == WHITE ? v : -v;
ss << "Final evaluation " << 0.01 * UCIEngine::to_cp(v, pos) << " (white side)";
ss << " [with scaled NNUE, ...]";
+3 -1
View File
@@ -39,14 +39,16 @@ namespace Eval {
namespace NNUE {
struct Networks;
struct AccumulatorCaches;
class AccumulatorStack;
}
std::string trace(Position& pos, const Eval::NNUE::Networks& networks);
int simple_eval(const Position& pos, Color c);
int simple_eval(const Position& pos);
bool use_smallnet(const Position& pos);
Value evaluate(const NNUE::Networks& networks,
const Position& pos,
Eval::NNUE::AccumulatorStack& accumulators,
Eval::NNUE::AccumulatorCaches& caches,
int optimism);
} // namespace Eval
+6
View File
@@ -155,6 +155,12 @@ struct CorrHistTypedef<Continuation> {
using type = MultiArray<CorrHistTypedef<PieceTo>::type, PIECE_NB, SQUARE_NB>;
};
template<>
struct CorrHistTypedef<NonPawn> {
using type =
Stats<std::int16_t, CORRECTION_HISTORY_LIMIT, CORRECTION_HISTORY_SIZE, COLOR_NB, COLOR_NB>;
};
}
template<CorrHistType T>
+14 -1
View File
@@ -287,12 +287,18 @@ namespace {
template<size_t N>
struct DebugInfo {
std::atomic<int64_t> data[N] = {0};
std::array<std::atomic<int64_t>, N> data = {0};
[[nodiscard]] constexpr std::atomic<int64_t>& operator[](size_t index) {
assert(index < N);
return data[index];
}
constexpr DebugInfo& operator=(const DebugInfo& other) {
for (size_t i = 0; i < N; i++)
data[i].store(other.data[i].load());
return *this;
}
};
struct DebugExtremes: public DebugInfo<3> {
@@ -393,6 +399,13 @@ void dbg_print() {
}
}
void dbg_clear() {
hit.fill({});
mean.fill({});
stdev.fill({});
correl.fill({});
extremes.fill({});
}
// Used to serialize access to std::cout
// to avoid multiple threads writing at the same time.
+1
View File
@@ -73,6 +73,7 @@ void dbg_stdev_of(int64_t value, int slot = 0);
void dbg_extremes_of(int64_t value, int slot = 0);
void dbg_correl_of(int64_t value1, int64_t value2, int slot = 0);
void dbg_print();
void dbg_clear();
using TimePoint = std::chrono::milliseconds::rep; // A value in milliseconds
static_assert(sizeof(TimePoint) == sizeof(int64_t), "TimePoint should be 64 bits");
+1 -1
View File
@@ -134,7 +134,7 @@ ExtMove* generate_pawn_moves(const Position& pos, ExtMove* moveList, Bitboard ta
if (Type == EVASIONS && (target & (pos.ep_square() + Up)))
return moveList;
b1 = pawnsNotOn7 & pawn_attacks_bb(Them, pos.ep_square());
b1 = pawnsNotOn7 & attacks_bb<PAWN>(pos.ep_square(), Them);
assert(b1);
+1 -1
View File
@@ -176,7 +176,7 @@ void MovePicker::score() {
: 0;
// malus for putting piece en prise
m.value -= (pt == QUEEN ? bool(to & threatenedByRook) * 49000
m.value -= (pt == QUEEN && bool(to & threatenedByRook) ? 49000
: pt == ROOK && bool(to & threatenedByMinor) ? 24335
: 0);
+2 -6
View File
@@ -77,12 +77,8 @@ template void HalfKAv2_hm::append_changed_indices<BLACK>(Square ksq,
IndexList& removed,
IndexList& added);
int HalfKAv2_hm::update_cost(const StateInfo* st) { return st->dirtyPiece.dirty_num; }
int HalfKAv2_hm::refresh_cost(const Position& pos) { return pos.count<ALL_PIECES>(); }
bool HalfKAv2_hm::requires_refresh(const StateInfo* st, Color perspective) {
return st->dirtyPiece.piece[0] == make_piece(perspective, KING);
bool HalfKAv2_hm::requires_refresh(const DirtyPiece& dirtyPiece, Color perspective) {
return dirtyPiece.piece[0] == make_piece(perspective, KING);
}
} // namespace Stockfish::Eval::NNUE::Features
+2 -8
View File
@@ -28,7 +28,6 @@
#include "../nnue_common.h"
namespace Stockfish {
struct StateInfo;
class Position;
}
@@ -135,14 +134,9 @@ class HalfKAv2_hm {
static void
append_changed_indices(Square ksq, const DirtyPiece& dp, IndexList& removed, IndexList& added);
// Returns the cost of updating one perspective, the most costly one.
// Assumes no refresh needed.
static int update_cost(const StateInfo* st);
static int refresh_cost(const Position& pos);
// Returns whether the change stored in this StateInfo means
// Returns whether the change stored in this DirtyPiece means
// that a full accumulator refresh is required.
static bool requires_refresh(const StateInfo* st, Color perspective);
static bool requires_refresh(const DirtyPiece& dirtyPiece, Color perspective);
};
} // namespace Stockfish::Eval::NNUE::Features
-4
View File
@@ -245,19 +245,16 @@ class AffineTransform {
#if defined(USE_AVX2)
using vec_t = __m256i;
#define vec_setzero() _mm256_setzero_si256()
#define vec_set_32 _mm256_set1_epi32
#define vec_add_dpbusd_32 Simd::m256_add_dpbusd_epi32
#define vec_hadd Simd::m256_hadd
#elif defined(USE_SSSE3)
using vec_t = __m128i;
#define vec_setzero() _mm_setzero_si128()
#define vec_set_32 _mm_set1_epi32
#define vec_add_dpbusd_32 Simd::m128_add_dpbusd_epi32
#define vec_hadd Simd::m128_hadd
#elif defined(USE_NEON_DOTPROD)
using vec_t = int32x4_t;
#define vec_setzero() vdupq_n_s32(0)
#define vec_set_32 vdupq_n_s32
#define vec_add_dpbusd_32(acc, a, b) \
Simd::dotprod_m128_add_dpbusd_epi32(acc, vreinterpretq_s8_s32(a), \
vreinterpretq_s8_s32(b))
@@ -282,7 +279,6 @@ class AffineTransform {
output[0] = vec_hadd(sum0, biases[0]);
#undef vec_setzero
#undef vec_set_32
#undef vec_add_dpbusd_32
#undef vec_hadd
}
+9 -6
View File
@@ -211,6 +211,7 @@ bool Network<Arch, Transformer>::save(const std::optional<std::string>& filename
template<typename Arch, typename Transformer>
NetworkOutput
Network<Arch, Transformer>::evaluate(const Position& pos,
AccumulatorStack& accumulatorStack,
AccumulatorCaches::Cache<FTDimensions>* cache) const {
// We manually align the arrays on the stack because with gcc < 9.3
// overaligning stack variables with alignas() doesn't work correctly.
@@ -230,8 +231,9 @@ Network<Arch, Transformer>::evaluate(const Position& pos
ASSERT_ALIGNED(transformedFeatures, alignment);
const int bucket = (pos.count<ALL_PIECES>() - 1) / 4;
const auto psqt = featureTransformer->transform(pos, cache, transformedFeatures, bucket);
const int bucket = (pos.count<ALL_PIECES>() - 1) / 4;
const auto psqt =
featureTransformer->transform(pos, accumulatorStack, cache, transformedFeatures, bucket);
const auto positional = network[bucket].propagate(transformedFeatures);
return {static_cast<Value>(psqt / OutputScale), static_cast<Value>(positional / OutputScale)};
}
@@ -282,6 +284,7 @@ void Network<Arch, Transformer>::verify(std::string
template<typename Arch, typename Transformer>
NnueEvalTrace
Network<Arch, Transformer>::trace_evaluate(const Position& pos,
AccumulatorStack& accumulatorStack,
AccumulatorCaches::Cache<FTDimensions>* cache) const {
// We manually align the arrays on the stack because with gcc < 9.3
// overaligning stack variables with alignas() doesn't work correctly.
@@ -305,7 +308,7 @@ Network<Arch, Transformer>::trace_evaluate(const Position&
for (IndexType bucket = 0; bucket < LayerStacks; ++bucket)
{
const auto materialist =
featureTransformer->transform(pos, cache, transformedFeatures, bucket);
featureTransformer->transform(pos, accumulatorStack, cache, transformedFeatures, bucket);
const auto positional = network[bucket].propagate(transformedFeatures);
t.psqt[bucket] = static_cast<Value>(materialist / OutputScale);
@@ -449,14 +452,14 @@ bool Network<Arch, Transformer>::write_parameters(std::ostream& stream,
return bool(stream);
}
// Explicit template instantiation
// Explicit template instantiations
template class Network<
NetworkArchitecture<TransformedFeatureDimensionsBig, L2Big, L3Big>,
FeatureTransformer<TransformedFeatureDimensionsBig, &StateInfo::accumulatorBig>>;
FeatureTransformer<TransformedFeatureDimensionsBig, &AccumulatorState::accumulatorBig>>;
template class Network<
NetworkArchitecture<TransformedFeatureDimensionsSmall, L2Small, L3Small>,
FeatureTransformer<TransformedFeatureDimensionsSmall, &StateInfo::accumulatorSmall>>;
FeatureTransformer<TransformedFeatureDimensionsSmall, &AccumulatorState::accumulatorSmall>>;
} // namespace Stockfish::Eval::NNUE
+10 -3
View File
@@ -29,13 +29,16 @@
#include <utility>
#include "../memory.h"
#include "../position.h"
#include "../types.h"
#include "nnue_accumulator.h"
#include "nnue_architecture.h"
#include "nnue_feature_transformer.h"
#include "nnue_misc.h"
namespace Stockfish {
class Position;
}
namespace Stockfish::Eval::NNUE {
enum class EmbeddedNNUEType {
@@ -64,11 +67,13 @@ class Network {
bool save(const std::optional<std::string>& filename) const;
NetworkOutput evaluate(const Position& pos,
AccumulatorStack& accumulatorStack,
AccumulatorCaches::Cache<FTDimensions>* cache) const;
void verify(std::string evalfilePath, const std::function<void(std::string_view)>&) const;
NnueEvalTrace trace_evaluate(const Position& pos,
AccumulatorStack& accumulatorStack,
AccumulatorCaches::Cache<FTDimensions>* cache) const;
private:
@@ -100,16 +105,18 @@ class Network {
template<IndexType Size>
friend struct AccumulatorCaches::Cache;
friend class AccumulatorStack;
};
// Definitions of the network types
using SmallFeatureTransformer =
FeatureTransformer<TransformedFeatureDimensionsSmall, &StateInfo::accumulatorSmall>;
FeatureTransformer<TransformedFeatureDimensionsSmall, &AccumulatorState::accumulatorSmall>;
using SmallNetworkArchitecture =
NetworkArchitecture<TransformedFeatureDimensionsSmall, L2Small, L3Small>;
using BigFeatureTransformer =
FeatureTransformer<TransformedFeatureDimensionsBig, &StateInfo::accumulatorBig>;
FeatureTransformer<TransformedFeatureDimensionsBig, &AccumulatorState::accumulatorBig>;
using BigNetworkArchitecture = NetworkArchitecture<TransformedFeatureDimensionsBig, L2Big, L3Big>;
using NetworkBig = Network<BigNetworkArchitecture, BigFeatureTransformer>;
+565
View File
@@ -0,0 +1,565 @@
/*
Stockfish, a UCI chess playing engine derived from Glaurung 2.1
Copyright (C) 2004-2025 The Stockfish developers (see AUTHORS file)
Stockfish is free software: you can redistribute it and/or modify
it under the terms of the GNU General Public License as published by
the Free Software Foundation, either version 3 of the License, or
(at your option) any later version.
Stockfish is distributed in the hope that it will be useful,
but WITHOUT ANY WARRANTY; without even the implied warranty of
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
GNU General Public License for more details.
You should have received a copy of the GNU General Public License
along with this program. If not, see <http://www.gnu.org/licenses/>.
*/
#include "nnue_accumulator.h"
#include <cassert>
#include <initializer_list>
#include <memory>
#include <type_traits>
#include "../bitboard.h"
#include "../misc.h"
#include "../position.h"
#include "../types.h"
#include "network.h"
#include "nnue_architecture.h"
#include "nnue_common.h"
#include "nnue_feature_transformer.h"
namespace Stockfish::Eval::NNUE {
#if defined(__GNUC__) && !defined(__clang__)
#define sf_assume(cond) \
do \
{ \
if (!(cond)) \
__builtin_unreachable(); \
} while (0)
#else
// do nothing for other compilers
#define sf_assume(cond)
#endif
namespace {
template<Color Perspective,
IncUpdateDirection Direction = FORWARD,
IndexType TransformedFeatureDimensions,
Accumulator<TransformedFeatureDimensions> AccumulatorState::* accPtr>
void update_accumulator_incremental(
const FeatureTransformer<TransformedFeatureDimensions, accPtr>& featureTransformer,
const Square ksq,
AccumulatorState& target_state,
const AccumulatorState& computed);
template<Color Perspective,
IndexType Dimensions,
Accumulator<Dimensions> AccumulatorState::* accPtr>
void update_accumulator_refresh_cache(
const FeatureTransformer<Dimensions, accPtr>& featureTransformer,
const Position& pos,
AccumulatorState& accumulatorState,
AccumulatorCaches::Cache<Dimensions>& cache);
}
void AccumulatorState::reset(const DirtyPiece& dp) noexcept {
dirtyPiece = dp;
accumulatorBig.computed.fill(false);
accumulatorSmall.computed.fill(false);
}
const AccumulatorState& AccumulatorStack::latest() const noexcept {
return m_accumulators[m_current_idx - 1];
}
AccumulatorState& AccumulatorStack::mut_latest() noexcept {
return m_accumulators[m_current_idx - 1];
}
void AccumulatorStack::reset(const Position& rootPos,
const Networks& networks,
AccumulatorCaches& caches) noexcept {
m_current_idx = 1;
update_accumulator_refresh_cache<WHITE, TransformedFeatureDimensionsBig,
&AccumulatorState::accumulatorBig>(
*networks.big.featureTransformer, rootPos, m_accumulators[0], caches.big);
update_accumulator_refresh_cache<BLACK, TransformedFeatureDimensionsBig,
&AccumulatorState::accumulatorBig>(
*networks.big.featureTransformer, rootPos, m_accumulators[0], caches.big);
update_accumulator_refresh_cache<WHITE, TransformedFeatureDimensionsSmall,
&AccumulatorState::accumulatorSmall>(
*networks.small.featureTransformer, rootPos, m_accumulators[0], caches.small);
update_accumulator_refresh_cache<BLACK, TransformedFeatureDimensionsSmall,
&AccumulatorState::accumulatorSmall>(
*networks.small.featureTransformer, rootPos, m_accumulators[0], caches.small);
}
void AccumulatorStack::push(const DirtyPiece& dirtyPiece) noexcept {
assert(m_current_idx + 1 < m_accumulators.size());
m_accumulators[m_current_idx].reset(dirtyPiece);
m_current_idx++;
}
void AccumulatorStack::pop() noexcept {
assert(m_current_idx > 1);
m_current_idx--;
}
template<IndexType Dimensions, Accumulator<Dimensions> AccumulatorState::* accPtr>
void AccumulatorStack::evaluate(const Position& pos,
const FeatureTransformer<Dimensions, accPtr>& featureTransformer,
AccumulatorCaches::Cache<Dimensions>& cache) noexcept {
evaluate_side<WHITE>(pos, featureTransformer, cache);
evaluate_side<BLACK>(pos, featureTransformer, cache);
}
template<Color Perspective,
IndexType Dimensions,
Accumulator<Dimensions> AccumulatorState::* accPtr>
void AccumulatorStack::evaluate_side(
const Position& pos,
const FeatureTransformer<Dimensions, accPtr>& featureTransformer,
AccumulatorCaches::Cache<Dimensions>& cache) noexcept {
const auto last_usable_accum = find_last_usable_accumulator<Perspective, Dimensions, accPtr>();
if ((m_accumulators[last_usable_accum].*accPtr).computed[Perspective])
forward_update_incremental<Perspective>(pos, featureTransformer, last_usable_accum);
else
{
update_accumulator_refresh_cache<Perspective>(featureTransformer, pos, mut_latest(), cache);
backward_update_incremental<Perspective>(pos, featureTransformer, last_usable_accum);
}
}
// Find the earliest usable accumulator, this can either be a computed accumulator or the accumulator
// state just before a change that requires full refresh.
template<Color Perspective,
IndexType Dimensions,
Accumulator<Dimensions> AccumulatorState::* accPtr>
std::size_t AccumulatorStack::find_last_usable_accumulator() const noexcept {
for (std::size_t curr_idx = m_current_idx - 1; curr_idx > 0; curr_idx--)
{
if ((m_accumulators[curr_idx].*accPtr).computed[Perspective])
return curr_idx;
if (FeatureSet::requires_refresh(m_accumulators[curr_idx].dirtyPiece, Perspective))
return curr_idx;
}
return 0;
}
template<Color Perspective,
IndexType Dimensions,
Accumulator<Dimensions> AccumulatorState::* accPtr>
void AccumulatorStack::forward_update_incremental(
const Position& pos,
const FeatureTransformer<Dimensions, accPtr>& featureTransformer,
const std::size_t begin) noexcept {
assert(begin < m_accumulators.size());
assert((m_accumulators[begin].*accPtr).computed[Perspective]);
const Square ksq = pos.square<KING>(Perspective);
for (std::size_t next = begin + 1; next < m_current_idx; next++)
update_accumulator_incremental<Perspective>(featureTransformer, ksq, m_accumulators[next],
m_accumulators[next - 1]);
assert((latest().*accPtr).computed[Perspective]);
}
template<Color Perspective,
IndexType Dimensions,
Accumulator<Dimensions> AccumulatorState::* accPtr>
void AccumulatorStack::backward_update_incremental(
const Position& pos,
const FeatureTransformer<Dimensions, accPtr>& featureTransformer,
const std::size_t end) noexcept {
assert(end < m_accumulators.size());
assert(end < m_current_idx);
assert((latest().*accPtr).computed[Perspective]);
const Square ksq = pos.square<KING>(Perspective);
for (std::size_t next = m_current_idx - 2; next >= end; next--)
update_accumulator_incremental<Perspective, BACKWARD>(
featureTransformer, ksq, m_accumulators[next], m_accumulators[next + 1]);
assert((m_accumulators[end].*accPtr).computed[Perspective]);
}
// Explicit template instantiations
template void
AccumulatorStack::evaluate<TransformedFeatureDimensionsBig, &AccumulatorState::accumulatorBig>(
const Position& pos,
const FeatureTransformer<TransformedFeatureDimensionsBig, &AccumulatorState::accumulatorBig>&
featureTransformer,
AccumulatorCaches::Cache<TransformedFeatureDimensionsBig>& cache) noexcept;
template void
AccumulatorStack::evaluate<TransformedFeatureDimensionsSmall, &AccumulatorState::accumulatorSmall>(
const Position& pos,
const FeatureTransformer<TransformedFeatureDimensionsSmall, &AccumulatorState::accumulatorSmall>&
featureTransformer,
AccumulatorCaches::Cache<TransformedFeatureDimensionsSmall>& cache) noexcept;
namespace {
template<typename VectorWrapper,
IndexType Width,
UpdateOperation... ops,
typename ElementType,
typename... Ts,
std::enable_if_t<is_all_same_v<ElementType, Ts...>, bool> = true>
void fused_row_reduce(const ElementType* in, ElementType* out, const Ts* const... rows) {
constexpr IndexType size = Width * sizeof(ElementType) / sizeof(typename VectorWrapper::type);
auto* vecIn = reinterpret_cast<const typename VectorWrapper::type*>(in);
auto* vecOut = reinterpret_cast<typename VectorWrapper::type*>(out);
for (IndexType i = 0; i < size; ++i)
vecOut[i] = fused<VectorWrapper, ops...>(
vecIn[i], reinterpret_cast<const typename VectorWrapper::type*>(rows)[i]...);
}
template<Color Perspective,
IndexType Dimensions,
Accumulator<Dimensions> AccumulatorState::* accPtr>
struct AccumulatorUpdateContext {
const FeatureTransformer<Dimensions, accPtr>& featureTransformer;
const AccumulatorState& from;
AccumulatorState& to;
AccumulatorUpdateContext(const FeatureTransformer<Dimensions, accPtr>& ft,
const AccumulatorState& accF,
AccumulatorState& accT) noexcept :
featureTransformer{ft},
from{accF},
to{accT} {}
template<UpdateOperation... ops,
typename... Ts,
std::enable_if_t<is_all_same_v<IndexType, Ts...>, bool> = true>
void apply(const Ts... indices) {
auto to_weight_vector = [&](const IndexType index) {
return &featureTransformer.weights[index * Dimensions];
};
auto to_psqt_weight_vector = [&](const IndexType index) {
return &featureTransformer.psqtWeights[index * PSQTBuckets];
};
fused_row_reduce<Vec16Wrapper, Dimensions, ops...>((from.*accPtr).accumulation[Perspective],
(to.*accPtr).accumulation[Perspective],
to_weight_vector(indices)...);
fused_row_reduce<Vec32Wrapper, PSQTBuckets, ops...>(
(from.*accPtr).psqtAccumulation[Perspective], (to.*accPtr).psqtAccumulation[Perspective],
to_psqt_weight_vector(indices)...);
}
};
template<Color Perspective,
IndexType Dimensions,
Accumulator<Dimensions> AccumulatorState::* accPtr>
auto make_accumulator_update_context(
const FeatureTransformer<Dimensions, accPtr>& featureTransformer,
const AccumulatorState& accumulatorFrom,
AccumulatorState& accumulatorTo) noexcept {
return AccumulatorUpdateContext<Perspective, Dimensions, accPtr>{
featureTransformer, accumulatorFrom, accumulatorTo};
}
template<Color Perspective,
IncUpdateDirection Direction,
IndexType TransformedFeatureDimensions,
Accumulator<TransformedFeatureDimensions> AccumulatorState::* accPtr>
void update_accumulator_incremental(
const FeatureTransformer<TransformedFeatureDimensions, accPtr>& featureTransformer,
const Square ksq,
AccumulatorState& target_state,
const AccumulatorState& computed) {
[[maybe_unused]] constexpr bool Forward = Direction == FORWARD;
[[maybe_unused]] constexpr bool Backward = Direction == BACKWARD;
assert(Forward != Backward);
assert((computed.*accPtr).computed[Perspective]);
assert(!(target_state.*accPtr).computed[Perspective]);
// The size must be enough to contain the largest possible update.
// That might depend on the feature set and generally relies on the
// feature set's update cost calculation to be correct and never allow
// updates with more added/removed features than MaxActiveDimensions.
// In this case, the maximum size of both feature addition and removal
// is 2, since we are incrementally updating one move at a time.
FeatureSet::IndexList removed, added;
if constexpr (Forward)
FeatureSet::append_changed_indices<Perspective>(ksq, target_state.dirtyPiece, removed,
added);
else
FeatureSet::append_changed_indices<Perspective>(ksq, computed.dirtyPiece, added, removed);
assert(added.size() == 1 || added.size() == 2);
assert(removed.size() == 1 || removed.size() == 2);
if (Forward)
assert(added.size() <= removed.size());
else
assert(removed.size() <= added.size());
// Workaround compiler warning for uninitialized variables, replicated on
// profile builds on windows with gcc 14.2.0.
// TODO remove once unneeded
sf_assume(added.size() == 1 || added.size() == 2);
sf_assume(removed.size() == 1 || removed.size() == 2);
auto updateContext =
make_accumulator_update_context<Perspective>(featureTransformer, computed, target_state);
if ((Forward && removed.size() == 1) || (Backward && added.size() == 1))
{
assert(added.size() == 1 && removed.size() == 1);
updateContext.template apply<Add, Sub>(added[0], removed[0]);
}
else if (Forward && added.size() == 1)
{
assert(removed.size() == 2);
updateContext.template apply<Add, Sub, Sub>(added[0], removed[0], removed[1]);
}
else if (Backward && removed.size() == 1)
{
assert(added.size() == 2);
updateContext.template apply<Add, Add, Sub>(added[0], added[1], removed[0]);
}
else
{
assert(added.size() == 2 && removed.size() == 2);
updateContext.template apply<Add, Add, Sub, Sub>(added[0], added[1], removed[0],
removed[1]);
}
(target_state.*accPtr).computed[Perspective] = true;
}
template<Color Perspective,
IndexType Dimensions,
Accumulator<Dimensions> AccumulatorState::* accPtr>
void update_accumulator_refresh_cache(
const FeatureTransformer<Dimensions, accPtr>& featureTransformer,
const Position& pos,
AccumulatorState& accumulatorState,
AccumulatorCaches::Cache<Dimensions>& cache) {
using Tiling [[maybe_unused]] = SIMDTiling<Dimensions, Dimensions>;
const Square ksq = pos.square<KING>(Perspective);
auto& entry = cache[ksq][Perspective];
FeatureSet::IndexList removed, added;
for (Color c : {WHITE, BLACK})
{
for (PieceType pt = PAWN; pt <= KING; ++pt)
{
const Piece piece = make_piece(c, pt);
const Bitboard oldBB = entry.byColorBB[c] & entry.byTypeBB[pt];
const Bitboard newBB = pos.pieces(c, pt);
Bitboard toRemove = oldBB & ~newBB;
Bitboard toAdd = newBB & ~oldBB;
while (toRemove)
{
Square sq = pop_lsb(toRemove);
removed.push_back(FeatureSet::make_index<Perspective>(sq, piece, ksq));
}
while (toAdd)
{
Square sq = pop_lsb(toAdd);
added.push_back(FeatureSet::make_index<Perspective>(sq, piece, ksq));
}
}
}
auto& accumulator = accumulatorState.*accPtr;
accumulator.computed[Perspective] = true;
#ifdef VECTOR
const bool combineLast3 =
std::abs((int) removed.size() - (int) added.size()) == 1 && removed.size() + added.size() > 2;
vec_t acc[Tiling::NumRegs];
psqt_vec_t psqt[Tiling::NumPsqtRegs];
for (IndexType j = 0; j < Dimensions / Tiling::TileHeight; ++j)
{
auto* accTile =
reinterpret_cast<vec_t*>(&accumulator.accumulation[Perspective][j * Tiling::TileHeight]);
auto* entryTile = reinterpret_cast<vec_t*>(&entry.accumulation[j * Tiling::TileHeight]);
for (IndexType k = 0; k < Tiling::NumRegs; ++k)
acc[k] = entryTile[k];
std::size_t i = 0;
for (; i < std::min(removed.size(), added.size()) - combineLast3; ++i)
{
IndexType indexR = removed[i];
const IndexType offsetR = Dimensions * indexR + j * Tiling::TileHeight;
auto* columnR = reinterpret_cast<const vec_t*>(&featureTransformer.weights[offsetR]);
IndexType indexA = added[i];
const IndexType offsetA = Dimensions * indexA + j * Tiling::TileHeight;
auto* columnA = reinterpret_cast<const vec_t*>(&featureTransformer.weights[offsetA]);
for (IndexType k = 0; k < Tiling::NumRegs; ++k)
acc[k] = fused<Vec16Wrapper, Add, Sub>(acc[k], columnA[k], columnR[k]);
}
if (combineLast3)
{
IndexType indexR = removed[i];
const IndexType offsetR = Dimensions * indexR + j * Tiling::TileHeight;
auto* columnR = reinterpret_cast<const vec_t*>(&featureTransformer.weights[offsetR]);
IndexType indexA = added[i];
const IndexType offsetA = Dimensions * indexA + j * Tiling::TileHeight;
auto* columnA = reinterpret_cast<const vec_t*>(&featureTransformer.weights[offsetA]);
if (removed.size() > added.size())
{
IndexType indexR2 = removed[i + 1];
const IndexType offsetR2 = Dimensions * indexR2 + j * Tiling::TileHeight;
auto* columnR2 =
reinterpret_cast<const vec_t*>(&featureTransformer.weights[offsetR2]);
for (IndexType k = 0; k < Tiling::NumRegs; ++k)
acc[k] = fused<Vec16Wrapper, Add, Sub, Sub>(acc[k], columnA[k], columnR[k],
columnR2[k]);
}
else
{
IndexType indexA2 = added[i + 1];
const IndexType offsetA2 = Dimensions * indexA2 + j * Tiling::TileHeight;
auto* columnA2 =
reinterpret_cast<const vec_t*>(&featureTransformer.weights[offsetA2]);
for (IndexType k = 0; k < Tiling::NumRegs; ++k)
acc[k] = fused<Vec16Wrapper, Add, Add, Sub>(acc[k], columnA[k], columnA2[k],
columnR[k]);
}
}
else
{
for (; i < removed.size(); ++i)
{
IndexType index = removed[i];
const IndexType offset = Dimensions * index + j * Tiling::TileHeight;
auto* column = reinterpret_cast<const vec_t*>(&featureTransformer.weights[offset]);
for (IndexType k = 0; k < Tiling::NumRegs; ++k)
acc[k] = vec_sub_16(acc[k], column[k]);
}
for (; i < added.size(); ++i)
{
IndexType index = added[i];
const IndexType offset = Dimensions * index + j * Tiling::TileHeight;
auto* column = reinterpret_cast<const vec_t*>(&featureTransformer.weights[offset]);
for (IndexType k = 0; k < Tiling::NumRegs; ++k)
acc[k] = vec_add_16(acc[k], column[k]);
}
}
for (IndexType k = 0; k < Tiling::NumRegs; k++)
vec_store(&entryTile[k], acc[k]);
for (IndexType k = 0; k < Tiling::NumRegs; k++)
vec_store(&accTile[k], acc[k]);
}
for (IndexType j = 0; j < PSQTBuckets / Tiling::PsqtTileHeight; ++j)
{
auto* accTilePsqt = reinterpret_cast<psqt_vec_t*>(
&accumulator.psqtAccumulation[Perspective][j * Tiling::PsqtTileHeight]);
auto* entryTilePsqt =
reinterpret_cast<psqt_vec_t*>(&entry.psqtAccumulation[j * Tiling::PsqtTileHeight]);
for (std::size_t k = 0; k < Tiling::NumPsqtRegs; ++k)
psqt[k] = entryTilePsqt[k];
for (std::size_t i = 0; i < removed.size(); ++i)
{
IndexType index = removed[i];
const IndexType offset = PSQTBuckets * index + j * Tiling::PsqtTileHeight;
auto* columnPsqt =
reinterpret_cast<const psqt_vec_t*>(&featureTransformer.psqtWeights[offset]);
for (std::size_t k = 0; k < Tiling::NumPsqtRegs; ++k)
psqt[k] = vec_sub_psqt_32(psqt[k], columnPsqt[k]);
}
for (std::size_t i = 0; i < added.size(); ++i)
{
IndexType index = added[i];
const IndexType offset = PSQTBuckets * index + j * Tiling::PsqtTileHeight;
auto* columnPsqt =
reinterpret_cast<const psqt_vec_t*>(&featureTransformer.psqtWeights[offset]);
for (std::size_t k = 0; k < Tiling::NumPsqtRegs; ++k)
psqt[k] = vec_add_psqt_32(psqt[k], columnPsqt[k]);
}
for (std::size_t k = 0; k < Tiling::NumPsqtRegs; ++k)
vec_store_psqt(&entryTilePsqt[k], psqt[k]);
for (std::size_t k = 0; k < Tiling::NumPsqtRegs; ++k)
vec_store_psqt(&accTilePsqt[k], psqt[k]);
}
#else
for (const auto index : removed)
{
const IndexType offset = Dimensions * index;
for (IndexType j = 0; j < Dimensions; ++j)
entry.accumulation[j] -= featureTransformer.weights[offset + j];
for (std::size_t k = 0; k < PSQTBuckets; ++k)
entry.psqtAccumulation[k] -= featureTransformer.psqtWeights[index * PSQTBuckets + k];
}
for (const auto index : added)
{
const IndexType offset = Dimensions * index;
for (IndexType j = 0; j < Dimensions; ++j)
entry.accumulation[j] += featureTransformer.weights[offset + j];
for (std::size_t k = 0; k < PSQTBuckets; ++k)
entry.psqtAccumulation[k] += featureTransformer.psqtWeights[index * PSQTBuckets + k];
}
// The accumulator of the refresh entry has been updated.
// Now copy its content to the actual accumulator we were refreshing.
std::memcpy(accumulator.accumulation[Perspective], entry.accumulation,
sizeof(BiasType) * Dimensions);
std::memcpy(accumulator.psqtAccumulation[Perspective], entry.psqtAccumulation,
sizeof(int32_t) * PSQTBuckets);
#endif
for (Color c : {WHITE, BLACK})
entry.byColorBB[c] = pos.pieces(c);
for (PieceType pt = PAWN; pt <= KING; ++pt)
entry.byTypeBB[pt] = pos.pieces(pt);
}
}
}
+86 -3
View File
@@ -21,23 +21,43 @@
#ifndef NNUE_ACCUMULATOR_H_INCLUDED
#define NNUE_ACCUMULATOR_H_INCLUDED
#include <array>
#include <cstddef>
#include <cstdint>
#include <cstring>
#include <vector>
#include "../types.h"
#include "nnue_architecture.h"
#include "nnue_common.h"
namespace Stockfish {
class Position;
}
namespace Stockfish::Eval::NNUE {
using BiasType = std::int16_t;
using PSQTWeightType = std::int32_t;
using IndexType = std::uint32_t;
struct Networks;
template<IndexType Size>
struct alignas(CacheLineSize) Accumulator;
struct AccumulatorState;
template<IndexType TransformedFeatureDimensions,
Accumulator<TransformedFeatureDimensions> AccumulatorState::* accPtr>
class FeatureTransformer;
// Class that holds the result of affine transformation of input features
template<IndexType Size>
struct alignas(CacheLineSize) Accumulator {
std::int16_t accumulation[COLOR_NB][Size];
std::int32_t psqtAccumulation[COLOR_NB][PSQTBuckets];
bool computed[COLOR_NB];
std::int16_t accumulation[COLOR_NB][Size];
std::int32_t psqtAccumulation[COLOR_NB][PSQTBuckets];
std::array<bool, COLOR_NB> computed;
};
@@ -95,6 +115,69 @@ struct AccumulatorCaches {
Cache<TransformedFeatureDimensionsSmall> small;
};
struct AccumulatorState {
Accumulator<TransformedFeatureDimensionsBig> accumulatorBig;
Accumulator<TransformedFeatureDimensionsSmall> accumulatorSmall;
DirtyPiece dirtyPiece;
void reset(const DirtyPiece& dp) noexcept;
};
class AccumulatorStack {
public:
AccumulatorStack() :
m_accumulators(MAX_PLY + 1),
m_current_idx{} {}
[[nodiscard]] const AccumulatorState& latest() const noexcept;
void
reset(const Position& rootPos, const Networks& networks, AccumulatorCaches& caches) noexcept;
void push(const DirtyPiece& dirtyPiece) noexcept;
void pop() noexcept;
template<IndexType Dimensions, Accumulator<Dimensions> AccumulatorState::* accPtr>
void evaluate(const Position& pos,
const FeatureTransformer<Dimensions, accPtr>& featureTransformer,
AccumulatorCaches::Cache<Dimensions>& cache) noexcept;
private:
[[nodiscard]] AccumulatorState& mut_latest() noexcept;
template<Color Perspective,
IndexType Dimensions,
Accumulator<Dimensions> AccumulatorState::* accPtr>
void evaluate_side(const Position& pos,
const FeatureTransformer<Dimensions, accPtr>& featureTransformer,
AccumulatorCaches::Cache<Dimensions>& cache) noexcept;
template<Color Perspective,
IndexType Dimensions,
Accumulator<Dimensions> AccumulatorState::* accPtr>
[[nodiscard]] std::size_t find_last_usable_accumulator() const noexcept;
template<Color Perspective,
IndexType Dimensions,
Accumulator<Dimensions> AccumulatorState::* accPtr>
void
forward_update_incremental(const Position& pos,
const FeatureTransformer<Dimensions, accPtr>& featureTransformer,
const std::size_t begin) noexcept;
template<Color Perspective,
IndexType Dimensions,
Accumulator<Dimensions> AccumulatorState::* accPtr>
void
backward_update_incremental(const Position& pos,
const FeatureTransformer<Dimensions, accPtr>& featureTransformer,
const std::size_t end) noexcept;
std::vector<AccumulatorState> m_accumulators;
std::size_t m_current_idx;
};
} // namespace Stockfish::Eval::NNUE
#endif // NNUE_ACCUMULATOR_H_INCLUDED
+5
View File
@@ -279,6 +279,11 @@ inline void write_leb_128(std::ostream& stream, const IntType* values, std::size
flush();
}
enum IncUpdateDirection {
FORWARD,
BACKWARD
};
} // namespace Stockfish::Eval::NNUE
#endif // #ifndef NNUE_COMMON_H_INCLUDED
+61 -414
View File
@@ -22,12 +22,9 @@
#define NNUE_FEATURE_TRANSFORMER_H_INCLUDED
#include <algorithm>
#include <cassert>
#include <cstdint>
#include <cstring>
#include <iosfwd>
#include <type_traits>
#include <utility>
#include "../position.h"
#include "../types.h"
@@ -41,11 +38,6 @@ using BiasType = std::int16_t;
using WeightType = std::int16_t;
using PSQTWeightType = std::int32_t;
enum IncUpdateDirection {
FORWARD,
BACKWARDS
};
// If vector instructions are enabled, we update and refresh the
// accumulator tile by tile such that each tile fits in the CPU's
// vector registers.
@@ -149,6 +141,60 @@ using psqt_vec_t = int32x4_t;
#endif
struct Vec16Wrapper {
#ifdef VECTOR
using type = vec_t;
static type add(const type& lhs, const type& rhs) { return vec_add_16(lhs, rhs); }
static type sub(const type& lhs, const type& rhs) { return vec_sub_16(lhs, rhs); }
#else
using type = BiasType;
static type add(const type& lhs, const type& rhs) { return lhs + rhs; }
static type sub(const type& lhs, const type& rhs) { return lhs - rhs; }
#endif
};
struct Vec32Wrapper {
#ifdef VECTOR
using type = psqt_vec_t;
static type add(const type& lhs, const type& rhs) { return vec_add_psqt_32(lhs, rhs); }
static type sub(const type& lhs, const type& rhs) { return vec_sub_psqt_32(lhs, rhs); }
#else
using type = PSQTWeightType;
static type add(const type& lhs, const type& rhs) { return lhs + rhs; }
static type sub(const type& lhs, const type& rhs) { return lhs - rhs; }
#endif
};
enum UpdateOperation {
Add,
Sub
};
template<typename VecWrapper,
UpdateOperation... ops,
std::enable_if_t<sizeof...(ops) == 0, bool> = true>
typename VecWrapper::type fused(const typename VecWrapper::type& in) {
return in;
}
template<typename VecWrapper,
UpdateOperation update_op,
UpdateOperation... ops,
typename T,
typename... Ts,
std::enable_if_t<is_all_same_v<typename VecWrapper::type, T, Ts...>, bool> = true,
std::enable_if_t<sizeof...(ops) == sizeof...(Ts), bool> = true>
typename VecWrapper::type
fused(const typename VecWrapper::type& in, const T& operand, const Ts&... operands) {
switch (update_op)
{
case Add :
return fused<VecWrapper, ops...>(VecWrapper::add(in, operand), operands...);
case Sub :
return fused<VecWrapper, ops...>(VecWrapper::sub(in, operand), operands...);
}
}
// Returns the inverse of a permutation
template<std::size_t Len>
constexpr std::array<std::size_t, Len>
@@ -247,15 +293,11 @@ class SIMDTiling {
// Input feature converter
template<IndexType TransformedFeatureDimensions,
Accumulator<TransformedFeatureDimensions> StateInfo::* accPtr>
Accumulator<TransformedFeatureDimensions> AccumulatorState::* accPtr>
class FeatureTransformer {
// Number of output dimensions for one side
static constexpr IndexType HalfDimensions = TransformedFeatureDimensions;
static constexpr bool Big = TransformedFeatureDimensions == TransformedFeatureDimensionsBig;
private:
using Tiling = SIMDTiling<TransformedFeatureDimensions, HalfDimensions>;
public:
// Output type
@@ -347,19 +389,21 @@ class FeatureTransformer {
// Convert input features
std::int32_t transform(const Position& pos,
AccumulatorStack& accumulatorStack,
AccumulatorCaches::Cache<HalfDimensions>* cache,
OutputType* output,
int bucket) const {
update_accumulator<WHITE>(pos, cache);
update_accumulator<BLACK>(pos, cache);
accumulatorStack.evaluate(pos, *this, *cache);
const auto& accumulatorState = accumulatorStack.latest();
const Color perspectives[2] = {pos.side_to_move(), ~pos.side_to_move()};
const auto& psqtAccumulation = (pos.state()->*accPtr).psqtAccumulation;
const auto& psqtAccumulation = (accumulatorState.*accPtr).psqtAccumulation;
const auto psqt =
(psqtAccumulation[perspectives[0]][bucket] - psqtAccumulation[perspectives[1]][bucket])
/ 2;
const auto& accumulation = (pos.state()->*accPtr).accumulation;
const auto& accumulation = (accumulatorState.*accPtr).accumulation;
for (IndexType p = 0; p < 2; ++p)
{
@@ -472,403 +516,6 @@ class FeatureTransformer {
return psqt;
} // end of function transform()
private:
// Given a computed accumulator, computes the accumulator of another position.
template<Color Perspective, IncUpdateDirection Direction = FORWARD>
void update_accumulator_incremental(const Square ksq,
StateInfo* target_state,
const StateInfo* computed) const {
[[maybe_unused]] constexpr bool Forward = Direction == FORWARD;
[[maybe_unused]] constexpr bool Backwards = Direction == BACKWARDS;
assert((computed->*accPtr).computed[Perspective]);
StateInfo* next = Forward ? computed->next : computed->previous;
assert(next != nullptr);
assert(!(next->*accPtr).computed[Perspective]);
// The size must be enough to contain the largest possible update.
// That might depend on the feature set and generally relies on the
// feature set's update cost calculation to be correct and never allow
// updates with more added/removed features than MaxActiveDimensions.
// In this case, the maximum size of both feature addition and removal
// is 2, since we are incrementally updating one move at a time.
FeatureSet::IndexList removed, added;
if constexpr (Forward)
FeatureSet::append_changed_indices<Perspective>(ksq, next->dirtyPiece, removed, added);
else
FeatureSet::append_changed_indices<Perspective>(ksq, computed->dirtyPiece, added,
removed);
if (removed.size() == 0 && added.size() == 0)
{
std::memcpy((next->*accPtr).accumulation[Perspective],
(computed->*accPtr).accumulation[Perspective],
HalfDimensions * sizeof(BiasType));
std::memcpy((next->*accPtr).psqtAccumulation[Perspective],
(computed->*accPtr).psqtAccumulation[Perspective],
PSQTBuckets * sizeof(PSQTWeightType));
}
else
{
assert(added.size() == 1 || added.size() == 2);
assert(removed.size() == 1 || removed.size() == 2);
if (Forward)
assert(added.size() <= removed.size());
else
assert(removed.size() <= added.size());
#ifdef VECTOR
auto* accIn =
reinterpret_cast<const vec_t*>(&(computed->*accPtr).accumulation[Perspective][0]);
auto* accOut = reinterpret_cast<vec_t*>(&(next->*accPtr).accumulation[Perspective][0]);
const IndexType offsetA0 = HalfDimensions * added[0];
auto* columnA0 = reinterpret_cast<const vec_t*>(&weights[offsetA0]);
const IndexType offsetR0 = HalfDimensions * removed[0];
auto* columnR0 = reinterpret_cast<const vec_t*>(&weights[offsetR0]);
if ((Forward && removed.size() == 1) || (Backwards && added.size() == 1))
{
assert(added.size() == 1 && removed.size() == 1);
for (IndexType i = 0; i < HalfDimensions * sizeof(WeightType) / sizeof(vec_t); ++i)
accOut[i] = vec_add_16(vec_sub_16(accIn[i], columnR0[i]), columnA0[i]);
}
else if (Forward && added.size() == 1)
{
assert(removed.size() == 2);
const IndexType offsetR1 = HalfDimensions * removed[1];
auto* columnR1 = reinterpret_cast<const vec_t*>(&weights[offsetR1]);
for (IndexType i = 0; i < HalfDimensions * sizeof(WeightType) / sizeof(vec_t); ++i)
accOut[i] = vec_sub_16(vec_add_16(accIn[i], columnA0[i]),
vec_add_16(columnR0[i], columnR1[i]));
}
else if (Backwards && removed.size() == 1)
{
assert(added.size() == 2);
const IndexType offsetA1 = HalfDimensions * added[1];
auto* columnA1 = reinterpret_cast<const vec_t*>(&weights[offsetA1]);
for (IndexType i = 0; i < HalfDimensions * sizeof(WeightType) / sizeof(vec_t); ++i)
accOut[i] = vec_add_16(vec_add_16(accIn[i], columnA0[i]),
vec_sub_16(columnA1[i], columnR0[i]));
}
else
{
assert(added.size() == 2 && removed.size() == 2);
const IndexType offsetA1 = HalfDimensions * added[1];
auto* columnA1 = reinterpret_cast<const vec_t*>(&weights[offsetA1]);
const IndexType offsetR1 = HalfDimensions * removed[1];
auto* columnR1 = reinterpret_cast<const vec_t*>(&weights[offsetR1]);
for (IndexType i = 0; i < HalfDimensions * sizeof(WeightType) / sizeof(vec_t); ++i)
accOut[i] =
vec_add_16(accIn[i], vec_sub_16(vec_add_16(columnA0[i], columnA1[i]),
vec_add_16(columnR0[i], columnR1[i])));
}
auto* accPsqtIn = reinterpret_cast<const psqt_vec_t*>(
&(computed->*accPtr).psqtAccumulation[Perspective][0]);
auto* accPsqtOut =
reinterpret_cast<psqt_vec_t*>(&(next->*accPtr).psqtAccumulation[Perspective][0]);
const IndexType offsetPsqtA0 = PSQTBuckets * added[0];
auto* columnPsqtA0 = reinterpret_cast<const psqt_vec_t*>(&psqtWeights[offsetPsqtA0]);
const IndexType offsetPsqtR0 = PSQTBuckets * removed[0];
auto* columnPsqtR0 = reinterpret_cast<const psqt_vec_t*>(&psqtWeights[offsetPsqtR0]);
if ((Forward && removed.size() == 1)
|| (Backwards && added.size() == 1)) // added.size() == removed.size() == 1
{
for (std::size_t i = 0;
i < PSQTBuckets * sizeof(PSQTWeightType) / sizeof(psqt_vec_t); ++i)
accPsqtOut[i] = vec_add_psqt_32(vec_sub_psqt_32(accPsqtIn[i], columnPsqtR0[i]),
columnPsqtA0[i]);
}
else if (Forward && added.size() == 1)
{
const IndexType offsetPsqtR1 = PSQTBuckets * removed[1];
auto* columnPsqtR1 =
reinterpret_cast<const psqt_vec_t*>(&psqtWeights[offsetPsqtR1]);
for (std::size_t i = 0;
i < PSQTBuckets * sizeof(PSQTWeightType) / sizeof(psqt_vec_t); ++i)
accPsqtOut[i] =
vec_sub_psqt_32(vec_add_psqt_32(accPsqtIn[i], columnPsqtA0[i]),
vec_add_psqt_32(columnPsqtR0[i], columnPsqtR1[i]));
}
else if (Backwards && removed.size() == 1)
{
const IndexType offsetPsqtA1 = PSQTBuckets * added[1];
auto* columnPsqtA1 =
reinterpret_cast<const psqt_vec_t*>(&psqtWeights[offsetPsqtA1]);
for (std::size_t i = 0;
i < PSQTBuckets * sizeof(PSQTWeightType) / sizeof(psqt_vec_t); ++i)
accPsqtOut[i] =
vec_add_psqt_32(vec_add_psqt_32(accPsqtIn[i], columnPsqtA0[i]),
vec_sub_psqt_32(columnPsqtA1[i], columnPsqtR0[i]));
}
else
{
const IndexType offsetPsqtA1 = PSQTBuckets * added[1];
auto* columnPsqtA1 =
reinterpret_cast<const psqt_vec_t*>(&psqtWeights[offsetPsqtA1]);
const IndexType offsetPsqtR1 = PSQTBuckets * removed[1];
auto* columnPsqtR1 =
reinterpret_cast<const psqt_vec_t*>(&psqtWeights[offsetPsqtR1]);
for (std::size_t i = 0;
i < PSQTBuckets * sizeof(PSQTWeightType) / sizeof(psqt_vec_t); ++i)
accPsqtOut[i] = vec_add_psqt_32(
accPsqtIn[i],
vec_sub_psqt_32(vec_add_psqt_32(columnPsqtA0[i], columnPsqtA1[i]),
vec_add_psqt_32(columnPsqtR0[i], columnPsqtR1[i])));
}
#else
std::memcpy((next->*accPtr).accumulation[Perspective],
(computed->*accPtr).accumulation[Perspective],
HalfDimensions * sizeof(BiasType));
std::memcpy((next->*accPtr).psqtAccumulation[Perspective],
(computed->*accPtr).psqtAccumulation[Perspective],
PSQTBuckets * sizeof(PSQTWeightType));
// Difference calculation for the deactivated features
for (const auto index : removed)
{
const IndexType offset = HalfDimensions * index;
for (IndexType i = 0; i < HalfDimensions; ++i)
(next->*accPtr).accumulation[Perspective][i] -= weights[offset + i];
for (std::size_t i = 0; i < PSQTBuckets; ++i)
(next->*accPtr).psqtAccumulation[Perspective][i] -=
psqtWeights[index * PSQTBuckets + i];
}
// Difference calculation for the activated features
for (const auto index : added)
{
const IndexType offset = HalfDimensions * index;
for (IndexType i = 0; i < HalfDimensions; ++i)
(next->*accPtr).accumulation[Perspective][i] += weights[offset + i];
for (std::size_t i = 0; i < PSQTBuckets; ++i)
(next->*accPtr).psqtAccumulation[Perspective][i] +=
psqtWeights[index * PSQTBuckets + i];
}
#endif
}
(next->*accPtr).computed[Perspective] = true;
if (next != target_state)
update_accumulator_incremental<Perspective, Direction>(ksq, target_state, next);
}
template<Color Perspective>
void update_accumulator_refresh_cache(const Position& pos,
AccumulatorCaches::Cache<HalfDimensions>* cache) const {
assert(cache != nullptr);
Square ksq = pos.square<KING>(Perspective);
auto& entry = (*cache)[ksq][Perspective];
FeatureSet::IndexList removed, added;
for (Color c : {WHITE, BLACK})
{
for (PieceType pt = PAWN; pt <= KING; ++pt)
{
const Piece piece = make_piece(c, pt);
const Bitboard oldBB = entry.byColorBB[c] & entry.byTypeBB[pt];
const Bitboard newBB = pos.pieces(c, pt);
Bitboard toRemove = oldBB & ~newBB;
Bitboard toAdd = newBB & ~oldBB;
while (toRemove)
{
Square sq = pop_lsb(toRemove);
removed.push_back(FeatureSet::make_index<Perspective>(sq, piece, ksq));
}
while (toAdd)
{
Square sq = pop_lsb(toAdd);
added.push_back(FeatureSet::make_index<Perspective>(sq, piece, ksq));
}
}
}
auto& accumulator = pos.state()->*accPtr;
accumulator.computed[Perspective] = true;
#ifdef VECTOR
vec_t acc[Tiling::NumRegs];
psqt_vec_t psqt[Tiling::NumPsqtRegs];
for (IndexType j = 0; j < HalfDimensions / Tiling::TileHeight; ++j)
{
auto* accTile = reinterpret_cast<vec_t*>(
&accumulator.accumulation[Perspective][j * Tiling::TileHeight]);
auto* entryTile = reinterpret_cast<vec_t*>(&entry.accumulation[j * Tiling::TileHeight]);
for (IndexType k = 0; k < Tiling::NumRegs; ++k)
acc[k] = entryTile[k];
std::size_t i = 0;
for (; i < std::min(removed.size(), added.size()); ++i)
{
IndexType indexR = removed[i];
const IndexType offsetR = HalfDimensions * indexR + j * Tiling::TileHeight;
auto* columnR = reinterpret_cast<const vec_t*>(&weights[offsetR]);
IndexType indexA = added[i];
const IndexType offsetA = HalfDimensions * indexA + j * Tiling::TileHeight;
auto* columnA = reinterpret_cast<const vec_t*>(&weights[offsetA]);
for (IndexType k = 0; k < Tiling::NumRegs; ++k)
acc[k] = vec_add_16(acc[k], vec_sub_16(columnA[k], columnR[k]));
}
for (; i < removed.size(); ++i)
{
IndexType index = removed[i];
const IndexType offset = HalfDimensions * index + j * Tiling::TileHeight;
auto* column = reinterpret_cast<const vec_t*>(&weights[offset]);
for (IndexType k = 0; k < Tiling::NumRegs; ++k)
acc[k] = vec_sub_16(acc[k], column[k]);
}
for (; i < added.size(); ++i)
{
IndexType index = added[i];
const IndexType offset = HalfDimensions * index + j * Tiling::TileHeight;
auto* column = reinterpret_cast<const vec_t*>(&weights[offset]);
for (IndexType k = 0; k < Tiling::NumRegs; ++k)
acc[k] = vec_add_16(acc[k], column[k]);
}
for (IndexType k = 0; k < Tiling::NumRegs; k++)
vec_store(&entryTile[k], acc[k]);
for (IndexType k = 0; k < Tiling::NumRegs; k++)
vec_store(&accTile[k], acc[k]);
}
for (IndexType j = 0; j < PSQTBuckets / Tiling::PsqtTileHeight; ++j)
{
auto* accTilePsqt = reinterpret_cast<psqt_vec_t*>(
&accumulator.psqtAccumulation[Perspective][j * Tiling::PsqtTileHeight]);
auto* entryTilePsqt =
reinterpret_cast<psqt_vec_t*>(&entry.psqtAccumulation[j * Tiling::PsqtTileHeight]);
for (std::size_t k = 0; k < Tiling::NumPsqtRegs; ++k)
psqt[k] = entryTilePsqt[k];
for (std::size_t i = 0; i < removed.size(); ++i)
{
IndexType index = removed[i];
const IndexType offset = PSQTBuckets * index + j * Tiling::PsqtTileHeight;
auto* columnPsqt = reinterpret_cast<const psqt_vec_t*>(&psqtWeights[offset]);
for (std::size_t k = 0; k < Tiling::NumPsqtRegs; ++k)
psqt[k] = vec_sub_psqt_32(psqt[k], columnPsqt[k]);
}
for (std::size_t i = 0; i < added.size(); ++i)
{
IndexType index = added[i];
const IndexType offset = PSQTBuckets * index + j * Tiling::PsqtTileHeight;
auto* columnPsqt = reinterpret_cast<const psqt_vec_t*>(&psqtWeights[offset]);
for (std::size_t k = 0; k < Tiling::NumPsqtRegs; ++k)
psqt[k] = vec_add_psqt_32(psqt[k], columnPsqt[k]);
}
for (std::size_t k = 0; k < Tiling::NumPsqtRegs; ++k)
vec_store_psqt(&entryTilePsqt[k], psqt[k]);
for (std::size_t k = 0; k < Tiling::NumPsqtRegs; ++k)
vec_store_psqt(&accTilePsqt[k], psqt[k]);
}
#else
for (const auto index : removed)
{
const IndexType offset = HalfDimensions * index;
for (IndexType j = 0; j < HalfDimensions; ++j)
entry.accumulation[j] -= weights[offset + j];
for (std::size_t k = 0; k < PSQTBuckets; ++k)
entry.psqtAccumulation[k] -= psqtWeights[index * PSQTBuckets + k];
}
for (const auto index : added)
{
const IndexType offset = HalfDimensions * index;
for (IndexType j = 0; j < HalfDimensions; ++j)
entry.accumulation[j] += weights[offset + j];
for (std::size_t k = 0; k < PSQTBuckets; ++k)
entry.psqtAccumulation[k] += psqtWeights[index * PSQTBuckets + k];
}
// The accumulator of the refresh entry has been updated.
// Now copy its content to the actual accumulator we were refreshing.
std::memcpy(accumulator.accumulation[Perspective], entry.accumulation,
sizeof(BiasType) * HalfDimensions);
std::memcpy(accumulator.psqtAccumulation[Perspective], entry.psqtAccumulation,
sizeof(int32_t) * PSQTBuckets);
#endif
for (Color c : {WHITE, BLACK})
entry.byColorBB[c] = pos.pieces(c);
for (PieceType pt = PAWN; pt <= KING; ++pt)
entry.byTypeBB[pt] = pos.pieces(pt);
}
template<Color Perspective>
void update_accumulator(const Position& pos,
AccumulatorCaches::Cache<HalfDimensions>* cache) const {
StateInfo* st = pos.state();
if ((st->*accPtr).computed[Perspective])
return; // nothing to do
[[maybe_unused]] // only used when !Big
int gain = FeatureSet::refresh_cost(pos);
// Look for a usable already computed accumulator of an earlier position.
// When computing the small accumulator, we keep track of the estimated gain in
// terms of features to be added/subtracted.
// When computing the big accumulator, we expect to be able to reuse any
// accumulators, so we always try to do an incremental update.
do
{
if (FeatureSet::requires_refresh(st, Perspective)
|| (!Big && (gain -= FeatureSet::update_cost(st) < 0)) || !st->previous
|| st->previous->next != st)
{
// compute accumulator from scratch for this position
update_accumulator_refresh_cache<Perspective>(pos, cache);
if (Big && st != pos.state())
// when computing a big accumulator from scratch we can use it to
// efficiently compute the accumulator backwards, until we get to a king
// move. We expect that we will need these accumulators later anyway, so
// computing them now will save some work.
update_accumulator_incremental<Perspective, BACKWARDS>(
pos.square<KING>(Perspective), st, pos.state());
return;
}
st = st->previous;
} while (!(st->*accPtr).computed[Perspective]);
// Start from the oldest computed accumulator, update all the
// accumulators up to the current position.
update_accumulator_incremental<Perspective>(pos.square<KING>(Perspective), pos.state(), st);
}
template<IndexType Size>
friend struct AccumulatorCaches::Cache;
alignas(CacheLineSize) BiasType biases[HalfDimensions];
alignas(CacheLineSize) WeightType weights[HalfDimensions * InputDimensions];
alignas(CacheLineSize) PSQTWeightType psqtWeights[InputDimensions * PSQTBuckets];
+8 -7
View File
@@ -120,9 +120,12 @@ trace(Position& pos, const Eval::NNUE::Networks& networks, Eval::NNUE::Accumulat
format_cp_compact(value, &board[y + 2][x + 2], pos);
};
AccumulatorStack accumulators;
accumulators.reset(pos, networks, caches);
// We estimate the value of each piece by doing a differential evaluation from
// the current base eval, simulating the removal of the piece from its square.
auto [psqt, positional] = networks.big.evaluate(pos, &caches.big);
auto [psqt, positional] = networks.big.evaluate(pos, accumulators, &caches.big);
Value base = psqt + positional;
base = pos.side_to_move() == WHITE ? base : -base;
@@ -135,18 +138,15 @@ trace(Position& pos, const Eval::NNUE::Networks& networks, Eval::NNUE::Accumulat
if (pc != NO_PIECE && type_of(pc) != KING)
{
auto st = pos.state();
pos.remove_piece(sq);
st->accumulatorBig.computed[WHITE] = st->accumulatorBig.computed[BLACK] = false;
std::tie(psqt, positional) = networks.big.evaluate(pos, &caches.big);
accumulators.reset(pos, networks, caches);
std::tie(psqt, positional) = networks.big.evaluate(pos, accumulators, &caches.big);
Value eval = psqt + positional;
eval = pos.side_to_move() == WHITE ? eval : -eval;
v = base - eval;
pos.put_piece(pc, sq);
st->accumulatorBig.computed[WHITE] = st->accumulatorBig.computed[BLACK] = false;
}
writeSquare(f, r, pc, v);
@@ -157,7 +157,8 @@ trace(Position& pos, const Eval::NNUE::Networks& networks, Eval::NNUE::Accumulat
ss << board[row] << '\n';
ss << '\n';
auto t = networks.big.trace_evaluate(pos, &caches.big);
accumulators.reset(pos, networks, caches);
auto t = networks.big.trace_evaluate(pos, accumulators, &caches.big);
ss << " NNUE network contributions "
<< (pos.side_to_move() == WHITE ? "(White to move)" : "(Black to move)") << std::endl
-1
View File
@@ -35,7 +35,6 @@ template<bool Root>
uint64_t perft(Position& pos, Depth depth) {
StateInfo st;
ASSERT_ALIGNED(&st, Eval::NNUE::CacheLineSize);
uint64_t cnt, nodes = 0;
const bool leaf = (depth == 2);
+35 -43
View File
@@ -34,7 +34,6 @@
#include "bitboard.h"
#include "misc.h"
#include "movegen.h"
#include "nnue/nnue_common.h"
#include "syzygy/tbprobe.h"
#include "tt.h"
#include "uci.h"
@@ -55,8 +54,8 @@ namespace {
constexpr std::string_view PieceToChar(" PNBRQK pnbrqk");
constexpr Piece Pieces[] = {W_PAWN, W_KNIGHT, W_BISHOP, W_ROOK, W_QUEEN, W_KING,
B_PAWN, B_KNIGHT, B_BISHOP, B_ROOK, B_QUEEN, B_KING};
static constexpr Piece Pieces[] = {W_PAWN, W_KNIGHT, W_BISHOP, W_ROOK, W_QUEEN, W_KING,
B_PAWN, B_KNIGHT, B_BISHOP, B_ROOK, B_QUEEN, B_KING};
} // namespace
@@ -83,7 +82,6 @@ std::ostream& operator<<(std::ostream& os, const Position& pos) {
if (int(Tablebases::MaxCardinality) >= popcount(pos.pieces()) && !pos.can_castle(ANY_CASTLING))
{
StateInfo st;
ASSERT_ALIGNED(&st, Eval::NNUE::CacheLineSize);
Position p;
p.set(pos.fen(), pos.is_chess960(), &st);
@@ -272,7 +270,7 @@ Position& Position::set(const string& fenStr, bool isChess960, StateInfo* si) {
// a) side to move have a pawn threatening epSquare
// b) there is an enemy pawn in front of epSquare
// c) there is no piece on epSquare or behind epSquare
enpassant = pawn_attacks_bb(~sideToMove, st->epSquare) & pieces(sideToMove, PAWN)
enpassant = attacks_bb<PAWN>(st->epSquare, ~sideToMove) & pieces(sideToMove, PAWN)
&& (pieces(~sideToMove, PAWN) & (st->epSquare + pawn_push(~sideToMove)))
&& !(pieces() & (st->epSquare | (st->epSquare + pawn_push(sideToMove))));
}
@@ -323,7 +321,7 @@ void Position::set_check_info() const {
Square ksq = square<KING>(~sideToMove);
st->checkSquares[PAWN] = pawn_attacks_bb(~sideToMove, ksq);
st->checkSquares[PAWN] = attacks_bb<PAWN>(ksq, ~sideToMove);
st->checkSquares[KNIGHT] = attacks_bb<KNIGHT>(ksq);
st->checkSquares[BISHOP] = attacks_bb<BISHOP>(ksq, pieces());
st->checkSquares[ROOK] = attacks_bb<ROOK>(ksq, pieces());
@@ -489,8 +487,8 @@ Bitboard Position::attackers_to(Square s, Bitboard occupied) const {
return (attacks_bb<ROOK>(s, occupied) & pieces(ROOK, QUEEN))
| (attacks_bb<BISHOP>(s, occupied) & pieces(BISHOP, QUEEN))
| (pawn_attacks_bb(BLACK, s) & pieces(WHITE, PAWN))
| (pawn_attacks_bb(WHITE, s) & pieces(BLACK, PAWN))
| (attacks_bb<PAWN>(s, BLACK) & pieces(WHITE, PAWN))
| (attacks_bb<PAWN>(s, WHITE) & pieces(BLACK, PAWN))
| (attacks_bb<KNIGHT>(s) & pieces(KNIGHT)) | (attacks_bb<KING>(s) & pieces(KING));
}
@@ -500,7 +498,7 @@ bool Position::attackers_to_exist(Square s, Bitboard occupied, Color c) const {
&& (attacks_bb<ROOK>(s, occupied) & pieces(c, ROOK, QUEEN)))
|| ((attacks_bb<BISHOP>(s) & pieces(c, BISHOP, QUEEN))
&& (attacks_bb<BISHOP>(s, occupied) & pieces(c, BISHOP, QUEEN)))
|| (((pawn_attacks_bb(~c, s) & pieces(PAWN)) | (attacks_bb<KNIGHT>(s) & pieces(KNIGHT))
|| (((attacks_bb<PAWN>(s, ~c) & pieces(PAWN)) | (attacks_bb<KNIGHT>(s) & pieces(KNIGHT))
| (attacks_bb<KING>(s) & pieces(KING)))
& pieces(c));
}
@@ -560,7 +558,7 @@ bool Position::legal(Move m) const {
// A non-king move is legal if and only if it is not pinned or it
// is moving along the ray towards or away from the king.
return !(blockers_for_king(us) & from) || aligned(from, to, square<KING>(us));
return !(blockers_for_king(us) & from) || line_bb(from, to) & pieces(us, KING);
}
@@ -599,9 +597,9 @@ bool Position::pseudo_legal(const Move m) const {
if ((Rank8BB | Rank1BB) & to)
return false;
if (!(pawn_attacks_bb(us, from) & pieces(~us) & to) // Not a capture
&& !((from + pawn_push(us) == to) && empty(to)) // Not a single push
&& !((from + 2 * pawn_push(us) == to) // Not a double push
if (!(attacks_bb<PAWN>(from, us) & pieces(~us) & to) // Not a capture
&& !((from + pawn_push(us) == to) && empty(to)) // Not a single push
&& !((from + 2 * pawn_push(us) == to) // Not a double push
&& (relative_rank(us, from) == RANK_2) && empty(to) && empty(to - pawn_push(us))))
return false;
}
@@ -648,7 +646,7 @@ bool Position::gives_check(Move m) const {
// Is there a discovered check?
if (blockers_for_king(~sideToMove) & from)
return !aligned(from, to, square<KING>(~sideToMove)) || m.type_of() == CASTLING;
return !(line_bb(from, to) & pieces(~sideToMove, KING)) || m.type_of() == CASTLING;
switch (m.type_of())
{
@@ -656,7 +654,7 @@ bool Position::gives_check(Move m) const {
return false;
case PROMOTION :
return attacks_bb(m.promotion_type(), to, pieces() ^ from) & square<KING>(~sideToMove);
return attacks_bb(m.promotion_type(), to, pieces() ^ from) & pieces(~sideToMove, KING);
// En passant capture with check? We have already handled the case of direct
// checks and ordinary discovered check, so the only case we need to handle
@@ -685,10 +683,10 @@ bool Position::gives_check(Move m) const {
// moves should be filtered out before this function is called.
// If a pointer to the TT table is passed, the entry for the new position
// will be prefetched
void Position::do_move(Move m,
StateInfo& newSt,
bool givesCheck,
const TranspositionTable* tt = nullptr) {
DirtyPiece Position::do_move(Move m,
StateInfo& newSt,
bool givesCheck,
const TranspositionTable* tt = nullptr) {
assert(m.is_ok());
assert(&newSt != st);
@@ -709,11 +707,7 @@ void Position::do_move(Move m,
++st->rule50;
++st->pliesFromNull;
// Used by NNUE
st->accumulatorBig.computed[WHITE] = st->accumulatorBig.computed[BLACK] =
st->accumulatorSmall.computed[WHITE] = st->accumulatorSmall.computed[BLACK] = false;
auto& dp = st->dirtyPiece;
DirtyPiece dp;
dp.dirty_num = 1;
Color us = sideToMove;
@@ -733,14 +727,13 @@ void Position::do_move(Move m,
assert(captured == make_piece(us, ROOK));
Square rfrom, rto;
do_castling<true>(us, from, to, rfrom, rto);
do_castling<true>(us, from, to, rfrom, rto, &dp);
k ^= Zobrist::psq[captured][rfrom] ^ Zobrist::psq[captured][rto];
st->nonPawnKey[us] ^= Zobrist::psq[captured][rfrom] ^ Zobrist::psq[captured][rto];
captured = NO_PIECE;
}
if (captured)
else if (captured)
{
Square capsq = to;
@@ -818,7 +811,7 @@ void Position::do_move(Move m,
{
// Set en passant square if the moved pawn can be captured
if ((int(to) ^ int(from)) == 16
&& (pawn_attacks_bb(us, to - pawn_push(us)) & pieces(them, PAWN)))
&& (attacks_bb<PAWN>(to - pawn_push(us), us) & pieces(them, PAWN)))
{
st->epSquare = to - pawn_push(us);
k ^= Zobrist::enpassant[file_of(st->epSquare)];
@@ -906,6 +899,8 @@ void Position::do_move(Move m,
}
assert(pos_is_ok());
return dp;
}
@@ -975,23 +970,25 @@ void Position::undo_move(Move m) {
// Helper used to do/undo a castling move. This is a bit
// tricky in Chess960 where from/to squares can overlap.
template<bool Do>
void Position::do_castling(Color us, Square from, Square& to, Square& rfrom, Square& rto) {
void Position::do_castling(
Color us, Square from, Square& to, Square& rfrom, Square& rto, DirtyPiece* const dp) {
bool kingSide = to > from;
rfrom = to; // Castling is encoded as "king captures friendly rook"
rto = relative_square(us, kingSide ? SQ_F1 : SQ_D1);
to = relative_square(us, kingSide ? SQ_G1 : SQ_C1);
assert(!Do || dp);
if (Do)
{
auto& dp = st->dirtyPiece;
dp.piece[0] = make_piece(us, KING);
dp.from[0] = from;
dp.to[0] = to;
dp.piece[1] = make_piece(us, ROOK);
dp.from[1] = rfrom;
dp.to[1] = rto;
dp.dirty_num = 2;
dp->piece[0] = make_piece(us, KING);
dp->from[0] = from;
dp->to[0] = to;
dp->piece[1] = make_piece(us, ROOK);
dp->from[1] = rfrom;
dp->to[1] = rto;
dp->dirty_num = 2;
}
// Remove both pieces first since squares could overlap in Chess960
@@ -1011,7 +1008,7 @@ void Position::do_null_move(StateInfo& newSt, const TranspositionTable& tt) {
assert(!checkers());
assert(&newSt != st);
std::memcpy(&newSt, st, offsetof(StateInfo, accumulatorBig));
std::memcpy(&newSt, st, sizeof(StateInfo));
newSt.previous = st;
st->next = &newSt;
@@ -1026,11 +1023,6 @@ void Position::do_null_move(StateInfo& newSt, const TranspositionTable& tt) {
st->key ^= Zobrist::side;
prefetch(tt.first_entry(key()));
st->dirtyPiece.dirty_num = 0;
st->dirtyPiece.piece[0] = NO_PIECE; // Avoid checks in UpdateAccumulator()
st->accumulatorBig.computed[WHITE] = st->accumulatorBig.computed[BLACK] =
st->accumulatorSmall.computed[WHITE] = st->accumulatorSmall.computed[BLACK] = false;
st->pliesFromNull = 0;
sideToMove = ~sideToMove;
+11 -13
View File
@@ -26,8 +26,6 @@
#include <string>
#include "bitboard.h"
#include "nnue/nnue_accumulator.h"
#include "nnue/nnue_architecture.h"
#include "types.h"
namespace Stockfish {
@@ -61,11 +59,6 @@ struct StateInfo {
Bitboard checkSquares[PIECE_TYPE_NB];
Piece capturedPiece;
int repetition;
// Used by NNUE
DirtyPiece dirtyPiece;
Eval::NNUE::Accumulator<Eval::NNUE::TransformedFeatureDimensionsBig> accumulatorBig;
Eval::NNUE::Accumulator<Eval::NNUE::TransformedFeatureDimensionsSmall> accumulatorSmall;
};
@@ -140,11 +133,11 @@ class Position {
Piece captured_piece() const;
// Doing and undoing moves
void do_move(Move m, StateInfo& newSt, const TranspositionTable* tt);
void do_move(Move m, StateInfo& newSt, bool givesCheck, const TranspositionTable* tt);
void undo_move(Move m);
void do_null_move(StateInfo& newSt, const TranspositionTable& tt);
void undo_null_move();
void do_move(Move m, StateInfo& newSt, const TranspositionTable* tt);
DirtyPiece do_move(Move m, StateInfo& newSt, bool givesCheck, const TranspositionTable* tt);
void undo_move(Move m);
void do_null_move(StateInfo& newSt, const TranspositionTable& tt);
void undo_null_move();
// Static Exchange Evaluation
bool see_ge(Move m, int threshold = 0) const;
@@ -187,7 +180,12 @@ class Position {
// Other helpers
void move_piece(Square from, Square to);
template<bool Do>
void do_castling(Color us, Square from, Square& to, Square& rfrom, Square& rto);
void do_castling(Color us,
Square from,
Square& to,
Square& rfrom,
Square& rto,
DirtyPiece* const dp = nullptr);
template<bool AfterMove>
Key adjust_key50(Key k) const;
+179 -138
View File
@@ -28,6 +28,7 @@
#include <cstdlib>
#include <initializer_list>
#include <iostream>
#include <limits>
#include <list>
#include <ratio>
#include <string>
@@ -41,7 +42,6 @@
#include "movepick.h"
#include "nnue/network.h"
#include "nnue/nnue_accumulator.h"
#include "nnue/nnue_common.h"
#include "position.h"
#include "syzygy/tbprobe.h"
#include "thread.h"
@@ -72,7 +72,7 @@ namespace {
// Futility margin
Value futility_margin(Depth d, bool noTtCutNode, bool improving, bool oppWorsening) {
Value futilityMult = 112 - 26 * noTtCutNode;
Value futilityMult = 110 - 25 * noTtCutNode;
Value improvingDeduction = improving * futilityMult * 2;
Value worseningDeduction = oppWorsening * futilityMult / 3;
@@ -88,13 +88,40 @@ int correction_value(const Worker& w, const Position& pos, const Stack* const ss
const auto m = (ss - 1)->currentMove;
const auto pcv = w.pawnCorrectionHistory[pawn_structure_index<Correction>(pos)][us];
const auto micv = w.minorPieceCorrectionHistory[minor_piece_index(pos)][us];
const auto wnpcv = w.nonPawnCorrectionHistory[WHITE][non_pawn_index<WHITE>(pos)][us];
const auto bnpcv = w.nonPawnCorrectionHistory[BLACK][non_pawn_index<BLACK>(pos)][us];
const auto wnpcv = w.nonPawnCorrectionHistory[non_pawn_index<WHITE>(pos)][WHITE][us];
const auto bnpcv = w.nonPawnCorrectionHistory[non_pawn_index<BLACK>(pos)][BLACK][us];
const auto cntcv =
m.is_ok() ? (*(ss - 2)->continuationCorrectionHistory)[pos.piece_on(m.to_sq())][m.to_sq()]
: 0;
return 6995 * pcv + 6593 * micv + 7753 * (wnpcv + bnpcv) + 6049 * cntcv;
return 7685 * pcv + 7495 * micv + 9144 * (wnpcv + bnpcv) + 6469 * cntcv;
}
int risk_tolerance(const Position& pos, Value v) {
// Returns (some constant of) second derivative of sigmoid.
static constexpr auto sigmoid_d2 = [](int x, int y) {
return 644800 * x / ((x * x + 3 * y * y) * y);
};
int m = (67 * pos.count<PAWN>() + 182 * pos.count<KNIGHT>() + 182 * pos.count<BISHOP>()
+ 337 * pos.count<ROOK>() + 553 * pos.count<QUEEN>())
/ 64;
// a and b are the crude approximation of the wdl model.
// The win rate is: 1/(1+exp((a-v)/b))
// The loss rate is 1/(1+exp((v+a)/b))
int a = 356;
int b = ((65 * m - 3172) * m + 240578) / 2048;
// guard against overflow
assert(abs(v) + a <= std::numeric_limits<int>::max() / 644800);
// The risk utility is therefore d/dv^2 (1/(1+exp(-(v-a)/b)) -1/(1+exp(-(-v-a)/b)))
// -115200x/(x^2+3) = -345600(ab) / (a^2+3b^2) (multiplied by some constant) (second degree pade approximant)
int winning_risk = sigmoid_d2(v - a, b);
int losing_risk = sigmoid_d2(v + a, b);
return -(winning_risk + losing_risk) * 32;
}
// Add correctionHistory value to raw staticEval and guarantee evaluation
@@ -110,19 +137,19 @@ void update_correction_history(const Position& pos,
const Move m = (ss - 1)->currentMove;
const Color us = pos.side_to_move();
static constexpr int nonPawnWeight = 165;
static constexpr int nonPawnWeight = 162;
workerThread.pawnCorrectionHistory[pawn_structure_index<Correction>(pos)][us]
<< bonus * 109 / 128;
workerThread.minorPieceCorrectionHistory[minor_piece_index(pos)][us] << bonus * 141 / 128;
workerThread.nonPawnCorrectionHistory[WHITE][non_pawn_index<WHITE>(pos)][us]
<< bonus * 111 / 128;
workerThread.minorPieceCorrectionHistory[minor_piece_index(pos)][us] << bonus * 146 / 128;
workerThread.nonPawnCorrectionHistory[non_pawn_index<WHITE>(pos)][WHITE][us]
<< bonus * nonPawnWeight / 128;
workerThread.nonPawnCorrectionHistory[BLACK][non_pawn_index<BLACK>(pos)][us]
workerThread.nonPawnCorrectionHistory[non_pawn_index<BLACK>(pos)][BLACK][us]
<< bonus * nonPawnWeight / 128;
if (m.is_ok())
(*(ss - 2)->continuationCorrectionHistory)[pos.piece_on(m.to_sq())][m.to_sq()]
<< bonus * 138 / 128;
<< bonus * 143 / 128;
}
// Add a small random component to draw evaluations to avoid 3-fold blindness
@@ -170,6 +197,8 @@ void Search::Worker::ensure_network_replicated() {
void Search::Worker::start_searching() {
accumulatorStack.reset(rootPos, networks[numaAccessToken], refreshTable);
// Non-main threads go directly to iterative_deepening()
if (!is_mainthread())
{
@@ -311,14 +340,10 @@ void Search::Worker::iterative_deepening() {
&this->continuationHistory[0][0][NO_PIECE][0]; // Use as a sentinel
(ss - i)->continuationCorrectionHistory = &this->continuationCorrectionHistory[NO_PIECE][0];
(ss - i)->staticEval = VALUE_NONE;
(ss - i)->reduction = 0;
}
for (int i = 0; i <= MAX_PLY + 2; ++i)
{
(ss + i)->ply = i;
(ss + i)->reduction = 0;
}
(ss + i)->ply = i;
ss->pv = pv;
@@ -342,7 +367,7 @@ void Search::Worker::iterative_deepening() {
int searchAgainCounter = 0;
lowPlyHistory.fill(95);
lowPlyHistory.fill(92);
// Iterative deepening loop until requested to stop or the target depth is reached
while (++rootDepth < MAX_PLY && !threads.stop
@@ -378,13 +403,13 @@ void Search::Worker::iterative_deepening() {
selDepth = 0;
// Reset aspiration window starting size
delta = 5 + std::abs(rootMoves[pvIdx].meanSquaredScore) / 13000;
delta = 5 + std::abs(rootMoves[pvIdx].meanSquaredScore) / 11834;
Value avg = rootMoves[pvIdx].averageScore;
alpha = std::max(avg - delta, -VALUE_INFINITE);
beta = std::min(avg + delta, VALUE_INFINITE);
// Adjust optimism based on root move's averageScore
optimism[us] = 138 * avg / (std::abs(avg) + 81);
optimism[us] = 138 * avg / (std::abs(avg) + 84);
optimism[~us] = -optimism[us];
// Start with a small aspiration window and, in the case of a fail
@@ -515,7 +540,8 @@ void Search::Worker::iterative_deepening() {
// Do we have time for the next iteration? Can we stop searching now?
if (limits.use_time_management() && !threads.stop && !mainThread->stopOnPonderhit)
{
int nodesEffort = rootMoves[0].effort * 100000 / std::max(size_t(1), size_t(nodes));
uint64_t nodesEffort =
rootMoves[0].effort * 100000 / std::max(size_t(1), size_t(nodes));
double fallingEval =
(11.396 + 2.035 * (mainThread->bestPreviousAverageScore - bestValue)
@@ -542,8 +568,8 @@ void Search::Worker::iterative_deepening() {
&& !mainThread->ponder)
threads.stop = true;
// Stop the search if we have exceeded the totalTime
if (elapsedTime > totalTime)
// Stop the search if we have exceeded the totalTime or maximum
if (elapsedTime > std::min(totalTime, double(mainThread->tm.maximum())))
{
// If we are allowed to ponder do not stop the search now but
// keep pondering until the GUI sends "ponderhit" or "stop".
@@ -572,29 +598,48 @@ void Search::Worker::iterative_deepening() {
skill.best ? skill.best : skill.pick_best(rootMoves, multiPV)));
}
void Search::Worker::do_move(Position& pos, const Move move, StateInfo& st) {
do_move(pos, move, st, pos.gives_check(move));
}
void Search::Worker::do_move(Position& pos, const Move move, StateInfo& st, const bool givesCheck) {
DirtyPiece dp = pos.do_move(move, st, givesCheck, &tt);
accumulatorStack.push(dp);
}
void Search::Worker::do_null_move(Position& pos, StateInfo& st) { pos.do_null_move(st, tt); }
void Search::Worker::undo_move(Position& pos, const Move move) {
pos.undo_move(move);
accumulatorStack.pop();
}
void Search::Worker::undo_null_move(Position& pos) { pos.undo_null_move(); }
// Reset histories, usually before a new game
void Search::Worker::clear() {
mainHistory.fill(65);
lowPlyHistory.fill(107);
captureHistory.fill(-655);
pawnHistory.fill(-1215);
pawnCorrectionHistory.fill(4);
mainHistory.fill(66);
lowPlyHistory.fill(105);
captureHistory.fill(-646);
pawnHistory.fill(-1262);
pawnCorrectionHistory.fill(6);
minorPieceCorrectionHistory.fill(0);
nonPawnCorrectionHistory[WHITE].fill(0);
nonPawnCorrectionHistory[BLACK].fill(0);
nonPawnCorrectionHistory.fill(0);
for (auto& to : continuationCorrectionHistory)
for (auto& h : to)
h.fill(0);
h.fill(5);
for (bool inCheck : {false, true})
for (StatsType c : {NoCaptures, Captures})
for (auto& to : continuationHistory[inCheck][c])
for (auto& h : to)
h.fill(-493);
h.fill(-468);
for (size_t i = 1; i < reductions.size(); ++i)
reductions[i] = int(2937 / 128.0 * std::log(i));
reductions[i] = int(2954 / 128.0 * std::log(i));
refreshTable.clear(networks[numaAccessToken]);
}
@@ -634,7 +679,6 @@ Value Search::Worker::search(
Move pv[MAX_PLY + 1];
StateInfo st;
ASSERT_ALIGNED(&st, Eval::NNUE::CacheLineSize);
Key posKey;
Move move, excludedMove, bestMove;
@@ -721,13 +765,11 @@ Value Search::Worker::search(
// Bonus for a quiet ttMove that fails high
if (!ttCapture)
update_quiet_histories(pos, ss, *this, ttData.move,
std::min(117600 * depth - 71344, 1244992) / 1024);
std::min(120 * depth - 75, 1241));
// Extra penalty for early quiet moves of the previous ply
if (prevSq != SQ_NONE && (ss - 1)->moveCount <= 3 && !priorCapture)
update_continuation_histories(ss - 1, pos.piece_on(prevSq), prevSq,
-std::min(779788 * (depth + 1) - 271806, 2958308)
/ 1024);
update_continuation_histories(ss - 1, pos.piece_on(prevSq), prevSq, -2200);
}
// Partial workaround for the graph history interaction problem
@@ -833,11 +875,11 @@ Value Search::Worker::search(
// Use static evaluation difference to improve quiet move ordering
if (((ss - 1)->currentMove).is_ok() && !(ss - 1)->inCheck && !priorCapture)
{
int bonus = std::clamp(-10 * int((ss - 1)->staticEval + ss->staticEval), -1906, 1450) + 638;
thisThread->mainHistory[~us][((ss - 1)->currentMove).from_to()] << bonus * 1136 / 1024;
int bonus = std::clamp(-10 * int((ss - 1)->staticEval + ss->staticEval), -1950, 1416) + 655;
thisThread->mainHistory[~us][((ss - 1)->currentMove).from_to()] << bonus * 1124 / 1024;
if (type_of(pos.piece_on(prevSq)) != PAWN && ((ss - 1)->currentMove).type_of() != PROMOTION)
thisThread->pawnHistory[pawn_structure_index(pos)][pos.piece_on(prevSq)][prevSq]
<< bonus * 1195 / 1024;
<< bonus * 1196 / 1024;
}
// Set up the improving flag, which is true if current static evaluation is
@@ -850,43 +892,43 @@ Value Search::Worker::search(
if (priorReduction >= 3 && !opponentWorsening)
depth++;
if (priorReduction >= 1 && depth >= 2 && ss->staticEval + (ss - 1)->staticEval > 200)
if (priorReduction >= 1 && depth >= 2 && ss->staticEval + (ss - 1)->staticEval > 188)
depth--;
// Step 7. Razoring
// If eval is really low, skip search entirely and return the qsearch value.
// For PvNodes, we must have a guard against mates being returned.
if (!PvNode && eval < alpha - 446 - 303 * depth * depth)
if (!PvNode && eval < alpha - 461 - 315 * depth * depth)
return qsearch<NonPV>(pos, ss, alpha, beta);
// Step 8. Futility pruning: child node
// The depth condition is important for mate finding.
if (!ss->ttPv && depth < 14
&& eval - futility_margin(depth, cutNode && !ss->ttHit, improving, opponentWorsening)
- (ss - 1)->statScore / 326 + 37 - std::abs(correctionValue) / 132821
- (ss - 1)->statScore / 301 + 37 - std::abs(correctionValue) / 139878
>= beta
&& eval >= beta && (!ttData.move || ttCapture) && !is_loss(beta) && !is_win(eval))
return beta + (eval - beta) / 3;
// Step 9. Null move search with verification search
if (cutNode && (ss - 1)->currentMove != Move::null() && eval >= beta
&& ss->staticEval >= beta - 21 * depth + 455 - 60 * improving && !excludedMove
&& pos.non_pawn_material(us) && ss->ply >= thisThread->nmpMinPly && !is_loss(beta))
&& ss->staticEval >= beta - 19 * depth + 418 && !excludedMove && pos.non_pawn_material(us)
&& ss->ply >= thisThread->nmpMinPly && !is_loss(beta))
{
assert(eval - beta >= 0);
// Null move dynamic reduction based on depth and eval
Depth R = std::min(int(eval - beta) / 237, 6) + depth / 3 + 5;
Depth R = std::min(int(eval - beta) / 232, 6) + depth / 3 + 5;
ss->currentMove = Move::null();
ss->continuationHistory = &thisThread->continuationHistory[0][0][NO_PIECE][0];
ss->continuationCorrectionHistory = &thisThread->continuationCorrectionHistory[NO_PIECE][0];
pos.do_null_move(st, tt);
do_null_move(pos, st);
Value nullValue = -search<NonPV>(pos, ss + 1, -beta, -beta + 1, depth - R, false);
pos.undo_null_move();
undo_null_move(pos);
// Do not return unproven mate or TB scores
if (nullValue >= beta && !is_win(nullValue))
@@ -909,7 +951,7 @@ Value Search::Worker::search(
}
}
improving |= ss->staticEval >= beta + 97;
improving |= ss->staticEval >= beta + 94;
// Step 10. Internal iterative reductions
// For PV nodes without a ttMove as well as for deep enough cutNodes, we decrease depth.
@@ -920,7 +962,7 @@ Value Search::Worker::search(
// Step 11. ProbCut
// If we have a good enough capture (or queen promotion) and a reduced search
// returns a value much above beta, we can (almost) safely prune the previous move.
probCutBeta = beta + 187 - 55 * improving;
probCutBeta = beta + 185 - 58 * improving;
if (depth >= 3
&& !is_decisive(beta)
// If value from transposition table is lower than probCutBeta, don't attempt
@@ -938,17 +980,14 @@ Value Search::Worker::search(
{
assert(move.is_ok());
if (move == excludedMove)
continue;
if (!pos.legal(move))
if (move == excludedMove || !pos.legal(move))
continue;
assert(pos.capture_stage(move));
movedPiece = pos.moved_piece(move);
pos.do_move(move, st, &tt);
do_move(pos, move, st);
thisThread->nodes.fetch_add(1, std::memory_order_relaxed);
ss->currentMove = move;
@@ -966,7 +1005,7 @@ Value Search::Worker::search(
value = -search<NonPV>(pos, ss + 1, -probCutBeta, -probCutBeta + 1, probCutDepth,
!cutNode);
pos.undo_move(move);
undo_move(pos, move);
if (value >= probCutBeta)
{
@@ -984,7 +1023,7 @@ Value Search::Worker::search(
moves_loop: // When in check, search starts here
// Step 12. A small Probcut idea
probCutBeta = beta + 413;
probCutBeta = beta + 415;
if ((ttData.bound & BOUND_LOWER) && ttData.depth >= depth - 4 && ttData.value >= probCutBeta
&& !is_decisive(beta) && is_valid(ttData.value) && !is_decisive(ttData.value))
return probCutBeta;
@@ -1050,7 +1089,7 @@ moves_loop: // When in check, search starts here
// Smaller or even negative value is better for short time controls
// Bigger value is better for long time controls
if (ss->ttPv)
r += 1031;
r += 979;
// Step 14. Pruning at shallow depth.
// Depth conditions are important for mate finding.
@@ -1072,15 +1111,15 @@ moves_loop: // When in check, search starts here
// Futility pruning for captures
if (!givesCheck && lmrDepth < 7 && !ss->inCheck)
{
Value futilityValue = ss->staticEval + 242 + 238 * lmrDepth
+ PieceValue[capturedPiece] + 95 * captHist / 700;
Value futilityValue = ss->staticEval + 242 + 230 * lmrDepth
+ PieceValue[capturedPiece] + 133 * captHist / 1024;
if (futilityValue <= alpha)
continue;
}
// SEE based pruning for captures and checks
int seeHist = std::clamp(captHist / 36, -153 * depth, 134 * depth);
if (!pos.see_ge(move, -157 * depth - seeHist))
int seeHist = std::clamp(captHist / 32, -138 * depth, 135 * depth);
if (!pos.see_ge(move, -154 * depth - seeHist))
continue;
}
else
@@ -1091,17 +1130,15 @@ moves_loop: // When in check, search starts here
+ thisThread->pawnHistory[pawn_structure_index(pos)][movedPiece][move.to_sq()];
// Continuation history based pruning
if (history < -4107 * depth)
if (history < -4348 * depth)
continue;
history += 68 * thisThread->mainHistory[us][move.from_to()] / 32;
lmrDepth += history / 3576;
lmrDepth += history / 3593;
Value futilityValue = ss->staticEval + (bestMove ? 49 : 143) + 116 * lmrDepth;
if (bestValue < ss->staticEval - 150 && lmrDepth < 7)
futilityValue += 108;
Value futilityValue = ss->staticEval + (bestMove ? 48 : 146) + 116 * lmrDepth
+ 103 * (bestValue < ss->staticEval - 128);
// Futility pruning: parent node
// (*Scaler): Generally, more frequent futility pruning
@@ -1117,7 +1154,7 @@ moves_loop: // When in check, search starts here
lmrDepth = std::max(lmrDepth, 0);
// Prune moves with negative SEE
if (!pos.see_ge(move, -26 * lmrDepth * lmrDepth))
if (!pos.see_ge(move, -27 * lmrDepth * lmrDepth))
continue;
}
}
@@ -1137,11 +1174,11 @@ moves_loop: // When in check, search starts here
// and lower extension margins scale well.
if (!rootNode && move == ttData.move && !excludedMove
&& depth >= 5 - (thisThread->completedDepth > 32) + ss->ttPv
&& depth >= 6 - (thisThread->completedDepth > 29) + ss->ttPv
&& is_valid(ttData.value) && !is_decisive(ttData.value)
&& (ttData.bound & BOUND_LOWER) && ttData.depth >= depth - 3)
{
Value singularBeta = ttData.value - (55 + 81 * (ss->ttPv && !PvNode)) * depth / 58;
Value singularBeta = ttData.value - (59 + 77 * (ss->ttPv && !PvNode)) * depth / 54;
Depth singularDepth = newDepth / 2;
ss->excludedMove = move;
@@ -1151,11 +1188,11 @@ moves_loop: // When in check, search starts here
if (value < singularBeta)
{
int corrValAdj1 = std::abs(correctionValue) / 265083;
int corrValAdj2 = std::abs(correctionValue) / 253680;
int doubleMargin = 267 * PvNode - 181 * !ttCapture - corrValAdj1;
int corrValAdj1 = std::abs(correctionValue) / 248873;
int corrValAdj2 = std::abs(correctionValue) / 255331;
int doubleMargin = 262 * PvNode - 188 * !ttCapture - corrValAdj1;
int tripleMargin =
96 + 282 * PvNode - 250 * !ttCapture + 103 * ss->ttPv - corrValAdj2;
88 + 265 * PvNode - 256 * !ttCapture + 93 * ss->ttPv - corrValAdj2;
extension = 1 + (value < singularBeta - doubleMargin)
+ (value < singularBeta - tripleMargin);
@@ -1191,7 +1228,7 @@ moves_loop: // When in check, search starts here
}
// Step 16. Make the move
pos.do_move(move, st, givesCheck, &tt);
do_move(pos, move, st, givesCheck);
thisThread->nodes.fetch_add(1, std::memory_order_relaxed);
// Add extension to new depth
@@ -1208,43 +1245,49 @@ moves_loop: // When in check, search starts here
// Decrease reduction for PvNodes (*Scaler)
if (ss->ttPv)
r -= 2230 + PvNode * 1013 + (ttData.value > alpha) * 925
+ (ttData.depth >= depth) * (971 + cutNode * 1159);
r -= 2381 + PvNode * 1008 + (ttData.value > alpha) * 880
+ (ttData.depth >= depth) * (1022 + cutNode * 1140);
// These reduction adjustments have no proven non-linear scaling
r += 316 - moveCount * 32;
r += 306 - moveCount * 34;
r -= std::abs(correctionValue) / 31568;
r -= std::abs(correctionValue) / 29696;
if (PvNode && std::abs(bestValue) <= 2000)
r -= risk_tolerance(pos, bestValue);
// Increase reduction for cut nodes
if (cutNode)
r += 2608 + 1024 * !ttData.move;
r += 2784 + 1038 * !ttData.move;
// Increase reduction if ttMove is a capture but the current move is not a capture
if (ttCapture && !capture)
r += 1123 + (depth < 8) * 982;
r += 1171 + (depth < 8) * 985;
// Increase reduction if next ply has a lot of fail high
if ((ss + 1)->cutoffCnt > 3)
r += 981 + allNode * 833;
if ((ss + 1)->cutoffCnt > 2)
r += 1042 + allNode * 864;
// For first picked move (ttMove) reduce reduction
else if (move == ttData.move)
r -= 1982;
r -= 1937;
if (capture)
ss->statScore =
688 * int(PieceValue[pos.captured_piece()]) / 100
846 * int(PieceValue[pos.captured_piece()]) / 128
+ thisThread->captureHistory[movedPiece][move.to_sq()][type_of(pos.captured_piece())]
- 4653;
- 4822;
else if (ss->inCheck)
ss->statScore = thisThread->mainHistory[us][move.from_to()]
+ (*contHist[0])[movedPiece][move.to_sq()] - 2771;
else
ss->statScore = 2 * thisThread->mainHistory[us][move.from_to()]
+ (*contHist[0])[movedPiece][move.to_sq()]
+ (*contHist[1])[movedPiece][move.to_sq()] - 3591;
+ (*contHist[1])[movedPiece][move.to_sq()] - 3271;
// Decrease/increase reduction for moves with a good/bad history
r -= ss->statScore * 1407 / 16384;
r -= ss->statScore * 1582 / 16384;
// Step 17. Late moves reduction / extension (LMR)
if (depth >= 2 && moveCount > 1)
@@ -1270,7 +1313,7 @@ moves_loop: // When in check, search starts here
{
// Adjust full-depth search based on LMR results - if the result was
// good enough search deeper, if it was bad enough search shallower.
const bool doDeeperSearch = value > (bestValue + 41 + 2 * newDepth);
const bool doDeeperSearch = value > (bestValue + 43 + 2 * newDepth);
const bool doShallowerSearch = value < bestValue + 9;
newDepth += doDeeperSearch - doShallowerSearch;
@@ -1279,9 +1322,10 @@ moves_loop: // When in check, search starts here
value = -search<NonPV>(pos, ss + 1, -(alpha + 1), -alpha, newDepth, !cutNode);
// Post LMR continuation history updates
int bonus = (value >= beta) * 2010;
update_continuation_histories(ss, movedPiece, move.to_sq(), bonus);
update_continuation_histories(ss, movedPiece, move.to_sq(), 1600);
}
else if (value > alpha && value < bestValue + 9)
newDepth--;
}
// Step 18. Full-depth search when LMR is skipped
@@ -1289,11 +1333,11 @@ moves_loop: // When in check, search starts here
{
// Increase reduction if ttMove is not present
if (!ttData.move)
r += 1111;
r += 1156;
// Note that if expected reduction is high, we reduce search depth here
value = -search<NonPV>(pos, ss + 1, -(alpha + 1), -alpha,
newDepth - (r > 3554) - (r > 5373 && newDepth > 2), !cutNode);
newDepth - (r > 3495) - (r > 5510 && newDepth > 2), !cutNode);
}
// For PV nodes only, do a full PV search on the first move or after a fail high,
@@ -1311,7 +1355,7 @@ moves_loop: // When in check, search starts here
}
// Step 19. Undo move
pos.undo_move(move);
undo_move(pos, move);
assert(value > -VALUE_INFINITE && value < VALUE_INFINITE);
@@ -1400,7 +1444,7 @@ moves_loop: // When in check, search starts here
else
{
// Reduce other moves if we have found at least one score improvement
if (depth > 2 && depth < 15 && !is_decisive(value))
if (depth > 2 && depth < 16 && !is_decisive(value))
depth -= 2;
assert(depth > 0);
@@ -1427,9 +1471,8 @@ moves_loop: // When in check, search starts here
assert(moveCount || !ss->inCheck || excludedMove || !MoveList<LEGAL>(pos).size());
// Adjust best value for fail high cases at non-pv nodes
if (!PvNode && bestValue >= beta && !is_decisive(bestValue) && !is_decisive(beta)
&& !is_decisive(alpha))
// Adjust best value for fail high cases
if (bestValue >= beta && !is_decisive(bestValue) && !is_decisive(beta) && !is_decisive(alpha))
bestValue = (bestValue * depth + beta) / (depth + 1);
if (!moveCount)
@@ -1441,37 +1484,37 @@ moves_loop: // When in check, search starts here
update_all_stats(pos, ss, *this, bestMove, prevSq, quietsSearched, capturesSearched, depth,
bestMove == ttData.move, moveCount);
// Bonus for prior countermove that caused the fail low
// Bonus for prior quiet countermove that caused the fail low
else if (!priorCapture && prevSq != SQ_NONE)
{
int bonusScale = (118 * (depth > 5) + 36 * !allNode + 161 * ((ss - 1)->moveCount > 8)
+ 133 * (!ss->inCheck && bestValue <= ss->staticEval - 107)
+ 120 * (!(ss - 1)->inCheck && bestValue <= -(ss - 1)->staticEval - 84)
+ 81 * ((ss - 1)->isTTMove) + 100 * (ss->cutoffCnt <= 3)
+ std::min(-(ss - 1)->statScore / 108, 320));
int bonusScale =
(std::clamp(80 * depth - 320, 0, 200) + 34 * !allNode + 164 * ((ss - 1)->moveCount > 8)
+ 141 * (!ss->inCheck && bestValue <= ss->staticEval - 100)
+ 121 * (!(ss - 1)->inCheck && bestValue <= -(ss - 1)->staticEval - 75)
+ 86 * ((ss - 1)->isTTMove) + 86 * (ss->cutoffCnt <= 3)
+ std::min(-(ss - 1)->statScore / 112, 303));
bonusScale = std::max(bonusScale, 0);
const int scaledBonus = std::min(160 * depth - 106, 1523) * bonusScale;
const int scaledBonus = std::min(160 * depth - 99, 1492) * bonusScale;
update_continuation_histories(ss - 1, pos.piece_on(prevSq), prevSq,
scaledBonus * 416 / 32768);
scaledBonus * 388 / 32768);
thisThread->mainHistory[~us][((ss - 1)->currentMove).from_to()]
<< scaledBonus * 219 / 32768;
<< scaledBonus * 212 / 32768;
if (type_of(pos.piece_on(prevSq)) != PAWN && ((ss - 1)->currentMove).type_of() != PROMOTION)
thisThread->pawnHistory[pawn_structure_index(pos)][pos.piece_on(prevSq)][prevSq]
<< scaledBonus * 1103 / 32768;
<< scaledBonus * 1055 / 32768;
}
// Bonus for prior capture countermove that caused the fail low
else if (priorCapture && prevSq != SQ_NONE)
{
// bonus for prior countermoves that caused the fail low
Piece capturedPiece = pos.captured_piece();
assert(capturedPiece != NO_PIECE);
thisThread->captureHistory[pos.piece_on(prevSq)][prevSq][type_of(capturedPiece)]
<< std::min(330 * depth - 198, 3320);
thisThread->captureHistory[pos.piece_on(prevSq)][prevSq][type_of(capturedPiece)] << 1100;
}
if (PvNode)
@@ -1533,7 +1576,6 @@ Value Search::Worker::qsearch(Position& pos, Stack* ss, Value alpha, Value beta)
Move pv[MAX_PLY + 1];
StateInfo st;
ASSERT_ALIGNED(&st, Eval::NNUE::CacheLineSize);
Key posKey;
Move move, bestMove;
@@ -1579,12 +1621,13 @@ Value Search::Worker::qsearch(Position& pos, Stack* ss, Value alpha, Value beta)
return ttData.value;
// Step 4. Static evaluation of the position
Value unadjustedStaticEval = VALUE_NONE;
const auto correctionValue = correction_value(*thisThread, pos, ss);
Value unadjustedStaticEval = VALUE_NONE;
if (ss->inCheck)
bestValue = futilityBase = -VALUE_INFINITE;
else
{
const auto correctionValue = correction_value(*thisThread, pos, ss);
if (ss->ttHit)
{
// Never assume anything about values stored in TT
@@ -1625,7 +1668,7 @@ Value Search::Worker::qsearch(Position& pos, Stack* ss, Value alpha, Value beta)
if (bestValue > alpha)
alpha = bestValue;
futilityBase = ss->staticEval + 325;
futilityBase = ss->staticEval + 359;
}
const PieceToHistory* contHist[] = {(ss - 1)->continuationHistory,
@@ -1685,10 +1728,9 @@ Value Search::Worker::qsearch(Position& pos, Stack* ss, Value alpha, Value beta)
// Continuation history based pruning
if (!capture
&& (*contHist[0])[pos.moved_piece(move)][move.to_sq()]
+ (*contHist[1])[pos.moved_piece(move)][move.to_sq()]
+ thisThread->pawnHistory[pawn_structure_index(pos)][pos.moved_piece(move)]
[move.to_sq()]
<= 5389)
<= 6290)
continue;
// Do not search moves with bad enough SEE values
@@ -1699,7 +1741,7 @@ Value Search::Worker::qsearch(Position& pos, Stack* ss, Value alpha, Value beta)
// Step 7. Make and search the move
Piece movedPiece = pos.moved_piece(move);
pos.do_move(move, st, givesCheck, &tt);
do_move(pos, move, st, givesCheck);
thisThread->nodes.fetch_add(1, std::memory_order_relaxed);
// Update the current move
@@ -1710,7 +1752,7 @@ Value Search::Worker::qsearch(Position& pos, Stack* ss, Value alpha, Value beta)
&thisThread->continuationCorrectionHistory[movedPiece][move.to_sq()];
value = -qsearch<nodeType>(pos, ss + 1, -beta, -alpha);
pos.undo_move(move);
undo_move(pos, move);
assert(value > -VALUE_INFINITE && value < VALUE_INFINITE);
@@ -1743,8 +1785,8 @@ Value Search::Worker::qsearch(Position& pos, Stack* ss, Value alpha, Value beta)
return mated_in(ss->ply); // Plies to mate from the root
}
if (!is_decisive(bestValue) && bestValue >= beta)
bestValue = (3 * bestValue + beta) / 4;
if (!is_decisive(bestValue) && bestValue > beta)
bestValue = (bestValue + beta) / 2;
// Save gathered info in transposition table. The static evaluation
// is saved as it was before adjustment by correction history.
@@ -1759,7 +1801,7 @@ Value Search::Worker::qsearch(Position& pos, Stack* ss, Value alpha, Value beta)
Depth Search::Worker::reduction(bool i, Depth d, int mn, int delta) const {
int reductionScale = reductions[d] * reductions[mn];
return reductionScale - delta * 735 / rootDelta + !i * reductionScale * 191 / 512 + 1132;
return reductionScale - delta * 764 / rootDelta + !i * reductionScale * 191 / 512 + 1087;
}
// elapsed() returns the time elapsed since the search started. If the
@@ -1777,7 +1819,7 @@ TimePoint Search::Worker::elapsed() const {
TimePoint Search::Worker::elapsed_time() const { return main_manager()->tm.elapsed_time(); }
Value Search::Worker::evaluate(const Position& pos) {
return Eval::evaluate(networks[numaAccessToken], pos, refreshTable,
return Eval::evaluate(networks[numaAccessToken], pos, accumulatorStack, refreshTable,
optimism[pos.side_to_move()]);
}
@@ -1855,35 +1897,35 @@ void update_all_stats(const Position& pos,
Piece moved_piece = pos.moved_piece(bestMove);
PieceType captured;
int bonus = std::min(162 * depth - 92, 1587) + 298 * isTTMove;
int malus = std::min(694 * depth - 230, 2503) - 32 * (moveCount - 1);
int bonus = std::min(141 * depth - 89, 1613) + 311 * isTTMove;
int malus = std::min(695 * depth - 215, 2808) - 31 * (moveCount - 1);
if (!pos.capture_stage(bestMove))
{
update_quiet_histories(pos, ss, workerThread, bestMove, bonus * 1202 / 1024);
update_quiet_histories(pos, ss, workerThread, bestMove, bonus * 1129 / 1024);
// Decrease stats for all non-best quiet moves
for (Move move : quietsSearched)
update_quiet_histories(pos, ss, workerThread, move, -malus * 1152 / 1024);
update_quiet_histories(pos, ss, workerThread, move, -malus * 1246 / 1024);
}
else
{
// Increase stats for the best move in case it was a capture move
captured = type_of(pos.piece_on(bestMove.to_sq()));
captureHistory[moved_piece][bestMove.to_sq()][captured] << bonus * 1236 / 1024;
captureHistory[moved_piece][bestMove.to_sq()][captured] << bonus * 1187 / 1024;
}
// Extra penalty for a quiet early move that was not a TT move in
// previous ply when it gets refuted.
if (prevSq != SQ_NONE && ((ss - 1)->moveCount == 1 + (ss - 1)->ttHit) && !pos.captured_piece())
update_continuation_histories(ss - 1, pos.piece_on(prevSq), prevSq, -malus * 976 / 1024);
update_continuation_histories(ss - 1, pos.piece_on(prevSq), prevSq, -malus * 987 / 1024);
// Decrease stats for all non-best capture moves
for (Move move : capturesSearched)
{
moved_piece = pos.moved_piece(move);
captured = type_of(pos.piece_on(move.to_sq()));
captureHistory[moved_piece][move.to_sq()][captured] << -malus * 1224 / 1024;
captureHistory[moved_piece][move.to_sq()][captured] << -malus * 1377 / 1024;
}
}
@@ -1892,7 +1934,7 @@ void update_all_stats(const Position& pos,
// at ply -1, -2, -3, -4, and -6 with current move.
void update_continuation_histories(Stack* ss, Piece pc, Square to, int bonus) {
static constexpr std::array<ConthistBonus, 6> conthist_bonuses = {
{{1, 1029}, {2, 656}, {3, 326}, {4, 536}, {5, 120}, {6, 537}}};
{{1, 1103}, {2, 659}, {3, 323}, {4, 533}, {5, 121}, {6, 474}}};
for (const auto [i, weight] : conthist_bonuses)
{
@@ -1913,12 +1955,12 @@ void update_quiet_histories(
workerThread.mainHistory[us][move.from_to()] << bonus; // Untuned to prevent duplicate effort
if (ss->ply < LOW_PLY_HISTORY_SIZE)
workerThread.lowPlyHistory[ss->ply][move.from_to()] << bonus * 844 / 1024;
workerThread.lowPlyHistory[ss->ply][move.from_to()] << bonus * 829 / 1024;
update_continuation_histories(ss, pos.moved_piece(move), move.to_sq(), bonus * 964 / 1024);
update_continuation_histories(ss, pos.moved_piece(move), move.to_sq(), bonus * 1004 / 1024);
int pIndex = pawn_structure_index(pos);
workerThread.pawnHistory[pIndex][pos.moved_piece(move)][move.to_sq()] << bonus * 615 / 1024;
workerThread.pawnHistory[pIndex][pos.moved_piece(move)][move.to_sq()] << bonus * 587 / 1024;
}
}
@@ -1940,8 +1982,8 @@ Move Skill::pick_best(const RootMoves& rootMoves, size_t multiPV) {
for (size_t i = 0; i < multiPV; ++i)
{
// This is our magic formula
int push = (weakness * int(topScore - rootMoves[i].score)
+ delta * (rng.rand<unsigned>() % int(weakness)))
int push = int(weakness * int(topScore - rootMoves[i].score)
+ delta * (rng.rand<unsigned>() % int(weakness)))
/ 128;
if (rootMoves[i].score + push >= maxScore)
@@ -2209,7 +2251,6 @@ void SearchManager::pv(Search::Worker& worker,
bool RootMove::extract_ponder_from_tt(const TranspositionTable& tt, Position& pos) {
StateInfo st;
ASSERT_ALIGNED(&st, Eval::NNUE::CacheLineSize);
assert(pv.size() == 1);
if (pv[0] == Move::none())
+8 -1
View File
@@ -301,7 +301,7 @@ class Worker {
CorrectionHistory<Pawn> pawnCorrectionHistory;
CorrectionHistory<Minor> minorPieceCorrectionHistory;
CorrectionHistory<NonPawn> nonPawnCorrectionHistory[COLOR_NB];
CorrectionHistory<NonPawn> nonPawnCorrectionHistory;
CorrectionHistory<Continuation> continuationCorrectionHistory;
#ifdef USE_MPI
@@ -329,6 +329,12 @@ class Worker {
private:
void iterative_deepening();
void do_move(Position& pos, const Move move, StateInfo& st);
void do_move(Position& pos, const Move move, StateInfo& st, const bool givesCheck);
void do_null_move(Position& pos, StateInfo& st);
void undo_move(Position& pos, const Move move);
void undo_null_move(Position& pos);
// This is the main search function, for both PV and non-PV nodes
template<NodeType nodeType>
Value search(Position& pos, Stack* ss, Value alpha, Value beta, Depth depth, bool cutNode);
@@ -381,6 +387,7 @@ class Worker {
const LazyNumaReplicated<Eval::NNUE::Networks>& networks;
// Used by NNUE
Eval::NNUE::AccumulatorStack accumulatorStack;
Eval::NNUE::AccumulatorCaches refreshTable;
friend class Stockfish::ThreadPool;
+16 -7
View File
@@ -38,6 +38,7 @@
#include <cassert>
#include <cstdint>
#include <type_traits>
#if defined(_MSC_VER)
// Disable some silly and noisy warnings from MSVC compiler
@@ -289,8 +290,8 @@ struct DirtyPiece {
};
#define ENABLE_INCR_OPERATORS_ON(T) \
inline T& operator++(T& d) { return d = T(int(d) + 1); } \
inline T& operator--(T& d) { return d = T(int(d) - 1); }
constexpr T& operator++(T& d) { return d = T(int(d) + 1); } \
constexpr T& operator--(T& d) { return d = T(int(d) - 1); }
ENABLE_INCR_OPERATORS_ON(PieceType)
ENABLE_INCR_OPERATORS_ON(Square)
@@ -303,10 +304,10 @@ constexpr Direction operator+(Direction d1, Direction d2) { return Direction(int
constexpr Direction operator*(int i, Direction d) { return Direction(i * int(d)); }
// Additional operators to add a Direction to a Square
constexpr Square operator+(Square s, Direction d) { return Square(int(s) + int(d)); }
constexpr Square operator-(Square s, Direction d) { return Square(int(s) - int(d)); }
inline Square& operator+=(Square& s, Direction d) { return s = s + d; }
inline Square& operator-=(Square& s, Direction d) { return s = s - d; }
constexpr Square operator+(Square s, Direction d) { return Square(int(s) + int(d)); }
constexpr Square operator-(Square s, Direction d) { return Square(int(s) - int(d)); }
constexpr Square& operator+=(Square& s, Direction d) { return s = s + d; }
constexpr Square& operator-=(Square& s, Direction d) { return s = s - d; }
// Toggle color
constexpr Color operator~(Color c) { return Color(c ^ BLACK); }
@@ -334,7 +335,7 @@ constexpr Piece make_piece(Color c, PieceType pt) { return Piece((c << 3) + pt);
constexpr PieceType type_of(Piece pc) { return PieceType(pc & 7); }
inline Color color_of(Piece pc) {
constexpr Color color_of(Piece pc) {
assert(pc != NO_PIECE);
return Color(pc >> 3);
}
@@ -429,6 +430,14 @@ class Move {
std::uint16_t data;
};
template<typename T, typename... Ts>
struct is_all_same {
static constexpr bool value = (std::is_same_v<T, Ts> && ...);
};
template<typename... Ts>
constexpr auto is_all_same_v = is_all_same<Ts...>::value;
} // namespace Stockfish
#endif // #ifndef TYPES_H_INCLUDED
+2 -2
View File
@@ -520,8 +520,8 @@ WinRateParams win_rate_params(const Position& pos) {
double m = std::clamp(material, 17, 78) / 58.0;
// Return a = p_a(material) and b = p_b(material), see github.com/official-stockfish/WDL_model
constexpr double as[] = {-37.45051876, 121.19101539, -132.78783573, 420.70576692};
constexpr double bs[] = {90.26261072, -137.26549898, 71.10130540, 51.35259597};
constexpr double as[] = {-13.50030198, 40.92780883, -36.82753545, 386.83004070};
constexpr double bs[] = {96.53354896, -165.79058388, 90.89679019, 49.29561889};
double a = (((as[0] * m + as[1]) * m + as[2]) * m) + as[3];
double b = (((bs[0] * m + bs[1]) * m + bs[2]) * m) + bs[3];
+77 -16
View File
@@ -1,6 +1,8 @@
#!/bin/bash
# verify perft numbers (positions from https://www.chessprogramming.org/Perft_Results)
TESTS_FAILED=0
error()
{
echo "perft testing failed on line $1"
@@ -10,23 +12,82 @@ trap 'error ${LINENO}' ERR
echo "perft testing started"
cat << EOF > perft.exp
set timeout 10
lassign \$argv pos depth result
spawn ./stockfish
send "position \$pos\\ngo perft \$depth\\n"
expect "Nodes searched? \$result" {} timeout {exit 1}
send "quit\\n"
expect eof
EXPECT_SCRIPT=$(mktemp)
cat << 'EOF' > $EXPECT_SCRIPT
#!/usr/bin/expect -f
set timeout 30
lassign [lrange $argv 0 4] pos depth result chess960 logfile
log_file -noappend $logfile
spawn ./stockfish
if {$chess960 == "true"} {
send "setoption name UCI_Chess960 value true\n"
}
send "position $pos\ngo perft $depth\n"
expect {
"Nodes searched: $result" {}
timeout {puts "TIMEOUT: Expected $result nodes"; exit 1}
eof {puts "EOF: Stockfish crashed"; exit 2}
}
send "quit\n"
expect eof
EOF
expect perft.exp startpos 5 4865609 > /dev/null
expect perft.exp "fen r3k2r/p1ppqpb1/bn2pnp1/3PN3/1p2P3/2N2Q1p/PPPBBPPP/R3K2R w KQkq -" 5 193690690 > /dev/null
expect perft.exp "fen 8/2p5/3p4/KP5r/1R3p1k/8/4P1P1/8 w - -" 6 11030083 > /dev/null
expect perft.exp "fen r3k2r/Pppp1ppp/1b3nbN/nP6/BBP1P3/q4N2/Pp1P2PP/R2Q1RK1 w kq - 0 1" 5 15833292 > /dev/null
expect perft.exp "fen rnbq1k1r/pp1Pbppp/2p5/8/2B5/8/PPP1NnPP/RNBQK2R w KQ - 1 8" 5 89941194 > /dev/null
expect perft.exp "fen r4rk1/1pp1qppp/p1np1n2/2b1p1B1/2B1P1b1/P1NP1N2/1PP1QPPP/R4RK1 w - - 0 10" 5 164075551 > /dev/null
chmod +x $EXPECT_SCRIPT
rm perft.exp
run_test() {
local pos="$1"
local depth="$2"
local expected="$3"
local chess960="$4"
local tmp_file=$(mktemp)
echo "perft testing OK"
echo -n "Testing depth $depth: ${pos:0:40}... "
if $EXPECT_SCRIPT "$pos" "$depth" "$expected" "$chess960" "$tmp_file" > /dev/null 2>&1; then
echo "OK"
rm -f "$tmp_file"
else
local exit_code=$?
echo "FAILED (exit code: $exit_code)"
echo "===== Output for failed test ====="
cat "$tmp_file"
echo "=================================="
rm -f "$tmp_file"
TESTS_FAILED=1
fi
}
# standard positions
run_test "startpos" 7 3195901860 "false"
run_test "fen r3k2r/p1ppqpb1/bn2pnp1/3PN3/1p2P3/2N2Q1p/PPPBBPPP/R3K2R w KQkq -" 5 193690690 "false"
run_test "fen 8/2p5/3p4/KP5r/1R3p1k/8/4P1P1/8 w - -" 7 178633661 "false"
run_test "fen r3k2r/Pppp1ppp/1b3nbN/nP6/BBP1P3/q4N2/Pp1P2PP/R2Q1RK1 w kq - 0 1" 6 706045033 "false"
run_test "fen rnbq1k1r/pp1Pbppp/2p5/8/2B5/8/PPP1NnPP/RNBQK2R w KQ - 1 8" 5 89941194 "false"
run_test "fen r4rk1/1pp1qppp/p1np1n2/2b1p1B1/2B1P1b1/P1NP1N2/1PP1QPPP/R4RK1 w - - 0 10" 5 164075551 "false"
run_test "fen r7/4p3/5p1q/3P4/4pQ2/4pP2/6pp/R3K1kr w Q - 1 3" 5 11609488 "false"
# chess960 positions
run_test "fen rnbqkbnr/pppppppp/8/8/8/8/PPPPPPPP/RNBQKBNR w AHah - 0 1" 6 119060324 "true"
run_test "fen 1rqbkrbn/1ppppp1p/1n6/p1N3p1/8/2P4P/PP1PPPP1/1RQBKRBN w FBfb - 0 9" 6 191762235 "true"
run_test "fen rbbqn1kr/pp2p1pp/6n1/2pp1p2/2P4P/P7/BP1PPPP1/R1BQNNKR w HAha - 0 9" 6 924181432 "true"
run_test "fen rqbbknr1/1ppp2pp/p5n1/4pp2/P7/1PP5/1Q1PPPPP/R1BBKNRN w GAga - 0 9" 6 308553169 "true"
run_test "fen 4rrb1/1kp3b1/1p1p4/pP1Pn2p/5p2/1PR2P2/2P1NB1P/2KR1B2 w D - 0 21" 6 872323796 "true"
run_test "fen 1rkr3b/1ppn3p/3pB1n1/6q1/R2P4/4N1P1/1P5P/2KRQ1B1 b Dbd - 0 14" 6 2678022813 "true"
run_test "fen qbbnrkr1/p1pppppp/1p4n1/8/2P5/6N1/PPNPPPPP/1BRKBRQ1 b FCge - 1 3" 6 521301336 "true"
run_test "fen rr6/2kpp3/1ppn2p1/p2b1q1p/P4P1P/1PNN2P1/2PP4/1K2R2R b E - 1 20" 2 1438 "true"
run_test "fen rr6/2kpp3/1ppn2p1/p2b1q1p/P4P1P/1PNN2P1/2PP4/1K2RR2 w E - 0 20" 3 37340 "true"
run_test "fen rr6/2kpp3/1ppnb1p1/p2Q1q1p/P4P1P/1PNN2P1/2PP4/1K2RR2 b E - 2 19" 4 2237725 "true"
run_test "fen rr6/2kpp3/1ppnb1p1/p4q1p/P4P1P/1PNN2P1/2PP2Q1/1K2RR2 w E - 1 19" 4 2098209 "true"
run_test "fen rr6/2kpp3/1ppnb1p1/p4q1p/P4P1P/1PNN2P1/2PP2Q1/1K2RR2 w E - 1 19" 5 79014522 "true"
run_test "fen rr6/2kpp3/1ppnb1p1/p4q1p/P4P1P/1PNN2P1/2PP2Q1/1K2RR2 w E - 1 19" 6 2998685421 "true"
rm -f $EXPECT_SCRIPT
echo "perft testing completed"
if [ $TESTS_FAILED -ne 0 ]; then
echo "Some tests failed"
exit 1
fi