crypton-2.2.0: cbits/tests/bearssl_diff.c
/*
* crypton's portable AES and GHASH against the BearSSL they are written on.
*
* This began as the migration check: for one commit, cbits/aes/generic.c and
* cbits/aes/gf.c were still the table-driven implementations, and this said
* the vendored code computed what they computed before they were replaced.
*
* Since the replacement the two sides are no longer independent, and what is
* left is still worth checking: everything in generic.c and gf.c is now
* crypton's own glue -- a schedule kept compressed and carried through
* memcpy, the interleave-and-ortho idiom around a single block, three of four
* lanes left idle, and a GHASH entry that reaches the same multiply by handing
* it a block of zeros. Each of those is somewhere a mistake would live, and
* each is compared here against calling BearSSL directly.
*
* FIPS-197's own vectors come first either way, so that agreement means AES
* rather than two callers agreeing on something that is not.
*
* Run with an argument to corrupt results on purpose and see the comparison
* notice: a differential test that cannot fail has said nothing.
*/
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <stdint.h>
#include "bearssl/inner.h"
#include "crypton_aes.h"
#include "aes/generic.h"
#include "aes/gf.h"
#include "aes/block128.h"
/*
* crypton_aes.c's table, which is a global, and the three key sizes of the
* two CTR entries this looks for in it. Searching rather than indexing
* because the index is an enum private to that file.
*/
#define BRANCH_TABLE_SEARCH 64
extern void *crypton_aes_branch_table[];
/* the two GCM implementations, which crypton_aes.c declares but no header
* does: in this build the generic one reaches the portable block function
* too, so the two have to agree exactly, tag and all */
void crypton_aes_generic_gcm_encrypt(uint8_t *, aes_gcm *, aes_key *, uint8_t *, uint32_t);
void crypton_aes_generic_gcm_decrypt(uint8_t *, aes_gcm *, aes_key *, uint8_t *, uint32_t);
void crypton_aes_bitsliced_gcm_encrypt(uint8_t *, aes_gcm *, aes_key *, uint8_t *, uint32_t);
void crypton_aes_bitsliced_gcm_decrypt(uint8_t *, aes_gcm *, aes_key *, uint8_t *, uint32_t);
/* ---- the vendored code, one block at a time, as aes_ct64_cbcenc.c does ---- */
static void bear_key(uint64_t *comp_skey, unsigned *nr,
const uint8_t *key, size_t len)
{
*nr = br_aes_ct64_keysched(comp_skey, key, len);
}
static void bear_block(uint8_t *out, const uint64_t *comp_skey, unsigned nr,
const uint8_t *in, int decrypt)
{
uint64_t sk_exp[120];
uint32_t w[4];
uint64_t q[8];
br_aes_ct64_skey_expand(sk_exp, nr, comp_skey);
w[0] = br_dec32le(in);
w[1] = br_dec32le(in + 4);
w[2] = br_dec32le(in + 8);
w[3] = br_dec32le(in + 12);
memset(q, 0, sizeof q);
br_aes_ct64_interleave_in(&q[0], &q[4], w);
br_aes_ct64_ortho(q);
if (decrypt)
br_aes_ct64_bitslice_decrypt(nr, sk_exp, q);
else
br_aes_ct64_bitslice_encrypt(nr, sk_exp, q);
br_aes_ct64_ortho(q);
br_aes_ct64_interleave_out(w, q[0], q[4]);
br_enc32le(out, w[0]);
br_enc32le(out + 4, w[1]);
br_enc32le(out + 8, w[2]);
br_enc32le(out + 12, w[3]);
}
/* ---- crypton's GHASH, driven the way crypton_aes.c drives it ---- */
static void crypton_ghash(uint8_t *y, const uint8_t *h,
const uint8_t *data, size_t len)
{
table_4bit ht;
block128 acc;
size_t i;
crypton_aes_generic_hinit(ht, (const block128 *) h);
block128_zero(&acc);
for (i = 0; i < len; i += 16) {
block128_xor_bytes(&acc, data + i, 16);
crypton_aes_generic_gf_mul(&acc, ht);
}
memcpy(y, &acc, 16);
}
/* ---- the comparison ---- */
static int failures;
static int sabotage;
static void same(const char *what, const uint8_t *a, const uint8_t *b, size_t n)
{
if (memcmp(a, b, n) != 0) {
size_t i;
printf(" MISMATCH %s\n bearssl ", what);
for (i = 0; i < n; i++) printf("%02x", a[i]);
printf("\n crypton ");
for (i = 0; i < n; i++) printf("%02x", b[i]);
printf("\n");
failures++;
}
}
static uint32_t rnd_state = 1;
static uint8_t rnd(void)
{
rnd_state = rnd_state * 1103515245u + 12345u;
return (uint8_t)(rnd_state >> 16);
}
static void rnd_fill(uint8_t *p, size_t n)
{
while (n--) *p++ = rnd();
}
int main(int argc, char **argv)
{
/* FIPS-197 C.1, C.2, C.3 */
static const uint8_t pt[16] = {
0x00,0x11,0x22,0x33,0x44,0x55,0x66,0x77,
0x88,0x99,0xaa,0xbb,0xcc,0xdd,0xee,0xff };
static const uint8_t k128[16] = {
0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15 };
static const uint8_t k192[24] = {
0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22,23 };
static const uint8_t k256[32] = {
0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,
16,17,18,19,20,21,22,23,24,25,26,27,28,29,30,31 };
static const uint8_t c128[16] = {
0x69,0xc4,0xe0,0xd8,0x6a,0x7b,0x04,0x30,
0xd8,0xcd,0xb7,0x80,0x70,0xb4,0xc5,0x5a };
static const uint8_t c192[16] = {
0xdd,0xa9,0x7c,0xa4,0x86,0x4c,0xdf,0xe0,
0x6e,0xaf,0x70,0xa0,0xec,0x0d,0x71,0x91 };
static const uint8_t c256[16] = {
0x8e,0xa2,0xb7,0xca,0x51,0x67,0x45,0xbf,
0xea,0xfc,0x49,0x90,0x4b,0x49,0x60,0x89 };
const uint8_t *keys[3] = { k128, k192, k256 };
const uint8_t *cts[3] = { c128, c192, c256 };
size_t klens[3] = { 16, 24, 32 };
uint64_t comp[30];
unsigned nr;
uint8_t a[16], b[16];
int i, round;
sabotage = argc > 1;
printf("== FIPS-197, so that agreement means AES ==\n");
for (i = 0; i < 3; i++) {
bear_key(comp, &nr, keys[i], klens[i]);
bear_block(a, comp, nr, pt, 0);
if (sabotage && i == 1) a[0] ^= 1;
same("bearssl vs FIPS-197", a, cts[i], 16);
bear_block(b, comp, nr, cts[i], 1);
same("bearssl decrypt vs plaintext", b, pt, 16);
}
printf("== crypton's glue vs BearSSL direct, 2000 random keys and blocks ==\n");
for (round = 0; round < 2000; round++) {
uint8_t key[32], in[16], e1[16], e2[16], d1[16], d2[16];
size_t kl = klens[round % 3];
aes_key ck;
rnd_fill(key, kl);
rnd_fill(in, 16);
bear_key(comp, &nr, key, kl);
bear_block(e1, comp, nr, in, 0);
bear_block(d1, comp, nr, in, 1);
crypton_aes_generic_init(&ck, key, (uint8_t) kl);
crypton_aes_generic_encrypt_block((aes_block *) e2, &ck,
(aes_block *) in);
crypton_aes_generic_decrypt_block((aes_block *) d2, &ck,
(aes_block *) in);
if (sabotage && round == 7) e1[3] ^= 0x10;
same("encrypt", e1, e2, 16);
same("decrypt", d1, d2, 16);
}
printf("== GHASH: crypton's entries vs br_ghash_ctmul64, 500 messages ==\n");
for (round = 0; round < 500; round++) {
uint8_t h[16], data[256], y1[16], y2[16];
size_t len = 16u * (size_t)(1 + (round % 16));
rnd_fill(h, 16);
rnd_fill(data, len);
memset(y1, 0, 16);
br_ghash_ctmul64(y1, h, data, len);
crypton_ghash(y2, h, data, len);
if (sabotage && round == 3) y1[15] ^= 0x80;
same("ghash", y1, y2, 16);
}
printf("== gf_mul4: the four-block entry against the same four blocks ==\n");
for (round = 0; round < 500; round++) {
uint8_t h[16], data[64], y1[16];
table_4bit ht;
block128 acc;
rnd_fill(h, 16);
rnd_fill(data, sizeof data);
memset(y1, 0, 16);
br_ghash_ctmul64(y1, h, data, sizeof data);
crypton_aes_generic_hinit(ht, (const block128 *) h);
block128_zero(&acc);
crypton_aes_generic_gf_mul4(&acc, (const block128 *) data, ht);
if (sabotage && round == 11) y1[0] ^= 0x40;
same("gf_mul4", y1, (const uint8_t *) &acc, 16);
}
/*
* And that the wide entries are the ones installed. Nothing else
* here would notice if they were not: the entries they replace are
* correct too, just a block at a time, so every answer above would
* be the same and only the speed would be gone.
*/
/*
* The Haskell suite reaches the four-block pass, but thinly: its
* vectors are mostly a block or three long, and breaking that pass
* alone fails sixteen of its examples where breaking every pass
* fails seven hundred. So the lane logic and the counter are
* covered here instead, where the lengths can be chosen.
*/
printf("== many blocks at a pass against one at a time ==\n");
for (round = 0; round < 200; round++) {
uint8_t key[32], in[16 * 9], wide[16 * 9], single[16 * 9];
size_t kl = klens[round % 3];
uint32_t nb = 1 + (round % 9);
aes_sched sched;
aes_key ck;
uint32_t i;
int dec = round & 1;
rnd_fill(key, kl);
rnd_fill(in, nb * 16);
crypton_aes_generic_init(&ck, key, (uint8_t) kl);
crypton_aes_generic_schedule(&sched, &ck);
crypton_aes_generic_blocks(wide, in, nb, &sched, dec);
for (i = 0; i < nb; i++) {
if (dec)
crypton_aes_generic_decrypt_block(
(aes_block *) (single + 16 * i), &ck,
(aes_block *) (in + 16 * i));
else
crypton_aes_generic_encrypt_block(
(aes_block *) (single + 16 * i), &ck,
(aes_block *) (in + 16 * i));
}
if (sabotage && round == 5) wide[16] ^= 2;
same("wide pass", wide, single, nb * 16);
}
printf("== the four-block CTR against one block at a time ==\n");
for (round = 0; round < 200; round++) {
uint8_t key[32], iv[16], in[200], got[200], want[200];
size_t kl = klens[round % 3];
/* lengths that land on, before and after a group of four */
uint32_t len = 1 + (round % 200);
aes_key ck;
aes_block counter, ks;
uint32_t done, n, i;
int c32 = round & 1;
rnd_fill(key, kl);
rnd_fill(iv, 16);
rnd_fill(in, len);
crypton_aes_generic_init(&ck, key, (uint8_t) kl);
if (c32)
crypton_aes_bitsliced_encrypt_c32(got, &ck,
(aes_block *) iv, in, len);
else
crypton_aes_bitsliced_encrypt_ctr(got, &ck,
(aes_block *) iv, in, len);
block128_copy(&counter, (block128 *) iv);
for (done = 0; done < len; done += 16) {
crypton_aes_generic_encrypt_block(&ks, &ck, &counter);
n = len - done < 16 ? len - done : 16;
for (i = 0; i < n; i++)
want[done + i] = ((uint8_t *) &ks)[i]
^ in[done + i];
if (c32)
block128_inc32_le(&counter);
else
block128_inc_be(&counter);
}
if (sabotage && round == 9) got[len - 1] ^= 4;
same("ctr", got, want, len);
}
printf("== XTS four at a pass against one at a time ==\n");
for (round = 0; round < 200; round++) {
uint8_t key[32], key2[32], du[16], in[16 * 11];
uint8_t got[16 * 11], want[16 * 11];
size_t kl = klens[round % 3];
uint32_t nb = 1 + (round % 11);
uint32_t spoint = round % 3;
aes_key k1, k2;
aes_block tweak;
uint32_t i, j;
int dec = round & 1;
rnd_fill(key, kl);
rnd_fill(key2, kl);
rnd_fill(du, 16);
rnd_fill(in, nb * 16);
crypton_aes_initkey(&k1, key, (uint8_t) kl);
crypton_aes_initkey(&k2, key2, (uint8_t) kl);
{
aes_block d;
memcpy(&d, du, 16);
if (dec)
crypton_aes_decrypt_xts((aes_block *) got, &k1, &k2,
&d, spoint, (aes_block *) in, nb);
else
crypton_aes_encrypt_xts((aes_block *) got, &k1, &k2,
&d, spoint, (aes_block *) in, nb);
}
/* the same thing a block at a time */
memcpy(&tweak, du, 16);
crypton_aes_generic_encrypt_block(&tweak, &k2, &tweak);
for (j = 0; j < spoint; j++)
crypton_aes_generic_gf_mulx((block128 *) &tweak);
for (i = 0; i < nb; i++) {
aes_block t;
block128_vxor(&t, (block128 *) (in + 16 * i), &tweak);
if (dec)
crypton_aes_generic_decrypt_block(&t, &k1, &t);
else
crypton_aes_generic_encrypt_block(&t, &k1, &t);
block128_vxor((block128 *) (want + 16 * i), &t, &tweak);
crypton_aes_generic_gf_mulx((block128 *) &tweak);
}
if (sabotage && round == 17) got[16] ^= 8;
same("xts", got, want, nb * 16);
}
printf("== the four-block GCM against the one-block GCM ==\n");
for (round = 0; round < 200; round++) {
uint8_t key[32], iv[12], in[300], ga[300], gb[300];
size_t kl = klens[round % 3];
uint32_t len = 1 + (round % 300);
aes_key ck;
aes_gcm g1, g2;
int dec = round & 1;
rnd_fill(key, kl);
rnd_fill(iv, sizeof iv);
rnd_fill(in, len);
crypton_aes_initkey(&ck, key, (uint8_t) kl);
crypton_aes_gcm_init(&g1, &ck, iv, sizeof iv);
memcpy(&g2, &g1, sizeof g1);
if (dec) {
crypton_aes_generic_gcm_decrypt(ga, &g1, &ck, in, len);
crypton_aes_bitsliced_gcm_decrypt(gb, &g2, &ck, in, len);
} else {
crypton_aes_generic_gcm_encrypt(ga, &g1, &ck, in, len);
crypton_aes_bitsliced_gcm_encrypt(gb, &g2, &ck, in, len);
}
if (sabotage && round == 13) gb[0] ^= 1;
same("gcm text", ga, gb, len);
/* the running GHASH and counter, which the tag is made from */
same("gcm state", (const uint8_t *) &g1, (const uint8_t *) &g2,
sizeof g1);
}
printf("== which implementation the build chose ==\n");
#if !defined(WITH_AESNI) && !defined(WITH_ARMV8_CRYPTO)
/*
* With no accelerator compiled in, crypton_aes.c does not read the
* branch table at all -- its GET_ macros name the portable entries
* directly -- so there is nothing here to look at, and which
* implementation runs is settled by the preprocessor. Scanning the
* table in this build was a check that passed while telling nothing,
* which is how the four-block CTR came to be written, installed, and
* never called.
*/
printf(" named at compile time; the table is not read in this build\n");
(void) crypton_aes_branch_table;
if (sabotage) { /* nothing to corrupt here */ }
#else
{
int i, ctr = 0, c32 = 0;
for (i = 0; i < BRANCH_TABLE_SEARCH; i++) {
if (crypton_aes_branch_table[i] ==
(void *) crypton_aes_bitsliced_encrypt_ctr)
ctr++;
if (crypton_aes_branch_table[i] ==
(void *) crypton_aes_bitsliced_encrypt_c32)
c32++;
}
printf(" CTR entries %d, C32 entries %d\n", ctr, c32);
if (sabotage) { ctr = 0; }
if (ctr != 3 || c32 != 3) {
printf(" MISMATCH the portable CTR entries were not installed\n");
failures++;
}
}
#endif
printf("%s: %d mismatch(es)%s\n",
failures ? "FAIL" : "ok", failures,
sabotage ? " (sabotage was asked for)" : "");
return failures != 0;
}