diff --git a/docs/README.md b/docs/README.md index 62475d9..7f7361a 100644 --- a/docs/README.md +++ b/docs/README.md @@ -7,3 +7,6 @@ Craig and I started by trying to create a simple e-commerce website that used Aeternity as its payment system. We are also currently developing a wallet. We have run into many weird stupid technical pitfalls or weird things you have to learn about. This file tree documents the ones we cared to write about. + +- How to install NPM on Linux without getting AIDS +- diff --git a/docs/baseN/README.md b/docs/baseN/README.md index 4d723c2..f6022c0 100644 --- a/docs/baseN/README.md +++ b/docs/baseN/README.md @@ -1,12 +1,128 @@ # Base58/Base64 Number Encoding Schema in Detail +Base64 and Base58 are two algorithms for encoding byte arrays in plain text. I +initially assumed these were two instances of the same "Base N" algorithm. **This +is not the case. These are two fundamentally different algorithms.** + ## tldr ```erlang -b58_enc(Bits) -> - NBits = bit_size(Bits), - <> = Bits, - b58_enc(BitNum, []). +-spec b64_enc(Bytes) -> Base64 + when Bytes :: binary(). + Base64 :: string(). +%% @doc +%% Encode a byte array into a base64 string +%% @end + +%% general case: at least 3 bytes (24 bits) remaining +%% encode into 4 characters +b64_enc(<>) -> + CA = b64_int2char(A), + CB = b64_int2char(B), + CC = b64_int2char(C), + CD = b64_int2char(D), + [CA, CB, CC, CD | b64_enc(Rest)], +%% terminal case: 2 bytes (= 16 bits) remaining +%% encode into 3 characters and a single padding character +b64_enc(<>) -> + CA = b64_int2char(A), + CB = b64_int2char(B), + CC = b64_int2char(C bsl 2), + [CA, CB, CC, $=]; +%% terminal case: 1 bytes (= 8 bits) remaining +%% encode into 2 characters and two padding characters +b64_enc(<>) -> + CA = b64_int2char(A), + CB = b64_int2char(B bsl 4), + [CA, CB, $=, $=]; +%% terminal case: empty byte array +b64_enc(<<>>) -> + []. + + + +-spec b64_dec(Base64) -> Bytes + when Base64 :: string(), + Bytes :: binary(). +%% @doc +%% Decode a base64 string into a byte array +%% @end + +b64_dec(Base64_String) -> + b64_dec(Base64_String, <<>>). + + +%% terminal case: two equals signs at the end (decode 1 byte) +b64_dec([W, X, $=, $=], Acc) -> + NW = b64_char2int(W), + NX = b64_char2int(X), + <> = <>, + <>; +%% terminal case: one equals sign at the end (decode 2 bytes) +b64_dec([W, X, Y, $=], Acc) -> + NW = b64_char2int(W), + NX = b64_char2int(X), + NY = b64_char2int(Y), + <> = <>, + <>; +%% terminal case: end of string +b64_dec([], Acc) -> + Acc; +%% general case: 4 or more chars remaining (decode 3 bytes) +b64_dec([W, X, Y, Z | Rest], Acc) -> + NW = b64_char2int(W), + NX = b64_char2int(X), + NY = b64_char2int(Y), + NZ = b64_char2int(Z), + NewAcc = <>, + b64_dec(Rest, NewAcc). + + + +-spec b58_enc(Bytes) -> Base58 + when Bytes :: binary(), + Base58 :: string(). +%% @doc +%% Encode a bytestring into base58 notation + +b58_enc(Bytes) -> + %% grab leading 0s + {ZerosBase58, Rest} = split_zeros(Bytes, []), + NBitsInRest = bit_size(Rest), + <> = Rest, + RestBase58 = b58_enc(RestBigNum, []), + ZerosBase58 ++ RestBase58. + + + +-spec split_zeros(Bytes, B58_Zeros_Acc) -> {B58_Zeros, Rest} + when Bytes :: binary(), + B58_Zeros_Acc :: string(), + B58_Zeros :: string(), + Rest :: binary(). +%% @private +%% Base58 thinks of your byte array as a big integer, and therefore has no way +%% to distinguish between say <<1,2,3>> and <<0, 0, 0, 1, 2, 3>>. To resolve +%% this, we prepend ASCII `1`s at the beginning of the result, one for each +%% leading zero byte in the input byte array. +%% +%% The ASCII `0` (numeral zero) character is not used in order to avoid +%% ambiguity with ASCII `O` (uppercase letter O) + +split_zeros(<<0:8, Rest/binary>>, B58_Zeros) -> + split_zeros(Rest, [$1 | B58_Zeros]); +split_zeros(Rest, B58_Zeros) -> + {B58_Zeros, Rest}. + + + +-spec b58_enc(BytesBigNum, Base58Acc) -> Base58 + when BytesBigNum :: integer(), + Base58Acc :: [0..57], + Base58 :: string(). +%% @private +%% Encode a number into base58 notation using the standard quotient-remainder +%% algorithm you would use for any other base. b58_enc(0, Acc) -> lists:map(fun b58_int2char/1, Acc); @@ -17,229 +133,49 @@ b58_enc(BitNum, Acc) -> -b58_dec(Str) -> - Ns = lists:map(fun b58_char2int/1, Str), - b58_dec(Ns, 0). +-spec b58_dec(Base58) -> DecodedBytes + when Base58 :: string(), + DecodedBytes :: binary(). +%% @doc +%% Decode a Base58-encoded string into a bytestring +%% @end +%% this works by parsing the string as an integer and then converting the +%% integer to "base 256" (256 = 2^8) +b58_dec(Str) -> + %% the number of leading 1 in the input plain-text string tells us the + %% number of leading zeros in the output byte array + {LeadingZeros, RestStr} = split_ones(Str, <<>>), + %% you could make this more efficient by converting this to a single pass + %% over the input string. steps shown separately because this is tutorial + %% code + RestNs = lists:map(fun b58_char2int/1, RestStr), + RestBytes = b58_dec(RestNs, 0), + <>. + +%% this is basically the oppsite of split_zeros/2 above +split_ones([$1 | Rest], LeadingZeros) -> + split_ones(Rest, <>); +split_ones(B58Str, LeadingZeros) -> + {LeadingZeros, B58Str}. + + +%% this parses the input symbols into a big integer b58_dec([N | Ns], Acc) -> NewAcc = (Acc*58) + N, b58_dec(Ns, NewAcc); +%% then converts that big integer into "base 256" (i.e. a byte array) b58_dec([], FinalAccN) -> - MinNBits = trunc(math:log2(FinalAccN) + 1), - NBytes = ceil(MinNBits / 8), - NBits = NBytes * 8, - <>. + bignum_to_binary_bige(FinalAccN, <<>>). - - -b64_enc(<>) -> - CA = b64_int2char(A), - CB = b64_int2char(B), - CC = b64_int2char(C), - CD = b64_int2char(D), - [CA, CB, CC, CD | b64_enc(Rest)], -b64_enc(<>) -> - CA = b64_int2char(A), - CB = b64_int2char(B), - CC = b64_int2char(C bsl 2), - [CA, CB, CC, $=]; -b64_enc(<>) -> - CA = b64_int2char(A), - CB = b64_int2char(B bsl 4), - [CA, CB, $=, $=]; -b64_enc(<<>>) -> - []. - - - -b64_dec(Base64_String) -> - b64_dec(Base64_String, <<>>). - -b64_dec([W, X, $=, $=], Acc) -> - NW = b64_char2int(W), - NX = b64_char2int(X), - <> = <>, - <>; -b64_dec([W, X, Y, $=], Acc) -> - NW = b64_char2int(W), - NX = b64_char2int(X), - NY = b64_char2int(Y), - <> = <>, - <>; -b64_dec([], Acc) -> +%% in the encode step, we were converting the number to a base58 "string" +%% here we are doing essentially the same thing, but instead converting the +%% number to a "base 256 string" (i.e. byte array) +bignum_to_binary_bige(0, Acc) -> Acc; -b64_dec([W, X, Y, Z | Rest], Acc) -> - NW = b64_char2int(W), - NX = b64_char2int(X), - NY = b64_char2int(Y), - NZ = b64_char2int(Z), - NewAcc = <>, - b64_dec(Rest, NewAcc). +bignum_to_binary_bige(N, Acc) -> + Q = N div 256, + R = N rem 256, + NewAcc = <>, + bignum_to_binary_bige(Q, NewAcc). ``` - -## Introduction - -This document explains the Base58 and Base64 notations, and the algorithms -for working with them. I wrote this document because I had a fair bit of -difficulty working this out for myself, even with a strong math background. -I couldn't find any resource on the internet explaining all of this simply. - -Code examples are given in Erlang and TypeScript. These are the two most -common languages used within the Aeternity project, and both happen to be -languges that make these tasks easy. This document assumes you are familiar -with either/both languages. Even if that's not true, Erlang is a very simple -and clean language, so the code should be pretty self-explanatory if you read -it. - -Base64 is kind of annoying but it's pretty straightforward to code. My -initial assumption was that Base58 was in some way "the same" algorithm but -with `n = 58` instead of `n = 64`. When I went to look up the spec, I found -this ([source](https://digitalbazaar.github.io/base58-spec/)) - -> ### 3. The Base58 Encoding Algorithm -> -> To encode an array of bytes to a Base58 encoded value, run the following -> algorithm. All mathematical operations MUST be performed using integer -> arithmetic. Start by initializing a `zero_counter` to zero (`0x0`), an -> `encoding_flag` to zero (`0x0`), a `b58_bytes` array, a `b58_encoding` -> array, and a `carry` value to zero (`0x0`). For each byte in the array of -> bytes and while `carry` does not equal zero (`0x0`) after the first -> iteration: -> -> 1. If `encoding_flag` is not set, and if the byte is a zero (`0x0`), -> increment the value of `zero_counter`. If the value is not zero `(0x0)`, -> set `encoding_flag` to true `(0x1)`. -> 2. If `encoding_flag` is set, multiply the current byte value by 256 and add -> it to `carry`. -> 3. Set the corresponding byte value in `b58_bytes` to the value of `carry` -> modulus 58. -> 4. Set `carry` to the value of `carry` divided by 58. -> -> Once the `b58_bytes` array has been constructed, generate the final -> `b58_encoding` using the following algorithm. Set the first `zero_counter` -> bytes in `b58_encoding` to `1`. Then, for every byte in `b58_array`, map the -> byte value using the Base58 alphabet in the previous section to its -> corresponding character in `b58_encoding`. Return `b58_encoding` as the -> Base58 representation of the input array of bytes. - -I personally have no idea what that does. I found a YouTube video that -explained the Base58 algorithm in a way that made a lot more sense. -([source](https://youtu.be/GedV3S9X89c)). The video gave a clear enough -explanation of the Base58 algorithm that I could _figure out_ what is going on -and why it makes sense. I was able to relate what I was seeing in the video to -background context I happen to have from mathematics. But the video didn't -provide that context. - -I want this document to explain what both algorithms do, why they make sense, -how they are different, and why they _have_ to be different. All with code -examples and sufficient mathematical context. - -Let's get started. - -Any data stored in a computer is represented as an integer. For the purposes of -this discussion, we're going to assume all integers are non-negative (greater -than or equal to 0). The discussion below can easily be modified to accomodate -negative integers. This would add a small amount of annoying complexity in -exchange for no gain in conceptual clarity. Nothing we are doing requires -dealing with negative numbers. - -The problem we are interested in is _how do we represent really big integers in -plain text?_. - -The first point I want you to take away is that **these are two totally -different solutions**. Do not be fooled by the name. It is **NOT** the -case that these are two instances of the same "Base N" -algorithm, just one is `N = 64` and one is `N = 58`. **These are two totally -different approaches to solving the same problem.** - -More precisely, the underlying mathematics behind the two notations is very -similar, but the algorithms for producing them are very different. More detail -later. - -Like I said, any given piece of data is---from the perspective of your -computer---just a very big integer. The difference between the two algorithms -is, roughly: - -1. The Base64 algorithm thinks of that integer as a "stream of digits" -2. The Base58 algorithm thinks of that integer as a "pure integer," kind of - the way math thinks of an integer: the integer _itself_ is a different - thing than the way the integer is _represented_. - -Base64 encoding/decoding involves a straightforward translation back and forth -from the machine representation of integers, without thinking too much (or at -all) about the math involved. - -Base58 encoding/decoding requires thinking about the integer from a more mathy -point of view. That weird arcane algorithm above is what happens when you try -to phrase the mathematics in terms of the machine representation of -really big integers. - -We're going to focus on the mathy point of view and then circle back to the -weird arcane algorithm later on. - -There is actually a good reason we don't use decimal notation for really big -integers: it's extremely wasteful. - -I'll explain the following in more detail in a later section. Roll with me. To -any piece of data there is associated a quantity called **information**. The -_unit_ of information is the _bit_, in the same sense that the unit of length -is the meter. - -1. There are 256 distinct bytes. A single byte (machine digit) - contains exactly 8 bits $8 = \log_2 256$ of information. - -2. There are 10 distinct decimal ("Base10") symbols. A single decimal digit - contains approximately 3.32 bits (`3.32 ~ log2(10)`) of information. - -3. There are 64 distinct Base64 symbols. A single Base64 digit contains - exactly $6$ bits (`6 = log2(64)`) of information. - -4. There are 58 distinct Base58 symbols. A single Base58 digit contains - approximately 5.86 bits (`5.86 ~ log2(58)`) of information. - -What this means is, in base64 notation, each symbol consumes 6 bits of -information, roughly twice the rate of decimal notation (~3.32 bits per -symbol). What this means in practice is that a number written in Base64 -notation is about half as long as a number written in decimal notation. - -For instance, the number `K = 90 682 877 680 429` - -1. requires 14 digits (count them!) in decimal notation - - $$ - \frac{(\log_2 K) \text{ bits}} - {(\log_2 10) \text{ bits per symbol}} - \approx - \frac{46.37 \text{ bits}} - { 3.37 \text{ bits per symbol}} - \approx 13.96 \text{ symbols} - $$ - -2. requires 8 digits in Base64 notation (`UnnAthst`) - - $$ - \frac{(\log_2 K) \text{ bits}} - {(\log_2 64) \text{ bits per symbol}} - \approx - \frac{46.37 \text{ bits}} - { 6 \text{ bits per symbol}} - \approx 7.73 \text{ symbols} - $$ - -3. requires 8 digits in Base58 notation (`i55xNZNt`) - - $$ - \frac{(\log_2 K) \text{ bits}} - {(\log_2 58) \text{ bits per symbol}} - \approx - \frac{46.37 \text{ bits}} - { 5.86 \text{ bits per symbol}} - \approx 7.91 \text{ symbols} - $$ - -As you can see, the difference in space complexity between Base58 and Base64 is -pretty small, but the difference between Base10 is pretty large. Base58 has -the same alphabet (set of symbols) as Base64, minus a handful that can cause -readability or manual input issues. For instance, the Base64 alphabet contains -both the symbol `0` (numeral zero) and `O` (uppercase letter `o`). The Base58 -alphabet contains neither. diff --git a/docs/ecc/README.md b/docs/ecc/README.md index 24713eb..8ec84da 100644 --- a/docs/ecc/README.md +++ b/docs/ecc/README.md @@ -148,3 +148,6 @@ call it `G`) has the property that if we `ec_grop` it with itself repeatedly (`ec_grop(G, ec_grop(G, ec_grop(G, ...)))`), the resulting **\[orbit\]** cycles through every point on the curve. +### The EC group operation + +### Projective Geometry