bip39.py
python
sha256:2fa778aba8ab0ec15295b8624c6480a573482ffc9c206a6d9546f1c41d2c2b7b
feat: supercharge muse blame + remove --porcelain everywhere
Human
patch
164 days ago
| 1 | """muse.core.bip39 — BIP39 mnemonic generation, validation, and seed derivation. |
| 2 | |
| 3 | BIP39 defines a standard for converting a random bit-string into a human-readable |
| 4 | word sequence (the *mnemonic*) and then into a cryptographic seed via |
| 5 | PBKDF2-HMAC-SHA512. That seed feeds into HD wallet derivation (SLIP-0010 for |
| 6 | Ed25519, BIP32 for secp256k1). |
| 7 | |
| 8 | The mnemonic IS the root secret — whoever holds it controls every key derived |
| 9 | from it. Write it down on paper. Never store it digitally without encryption. |
| 10 | |
| 11 | Supported strengths |
| 12 | ------------------- |
| 13 | All five BIP39 entropy levels are supported: |
| 14 | |
| 15 | .. list-table:: |
| 16 | :widths: 15 15 70 |
| 17 | :header-rows: 1 |
| 18 | |
| 19 | * - Bits |
| 20 | - Words |
| 21 | - Constant / use case |
| 22 | * - 128 |
| 23 | - 12 |
| 24 | - :data:`STRENGTH_STANDARD` — standard security, matches most hardware wallets |
| 25 | * - 160 |
| 26 | - 15 |
| 27 | - :data:`STRENGTH_LOW` — slightly higher entropy than 12-word |
| 28 | * - 192 |
| 29 | - 18 |
| 30 | - :data:`STRENGTH_MEDIUM` — strong middle ground |
| 31 | * - 224 |
| 32 | - 21 |
| 33 | - :data:`STRENGTH_HIGH` — high security without the full 24-word burden |
| 34 | * - 256 |
| 35 | - 24 |
| 36 | - :data:`STRENGTH_PARANOID` — maximum entropy for highest-value root identities |
| 37 | |
| 38 | Supported languages |
| 39 | ------------------- |
| 40 | All 12 official BIP39 wordlists are supported: |
| 41 | |
| 42 | ``"english"``, ``"spanish"``, ``"french"``, ``"italian"``, ``"portuguese"``, |
| 43 | ``"czech"``, ``"japanese"``, ``"korean"``, ``"chinese_simplified"``, |
| 44 | ``"chinese_traditional"``, ``"russian"``, ``"turkish"`` |
| 45 | |
| 46 | Language is a *generation and validation* concern only. Seed derivation |
| 47 | (PBKDF2-HMAC-SHA512) is performed on the raw normalised words and is |
| 48 | language-agnostic — a Japanese mnemonic and an English mnemonic with the same |
| 49 | underlying entropy produce the same seed. |
| 50 | |
| 51 | Language detection |
| 52 | ------------------ |
| 53 | Pass ``language="auto"`` to :func:`validate_mnemonic` to auto-detect the |
| 54 | language from the words. Detection is performed by the ``mnemonic`` library |
| 55 | using wordlist membership; it is unambiguous for all 12 official lists. |
| 56 | |
| 57 | Implementation |
| 58 | -------------- |
| 59 | Delegates all entropy generation, wordlist lookup, checksum computation, and |
| 60 | PBKDF2 derivation to the ``mnemonic`` package (official Trezor implementation, |
| 61 | pure Python, production-grade). This module is a typed façade that: |
| 62 | |
| 63 | - Provides all five entropy strengths as named constants |
| 64 | - Exposes all 12 official BIP39 language wordlists |
| 65 | - Enforces NFKD normalisation per spec (hardware-wallet interoperability) |
| 66 | - Offers language auto-detection for validation |
| 67 | - Raises :class:`Bip39Error` instead of bare exceptions |
| 68 | |
| 69 | Security properties |
| 70 | ------------------- |
| 71 | - ``generate_mnemonic()`` reads from the OS CSPRNG (``os.urandom`` via |
| 72 | ``secrets`` inside the ``mnemonic`` library) — never the ``random`` module. |
| 73 | - Passphrase support: BIP39 allows an optional passphrase ("25th word"). |
| 74 | When used, a different seed is derived from the same mnemonic. The |
| 75 | passphrase is **never stored** — it must be supplied on every derivation. |
| 76 | Loss of the passphrase means permanent loss of access; back it up separately. |
| 77 | - For Japanese mnemonics the separator is ideographic space (U+3000); NFKD |
| 78 | normalisation handles this transparently. |
| 79 | |
| 80 | References |
| 81 | ---------- |
| 82 | - BIP39 specification: https://github.com/bitcoin/bips/blob/master/bip-0039.mediawiki |
| 83 | - Trezor ``mnemonic`` library: https://github.com/trezor/python-mnemonic |
| 84 | |
| 85 | Examples |
| 86 | -------- |
| 87 | :: |
| 88 | |
| 89 | from muse.core.bip39 import ( |
| 90 | generate_mnemonic, validate_mnemonic, mnemonic_to_seed, |
| 91 | STRENGTH_STANDARD, STRENGTH_PARANOID, |
| 92 | ) |
| 93 | |
| 94 | # Generate a new 12-word English mnemonic (128-bit entropy) |
| 95 | words = generate_mnemonic() |
| 96 | |
| 97 | # 24-word paranoid-security mnemonic |
| 98 | words_24 = generate_mnemonic(strength=STRENGTH_PARANOID) |
| 99 | |
| 100 | # Japanese 12-word mnemonic |
| 101 | words_ja = generate_mnemonic(language="japanese") |
| 102 | |
| 103 | # Validate — language auto-detected |
| 104 | assert validate_mnemonic(words_ja) |
| 105 | |
| 106 | # Derive the 512-bit seed (input to HD derivation) |
| 107 | seed = mnemonic_to_seed(words) # no passphrase |
| 108 | seed = mnemonic_to_seed(words, "my secret") # with BIP39 passphrase |
| 109 | """ |
| 110 | |
| 111 | from __future__ import annotations |
| 112 | |
| 113 | import unicodedata |
| 114 | from typing import Literal |
| 115 | |
| 116 | from mnemonic import Mnemonic as _Mnemonic |
| 117 | |
| 118 | __all__ = [ |
| 119 | "Bip39Error", |
| 120 | "Bip39Strength", |
| 121 | "STRENGTH_STANDARD", |
| 122 | "STRENGTH_LOW", |
| 123 | "STRENGTH_MEDIUM", |
| 124 | "STRENGTH_HIGH", |
| 125 | "STRENGTH_PARANOID", |
| 126 | "SUPPORTED_LANGUAGES", |
| 127 | "FUNCTIONAL_LANGUAGES", |
| 128 | "generate_mnemonic", |
| 129 | "validate_mnemonic", |
| 130 | "mnemonic_to_seed", |
| 131 | "detect_language", |
| 132 | "word_count", |
| 133 | ] |
| 134 | |
| 135 | # --------------------------------------------------------------------------- |
| 136 | # Strength constants |
| 137 | # --------------------------------------------------------------------------- |
| 138 | |
| 139 | #: 128-bit entropy → 12 words. Standard security; matches most hardware wallets. |
| 140 | STRENGTH_STANDARD: Literal[128] = 128 |
| 141 | |
| 142 | #: 160-bit entropy → 15 words. Slightly above standard; rarely used in practice. |
| 143 | STRENGTH_LOW: Literal[160] = 160 |
| 144 | |
| 145 | #: 192-bit entropy → 18 words. Strong middle ground. |
| 146 | STRENGTH_MEDIUM: Literal[192] = 192 |
| 147 | |
| 148 | #: 224-bit entropy → 21 words. High security without the full 24-word burden. |
| 149 | STRENGTH_HIGH: Literal[224] = 224 |
| 150 | |
| 151 | #: 256-bit entropy → 24 words. Maximum entropy for highest-value root identities. |
| 152 | STRENGTH_PARANOID: Literal[256] = 256 |
| 153 | |
| 154 | #: Type alias for all supported entropy strengths. |
| 155 | Bip39Strength = Literal[128, 160, 192, 224, 256] |
| 156 | |
| 157 | #: Map from entropy bits to mnemonic word count. |
| 158 | _WORDS_FOR_STRENGTH: dict[int, int] = { |
| 159 | 128: 12, |
| 160 | 160: 15, |
| 161 | 192: 18, |
| 162 | 224: 21, |
| 163 | 256: 24, |
| 164 | } |
| 165 | |
| 166 | # --------------------------------------------------------------------------- |
| 167 | # Language constants |
| 168 | # --------------------------------------------------------------------------- |
| 169 | |
| 170 | #: All language identifiers shipped with the installed ``mnemonic`` package. |
| 171 | #: Note: some entries (currently ``"turkish"`` and ``"russian"``) have |
| 172 | #: incomplete wordlist data in this version of the library and cannot generate |
| 173 | #: valid checksums. Use :data:`FUNCTIONAL_LANGUAGES` for languages that are |
| 174 | #: fully operational (generate, validate, and detect). |
| 175 | SUPPORTED_LANGUAGES: list[str] = _Mnemonic.list_languages() |
| 176 | |
| 177 | #: Languages that are fully operational: generation, checksum validation, |
| 178 | #: and auto-detection all work correctly. Use this set when iterating over |
| 179 | #: languages for production key generation. |
| 180 | FUNCTIONAL_LANGUAGES: list[str] = [ |
| 181 | lang for lang in SUPPORTED_LANGUAGES |
| 182 | if lang not in ("turkish", "russian") |
| 183 | ] |
| 184 | |
| 185 | #: Sentinel value for language auto-detection in :func:`validate_mnemonic`. |
| 186 | _LANG_AUTO = "auto" |
| 187 | |
| 188 | #: Per-language Mnemonic singletons — created lazily, one per language. |
| 189 | _MNEMONIC_CACHE: dict[str, _Mnemonic] = {} |
| 190 | |
| 191 | |
| 192 | def _get_mnemonic(language: str) -> _Mnemonic: |
| 193 | """Return a cached :class:`_Mnemonic` instance for *language*.""" |
| 194 | if language not in _MNEMONIC_CACHE: |
| 195 | if language not in SUPPORTED_LANGUAGES: |
| 196 | raise Bip39Error( |
| 197 | f"Unsupported BIP39 language: {language!r}. " |
| 198 | f"Supported: {sorted(SUPPORTED_LANGUAGES)}" |
| 199 | ) |
| 200 | _MNEMONIC_CACHE[language] = _Mnemonic(language) |
| 201 | return _MNEMONIC_CACHE[language] |
| 202 | |
| 203 | |
| 204 | # --------------------------------------------------------------------------- |
| 205 | # Errors |
| 206 | # --------------------------------------------------------------------------- |
| 207 | |
| 208 | |
| 209 | class Bip39Error(ValueError): |
| 210 | """Raised when a BIP39 operation fails. |
| 211 | |
| 212 | Subclasses :class:`ValueError` so callers that catch ``ValueError`` still |
| 213 | work correctly. Use ``except Bip39Error`` for precise handling. |
| 214 | |
| 215 | Common causes: |
| 216 | |
| 217 | - Unsupported entropy strength (not one of 128, 160, 192, 224, 256). |
| 218 | - Unsupported or misspelled language name. |
| 219 | - Language detection failure (words not from any known wordlist). |
| 220 | |
| 221 | Examples |
| 222 | -------- |
| 223 | :: |
| 224 | |
| 225 | try: |
| 226 | generate_mnemonic(strength=64) |
| 227 | except Bip39Error as exc: |
| 228 | print(f"bad strength: {exc}") |
| 229 | """ |
| 230 | |
| 231 | |
| 232 | # --------------------------------------------------------------------------- |
| 233 | # Public API |
| 234 | # --------------------------------------------------------------------------- |
| 235 | |
| 236 | |
| 237 | def generate_mnemonic( |
| 238 | strength: Bip39Strength = STRENGTH_STANDARD, |
| 239 | language: str = "english", |
| 240 | ) -> str: |
| 241 | """Generate a new BIP39 mnemonic from OS CSPRNG entropy. |
| 242 | |
| 243 | Parameters |
| 244 | ---------- |
| 245 | strength: |
| 246 | Entropy bit-length. One of :data:`STRENGTH_STANDARD` (128), |
| 247 | :data:`STRENGTH_LOW` (160), :data:`STRENGTH_MEDIUM` (192), |
| 248 | :data:`STRENGTH_HIGH` (224), or :data:`STRENGTH_PARANOID` (256). |
| 249 | Default: :data:`STRENGTH_STANDARD`. |
| 250 | language: |
| 251 | BIP39 wordlist language. One of the strings in |
| 252 | :data:`SUPPORTED_LANGUAGES`. Default: ``"english"``. |
| 253 | |
| 254 | Returns |
| 255 | ------- |
| 256 | str |
| 257 | Space-separated mnemonic phrase in the requested language. |
| 258 | All words are from the official BIP39 wordlist for that language. |
| 259 | The checksum word is included as the final word. |
| 260 | |
| 261 | .. note:: |
| 262 | Japanese mnemonics use ideographic space (U+3000) as the word |
| 263 | separator, as required by the BIP39 Japanese wordlist spec. |
| 264 | |
| 265 | Raises |
| 266 | ------ |
| 267 | Bip39Error |
| 268 | If *strength* is not a supported value, or *language* is not |
| 269 | a supported BIP39 wordlist language. |
| 270 | |
| 271 | Security |
| 272 | -------- |
| 273 | Entropy is read from the OS CSPRNG (``os.urandom`` inside the ``mnemonic`` |
| 274 | library — the same source used by ``secrets.token_bytes``). The Python |
| 275 | ``random`` module is never used. |
| 276 | |
| 277 | Examples |
| 278 | -------- |
| 279 | :: |
| 280 | |
| 281 | words = generate_mnemonic() # 12-word English |
| 282 | words_24 = generate_mnemonic(strength=STRENGTH_PARANOID) # 24-word English |
| 283 | words_15 = generate_mnemonic(strength=STRENGTH_LOW) # 15-word English |
| 284 | words_ja = generate_mnemonic(language="japanese") # 12-word Japanese |
| 285 | words_es = generate_mnemonic(strength=STRENGTH_HIGH, language="spanish") # 21-word Spanish |
| 286 | """ |
| 287 | if strength not in _WORDS_FOR_STRENGTH: |
| 288 | raise Bip39Error( |
| 289 | f"Unsupported BIP39 strength: {strength}. " |
| 290 | f"Must be one of {sorted(_WORDS_FOR_STRENGTH)}." |
| 291 | ) |
| 292 | return _get_mnemonic(language).generate(strength=strength) |
| 293 | |
| 294 | |
| 295 | def validate_mnemonic(words: str, language: str = _LANG_AUTO) -> bool: |
| 296 | """Return ``True`` when *words* is a valid BIP39 mnemonic. |
| 297 | |
| 298 | Validation checks (performed by the ``mnemonic`` library): |
| 299 | |
| 300 | 1. Word count is 12, 15, 18, 21, or 24. |
| 301 | 2. Every word appears in the BIP39 wordlist for the given (or detected) language. |
| 302 | 3. The embedded checksum (last ``entropy_bits / 32`` bits of SHA-256(entropy)) |
| 303 | matches — detects single-word transcription errors. |
| 304 | |
| 305 | Parameters |
| 306 | ---------- |
| 307 | words: |
| 308 | The mnemonic phrase to validate. Leading/trailing whitespace and |
| 309 | runs of internal whitespace are normalised before checking. |
| 310 | language: |
| 311 | Wordlist language to validate against. Pass ``"auto"`` (default) |
| 312 | to auto-detect the language from the words. Pass an explicit language |
| 313 | string (e.g. ``"japanese"``) to skip detection and validate against |
| 314 | that wordlist directly. |
| 315 | |
| 316 | Returns |
| 317 | ------- |
| 318 | bool |
| 319 | ``True`` if and only if the mnemonic passes all BIP39 checks. |
| 320 | ``False`` for any structural, wordlist, or checksum failure. |
| 321 | |
| 322 | Examples |
| 323 | -------- |
| 324 | :: |
| 325 | |
| 326 | assert validate_mnemonic("abandon " * 11 + "about") # classic EN test vector |
| 327 | assert not validate_mnemonic("abandon " * 12) # bad checksum |
| 328 | |
| 329 | words_ja = generate_mnemonic(language="japanese") |
| 330 | assert validate_mnemonic(words_ja) # auto-detect Japanese |
| 331 | assert validate_mnemonic(words_ja, "japanese") # explicit language |
| 332 | """ |
| 333 | normalized = " ".join(words.strip().split()) |
| 334 | if language == _LANG_AUTO: |
| 335 | try: |
| 336 | detected = detect_language(normalized) |
| 337 | except Bip39Error: |
| 338 | return False |
| 339 | m = _get_mnemonic(detected) |
| 340 | else: |
| 341 | m = _get_mnemonic(language) |
| 342 | return bool(m.check(normalized)) |
| 343 | |
| 344 | |
| 345 | def mnemonic_to_seed(words: str, passphrase: str = "") -> bytes: |
| 346 | """Derive the 512-bit BIP39 root seed from a mnemonic and optional passphrase. |
| 347 | |
| 348 | Seed derivation is **language-agnostic** — only the raw normalised words |
| 349 | and passphrase matter. A Japanese and an English mnemonic with identical |
| 350 | underlying entropy bits produce the same seed. |
| 351 | |
| 352 | Implements the BIP39 seed derivation:: |
| 353 | |
| 354 | seed = PBKDF2-HMAC-SHA512( |
| 355 | password = NFKD(mnemonic), |
| 356 | salt = "mnemonic" + NFKD(passphrase), |
| 357 | iterations = 2048, |
| 358 | dklen = 64, # 512 bits |
| 359 | ) |
| 360 | |
| 361 | Parameters |
| 362 | ---------- |
| 363 | words: |
| 364 | BIP39 mnemonic phrase in any supported language. Should be validated |
| 365 | with :func:`validate_mnemonic` before calling this function. An |
| 366 | invalid mnemonic still produces a seed (BIP39 does not error at this |
| 367 | stage), but the seed has no well-defined relationship to any standard |
| 368 | HD wallet. |
| 369 | passphrase: |
| 370 | Optional BIP39 extension passphrase ("25th word"). Default: ``""``. |
| 371 | |
| 372 | .. warning:: |
| 373 | The passphrase is **never stored**. A different passphrase |
| 374 | produces a completely different seed and therefore completely |
| 375 | different keys. Back it up separately from the mnemonic — losing |
| 376 | either means losing all derived keys permanently. |
| 377 | |
| 378 | Returns |
| 379 | ------- |
| 380 | bytes |
| 381 | 64 bytes (512 bits) of deterministic seed material. Feed into |
| 382 | :mod:`muse.core.slip010` (Ed25519) or the BIP32 secp256k1 master |
| 383 | key function. |
| 384 | |
| 385 | Security |
| 386 | -------- |
| 387 | NFKD normalisation is applied to both the mnemonic and passphrase as |
| 388 | required by BIP39. This ensures hardware-wallet compatibility: a Ledger |
| 389 | or Trezor with the same words and passphrase produces the same seed. |
| 390 | |
| 391 | Examples |
| 392 | -------- |
| 393 | :: |
| 394 | |
| 395 | seed = mnemonic_to_seed("abandon " * 11 + "about") |
| 396 | assert len(seed) == 64 |
| 397 | |
| 398 | seed_ja = mnemonic_to_seed(generate_mnemonic(language="japanese")) |
| 399 | assert len(seed_ja) == 64 |
| 400 | |
| 401 | # With passphrase — completely different seed: |
| 402 | seed2 = mnemonic_to_seed("abandon " * 11 + "about", passphrase="TREZOR") |
| 403 | assert seed != seed2 |
| 404 | """ |
| 405 | normalized_words = unicodedata.normalize("NFKD", " ".join(words.strip().split())) |
| 406 | normalized_pass = unicodedata.normalize("NFKD", passphrase) |
| 407 | return bytes(_Mnemonic.to_seed(normalized_words, normalized_pass)) |
| 408 | |
| 409 | |
| 410 | def detect_language(words: str) -> str: |
| 411 | """Detect the BIP39 language of a mnemonic phrase. |
| 412 | |
| 413 | Inspects the words against all 12 official BIP39 wordlists and returns |
| 414 | the name of the matching language. |
| 415 | |
| 416 | Parameters |
| 417 | ---------- |
| 418 | words: |
| 419 | Mnemonic phrase. At least one word must be present. |
| 420 | |
| 421 | Returns |
| 422 | ------- |
| 423 | str |
| 424 | Language name as returned by :data:`SUPPORTED_LANGUAGES`, e.g. |
| 425 | ``"english"``, ``"japanese"``, ``"korean"``. |
| 426 | |
| 427 | Raises |
| 428 | ------ |
| 429 | Bip39Error |
| 430 | If the language cannot be determined (words not from any known BIP39 |
| 431 | wordlist, or the phrase is ambiguous). |
| 432 | |
| 433 | Examples |
| 434 | -------- |
| 435 | :: |
| 436 | |
| 437 | words_fr = generate_mnemonic(language="french") |
| 438 | assert detect_language(words_fr) == "french" |
| 439 | |
| 440 | detect_language("not bip39 words") # raises Bip39Error |
| 441 | """ |
| 442 | normalized = " ".join(words.strip().split()) |
| 443 | try: |
| 444 | return _Mnemonic.detect_language(normalized) |
| 445 | except Exception as exc: |
| 446 | raise Bip39Error( |
| 447 | f"Cannot detect BIP39 language for the given mnemonic: {exc}" |
| 448 | ) from exc |
| 449 | |
| 450 | |
| 451 | def word_count(strength: Bip39Strength = STRENGTH_STANDARD) -> int: |
| 452 | """Return the number of mnemonic words for the given entropy *strength*. |
| 453 | |
| 454 | Parameters |
| 455 | ---------- |
| 456 | strength: |
| 457 | Entropy bit-length. One of 128, 160, 192, 224, or 256. |
| 458 | |
| 459 | Returns |
| 460 | ------- |
| 461 | int |
| 462 | 12 / 15 / 18 / 21 / 24 for 128 / 160 / 192 / 224 / 256 bits. |
| 463 | |
| 464 | Raises |
| 465 | ------ |
| 466 | Bip39Error |
| 467 | If *strength* is not a supported value. |
| 468 | |
| 469 | Examples |
| 470 | -------- |
| 471 | :: |
| 472 | |
| 473 | assert word_count(128) == 12 |
| 474 | assert word_count(160) == 15 |
| 475 | assert word_count(192) == 18 |
| 476 | assert word_count(224) == 21 |
| 477 | assert word_count(256) == 24 |
| 478 | """ |
| 479 | if strength not in _WORDS_FOR_STRENGTH: |
| 480 | raise Bip39Error( |
| 481 | f"Unsupported BIP39 strength: {strength}. " |
| 482 | f"Must be one of {sorted(_WORDS_FOR_STRENGTH)}." |
| 483 | ) |
| 484 | return _WORDS_FOR_STRENGTH[strength] |
File History
1 commit
sha256:2fa778aba8ab0ec15295b8624c6480a573482ffc9c206a6d9546f1c41d2c2b7b
feat: supercharge muse blame + remove --porcelain everywhere
Human
patch
164 days ago