gabriel / muse public
bip39.py python
484 lines 16.2 KB
Raw
sha256:2fa778aba8ab0ec15295b8624c6480a573482ffc9c206a6d9546f1c41d2c2b7b feat: supercharge muse blame + remove --porcelain everywhere Human patch 164 days ago
1 """muse.core.bip39 — BIP39 mnemonic generation, validation, and seed derivation.
2
3 BIP39 defines a standard for converting a random bit-string into a human-readable
4 word sequence (the *mnemonic*) and then into a cryptographic seed via
5 PBKDF2-HMAC-SHA512. That seed feeds into HD wallet derivation (SLIP-0010 for
6 Ed25519, BIP32 for secp256k1).
7
8 The mnemonic IS the root secret — whoever holds it controls every key derived
9 from it. Write it down on paper. Never store it digitally without encryption.
10
11 Supported strengths
12 -------------------
13 All five BIP39 entropy levels are supported:
14
15 .. list-table::
16 :widths: 15 15 70
17 :header-rows: 1
18
19 * - Bits
20 - Words
21 - Constant / use case
22 * - 128
23 - 12
24 - :data:`STRENGTH_STANDARD` — standard security, matches most hardware wallets
25 * - 160
26 - 15
27 - :data:`STRENGTH_LOW` — slightly higher entropy than 12-word
28 * - 192
29 - 18
30 - :data:`STRENGTH_MEDIUM` — strong middle ground
31 * - 224
32 - 21
33 - :data:`STRENGTH_HIGH` — high security without the full 24-word burden
34 * - 256
35 - 24
36 - :data:`STRENGTH_PARANOID` — maximum entropy for highest-value root identities
37
38 Supported languages
39 -------------------
40 All 12 official BIP39 wordlists are supported:
41
42 ``"english"``, ``"spanish"``, ``"french"``, ``"italian"``, ``"portuguese"``,
43 ``"czech"``, ``"japanese"``, ``"korean"``, ``"chinese_simplified"``,
44 ``"chinese_traditional"``, ``"russian"``, ``"turkish"``
45
46 Language is a *generation and validation* concern only. Seed derivation
47 (PBKDF2-HMAC-SHA512) is performed on the raw normalised words and is
48 language-agnostic — a Japanese mnemonic and an English mnemonic with the same
49 underlying entropy produce the same seed.
50
51 Language detection
52 ------------------
53 Pass ``language="auto"`` to :func:`validate_mnemonic` to auto-detect the
54 language from the words. Detection is performed by the ``mnemonic`` library
55 using wordlist membership; it is unambiguous for all 12 official lists.
56
57 Implementation
58 --------------
59 Delegates all entropy generation, wordlist lookup, checksum computation, and
60 PBKDF2 derivation to the ``mnemonic`` package (official Trezor implementation,
61 pure Python, production-grade). This module is a typed façade that:
62
63 - Provides all five entropy strengths as named constants
64 - Exposes all 12 official BIP39 language wordlists
65 - Enforces NFKD normalisation per spec (hardware-wallet interoperability)
66 - Offers language auto-detection for validation
67 - Raises :class:`Bip39Error` instead of bare exceptions
68
69 Security properties
70 -------------------
71 - ``generate_mnemonic()`` reads from the OS CSPRNG (``os.urandom`` via
72 ``secrets`` inside the ``mnemonic`` library) — never the ``random`` module.
73 - Passphrase support: BIP39 allows an optional passphrase ("25th word").
74 When used, a different seed is derived from the same mnemonic. The
75 passphrase is **never stored** — it must be supplied on every derivation.
76 Loss of the passphrase means permanent loss of access; back it up separately.
77 - For Japanese mnemonics the separator is ideographic space (U+3000); NFKD
78 normalisation handles this transparently.
79
80 References
81 ----------
82 - BIP39 specification: https://github.com/bitcoin/bips/blob/master/bip-0039.mediawiki
83 - Trezor ``mnemonic`` library: https://github.com/trezor/python-mnemonic
84
85 Examples
86 --------
87 ::
88
89 from muse.core.bip39 import (
90 generate_mnemonic, validate_mnemonic, mnemonic_to_seed,
91 STRENGTH_STANDARD, STRENGTH_PARANOID,
92 )
93
94 # Generate a new 12-word English mnemonic (128-bit entropy)
95 words = generate_mnemonic()
96
97 # 24-word paranoid-security mnemonic
98 words_24 = generate_mnemonic(strength=STRENGTH_PARANOID)
99
100 # Japanese 12-word mnemonic
101 words_ja = generate_mnemonic(language="japanese")
102
103 # Validate — language auto-detected
104 assert validate_mnemonic(words_ja)
105
106 # Derive the 512-bit seed (input to HD derivation)
107 seed = mnemonic_to_seed(words) # no passphrase
108 seed = mnemonic_to_seed(words, "my secret") # with BIP39 passphrase
109 """
110
111 from __future__ import annotations
112
113 import unicodedata
114 from typing import Literal
115
116 from mnemonic import Mnemonic as _Mnemonic
117
118 __all__ = [
119 "Bip39Error",
120 "Bip39Strength",
121 "STRENGTH_STANDARD",
122 "STRENGTH_LOW",
123 "STRENGTH_MEDIUM",
124 "STRENGTH_HIGH",
125 "STRENGTH_PARANOID",
126 "SUPPORTED_LANGUAGES",
127 "FUNCTIONAL_LANGUAGES",
128 "generate_mnemonic",
129 "validate_mnemonic",
130 "mnemonic_to_seed",
131 "detect_language",
132 "word_count",
133 ]
134
135 # ---------------------------------------------------------------------------
136 # Strength constants
137 # ---------------------------------------------------------------------------
138
139 #: 128-bit entropy → 12 words. Standard security; matches most hardware wallets.
140 STRENGTH_STANDARD: Literal[128] = 128
141
142 #: 160-bit entropy → 15 words. Slightly above standard; rarely used in practice.
143 STRENGTH_LOW: Literal[160] = 160
144
145 #: 192-bit entropy → 18 words. Strong middle ground.
146 STRENGTH_MEDIUM: Literal[192] = 192
147
148 #: 224-bit entropy → 21 words. High security without the full 24-word burden.
149 STRENGTH_HIGH: Literal[224] = 224
150
151 #: 256-bit entropy → 24 words. Maximum entropy for highest-value root identities.
152 STRENGTH_PARANOID: Literal[256] = 256
153
154 #: Type alias for all supported entropy strengths.
155 Bip39Strength = Literal[128, 160, 192, 224, 256]
156
157 #: Map from entropy bits to mnemonic word count.
158 _WORDS_FOR_STRENGTH: dict[int, int] = {
159 128: 12,
160 160: 15,
161 192: 18,
162 224: 21,
163 256: 24,
164 }
165
166 # ---------------------------------------------------------------------------
167 # Language constants
168 # ---------------------------------------------------------------------------
169
170 #: All language identifiers shipped with the installed ``mnemonic`` package.
171 #: Note: some entries (currently ``"turkish"`` and ``"russian"``) have
172 #: incomplete wordlist data in this version of the library and cannot generate
173 #: valid checksums. Use :data:`FUNCTIONAL_LANGUAGES` for languages that are
174 #: fully operational (generate, validate, and detect).
175 SUPPORTED_LANGUAGES: list[str] = _Mnemonic.list_languages()
176
177 #: Languages that are fully operational: generation, checksum validation,
178 #: and auto-detection all work correctly. Use this set when iterating over
179 #: languages for production key generation.
180 FUNCTIONAL_LANGUAGES: list[str] = [
181 lang for lang in SUPPORTED_LANGUAGES
182 if lang not in ("turkish", "russian")
183 ]
184
185 #: Sentinel value for language auto-detection in :func:`validate_mnemonic`.
186 _LANG_AUTO = "auto"
187
188 #: Per-language Mnemonic singletons — created lazily, one per language.
189 _MNEMONIC_CACHE: dict[str, _Mnemonic] = {}
190
191
192 def _get_mnemonic(language: str) -> _Mnemonic:
193 """Return a cached :class:`_Mnemonic` instance for *language*."""
194 if language not in _MNEMONIC_CACHE:
195 if language not in SUPPORTED_LANGUAGES:
196 raise Bip39Error(
197 f"Unsupported BIP39 language: {language!r}. "
198 f"Supported: {sorted(SUPPORTED_LANGUAGES)}"
199 )
200 _MNEMONIC_CACHE[language] = _Mnemonic(language)
201 return _MNEMONIC_CACHE[language]
202
203
204 # ---------------------------------------------------------------------------
205 # Errors
206 # ---------------------------------------------------------------------------
207
208
209 class Bip39Error(ValueError):
210 """Raised when a BIP39 operation fails.
211
212 Subclasses :class:`ValueError` so callers that catch ``ValueError`` still
213 work correctly. Use ``except Bip39Error`` for precise handling.
214
215 Common causes:
216
217 - Unsupported entropy strength (not one of 128, 160, 192, 224, 256).
218 - Unsupported or misspelled language name.
219 - Language detection failure (words not from any known wordlist).
220
221 Examples
222 --------
223 ::
224
225 try:
226 generate_mnemonic(strength=64)
227 except Bip39Error as exc:
228 print(f"bad strength: {exc}")
229 """
230
231
232 # ---------------------------------------------------------------------------
233 # Public API
234 # ---------------------------------------------------------------------------
235
236
237 def generate_mnemonic(
238 strength: Bip39Strength = STRENGTH_STANDARD,
239 language: str = "english",
240 ) -> str:
241 """Generate a new BIP39 mnemonic from OS CSPRNG entropy.
242
243 Parameters
244 ----------
245 strength:
246 Entropy bit-length. One of :data:`STRENGTH_STANDARD` (128),
247 :data:`STRENGTH_LOW` (160), :data:`STRENGTH_MEDIUM` (192),
248 :data:`STRENGTH_HIGH` (224), or :data:`STRENGTH_PARANOID` (256).
249 Default: :data:`STRENGTH_STANDARD`.
250 language:
251 BIP39 wordlist language. One of the strings in
252 :data:`SUPPORTED_LANGUAGES`. Default: ``"english"``.
253
254 Returns
255 -------
256 str
257 Space-separated mnemonic phrase in the requested language.
258 All words are from the official BIP39 wordlist for that language.
259 The checksum word is included as the final word.
260
261 .. note::
262 Japanese mnemonics use ideographic space (U+3000) as the word
263 separator, as required by the BIP39 Japanese wordlist spec.
264
265 Raises
266 ------
267 Bip39Error
268 If *strength* is not a supported value, or *language* is not
269 a supported BIP39 wordlist language.
270
271 Security
272 --------
273 Entropy is read from the OS CSPRNG (``os.urandom`` inside the ``mnemonic``
274 library — the same source used by ``secrets.token_bytes``). The Python
275 ``random`` module is never used.
276
277 Examples
278 --------
279 ::
280
281 words = generate_mnemonic() # 12-word English
282 words_24 = generate_mnemonic(strength=STRENGTH_PARANOID) # 24-word English
283 words_15 = generate_mnemonic(strength=STRENGTH_LOW) # 15-word English
284 words_ja = generate_mnemonic(language="japanese") # 12-word Japanese
285 words_es = generate_mnemonic(strength=STRENGTH_HIGH, language="spanish") # 21-word Spanish
286 """
287 if strength not in _WORDS_FOR_STRENGTH:
288 raise Bip39Error(
289 f"Unsupported BIP39 strength: {strength}. "
290 f"Must be one of {sorted(_WORDS_FOR_STRENGTH)}."
291 )
292 return _get_mnemonic(language).generate(strength=strength)
293
294
295 def validate_mnemonic(words: str, language: str = _LANG_AUTO) -> bool:
296 """Return ``True`` when *words* is a valid BIP39 mnemonic.
297
298 Validation checks (performed by the ``mnemonic`` library):
299
300 1. Word count is 12, 15, 18, 21, or 24.
301 2. Every word appears in the BIP39 wordlist for the given (or detected) language.
302 3. The embedded checksum (last ``entropy_bits / 32`` bits of SHA-256(entropy))
303 matches — detects single-word transcription errors.
304
305 Parameters
306 ----------
307 words:
308 The mnemonic phrase to validate. Leading/trailing whitespace and
309 runs of internal whitespace are normalised before checking.
310 language:
311 Wordlist language to validate against. Pass ``"auto"`` (default)
312 to auto-detect the language from the words. Pass an explicit language
313 string (e.g. ``"japanese"``) to skip detection and validate against
314 that wordlist directly.
315
316 Returns
317 -------
318 bool
319 ``True`` if and only if the mnemonic passes all BIP39 checks.
320 ``False`` for any structural, wordlist, or checksum failure.
321
322 Examples
323 --------
324 ::
325
326 assert validate_mnemonic("abandon " * 11 + "about") # classic EN test vector
327 assert not validate_mnemonic("abandon " * 12) # bad checksum
328
329 words_ja = generate_mnemonic(language="japanese")
330 assert validate_mnemonic(words_ja) # auto-detect Japanese
331 assert validate_mnemonic(words_ja, "japanese") # explicit language
332 """
333 normalized = " ".join(words.strip().split())
334 if language == _LANG_AUTO:
335 try:
336 detected = detect_language(normalized)
337 except Bip39Error:
338 return False
339 m = _get_mnemonic(detected)
340 else:
341 m = _get_mnemonic(language)
342 return bool(m.check(normalized))
343
344
345 def mnemonic_to_seed(words: str, passphrase: str = "") -> bytes:
346 """Derive the 512-bit BIP39 root seed from a mnemonic and optional passphrase.
347
348 Seed derivation is **language-agnostic** — only the raw normalised words
349 and passphrase matter. A Japanese and an English mnemonic with identical
350 underlying entropy bits produce the same seed.
351
352 Implements the BIP39 seed derivation::
353
354 seed = PBKDF2-HMAC-SHA512(
355 password = NFKD(mnemonic),
356 salt = "mnemonic" + NFKD(passphrase),
357 iterations = 2048,
358 dklen = 64, # 512 bits
359 )
360
361 Parameters
362 ----------
363 words:
364 BIP39 mnemonic phrase in any supported language. Should be validated
365 with :func:`validate_mnemonic` before calling this function. An
366 invalid mnemonic still produces a seed (BIP39 does not error at this
367 stage), but the seed has no well-defined relationship to any standard
368 HD wallet.
369 passphrase:
370 Optional BIP39 extension passphrase ("25th word"). Default: ``""``.
371
372 .. warning::
373 The passphrase is **never stored**. A different passphrase
374 produces a completely different seed and therefore completely
375 different keys. Back it up separately from the mnemonic — losing
376 either means losing all derived keys permanently.
377
378 Returns
379 -------
380 bytes
381 64 bytes (512 bits) of deterministic seed material. Feed into
382 :mod:`muse.core.slip010` (Ed25519) or the BIP32 secp256k1 master
383 key function.
384
385 Security
386 --------
387 NFKD normalisation is applied to both the mnemonic and passphrase as
388 required by BIP39. This ensures hardware-wallet compatibility: a Ledger
389 or Trezor with the same words and passphrase produces the same seed.
390
391 Examples
392 --------
393 ::
394
395 seed = mnemonic_to_seed("abandon " * 11 + "about")
396 assert len(seed) == 64
397
398 seed_ja = mnemonic_to_seed(generate_mnemonic(language="japanese"))
399 assert len(seed_ja) == 64
400
401 # With passphrase — completely different seed:
402 seed2 = mnemonic_to_seed("abandon " * 11 + "about", passphrase="TREZOR")
403 assert seed != seed2
404 """
405 normalized_words = unicodedata.normalize("NFKD", " ".join(words.strip().split()))
406 normalized_pass = unicodedata.normalize("NFKD", passphrase)
407 return bytes(_Mnemonic.to_seed(normalized_words, normalized_pass))
408
409
410 def detect_language(words: str) -> str:
411 """Detect the BIP39 language of a mnemonic phrase.
412
413 Inspects the words against all 12 official BIP39 wordlists and returns
414 the name of the matching language.
415
416 Parameters
417 ----------
418 words:
419 Mnemonic phrase. At least one word must be present.
420
421 Returns
422 -------
423 str
424 Language name as returned by :data:`SUPPORTED_LANGUAGES`, e.g.
425 ``"english"``, ``"japanese"``, ``"korean"``.
426
427 Raises
428 ------
429 Bip39Error
430 If the language cannot be determined (words not from any known BIP39
431 wordlist, or the phrase is ambiguous).
432
433 Examples
434 --------
435 ::
436
437 words_fr = generate_mnemonic(language="french")
438 assert detect_language(words_fr) == "french"
439
440 detect_language("not bip39 words") # raises Bip39Error
441 """
442 normalized = " ".join(words.strip().split())
443 try:
444 return _Mnemonic.detect_language(normalized)
445 except Exception as exc:
446 raise Bip39Error(
447 f"Cannot detect BIP39 language for the given mnemonic: {exc}"
448 ) from exc
449
450
451 def word_count(strength: Bip39Strength = STRENGTH_STANDARD) -> int:
452 """Return the number of mnemonic words for the given entropy *strength*.
453
454 Parameters
455 ----------
456 strength:
457 Entropy bit-length. One of 128, 160, 192, 224, or 256.
458
459 Returns
460 -------
461 int
462 12 / 15 / 18 / 21 / 24 for 128 / 160 / 192 / 224 / 256 bits.
463
464 Raises
465 ------
466 Bip39Error
467 If *strength* is not a supported value.
468
469 Examples
470 --------
471 ::
472
473 assert word_count(128) == 12
474 assert word_count(160) == 15
475 assert word_count(192) == 18
476 assert word_count(224) == 21
477 assert word_count(256) == 24
478 """
479 if strength not in _WORDS_FOR_STRENGTH:
480 raise Bip39Error(
481 f"Unsupported BIP39 strength: {strength}. "
482 f"Must be one of {sorted(_WORDS_FOR_STRENGTH)}."
483 )
484 return _WORDS_FOR_STRENGTH[strength]
File History 1 commit
sha256:2fa778aba8ab0ec15295b8624c6480a573482ffc9c206a6d9546f1c41d2c2b7b feat: supercharge muse blame + remove --porcelain everywhere Human patch 164 days ago