diff --git a/.coafile b/.coafile new file mode 100644 index 000000000..666eefe95 --- /dev/null +++ b/.coafile @@ -0,0 +1,5 @@ +[Default] +bears = SpaceConsistencyBear,LineLengthBear,PyUnusedCodeBear +use_spaces = true +max_line_length = 120 +files = babel/**/*.py diff --git a/.coveragerc b/.coveragerc new file mode 100644 index 000000000..a3d8ae65e --- /dev/null +++ b/.coveragerc @@ -0,0 +1,5 @@ +[report] +exclude_lines = + NotImplemented + pragma: no cover + warnings.warn \ No newline at end of file diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml new file mode 100644 index 000000000..e9c411862 --- /dev/null +++ b/.github/workflows/test.yml @@ -0,0 +1,38 @@ +name: Test + +on: + push: + branches: [ master ] + pull_request: + branches: [ master ] + +jobs: + test: + runs-on: ${{ matrix.os }} + strategy: + matrix: + os: [ubuntu-18.04, windows-2019, macos-10.15] + python-version: [3.6, 3.7, 3.8, 3.9, pypy3] + exclude: + - os: windows-2019 + python-version: pypy3 + # TODO: Remove this; see: + # https://github.com/actions/setup-python/issues/151 + # https://github.com/tox-dev/tox/issues/1704 + # https://foss.heptapod.net/pypy/pypy/-/issues/3331 + env: + BABEL_CLDR_NO_DOWNLOAD_PROGRESS: "1" + BABEL_CLDR_QUIET: "1" + steps: + - uses: actions/checkout@v2 + - name: Set up Python ${{ matrix.python-version }} + uses: actions/setup-python@v2 + with: + python-version: ${{ matrix.python-version }} + - name: Install dependencies + run: | + python -m pip install --upgrade pip setuptools wheel + python -m pip install tox tox-gh-actions==2.1.0 + - name: Run test via Tox + run: tox --skip-missing-interpreters + - uses: codecov/codecov-action@v1 diff --git a/.gitignore b/.gitignore index 7effb2fcc..2886dec52 100644 --- a/.gitignore +++ b/.gitignore @@ -1,16 +1,23 @@ -*.so -docs/_build +**/__pycache__ +*.egg +*.egg-info *.pyc *.pyo -*.egg-info -*.egg -build -dist +*.so +*.swp +*~ +.*cache .DS_Store +.idea .tox -test-env -**/__pycache__ +/venv* babel/global.dat -tests/messages/data/project/i18n/long_messages.pot -tests/messages/data/project/i18n/temp.pot +babel/global.dat.json +build +dist +docs/_build +test-env tests/messages/data/project/i18n/en_US +tests/messages/data/project/i18n/long_messages.pot +tests/messages/data/project/i18n/temp* +tests/messages/data/project/i18n/fi_BUGGY/LC_MESSAGES/*.mo diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml new file mode 100644 index 000000000..1fd8b5cb5 --- /dev/null +++ b/.pre-commit-config.yaml @@ -0,0 +1,19 @@ +- repo: https://github.com/pre-commit/pre-commit-hooks + sha: 97b88d9610bcc03982ddac33caba98bb2b751f5f + hooks: + - id: autopep8-wrapper + exclude: (docs/conf.py|tests/messages/data/) + - id: check-added-large-files + - id: check-docstring-first + exclude: (docs/conf.py) + - id: check-json + - id: check-yaml + - id: debug-statements + - id: end-of-file-fixer + - id: flake8 + exclude: (docs/conf.py|babel/messages/__init__.py|babel/__init__.py|babel/_compat.py|tests/messages/data|scripts/import_cldr.py) + - id: name-tests-test + args: ['--django'] + exclude: (tests/messages/data/) + - id: requirements-txt-fixer + - id: trailing-whitespace diff --git a/.rultor.yml b/.rultor.yml new file mode 100644 index 000000000..6267e49ac --- /dev/null +++ b/.rultor.yml @@ -0,0 +1,12 @@ +install: + - pip install pytest + +docker: + as_root: true # for pip installation + image: "coala/rultor-python" + +merge: + fast-forward: only + rebase: true + script: + - echo "Nothing to do (yet!)" diff --git a/.travis.yml b/.travis.yml deleted file mode 100644 index 94425a607..000000000 --- a/.travis.yml +++ /dev/null @@ -1,24 +0,0 @@ -language: python - -python: - - "2.6" - - "2.7" - - "pypy" - - "3.3" - -install: - - pip install --upgrade pip - - pip install pytest - - pip install --editable . - -script: make test - -notifications: - email: false - irc: - channels: - - "chat.freenode.net#pocoo" - on_success: change - on_failure: always - use_notice: true - skip_join: true diff --git a/AUTHORS b/AUTHORS index 09d0bc03b..9cf8f4e7d 100644 --- a/AUTHORS +++ b/AUTHORS @@ -1,23 +1,126 @@ -Babel is written and maintained by the Babel team and various contributors: - -Maintainer and Current Project Lead: -- Armin Ronacher - -Contributors: +Babel is written and maintained by the Babel team and various contributors: -- Christopher Lenz -- Alex Morega -- Felix Schwarz -- Pedro Algarvio -- Jeroen Ruigrok van der Werven -- Philip Jenvey -- Tobias Bieniek -- Jonas Borgström -- Daniel Neuhäuser -- Nick Retallack -- Thomas Waldmann -- Lennart Regebro +- Aarni Koskela +- Christopher Lenz +- Armin Ronacher +- Alex Morega +- Lasse Schuirmann +- Felix Schwarz +- Pedro Algarvio +- Jeroen Ruigrok van der Werven +- Philip Jenvey +- benselme +- Isaac Jurado +- Tobias Bieniek +- Erick Wilder +- Michael Birtwell +- Jonas Borgström +- Kevin Deldycke +- Jon Dufresne +- Ville Skyttä +- Hugo +- Heungsub Lee +- Jakob Schnitzer +- Sachin Paliwal +- Alex Willmer +- Daniel Neuhäuser +- Miro Hrončok +- Cédric Krier +- Luke Plant +- Jennifer Wang +- Lukas Balaga +- sudheesh001 +- Niklas Hambüchen +- Changaco +- Xavier Fernandez +- KO. Mattsson +- Sébastien Diemer +- alexbodn@gmail.com +- saurabhiiit +- srisankethu +- Erik Romijn +- Lukas B +- Ryan J Ollos +- Arturas Moskvinas +- Leonardo Pistone +- Jun Omae +- Hyunjun Kim +- Alessio Bogon +- Nikiforov Konstantin +- Abdullah Javed Nesar +- Brad Martin +- Tyler Kennedy +- CyanNani123 +- sebleblanc +- He Chen +- Steve (Gadget) Barnes +- Romuald Brunet +- Mario Frasca +- BT-sschmid +- Alberto Mardegan +- mondeja +- NotAFile +- Julien Palard +- Brian Cappello +- Serban Constantin +- Bryn Truscott +- Chris +- Charly C +- PTrottier +- xmo-odoo +- StevenJ +- Jungmo Ku +- Simeon Visser +- Narendra Vardi +- Stefane Fermigier +- Narayan Acharya +- François Magimel +- Wolfgang Doll +- Roy Williams +- Marc-André Dufresne +- Abhishek Tiwari +- David Baumgold +- Alex Kuzmenko +- Georg Schölly +- ldwoolley +- Rodrigo Ramírez Norambuena +- Jakub Wilk +- Roman Rader +- Max Shenfield +- Nicolas Grilly +- Kenny Root +- Adam Chainz +- Sébastien Fievet +- Anthony Sottile +- Yuriy Shatrov +- iamshubh22 +- Sven Anderson +- Eoin Nugent +- Roman Imankulov +- David Stanek +- Roy Wellington Ⅳ +- Florian Schulze +- Todd M. Guerra +- Joseph Breihan +- Craig Loftus +- The Gitter Badger +- Régis Behmo +- Julen Ruiz Aizpuru +- astaric +- Felix Yan +- Philip_Tzou +- Jesús Espino +- Jeremy Weinstein +- James Page +- masklinn +- Sjoerd Langkemper +- Matt Iversen +- Alexander A. Dyshev +- Dirkjan Ochtman +- Nick Retallack +- Thomas Waldmann +- xen Babel was previously developed under the Copyright of Edgewall Software. The following copyright notice holds true for releases before 2013: "Copyright (c) diff --git a/CHANGES b/CHANGES index 496f519bb..e3c54bfc8 100644 --- a/CHANGES +++ b/CHANGES @@ -1,6 +1,419 @@ Babel Changelog =============== +Version 2.9.1 +------------- + +Bugfixes +~~~~~~~~ + +* The internal locale-data loading functions now validate the name of the locale file to be loaded and only + allow files within Babel's data directory. Thank you to Chris Lyne of Tenable, Inc. for discovering the issue! + +Version 2.9.0 +------------- + +Upcoming version support changes +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +* This version, Babel 2.9, is the last version of Babel to support Python 2.7, Python 3.4, and Python 3.5. + +Improvements +~~~~~~~~~~~~ + +* CLDR: Use CLDR 37 – Aarni Koskela (#734) +* Dates: Handle ZoneInfo objects in get_timezone_location, get_timezone_name - Alessio Bogon (#741) +* Numbers: Add group_separator feature in number formatting - Abdullah Javed Nesar (#726) + +Bugfixes +~~~~~~~~ + +* Dates: Correct default Format().timedelta format to 'long' to mute deprecation warnings – Aarni Koskela +* Import: Simplify iteration code in "import_cldr.py" – Felix Schwarz +* Import: Stop using deprecated ElementTree methods "getchildren()" and "getiterator()" – Felix Schwarz +* Messages: Fix unicode printing error on Python 2 without TTY. – Niklas Hambüchen +* Messages: Introduce invariant that _invalid_pofile() takes unicode line. – Niklas Hambüchen +* Tests: fix tests when using Python 3.9 – Felix Schwarz +* Tests: Remove deprecated 'sudo: false' from Travis configuration – Jon Dufresne +* Tests: Support Py.test 6.x – Aarni Koskela +* Utilities: LazyProxy: Handle AttributeError in specified func – Nikiforov Konstantin (#724) +* Utilities: Replace usage of parser.suite with ast.parse – Miro Hrončok + +Documentation +~~~~~~~~~~~~~ + +* Update parse_number comments – Brad Martin (#708) +* Add __iter__ to Catalog documentation – @CyanNani123 + +Version 2.8.1 +------------- + +This is solely a patch release to make running tests on Py.test 6+ possible. + +Bugfixes +~~~~~~~~ + +* Support Py.test 6 - Aarni Koskela (#747, #750, #752) + +Version 2.8.0 +------------- + +Improvements +~~~~~~~~~~~~ + +* CLDR: Upgrade to CLDR 36.0 - Aarni Koskela (#679) +* Messages: Don't even open files with the "ignore" extraction method - @sebleblanc (#678) + +Bugfixes +~~~~~~~~ + +* Numbers: Fix formatting very small decimals when quantization is disabled - Lev Lybin, @miluChen (#662) +* Messages: Attempt to sort all messages – Mario Frasca (#651, #606) + +Docs +~~~~ + +* Add years to changelog - Romuald Brunet +* Note that installation requires pytz - Steve (Gadget) Barnes + +Version 2.7.0 +------------- + +Possibly incompatible changes +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +These may be backward incompatible in some cases, as some more-or-less internal +APIs have changed. Please feel free to file issues if you bump into anything +strange and we'll try to help! + +* General: Internal uses of ``babel.util.odict`` have been replaced with + ``collections.OrderedDict`` from The Python standard library. + +Improvements +~~~~~~~~~~~~ + +* CLDR: Upgrade to CLDR 35.1 - Alberto Mardegan, Aarni Koskela (#626, #643) +* General: allow anchoring path patterns to the start of a string - Brian Cappello (#600) +* General: Bumped version requirement on pytz - @chrisbrake (#592) +* Messages: `pybabel compile`: exit with code 1 if errors were encountered - Aarni Koskela (#647) +* Messages: Add omit-header to update_catalog - Cédric Krier (#633) +* Messages: Catalog update: keep user comments from destination by default - Aarni Koskela (#648) +* Messages: Skip empty message when writing mo file - Cédric Krier (#564) +* Messages: Small fixes to avoid crashes on badly formatted .po files - Bryn Truscott (#597) +* Numbers: `parse_decimal()` `strict` argument and `suggestions` - Charly C (#590) +* Numbers: don't repeat suggestions in parse_decimal strict - Serban Constantin (#599) +* Numbers: implement currency formatting with long display names - Luke Plant (#585) +* Numbers: parse_decimal(): assume spaces are equivalent to non-breaking spaces when not in strict mode - Aarni Koskela (#649) +* Performance: Cache locale_identifiers() - Aarni Koskela (#644) + +Bugfixes +~~~~~~~~ + +* CLDR: Skip alt=... for week data (minDays, firstDay, weekendStart, weekendEnd) - Aarni Koskela (#634) +* Dates: Fix wrong weeknumber for 31.12.2018 - BT-sschmid (#621) +* Locale: Avoid KeyError trying to get data on WindowsXP - mondeja (#604) +* Locale: get_display_name(): Don't attempt to concatenate variant information to None - Aarni Koskela (#645) +* Messages: pofile: Add comparison operators to _NormalizedString - Aarni Koskela (#646) +* Messages: pofile: don't crash when message.locations can't be sorted - Aarni Koskela (#646) + +Tooling & docs +~~~~~~~~~~~~~~ + +* Docs: Remove all references to deprecated easy_install - Jon Dufresne (#610) +* Docs: Switch print statement in docs to print function - NotAFile +* Docs: Update all pypi.python.org URLs to pypi.org - Jon Dufresne (#587) +* Docs: Use https URLs throughout project where available - Jon Dufresne (#588) +* Support: Add testing and document support for Python 3.7 - Jon Dufresne (#611) +* Support: Test on Python 3.8-dev - Aarni Koskela (#642) +* Support: Using ABCs from collections instead of collections.abc is deprecated. - Julien Palard (#609) +* Tests: Fix conftest.py compatibility with pytest 4.3 - Miro Hrončok (#635) +* Tests: Update pytest and pytest-cov - Miro Hrončok (#635) + +Version 2.6.0 +------------- + +Possibly incompatible changes +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +These may be backward incompatible in some cases, as some more-or-less internal APIs have changed. +Please feel free to file issues if you bump into anything strange and we'll try to help! + +* Numbers: Refactor decimal handling code and allow bypass of decimal quantization. (@kdeldycke) (PR #538) +* Messages: allow processing files that are in locales unknown to Babel (@akx) (PR #557) +* General: Drop support for EOL Python 2.6 and 3.3 (@hugovk) (PR #546) + +Other changes +~~~~~~~~~~~~~ + +* CLDR: Use CLDR 33 (@akx) (PR #581) +* Lists: Add support for various list styles other than the default (@akx) (#552) +* Messages: Add new PoFileError exception (@Bedrock02) (PR #532) +* Times: Simplify Linux distro specific explicit timezone setting search (@scop) (PR #528) + +Bugfixes +~~~~~~~~ + +* CLDR: avoid importing alt=narrow currency symbols (@akx) (PR #558) +* CLDR: ignore non-Latin numbering systems (@akx) (PR #579) +* Docs: Fix improper example for date formatting (@PTrottier) (PR #574) +* Tooling: Fix some deprecation warnings (@akx) (PR #580) + +Tooling & docs +~~~~~~~~~~~~~~ + +* Add explicit signatures to some date autofunctions (@xmo-odoo) (PR #554) +* Include license file in the generated wheel package (@jdufresne) (PR #539) +* Python 3.6 invalid escape sequence deprecation fixes (@scop) (PR #528) +* Test and document all supported Python versions (@jdufresne) (PR #540) +* Update copyright header years and authors file (@akx) (PR #559) + + +Version 2.5.3 +------------- + +This is a maintenance release that reverts undesired API-breaking changes that slipped into 2.5.2 +(see https://github.com/python-babel/babel/issues/550). + +It is based on v2.5.1 (f29eccd) with commits 7cedb84, 29da2d2 and edfb518 cherry-picked on top. + +Version 2.5.2 +------------- + +Bugfixes +~~~~~~~~ + +* Revert the unnecessary PyInstaller fixes from 2.5.0 and 2.5.1 (#533) (@yagebu) + +Version 2.5.1 +------------- + +Minor Improvements and bugfixes +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +* Use a fixed datetime to avoid test failures (#520) (@narendravardi) +* Parse multi-line __future__ imports better (#519) (@akx) +* Fix validate_currency docstring (#522) +* Allow normalize_locale and exists to handle various unexpected inputs (#523) (@suhojm) +* Make PyInstaller support more robust (#525, #526) (@thijstriemstra, @akx) + + +Version 2.5.0 +------------- + +New Features +~~~~~~~~~~~~ + +* Numbers: Add currency utilities and helpers (#491) (@kdeldycke) +* Support PyInstaller (#500, #505) (@wodo) + +Minor Improvements and bugfixes +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +* Dates: Add __str__ to DateTimePattern (#515) (@sfermigier) +* Dates: Fix an invalid string to bytes comparison when parsing TZ files on Py3 (#498) (@rowillia) +* Dates: Formatting zero-padded components of dates is faster (#517) (@akx) +* Documentation: Fix "Good Commits" link in CONTRIBUTING.md (#511) (@naryanacharya6) +* Documentation: Fix link to Python gettext module (#512) (@Linkid) +* Messages: Allow both dash and underscore separated locale identifiers in pofiles (#489, #490) (@akx) +* Messages: Extract Python messages in nested gettext calls (#488) (@sublee) +* Messages: Fix in-place editing of dir list while iterating (#476, #492) (@MarcDufresne) +* Messages: Stabilize sort order (#482) (@xavfernandez) +* Time zones: Honor the no-inherit marker for metazone names (#405) (@akx) + + +Version 2.4.0 +------------- + +New Features +~~~~~~~~~~~~ + +Some of these changes might break your current code and/or tests. + +* CLDR: CLDR 29 is now used instead of CLDR 28 (#405) (@akx) +* Messages: Add option 'add_location' for location line formatting (#438, #459) (@rrader, @alxpy) +* Numbers: Allow full control of decimal behavior (#410) (@etanol) + +Minor Improvements and bugfixes +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +* Documentation: Improve Date Fields descriptions (#450) (@ldwoolley) +* Documentation: Typo fixes and documentation improvements (#406, #412, #403, #440, #449, #463) (@zyegfryed, @adamchainz, @jwilk, @akx, @roramirez, @abhishekcs10) +* Messages: Default to UTF-8 source encoding instead of ISO-8859-1 (#399) (@asottile) +* Messages: Ensure messages are extracted in the order they were passed in (#424) (@ngrilly) +* Messages: Message extraction for JSX files is improved (#392, #396, #425) (@karloskar, @georgschoelly) +* Messages: PO file reading supports multi-line obsolete units (#429) (@mbirtwell) +* Messages: Python message extractor respects unicode_literals in __future__ (#427) (@sublee) +* Messages: Roundtrip Language headers (#420) (@kruton) +* Messages: units before obsolete units are no longer erroneously marked obsolete (#452) (@mbirtwell) +* Numbers: `parse_pattern` now preserves the full original pattern (#414) (@jtwang) +* Numbers: Fix float conversion in `extract_operands` (#435) (@akx) +* Plurals: Fix plural forms for Czech and Slovak locales (#373) (@ykshatroff) +* Plurals: More plural form fixes based on Mozilla and CLDR references (#431) (@mshenfield) + + +Internal improvements +~~~~~~~~~~~~~~~~~~~~~ + +* Local times are constructed correctly in tests (#411) (@etanol) +* Miscellaneous small improvements (#437) (@scop) +* Regex flags are extracted from the regex strings (#462) (@singingwolfboy) +* The PO file reader is now a class and has seen some refactoring (#429, #452) (@mbirtwell) + + +Version 2.3.4 +------------- + +(Bugfix release, released on April 22th 2016) + +Bugfixes +~~~~~~~~ + +* CLDR: The lxml library is no longer used for CLDR importing, so it should not cause strange failures either. Thanks to @aronbierbaum for the bug report and @jtwang for the fix. (https://github.com/python-babel/babel/pull/393) +* CLI: Every last single CLI usage regression should now be gone, and both distutils and stand-alone CLIs should work as they have in the past. Thanks to @paxswill and @ajaeger for bug reports. (https://github.com/python-babel/babel/pull/389) + +Version 2.3.3 +------------- + +(Bugfix release, released on April 12th 2016) + +Bugfixes +~~~~~~~~ + +* CLI: Usage regressions that had snuck in between 2.2 and 2.3 should be no more. (https://github.com/python-babel/babel/pull/386) Thanks to @ajaeger, @sebdiem and @jcristovao for bug reports and patches. + +Version 2.3.2 +------------- + +(Bugfix release, released on April 9th 2016) + +Bugfixes +~~~~~~~~ + +* Dates: Period (am/pm) formatting was broken in certain locales (namely zh_TW). Thanks to @jun66j5 for the bug report. (https://github.com/python-babel/babel/issues/378, https://github.com/python-babel/babel/issues/379) + +Version 2.3.1 +------------- + +(Bugfix release because of deployment problems, released on April 8th 2016) + +Version 2.3 +----------- + +(Feature release, released on April 8th 2016) + +Internal improvements +~~~~~~~~~~~~~~~~~~~~~ + +* The CLI frontend and Distutils commands use a shared implementation (https://github.com/python-babel/babel/pull/311) +* PyPy3 is supported (https://github.com/python-babel/babel/pull/343) + +Features +~~~~~~~~ + +* CLDR: Add an API for territory language data (https://github.com/python-babel/babel/pull/315) +* Core: Character order and measurement system data is imported and exposed (https://github.com/python-babel/babel/pull/368) +* Dates: Add an API for time interval formatting (https://github.com/python-babel/babel/pull/316) +* Dates: More pattern formats and lengths are supported (https://github.com/python-babel/babel/pull/347) +* Dates: Period IDs are imported and exposed (https://github.com/python-babel/babel/pull/349) +* Dates: Support for date-time skeleton formats has been added (https://github.com/python-babel/babel/pull/265) +* Dates: Timezone formatting has been improved (https://github.com/python-babel/babel/pull/338) +* Messages: JavaScript extraction now supports dotted names, ES6 template strings and JSX tags (https://github.com/python-babel/babel/pull/332) +* Messages: npgettext is recognized by default (https://github.com/python-babel/babel/pull/341) +* Messages: The CLI learned to accept multiple domains (https://github.com/python-babel/babel/pull/335) +* Messages: The extraction commands now accept filenames in addition to directories (https://github.com/python-babel/babel/pull/324) +* Units: A new API for unit formatting is implemented (https://github.com/python-babel/babel/pull/369) + +Bugfixes +~~~~~~~~ + +* Core: Mixed-case locale IDs work more reliably (https://github.com/python-babel/babel/pull/361) +* Dates: S...S formats work correctly now (https://github.com/python-babel/babel/pull/360) +* Messages: All messages are now sorted correctly if sorting has been specified (https://github.com/python-babel/babel/pull/300) +* Messages: Fix the unexpected behavior caused by catalog header updating (e0e7ef1) (https://github.com/python-babel/babel/pull/320) +* Messages: Gettext operands are now generated correctly (https://github.com/python-babel/babel/pull/295) +* Messages: Message extraction has been taught to detect encodings better (https://github.com/python-babel/babel/pull/274) + +Version 2.2 +----------- + +(Feature release, released on January 2nd 2016) + +Bugfixes +~~~~~~~~ + +* General: Add __hash__ to Locale. (#303) (2aa8074) +* General: Allow files with BOM if they're UTF-8 (#189) (da87edd) +* General: localedata directory is now locale-data (#109) (2d1882e) +* General: odict: Fix pop method (0a9e97e) +* General: Removed uses of datetime.date class from *.dat files (#174) (94f6830) +* Messages: Fix plural selection for Chinese (531f666) +* Messages: Fix typo and add semicolon in plural_forms (5784501) +* Messages: Flatten NullTranslations.files into a list (ad11101) +* Times: FixedOffsetTimezone: fix display of negative offsets (d816803) + +Features +~~~~~~~~ + +* CLDR: Update to CLDR 28 (#292) (9f7f4d0) +* General: Add __copy__ and __deepcopy__ to LazyProxy. (a1cc3f1) +* General: Add official support for Python 3.4 and 3.5 +* General: Improve odict performance by making key search O(1) (6822b7f) +* Locale: Add an ordinal_form property to Locale (#270) (b3f3430) +* Locale: Add support for list formatting (37ce4fa, be6e23d) +* Locale: Check inheritance exceptions first (3ef0d6d) +* Messages: Allow file locations without line numbers (#279) (79bc781) +* Messages: Allow passing a callable to `extract()` (#289) (3f58516) +* Messages: Support 'Language' header field of PO files (#76) (3ce842b) +* Messages: Update catalog headers from templates (e0e7ef1) +* Numbers: Properly load and expose currency format types (#201) (df676ab) +* Numbers: Use cdecimal by default when available (b6169be) +* Numbers: Use the CLDR's suggested number of decimals for format_currency (#139) (201ed50) +* Times: Add format_timedelta(format='narrow') support (edc5eb5) + +Version 2.1 +----------- + +(Bugfix/minor feature release, released on September 25th 2015) + +- Parse and honour the locale inheritance exceptions + (https://github.com/python-babel/babel/issues/97) +- Fix Locale.parse using ``global.dat`` incompatible types + (https://github.com/python-babel/babel/issues/174) +- Fix display of negative offsets in ``FixedOffsetTimezone`` + (https://github.com/python-babel/babel/issues/214) +- Improved odict performance which is used during localization file + build, should improve compilation time for large projects +- Add support for "narrow" format for ``format_timedelta`` +- Add universal wheel support +- Support 'Language' header field in .PO files + (fixes https://github.com/python-babel/babel/issues/76) +- Test suite enhancements (coverage, broken tests fixed, etc) +- Documentation updated + +Version 2.0 +----------- + +(Released on July 27th 2015, codename Second Coming) + +- Added support for looking up currencies that belong to a territory + through the :func:`babel.numbers.get_territory_currencies` + function. +- Improved Python 3 support. +- Fixed some broken tests for timezone behavior. +- Improved various smaller things for dealing with dates. + +Version 1.4 +----------- + +(bugfix release, release date to be decided) + +- Fixed a bug that caused deprecated territory codes not being + converted properly by the subtag resolving. This for instance + showed up when trying to use ``und_UK`` as a language code + which now properly resolves to ``en_GB``. +- Fixed a bug that made it impossible to import the CLDR data + from scratch on windows systems. + Version 1.3 ----------- @@ -9,7 +422,7 @@ Version 1.3 - Fixed a bug in likely-subtag resolving for some common locales. This primarily makes ``zh_CN`` work again which was broken due to how it was defined in the likely subtags combined with - our broken resolving. This fixes #37. + our broken resolving. This fixes :gh:`37`. - Fixed a bug that caused pybabel to break when writing to stdout on Python 3. - Removed a stray print that was causing issues when writing to @@ -44,8 +457,8 @@ Version 1.0 - use tox for testing on different pythons - Added support for the locale plural rules defined by the CLDR. - Added `format_timedelta` function to support localized formatting of - relative times with strings such as "2 days" or "1 month" (ticket #126). -- Fixed negative offset handling of Catalog._set_mime_headers (ticket #165). + relative times with strings such as "2 days" or "1 month" (:trac:`126`). +- Fixed negative offset handling of Catalog._set_mime_headers (:trac:`165`). - Fixed the case where messages containing square brackets would break with an unpack error. - updated to CLDR 23 @@ -53,52 +466,56 @@ Version 1.0 - Fix various typos. - Sort output of list-locales. - Make the POT-Creation-Date of the catalog being updated equal to - POT-Creation-Date of the template used to update (ticket #148). + POT-Creation-Date of the template used to update (:trac:`148`). - Use a more explicit error message if no option or argument (command) is - passed to pybabel (ticket #81). -- Keep the PO-Revision-Date if it is not the default value (ticket #148). + passed to pybabel (:trac:`81`). +- Keep the PO-Revision-Date if it is not the default value (:trac:`148`). - Make --no-wrap work by reworking --width's default and mimic xgettext's - behaviour of always wrapping comments (ticket #145). -- Add --project and --version options for commandline (ticket #173). + behaviour of always wrapping comments (:trac:`145`). +- Add --project and --version options for commandline (:trac:`173`). - Add a __ne__() method to the Local class. - Explicitly sort instead of using sorted() and don't assume ordering (Jython compatibility). - Removed ValueError raising for string formatting message checkers if the - string does not contain any string formattings (ticket #150). -- Fix Serbian plural forms (ticket #213). -- Small speed improvement in format_date() (ticket #216). -- Fix so frontend.CommandLineInterface.run does not accumulate logging - handlers (#227, reported with initial patch by dfraser) -- Fix exception if environment contains an invalid locale setting (#200) -- use cPickle instead of pickle for better performance (#225) + string does not contain any string formattings (:trac:`150`). +- Fix Serbian plural forms (:trac:`213`). +- Small speed improvement in format_date() (:trac:`216`). +- Fix so frontend.CommandLineInterface.run does not accumulate logging + handlers (:trac:`227`, reported with initial patch by dfraser) +- Fix exception if environment contains an invalid locale setting + (:trac:`200`) +- use cPickle instead of pickle for better performance (:trac:`225`) - Only use bankers round algorithm as a tie breaker if there are two nearest - numbers, round as usual if there is only one nearest number (#267, patch by - Martin) -- Allow disabling cache behaviour in LazyProxy (#208, initial patch from Pedro - Algarvio) -- Support for context-aware methods during message extraction (#229, patch - from David Rios) -- "init" and "update" commands support "--no-wrap" option (#289) + numbers, round as usual if there is only one nearest number (:trac:`267`, + patch by Martin) +- Allow disabling cache behaviour in LazyProxy (:trac:`208`, initial patch + from Pedro Algarvio) +- Support for context-aware methods during message extraction (:trac:`229`, + patch from David Rios) +- "init" and "update" commands support "--no-wrap" option (:trac:`289`) - fix formatting of fraction in format_decimal() if the input value is a float - with more than 7 significant digits (#183) -- fix format_date() with datetime parameter (#282, patch from Xavier Morel) -- fix format_decimal() with small Decimal values (#214, patch from George Lund) -- fix handling of messages containing '\\n' (#198) -- handle irregular multi-line msgstr (no "" as first line) gracefully (#171) -- parse_decimal() now returns Decimals not floats, API change (#178) -- no warnings when running setup.py without installed setuptools (#262) + with more than 7 significant digits (:trac:`183`) +- fix format_date() with datetime parameter (:trac:`282`, patch from Xavier + Morel) +- fix format_decimal() with small Decimal values (:trac:`214`, patch from + George Lund) +- fix handling of messages containing '\\n' (:trac:`198`) +- handle irregular multi-line msgstr (no "" as first line) gracefully + (:trac:`171`) +- parse_decimal() now returns Decimals not floats, API change (:trac:`178`) +- no warnings when running setup.py without installed setuptools (:trac:`262`) - modified Locale.__eq__ method so Locales are only equal if all of their attributes (language, territory, script, variant) are equal -- resort to hard-coded message extractors/checkers if pkg_resources is - installed but no egg-info was found (#230) -- format_time() and format_datetime() now accept also floats (#242) +- resort to hard-coded message extractors/checkers if pkg_resources is + installed but no egg-info was found (:trac:`230`) +- format_time() and format_datetime() now accept also floats (:trac:`242`) - add babel.support.NullTranslations class similar to gettext.NullTranslations - but with all of Babel's new gettext methods (#277) -- "init" and "update" commands support "--width" option (#284) -- fix 'input_dirs' option for setuptools integration (#232, initial patch by - Étienne Bersac) -- ensure .mo file header contains the same information as the source .po file - (#199) + but with all of Babel's new gettext methods (:trac:`277`) +- "init" and "update" commands support "--width" option (:trac:`284`) +- fix 'input_dirs' option for setuptools integration (:trac:`232`, initial + patch by Étienne Bersac) +- ensure .mo file header contains the same information as the source .po file + (:trac:`199`) - added support for get_language_name() on the locale objects. - added support for get_territory_name() on the locale objects. - added support for get_script_name() on the locale objects. @@ -122,43 +539,44 @@ Version 0.9.6 - Backport r493-494: documentation typo fixes. - Make the CLDR import script work with Python 2.7. - Fix various typos. -- Fixed Python 2.3 compatibility (ticket #146, #233). +- Fixed Python 2.3 compatibility (:trac:`146`, :trac:`233`). - Sort output of list-locales. - Make the POT-Creation-Date of the catalog being updated equal to - POT-Creation-Date of the template used to update (ticket #148). + POT-Creation-Date of the template used to update (:trac:`148`). - Use a more explicit error message if no option or argument (command) is - passed to pybabel (ticket #81). -- Keep the PO-Revision-Date if it is not the default value (ticket #148). + passed to pybabel (:trac:`81`). +- Keep the PO-Revision-Date if it is not the default value (:trac:`148`). - Make --no-wrap work by reworking --width's default and mimic xgettext's - behaviour of always wrapping comments (ticket #145). -- Fixed negative offset handling of Catalog._set_mime_headers (ticket #165). -- Add --project and --version options for commandline (ticket #173). + behaviour of always wrapping comments (:trac:`145`). +- Fixed negative offset handling of Catalog._set_mime_headers (:trac:`165`). +- Add --project and --version options for commandline (:trac:`173`). - Add a __ne__() method to the Local class. - Explicitly sort instead of using sorted() and don't assume ordering (Python 2.3 and Jython compatibility). - Removed ValueError raising for string formatting message checkers if the - string does not contain any string formattings (ticket #150). -- Fix Serbian plural forms (ticket #213). -- Small speed improvement in format_date() (ticket #216). -- Fix number formatting for locales where CLDR specifies alt or draft - items (ticket #217) -- Fix bad check in format_time (ticket #257, reported with patch and tests by + string does not contain any string formattings (:trac:`150`). +- Fix Serbian plural forms (:trac:`213`). +- Small speed improvement in format_date() (:trac:`216`). +- Fix number formatting for locales where CLDR specifies alt or draft + items (:trac:`217`) +- Fix bad check in format_time (:trac:`257`, reported with patch and tests by jomae) -- Fix so frontend.CommandLineInterface.run does not accumulate logging - handlers (#227, reported with initial patch by dfraser) -- Fix exception if environment contains an invalid locale setting (#200) +- Fix so frontend.CommandLineInterface.run does not accumulate logging + handlers (:trac:`227`, reported with initial patch by dfraser) +- Fix exception if environment contains an invalid locale setting + (:trac:`200`) Version 0.9.5 ------------- -(relased on April 6th 2010) +(released on April 6th 2010) - Fixed the case where messages containing square brackets would break with an unpack error. - Backport of r467: Fuzzy matching regarding plurals should *NOT* be checked against len(message.id) because this is always 2, instead, it's should be - checked against catalog.num_plurals (ticket #212). + checked against catalog.num_plurals (:trac:`212`). Version 0.9.4 @@ -170,17 +588,17 @@ Version 0.9.4 CLDR data are no longer imported, so the symbol code will be used instead. - Fixed quarter support in date formatting. - Fixed a serious memory leak that was introduces by the support for CLDR - aliases in 0.9.3 (ticket #128). + aliases in 0.9.3 (:trac:`128`). - Locale modifiers such as "@euro" are now stripped from locale identifiers - when parsing (ticket #136). + when parsing (:trac:`136`). - The system locales "C" and "POSIX" are now treated as aliases for "en_US_POSIX", for which the CLDR provides the appropriate data. Thanks to Manlio Perillo for the suggestion. -- Fixed JavaScript extraction for regular expression literals (ticket #138) +- Fixed JavaScript extraction for regular expression literals (:trac:`138`) and concatenated strings. - The `Translation` class in `babel.support` can now manage catalogs with different message domains, and exposes the family of `d*gettext` functions - (ticket #137). + (:trac:`137`). Version 0.9.3 @@ -190,11 +608,11 @@ Version 0.9.3 - Fixed invalid message extraction methods causing an UnboundLocalError. - Extraction method specification can now use a dot instead of the colon to - separate module and function name (ticket #105). + separate module and function name (:trac:`105`). - Fixed message catalog compilation for locales with more than two plural - forms (ticket #95). + forms (:trac:`95`). - Fixed compilation of message catalogs for locales with more than two plural - forms where the translations were empty (ticket #97). + forms where the translations were empty (:trac:`97`). - The stripping of the comment tags in comments is optional now and is done for each line in a comment. - Added a JavaScript message extractor. @@ -204,7 +622,7 @@ Version 0.9.3 correct plural forms for a locale as tuple. - Added support for alias definitions in the CLDR data files, meaning that the chance for items missing in certain locales should be greatly reduced - (ticket #68). + (:trac:`68`). Version 0.9.2 @@ -212,15 +630,15 @@ Version 0.9.2 (released on February 4th 2008) -- Fixed catalogs' charset values not being recognized (ticket #66). +- Fixed catalogs' charset values not being recognized (:trac:`66`). - Numerous improvements to the default plural forms. -- Fixed fuzzy matching when updating message catalogs (ticket #82). +- Fixed fuzzy matching when updating message catalogs (:trac:`82`). - Fixed bug in catalog updating, that in some cases pulled in translations from different catalogs based on the same template. - Location lines in PO files do no longer get wrapped at hyphens in file - names (ticket #79). + names (:trac:`79`). - Fixed division by zero error in catalog compilation on empty catalogs - (ticket #60). + (:trac:`60`). Version 0.9.1 @@ -233,7 +651,7 @@ Version 0.9.1 `ngettext`, or vice versa. - Fixed time formatting for 12 am and 12 pm. - Fixed output encoding of the `pybabel --list-locales` command. -- MO files are now written in binary mode on windows (ticket #61). +- MO files are now written in binary mode on windows (:trac:`61`). Version 0.9 @@ -243,23 +661,24 @@ Version 0.9 - The `new_catalog` distutils command has been renamed to `init_catalog` for consistency with the command-line frontend. -- Added compilation of message catalogs to MO files (ticket #21). -- Added updating of message catalogs from POT files (ticket #22). +- Added compilation of message catalogs to MO files (:trac:`21`). +- Added updating of message catalogs from POT files (:trac:`22`). - Support for significant digits in number formatting. - Apply proper "banker's rounding" in number formatting in a cross-platform manner. - The number formatting functions now also work with numbers represented by - Python `Decimal` objects (ticket #53). + Python `Decimal` objects (:trac:`53`). - Added extensible infrastructure for validating translation catalogs. - Fixed the extractor not filtering out messages that didn't validate against - the keyword's specification (ticket #39). + the keyword's specification (:trac:`39`). - Fixed the extractor raising an exception when encountering an empty string msgid. It now emits a warning to stderr. - Numerous Python message extractor fixes: it now handles nested function calls within a gettext function call correctly, uses the correct line number - for multi-line function calls, and other small fixes (tickets #38 and #39). + for multi-line function calls, and other small fixes (tickets :trac:`38` and + :trac:`39`). - Improved support for detecting Python string formatting fields in message - strings (ticket #57). + strings (:trac:`57`). - CLDR upgraded to the 1.5 release. - Improved timezone formatting. - Implemented scientific number formatting. @@ -281,21 +700,21 @@ Version 0.8.1 that way. - The character set specified in PO template files is now respected when creating new catalog files based on that template. This allows the use of - characters outside the ASCII range in POT files (ticket #17). + characters outside the ASCII range in POT files (:trac:`17`). - The default ordering of messages in generated POT files, which is based on the order those messages are found when walking the source tree, is no longer subject to differences between platforms; directory and file names are now always sorted alphabetically. - The Python message extractor now respects the special encoding comment to be - able to handle files containing non-ASCII characters (ticket #23). + able to handle files containing non-ASCII characters (:trac:`23`). - Added ``N_`` (gettext noop) to the extractor's default keywords. - Made locale string parsing more robust, and also take the script part into - account (ticket #27). + account (:trac:`27`). - Added a function to list all locales for which locale data is available. - Added a command-line option to the `pybabel` command which prints out all - available locales (ticket #24). + available locales (:trac:`24`). - The name of the command-line script has been changed from just `babel` to - `pybabel` to avoid a conflict with the OpenBabel project (ticket #34). + `pybabel` to avoid a conflict with the OpenBabel project (:trac:`34`). Version 0.8 diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md new file mode 100644 index 000000000..079ef06b2 --- /dev/null +++ b/CONTRIBUTING.md @@ -0,0 +1,55 @@ +# Babel Contribution Guidelines + +Welcome to Babel! These guidelines will give you a short overview over how we +handle issues and PRs in this repository. Note that they are preliminary and +still need proper phrasing - if you'd like to help - be sure to make a PR. + +Please know that we do appreciate all contributions - bug reports as well as +Pull Requests. + +## Filing Issues + +When filing an issue, please use this template: + +``` +# Overview Description + +# Steps to Reproduce + +1. +2. +3. + +# Actual Results + +# Expected Results + +# Reproducibility + +# Additional Information: + +``` + +## PR Merge Criteria + +For a PR to be merged, the following statements must hold true: + +- All CI services pass. (Windows build, linux build, sufficient test coverage.) +- All commits must have been reviewed and approved by a babel maintainer who is + not the author of the PR. Commits shall comply to the "Good Commits" standards + outlined below. + +To begin contributing have a look at the open [easy issues](https://github.com/python-babel/babel/issues?q=is%3Aopen+is%3Aissue+label%3Adifficulty%2Flow) +which could be fixed. + +## Correcting PRs + +Rebasing PRs is preferred over merging master into the source branches again +and again cluttering our history. If a reviewer has suggestions, the commit +shall be amended so the history is not cluttered by "fixup commits". + +## Writing Good Commits + +Please see +https://api.coala.io/en/latest/Developers/Writing_Good_Commits.html +for guidelines on how to write good commits and proper commit messages. diff --git a/LICENSE b/LICENSE index 1f1f55b60..693e1a187 100644 --- a/LICENSE +++ b/LICENSE @@ -1,4 +1,4 @@ -Copyright (C) 2013 by the Babel Team, see AUTHORS for more information. +Copyright (c) 2013-2021 by the Babel Team, see AUTHORS for more information. All rights reserved. diff --git a/MANIFEST.in b/MANIFEST.in index a53a51f14..6c1e7aff4 100644 --- a/MANIFEST.in +++ b/MANIFEST.in @@ -1,7 +1,7 @@ include Makefile CHANGES LICENSE AUTHORS include conftest.py tox.ini include babel/global.dat -include babel/localedata/*.dat +include babel/locale-data/*.dat recursive-include docs * recursive-exclude docs/_build * include scripts/* diff --git a/Makefile b/Makefile index f2cdc967c..eafaeb48d 100644 --- a/Makefile +++ b/Makefile @@ -1,5 +1,8 @@ test: import-cldr - @py.test tests + @PYTHONWARNINGS=default python ${PYTHON_TEST_FLAGS} -m pytest + +test-cov: import-cldr + @PYTHONWARNINGS=default python ${PYTHON_TEST_FLAGS} -m pytest --cov=babel test-env: @virtualenv test-env @@ -18,8 +21,8 @@ import-cldr: @python scripts/download_import_cldr.py clean-cldr: - @rm babel/localedata/*.dat - @rm babel/global.dat + @rm -f babel/locale-data/*.dat + @rm -f babel/global.dat clean-pyc: @find . -name '*.pyc' -exec rm {} \; diff --git a/README b/README deleted file mode 100644 index 8e783e1d0..000000000 --- a/README +++ /dev/null @@ -1,12 +0,0 @@ -About Babel -=========== - -Babel is a Python library that provides an integrated collection of -utilities that assist with internationalizing and localizing Python -applications (in particular web-based applications.) - -Details can be found in the HTML files in the `docs` folder. - -For more information please visit the Babel web site: - - diff --git a/README.rst b/README.rst new file mode 100644 index 000000000..c708f9da6 --- /dev/null +++ b/README.rst @@ -0,0 +1,24 @@ +About Babel +=========== + +Babel is a Python library that provides an integrated collection of +utilities that assist with internationalizing and localizing Python +applications (in particular web-based applications.) + +Details can be found in the HTML files in the ``docs`` folder. + +For more information please visit the Babel web site: + +http://babel.pocoo.org/ + +Join the chat at https://gitter.im/python-babel/babel + +Contributing to Babel +===================== + +If you want to contribute code to Babel, please take a look at our +`CONTRIBUTING.md `__. + +If you know your way around Babels codebase a bit and like to help +further, we would appreciate any help in reviewing pull requests. Please +contact us at https://gitter.im/python-babel/babel if you're interested! diff --git a/babel/__init__.py b/babel/__init__.py index dd9f17e04..3e20e4bd1 100644 --- a/babel/__init__.py +++ b/babel/__init__.py @@ -13,12 +13,12 @@ access to various locale display names, localized number and date formatting, etc. - :copyright: (c) 2013 by the Babel Team. + :copyright: (c) 2013-2021 by the Babel Team. :license: BSD, see LICENSE for more details. """ from babel.core import UnknownLocaleError, Locale, default_locale, \ - negotiate_locale, parse_locale, get_locale_identifier + negotiate_locale, parse_locale, get_locale_identifier -__version__ = '1.3' +__version__ = '2.9.1' diff --git a/babel/_compat.py b/babel/_compat.py index 86096daa6..11b4d7a6b 100644 --- a/babel/_compat.py +++ b/babel/_compat.py @@ -1,4 +1,5 @@ import sys +import array PY2 = sys.version_info[0] == 2 @@ -9,9 +10,9 @@ text_type = str string_types = (str,) integer_types = (int, ) - unichr = chr text_to_native = lambda s, enc: s + unichr = chr iterkeys = lambda d: iter(d.keys()) itervalues = lambda d: iter(d.values()) @@ -26,6 +27,9 @@ cmp = lambda a, b: (a > b) - (a < b) + array_tobytes = array.array.tobytes + from collections import abc + else: text_type = unicode string_types = (str, unicode) @@ -42,10 +46,34 @@ from StringIO import StringIO import cPickle as pickle - from itertools import izip, imap + from itertools import imap + from itertools import izip range_type = xrange cmp = cmp + array_tobytes = array.array.tostring + import collections as abc number_types = integer_types + (float,) + + +def force_text(s, encoding='utf-8', errors='strict'): + if isinstance(s, text_type): + return s + if isinstance(s, bytes): + return s.decode(encoding, errors) + return text_type(s) + + +# +# Since Python 3.3, a fast decimal implementation is already included in the +# standard library. Otherwise use cdecimal when available +# +if sys.version_info[:2] >= (3, 3): + import decimal +else: + try: + import cdecimal as decimal + except ImportError: + import decimal diff --git a/babel/core.py b/babel/core.py index 6e6e6d619..a323a7295 100644 --- a/babel/core.py +++ b/babel/core.py @@ -5,7 +5,7 @@ Core locale representation and locale data access. - :copyright: (c) 2013 by the Babel Team. + :copyright: (c) 2013-2021 by the Babel Team. :license: BSD, see LICENSE for more details. """ @@ -13,12 +13,14 @@ from babel import localedata from babel._compat import pickle, string_types +from babel.plural import PluralRule __all__ = ['UnknownLocaleError', 'Locale', 'default_locale', 'negotiate_locale', 'parse_locale'] _global_data = None +_default_plural_rule = PluralRule({}) def _raise_no_data_error(): @@ -37,10 +39,29 @@ def get_global(key): information independent of individual locales. >>> get_global('zone_aliases')['UTC'] - u'Etc/GMT' + u'Etc/UTC' >>> get_global('zone_territories')['Europe/Berlin'] u'DE' + The keys available are: + + - ``all_currencies`` + - ``currency_fractions`` + - ``language_aliases`` + - ``likely_subtags`` + - ``parent_exceptions`` + - ``script_aliases`` + - ``territory_aliases`` + - ``territory_currencies`` + - ``territory_languages`` + - ``territory_zones`` + - ``variant_aliases`` + - ``windows_zone_mapping`` + - ``zone_aliases`` + - ``zone_territories`` + + .. note:: The internal structure of the data may change between versions. + .. versionadded:: 0.9 :param key: the data key @@ -51,11 +72,8 @@ def get_global(key): filename = os.path.join(dirname, 'global.dat') if not os.path.isfile(filename): _raise_no_data_error() - fileobj = open(filename, 'rb') - try: + with open(filename, 'rb') as fileobj: _global_data = pickle.load(fileobj) - finally: - fileobj.close() return _global_data.get(key, {}) @@ -111,10 +129,10 @@ class Locale(object): If a locale is requested for which no locale data is available, an `UnknownLocaleError` is raised: - >>> Locale.parse('en_DE') + >>> Locale.parse('en_XX') Traceback (most recent call last): ... - UnknownLocaleError: unknown locale 'en_DE' + UnknownLocaleError: unknown locale 'en_XX' For more information see :rfc:`3066`. """ @@ -245,7 +263,7 @@ def parse(cls, identifier, sep='_', resolve_likely_subtags=True): elif isinstance(identifier, Locale): return identifier elif not isinstance(identifier, string_types): - raise TypeError('Unxpected value for identifier: %r' % (identifier,)) + raise TypeError('Unexpected value for identifier: %r' % (identifier,)) parts = parse_locale(identifier, sep=sep) input_id = get_locale_identifier(parts) @@ -282,7 +300,7 @@ def _try_load_reducing(parts): language, territory, script, variant = parts language = get_global('language_aliases').get(language, language) - territory = get_global('territory_aliases').get(territory, territory) + territory = get_global('territory_aliases').get(territory, (territory,))[0] script = get_global('script_aliases').get(script, script) variant = get_global('variant_aliases').get(variant, variant) @@ -324,6 +342,9 @@ def __eq__(self, other): def __ne__(self, other): return not self.__eq__(other) + def __hash__(self): + return hash((self.language, self.territory, self.script, self.variant)) + def __repr__(self): parameters = [''] for key in ('territory', 'script', 'variant'): @@ -358,7 +379,7 @@ def get_display_name(self, locale=None): locale = self locale = Locale.parse(locale) retval = locale.languages.get(self.language) - if self.territory or self.script or self.variant: + if retval and (self.territory or self.script or self.variant): details = [] if self.script: details.append(locale.scripts.get(self.script)) @@ -430,8 +451,8 @@ def get_script_name(self, locale=None): script_name = property(get_script_name, doc="""\ The localized script name of the locale if available. - >>> Locale('ms', 'SG', script='Latn').script_name - u'Latin' + >>> Locale('sr', 'ME', script='Latn').script_name + u'latinica' """) @property @@ -446,7 +467,7 @@ def english_name(self): :type: `unicode`""" return self.get_display_name(Locale('en')) - #{ General Locale Display Names + # { General Locale Display Names @property def languages(self): @@ -493,7 +514,7 @@ def variants(self): """ return self._data['variants'] - #{ Number Formatting + # { Number Formatting @property def currencies(self): @@ -524,6 +545,9 @@ def currency_symbols(self): def number_symbols(self): """Symbols used in number formatting. + .. note:: The format of the value returned may change between + Babel versions. + >>> Locale('fr', 'FR').number_symbols['decimal'] u',' """ @@ -533,6 +557,9 @@ def number_symbols(self): def decimal_formats(self): """Locale patterns for decimal number formatting. + .. note:: The format of the value returned may change between + Babel versions. + >>> Locale('en', 'US').decimal_formats[None] """ @@ -542,8 +569,13 @@ def decimal_formats(self): def currency_formats(self): """Locale patterns for currency number formatting. - >>> print Locale('en', 'US').currency_formats[None] + .. note:: The format of the value returned may change between + Babel versions. + + >>> Locale('en', 'US').currency_formats['standard'] + >>> Locale('en', 'US').currency_formats['accounting'] + """ return self._data['currency_formats'] @@ -551,6 +583,9 @@ def currency_formats(self): def percent_formats(self): """Locale patterns for percent number formatting. + .. note:: The format of the value returned may change between + Babel versions. + >>> Locale('en', 'US').percent_formats[None] """ @@ -560,12 +595,15 @@ def percent_formats(self): def scientific_formats(self): """Locale patterns for scientific number formatting. + .. note:: The format of the value returned may change between + Babel versions. + >>> Locale('en', 'US').scientific_formats[None] """ return self._data['scientific_formats'] - #{ Calendar Information and Date Formatting + # { Calendar Information and Date Formatting @property def periods(self): @@ -574,7 +612,24 @@ def periods(self): >>> Locale('en', 'US').periods['am'] u'AM' """ - return self._data['periods'] + try: + return self._data['day_periods']['stand-alone']['wide'] + except KeyError: + return {} + + @property + def day_periods(self): + """Locale display names for various day periods (not necessarily only AM/PM). + + These are not meant to be used without the relevant `day_period_rules`. + """ + return self._data['day_periods'] + + @property + def day_period_rules(self): + """Day period rules for the locale. Used by `get_period_id`. + """ + return self._data.get('day_period_rules', {}) @property def days(self): @@ -607,6 +662,9 @@ def quarters(self): def eras(self): """Locale display names for eras. + .. note:: The format of the value returned may change between + Babel versions. + >>> Locale('en', 'US').eras['wide'][1] u'Anno Domini' >>> Locale('en', 'US').eras['abbreviated'][0] @@ -618,6 +676,9 @@ def eras(self): def time_zones(self): """Locale display names for time zones. + .. note:: The format of the value returned may change between + Babel versions. + >>> Locale('en', 'US').time_zones['Europe/London']['long']['daylight'] u'British Summer Time' >>> Locale('en', 'US').time_zones['America/St_Johns']['city'] @@ -632,6 +693,9 @@ def meta_zones(self): Meta time zones are basically groups of different Olson time zones that have the same GMT offset and daylight savings time. + .. note:: The format of the value returned may change between + Babel versions. + >>> Locale('en', 'US').meta_zones['Europe_Central']['long']['daylight'] u'Central European Summer Time' @@ -643,6 +707,9 @@ def meta_zones(self): def zone_formats(self): """Patterns related to the formatting of time zones. + .. note:: The format of the value returned may change between + Babel versions. + >>> Locale('en', 'US').zone_formats['fallback'] u'%(1)s (%(0)s)' >>> Locale('pt', 'BR').zone_formats['region'] @@ -695,6 +762,9 @@ def min_week_days(self): def date_formats(self): """Locale patterns for date formatting. + .. note:: The format of the value returned may change between + Babel versions. + >>> Locale('en', 'US').date_formats['short'] >>> Locale('fr', 'FR').date_formats['long'] @@ -706,6 +776,9 @@ def date_formats(self): def time_formats(self): """Locale patterns for time formatting. + .. note:: The format of the value returned may change between + Babel versions. + >>> Locale('en', 'US').time_formats['short'] >>> Locale('fr', 'FR').time_formats['long'] @@ -717,13 +790,51 @@ def time_formats(self): def datetime_formats(self): """Locale patterns for datetime formatting. + .. note:: The format of the value returned may change between + Babel versions. + >>> Locale('en').datetime_formats['full'] u"{1} 'at' {0}" >>> Locale('th').datetime_formats['medium'] - u'{1}, {0}' + u'{1} {0}' """ return self._data['datetime_formats'] + @property + def datetime_skeletons(self): + """Locale patterns for formatting parts of a datetime. + + >>> Locale('en').datetime_skeletons['MEd'] + + >>> Locale('fr').datetime_skeletons['MEd'] + + >>> Locale('fr').datetime_skeletons['H'] + + """ + return self._data['datetime_skeletons'] + + @property + def interval_formats(self): + """Locale patterns for interval formatting. + + .. note:: The format of the value returned may change between + Babel versions. + + How to format date intervals in Finnish when the day is the + smallest changing component: + + >>> Locale('fi_FI').interval_formats['MEd']['d'] + [u'E d. \u2013 ', u'E d.M.'] + + .. seealso:: + + The primary API to use this data is :py:func:`babel.dates.format_interval`. + + + :rtype: dict[str, dict[str, list[str]]] + """ + return self._data['interval_formats'] + @property def plural_form(self): """Plural rules for the locale. @@ -737,7 +848,88 @@ def plural_form(self): >>> Locale('ru').plural_form(100) 'many' """ - return self._data['plural_form'] + return self._data.get('plural_form', _default_plural_rule) + + @property + def list_patterns(self): + """Patterns for generating lists + + .. note:: The format of the value returned may change between + Babel versions. + + >>> Locale('en').list_patterns['standard']['start'] + u'{0}, {1}' + >>> Locale('en').list_patterns['standard']['end'] + u'{0}, and {1}' + >>> Locale('en_GB').list_patterns['standard']['end'] + u'{0} and {1}' + """ + return self._data['list_patterns'] + + @property + def ordinal_form(self): + """Plural rules for the locale. + + >>> Locale('en').ordinal_form(1) + 'one' + >>> Locale('en').ordinal_form(2) + 'two' + >>> Locale('en').ordinal_form(3) + 'few' + >>> Locale('fr').ordinal_form(2) + 'other' + >>> Locale('ru').ordinal_form(100) + 'other' + """ + return self._data.get('ordinal_form', _default_plural_rule) + + @property + def measurement_systems(self): + """Localized names for various measurement systems. + + >>> Locale('fr', 'FR').measurement_systems['US'] + u'am\\xe9ricain' + >>> Locale('en', 'US').measurement_systems['US'] + u'US' + + """ + return self._data['measurement_systems'] + + @property + def character_order(self): + """The text direction for the language. + + >>> Locale('de', 'DE').character_order + 'left-to-right' + >>> Locale('ar', 'SA').character_order + 'right-to-left' + """ + return self._data['character_order'] + + @property + def text_direction(self): + """The text direction for the language in CSS short-hand form. + + >>> Locale('de', 'DE').text_direction + 'ltr' + >>> Locale('ar', 'SA').text_direction + 'rtl' + """ + return ''.join(word[0] for word in self.character_order.split('-')) + + @property + def unit_display_names(self): + """Display names for units of measurement. + + .. seealso:: + + You may want to use :py:func:`babel.units.get_unit_name` instead. + + .. note:: The format of the value returned may change between + Babel versions. + + """ + return self._data['unit_display_names'] def default_locale(category=None, aliases=LOCALE_ALIASES): @@ -775,7 +967,7 @@ def default_locale(category=None, aliases=LOCALE_ALIASES): # the LANGUAGE variable may contain a colon-separated list of # language codes; we just pick the language on the list locale = locale.split(':')[0] - if locale in ('C', 'POSIX'): + if locale.split('.')[0] in ('C', 'POSIX'): locale = 'en_US_POSIX' elif aliases and locale in aliases: locale = aliases[locale] @@ -926,7 +1118,7 @@ def parse_locale(identifier, sep='_'): def get_locale_identifier(tup, sep='_'): """The reverse of :func:`parse_locale`. It creates a locale identifier out of a ``(language, territory, script, variant)`` tuple. Items can be set to - ``None`` and trailing ``None``\s can also be left out of the tuple. + ``None`` and trailing ``None``\\s can also be left out of the tuple. >>> get_locale_identifier(('de', 'DE', None, '1999')) 'de_DE_1999' diff --git a/babel/dates.py b/babel/dates.py index 72674e8aa..75e8f3501 100644 --- a/babel/dates.py +++ b/babel/dates.py @@ -12,13 +12,14 @@ * ``LC_ALL``, and * ``LANG`` - :copyright: (c) 2013 by the Babel Team. + :copyright: (c) 2013-2021 by the Babel Team. :license: BSD, see LICENSE for more details. """ from __future__ import division import re +import warnings import pytz as _pytz from datetime import date, datetime, time, timedelta @@ -26,7 +27,16 @@ from babel.core import default_locale, get_global, Locale from babel.util import UTC, LOCALTZ -from babel._compat import string_types, integer_types, number_types +from babel._compat import string_types, integer_types, number_types, PY2 + +# "If a given short metazone form is known NOT to be understood in a given +# locale and the parent locale has this value such that it would normally +# be inherited, the inheritance of this value can be explicitly disabled by +# use of the 'no inheritance marker' as the value, which is 3 simultaneous [sic] +# empty set characters ( U+2205 )." +# - https://www.unicode.org/reports/tr35/tr35-dates.html#Metazone_Names + +NO_INHERITANCE_MARKER = u'\u2205\u2205\u2205' LC_TIME = default_locale('LC_TIME') @@ -37,6 +47,147 @@ time_ = time +def _get_dt_and_tzinfo(dt_or_tzinfo): + """ + Parse a `dt_or_tzinfo` value into a datetime and a tzinfo. + + See the docs for this function's callers for semantics. + + :rtype: tuple[datetime, tzinfo] + """ + if dt_or_tzinfo is None: + dt = datetime.now() + tzinfo = LOCALTZ + elif isinstance(dt_or_tzinfo, string_types): + dt = None + tzinfo = get_timezone(dt_or_tzinfo) + elif isinstance(dt_or_tzinfo, integer_types): + dt = None + tzinfo = UTC + elif isinstance(dt_or_tzinfo, (datetime, time)): + dt = _get_datetime(dt_or_tzinfo) + if dt.tzinfo is not None: + tzinfo = dt.tzinfo + else: + tzinfo = UTC + else: + dt = None + tzinfo = dt_or_tzinfo + return dt, tzinfo + + +def _get_tz_name(dt_or_tzinfo): + """ + Get the timezone name out of a time, datetime, or tzinfo object. + + :rtype: str + """ + dt, tzinfo = _get_dt_and_tzinfo(dt_or_tzinfo) + if hasattr(tzinfo, 'zone'): # pytz object + return tzinfo.zone + elif hasattr(tzinfo, 'key') and tzinfo.key is not None: # ZoneInfo object + return tzinfo.key + else: + return tzinfo.tzname(dt or datetime.utcnow()) + + +def _get_datetime(instant): + """ + Get a datetime out of an "instant" (date, time, datetime, number). + + .. warning:: The return values of this function may depend on the system clock. + + If the instant is None, the current moment is used. + If the instant is a time, it's augmented with today's date. + + Dates are converted to naive datetimes with midnight as the time component. + + >>> _get_datetime(date(2015, 1, 1)) + datetime.datetime(2015, 1, 1, 0, 0) + + UNIX timestamps are converted to datetimes. + + >>> _get_datetime(1400000000) + datetime.datetime(2014, 5, 13, 16, 53, 20) + + Other values are passed through as-is. + + >>> x = datetime(2015, 1, 1) + >>> _get_datetime(x) is x + True + + :param instant: date, time, datetime, integer, float or None + :type instant: date|time|datetime|int|float|None + :return: a datetime + :rtype: datetime + """ + if instant is None: + return datetime_.utcnow() + elif isinstance(instant, integer_types) or isinstance(instant, float): + return datetime_.utcfromtimestamp(instant) + elif isinstance(instant, time): + return datetime_.combine(date.today(), instant) + elif isinstance(instant, date) and not isinstance(instant, datetime): + return datetime_.combine(instant, time()) + # TODO (3.x): Add an assertion/type check for this fallthrough branch: + return instant + + +def _ensure_datetime_tzinfo(datetime, tzinfo=None): + """ + Ensure the datetime passed has an attached tzinfo. + + If the datetime is tz-naive to begin with, UTC is attached. + + If a tzinfo is passed in, the datetime is normalized to that timezone. + + >>> _ensure_datetime_tzinfo(datetime(2015, 1, 1)).tzinfo.zone + 'UTC' + + >>> tz = get_timezone("Europe/Stockholm") + >>> _ensure_datetime_tzinfo(datetime(2015, 1, 1, 13, 15, tzinfo=UTC), tzinfo=tz).hour + 14 + + :param datetime: Datetime to augment. + :param tzinfo: Optional tznfo. + :return: datetime with tzinfo + :rtype: datetime + """ + if datetime.tzinfo is None: + datetime = datetime.replace(tzinfo=UTC) + if tzinfo is not None: + datetime = datetime.astimezone(get_timezone(tzinfo)) + if hasattr(tzinfo, 'normalize'): # pytz + datetime = tzinfo.normalize(datetime) + return datetime + + +def _get_time(time, tzinfo=None): + """ + Get a timezoned time from a given instant. + + .. warning:: The return values of this function may depend on the system clock. + + :param time: time, datetime or None + :rtype: time + """ + if time is None: + time = datetime.utcnow() + elif isinstance(time, number_types): + time = datetime.utcfromtimestamp(time) + if time.tzinfo is None: + time = time.replace(tzinfo=UTC) + if isinstance(time, datetime): + if tzinfo is not None: + time = time.astimezone(tzinfo) + if hasattr(tzinfo, 'normalize'): # pytz + time = tzinfo.normalize(time) + time = time.timetz() + elif tzinfo is not None: + time = time.replace(tzinfo=tzinfo) + return time + + def get_timezone(zone=None): """Looks up a timezone by name and returns it. The timezone object returned comes from ``pytz`` and corresponds to the `tzinfo` interface and @@ -77,10 +228,7 @@ def get_next_timezone_transition(zone=None, dt=None): If not given the current time is assumed. """ zone = get_timezone(zone) - if dt is None: - dt = datetime.utcnow() - else: - dt = dt.replace(tzinfo=None) + dt = _get_datetime(dt).replace(tzinfo=None) if not hasattr(zone, '_utc_transition_times'): raise TypeError('Given timezone does not have UTC transition ' @@ -149,15 +297,17 @@ def __repr__(self): ) -def get_period_names(locale=LC_TIME): +def get_period_names(width='wide', context='stand-alone', locale=LC_TIME): """Return the names for day periods (AM/PM) used by the locale. >>> get_period_names(locale='en_US')['am'] u'AM' + :param width: the width to use, one of "abbreviated", "narrow", or "wide" + :param context: the context, either "format" or "stand-alone" :param locale: the `Locale` object, or a locale string """ - return Locale.parse(locale).periods + return Locale.parse(locale).day_periods[context][width] def get_day_names(width='wide', context='format', locale=LC_TIME): @@ -165,12 +315,14 @@ def get_day_names(width='wide', context='format', locale=LC_TIME): >>> get_day_names('wide', locale='en_US')[1] u'Tuesday' + >>> get_day_names('short', locale='en_US')[1] + u'Tu' >>> get_day_names('abbreviated', locale='es')[1] - u'mar' + u'mar.' >>> get_day_names('narrow', context='stand-alone', locale='de_DE')[1] u'D' - :param width: the width to use, one of "wide", "abbreviated", or "narrow" + :param width: the width to use, one of "wide", "abbreviated", "short" or "narrow" :param context: the context, either "format" or "stand-alone" :param locale: the `Locale` object, or a locale string """ @@ -183,7 +335,7 @@ def get_month_names(width='wide', context='format', locale=LC_TIME): >>> get_month_names('wide', locale='en_US')[1] u'January' >>> get_month_names('abbreviated', locale='es')[1] - u'ene' + u'ene.' >>> get_month_names('narrow', context='stand-alone', locale='de_DE')[1] u'J' @@ -201,6 +353,8 @@ def get_quarter_names(width='wide', context='format', locale=LC_TIME): u'1st quarter' >>> get_quarter_names('abbreviated', locale='de_DE')[1] u'Q1' + >>> get_quarter_names('narrow', locale='de_DE')[1] + u'1' :param width: the width to use, one of "wide", "abbreviated", or "narrow" :param context: the context, either "format" or "stand-alone" @@ -272,61 +426,73 @@ def get_time_format(format='medium', locale=LC_TIME): return Locale.parse(locale).time_formats[format] -def get_timezone_gmt(datetime=None, width='long', locale=LC_TIME): +def get_timezone_gmt(datetime=None, width='long', locale=LC_TIME, return_z=False): """Return the timezone associated with the given `datetime` object formatted as string indicating the offset from GMT. >>> dt = datetime(2007, 4, 1, 15, 30) >>> get_timezone_gmt(dt, locale='en') u'GMT+00:00' - + >>> get_timezone_gmt(dt, locale='en', return_z=True) + 'Z' + >>> get_timezone_gmt(dt, locale='en', width='iso8601_short') + u'+00' >>> tz = get_timezone('America/Los_Angeles') - >>> dt = datetime(2007, 4, 1, 15, 30, tzinfo=tz) + >>> dt = tz.localize(datetime(2007, 4, 1, 15, 30)) >>> get_timezone_gmt(dt, locale='en') - u'GMT-08:00' + u'GMT-07:00' >>> get_timezone_gmt(dt, 'short', locale='en') - u'-0800' + u'-0700' + >>> get_timezone_gmt(dt, locale='en', width='iso8601_short') + u'-07' The long format depends on the locale, for example in France the acronym UTC string is used instead of GMT: >>> get_timezone_gmt(dt, 'long', locale='fr_FR') - u'UTC-08:00' + u'UTC-07:00' .. versionadded:: 0.9 :param datetime: the ``datetime`` object; if `None`, the current date and time in UTC is used - :param width: either "long" or "short" + :param width: either "long" or "short" or "iso8601" or "iso8601_short" :param locale: the `Locale` object, or a locale string + :param return_z: True or False; Function returns indicator "Z" + when local time offset is 0 """ - if datetime is None: - datetime = datetime_.utcnow() - elif isinstance(datetime, integer_types): - datetime = datetime_.utcfromtimestamp(datetime).time() - if datetime.tzinfo is None: - datetime = datetime.replace(tzinfo=UTC) + datetime = _ensure_datetime_tzinfo(_get_datetime(datetime)) locale = Locale.parse(locale) offset = datetime.tzinfo.utcoffset(datetime) seconds = offset.days * 24 * 60 * 60 + offset.seconds hours, seconds = divmod(seconds, 3600) - if width == 'short': + if return_z and hours == 0 and seconds == 0: + return 'Z' + elif seconds == 0 and width == 'iso8601_short': + return u'%+03d' % hours + elif width == 'short' or width == 'iso8601_short': pattern = u'%+03d%02d' + elif width == 'iso8601': + pattern = u'%+03d:%02d' else: pattern = locale.zone_formats['gmt'] % '%+03d:%02d' return pattern % (hours, seconds // 60) -def get_timezone_location(dt_or_tzinfo=None, locale=LC_TIME): - """Return a representation of the given timezone using "location format". +def get_timezone_location(dt_or_tzinfo=None, locale=LC_TIME, return_city=False): + u"""Return a representation of the given timezone using "location format". The result depends on both the local display name of the country and the city associated with the time zone: >>> tz = get_timezone('America/St_Johns') - >>> get_timezone_location(tz, locale='de_DE') - u"Kanada (St. John's) Zeit" + >>> print(get_timezone_location(tz, locale='de_DE')) + Kanada (St. John’s) Zeit + >>> print(get_timezone_location(tz, locale='en')) + Canada (St. John’s) Time + >>> print(get_timezone_location(tz, locale='en', return_city=True)) + St. John’s >>> tz = get_timezone('America/Mexico_City') >>> get_timezone_location(tz, locale='de_DE') u'Mexiko (Mexiko-Stadt) Zeit' @@ -344,32 +510,14 @@ def get_timezone_location(dt_or_tzinfo=None, locale=LC_TIME): the timezone; if `None`, the current date and time in UTC is assumed :param locale: the `Locale` object, or a locale string + :param return_city: True or False, if True then return exemplar city (location) + for the time zone :return: the localized timezone name using location format + """ - if dt_or_tzinfo is None: - dt = datetime.now() - tzinfo = LOCALTZ - elif isinstance(dt_or_tzinfo, string_types): - dt = None - tzinfo = get_timezone(dt_or_tzinfo) - elif isinstance(dt_or_tzinfo, integer_types): - dt = None - tzinfo = UTC - elif isinstance(dt_or_tzinfo, (datetime, time)): - dt = dt_or_tzinfo - if dt.tzinfo is not None: - tzinfo = dt.tzinfo - else: - tzinfo = UTC - else: - dt = None - tzinfo = dt_or_tzinfo locale = Locale.parse(locale) - if hasattr(tzinfo, 'zone'): - zone = tzinfo.zone - else: - zone = tzinfo.tzname(dt or datetime.utcnow()) + zone = _get_tz_name(dt_or_tzinfo) # Get the canonical time-zone code zone = get_global('zone_aliases').get(zone, zone) @@ -381,10 +529,10 @@ def get_timezone_location(dt_or_tzinfo=None, locale=LC_TIME): region_format = locale.zone_formats['region'] territory = get_global('zone_territories').get(zone) if territory not in locale.territories: - territory = 'ZZ' # invalid/unknown + territory = 'ZZ' # invalid/unknown territory_name = locale.territories[territory] - if territory and len(get_global('territory_zones').get(territory, [])) == 1: - return region_format % (territory_name) + if not return_city and territory and len(get_global('territory_zones').get(territory, [])) == 1: + return region_format % territory_name # Otherwise, include the city in the output fallback_format = locale.zone_formats['fallback'] @@ -400,6 +548,8 @@ def get_timezone_location(dt_or_tzinfo=None, locale=LC_TIME): else: city_name = zone.replace('_', ' ') + if return_city: + return city_name return region_format % (fallback_format % { '0': city_name, '1': territory_name @@ -407,13 +557,15 @@ def get_timezone_location(dt_or_tzinfo=None, locale=LC_TIME): def get_timezone_name(dt_or_tzinfo=None, width='long', uncommon=False, - locale=LC_TIME, zone_variant=None): + locale=LC_TIME, zone_variant=None, return_zone=False): r"""Return the localized display name for the given timezone. The timezone may be specified using a ``datetime`` or `tzinfo` object. >>> dt = time(15, 30, tzinfo=get_timezone('America/Los_Angeles')) >>> get_timezone_name(dt, locale='en_US') u'Pacific Standard Time' + >>> get_timezone_name(dt, locale='en_US', return_zone=True) + 'America/Los_Angeles' >>> get_timezone_name(dt, width='short', locale='en_US') u'PST' @@ -451,7 +603,7 @@ def get_timezone_name(dt_or_tzinfo=None, width='long', uncommon=False, format. For more information see `LDML Appendix J: Time Zone Display Names - `_ + `_ .. versionadded:: 0.9 @@ -472,31 +624,13 @@ def get_timezone_name(dt_or_tzinfo=None, width='long', uncommon=False, values are valid: ``'generic'``, ``'daylight'`` and ``'standard'``. :param locale: the `Locale` object, or a locale string + :param return_zone: True or False. If true then function + returns long time zone ID """ - if dt_or_tzinfo is None: - dt = datetime.now() - tzinfo = LOCALTZ - elif isinstance(dt_or_tzinfo, string_types): - dt = None - tzinfo = get_timezone(dt_or_tzinfo) - elif isinstance(dt_or_tzinfo, integer_types): - dt = None - tzinfo = UTC - elif isinstance(dt_or_tzinfo, (datetime, time)): - dt = dt_or_tzinfo - if dt.tzinfo is not None: - tzinfo = dt.tzinfo - else: - tzinfo = UTC - else: - dt = None - tzinfo = dt_or_tzinfo + dt, tzinfo = _get_dt_and_tzinfo(dt_or_tzinfo) locale = Locale.parse(locale) - if hasattr(tzinfo, 'zone'): - zone = tzinfo.zone - else: - zone = tzinfo.tzname(dt) + zone = _get_tz_name(dt_or_tzinfo) if zone_variant is None: if dt is None: @@ -513,7 +647,8 @@ def get_timezone_name(dt_or_tzinfo=None, width='long', uncommon=False, # Get the canonical time-zone code zone = get_global('zone_aliases').get(zone, zone) - + if return_zone: + return zone info = locale.time_zones.get(zone, {}) # Try explicitly translated zone names first if width in info: @@ -524,8 +659,13 @@ def get_timezone_name(dt_or_tzinfo=None, width='long', uncommon=False, if metazone: metazone_info = locale.meta_zones.get(metazone, {}) if width in metazone_info: - if zone_variant in metazone_info[width]: - return metazone_info[width][zone_variant] + name = metazone_info[width].get(zone_variant) + if width == 'short' and name == NO_INHERITANCE_MARKER: + # If the short form is marked no-inheritance, + # try to fall back to the long name instead. + name = metazone_info.get('long', {}).get(zone_variant) + if name: + return name # If we have a concrete datetime, we assume that the result can't be # independent of daylight savings time, so we return the GMT offset @@ -538,7 +678,7 @@ def get_timezone_name(dt_or_tzinfo=None, width='long', uncommon=False, def format_date(date=None, format='medium', locale=LC_TIME): """Return a date formatted according to the given pattern. - >>> d = date(2007, 04, 01) + >>> d = date(2007, 4, 1) >>> format_date(d, locale='en_US') u'Apr 1, 2007' >>> format_date(d, format='full', locale='de_DE') @@ -572,7 +712,7 @@ def format_datetime(datetime=None, format='medium', tzinfo=None, locale=LC_TIME): r"""Return a date formatted according to the given pattern. - >>> dt = datetime(2007, 04, 01, 15, 30) + >>> dt = datetime(2007, 4, 1, 15, 30) >>> format_datetime(dt, locale='en_US') u'Apr 1, 2007, 3:30:00 PM' @@ -581,7 +721,7 @@ def format_datetime(datetime=None, format='medium', tzinfo=None, >>> format_datetime(dt, 'full', tzinfo=get_timezone('Europe/Paris'), ... locale='fr_FR') - u'dimanche 1 avril 2007 17:30:00 heure avanc\xe9e d\u2019Europe centrale' + u'dimanche 1 avril 2007 \xe0 17:30:00 heure d\u2019\xe9t\xe9 d\u2019Europe centrale' >>> format_datetime(dt, "yyyy.MM.dd G 'at' HH:mm:ss zzz", ... tzinfo=get_timezone('US/Eastern'), locale='en') u'2007.04.01 AD at 11:30:00 EDT' @@ -593,18 +733,7 @@ def format_datetime(datetime=None, format='medium', tzinfo=None, :param tzinfo: the timezone to apply to the time for display :param locale: a `Locale` object or a locale identifier """ - if datetime is None: - datetime = datetime_.utcnow() - elif isinstance(datetime, number_types): - datetime = datetime_.utcfromtimestamp(datetime) - elif isinstance(datetime, time): - datetime = datetime_.combine(date.today(), datetime) - if datetime.tzinfo is None: - datetime = datetime.replace(tzinfo=UTC) - if tzinfo is not None: - datetime = datetime.astimezone(get_timezone(tzinfo)) - if hasattr(tzinfo, 'normalize'): # pytz - datetime = tzinfo.normalize(datetime) + datetime = _ensure_datetime_tzinfo(_get_datetime(datetime), tzinfo) locale = Locale.parse(locale) if format in ('full', 'long', 'medium', 'short'): @@ -639,7 +768,7 @@ def format_time(time=None, format='medium', tzinfo=None, locale=LC_TIME): >>> tzinfo = get_timezone('Europe/Paris') >>> t = tzinfo.localize(t) >>> format_time(t, format='full', tzinfo=tzinfo, locale='fr_FR') - u'15:30:00 heure avanc\xe9e d\u2019Europe centrale' + u'15:30:00 heure d\u2019\xe9t\xe9 d\u2019Europe centrale' >>> format_time(t, "hh 'o''clock' a, zzzz", tzinfo=get_timezone('US/Eastern'), ... locale='en') u"09 o'clock AM, Eastern Daylight Time" @@ -660,7 +789,7 @@ def format_time(time=None, format='medium', tzinfo=None, locale=LC_TIME): >>> t = time(15, 30) >>> format_time(t, format='full', tzinfo=get_timezone('Europe/Paris'), ... locale='fr_FR') - u'15:30:00 heure normale de l\u2019Europe centrale' + u'15:30:00 heure normale d\u2019Europe centrale' >>> format_time(t, format='full', tzinfo=get_timezone('US/Eastern'), ... locale='en_US') u'3:30:00 PM Eastern Standard Time' @@ -672,20 +801,7 @@ def format_time(time=None, format='medium', tzinfo=None, locale=LC_TIME): :param tzinfo: the time-zone to apply to the time for display :param locale: a `Locale` object or a locale identifier """ - if time is None: - time = datetime.utcnow() - elif isinstance(time, number_types): - time = datetime.utcfromtimestamp(time) - if time.tzinfo is None: - time = time.replace(tzinfo=UTC) - if isinstance(time, datetime): - if tzinfo is not None: - time = time.astimezone(tzinfo) - if hasattr(tzinfo, 'normalize'): # pytz - time = tzinfo.normalize(time) - time = time.timetz() - elif tzinfo is not None: - time = time.replace(tzinfo=tzinfo) + time = _get_time(time, tzinfo) locale = Locale.parse(locale) if format in ('full', 'long', 'medium', 'short'): @@ -693,19 +809,57 @@ def format_time(time=None, format='medium', tzinfo=None, locale=LC_TIME): return parse_pattern(format).apply(time, locale) +def format_skeleton(skeleton, datetime=None, tzinfo=None, fuzzy=True, locale=LC_TIME): + r"""Return a time and/or date formatted according to the given pattern. + + The skeletons are defined in the CLDR data and provide more flexibility + than the simple short/long/medium formats, but are a bit harder to use. + The are defined using the date/time symbols without order or punctuation + and map to a suitable format for the given locale. + + >>> t = datetime(2007, 4, 1, 15, 30) + >>> format_skeleton('MMMEd', t, locale='fr') + u'dim. 1 avr.' + >>> format_skeleton('MMMEd', t, locale='en') + u'Sun, Apr 1' + >>> format_skeleton('yMMd', t, locale='fi') # yMMd is not in the Finnish locale; yMd gets used + u'1.4.2007' + >>> format_skeleton('yMMd', t, fuzzy=False, locale='fi') # yMMd is not in the Finnish locale, an error is thrown + Traceback (most recent call last): + ... + KeyError: yMMd + + After the skeleton is resolved to a pattern `format_datetime` is called so + all timezone processing etc is the same as for that. + + :param skeleton: A date time skeleton as defined in the cldr data. + :param datetime: the ``time`` or ``datetime`` object; if `None`, the current + time in UTC is used + :param tzinfo: the time-zone to apply to the time for display + :param fuzzy: If the skeleton is not found, allow choosing a skeleton that's + close enough to it. + :param locale: a `Locale` object or a locale identifier + """ + locale = Locale.parse(locale) + if fuzzy and skeleton not in locale.datetime_skeletons: + skeleton = match_skeleton(skeleton, locale.datetime_skeletons) + format = locale.datetime_skeletons[skeleton] + return format_datetime(datetime, format, tzinfo, locale) + + TIMEDELTA_UNITS = ( - ('year', 3600 * 24 * 365), - ('month', 3600 * 24 * 30), - ('week', 3600 * 24 * 7), - ('day', 3600 * 24), - ('hour', 3600), + ('year', 3600 * 24 * 365), + ('month', 3600 * 24 * 30), + ('week', 3600 * 24 * 7), + ('day', 3600 * 24), + ('hour', 3600), ('minute', 60), ('second', 1) ) def format_timedelta(delta, granularity='second', threshold=.85, - add_direction=False, format='medium', + add_direction=False, format='long', locale=LC_TIME): """Return a time delta according to the rules of the given locale. @@ -733,11 +887,18 @@ def format_timedelta(delta, granularity='second', threshold=.85, In addition directional information can be provided that informs the user if the date is in the past or in the future: - >>> format_timedelta(timedelta(hours=1), add_direction=True) - u'In 1 hour' - >>> format_timedelta(timedelta(hours=-1), add_direction=True) + >>> format_timedelta(timedelta(hours=1), add_direction=True, locale='en') + u'in 1 hour' + >>> format_timedelta(timedelta(hours=-1), add_direction=True, locale='en') u'1 hour ago' + The format parameter controls how compact or wide the presentation is: + + >>> format_timedelta(timedelta(hours=3), format='short', locale='en') + u'3 hr' + >>> format_timedelta(timedelta(hours=3), format='narrow', locale='en') + u'3h' + :param delta: a ``timedelta`` object representing the time difference to format, or the delta in seconds as an `int` value :param granularity: determines the smallest unit that should be displayed, @@ -750,25 +911,33 @@ def format_timedelta(delta, granularity='second', threshold=.85, positive timedelta will include the information about it being in the future, a negative will be information about the value being in the past. - :param format: the format (currently only "medium" and "short" are supported) + :param format: the format, can be "narrow", "short" or "long". ( + "medium" is deprecated, currently converted to "long" to + maintain compatibility) :param locale: a `Locale` object or a locale identifier """ - if format not in ('short', 'medium'): - raise TypeError('Format can only be one of "short" or "medium"') + if format not in ('narrow', 'short', 'medium', 'long'): + raise TypeError('Format must be one of "narrow", "short" or "long"') + if format == 'medium': + warnings.warn('"medium" value for format param of format_timedelta' + ' is deprecated. Use "long" instead', + category=DeprecationWarning) + format = 'long' if isinstance(delta, timedelta): seconds = int((delta.days * 86400) + delta.seconds) else: seconds = delta locale = Locale.parse(locale) - def _iter_choices(unit): + def _iter_patterns(a_unit): if add_direction: + unit_rel_patterns = locale._data['date_fields'][a_unit] if seconds >= 0: - yield unit + '-future' + yield unit_rel_patterns['future'] else: - yield unit + '-past' - yield unit + ':' + format - yield unit + yield unit_rel_patterns['past'] + a_unit = 'duration-' + a_unit + yield locale._data['unit_patterns'].get(a_unit, {}).get(format) for unit, secs_per_unit in TIMEDELTA_UNITS: value = abs(seconds) / secs_per_unit @@ -778,8 +947,7 @@ def _iter_choices(unit): value = int(round(value)) plural_form = locale.plural_form(value) pattern = None - for choice in _iter_choices(unit): - patterns = locale._data['unit_patterns'].get(choice) + for patterns in _iter_patterns(unit): if patterns is not None: pattern = patterns[plural_form] break @@ -791,6 +959,181 @@ def _iter_choices(unit): return u'' +def _format_fallback_interval(start, end, skeleton, tzinfo, locale): + if skeleton in locale.datetime_skeletons: # Use the given skeleton + format = lambda dt: format_skeleton(skeleton, dt, tzinfo, locale=locale) + elif all((isinstance(d, date) and not isinstance(d, datetime)) for d in (start, end)): # Both are just dates + format = lambda dt: format_date(dt, locale=locale) + elif all((isinstance(d, time) and not isinstance(d, date)) for d in (start, end)): # Both are times + format = lambda dt: format_time(dt, tzinfo=tzinfo, locale=locale) + else: + format = lambda dt: format_datetime(dt, tzinfo=tzinfo, locale=locale) + + formatted_start = format(start) + formatted_end = format(end) + + if formatted_start == formatted_end: + return format(start) + + return ( + locale.interval_formats.get(None, "{0}-{1}"). + replace("{0}", formatted_start). + replace("{1}", formatted_end) + ) + + +def format_interval(start, end, skeleton=None, tzinfo=None, fuzzy=True, locale=LC_TIME): + """ + Format an interval between two instants according to the locale's rules. + + >>> format_interval(date(2016, 1, 15), date(2016, 1, 17), "yMd", locale="fi") + u'15.\u201317.1.2016' + + >>> format_interval(time(12, 12), time(16, 16), "Hm", locale="en_GB") + '12:12\u201316:16' + + >>> format_interval(time(5, 12), time(16, 16), "hm", locale="en_US") + '5:12 AM \u2013 4:16 PM' + + >>> format_interval(time(16, 18), time(16, 24), "Hm", locale="it") + '16:18\u201316:24' + + If the start instant equals the end instant, the interval is formatted like the instant. + + >>> format_interval(time(16, 18), time(16, 18), "Hm", locale="it") + '16:18' + + Unknown skeletons fall back to "default" formatting. + + >>> format_interval(date(2015, 1, 1), date(2017, 1, 1), "wzq", locale="ja") + '2015/01/01\uff5e2017/01/01' + + >>> format_interval(time(16, 18), time(16, 24), "xxx", locale="ja") + '16:18:00\uff5e16:24:00' + + >>> format_interval(date(2016, 1, 15), date(2016, 1, 17), "xxx", locale="de") + '15.01.2016 \u2013 17.01.2016' + + :param start: First instant (datetime/date/time) + :param end: Second instant (datetime/date/time) + :param skeleton: The "skeleton format" to use for formatting. + :param tzinfo: tzinfo to use (if none is already attached) + :param fuzzy: If the skeleton is not found, allow choosing a skeleton that's + close enough to it. + :param locale: A locale object or identifier. + :return: Formatted interval + """ + locale = Locale.parse(locale) + + # NB: The quote comments below are from the algorithm description in + # https://www.unicode.org/reports/tr35/tr35-dates.html#intervalFormats + + # > Look for the intervalFormatItem element that matches the "skeleton", + # > starting in the current locale and then following the locale fallback + # > chain up to, but not including root. + + interval_formats = locale.interval_formats + + if skeleton not in interval_formats or not skeleton: + # > If no match was found from the previous step, check what the closest + # > match is in the fallback locale chain, as in availableFormats. That + # > is, this allows for adjusting the string value field's width, + # > including adjusting between "MMM" and "MMMM", and using different + # > variants of the same field, such as 'v' and 'z'. + if skeleton and fuzzy: + skeleton = match_skeleton(skeleton, interval_formats) + else: + skeleton = None + if not skeleton: # Still no match whatsoever? + # > Otherwise, format the start and end datetime using the fallback pattern. + return _format_fallback_interval(start, end, skeleton, tzinfo, locale) + + skel_formats = interval_formats[skeleton] + + if start == end: + return format_skeleton(skeleton, start, tzinfo, fuzzy=fuzzy, locale=locale) + + start = _ensure_datetime_tzinfo(_get_datetime(start), tzinfo=tzinfo) + end = _ensure_datetime_tzinfo(_get_datetime(end), tzinfo=tzinfo) + + start_fmt = DateTimeFormat(start, locale=locale) + end_fmt = DateTimeFormat(end, locale=locale) + + # > If a match is found from previous steps, compute the calendar field + # > with the greatest difference between start and end datetime. If there + # > is no difference among any of the fields in the pattern, format as a + # > single date using availableFormats, and return. + + for field in PATTERN_CHAR_ORDER: # These are in largest-to-smallest order + if field in skel_formats: + if start_fmt.extract(field) != end_fmt.extract(field): + # > If there is a match, use the pieces of the corresponding pattern to + # > format the start and end datetime, as above. + return "".join( + parse_pattern(pattern).apply(instant, locale) + for pattern, instant + in zip(skel_formats[field], (start, end)) + ) + + # > Otherwise, format the start and end datetime using the fallback pattern. + + return _format_fallback_interval(start, end, skeleton, tzinfo, locale) + + +def get_period_id(time, tzinfo=None, type=None, locale=LC_TIME): + """ + Get the day period ID for a given time. + + This ID can be used as a key for the period name dictionary. + + >>> get_period_names(locale="de")[get_period_id(time(7, 42), locale="de")] + u'Morgen' + + :param time: The time to inspect. + :param tzinfo: The timezone for the time. See ``format_time``. + :param type: The period type to use. Either "selection" or None. + The selection type is used for selecting among phrases such as + “Your email arrived yesterday evening” or “Your email arrived last night”. + :param locale: the `Locale` object, or a locale string + :return: period ID. Something is always returned -- even if it's just "am" or "pm". + """ + time = _get_time(time, tzinfo) + seconds_past_midnight = int(time.hour * 60 * 60 + time.minute * 60 + time.second) + locale = Locale.parse(locale) + + # The LDML rules state that the rules may not overlap, so iterating in arbitrary + # order should be alright, though `at` periods should be preferred. + rulesets = locale.day_period_rules.get(type, {}).items() + + for rule_id, rules in rulesets: + for rule in rules: + if "at" in rule and rule["at"] == seconds_past_midnight: + return rule_id + + for rule_id, rules in rulesets: + for rule in rules: + start_ok = end_ok = False + + if "from" in rule and seconds_past_midnight >= rule["from"]: + start_ok = True + if "to" in rule and seconds_past_midnight <= rule["to"]: + # This rule type does not exist in the present CLDR data; + # excuse the lack of test coverage. + end_ok = True + if "before" in rule and seconds_past_midnight < rule["before"]: + end_ok = True + if "after" in rule: + raise NotImplementedError("'after' is deprecated as of CLDR 29.") + + if start_ok and end_ok: + return rule_id + + if seconds_past_midnight < 43200: + return "am" + else: + return "pm" + + def parse_date(string, locale=LC_TIME): """Parse a date from a string. @@ -820,7 +1163,7 @@ def parse_date(string, locale=LC_TIME): # FIXME: this currently only supports numbers, but should also support month # names, both in the requested locale, and english - numbers = re.findall('(\d+)', string) + numbers = re.findall(r'(\d+)', string) year = numbers[indexes['Y']] if len(year) == 2: year = 2000 + int(year) @@ -863,7 +1206,7 @@ def parse_time(string, locale=LC_TIME): # and seconds should be optional, maybe minutes too # oh, and time-zones, of course - numbers = re.findall('(\d+)', string) + numbers = re.findall(r'(\d+)', string) hour = int(numbers[indexes['H']]) minute = int(numbers[indexes['M']]) second = int(numbers[indexes['S']]) @@ -882,6 +1225,12 @@ def __repr__(self): def __unicode__(self): return self.pattern + def __str__(self): + pat = self.pattern + if PY2: + pat = pat.encode('utf-8') + return pat + def __mod__(self, other): if type(other) is not DateTimeFormat: return NotImplemented @@ -922,6 +1271,7 @@ def __getitem__(self, name): elif char in ('E', 'e', 'c'): return self.format_weekday(char, num) elif char == 'a': + # TODO: Add support for the rest of the period formats (a*, b*, B*) return self.format_period(char) elif char == 'h': if self.value.hour % 12 == 0: @@ -945,11 +1295,30 @@ def __getitem__(self, name): return self.format_frac_seconds(num) elif char == 'A': return self.format_milliseconds_in_day(num) - elif char in ('z', 'Z', 'v', 'V'): + elif char in ('z', 'Z', 'v', 'V', 'x', 'X', 'O'): return self.format_timezone(char, num) else: raise KeyError('Unsupported date/time field %r' % char) + def extract(self, char): + char = str(char)[0] + if char == 'y': + return self.value.year + elif char == 'M': + return self.value.month + elif char == 'd': + return self.value.day + elif char == 'H': + return self.value.hour + elif char == 'h': + return self.value.hour % 12 or 12 + elif char == 'm': + return self.value.minute + elif char == 'a': + return int(self.value.hour >= 12) # 0 for am, 1 for pm + else: + raise NotImplementedError("Not implemented: extracting %r from %r" % (char, self.value)) + def format_era(self, char, num): width = {3: 'abbreviated', 4: 'wide', 5: 'narrow'}[max(3, num)] era = int(self.value.year >= 0) @@ -958,9 +1327,7 @@ def format_era(self, char, num): def format_year(self, char, num): value = self.value.year if char.isupper(): - week = self.get_week_number(self.get_day_of_year()) - if week == 0: - value -= 1 + value = self.value.isocalendar()[0] year = self.format(value, num) if num == 2: year = year[-2:] @@ -969,20 +1336,20 @@ def format_year(self, char, num): def format_quarter(self, char, num): quarter = (self.value.month - 1) // 3 + 1 if num <= 2: - return ('%%0%dd' % num) % quarter + return '%0*d' % (num, quarter) width = {3: 'abbreviated', 4: 'wide', 5: 'narrow'}[num] context = {'Q': 'format', 'q': 'stand-alone'}[char] return get_quarter_names(width, context, self.locale)[quarter] def format_month(self, char, num): if num <= 2: - return ('%%0%dd' % num) % self.value.month + return '%0*d' % (num, self.value.month) width = {3: 'abbreviated', 4: 'wide', 5: 'narrow'}[num] context = {'M': 'format', 'L': 'stand-alone'}[char] return get_month_names(width, context, self.locale)[self.value.month] def format_week(self, char, num): - if char.islower(): # week of year + if char.islower(): # week of year day_of_year = self.get_day_of_year() week = self.get_week_number(day_of_year) if week == 0: @@ -990,23 +1357,51 @@ def format_week(self, char, num): week = self.get_week_number(self.get_day_of_year(date), date.weekday()) return self.format(week, num) - else: # week of month + else: # week of month week = self.get_week_number(self.value.day) if week == 0: date = self.value - timedelta(days=self.value.day) week = self.get_week_number(date.day, date.weekday()) - pass return '%d' % week - def format_weekday(self, char, num): + def format_weekday(self, char='E', num=4): + """ + Return weekday from parsed datetime according to format pattern. + + >>> format = DateTimeFormat(date(2016, 2, 28), Locale.parse('en_US')) + >>> format.format_weekday() + u'Sunday' + + 'E': Day of week - Use one through three letters for the abbreviated day name, four for the full (wide) name, + five for the narrow name, or six for the short name. + >>> format.format_weekday('E',2) + u'Sun' + + 'e': Local day of week. Same as E except adds a numeric value that will depend on the local starting day of the + week, using one or two letters. For this example, Monday is the first day of the week. + >>> format.format_weekday('e',2) + '01' + + 'c': Stand-Alone local day of week - Use one letter for the local numeric value (same as 'e'), three for the + abbreviated day name, four for the full (wide) name, five for the narrow name, or six for the short name. + >>> format.format_weekday('c',1) + '1' + + :param char: pattern format character ('e','E','c') + :param num: count of format character + + """ if num < 3: if char.islower(): value = 7 - self.locale.first_week_day + self.value.weekday() return self.format(value % 7 + 1, num) num = 3 weekday = self.value.weekday() - width = {3: 'abbreviated', 4: 'wide', 5: 'narrow'}[num] - context = {3: 'format', 4: 'format', 5: 'stand-alone'}[num] + width = {3: 'abbreviated', 4: 'wide', 5: 'narrow', 6: 'short'}[num] + if char == 'c': + context = 'stand-alone' + else: + context = 'format' return get_day_names(width, context, self.locale)[weekday] def format_day_of_year(self, num): @@ -1017,23 +1412,38 @@ def format_day_of_week_in_month(self): def format_period(self, char): period = {0: 'am', 1: 'pm'}[int(self.value.hour >= 12)] - return get_period_names(locale=self.locale)[period] + for width in ('wide', 'narrow', 'abbreviated'): + period_names = get_period_names(context='format', width=width, locale=self.locale) + if period in period_names: + return period_names[period] + raise ValueError('Could not format period %s in %s' % (period, self.locale)) def format_frac_seconds(self, num): - value = str(self.value.microsecond) - return self.format(round(float('.%s' % value), num) * 10**num, num) + """ Return fractional seconds. + + Rounds the time's microseconds to the precision given by the number \ + of digits passed in. + """ + value = self.value.microsecond / 1000000 + return self.format(round(value, num) * 10**num, num) def format_milliseconds_in_day(self, num): msecs = self.value.microsecond // 1000 + self.value.second * 1000 + \ - self.value.minute * 60000 + self.value.hour * 3600000 + self.value.minute * 60000 + self.value.hour * 3600000 return self.format(msecs, num) def format_timezone(self, char, num): - width = {3: 'short', 4: 'long'}[max(3, num)] + width = {3: 'short', 4: 'long', 5: 'iso8601'}[max(3, num)] if char == 'z': return get_timezone_name(self.value, width, locale=self.locale) elif char == 'Z': + if num == 5: + return get_timezone_gmt(self.value, width, locale=self.locale, return_z=True) return get_timezone_gmt(self.value, width, locale=self.locale) + elif char == 'O': + if num == 4: + return get_timezone_gmt(self.value, width, locale=self.locale) + # TODO: To add support for O:1 elif char == 'v': return get_timezone_name(self.value.tzinfo, width, locale=self.locale) @@ -1041,10 +1451,32 @@ def format_timezone(self, char, num): if num == 1: return get_timezone_name(self.value.tzinfo, width, uncommon=True, locale=self.locale) + elif num == 2: + return get_timezone_name(self.value.tzinfo, locale=self.locale, return_zone=True) + elif num == 3: + return get_timezone_location(self.value.tzinfo, locale=self.locale, return_city=True) return get_timezone_location(self.value.tzinfo, locale=self.locale) + # Included additional elif condition to add support for 'Xx' in timezone format + elif char == 'X': + if num == 1: + return get_timezone_gmt(self.value, width='iso8601_short', locale=self.locale, + return_z=True) + elif num in (2, 4): + return get_timezone_gmt(self.value, width='short', locale=self.locale, + return_z=True) + elif num in (3, 5): + return get_timezone_gmt(self.value, width='iso8601', locale=self.locale, + return_z=True) + elif char == 'x': + if num == 1: + return get_timezone_gmt(self.value, width='iso8601_short', locale=self.locale) + elif num in (2, 4): + return get_timezone_gmt(self.value, width='short', locale=self.locale) + elif num in (3, 5): + return get_timezone_gmt(self.value, width='iso8601', locale=self.locale) def format(self, value, length): - return ('%%0%dd' % length) % value + return '%0*d' % (length, value) def get_day_of_year(self, date=None): if date is None: @@ -1079,26 +1511,46 @@ def get_week_number(self, day_of_period, day_of_week=None): if first_day < 0: first_day += 7 week_number = (day_of_period + first_day - 1) // 7 + if 7 - first_day >= self.locale.min_week_days: week_number += 1 + + if self.locale.first_week_day == 0: + # Correct the weeknumber in case of iso-calendar usage (first_week_day=0). + # If the weeknumber exceeds the maximum number of weeks for the given year + # we must count from zero.For example the above calculation gives week 53 + # for 2018-12-31. By iso-calender definition 2018 has a max of 52 + # weeks, thus the weeknumber must be 53-52=1. + max_weeks = date(year=self.value.year, day=28, month=12).isocalendar()[1] + if week_number > max_weeks: + week_number -= max_weeks + return week_number PATTERN_CHARS = { - 'G': [1, 2, 3, 4, 5], # era - 'y': None, 'Y': None, 'u': None, # year - 'Q': [1, 2, 3, 4], 'q': [1, 2, 3, 4], # quarter - 'M': [1, 2, 3, 4, 5], 'L': [1, 2, 3, 4, 5], # month - 'w': [1, 2], 'W': [1], # week - 'd': [1, 2], 'D': [1, 2, 3], 'F': [1], 'g': None, # day - 'E': [1, 2, 3, 4, 5], 'e': [1, 2, 3, 4, 5], 'c': [1, 3, 4, 5], # week day - 'a': [1], # period - 'h': [1, 2], 'H': [1, 2], 'K': [1, 2], 'k': [1, 2], # hour - 'm': [1, 2], # minute - 's': [1, 2], 'S': None, 'A': None, # second - 'z': [1, 2, 3, 4], 'Z': [1, 2, 3, 4], 'v': [1, 4], 'V': [1, 4] # zone + 'G': [1, 2, 3, 4, 5], # era + 'y': None, 'Y': None, 'u': None, # year + 'Q': [1, 2, 3, 4, 5], 'q': [1, 2, 3, 4, 5], # quarter + 'M': [1, 2, 3, 4, 5], 'L': [1, 2, 3, 4, 5], # month + 'w': [1, 2], 'W': [1], # week + 'd': [1, 2], 'D': [1, 2, 3], 'F': [1], 'g': None, # day + 'E': [1, 2, 3, 4, 5, 6], 'e': [1, 2, 3, 4, 5, 6], 'c': [1, 3, 4, 5, 6], # week day + 'a': [1], # period + 'h': [1, 2], 'H': [1, 2], 'K': [1, 2], 'k': [1, 2], # hour + 'm': [1, 2], # minute + 's': [1, 2], 'S': None, 'A': None, # second + 'z': [1, 2, 3, 4], 'Z': [1, 2, 3, 4, 5], 'O': [1, 4], 'v': [1, 4], # zone + 'V': [1, 2, 3, 4], 'x': [1, 2, 3, 4, 5], 'X': [1, 2, 3, 4, 5] # zone } +#: The pattern characters declared in the Date Field Symbol Table +#: (https://www.unicode.org/reports/tr35/tr35-dates.html#Date_Field_Symbol_Table) +#: in order of decreasing magnitude. +PATTERN_CHAR_ORDER = "GyYuUQqMLlwWdDFgEecabBChHKkjJmsSAzZOvVXx" + +_pattern_cache = {} + def parse_pattern(pattern): """Parse date, time, and datetime format patterns. @@ -1124,6 +1576,44 @@ def parse_pattern(pattern): if type(pattern) is DateTimePattern: return pattern + if pattern in _pattern_cache: + return _pattern_cache[pattern] + + result = [] + + for tok_type, tok_value in tokenize_pattern(pattern): + if tok_type == "chars": + result.append(tok_value.replace('%', '%%')) + elif tok_type == "field": + fieldchar, fieldnum = tok_value + limit = PATTERN_CHARS[fieldchar] + if limit and fieldnum not in limit: + raise ValueError('Invalid length for field: %r' + % (fieldchar * fieldnum)) + result.append('%%(%s)s' % (fieldchar * fieldnum)) + else: + raise NotImplementedError("Unknown token type: %s" % tok_type) + + _pattern_cache[pattern] = pat = DateTimePattern(pattern, u''.join(result)) + return pat + + +def tokenize_pattern(pattern): + """ + Tokenize date format patterns. + + Returns a list of (token_type, token_value) tuples. + + ``token_type`` may be either "chars" or "field". + + For "chars" tokens, the value is the literal value. + + For "field" tokens, the value is a tuple of (field character, repetition count). + + :param pattern: Pattern string + :type pattern: str + :rtype: list[tuple] + """ result = [] quotebuf = None charbuf = [] @@ -1131,21 +1621,17 @@ def parse_pattern(pattern): fieldnum = [0] def append_chars(): - result.append(''.join(charbuf).replace('%', '%%')) + result.append(('chars', ''.join(charbuf).replace('\0', "'"))) del charbuf[:] def append_field(): - limit = PATTERN_CHARS[fieldchar[0]] - if limit and fieldnum[0] not in limit: - raise ValueError('Invalid length for field: %r' - % (fieldchar[0] * fieldnum[0])) - result.append('%%(%s)s' % (fieldchar[0] * fieldnum[0])) + result.append(('field', (fieldchar[0], fieldnum[0]))) fieldchar[0] = '' fieldnum[0] = 0 for idx, char in enumerate(pattern.replace("''", '\0')): if quotebuf is None: - if char == "'": # quote started + if char == "'": # quote started if fieldchar[0]: append_field() elif charbuf: @@ -1167,10 +1653,10 @@ def append_field(): charbuf.append(char) elif quotebuf is not None: - if char == "'": # end of quote + if char == "'": # end of quote charbuf.extend(quotebuf) quotebuf = None - else: # inside quote + else: # inside quote quotebuf.append(char) if fieldchar[0]: @@ -1178,4 +1664,133 @@ def append_field(): elif charbuf: append_chars() - return DateTimePattern(pattern, u''.join(result).replace('\0', "'")) + return result + + +def untokenize_pattern(tokens): + """ + Turn a date format pattern token stream back into a string. + + This is the reverse operation of ``tokenize_pattern``. + + :type tokens: Iterable[tuple] + :rtype: str + """ + output = [] + for tok_type, tok_value in tokens: + if tok_type == "field": + output.append(tok_value[0] * tok_value[1]) + elif tok_type == "chars": + if not any(ch in PATTERN_CHARS for ch in tok_value): # No need to quote + output.append(tok_value) + else: + output.append("'%s'" % tok_value.replace("'", "''")) + return "".join(output) + + +def split_interval_pattern(pattern): + """ + Split an interval-describing datetime pattern into multiple pieces. + + > The pattern is then designed to be broken up into two pieces by determining the first repeating field. + - https://www.unicode.org/reports/tr35/tr35-dates.html#intervalFormats + + >>> split_interval_pattern(u'E d.M. \u2013 E d.M.') + [u'E d.M. \u2013 ', 'E d.M.'] + >>> split_interval_pattern("Y 'text' Y 'more text'") + ["Y 'text '", "Y 'more text'"] + >>> split_interval_pattern(u"E, MMM d \u2013 E") + [u'E, MMM d \u2013 ', u'E'] + >>> split_interval_pattern("MMM d") + ['MMM d'] + >>> split_interval_pattern("y G") + ['y G'] + >>> split_interval_pattern(u"MMM d \u2013 d") + [u'MMM d \u2013 ', u'd'] + + :param pattern: Interval pattern string + :return: list of "subpatterns" + """ + + seen_fields = set() + parts = [[]] + + for tok_type, tok_value in tokenize_pattern(pattern): + if tok_type == "field": + if tok_value[0] in seen_fields: # Repeated field + parts.append([]) + seen_fields.clear() + seen_fields.add(tok_value[0]) + parts[-1].append((tok_type, tok_value)) + + return [untokenize_pattern(tokens) for tokens in parts] + + +def match_skeleton(skeleton, options, allow_different_fields=False): + """ + Find the closest match for the given datetime skeleton among the options given. + + This uses the rules outlined in the TR35 document. + + >>> match_skeleton('yMMd', ('yMd', 'yMMMd')) + 'yMd' + + >>> match_skeleton('yMMd', ('jyMMd',), allow_different_fields=True) + 'jyMMd' + + >>> match_skeleton('yMMd', ('qyMMd',), allow_different_fields=False) + + >>> match_skeleton('hmz', ('hmv',)) + 'hmv' + + :param skeleton: The skeleton to match + :type skeleton: str + :param options: An iterable of other skeletons to match against + :type options: Iterable[str] + :return: The closest skeleton match, or if no match was found, None. + :rtype: str|None + """ + + # TODO: maybe implement pattern expansion? + + # Based on the implementation in + # http://source.icu-project.org/repos/icu/icu4j/trunk/main/classes/core/src/com/ibm/icu/text/DateIntervalInfo.java + + # Filter out falsy values and sort for stability; when `interval_formats` is passed in, there may be a None key. + options = sorted(option for option in options if option) + + if 'z' in skeleton and not any('z' in option for option in options): + skeleton = skeleton.replace('z', 'v') + + get_input_field_width = dict(t[1] for t in tokenize_pattern(skeleton) if t[0] == "field").get + best_skeleton = None + best_distance = None + for option in options: + get_opt_field_width = dict(t[1] for t in tokenize_pattern(option) if t[0] == "field").get + distance = 0 + for field in PATTERN_CHARS: + input_width = get_input_field_width(field, 0) + opt_width = get_opt_field_width(field, 0) + if input_width == opt_width: + continue + if opt_width == 0 or input_width == 0: + if not allow_different_fields: # This one is not okay + option = None + break + distance += 0x1000 # Magic weight constant for "entirely different fields" + elif field == 'M' and ((input_width > 2 and opt_width <= 2) or (input_width <= 2 and opt_width > 2)): + distance += 0x100 # Magic weight for "text turns into a number" + else: + distance += abs(input_width - opt_width) + + if not option: # We lost the option along the way (probably due to "allow_different_fields") + continue + + if not best_skeleton or distance < best_distance: + best_skeleton = option + best_distance = distance + + if distance == 0: # Found a perfect match! + break + + return best_skeleton diff --git a/babel/languages.py b/babel/languages.py new file mode 100644 index 000000000..097436705 --- /dev/null +++ b/babel/languages.py @@ -0,0 +1,71 @@ +# -- encoding: UTF-8 -- +from babel.core import get_global + + +def get_official_languages(territory, regional=False, de_facto=False): + """ + Get the official language(s) for the given territory. + + The language codes, if any are known, are returned in order of descending popularity. + + If the `regional` flag is set, then languages which are regionally official are also returned. + + If the `de_facto` flag is set, then languages which are "de facto" official are also returned. + + .. warning:: Note that the data is as up to date as the current version of the CLDR used + by Babel. If you need scientifically accurate information, use another source! + + :param territory: Territory code + :type territory: str + :param regional: Whether to return regionally official languages too + :type regional: bool + :param de_facto: Whether to return de-facto official languages too + :type de_facto: bool + :return: Tuple of language codes + :rtype: tuple[str] + """ + + territory = str(territory).upper() + allowed_stati = {"official"} + if regional: + allowed_stati.add("official_regional") + if de_facto: + allowed_stati.add("de_facto_official") + + languages = get_global("territory_languages").get(territory, {}) + pairs = [ + (info['population_percent'], language) + for language, info in languages.items() + if info.get('official_status') in allowed_stati + ] + pairs.sort(reverse=True) + return tuple(lang for _, lang in pairs) + + +def get_territory_language_info(territory): + """ + Get a dictionary of language information for a territory. + + The dictionary is keyed by language code; the values are dicts with more information. + + The following keys are currently known for the values: + + * `population_percent`: The percentage of the territory's population speaking the + language. + * `official_status`: An optional string describing the officiality status of the language. + Known values are "official", "official_regional" and "de_facto_official". + + .. warning:: Note that the data is as up to date as the current version of the CLDR used + by Babel. If you need scientifically accurate information, use another source! + + .. note:: Note that the format of the dict returned may change between Babel versions. + + See https://www.unicode.org/cldr/charts/latest/supplemental/territory_language_information.html + + :param territory: Territory code + :type territory: str + :return: Language information dictionary + :rtype: dict[str, dict] + """ + territory = str(territory).upper() + return get_global("territory_languages").get(territory, {}).copy() diff --git a/babel/lists.py b/babel/lists.py new file mode 100644 index 000000000..8368b27a6 --- /dev/null +++ b/babel/lists.py @@ -0,0 +1,87 @@ +# -*- coding: utf-8 -*- +""" + babel.lists + ~~~~~~~~~~~ + + Locale dependent formatting of lists. + + The default locale for the functions in this module is determined by the + following environment variables, in that order: + + * ``LC_ALL``, and + * ``LANG`` + + :copyright: (c) 2015-2021 by the Babel Team. + :license: BSD, see LICENSE for more details. +""" + +from babel.core import Locale, default_locale + +DEFAULT_LOCALE = default_locale() + + +def format_list(lst, style='standard', locale=DEFAULT_LOCALE): + """ + Format the items in `lst` as a list. + + >>> format_list(['apples', 'oranges', 'pears'], locale='en') + u'apples, oranges, and pears' + >>> format_list(['apples', 'oranges', 'pears'], locale='zh') + u'apples\u3001oranges\u548cpears' + >>> format_list(['omena', 'peruna', 'aplari'], style='or', locale='fi') + u'omena, peruna tai aplari' + + These styles are defined, but not all are necessarily available in all locales. + The following text is verbatim from the Unicode TR35-49 spec [1]. + + * standard: + A typical 'and' list for arbitrary placeholders. + eg. "January, February, and March" + * standard-short: + A short version of a 'and' list, suitable for use with short or abbreviated placeholder values. + eg. "Jan., Feb., and Mar." + * or: + A typical 'or' list for arbitrary placeholders. + eg. "January, February, or March" + * or-short: + A short version of an 'or' list. + eg. "Jan., Feb., or Mar." + * unit: + A list suitable for wide units. + eg. "3 feet, 7 inches" + * unit-short: + A list suitable for short units + eg. "3 ft, 7 in" + * unit-narrow: + A list suitable for narrow units, where space on the screen is very limited. + eg. "3′ 7″" + + [1]: https://www.unicode.org/reports/tr35/tr35-49/tr35-general.html#ListPatterns + + :param lst: a sequence of items to format in to a list + :param style: the style to format the list with. See above for description. + :param locale: the locale + """ + locale = Locale.parse(locale) + if not lst: + return '' + if len(lst) == 1: + return lst[0] + + if style not in locale.list_patterns: + raise ValueError('Locale %s does not support list formatting style %r (supported are %s)' % ( + locale, + style, + list(sorted(locale.list_patterns)), + )) + patterns = locale.list_patterns[style] + + if len(lst) == 2: + return patterns['2'].format(*lst) + + result = patterns['start'].format(lst[0], lst[1]) + for elem in lst[2:-1]: + result = patterns['middle'].format(result, elem) + result = patterns['end'].format(result, lst[-1]) + + return result diff --git a/babel/localedata/.gitignore b/babel/locale-data/.gitignore similarity index 100% rename from babel/localedata/.gitignore rename to babel/locale-data/.gitignore diff --git a/babel/localedata.py b/babel/localedata.py index 88883ac80..438afb643 100644 --- a/babel/localedata.py +++ b/babel/localedata.py @@ -8,44 +8,92 @@ :note: The `Locale` class, which uses this module under the hood, provides a more convenient interface for accessing the locale data. - :copyright: (c) 2013 by the Babel Team. + :copyright: (c) 2013-2021 by the Babel Team. :license: BSD, see LICENSE for more details. """ import os +import re +import sys import threading -from collections import MutableMapping +from itertools import chain -from babel._compat import pickle +from babel._compat import pickle, string_types, abc _cache = {} _cache_lock = threading.RLock() -_dirname = os.path.join(os.path.dirname(__file__), 'localedata') +_dirname = os.path.join(os.path.dirname(__file__), 'locale-data') +_windows_reserved_name_re = re.compile("^(con|prn|aux|nul|com[0-9]|lpt[0-9])$", re.I) + + +def normalize_locale(name): + """Normalize a locale ID by stripping spaces and apply proper casing. + + Returns the normalized locale ID string or `None` if the ID is not + recognized. + """ + if not name or not isinstance(name, string_types): + return None + name = name.strip().lower() + for locale_id in chain.from_iterable([_cache, locale_identifiers()]): + if name == locale_id.lower(): + return locale_id + + +def resolve_locale_filename(name): + """ + Resolve a locale identifier to a `.dat` path on disk. + """ + + # Clean up any possible relative paths. + name = os.path.basename(name) + + # Ensure we're not left with one of the Windows reserved names. + if sys.platform == "win32" and _windows_reserved_name_re.match(os.path.splitext(name)[0]): + raise ValueError("Name %s is invalid on Windows" % name) + + # Build the path. + return os.path.join(_dirname, '%s.dat' % name) def exists(name): - """Check whether locale data is available for the given locale. Ther - return value is `True` if it exists, `False` otherwise. + """Check whether locale data is available for the given locale. + + Returns `True` if it exists, `False` otherwise. :param name: the locale identifier string """ + if not name or not isinstance(name, string_types): + return False if name in _cache: return True - return os.path.exists(os.path.join(_dirname, '%s.dat' % name)) + file_found = os.path.exists(resolve_locale_filename(name)) + return True if file_found else bool(normalize_locale(name)) def locale_identifiers(): """Return a list of all locale identifiers for which locale data is available. + This data is cached after the first invocation in `locale_identifiers.cache`. + + Removing the `locale_identifiers.cache` attribute or setting it to `None` + will cause this function to re-read the list from disk. + .. versionadded:: 0.8.1 :return: a list of locale identifiers (strings) """ - return [stem for stem, extension in [ - os.path.splitext(filename) for filename in os.listdir(_dirname) - ] if extension == '.dat' and stem != 'root'] + data = getattr(locale_identifiers, 'cache', None) + if data is None: + locale_identifiers.cache = data = [ + stem + for stem, extension in + (os.path.splitext(filename) for filename in os.listdir(_dirname)) + if extension == '.dat' and stem != 'root' + ] + return data def load(name, merge_inherited=True): @@ -73,6 +121,7 @@ def load(name, merge_inherited=True): :raise `IOError`: if no locale data file is found for the given locale identifer, or one of the locales it inherits from """ + name = os.path.basename(name) _cache_lock.acquire() try: data = _cache.get(name) @@ -81,22 +130,22 @@ def load(name, merge_inherited=True): if name == 'root' or not merge_inherited: data = {} else: - parts = name.split('_') - if len(parts) == 1: - parent = 'root' - else: - parent = '_'.join(parts[:-1]) + from babel.core import get_global + parent = get_global('parent_exceptions').get(name) + if not parent: + parts = name.split('_') + if len(parts) == 1: + parent = 'root' + else: + parent = '_'.join(parts[:-1]) data = load(parent).copy() - filename = os.path.join(_dirname, '%s.dat' % name) - fileobj = open(filename, 'rb') - try: + filename = resolve_locale_filename(name) + with open(filename, 'rb') as fileobj: if name != 'root' and merge_inherited: merge(data, pickle.load(fileobj)) else: data = pickle.load(fileobj) - _cache[name] = data - finally: - fileobj.close() + _cache[name] = data return data finally: _cache_lock.release() @@ -108,7 +157,7 @@ def merge(dict1, dict2): >>> d = {1: 'foo', 3: 'baz'} >>> merge(d, {1: 'Foo', 2: 'Bar'}) - >>> items = d.items(); items.sort(); items + >>> sorted(d.items()) [(1, 'Foo'), (2, 'Bar'), (3, 'baz')] :param dict1: the dictionary to merge into @@ -168,7 +217,7 @@ def resolve(self, data): return data -class LocaleDataDict(MutableMapping): +class LocaleDataDict(abc.MutableMapping): """Dictionary wrapper that automatically resolves aliases to the actual values. """ @@ -187,13 +236,13 @@ def __iter__(self): def __getitem__(self, key): orig = val = self._data[key] - if isinstance(val, Alias): # resolve an alias + if isinstance(val, Alias): # resolve an alias val = val.resolve(self.base) - if isinstance(val, tuple): # Merge a partial dict with an alias + if isinstance(val, tuple): # Merge a partial dict with an alias alias, others = val val = alias.resolve(self.base).copy() merge(val, others) - if type(val) is dict: # Return a nested alias-resolving dict + if type(val) is dict: # Return a nested alias-resolving dict val = LocaleDataDict(val, base=self.base) if val is not orig: self._data[key] = val diff --git a/babel/localtime/__init__.py b/babel/localtime/__init__.py index cdb3e9b5d..bd3954951 100644 --- a/babel/localtime/__init__.py +++ b/babel/localtime/__init__.py @@ -6,14 +6,14 @@ Babel specific fork of tzlocal to determine the local timezone of the system. - :copyright: (c) 2013 by the Babel Team. + :copyright: (c) 2013-2021 by the Babel Team. :license: BSD, see LICENSE for more details. """ import sys import pytz import time -from datetime import timedelta, datetime +from datetime import timedelta from datetime import tzinfo from threading import RLock @@ -26,9 +26,9 @@ _cached_tz = None _cache_lock = RLock() -STDOFFSET = timedelta(seconds = -time.timezone) +STDOFFSET = timedelta(seconds=-time.timezone) if time.daylight: - DSTOFFSET = timedelta(seconds = -time.altzone) + DSTOFFSET = timedelta(seconds=-time.altzone) else: DSTOFFSET = STDOFFSET diff --git a/babel/localtime/_unix.py b/babel/localtime/_unix.py index b4a3b599f..c2194694c 100644 --- a/babel/localtime/_unix.py +++ b/babel/localtime/_unix.py @@ -1,3 +1,4 @@ +# -*- coding: utf-8 -*- from __future__ import with_statement import os import re @@ -5,7 +6,7 @@ import pytz import subprocess -_systemconfig_tz = re.compile(r'^Time Zone: (.*)$(?m)') +_systemconfig_tz = re.compile(r'^Time Zone: (.*)$', re.MULTILINE) def _tz_from_env(tzenv): @@ -27,6 +28,7 @@ def _tz_from_env(tzenv): "tzlocal() does not support non-zoneinfo timezones like %s. \n" "Please use a timezone in the form of Continent/City") + def _get_localzone(_root='/'): """Tries to find the local timezone configuration. This method prefers finding the timezone name and passing that to pytz, @@ -86,7 +88,7 @@ def _get_localzone(_root='/'): # Issue #3 in tzlocal was that /etc/timezone was a zoneinfo file. # That's a misconfiguration, but we need to handle it gracefully: - if data[:5] != 'TZif2': + if data[:5] != b'TZif2': etctz = data.strip().decode() # Get rid of host definitions and comments: if ' ' in etctz: @@ -99,30 +101,19 @@ def _get_localzone(_root='/'): # OpenSUSE has a TIMEZONE setting in /etc/sysconfig/clock and # Gentoo has a TIMEZONE setting in /etc/conf.d/clock # We look through these files for a timezone: - zone_re = re.compile('\s*ZONE\s*=\s*\"') - timezone_re = re.compile('\s*TIMEZONE\s*=\s*\"') - end_re = re.compile('\"') + timezone_re = re.compile(r'\s*(TIME)?ZONE\s*=\s*"(?P.+)"') for filename in ('etc/sysconfig/clock', 'etc/conf.d/clock'): tzpath = os.path.join(_root, filename) if not os.path.exists(tzpath): continue with open(tzpath, 'rt') as tzfile: - data = tzfile.readlines() - - for line in data: - # Look for the ZONE= setting. - match = zone_re.match(line) - if match is None: - # No ZONE= setting. Look for the TIMEZONE= setting. + for line in tzfile: match = timezone_re.match(line) - if match is not None: - # Some setting existed - line = line[match.end():] - etctz = line[:end_re.search(line).start()] - - # We found a timezone - return pytz.timezone(etctz.replace(' ', '_')) + if match is not None: + # We found a timezone + etctz = match.group("etctz") + return pytz.timezone(etctz.replace(' ', '_')) # No explicit setting existed. Use localtime for filename in ('etc/localtime', 'usr/local/etc/localtime'): diff --git a/babel/localtime/_win32.py b/babel/localtime/_win32.py index 1f6ecc7c0..65cc0885d 100644 --- a/babel/localtime/_win32.py +++ b/babel/localtime/_win32.py @@ -10,7 +10,14 @@ import pytz -tz_names = get_global('windows_zone_mapping') +# When building the cldr data on windows this module gets imported. +# Because at that point there is no global.dat yet this call will +# fail. We want to catch it down in that case then and just assume +# the mapping was empty. +try: + tz_names = get_global('windows_zone_mapping') +except RuntimeError: + tz_names = {} def valuestodict(key): @@ -59,7 +66,7 @@ def get_localzone_name(): sub = winreg.OpenKey(tzkey, subkey) data = valuestodict(sub) sub.Close() - if data['Std'] == tzwin: + if data.get('Std', None) == tzwin: tzkeyname = subkey break diff --git a/babel/messages/__init__.py b/babel/messages/__init__.py index 1b63bae2e..7d2587f63 100644 --- a/babel/messages/__init__.py +++ b/babel/messages/__init__.py @@ -5,7 +5,7 @@ Support for ``gettext`` message catalogs. - :copyright: (c) 2013 by the Babel Team. + :copyright: (c) 2013-2021 by the Babel Team. :license: BSD, see LICENSE for more details. """ diff --git a/babel/messages/catalog.py b/babel/messages/catalog.py index 501763b58..a19a3e6d8 100644 --- a/babel/messages/catalog.py +++ b/babel/messages/catalog.py @@ -5,7 +5,7 @@ Data structures for message catalogs. - :copyright: (c) 2013 by the Babel Team. + :copyright: (c) 2013-2021 by the Babel Team. :license: BSD, see LICENSE for more details. """ @@ -13,22 +13,23 @@ import time from cgi import parse_header +from collections import OrderedDict from datetime import datetime, time as time_ from difflib import get_close_matches from email import message_from_string from copy import copy from babel import __version__ as VERSION -from babel.core import Locale +from babel.core import Locale, UnknownLocaleError from babel.dates import format_datetime from babel.messages.plurals import get_plural -from babel.util import odict, distinct, LOCALTZ, FixedOffsetTimezone -from babel._compat import string_types, number_types, PY2, cmp +from babel.util import distinct, LOCALTZ, FixedOffsetTimezone +from babel._compat import string_types, number_types, PY2, cmp, text_type, force_text __all__ = ['Message', 'Catalog', 'TranslationError'] -PYTHON_FORMAT = re.compile(r'''(?x) +PYTHON_FORMAT = re.compile(r''' \% (?:\(([\w]*)\))? ( @@ -37,7 +38,39 @@ [hlL]? ) ([diouxXeEfFgGcrs%]) -''') +''', re.VERBOSE) + + +def _parse_datetime_header(value): + match = re.match(r'^(?P.*?)(?P[+-]\d{4})?$', value) + + tt = time.strptime(match.group('datetime'), '%Y-%m-%d %H:%M') + ts = time.mktime(tt) + dt = datetime.fromtimestamp(ts) + + # Separate the offset into a sign component, hours, and # minutes + tzoffset = match.group('tzoffset') + if tzoffset is not None: + plus_minus_s, rest = tzoffset[0], tzoffset[1:] + hours_offset_s, mins_offset_s = rest[:2], rest[2:] + + # Make them all integers + plus_minus = int(plus_minus_s + '1') + hours_offset = int(hours_offset_s) + mins_offset = int(mins_offset_s) + + # Calculate net offset + net_mins_offset = hours_offset * 60 + net_mins_offset += mins_offset + net_mins_offset *= plus_minus + + # Create an offset object + tzoffset = FixedOffsetTimezone(net_mins_offset) + + # Store the offset in a datetime object + dt = dt.replace(tzinfo=tzoffset) + + return dt class Message(object): @@ -51,7 +84,7 @@ def __init__(self, id, string=u'', locations=(), flags=(), auto_comments=(), pluralizable messages :param string: the translated message string, or a ``(singular, plural)`` tuple for pluralizable messages - :param locations: a sequence of ``(filenname, lineno)`` tuples + :param locations: a sequence of ``(filename, lineno)`` tuples :param flags: a set or sequence of flags :param auto_comments: a sequence of automatic comments for the message :param user_comments: a sequence of user comments for the message @@ -61,10 +94,10 @@ def __init__(self, id, string=u'', locations=(), flags=(), auto_comments=(), PO file, if any :param context: the message context """ - self.id = id #: The message ID + self.id = id if not string and self.pluralizable: string = (u'', u'') - self.string = string #: The message translation + self.string = string self.locations = list(distinct(locations)) self.flags = set(flags) if id and self.python_format: @@ -84,21 +117,13 @@ def __repr__(self): return '<%s %r (flags: %r)>' % (type(self).__name__, self.id, list(self.flags)) - def __cmp__(self, obj): + def __cmp__(self, other): """Compare Messages, taking into account plural ids""" - def values_to_compare(): - if isinstance(obj, Message): - plural = self.pluralizable - obj_plural = obj.pluralizable - if plural and obj_plural: - return self.id[0], obj.id[0] - elif plural: - return self.id[0], obj.id - elif obj_plural: - return self.id, obj.id[0] - return self.id, obj.id - this, other = values_to_compare() - return cmp(this, other) + def values_to_compare(obj): + if isinstance(obj, Message) and obj.pluralizable: + return obj.id[0], obj.context or '' + return obj.id, obj.context or '' + return cmp(values_to_compare(self), values_to_compare(other)) def __gt__(self, other): return self.__cmp__(other) > 0 @@ -242,15 +267,13 @@ def __init__(self, locale=None, domain=None, header_comment=DEFAULT_HEADER, :param charset: the encoding to use in the output (defaults to utf-8) :param fuzzy: the fuzzy bit on the catalog header """ - self.domain = domain #: The message domain - if locale: - locale = Locale.parse(locale) - self.locale = locale #: The locale or `None` + self.domain = domain + self.locale = locale self._header_comment = header_comment - self._messages = odict() + self._messages = OrderedDict() - self.project = project or 'PROJECT' #: The project name - self.version = version or 'VERSION' #: The project version + self.project = project or 'PROJECT' + self.version = version or 'VERSION' self.copyright_holder = copyright_holder or 'ORGANIZATION' self.msgid_bugs_address = msgid_bugs_address or 'EMAIL@ADDRESS' @@ -265,18 +288,48 @@ def __init__(self, locale=None, domain=None, header_comment=DEFAULT_HEADER, creation_date = datetime.now(LOCALTZ) elif isinstance(creation_date, datetime) and not creation_date.tzinfo: creation_date = creation_date.replace(tzinfo=LOCALTZ) - self.creation_date = creation_date #: Creation date of the template + self.creation_date = creation_date if revision_date is None: revision_date = 'YEAR-MO-DA HO:MI+ZONE' elif isinstance(revision_date, datetime) and not revision_date.tzinfo: revision_date = revision_date.replace(tzinfo=LOCALTZ) - self.revision_date = revision_date #: Last revision date of the catalog - self.fuzzy = fuzzy #: Catalog header fuzzy bit (`True` or `False`) + self.revision_date = revision_date + self.fuzzy = fuzzy - self.obsolete = odict() #: Dictionary of obsolete messages + self.obsolete = OrderedDict() # Dictionary of obsolete messages self._num_plurals = None self._plural_expr = None + def _set_locale(self, locale): + if locale is None: + self._locale_identifier = None + self._locale = None + return + + if isinstance(locale, Locale): + self._locale_identifier = text_type(locale) + self._locale = locale + return + + if isinstance(locale, string_types): + self._locale_identifier = text_type(locale) + try: + self._locale = Locale.parse(locale) + except UnknownLocaleError: + self._locale = None + return + + raise TypeError('`locale` must be a Locale, a locale identifier string, or None; got %r' % locale) + + def _get_locale(self): + return self._locale + + def _get_locale_identifier(self): + return self._locale_identifier + + locale = property(_get_locale, _set_locale) + locale_identifier = property(_get_locale_identifier) + def _get_header_comment(self): comment = self._header_comment year = datetime.now(LOCALTZ).strftime('%Y') @@ -286,9 +339,9 @@ def _get_header_comment(self): .replace('VERSION', self.version) \ .replace('YEAR', year) \ .replace('ORGANIZATION', self.copyright_holder) - if self.locale: - comment = comment.replace('Translations template', '%s translations' - % self.locale.english_name) + locale_name = (self.locale.english_name if self.locale else self.locale_identifier) + if locale_name: + comment = comment.replace('Translations template', '%s translations' % locale_name) return comment def _set_header_comment(self, string): @@ -299,7 +352,7 @@ def _set_header_comment(self, string): >>> catalog = Catalog(project='Foobar', version='1.0', ... copyright_holder='Foo Company') - >>> print catalog.header_comment #doctest: +ELLIPSIS + >>> print(catalog.header_comment) #doctest: +ELLIPSIS # Translations template for Foobar. # Copyright (C) ... Foo Company # This file is distributed under the same license as the Foobar project. @@ -317,7 +370,7 @@ def _set_header_comment(self, string): ... # This file is distributed under the same license as the PROJECT ... # project. ... #''' - >>> print catalog.header_comment + >>> print(catalog.header_comment) # The POT for my really cool Foobar project. # Copyright (C) 1990-2003 Foo Company # This file is distributed under the same license as the Foobar @@ -342,10 +395,12 @@ def _get_mime_headers(self): else: headers.append(('PO-Revision-Date', self.revision_date)) headers.append(('Last-Translator', self.last_translator)) - if (self.locale is not None) and ('LANGUAGE' in self.language_team): + if self.locale_identifier: + headers.append(('Language', str(self.locale_identifier))) + if self.locale_identifier and ('LANGUAGE' in self.language_team): headers.append(('Language-Team', - self.language_team.replace('LANGUAGE', - str(self.locale)))) + self.language_team.replace('LANGUAGE', + str(self.locale_identifier)))) else: headers.append(('Language-Team', self.language_team)) if self.locale is not None: @@ -359,7 +414,8 @@ def _get_mime_headers(self): def _set_mime_headers(self, headers): for name, value in headers: - name = name.lower() + name = force_text(name.lower(), encoding=self.charset) + value = force_text(value, encoding=self.charset) if name == 'project-id-version': parts = value.split(' ') self.project = u' '.join(parts[:-1]) @@ -368,6 +424,9 @@ def _set_mime_headers(self, headers): self.msgid_bugs_address = value elif name == 'last-translator': self.last_translator = value + elif name == 'language': + value = value.replace('-', '_') + self._set_locale(value) elif name == 'language-team': self.language_team = value elif name == 'content-type': @@ -379,63 +438,11 @@ def _set_mime_headers(self, headers): self._num_plurals = int(params.get('nplurals', 2)) self._plural_expr = params.get('plural', '(n != 1)') elif name == 'pot-creation-date': - # FIXME: this should use dates.parse_datetime as soon as that - # is ready - value, tzoffset, _ = re.split('([+-]\d{4})$', value, 1) - - tt = time.strptime(value, '%Y-%m-%d %H:%M') - ts = time.mktime(tt) - - # Separate the offset into a sign component, hours, and minutes - plus_minus_s, rest = tzoffset[0], tzoffset[1:] - hours_offset_s, mins_offset_s = rest[:2], rest[2:] - - # Make them all integers - plus_minus = int(plus_minus_s + '1') - hours_offset = int(hours_offset_s) - mins_offset = int(mins_offset_s) - - # Calculate net offset - net_mins_offset = hours_offset * 60 - net_mins_offset += mins_offset - net_mins_offset *= plus_minus - - # Create an offset object - tzoffset = FixedOffsetTimezone(net_mins_offset) - - # Store the offset in a datetime object - dt = datetime.fromtimestamp(ts) - self.creation_date = dt.replace(tzinfo=tzoffset) + self.creation_date = _parse_datetime_header(value) elif name == 'po-revision-date': # Keep the value if it's not the default one if 'YEAR' not in value: - # FIXME: this should use dates.parse_datetime as soon as - # that is ready - value, tzoffset, _ = re.split('([+-]\d{4})$', value, 1) - tt = time.strptime(value, '%Y-%m-%d %H:%M') - ts = time.mktime(tt) - - # Separate the offset into a sign component, hours, and - # minutes - plus_minus_s, rest = tzoffset[0], tzoffset[1:] - hours_offset_s, mins_offset_s = rest[:2], rest[2:] - - # Make them all integers - plus_minus = int(plus_minus_s + '1') - hours_offset = int(hours_offset_s) - mins_offset = int(mins_offset_s) - - # Calculate net offset - net_mins_offset = hours_offset * 60 - net_mins_offset += mins_offset - net_mins_offset *= plus_minus - - # Create an offset object - tzoffset = FixedOffsetTimezone(net_mins_offset) - - # Store the offset in a datetime object - dt = datetime.fromtimestamp(ts) - self.revision_date = dt.replace(tzinfo=tzoffset) + self.revision_date = _parse_datetime_header(value) mime_headers = property(_get_mime_headers, _set_mime_headers, doc="""\ The MIME headers of the catalog, used for the special ``msgid ""`` entry. @@ -451,7 +458,7 @@ def _set_mime_headers(self, headers): >>> catalog = Catalog(project='Foobar', version='1.0', ... creation_date=created) >>> for name, value in catalog.mime_headers: - ... print '%s: %s' % (name, value) + ... print('%s: %s' % (name, value)) Project-Id-Version: Foobar 1.0 Report-Msgid-Bugs-To: EMAIL@ADDRESS POT-Creation-Date: 1990-04-01 15:30+0000 @@ -471,12 +478,13 @@ def _set_mime_headers(self, headers): ... last_translator='John Doe ', ... language_team='de_DE ') >>> for name, value in catalog.mime_headers: - ... print '%s: %s' % (name, value) + ... print('%s: %s' % (name, value)) Project-Id-Version: Foobar 1.0 Report-Msgid-Bugs-To: EMAIL@ADDRESS POT-Creation-Date: 1990-04-01 15:30+0000 PO-Revision-Date: 1990-08-03 12:00+0000 Last-Translator: John Doe + Language: de_DE Language-Team: de_DE Plural-Forms: nplurals=2; plural=(n != 1) MIME-Version: 1.0 @@ -494,7 +502,7 @@ def num_plurals(self): >>> Catalog(locale='en').num_plurals 2 >>> Catalog(locale='ga').num_plurals - 3 + 5 :type: `int`""" if self._num_plurals is None: @@ -511,7 +519,9 @@ def plural_expr(self): >>> Catalog(locale='en').plural_expr '(n != 1)' >>> Catalog(locale='ga').plural_expr - '(n==1 ? 0 : n==2 ? 1 : 2)' + '(n==1 ? 0 : n==2 ? 1 : n>=3 && n<=6 ? 2 : n>=7 && n<=10 ? 3 : 4)' + >>> Catalog(locale='ding').plural_expr # unknown locale + '(n != 1)' :type: `string_types`""" if self._plural_expr is None: @@ -553,7 +563,7 @@ def __iter__(self): buf.append('%s: %s' % (name, value)) flags = set() if self.fuzzy: - flags |= set(['fuzzy']) + flags |= {'fuzzy'} yield Message(u'', '\n'.join(buf), flags=flags) for key in self._messages: yield self._messages[key] @@ -642,7 +652,7 @@ def add(self, id, string=None, locations=(), flags=(), auto_comments=(), pluralizable messages :param string: the translated message string, or a ``(singular, plural)`` tuple for pluralizable messages - :param locations: a sequence of ``(filenname, lineno)`` tuples + :param locations: a sequence of ``(filename, lineno)`` tuples :param flags: a set or sequence of flags :param auto_comments: a sequence of automatic comments :param user_comments: a sequence of user comments @@ -690,7 +700,7 @@ def delete(self, id, context=None): if key in self._messages: del self._messages[key] - def update(self, template, no_fuzzy_matching=False): + def update(self, template, no_fuzzy_matching=False, update_header_comment=False, keep_user_comments=True): """Update the catalog based on the given template catalog. >>> from babel.messages import Catalog @@ -737,7 +747,7 @@ def update(self, template, no_fuzzy_matching=False): >>> 'head' in catalog False - >>> catalog.obsolete.values() + >>> list(catalog.obsolete.values()) [] :param template: the reference catalog, usually read from a POT file @@ -745,7 +755,7 @@ def update(self, template, no_fuzzy_matching=False): """ messages = self._messages remaining = messages.copy() - self._messages = odict() + self._messages = OrderedDict() # Prepare for fuzzy matching fuzzy_candidates = [] @@ -770,6 +780,10 @@ def _merge(message, oldkey, newkey): else: oldmsg = remaining.pop(oldkey, None) message.string = oldmsg.string + + if keep_user_comments: + message.user_comments = list(distinct(oldmsg.user_comments)) + if isinstance(message.id, (list, tuple)): if not isinstance(message.string, (list, tuple)): fuzzy = True @@ -784,7 +798,7 @@ def _merge(message, oldkey, newkey): message.string = message.string[0] message.flags |= oldmsg.flags if fuzzy: - message.flags |= set([u'fuzzy']) + message.flags |= {u'fuzzy'} self[message.id] = message for message in template: @@ -793,10 +807,10 @@ def _merge(message, oldkey, newkey): if key in messages: _merge(message, key, key) else: - if no_fuzzy_matching is False: + if not no_fuzzy_matching: # do some fuzzy matching with difflib if isinstance(key, tuple): - matchkey = key[0] # just the msgid, no context + matchkey = key[0] # just the msgid, no context else: matchkey = key matches = get_close_matches(matchkey.lower().strip(), @@ -814,6 +828,12 @@ def _merge(message, oldkey, newkey): for msgid in remaining: if no_fuzzy_matching or msgid not in fuzzy_matches: self.obsolete[msgid] = remaining[msgid] + + if update_header_comment: + # Allow the updated catalog's header to be rewritten based on the + # template's header + self.header_comment = template.header_comment + # Make updated catalog's POT-Creation-Date equal to the template # used to update the catalog self.creation_date = template.creation_date diff --git a/babel/messages/checkers.py b/babel/messages/checkers.py index 24ecdcfed..cba911d72 100644 --- a/babel/messages/checkers.py +++ b/babel/messages/checkers.py @@ -7,7 +7,7 @@ :since: version 0.9 - :copyright: (c) 2013 by the Babel Team. + :copyright: (c) 2013-2021 by the Babel Team. :license: BSD, see LICENSE for more details. """ @@ -17,9 +17,9 @@ #: list of format chars that are compatible to each other _string_format_compatibilities = [ - set(['i', 'd', 'u']), - set(['x', 'X']), - set(['f', 'F', 'g', 'G']) + {'i', 'd', 'u'}, + {'x', 'X'}, + {'f', 'F', 'g', 'G'} ] @@ -61,7 +61,7 @@ def python_format(catalog, message): def _validate_format(format, alternative): """Test format string `alternative` against `format`. `format` can be the - msgid of a message and `alternative` one of the `msgstr`\s. The two + msgid of a message and `alternative` one of the `msgstr`\\s. The two arguments are not interchangeable as `alternative` may contain less placeholders if `format` uses named placeholders. diff --git a/babel/messages/extract.py b/babel/messages/extract.py index 2f8084af5..64497762c 100644 --- a/babel/messages/extract.py +++ b/babel/messages/extract.py @@ -13,15 +13,16 @@ The main entry points into the extraction functionality are the functions `extract_from_dir` and `extract_from_file`. - :copyright: (c) 2013 by the Babel Team. + :copyright: (c) 2013-2021 by the Babel Team. :license: BSD, see LICENSE for more details. """ import os +from os.path import relpath import sys from tokenize import generate_tokens, COMMENT, NAME, OP, STRING -from babel.util import parse_encoding, pathmatch, relpath +from babel.util import parse_encoding, parse_future_flags, pathmatch from babel._compat import PY2, text_type from textwrap import dedent @@ -37,14 +38,15 @@ 'dgettext': (2,), 'dngettext': (2, 3), 'N_': None, - 'pgettext': ((1, 'c'), 2) + 'pgettext': ((1, 'c'), 2), + 'npgettext': ((1, 'c'), 2, 3) } DEFAULT_MAPPING = [('**.py', 'python')] empty_msgid_warning = ( -'%s: warning: Empty msgid. It is reserved by GNU gettext: gettext("") ' -'returns the header entry with meta information, not the empty string.') + '%s: warning: Empty msgid. It is reserved by GNU gettext: gettext("") ' + 'returns the header entry with meta information, not the empty string.') def _strip_comment_tags(comments, tags): @@ -135,42 +137,90 @@ def extract_from_dir(dirname=None, method_map=DEFAULT_MAPPING, absname = os.path.abspath(dirname) for root, dirnames, filenames in os.walk(absname): - for subdir in dirnames: - if subdir.startswith('.') or subdir.startswith('_'): - dirnames.remove(subdir) + dirnames[:] = [ + subdir for subdir in dirnames + if not (subdir.startswith('.') or subdir.startswith('_')) + ] dirnames.sort() filenames.sort() for filename in filenames: - filename = relpath( - os.path.join(root, filename).replace(os.sep, '/'), - dirname - ) - for pattern, method in method_map: - if pathmatch(pattern, filename): - filepath = os.path.join(absname, filename) - options = {} - for opattern, odict in options_map.items(): - if pathmatch(opattern, filename): - options = odict - if callback: - callback(filename, method, options) - for lineno, message, comments, context in \ - extract_from_file(method, filepath, - keywords=keywords, - comment_tags=comment_tags, - options=options, - strip_comment_tags= - strip_comment_tags): - yield filename, lineno, message, comments, context - break + filepath = os.path.join(root, filename).replace(os.sep, '/') + + for message_tuple in check_and_call_extract_file( + filepath, + method_map, + options_map, + callback, + keywords, + comment_tags, + strip_comment_tags, + dirpath=absname, + ): + yield message_tuple + + +def check_and_call_extract_file(filepath, method_map, options_map, + callback, keywords, comment_tags, + strip_comment_tags, dirpath=None): + """Checks if the given file matches an extraction method mapping, and if so, calls extract_from_file. + + Note that the extraction method mappings are based relative to dirpath. + So, given an absolute path to a file `filepath`, we want to check using + just the relative path from `dirpath` to `filepath`. + + Yields 5-tuples (filename, lineno, messages, comments, context). + + :param filepath: An absolute path to a file that exists. + :param method_map: a list of ``(pattern, method)`` tuples that maps of + extraction method names to extended glob patterns + :param options_map: a dictionary of additional options (optional) + :param callback: a function that is called for every file that message are + extracted from, just before the extraction itself is + performed; the function is passed the filename, the name + of the extraction method and and the options dictionary as + positional arguments, in that order + :param keywords: a dictionary mapping keywords (i.e. names of functions + that should be recognized as translation functions) to + tuples that specify which of their arguments contain + localizable strings + :param comment_tags: a list of tags of translator comments to search for + and include in the results + :param strip_comment_tags: a flag that if set to `True` causes all comment + tags to be removed from the collected comments. + :param dirpath: the path to the directory to extract messages from. + :return: iterable of 5-tuples (filename, lineno, messages, comments, context) + :rtype: Iterable[tuple[str, int, str|tuple[str], list[str], str|None] + """ + # filename is the relative path from dirpath to the actual file + filename = relpath(filepath, dirpath) + + for pattern, method in method_map: + if not pathmatch(pattern, filename): + continue + + options = {} + for opattern, odict in options_map.items(): + if pathmatch(opattern, filename): + options = odict + if callback: + callback(filename, method, options) + for message_tuple in extract_from_file( + method, filepath, + keywords=keywords, + comment_tags=comment_tags, + options=options, + strip_comment_tags=strip_comment_tags + ): + yield (filename, ) + message_tuple + + break def extract_from_file(method, filename, keywords=DEFAULT_KEYWORDS, comment_tags=(), options=None, strip_comment_tags=False): """Extract messages from a specific file. - This function returns a list of tuples of the form ``(lineno, funcname, - message)``. + This function returns a list of tuples of the form ``(lineno, message, comments, context)``. :param filename: the path to the file to extract messages from :param method: a string specifying the extraction method (.e.g. "python") @@ -183,13 +233,15 @@ def extract_from_file(method, filename, keywords=DEFAULT_KEYWORDS, :param strip_comment_tags: a flag that if set to `True` causes all comment tags to be removed from the collected comments. :param options: a dictionary of additional options (optional) + :returns: list of tuples of the form ``(lineno, message, comments, context)`` + :rtype: list[tuple[int, str|tuple[str], list[str], str|None] """ - fileobj = open(filename, 'rb') - try: - return list(extract(method, fileobj, keywords, comment_tags, options, - strip_comment_tags)) - finally: - fileobj.close() + if method == 'ignore': + return [] + + with open(filename, 'rb') as fileobj: + return list(extract(method, fileobj, keywords, comment_tags, + options, strip_comment_tags)) def extract(method, fileobj, keywords=DEFAULT_KEYWORDS, comment_tags=(), @@ -197,22 +249,23 @@ def extract(method, fileobj, keywords=DEFAULT_KEYWORDS, comment_tags=(), """Extract messages from the given file-like object using the specified extraction method. - This function returns tuples of the form ``(lineno, message, comments)``. + This function returns tuples of the form ``(lineno, message, comments, context)``. The implementation dispatches the actual extraction to plugins, based on the value of the ``method`` parameter. - >>> source = '''# foo module + >>> source = b'''# foo module ... def run(argv): - ... print _('Hello, world!') + ... print(_('Hello, world!')) ... ''' - >>> from StringIO import StringIO - >>> for message in extract('python', StringIO(source)): - ... print message + >>> from babel._compat import BytesIO + >>> for message in extract('python', BytesIO(source)): + ... print(message) (3, u'Hello, world!', [], None) - :param method: a string specifying the extraction method (.e.g. "python"); + :param method: an extraction method (a callable), or + a string specifying the extraction method (.e.g. "python"); if this is a simple name, the extraction function will be looked up by entry point; if it is an explicit reference to a function (of the form ``package.module:funcname`` or @@ -229,9 +282,13 @@ def extract(method, fileobj, keywords=DEFAULT_KEYWORDS, comment_tags=(), :param strip_comment_tags: a flag that if set to `True` causes all comment tags to be removed from the collected comments. :raise ValueError: if the extraction method is not registered + :returns: iterable of tuples of the form ``(lineno, message, comments, context)`` + :rtype: Iterable[tuple[int, str|tuple[str], list[str], str|None] """ func = None - if ':' in method or '.' in method: + if callable(method): + func = method + elif ':' in method or '.' in method: if ':' not in method: lastdot = method.rfind('.') module, attrname = method[:lastdot], method[lastdot + 1:] @@ -258,6 +315,7 @@ def extract(method, fileobj, keywords=DEFAULT_KEYWORDS, comment_tags=(), 'javascript': extract_javascript } func = builtin.get(method) + if func is None: raise ValueError('Unknown extraction method %r' % method) @@ -304,8 +362,8 @@ def extract(method, fileobj, keywords=DEFAULT_KEYWORDS, comment_tags=(), first_msg_index = spec[0] - 1 if not messages[first_msg_index]: # An empty string msgid isn't valid, emit a warning - where = '%s:%i' % (hasattr(fileobj, 'name') and \ - fileobj.name or '(unknown)', lineno) + where = '%s:%i' % (hasattr(fileobj, 'name') and + fileobj.name or '(unknown)', lineno) sys.stderr.write((empty_msgid_warning % where) + '\n') continue @@ -348,7 +406,8 @@ def extract_python(fileobj, keywords, comment_tags, options): in_def = in_translator_comments = False comment_tag = None - encoding = parse_encoding(fileobj) or options.get('encoding', 'iso-8859-1') + encoding = parse_encoding(fileobj) or options.get('encoding', 'UTF-8') + future_flags = parse_future_flags(fileobj, encoding) if PY2: next_line = fileobj.readline @@ -390,7 +449,8 @@ def extract_python(fileobj, keywords, comment_tags, options): translator_comments.append((lineno, value)) break elif funcname and call_stack == 0: - if tok == OP and value == ')': + nested = (tok == NAME and value in keywords) + if (tok == OP and value == ')') or nested: if buf: messages.append(''.join(buf)) del buf[:] @@ -415,13 +475,16 @@ def extract_python(fileobj, keywords, comment_tags, options): messages = [] translator_comments = [] in_translator_comments = False + if nested: + funcname = value elif tok == STRING: # Unwrap quotes in a safe manner, maintaining the string's # encoding # https://sourceforge.net/tracker/?func=detail&atid=355470& # aid=617979&group_id=5470 - value = eval('# coding=%s\n%s' % (str(encoding), value), - {'__builtins__':{}}, {}) + code = compile('# coding=%s\n%s' % (str(encoding), value), + '', 'eval', future_flags) + value = eval(code, {'__builtins__': {}}, {}) if PY2 and not isinstance(value, text_type): value = value.decode(encoding) buf.append(value) @@ -437,7 +500,7 @@ def extract_python(fileobj, keywords, comment_tags, options): # Let's increase the last comment's lineno in order # for the comment to still be a valid one old_lineno, old_comment = translator_comments.pop() - translator_comments.append((old_lineno+1, old_comment)) + translator_comments.append((old_lineno + 1, old_comment)) elif call_stack > 0 and tok == OP and value == ')': call_stack -= 1 elif funcname and call_stack == -1: @@ -456,8 +519,12 @@ def extract_javascript(fileobj, keywords, comment_tags, options): :param comment_tags: a list of translator tags to search for and include in the results :param options: a dictionary of additional options (optional) + Supported options are: + * `jsx` -- set to false to disable JSX/E4X support. + * `template_string` -- set to false to disable ES6 + template string support. """ - from babel.messages.jslexer import tokenize, unquote_string + from babel.messages.jslexer import Token, tokenize, unquote_string funcname = message_lineno = None messages = [] last_argument = None @@ -466,8 +533,24 @@ def extract_javascript(fileobj, keywords, comment_tags, options): encoding = options.get('encoding', 'utf-8') last_token = None call_stack = -1 + dotted = any('.' in kw for kw in keywords) + + for token in tokenize( + fileobj.read().decode(encoding), + jsx=options.get("jsx", True), + template_string=options.get("template_string", True), + dotted=dotted + ): + if ( # Turn keyword`foo` expressions into keyword("foo") calls: + funcname and # have a keyword... + (last_token and last_token.type == 'name') and # we've seen nothing after the keyword... + token.type == 'template_string' # this is a template string + ): + message_lineno = token.lineno + messages = [unquote_string(token.value)] + call_stack = 0 + token = Token('operator', ')', token.lineno) - for token in tokenize(fileobj.read().decode(encoding)): if token.type == 'operator' and token.value == '(': if funcname: message_lineno = token.lineno @@ -527,7 +610,7 @@ def extract_javascript(fileobj, keywords, comment_tags, options): messages = [] call_stack = -1 - elif token.type == 'string': + elif token.type in ('string', 'template_string'): new_value = unquote_string(token.value) if concatenate_next: last_argument = (last_argument or '') + new_value @@ -547,16 +630,16 @@ def extract_javascript(fileobj, keywords, comment_tags, options): concatenate_next = True elif call_stack > 0 and token.type == 'operator' \ - and token.value == ')': + and token.value == ')': call_stack -= 1 elif funcname and call_stack == -1: funcname = None elif call_stack == -1 and token.type == 'name' and \ - token.value in keywords and \ - (last_token is None or last_token.type != 'name' or - last_token.value != 'function'): + token.value in keywords and \ + (last_token is None or last_token.type != 'name' or + last_token.value != 'function'): funcname = token.value last_token = token diff --git a/babel/messages/frontend.py b/babel/messages/frontend.py old mode 100755 new mode 100644 index 144bc98a1..c5eb1dea9 --- a/babel/messages/frontend.py +++ b/babel/messages/frontend.py @@ -5,37 +5,125 @@ Frontends for the message extraction functionality. - :copyright: (c) 2013 by the Babel Team. + :copyright: (c) 2013-2021 by the Babel Team. :license: BSD, see LICENSE for more details. """ +from __future__ import print_function -try: - from ConfigParser import RawConfigParser -except ImportError: - from configparser import RawConfigParser -from datetime import datetime -from distutils import log -from distutils.cmd import Command -from distutils.errors import DistutilsOptionError, DistutilsSetupError -from locale import getpreferredencoding import logging -from optparse import OptionParser +import optparse import os import re import shutil import sys import tempfile +from collections import OrderedDict +from datetime import datetime +from locale import getpreferredencoding from babel import __version__ as VERSION from babel import Locale, localedata +from babel._compat import StringIO, string_types, text_type, PY2 from babel.core import UnknownLocaleError from babel.messages.catalog import Catalog -from babel.messages.extract import extract_from_dir, DEFAULT_KEYWORDS, \ - DEFAULT_MAPPING +from babel.messages.extract import DEFAULT_KEYWORDS, DEFAULT_MAPPING, check_and_call_extract_file, extract_from_dir from babel.messages.mofile import write_mo from babel.messages.pofile import read_po, write_po -from babel.util import odict, LOCALTZ -from babel._compat import string_types, BytesIO, PY2 +from babel.util import LOCALTZ +from distutils import log as distutils_log +from distutils.cmd import Command as _Command +from distutils.errors import DistutilsOptionError, DistutilsSetupError + +try: + from ConfigParser import RawConfigParser +except ImportError: + from configparser import RawConfigParser + + +po_file_read_mode = ('rU' if PY2 else 'r') + + +def listify_value(arg, split=None): + """ + Make a list out of an argument. + + Values from `distutils` argument parsing are always single strings; + values from `optparse` parsing may be lists of strings that may need + to be further split. + + No matter the input, this function returns a flat list of whitespace-trimmed + strings, with `None` values filtered out. + + >>> listify_value("foo bar") + ['foo', 'bar'] + >>> listify_value(["foo bar"]) + ['foo', 'bar'] + >>> listify_value([["foo"], "bar"]) + ['foo', 'bar'] + >>> listify_value([["foo"], ["bar", None, "foo"]]) + ['foo', 'bar', 'foo'] + >>> listify_value("foo, bar, quux", ",") + ['foo', 'bar', 'quux'] + + :param arg: A string or a list of strings + :param split: The argument to pass to `str.split()`. + :return: + """ + out = [] + + if not isinstance(arg, (list, tuple)): + arg = [arg] + + for val in arg: + if val is None: + continue + if isinstance(val, (list, tuple)): + out.extend(listify_value(val, split=split)) + continue + out.extend(s.strip() for s in text_type(val).split(split)) + assert all(isinstance(val, string_types) for val in out) + return out + + +class Command(_Command): + # This class is a small shim between Distutils commands and + # optparse option parsing in the frontend command line. + + #: Option name to be input as `args` on the script command line. + as_args = None + + #: Options which allow multiple values. + #: This is used by the `optparse` transmogrification code. + multiple_value_options = () + + #: Options which are booleans. + #: This is used by the `optparse` transmogrification code. + # (This is actually used by distutils code too, but is never + # declared in the base class.) + boolean_options = () + + #: Option aliases, to retain standalone command compatibility. + #: Distutils does not support option aliases, but optparse does. + #: This maps the distutils argument name to an iterable of aliases + #: that are usable with optparse. + option_aliases = {} + + #: Choices for options that needed to be restricted to specific + #: list of choices. + option_choices = {} + + #: Log object. To allow replacement in the script command line runner. + log = distutils_log + + def __init__(self, dist=None): + # A less strict version of distutils' `__init__`. + self.distribution = dist + self.initialize_options() + self._dry_run = None + self.verbose = False + self.force = None + self.help = 0 + self.finalized = 0 class compile_catalog(Command): @@ -58,14 +146,14 @@ class compile_catalog(Command): description = 'compile message catalogs to binary MO files' user_options = [ ('domain=', 'D', - "domain of PO file (default 'messages')"), + "domains of PO files (space separated list, default 'messages')"), ('directory=', 'd', 'path to base directory containing the catalogs'), ('input-file=', 'i', 'name of the input file'), ('output-file=', 'o', "name of the output file (default " - "'//LC_MESSAGES/.po')"), + "'//LC_MESSAGES/.mo')"), ('locale=', 'l', 'locale of the catalog to compile'), ('use-fuzzy', 'f', @@ -85,14 +173,24 @@ def initialize_options(self): self.statistics = False def finalize_options(self): + self.domain = listify_value(self.domain) if not self.input_file and not self.directory: raise DistutilsOptionError('you must specify either the input file ' 'or the base directory') if not self.output_file and not self.directory: - raise DistutilsOptionError('you must specify either the input file ' + raise DistutilsOptionError('you must specify either the output file ' 'or the base directory') def run(self): + n_errors = 0 + for domain in self.domain: + for catalog, errors in self._run_domain(domain).items(): + n_errors += len(errors) + if n_errors: + self.log.error('%d errors encountered.' % n_errors) + return (1 if n_errors else 0) + + def _run_domain(self, domain): po_files = [] mo_files = [] @@ -101,19 +199,19 @@ def run(self): po_files.append((self.locale, os.path.join(self.directory, self.locale, 'LC_MESSAGES', - self.domain + '.po'))) + domain + '.po'))) mo_files.append(os.path.join(self.directory, self.locale, 'LC_MESSAGES', - self.domain + '.mo')) + domain + '.mo')) else: for locale in os.listdir(self.directory): po_file = os.path.join(self.directory, locale, - 'LC_MESSAGES', self.domain + '.po') + 'LC_MESSAGES', domain + '.po') if os.path.exists(po_file): po_files.append((locale, po_file)) mo_files.append(os.path.join(self.directory, locale, 'LC_MESSAGES', - self.domain + '.mo')) + domain + '.mo')) else: po_files.append((self.locale, self.input_file)) if self.output_file: @@ -121,46 +219,48 @@ def run(self): else: mo_files.append(os.path.join(self.directory, self.locale, 'LC_MESSAGES', - self.domain + '.mo')) + domain + '.mo')) if not po_files: raise DistutilsOptionError('no message catalogs found') + catalogs_and_errors = {} + for idx, (locale, po_file) in enumerate(po_files): mo_file = mo_files[idx] - infile = open(po_file, 'r') - try: + with open(po_file, 'rb') as infile: catalog = read_po(infile, locale) - finally: - infile.close() if self.statistics: translated = 0 for message in list(catalog)[1:]: if message.string: - translated +=1 + translated += 1 percentage = 0 if len(catalog): percentage = translated * 100 // len(catalog) - log.info('%d of %d messages (%d%%) translated in %r', - translated, len(catalog), percentage, po_file) + self.log.info( + '%d of %d messages (%d%%) translated in %s', + translated, len(catalog), percentage, po_file + ) if catalog.fuzzy and not self.use_fuzzy: - log.warn('catalog %r is marked as fuzzy, skipping', po_file) + self.log.info('catalog %s is marked as fuzzy, skipping', po_file) continue - for message, errors in catalog.check(): + catalogs_and_errors[catalog] = catalog_errors = list(catalog.check()) + for message, errors in catalog_errors: for error in errors: - log.error('error: %s:%d: %s', po_file, message.lineno, - error) + self.log.error( + 'error: %s:%d: %s', po_file, message.lineno, error + ) - log.info('compiling catalog %r to %r', po_file, mo_file) + self.log.info('compiling catalog %s to %s', po_file, mo_file) - outfile = open(mo_file, 'wb') - try: + with open(mo_file, 'wb') as outfile: write_mo(outfile, catalog, use_fuzzy=self.use_fuzzy) - finally: - outfile.close() + + return catalogs_and_errors class extract_messages(Command): @@ -181,16 +281,21 @@ class extract_messages(Command): description = 'extract localizable strings from the project code' user_options = [ ('charset=', None, - 'charset to use in the output file'), + 'charset to use in the output file (default "utf-8")'), ('keywords=', 'k', 'space-separated list of keywords to look for in addition to the ' - 'defaults'), + 'defaults (may be repeated multiple times)'), ('no-default-keywords', None, 'do not include the default keywords'), ('mapping-file=', 'F', 'path to the mapping configuration file'), ('no-location', None, 'do not include location comments with filename and line number'), + ('add-location=', None, + 'location lines format. If it is not given or "full", it generates ' + 'the lines with both file name and line number. If it is "file", ' + 'the line number part is omitted. If it is "never", it completely ' + 'suppresses the lines (same as --no-location).'), ('omit-header', None, 'do not include msgid "" entry in header'), ('output-file=', 'o', @@ -208,48 +313,81 @@ class extract_messages(Command): 'set report address for msgid'), ('copyright-holder=', None, 'set copyright holder in output'), + ('project=', None, + 'set project name in output'), + ('version=', None, + 'set project version in output'), ('add-comments=', 'c', 'place comment block with TAG (or those preceding keyword lines) in ' - 'output file. Separate multiple TAGs with commas(,)'), - ('strip-comments', None, + 'output file. Separate multiple TAGs with commas(,)'), # TODO: Support repetition of this argument + ('strip-comments', 's', 'strip the comment TAGs from the comments.'), - ('input-dirs=', None, - 'directories that should be scanned for messages. Separate multiple ' - 'directories with commas(,)'), + ('input-paths=', None, + 'files or directories that should be scanned for messages. Separate multiple ' + 'files or directories with commas(,)'), # TODO: Support repetition of this argument + ('input-dirs=', None, # TODO (3.x): Remove me. + 'alias for input-paths (does allow files as well as directories).'), ] boolean_options = [ 'no-default-keywords', 'no-location', 'omit-header', 'no-wrap', 'sort-output', 'sort-by-file', 'strip-comments' ] + as_args = 'input-paths' + multiple_value_options = ('add-comments', 'keywords') + option_aliases = { + 'keywords': ('--keyword',), + 'mapping-file': ('--mapping',), + 'output-file': ('--output',), + 'strip-comments': ('--strip-comment-tags',), + } + option_choices = { + 'add-location': ('full', 'file', 'never',), + } def initialize_options(self): self.charset = 'utf-8' - self.keywords = '' - self._keywords = DEFAULT_KEYWORDS.copy() + self.keywords = None self.no_default_keywords = False self.mapping_file = None self.no_location = False + self.add_location = None self.omit_header = False self.output_file = None self.input_dirs = None + self.input_paths = None self.width = None self.no_wrap = False self.sort_output = False self.sort_by_file = False self.msgid_bugs_address = None self.copyright_holder = None + self.project = None + self.version = None self.add_comments = None - self._add_comments = [] self.strip_comments = False + self.include_lineno = True def finalize_options(self): - if self.no_default_keywords and not self.keywords: + if self.input_dirs: + if not self.input_paths: + self.input_paths = self.input_dirs + else: + raise DistutilsOptionError( + 'input-dirs and input-paths are mutually exclusive' + ) + + if self.no_default_keywords: + keywords = {} + else: + keywords = DEFAULT_KEYWORDS.copy() + + keywords.update(parse_keywords(listify_value(self.keywords))) + + self.keywords = keywords + + if not self.keywords: raise DistutilsOptionError('you must specify new keywords if you ' 'disable the default ones') - if self.no_default_keywords: - self._keywords = {} - if self.keywords: - self._keywords.update(parse_keywords(self.keywords.split())) if not self.output_file: raise DistutilsOptionError('no output file specified') @@ -265,84 +403,122 @@ def finalize_options(self): raise DistutilsOptionError("'--sort-output' and '--sort-by-file' " "are mutually exclusive") - if self.input_dirs: - self.input_dirs = re.split(',\s*', self.input_dirs) - else: - self.input_dirs = dict.fromkeys([k.split('.',1)[0] - for k in self.distribution.packages + if self.input_paths: + if isinstance(self.input_paths, string_types): + self.input_paths = re.split(r',\s*', self.input_paths) + elif self.distribution is not None: + self.input_paths = dict.fromkeys([ + k.split('.', 1)[0] + for k in (self.distribution.packages or ()) ]).keys() + else: + self.input_paths = [] + + if not self.input_paths: + raise DistutilsOptionError("no input files or directories specified") - if self.add_comments: - self._add_comments = self.add_comments.split(',') + for path in self.input_paths: + if not os.path.exists(path): + raise DistutilsOptionError("Input path: %s does not exist" % path) + + self.add_comments = listify_value(self.add_comments or (), ",") + + if self.distribution: + if not self.project: + self.project = self.distribution.get_name() + if not self.version: + self.version = self.distribution.get_version() + + if self.add_location == 'never': + self.no_location = True + elif self.add_location == 'file': + self.include_lineno = False def run(self): mappings = self._get_mappings() - outfile = open(self.output_file, 'wb') - try: - catalog = Catalog(project=self.distribution.get_name(), - version=self.distribution.get_version(), + with open(self.output_file, 'wb') as outfile: + catalog = Catalog(project=self.project, + version=self.version, msgid_bugs_address=self.msgid_bugs_address, copyright_holder=self.copyright_holder, charset=self.charset) - for dirname, (method_map, options_map) in mappings.items(): + for path, method_map, options_map in mappings: def callback(filename, method, options): if method == 'ignore': return - filepath = os.path.normpath(os.path.join(dirname, filename)) + + # If we explicitly provide a full filepath, just use that. + # Otherwise, path will be the directory path and filename + # is the relative path from that dir to the file. + # So we can join those to get the full filepath. + if os.path.isfile(path): + filepath = path + else: + filepath = os.path.normpath(os.path.join(path, filename)) + optstr = '' if options: optstr = ' (%s)' % ', '.join(['%s="%s"' % (k, v) for k, v in options.items()]) - log.info('extracting messages from %s%s', filepath, optstr) - - extracted = extract_from_dir(dirname, method_map, options_map, - keywords=self._keywords, - comment_tags=self._add_comments, - callback=callback, - strip_comment_tags= - self.strip_comments) + self.log.info('extracting messages from %s%s', filepath, optstr) + + if os.path.isfile(path): + current_dir = os.getcwd() + extracted = check_and_call_extract_file( + path, method_map, options_map, + callback, self.keywords, self.add_comments, + self.strip_comments, current_dir + ) + else: + extracted = extract_from_dir( + path, method_map, options_map, + keywords=self.keywords, + comment_tags=self.add_comments, + callback=callback, + strip_comment_tags=self.strip_comments + ) for filename, lineno, message, comments, context in extracted: - filepath = os.path.normpath(os.path.join(dirname, filename)) + if os.path.isfile(path): + filepath = filename # already normalized + else: + filepath = os.path.normpath(os.path.join(path, filename)) + catalog.add(message, None, [(filepath, lineno)], auto_comments=comments, context=context) - log.info('writing PO template file to %s' % self.output_file) + self.log.info('writing PO template file to %s', self.output_file) write_po(outfile, catalog, width=self.width, no_location=self.no_location, omit_header=self.omit_header, sort_output=self.sort_output, - sort_by_file=self.sort_by_file) - finally: - outfile.close() + sort_by_file=self.sort_by_file, + include_lineno=self.include_lineno) def _get_mappings(self): - mappings = {} + mappings = [] if self.mapping_file: - fileobj = open(self.mapping_file, 'U') - try: + with open(self.mapping_file, po_file_read_mode) as fileobj: method_map, options_map = parse_mapping(fileobj) - for dirname in self.input_dirs: - mappings[dirname] = method_map, options_map - finally: - fileobj.close() + for path in self.input_paths: + mappings.append((path, method_map, options_map)) elif getattr(self.distribution, 'message_extractors', None): message_extractors = self.distribution.message_extractors - for dirname, mapping in message_extractors.items(): + for path, mapping in message_extractors.items(): if isinstance(mapping, string_types): - method_map, options_map = parse_mapping(BytesIO(mapping)) + method_map, options_map = parse_mapping(StringIO(mapping)) else: method_map, options_map = [], {} for pattern, method, options in mapping: method_map.append((pattern, method)) options_map[pattern] = options or {} - mappings[dirname] = method_map, options_map + mappings.append((path, method_map, options_map)) else: - for dirname in self.input_dirs: - mappings[dirname] = DEFAULT_MAPPING, {} + for path in self.input_paths: + mappings.append((path, DEFAULT_MAPPING, {})) return mappings @@ -436,26 +612,21 @@ def finalize_options(self): self.width = int(self.width) def run(self): - log.info('creating catalog %r based on %r', self.output_file, - self.input_file) + self.log.info( + 'creating catalog %s based on %s', self.output_file, self.input_file + ) - infile = open(self.input_file, 'r') - try: + with open(self.input_file, 'rb') as infile: # Although reading from the catalog template, read_po must be fed # the locale in order to correctly calculate plurals catalog = read_po(infile, locale=self.locale) - finally: - infile.close() catalog.locale = self._locale catalog.revision_date = datetime.now(LOCALTZ) catalog.fuzzy = False - outfile = open(self.output_file, 'wb') - try: + with open(self.output_file, 'wb') as outfile: write_po(outfile, catalog, width=self.width) - finally: - outfile.close() class update_catalog(Command): @@ -486,6 +657,8 @@ class update_catalog(Command): ('output-file=', 'o', "name of the output file (default " "'//LC_MESSAGES/.po')"), + ('omit-header', None, + "do not include msgid "" entry in header"), ('locale=', 'l', 'locale of the catalog to compile'), ('width=', 'w', @@ -497,21 +670,28 @@ class update_catalog(Command): 'whether to omit obsolete messages from the output'), ('no-fuzzy-matching', 'N', 'do not use fuzzy matching'), + ('update-header-comment', None, + 'update target header comment'), ('previous', None, - 'keep previous msgids of translated messages') + 'keep previous msgids of translated messages'), + ] + boolean_options = [ + 'omit-header', 'no-wrap', 'ignore-obsolete', 'no-fuzzy-matching', + 'previous', 'update-header-comment', ] - boolean_options = ['ignore_obsolete', 'no_fuzzy_matching', 'previous'] def initialize_options(self): self.domain = 'messages' self.input_file = None self.output_dir = None self.output_file = None + self.omit_header = False self.locale = None self.width = None self.no_wrap = False self.ignore_obsolete = False self.no_fuzzy_matching = False + self.update_header_comment = False self.previous = False def finalize_options(self): @@ -550,41 +730,35 @@ def run(self): else: po_files.append((self.locale, self.output_file)) + if not po_files: + raise DistutilsOptionError('no message catalogs found') + domain = self.domain if not domain: domain = os.path.splitext(os.path.basename(self.input_file))[0] - infile = open(self.input_file, 'U') - try: + with open(self.input_file, 'rb') as infile: template = read_po(infile) - finally: - infile.close() - - if not po_files: - raise DistutilsOptionError('no message catalogs found') for locale, filename in po_files: - log.info('updating catalog %r based on %r', filename, - self.input_file) - infile = open(filename, 'U') - try: + self.log.info('updating catalog %s based on %s', filename, self.input_file) + with open(filename, 'rb') as infile: catalog = read_po(infile, locale=locale, domain=domain) - finally: - infile.close() - catalog.update(template, self.no_fuzzy_matching) + catalog.update( + template, self.no_fuzzy_matching, + update_header_comment=self.update_header_comment + ) tmpname = os.path.join(os.path.dirname(filename), tempfile.gettempprefix() + os.path.basename(filename)) - tmpfile = open(tmpname, 'w') try: - try: + with open(tmpname, 'wb') as tmpfile: write_po(tmpfile, catalog, + omit_header=self.omit_header, ignore_obsolete=self.ignore_obsolete, include_previous=self.previous, width=self.width) - finally: - tmpfile.close() except: os.remove(tmpname) raise @@ -614,17 +788,30 @@ class CommandLineInterface(object): commands = { 'compile': 'compile message catalogs to MO files', 'extract': 'extract messages from source files and generate a POT file', - 'init': 'create new message catalogs from a POT file', - 'update': 'update existing message catalogs from a POT file' + 'init': 'create new message catalogs from a POT file', + 'update': 'update existing message catalogs from a POT file' } - def run(self, argv=sys.argv): + command_classes = { + 'compile': compile_catalog, + 'extract': extract_messages, + 'init': init_catalog, + 'update': update_catalog, + } + + log = None # Replaced on instance level + + def run(self, argv=None): """Main entry point of the command-line interface. :param argv: list of arguments passed on the command-line """ - self.parser = OptionParser(usage=self.usage % ('command', '[args]'), - version=self.version) + + if argv is None: + argv = sys.argv + + self.parser = optparse.OptionParser(usage=self.usage % ('command', '[args]'), + version=self.version) self.parser.disable_interspersed_args() self.parser.print_help = self._help self.parser.add_option('--list-locales', dest='list_locales', @@ -662,7 +849,8 @@ def run(self, argv=sys.argv): if cmdname not in self.commands: self.parser.error('unknown command "%s"' % cmdname) - return getattr(self, cmdname)(args[1:]) + cmdinst = self._configure_command(cmdname, args[1:]) + return cmdinst.run() def _configure_logging(self, loglevel): self.log = logging.getLogger('babel') @@ -688,463 +876,53 @@ def _help(self): for name, description in commands: print(format % (name, description)) - def compile(self, argv): - """Subcommand for compiling a message catalog to a MO file. - - :param argv: the command arguments - :since: version 0.9 - """ - parser = OptionParser(usage=self.usage % ('compile', ''), - description=self.commands['compile']) - parser.add_option('--domain', '-D', dest='domain', - help="domain of MO and PO files (default '%default')") - parser.add_option('--directory', '-d', dest='directory', - metavar='DIR', help='base directory of catalog files') - parser.add_option('--locale', '-l', dest='locale', metavar='LOCALE', - help='locale of the catalog') - parser.add_option('--input-file', '-i', dest='input_file', - metavar='FILE', help='name of the input file') - parser.add_option('--output-file', '-o', dest='output_file', - metavar='FILE', - help="name of the output file (default " - "'//LC_MESSAGES/" - ".mo')") - parser.add_option('--use-fuzzy', '-f', dest='use_fuzzy', - action='store_true', - help='also include fuzzy translations (default ' - '%default)') - parser.add_option('--statistics', dest='statistics', - action='store_true', - help='print statistics about translations') - - parser.set_defaults(domain='messages', use_fuzzy=False, - compile_all=False, statistics=False) - options, args = parser.parse_args(argv) - - po_files = [] - mo_files = [] - if not options.input_file: - if not options.directory: - parser.error('you must specify either the input file or the ' - 'base directory') - if options.locale: - po_files.append((options.locale, - os.path.join(options.directory, - options.locale, 'LC_MESSAGES', - options.domain + '.po'))) - mo_files.append(os.path.join(options.directory, options.locale, - 'LC_MESSAGES', - options.domain + '.mo')) - else: - for locale in os.listdir(options.directory): - po_file = os.path.join(options.directory, locale, - 'LC_MESSAGES', options.domain + '.po') - if os.path.exists(po_file): - po_files.append((locale, po_file)) - mo_files.append(os.path.join(options.directory, locale, - 'LC_MESSAGES', - options.domain + '.mo')) - else: - po_files.append((options.locale, options.input_file)) - if options.output_file: - mo_files.append(options.output_file) - else: - if not options.directory: - parser.error('you must specify either the input file or ' - 'the base directory') - mo_files.append(os.path.join(options.directory, options.locale, - 'LC_MESSAGES', - options.domain + '.mo')) - if not po_files: - parser.error('no message catalogs found') - - for idx, (locale, po_file) in enumerate(po_files): - mo_file = mo_files[idx] - infile = open(po_file, 'r') - try: - catalog = read_po(infile, locale) - finally: - infile.close() - - if options.statistics: - translated = 0 - for message in list(catalog)[1:]: - if message.string: - translated +=1 - percentage = 0 - if len(catalog): - percentage = translated * 100 // len(catalog) - self.log.info("%d of %d messages (%d%%) translated in %r", - translated, len(catalog), percentage, po_file) - - if catalog.fuzzy and not options.use_fuzzy: - self.log.warning('catalog %r is marked as fuzzy, skipping', - po_file) - continue - - for message, errors in catalog.check(): - for error in errors: - self.log.error('error: %s:%d: %s', po_file, message.lineno, - error) - - self.log.info('compiling catalog %r to %r', po_file, mo_file) - - outfile = open(mo_file, 'wb') - try: - write_mo(outfile, catalog, use_fuzzy=options.use_fuzzy) - finally: - outfile.close() - - def extract(self, argv): - """Subcommand for extracting messages from source files and generating - a POT file. - - :param argv: the command arguments + def _configure_command(self, cmdname, argv): """ - parser = OptionParser(usage=self.usage % ('extract', 'dir1 ...'), - description=self.commands['extract']) - parser.add_option('--charset', dest='charset', - help='charset to use in the output (default ' - '"%default")') - parser.add_option('-k', '--keyword', dest='keywords', action='append', - help='keywords to look for in addition to the ' - 'defaults. You can specify multiple -k flags on ' - 'the command line.') - parser.add_option('--no-default-keywords', dest='no_default_keywords', - action='store_true', - help="do not include the default keywords") - parser.add_option('--mapping', '-F', dest='mapping_file', - help='path to the extraction mapping file') - parser.add_option('--no-location', dest='no_location', - action='store_true', - help='do not include location comments with filename ' - 'and line number') - parser.add_option('--omit-header', dest='omit_header', - action='store_true', - help='do not include msgid "" entry in header') - parser.add_option('-o', '--output', dest='output', - help='path to the output POT file') - parser.add_option('-w', '--width', dest='width', type='int', - help="set output line width (default 76)") - parser.add_option('--no-wrap', dest='no_wrap', action='store_true', - help='do not break long message lines, longer than ' - 'the output line width, into several lines') - parser.add_option('--sort-output', dest='sort_output', - action='store_true', - help='generate sorted output (default False)') - parser.add_option('--sort-by-file', dest='sort_by_file', - action='store_true', - help='sort output by file location (default False)') - parser.add_option('--msgid-bugs-address', dest='msgid_bugs_address', - metavar='EMAIL@ADDRESS', - help='set report address for msgid') - parser.add_option('--copyright-holder', dest='copyright_holder', - help='set copyright holder in output') - parser.add_option('--project', dest='project', - help='set project name in output') - parser.add_option('--version', dest='version', - help='set project version in output') - parser.add_option('--add-comments', '-c', dest='comment_tags', - metavar='TAG', action='append', - help='place comment block with TAG (or those ' - 'preceding keyword lines) in output file. One ' - 'TAG per argument call') - parser.add_option('--strip-comment-tags', '-s', - dest='strip_comment_tags', action='store_true', - help='Strip the comment tags from the comments.') - - parser.set_defaults(charset='utf-8', keywords=[], - no_default_keywords=False, no_location=False, - omit_header = False, width=None, no_wrap=False, - sort_output=False, sort_by_file=False, - comment_tags=[], strip_comment_tags=False) - options, args = parser.parse_args(argv) - if not args: - parser.error('incorrect number of arguments') - - keywords = DEFAULT_KEYWORDS.copy() - if options.no_default_keywords: - if not options.keywords: - parser.error('you must specify new keywords if you disable the ' - 'default ones') - keywords = {} - if options.keywords: - keywords.update(parse_keywords(options.keywords)) - - if options.mapping_file: - fileobj = open(options.mapping_file, 'U') - try: - method_map, options_map = parse_mapping(fileobj) - finally: - fileobj.close() - else: - method_map = DEFAULT_MAPPING - options_map = {} - - if options.width and options.no_wrap: - parser.error("'--no-wrap' and '--width' are mutually exclusive.") - elif not options.width and not options.no_wrap: - options.width = 76 - - if options.sort_output and options.sort_by_file: - parser.error("'--sort-output' and '--sort-by-file' are mutually " - "exclusive") - - catalog = Catalog(project=options.project, - version=options.version, - msgid_bugs_address=options.msgid_bugs_address, - copyright_holder=options.copyright_holder, - charset=options.charset) - - for dirname in args: - if not os.path.isdir(dirname): - parser.error('%r is not a directory' % dirname) - - def callback(filename, method, options): - if method == 'ignore': - return - filepath = os.path.normpath(os.path.join(dirname, filename)) - optstr = '' - if options: - optstr = ' (%s)' % ', '.join(['%s="%s"' % (k, v) for - k, v in options.items()]) - self.log.info('extracting messages from %s%s', filepath, - optstr) - - extracted = extract_from_dir(dirname, method_map, options_map, - keywords, options.comment_tags, - callback=callback, - strip_comment_tags= - options.strip_comment_tags) - for filename, lineno, message, comments, context in extracted: - filepath = os.path.normpath(os.path.join(dirname, filename)) - catalog.add(message, None, [(filepath, lineno)], - auto_comments=comments, context=context) - - catalog_charset = catalog.charset - if options.output not in (None, '-'): - self.log.info('writing PO template file to %s' % options.output) - outfile = open(options.output, 'wb') - close_output = True - else: - outfile = sys.stdout - - # This is a bit of a hack on Python 3. stdout is a text stream so - # we need to find the underlying file when we write the PO. In - # later versions of Babel we want the write_po function to accept - # text or binary streams and automatically adjust the encoding. - if not PY2 and hasattr(outfile, 'buffer'): - catalog.charset = outfile.encoding - outfile = outfile.buffer.raw - - close_output = False - - try: - write_po(outfile, catalog, width=options.width, - no_location=options.no_location, - omit_header=options.omit_header, - sort_output=options.sort_output, - sort_by_file=options.sort_by_file) - finally: - if close_output: - outfile.close() - catalog.charset = catalog_charset - - def init(self, argv): - """Subcommand for creating new message catalogs from a template. - - :param argv: the command arguments - """ - parser = OptionParser(usage=self.usage % ('init', ''), - description=self.commands['init']) - parser.add_option('--domain', '-D', dest='domain', - help="domain of PO file (default '%default')") - parser.add_option('--input-file', '-i', dest='input_file', - metavar='FILE', help='name of the input file') - parser.add_option('--output-dir', '-d', dest='output_dir', - metavar='DIR', help='path to output directory') - parser.add_option('--output-file', '-o', dest='output_file', - metavar='FILE', - help="name of the output file (default " - "'//LC_MESSAGES/" - ".po')") - parser.add_option('--locale', '-l', dest='locale', metavar='LOCALE', - help='locale for the new localized catalog') - parser.add_option('-w', '--width', dest='width', type='int', - help="set output line width (default 76)") - parser.add_option('--no-wrap', dest='no_wrap', action='store_true', - help='do not break long message lines, longer than ' - 'the output line width, into several lines') - - parser.set_defaults(domain='messages') - options, args = parser.parse_args(argv) - - if not options.locale: - parser.error('you must provide a locale for the new catalog') - try: - locale = Locale.parse(options.locale) - except UnknownLocaleError as e: - parser.error(e) - - if not options.input_file: - parser.error('you must specify the input file') - - if not options.output_file and not options.output_dir: - parser.error('you must specify the output file or directory') - - if not options.output_file: - options.output_file = os.path.join(options.output_dir, - options.locale, 'LC_MESSAGES', - options.domain + '.po') - if not os.path.exists(os.path.dirname(options.output_file)): - os.makedirs(os.path.dirname(options.output_file)) - if options.width and options.no_wrap: - parser.error("'--no-wrap' and '--width' are mutually exclusive.") - elif not options.width and not options.no_wrap: - options.width = 76 - - infile = open(options.input_file, 'r') - try: - # Although reading from the catalog template, read_po must be fed - # the locale in order to correctly calculate plurals - catalog = read_po(infile, locale=options.locale) - finally: - infile.close() - - catalog.locale = locale - catalog.revision_date = datetime.now(LOCALTZ) - - self.log.info('creating catalog %r based on %r', options.output_file, - options.input_file) - - outfile = open(options.output_file, 'wb') - try: - write_po(outfile, catalog, width=options.width) - finally: - outfile.close() - - def update(self, argv): - """Subcommand for updating existing message catalogs from a template. - - :param argv: the command arguments - :since: version 0.9 + :type cmdname: str + :type argv: list[str] """ - parser = OptionParser(usage=self.usage % ('update', ''), - description=self.commands['update']) - parser.add_option('--domain', '-D', dest='domain', - help="domain of PO file (default '%default')") - parser.add_option('--input-file', '-i', dest='input_file', - metavar='FILE', help='name of the input file') - parser.add_option('--output-dir', '-d', dest='output_dir', - metavar='DIR', help='path to output directory') - parser.add_option('--output-file', '-o', dest='output_file', - metavar='FILE', - help="name of the output file (default " - "'//LC_MESSAGES/" - ".po')") - parser.add_option('--locale', '-l', dest='locale', metavar='LOCALE', - help='locale of the translations catalog') - parser.add_option('-w', '--width', dest='width', type='int', - help="set output line width (default 76)") - parser.add_option('--no-wrap', dest='no_wrap', action = 'store_true', - help='do not break long message lines, longer than ' - 'the output line width, into several lines') - parser.add_option('--ignore-obsolete', dest='ignore_obsolete', - action='store_true', - help='do not include obsolete messages in the output ' - '(default %default)') - parser.add_option('--no-fuzzy-matching', '-N', dest='no_fuzzy_matching', - action='store_true', - help='do not use fuzzy matching (default %default)') - parser.add_option('--previous', dest='previous', action='store_true', - help='keep previous msgids of translated messages ' - '(default %default)') - - parser.set_defaults(domain='messages', ignore_obsolete=False, - no_fuzzy_matching=False, previous=False) + cmdclass = self.command_classes[cmdname] + cmdinst = cmdclass() + if self.log: + cmdinst.log = self.log # Use our logger, not distutils'. + assert isinstance(cmdinst, Command) + cmdinst.initialize_options() + + parser = optparse.OptionParser( + usage=self.usage % (cmdname, ''), + description=self.commands[cmdname] + ) + as_args = getattr(cmdclass, "as_args", ()) + for long, short, help in cmdclass.user_options: + name = long.strip("=") + default = getattr(cmdinst, name.replace('-', '_')) + strs = ["--%s" % name] + if short: + strs.append("-%s" % short) + strs.extend(cmdclass.option_aliases.get(name, ())) + choices = cmdclass.option_choices.get(name, None) + if name == as_args: + parser.usage += "<%s>" % name + elif name in cmdclass.boolean_options: + parser.add_option(*strs, action="store_true", help=help) + elif name in cmdclass.multiple_value_options: + parser.add_option(*strs, action="append", help=help, choices=choices) + else: + parser.add_option(*strs, help=help, default=default, choices=choices) options, args = parser.parse_args(argv) - if not options.input_file: - parser.error('you must specify the input file') - if not options.output_file and not options.output_dir: - parser.error('you must specify the output file or directory') - if options.output_file and not options.locale: - parser.error('you must specify the locale') - if options.no_fuzzy_matching and options.previous: - options.previous = False + if as_args: + setattr(options, as_args.replace('-', '_'), args) - po_files = [] - if not options.output_file: - if options.locale: - po_files.append((options.locale, - os.path.join(options.output_dir, - options.locale, 'LC_MESSAGES', - options.domain + '.po'))) - else: - for locale in os.listdir(options.output_dir): - po_file = os.path.join(options.output_dir, locale, - 'LC_MESSAGES', - options.domain + '.po') - if os.path.exists(po_file): - po_files.append((locale, po_file)) - else: - po_files.append((options.locale, options.output_file)) - - domain = options.domain - if not domain: - domain = os.path.splitext(os.path.basename(options.input_file))[0] + for key, value in vars(options).items(): + setattr(cmdinst, key, value) - infile = open(options.input_file, 'U') try: - template = read_po(infile) - finally: - infile.close() + cmdinst.ensure_finalized() + except DistutilsOptionError as err: + parser.error(str(err)) - if not po_files: - parser.error('no message catalogs found') - - if options.width and options.no_wrap: - parser.error("'--no-wrap' and '--width' are mutually exclusive.") - elif not options.width and not options.no_wrap: - options.width = 76 - for locale, filename in po_files: - self.log.info('updating catalog %r based on %r', filename, - options.input_file) - infile = open(filename, 'U') - try: - catalog = read_po(infile, locale=locale, domain=domain) - finally: - infile.close() - - catalog.update(template, options.no_fuzzy_matching) - - tmpname = os.path.join(os.path.dirname(filename), - tempfile.gettempprefix() + - os.path.basename(filename)) - tmpfile = open(tmpname, 'w') - try: - try: - write_po(tmpfile, catalog, - ignore_obsolete=options.ignore_obsolete, - include_previous=options.previous, - width=options.width) - finally: - tmpfile.close() - except: - os.remove(tmpname) - raise - - try: - os.rename(tmpname, filename) - except OSError: - # We're probably on Windows, which doesn't support atomic - # renames, at least not through Python - # If the error is in fact due to a permissions problem, that - # same error is going to be raised from one of the following - # operations - os.remove(filename) - shutil.copy(tmpname, filename) - os.remove(tmpname) + return cmdinst def main(): @@ -1154,7 +932,7 @@ def main(): def parse_mapping(fileobj, filename=None): """Parse an extraction method mapping from a file-like object. - >>> buf = BytesIO(b''' + >>> buf = StringIO(''' ... [extractors] ... custom = mypackage.module:myfunc ... @@ -1205,8 +983,13 @@ def parse_mapping(fileobj, filename=None): options_map = {} parser = RawConfigParser() - parser._sections = odict(parser._sections) # We need ordered sections - parser.readfp(fileobj, filename) + parser._sections = OrderedDict(parser._sections) # We need ordered sections + + if PY2: + parser.readfp(fileobj, filename) + else: + parser.read_file(fileobj, filename) + for section in parser.sections(): if section == 'extractors': extractors = dict(parser.items(section)) @@ -1221,16 +1004,15 @@ def parse_mapping(fileobj, filename=None): method = extractors[method] method_map[idx] = (pattern, method) - return (method_map, options_map) + return method_map, options_map def parse_keywords(strings=[]): """Parse keywords specifications from the given list of strings. - >>> kw = parse_keywords(['_', 'dgettext:2', 'dngettext:2,3', 'pgettext:1c,2']).items() - >>> kw.sort() + >>> kw = sorted(parse_keywords(['_', 'dgettext:2', 'dngettext:2,3', 'pgettext:1c,2']).items()) >>> for keyword, indices in kw: - ... print (keyword, indices) + ... print((keyword, indices)) ('_', None) ('dgettext', (2,)) ('dngettext', (2, 3)) diff --git a/babel/messages/jslexer.py b/babel/messages/jslexer.py index 22c6e1f9c..c57b1213f 100644 --- a/babel/messages/jslexer.py +++ b/babel/messages/jslexer.py @@ -6,60 +6,73 @@ A simple JavaScript 1.5 lexer which is used for the JavaScript extractor. - :copyright: (c) 2013 by the Babel Team. + :copyright: (c) 2013-2021 by the Babel Team. :license: BSD, see LICENSE for more details. """ - -from operator import itemgetter +from collections import namedtuple import re from babel._compat import unichr -operators = [ +operators = sorted([ '+', '-', '*', '%', '!=', '==', '<', '>', '<=', '>=', '=', '+=', '-=', '*=', '%=', '<<', '>>', '>>>', '<<=', '>>=', '>>>=', '&', '&=', '|', '|=', '&&', '||', '^', '^=', '(', ')', '[', ']', '{', '}', '!', '--', '++', '~', ',', ';', '.', ':' -] -operators.sort(key=lambda a: -len(a)) +], key=len, reverse=True) escapes = {'b': '\b', 'f': '\f', 'n': '\n', 'r': '\r', 't': '\t'} -rules = [ - (None, re.compile(r'\s+(?u)')), +name_re = re.compile(r'[\w$_][\w\d$_]*', re.UNICODE) +dotted_name_re = re.compile(r'[\w$_][\w\d$_.]*[\w\d$_.]', re.UNICODE) +division_re = re.compile(r'/=?') +regex_re = re.compile(r'/(?:[^/\\]*(?:\\.[^/\\]*)*)/[a-zA-Z]*', re.DOTALL) +line_re = re.compile(r'(\r\n|\n|\r)') +line_join_re = re.compile(r'\\' + line_re.pattern) +uni_escape_re = re.compile(r'[a-fA-F0-9]{1,4}') + +Token = namedtuple('Token', 'type value lineno') + +_rules = [ + (None, re.compile(r'\s+', re.UNICODE)), (None, re.compile(r'