From e8377a5b266ec8082ea2f4ff9ce78e5a6fbe1dcd Mon Sep 17 00:00:00 2001 From: mi804 <1576993271@qq.com> Date: Sun, 27 Sep 2026 15:28:20 +0800 Subject: [PATCH 1/2] update code and docs --- .github/workflows/publish.yml | 112 ++ .gitignore | 27 + LICENSE | 2 +- README.md | 140 ++ README_zh.md | 131 ++ docs/assets/entropack-pipeline.png | Bin 0 -> 214065 bytes docs/en/.readthedocs.yaml | 17 + docs/en/API_Reference/index.md | 171 +++ docs/en/Makefile | 14 + docs/en/Principles/DFloat11.md | 52 + docs/en/Principles/Lattice-rANS.md | 76 ++ docs/en/Principles/Tile-ANS.md | 57 + docs/en/Usage/Configuration.md | 100 ++ docs/en/Usage/Linear-layers.md | 163 +++ docs/en/Usage/Quick-start.md | 84 ++ docs/en/Usage/Tensor-compression.md | 116 ++ docs/en/conf.py | 43 + docs/en/index.rst | 27 + docs/requirements.txt | 7 + docs/zh/.readthedocs.yaml | 17 + docs/zh/API_Reference/index.md | 151 +++ docs/zh/Makefile | 14 + docs/zh/Principles/DFloat11.md | 45 + docs/zh/Principles/Lattice-rANS.md | 64 + docs/zh/Principles/Tile-ANS.md | 48 + docs/zh/Usage/Configuration.md | 97 ++ docs/zh/Usage/Linear-layers.md | 155 +++ docs/zh/Usage/Quick-start.md | 81 ++ docs/zh/Usage/Tensor-compression.md | 110 ++ docs/zh/conf.py | 43 + docs/zh/index.rst | 27 + entropack/__init__.py | 15 + entropack/backends/__init__.py | 19 + entropack/backends/cuda/__init__.py | 16 + entropack/backends/cuda/device.py | 96 ++ entropack/backends/cuda/kernels.py | 70 + entropack/compression/__init__.py | 4 + entropack/compression/api.py | 73 ++ entropack/compression/compressed_tensor.py | 164 +++ entropack/linear/__init__.py | 5 + entropack/linear/linear.py | 499 ++++++++ entropack/linear/quant_kernels.py | 108 ++ entropack/linear/utils.py | 41 + entropack/registry.py | 121 ++ entropack/schemes/__init__.py | 12 + entropack/schemes/base.py | 124 ++ entropack/schemes/checks.py | 33 + entropack/schemes/config.py | 111 ++ entropack/schemes/dfloat11/__init__.py | 52 + entropack/schemes/dfloat11/cuda.py | 223 ++++ entropack/schemes/dfloat11/dfloat11.cu | 740 +++++++++++ entropack/schemes/dfloat11/eager.py | 239 ++++ entropack/schemes/dfloat11/format.py | 146 +++ entropack/schemes/lattice_rans/__init__.py | 60 + entropack/schemes/lattice_rans/cuda.py | 670 ++++++++++ entropack/schemes/lattice_rans/eager.py | 405 ++++++ entropack/schemes/lattice_rans/format.py | 309 +++++ .../schemes/lattice_rans/lattice_rans.cu | 1140 +++++++++++++++++ entropack/schemes/lattice_rans/rans.py | 209 +++ entropack/schemes/lattice_rans/rdo.py | 162 +++ entropack/schemes/tile_ans/__init__.py | 55 + entropack/schemes/tile_ans/cuda.py | 170 +++ entropack/schemes/tile_ans/device.cuh | 90 ++ entropack/schemes/tile_ans/eager.py | 307 +++++ entropack/schemes/tile_ans/format.py | 164 +++ entropack/schemes/tile_ans/tile_ans.cu | 469 +++++++ pyproject.toml | 36 + 67 files changed, 9347 insertions(+), 1 deletion(-) create mode 100644 .github/workflows/publish.yml create mode 100644 .gitignore create mode 100644 README.md create mode 100644 README_zh.md create mode 100644 docs/assets/entropack-pipeline.png create mode 100644 docs/en/.readthedocs.yaml create mode 100644 docs/en/API_Reference/index.md create mode 100644 docs/en/Makefile create mode 100644 docs/en/Principles/DFloat11.md create mode 100644 docs/en/Principles/Lattice-rANS.md create mode 100644 docs/en/Principles/Tile-ANS.md create mode 100644 docs/en/Usage/Configuration.md create mode 100644 docs/en/Usage/Linear-layers.md create mode 100644 docs/en/Usage/Quick-start.md create mode 100644 docs/en/Usage/Tensor-compression.md create mode 100644 docs/en/conf.py create mode 100644 docs/en/index.rst create mode 100644 docs/requirements.txt create mode 100644 docs/zh/.readthedocs.yaml create mode 100644 docs/zh/API_Reference/index.md create mode 100644 docs/zh/Makefile create mode 100644 docs/zh/Principles/DFloat11.md create mode 100644 docs/zh/Principles/Lattice-rANS.md create mode 100644 docs/zh/Principles/Tile-ANS.md create mode 100644 docs/zh/Usage/Configuration.md create mode 100644 docs/zh/Usage/Linear-layers.md create mode 100644 docs/zh/Usage/Quick-start.md create mode 100644 docs/zh/Usage/Tensor-compression.md create mode 100644 docs/zh/conf.py create mode 100644 docs/zh/index.rst create mode 100644 entropack/__init__.py create mode 100644 entropack/backends/__init__.py create mode 100644 entropack/backends/cuda/__init__.py create mode 100644 entropack/backends/cuda/device.py create mode 100644 entropack/backends/cuda/kernels.py create mode 100644 entropack/compression/__init__.py create mode 100644 entropack/compression/api.py create mode 100644 entropack/compression/compressed_tensor.py create mode 100644 entropack/linear/__init__.py create mode 100644 entropack/linear/linear.py create mode 100644 entropack/linear/quant_kernels.py create mode 100644 entropack/linear/utils.py create mode 100644 entropack/registry.py create mode 100644 entropack/schemes/__init__.py create mode 100644 entropack/schemes/base.py create mode 100644 entropack/schemes/checks.py create mode 100644 entropack/schemes/config.py create mode 100644 entropack/schemes/dfloat11/__init__.py create mode 100644 entropack/schemes/dfloat11/cuda.py create mode 100644 entropack/schemes/dfloat11/dfloat11.cu create mode 100644 entropack/schemes/dfloat11/eager.py create mode 100644 entropack/schemes/dfloat11/format.py create mode 100644 entropack/schemes/lattice_rans/__init__.py create mode 100644 entropack/schemes/lattice_rans/cuda.py create mode 100644 entropack/schemes/lattice_rans/eager.py create mode 100644 entropack/schemes/lattice_rans/format.py create mode 100644 entropack/schemes/lattice_rans/lattice_rans.cu create mode 100644 entropack/schemes/lattice_rans/rans.py create mode 100644 entropack/schemes/lattice_rans/rdo.py create mode 100644 entropack/schemes/tile_ans/__init__.py create mode 100644 entropack/schemes/tile_ans/cuda.py create mode 100644 entropack/schemes/tile_ans/device.cuh create mode 100644 entropack/schemes/tile_ans/eager.py create mode 100644 entropack/schemes/tile_ans/format.py create mode 100644 entropack/schemes/tile_ans/tile_ans.cu create mode 100644 pyproject.toml diff --git a/.github/workflows/publish.yml b/.github/workflows/publish.yml new file mode 100644 index 0000000..a66ddcb --- /dev/null +++ b/.github/workflows/publish.yml @@ -0,0 +1,112 @@ +name: Build and publish distributions + +on: + push: + branches: [main] + tags: ["v*"] + pull_request: + branches: [main] + workflow_dispatch: + +permissions: + contents: read + +concurrency: + group: ${{ github.workflow }}-${{ github.ref }} + cancel-in-progress: false + +jobs: + build: + name: Build wheel and source distribution + runs-on: ubuntu-latest + timeout-minutes: 10 + steps: + - uses: actions/checkout@v7 + with: + persist-credentials: false + + - uses: actions/setup-python@v7 + with: + python-version: "3.11" + + - name: Check release version + if: github.event_name == 'push' && startsWith(github.ref, 'refs/tags/v') + run: | + python - <<'PY' + import os + import tomllib + from pathlib import Path + + config = tomllib.loads(Path("pyproject.toml").read_text()) + expected = "v" + config["project"]["version"] + actual = os.environ["GITHUB_REF_NAME"] + if actual != expected: + raise SystemExit(f"Tag {actual} does not match package version {expected}") + PY + + - name: Install build tools + run: python -m pip install --upgrade build twine + + - name: Build distributions + run: python -m build + + - name: Check package metadata + run: python -m twine check --strict dist/* + + - name: Check packaged CUDA sources + run: | + python - <<'PY' + from pathlib import Path + import tarfile + import zipfile + + required = { + path.as_posix() + for path in Path("entropack").rglob("*") + if path.suffix in {".cu", ".cuh"} + } + if not required: + raise SystemExit("No CUDA sources found in the checkout") + wheels = list(Path("dist").glob("*.whl")) + sources = list(Path("dist").glob("*.tar.gz")) + if len(wheels) != 1 or len(sources) != 1: + raise SystemExit("Expected one wheel and one source distribution") + with zipfile.ZipFile(wheels[0]) as archive: + wheel_files = set(archive.namelist()) + with tarfile.open(sources[0]) as archive: + source_files = {name.partition("/")[2] for name in archive.getnames()} + for name, files in (("wheel", wheel_files), ("source distribution", source_files)): + missing = required - files + if missing: + raise SystemExit(f"Missing CUDA sources in {name}: {sorted(missing)}") + print(f"Verified {len(required)} CUDA source files in both distributions") + PY + + - name: Upload distributions + uses: actions/upload-artifact@v7 + with: + name: python-distributions + path: dist/* + if-no-files-found: error + retention-days: 14 + + publish: + name: Publish to PyPI + needs: build + if: github.event_name == 'push' && startsWith(github.ref, 'refs/tags/v') + runs-on: ubuntu-latest + timeout-minutes: 10 + environment: + name: pypi + url: https://pypi.org/project/entropack/ + permissions: + id-token: write + steps: + - name: Download distributions + uses: actions/download-artifact@v8 + with: + name: python-distributions + path: dist/ + + - name: Publish to PyPI + uses: pypa/gh-action-pypi-publish@release/v1 diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..c71c93c --- /dev/null +++ b/.gitignore @@ -0,0 +1,27 @@ +# Python +__pycache__/ +*.py[cod] +*.egg-info/ +build/ +dist/ +.pytest_cache/ + +# Environments +.venv/ +venv/ + +# Editors / tools +.idea/ +.vscode/ +.qoder/ +.claude/ + +# Benchmark artifacts +*.log + +# Sphinx +docs/_build/ +docs/*/_build/ + +# Local tests +/tests/ diff --git a/LICENSE b/LICENSE index 261eeb9..84a34e9 100644 --- a/LICENSE +++ b/LICENSE @@ -186,7 +186,7 @@ same "printed page" as the copyright notice for easier identification within third-party archives. - Copyright [yyyy] [name of copyright owner] + Copyright [2026] [ModelScope] Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file except in compliance with the License. diff --git a/README.md b/README.md new file mode 100644 index 0000000..a5a120a --- /dev/null +++ b/README.md @@ -0,0 +1,140 @@ +# EntroPack + +### General-purpose tensor compression for PyTorch + +EntroPack is a general-purpose tensor compression library for PyTorch. It supports lossless +compression for exact recovery and lossy compression with a target bitrate to balance storage +and reconstruction accuracy. GPU encoding and decoding compress tensors and restore them +in their original shape and dtype. + +[![License](https://img.shields.io/badge/license-Apache_2.0-blue.svg)](LICENSE) +![Python](https://img.shields.io/badge/python-%3E%3D3.10-blue.svg) + +[Documentation](docs/en/index.rst) · [中文](README_zh.md) + +- **Flexible bitrates.** Compress each weight matrix at any non-integer target bitrate, + or preserve every input bit with a lossless scheme. +- **Dtype preservation.** Restore tensors in their input dtype, including BF16, FP16, FP8, + and INT8. +- **PyTorch integration.** Compress and restore tensors through a common API, and save + them with `state_dict`. Compressed linear layers provide an integration for model weights. + +## Installation + +Use Python 3.10 or later and install a CUDA-enabled build of PyTorch 2.10 or later +for your environment. + +### Install from source (recommended) + +```bash +git clone https://github.com/modelscope/entropack.git +cd entropack +pip install -e ".[cuda13]" +``` + +### Install from PyPI + +PyPI releases may lag behind source updates. Install from source for the latest features. + +```bash +pip install "entropack[cuda13]" +``` + +Both installation methods above use CUDA 13 and include the matching CuPy package. +For CUDA 12, replace `cuda13` with `cuda12` in either command. If a compatible CuPy is +already installed, use `pip install -e .` for source installation or `pip install entropack` for PyPI. + +See [Quick start](docs/en/Usage/Quick-start.md) for environment requirements and usage examples. + +## Get started + +### Direct tensor compression + +This example compresses a 2D BF16 tensor at a target of 3.5 bits per element, +then decompresses it to a tensor with the original shape and dtype: + +```python +import torch +import entropack as ep + +tensor = (torch.randn(256, 256, device="cuda") * 0.02).to(torch.bfloat16) +config = ep.LatticeRANSConfig(target_bpp=3.5) + +compressed = ep.compress(tensor, config) +restored = ep.decompress(compressed, config) + +print(f"Target: {config.target_bpp:.2f} bits per element") +print(f"Stored: {compressed.actual_bpp:.2f} bits per element") +print(restored.shape, restored.dtype) +``` + +`target_bpp` is the requested number of bits per element (bpp). `actual_bpp` reports the stored +rate, including metadata. Targets from 1 to 11 are supported, including non-integer values. + +### Compressed Linear + +`CompressedLinear.from_linear` compresses an existing `torch.nn.Linear`'s weights using +the supplied Config and returns a new Compressed Linear. Call it as `layer(x)` to compute +the output, just as with an ordinary linear layer: + +```python +import torch +import entropack as ep + +linear = torch.nn.Linear(256, 256, dtype=torch.bfloat16, device="cuda") +config = ep.LatticeRANSConfig(target_bpp=4.0) +layer = ep.CompressedLinear.from_linear(linear, config=config) +x = torch.randn(8, 256, dtype=torch.bfloat16, device="cuda") + +with torch.inference_mode(): + output = layer(x) +print(output.shape, f"{layer.compressed_bits:.2f} bits per weight") +``` + +See [Compressed Linear usage](docs/en/Usage/Linear-layers.md) for model replacement and checkpoint examples. + +## Configuration + +Config defines the compression scheme and its settings. It is passed to tensor compression +and decompression functions or to a compressed Linear layer's constructor. + +| Config | Compression and use cases | +| --- | --- | +| `DFloat11Config()` | Specialized lossless compression for BF16 tensors. Decompression restores every input bit, for applications requiring exact recovery. | +| `TileANSConfig()` | Lossless compression for BF16, FP16, FP32, FP8, INT8, and other supported dtypes. The compression ratio depends on the input data distribution. | +| `LatticeRANSConfig(target_bpp=...)` | Lossy compression of 2D floating-point and integer tensors. `target_bpp` specifies the target bits per element, from 1 to 11 including non-integer values, to balance storage size and reconstruction accuracy. | + +See [Compression configuration](docs/en/Usage/Configuration.md) for scheme selection +and the complete parameter reference. + +## Performance + +On one NVIDIA H20, EntroPack compresses the weights of all 276 linear layers in +Z-Image-Turbo's diffusion transformer at a 4 bpp target in **3.5 seconds**. The compressed +weights occupy **4.02 bpp**, with **7.18%** relative L2 reconstruction error. Inference with +these weights takes **544.7 ms** per denoising step, only **7.7%** above the original BF16 +model's 505.7 ms. + +## Documentation + +| Guide | Contents | +| --- | --- | +| [Quick start](docs/en/Usage/Quick-start.md) | Install and run tensor compression and Compressed Linear examples | +| [Compression configuration](docs/en/Usage/Configuration.md) | Choose a scheme and look up supported dtypes and parameters | +| [Tensor compression](docs/en/Usage/Tensor-compression.md) | Encode, decode, inspect storage, move data, and save or load tensors | +| [Compressed Linear usage](docs/en/Usage/Linear-layers.md) | Replace model layers, use low-precision computation, and manage checkpoints | +| [API reference](docs/en/API_Reference/index.md) | Look up functions, classes, and properties | + +Compression principles: [DFloat11](docs/en/Principles/DFloat11.md), [tile-ANS](docs/en/Principles/Tile-ANS.md), +and [EntroPack lattice quantization](docs/en/Principles/Lattice-rANS.md). + +## Acknowledgements + +EntroPack's design is inspired by [DFloat11](https://github.com/LeanModels/DFloat11), +[dahuffman](https://github.com/soxofaan/dahuffman), +[DietGPU](https://github.com/facebookresearch/dietgpu), and +[tile-ANS](https://arxiv.org/abs/2606.15789). + +## License + +[Apache License 2.0](LICENSE). diff --git a/README_zh.md b/README_zh.md new file mode 100644 index 0000000..8d9c1b5 --- /dev/null +++ b/README_zh.md @@ -0,0 +1,131 @@ +# EntroPack + +### 面向 PyTorch 的通用张量压缩 + +EntroPack 是一个面向 PyTorch 的通用张量压缩库,支持完整保留原始数据的无损压缩, +以及通过目标码率控制存储大小与重建精度的有损压缩。EntroPack 提供 GPU 编解码, +将压缩后的张量恢复为原来的形状和数据类型。 + +[![License](https://img.shields.io/badge/license-Apache_2.0-blue.svg)](LICENSE) +![Python](https://img.shields.io/badge/python-%3E%3D3.10-blue.svg) + +[文档](docs/zh/index.rst) · [English](README.md) + +- **灵活设置码率。** 支持每个权重矩阵以任意非整数目标码率压缩,也可以选择逐位保留输入的无损方案。 +- **保留数据类型。** 解压后保留输入的数据类型,包括 BF16、FP16、FP8、INT8 等。 +- **接入 PyTorch。** 通过统一接口压缩和恢复张量,使用 `state_dict` 保存;模型权重还可以通过 Compressed Linear 接入。 + +## 安装 + +需要 Python 3.10 及以上,并先安装与环境匹配的 CUDA 版 PyTorch 2.10 及以上。 + +### 源码安装(推荐) + +```bash +git clone https://github.com/modelscope/entropack.git +cd entropack +pip install -e ".[cuda13]" +``` + +### 从 PyPI 安装 + +PyPI 版本更新可能有所延迟,如需最新功能,推荐从源码安装。 + +```bash +pip install "entropack[cuda13]" +``` + +上述两种安装方式均以 CUDA 13 为例,并包含对应版本的 CuPy。使用 CUDA 12 时, +将命令中的 `cuda13` 改为 `cuda12`。如果已安装匹配的 CuPy,源码安装和 PyPI 安装 +可分别使用 `pip install -e .` 和 `pip install entropack`。 + +环境要求与使用示例见[快速上手](docs/zh/Usage/Quick-start.md)。 + +## 快速开始 + +### 直接压缩张量 + +以下示例将一个二维 BF16 张量以每元素 3.5 bit 为目标压缩, +再解压为相同形状和数据类型的张量: + +```python +import torch +import entropack as ep + +tensor = (torch.randn(256, 256, device="cuda") * 0.02).to(torch.bfloat16) +config = ep.LatticeRANSConfig(target_bpp=3.5) + +compressed = ep.compress(tensor, config) +restored = ep.decompress(compressed, config) + +print(f"Target: {config.target_bpp:.2f} bits per element") +print(f"Stored: {compressed.actual_bpp:.2f} bits per element") +print(restored.shape, restored.dtype) +``` + +`target_bpp` 表示期望的每元素比特数(bpp),`actual_bpp` 返回包含元数据的实际存储码率。 +目标范围为 1–11,支持非整数值。 + +### 使用 Compressed Linear + +`CompressedLinear.from_linear` 按传入的 Config 压缩现有 `torch.nn.Linear` 的权重, +返回一个新的 Compressed Linear。仍可像普通线性层一样,通过 `layer(x)` 计算输出: + +```python +import torch +import entropack as ep + +linear = torch.nn.Linear(256, 256, dtype=torch.bfloat16, device="cuda") +config = ep.LatticeRANSConfig(target_bpp=4.0) +layer = ep.CompressedLinear.from_linear(linear, config=config) +x = torch.randn(8, 256, dtype=torch.bfloat16, device="cuda") + +with torch.inference_mode(): + output = layer(x) +print(output.shape, f"{layer.compressed_bits:.2f} bits per weight") +``` + +模型中的层替换和检查点操作见 [Compressed Linear 使用指南](docs/zh/Usage/Linear-layers.md)。 + +## Config:压缩配置 + +Config 定义压缩方案及其参数,在调用张量编解码函数或构造 Compressed Linear 时传入。 + +| Config | 压缩方式与适用场景 | +| --- | --- | +| `DFloat11Config()` | 专用于 BF16 张量的无损压缩,解压后逐位恢复输入,适合要求精确恢复的场景。 | +| `TileANSConfig()` | 支持 BF16、FP16、FP32、FP8、INT8 等多种数据类型的无损压缩。压缩比取决于输入的数据分布。 | +| `LatticeRANSConfig(target_bpp=...)` | 支持浮点和整数二维张量的有损压缩。`target_bpp` 指定每元素的目标比特数,范围为 1–11,支持非整数值,用于调整存储大小与重建精度之间的取舍。 | + +方案选择和完整参数见[压缩配置](docs/zh/Usage/Configuration.md)。 + +## 性能 + +在单张 NVIDIA H20 上,EntroPack 以 4 bpp 为目标压缩 Z-Image-Turbo 扩散 Transformer +的 276 个线性层权重,耗时 **3.5 秒**。压缩后的实际存储为 **4.02 bpp**, +权重相对 L2 重建误差为 **7.18%**。使用压缩权重推理时,去噪单步耗时为 **544.7 ms**, +相对原始 BF16 模型的 505.7 ms 仅增加 **7.7%**。 + +## 文档 + +| 指南 | 内容 | +| --- | --- | +| [快速上手](docs/zh/Usage/Quick-start.md) | 安装并运行张量压缩与 Compressed Linear 示例 | +| [压缩配置](docs/zh/Usage/Configuration.md) | 选择方案、查看支持类型与完整参数 | +| [通用张量压缩](docs/zh/Usage/Tensor-compression.md) | 编解码、存储统计、设备迁移和保存加载 | +| [Compressed Linear 使用指南](docs/zh/Usage/Linear-layers.md) | 模型替换、低精度计算和检查点使用 | +| [API 参考](docs/zh/API_Reference/index.md) | 查询函数、类与属性 | + +压缩原理:[DFloat11](docs/zh/Principles/DFloat11.md)、[tile-ANS](docs/zh/Principles/Tile-ANS.md)、 +[EntroPack 格量化](docs/zh/Principles/Lattice-rANS.md)。 + +## 致谢 + +EntroPack 的设计受到 [DFloat11](https://github.com/LeanModels/DFloat11)、 +[dahuffman](https://github.com/soxofaan/dahuffman)、 +[DietGPU](https://github.com/facebookresearch/dietgpu) 和 +[tile-ANS](https://arxiv.org/abs/2606.15789) 的启发。 + +## 许可证 + +[Apache License 2.0](LICENSE)。 diff --git a/docs/assets/entropack-pipeline.png b/docs/assets/entropack-pipeline.png new file mode 100644 index 0000000000000000000000000000000000000000..ba30f74c8cec4ebeb91084c15ac4d7ad4cd645d0 GIT binary patch literal 214065 zcmeFZ_g7O}7d4EcVna|+M4HG21O$}cQL2LU5;_XfdoKZ^0-_>CTBJ+wgx-sabO^l@ zkQxXj^iY!UZtvA+yx(8&{qiz0h6c_#IcM*^)|zY1x${9?RgwJioy#O7B;?PY%4w32 zT>L^pa<2d4S@0V%_G4!7&n4%l`fem7Y~Uca%V(Gy)*47i?vgx{d!p??yBa^MK#i=OlbUaA78f^{G^Lx7u>UYOVac`DQMl$yRCesj)MNp4jHlwC zKiB*E`GI((ma&%GX*41$+bU1%72iaaV5q_J3S^Lv$$yTaH-^|)f4aAiXOjrh%zJ>y zZ6FrdIc%)FP~&aZ&0E!1YvNNACJZ-`tA<}Q_^>9lp;v|%F8Vppa)AfGqGE94&*$_% zjbrH0#}66w-VdiV8!T#>B&%|W&xoL(l)fh*LbQ((Tf(xiS7D_@FmY#8Jl@yvqjONF zJswyo_ZQgK)MjPgzf0qrjQ-Xg(WCS7vpj}bv4<;!A;Yq35t80;-7ft3Q0mL93|Ib+ zl|agQGS?=Py0Wr-1AdE@Z!-*UCo5kuC|q5o=^s_&Us5tyo|LW{DE)i<;{|=&yX@k^ zuq7v$E$a+yY%$JJDt>)NoZW_%$7n~nNao4x@7VhHC+t<8-6DQ(tu0~FpqCnEnhM6L z+E0ZIC5CS~A1cTq9Is!S=(V2gzUEvv*85eh?g@wh)PVBa`7lTwLb~KqxfC{Ep+N4h^S|l;;*C0XHbF05Cf@fy>9XUqaUfp!^ zZ+P_6Kj4cjUqTRCS+fj2m@8gu8g%tmmz`*@ ziJ!=P-~Y?f_Dvp@V)Nc&U@rD6I;CY=ZWr?cAy);3g}n5blI9}nz{eb#OEm3 zDw>Cl%~%5N&T%v!yWW7uzr7c$l!bg@WkuKDhEPU_4`|FSD3H6jc|0nWl0C+iRhAAU zU>pe9%?Z4bL$~V>+Udi?hLoj^rRQ_<>PHuzxdLDn__jC82841 zvl&{p>ToY8w>a;}4x5(IuErE0?#3!vURWrjR9#lK#BCSjej~ z*PW^jG3ULXt6JOWohu(RJ*&#fT-`i2howrqB8tJyU}$eT`nKLcvLPmFJkZ21>q&mG zN)~qJLseneFLR_$sSi0zieS;zR?1a=gRR}ogi`;$H!Q316R|oYj>SA1cz;E5I_IhJ zIbxHP%=`6ed=^*05Ni9V9a~*lI!Ai_&&c(J=@=U)Ta1V_f6c|NtQw3oIUPGVHErd7 zZoI}IDY`oVSCRf$6xKWZEWCl@s|4TR7W)#1sXgiTcEkFJ2<&xc~+i8orW z%V_K9tf|bp{ptv+XEHoFsk)FMBJ&>1IHMmXj6&+vBu)KJmzJ?HKl-RT;MgN0?N^pe zno-RmoZZw%El=A(w;KW(lh~P_QjR;#7XHYxkM3(a+MW*R?E#LVlVEr-t0f7lp@_U@1tb3Fgab-w$$W< zZ{;*vnsoZ94)xb0_IeWQxR`0HUmZE|Htc^+^D_0ponHG%FTB)`CCqxojqLC?5;_B@ z#HO+xpi+3VN2$Jl?%~Cg2Toy0(c4y)nhlKxKE5@kTFgSDqh}ll(?-Tg^bFKW78)+k zK;cmGKa=Rz2@gP(EH$lG)AE)mqpxLaSLUrtdRLU>zS-vCL8>Ao#oK?L?PD_*c!XXW z{*(-*2&0vHiYo#OC`2U`N(=WU{@8Vsu50oac9^0X%p+vr)W(!%o6x_Q480Qdo(5)w zC#WwN1`Jh6!Zye4%$NF*Kq?ACs;E)2+-+BoYJn;`A&&vcYAOcz>-1*q z)7{J_a~=L6BNomeI^N#B3vzq@e&X)V6^~tftvO=mG4dqU-}J@ZhGF5lshDjRmKIrw zX=5(V{6W6*@SiC&B}a(iy$7hCT`7XDlR&Q4$!kTXFEcEB(@5m9nbKsJ2aH&2tVXiX zE#J)^K{vOykQINp4OiaREetbVo4~2AO|t6oz}ZupOYE24U7)LqH!|%m(9T!8g^?DN zHpNcx=XhM9_}sm82^|F|lu515TS)EvnYnLedU|&i!O9AtmCpg>H(e><^3We=h+aWXF-@oFlX#7O&$cUh)XI0!_Lqt#A zZf(0|=c|3HO%%Teei545)73fD<$2uWgYs8P6>av%J&}~`F(A&Y^-W8?j1g-clC1y` zf>#Wa7h+{kv@>1LgB6jxEgr76l)BU{?XBE882WK^aNs$p%*5N*G=P!`6!)v!A}dM~ z6s(-(KG`pVGeUCcCH(M%O9Ci97MVae4EMP_+MLkM^SwHZZo6aG2`lM6tZ5K{Zu(O3 z$sTpKMMkXnaMPT|ihVy%g;CfHS?;t^owQ2)8%;mvM?0SN4es_#j*TL)fQSO_{UM%MC5WXIRooG_x2Tx-`es!i1uU4|LsJ#uEc>y8lq z3B3CXUkVEgrF>7$dr&KynL7~XrqK)x>)csOUYnrFbI9aMR`s3W zR@S(|st9TCYDY7MM@V6|t7s2Cfgj;?SVdkW}Ty5rv8y0R#cy`;k#JHxTyOD9zc4!`j~tbfJq zhVyhW#m35$4KJl;^GTy8ag{W zL%Ku&6gSd~Rc#?=F9c$vGrsdBGZ;ED$G27Z$y;R2xRa4Sxyu|B*x2|t4D_yDxw2S; zfnvb=dFh8xnZn)vFki9b?B?RB$t#$e91U?X-c5_uS;>1>P+o4aM`}lq{2nu)DsvGE zC6rH%3=f-V39j4k!?g_U?CE1?G;}dvW<6r^=x*`##R*e~<`?qao&<<~QR{!p;XPOm zAgQ=hmAmbPVfTj+bRJX=4)!Vjdrcfv3bxz|eCXP0&&|-<@H&g7#Z|WBBj4GpsE!tt zij%K`E>bXdsO#}>Z2r!Z;{oS@xQS9@(MNu??L zV{9yBC0yzf#fE#iJ4<|_le;i#uaPXmq%O(QSJbt(0iqq`c#LceiBZa$F$Tc$YqgYQ z#*Y!+jJkK2bU;caXNyJm6bTDH~FqQl6T(cg(LXDzR} zqxdewI_s6?Qe0dtU6+$}W909=@{>uo7{S-j^!_@UH*+P=R&7EV!`d|!(^AhNx<2Xj_%i>#@8@R;?*5e@C1bx<=l2d@BF{ z;Gw!PF>yAryWi2X;$q`!i2@4At~^JB=Db8icD8OcAH|GY{c`tI8~N#CK~~*@6G#~# zicp_ys9u#FzhBPYfSMf5p6d5+Jj`|KJ9y*Z5hB^jnncI%a3DFv1~Nrd=c^BBo(HD8 z?Ug+Ge_4x~Ie_T+-5>L!eu6sq?yM+S905%;byvn87}iQj@X`6T!RUO4+l(OBS3>t#Z zbIB-_5r80kWv=tcq-e8;1}b=Oz%gbxsMI{fk z-_AVSy(bf>h}uE8aHcSBMz>uvJvk97O5%*gz?fi-&Yis1jo&Wu>6;D8qXNv9K`Arw z@jXaF&kXu=i3^{eHkq_p{bgPB1iG&*R_x=#E|P~!CuYWb-(@>!WoLX?(Y>*t=wIm_ z?06h53B!)>R_mq*HX1Jps!HnswqWnjrs(H&+2aX4!&&WQo8y>L_Jux7oAN)j`l4*% z7kv(AY}!@ZK-t&E6%Fqd0%XNOc8XuK-IfA>^o zXG+@Wq-wCkA2P@V>*JpILpLKnEd5`#s^ z_-iT2;z}tPlW1D~)gFNFoPH1mu$4LNpK7x{2X`#(^vyymA-MN;r~y>(QoY*jCF`}E zlem!jHLG0{P{?gcD522$T3#tj{a)gxoth1~%%C0UH~XXmnrEurWljIarrV3pZ<0Qk zznhqtuw`{b`Fv9k08_?l+Ouo7;m|PdlJU;(@L+dLg9#;Dz3+rVK!#PuQvsw+?S(E8 zgi|~`UG1NX0)!2EMQdcC_>!wuy>*1+%VxykWQV&as3wQY(be<}>tWRMo6BW;p}2c3 zvJVO-914q?4}b0FFs8h`uW4arb<)LOF70~~{IQF(5z^~j{`dAAsY9G5jjKJ6ZT4h{PSerWbvUw1)t&bra*FPJCG3Ex zTc~_{i>H`L_>#jV(WaBy_shGFL7r$%`Ke33oH zFOv;@YQP#gsXPb||LDCRAA%zihh+J)FEjI%?_V{tTiR?>Wd8GPU)Z@CyIT{RLpky? za(YMe@$XWbKd#)c_iKNfKL5o$hsUT+Q`2C1etmIaiI$e}#;w8aI=h`IrYX|^v{o-Q zm?Z*?$7wM#=JZqgeAI%AmF7^nl2SK#z2b`xEst(SsO21-F_sZaU=&qua%bh4GcpNy zsqxIl!J#OzAhbTw4*VHa?>WGbGV&GwBORpvY%5E6Tr9M5$9iGb7i2^58K?LqZ~qx| zv*UPr>}`gdrMBBSdpduf;Gt~XosR1=61cXIW+zluJfF|>`-oYe(%^ze+bQkHr24P% zJL?%ZN#ZZm9`~2T2K*00HWWl(r^ug6QfRG|Y%|^e&y~Ji)!*7PRqKsUt9;!BIziW+ z9l(M;YFD+9cL8mU6-qEZJUFhxr5oubebnZS=H=n=$1iz#{O)`|%bpCq`*WQaDGx*< zsfLStQPmM+&+~Nd$MKksMR^=LP9O|8)cR`2ItSi>3u)^Z?=Iw%2e$QoA@ocG`X$ro z8KhYU0`XdPodntXQ=^;edgnQmzZ0GK^r5E0pR2#7>v)@^OC>`Mj|u;>^Z2F42eHPh*|pw8u%!r3|OivaLBwskrivI77<;VYb%Adh_9smFLh}k8Okf>|uumt7i=M@ze zX9@myXO4`FJmE4I4MAHVN-F)Y22%6vW=xFL@=dsoHkt-LaOb}Rg+;+JYQpTPy2V9? zwKs+zap`(^cvPGvnUrpl7gWShu%f{qMxYNX@+b!eS0YA+m>y^FdL+U-xKRouV~~tefKlune8GMCf79 z+_>q%Ex&0>L53exT>bA^hGUd;Of+mJ<3`F{?iFrArw2&A1~&Ah8ho|?Ad z6%KaX6xi`mRO@HU|LE=4CQ4bD^`+)gqR)u`o|3srJj0^6C4(KSFxb-L=lQ9vo{JAohnkr>el=9M=|@ z9exjd3}k>!_}rT$%k7c*n*%L53KXfw{rpZsvl=F;xLWs^F|@yGqTx?gI~gbiV=O1V zq#q}oLLnvwk_T_QWQSFw{v%LiuZac&_e zC(lh80D=Wg7h+-|PZc@15MR0ey_b%RxwO(A1JAcwgcC@~}hwa#L zPu5jtxd4~={9;v03;FVjhi(*$-c_y<(uw~#w6n8$gkBVk+3h*HQp!#>%!{L!Fq4Sf zJ&7}h5Ag+2rhR8f5c+7Y-(*~6J5<#u^FFsYZPx^vVEDHCSR<~l%`$OEz%{NI>k}M-rpx63z&J~bs#If#$oG|Ld?_JvIHTBB_3e8AU(UQI+?6w~V`auDM?b@(lLjo*J~ z{5v5%n34y%XA?~?8sPigOPWEP7NMO#`~d(m6ia=>_B0ESh&Cpf%6(f#0qoDRSi)RI zs)mC$?iAM*?SMB><~X^=U|8kq-!q^EsAJ5Bg%o?x zZGOLB?AV4^PEPHtr1lPlHh-3m5gCM?jQO%n2%WqoQAa!C3s`SYMG^~z`zA6&fyyNF z-tjs!>B&Ll4T|&Th5ZC==6~OS}28F>wrHzB@_=mH|&bS5{7U4`l#t)3`;f z`_+Y(7hv)R-JPqdgg}qf*4@W@f6{I1seyv0g+**cwUu-GH)3OY<{L#oot~F()f8$T zoyit}=&H^TW!phR#>L46n67$7&(#^ho=JnghuJ;rM$B^Z76i8$K;&@9N0&!u%;dF! z{ykMB`Ab5IzJaczg~igZ7qgSy{P(}A0Qwd}gKjh0L^1^K>|6<^jEy~JdR%R{;5F}_ zPI*rl@lBnX%*-s9vQNS_HLZ5Pcp>WMO}MtM&J||*^^Iwi5u&mZq2}^zb(2)FIMYH~ z+ZsnMTfADfynOS1j1qHDMqneybrsp8eIaCU3|B}*tHshX2epzZ69P2oUbajxj%!WZ z{mG0g@3`IkL%oM-cVXW-ox5^zF(}yG%AUmMuNT1G)AIpWfhn8SLF177ZZ6sE-sX&8 zt}z`GLRZx1d=6C~9HK1-I$I0fz5=$Om7yWMyJ(rNST!1+sakBLCi-PAB4TXF%~Ca4 zWU{ME$*bKdKyRNAQ4mZiSr=SnvJ;=Sb0t)1nOCoZ*w?I`4~xYDo))y3U-$wRXJi$b zpmQ)yvz=0hHagDc66;Me)6L%vr_GxU{H)K>hd2IE+a)Fe$9K!!~fM3XB zXBvL~LDUnI!3+P%n#8Xi#`yUqQ;&ZYXm|FFH(yr;fM$?6sH3NIBr&ZBMsO7oeMc&) z3@HRiPTB(hbSPr#7c10`2mt~GSI})?*N(RoCiSDEO}aN@b$+W=H|>bII`{Vu$YsS% zhAarbcI5$RXAN2zmU^+4Am3IHlbk4RJ$-f6?{{kOQe$w+b>Yc0^4UlAg{O*T=*aZt zZuuZ28Gh@aZ;DTv{?zE#r;Jw)9qNsEBEmmLRz)Pghj7Lk&;q+ec!~&jP)SuBZsitR zP~?TBr`?Myw{lY;y@zK-3r3w7q3EQIdJ4{cLZ&1aJ9flN8j2ag<=fd7V;jJ7Wt%)SM`*`jmv+M3<6W^0)AZ^X;4My=A$2*T@OnEZ&b8i3*Dp0K%Xq4aEOhZja9TNZ% zp^y%G0?-W&@|7V@C@~@`x}GUBSC?+a4l3B_`*>8Qw9QsZvq?Q#1q_DJT*sGNjvMn) zgW`^|aGw2yVBmAs~Fy`-=kaU~T+L_{3X=~Bnt~1N&`?SWp(bT>381)lg{jmgi zFJ`$ho@9pgjbKWDUpZo}KZ0`psr@vNE)BBs&@ZoCS+%xiEMbB;c;lj-G9AdE_0OuW zVW$cb5fP9g85b#0i{3T|phI6~uEJN&jaFh*JCj&n@Xj0T!EomtIj6P@G4m09!wE#t@^}kR;o$anJnHY&V6nNK_`^@ zQM%guXwqfkSZynT#||t`Wb#`id(K5>-J+?&@GqF65e&Iu*2X1s$43HI#*P8shY*dX z8v08UU4)dp1?c_?W_{&L+-CwhT++~skx~Y9SW@_D!!_8kCblMNxe5DU$s_qTucvS~ z$J9N>SDfu2M)xjyYx~$|WVm17#Fz`u&Qs!{ZM0mecMOj5P!EiwuwVHZ)R z#^E>g!7}j>_1uXanfE~XjvY1oytrsX=32EW0VL^QCn=wTBVEnfrcohOM1fFlNcrTidK|vYC#q;W4au z+vF~q(ANaHA0zWV{K-b#gD*MahmYB!e~&R-Ybfi?yP|807n#e-b_JN0Xl}(8>g~3(lyBlaClH!So6Yb}jXZ5rzf-rd z5+qB=fz#-<3BMxa0`=U!Vo>7&2sPk=+wC{PQ~b7mWO0e(S_L|5pVxUDhFrM?(bfgp zhvRs|F3`7Ed#h{-mix7K@|u-3BI4Y~>#(|v&uv$k_VDI7*xu3&=AR#iKAf1FRVs7~ z;{0=TL1>N5$O4St+kwn0(pj;*EpE1DTph8n_V4h%K7E}&xKrJ1IMyNaZf~9MD9-v0 z)weJ++Bfv^&qulP*EUH3epggW8BK6ezH{zwbZ6#y*MA03EWq28tA5{Hjm^4P0>PMJ zZv7_d@aI4KX3Pkw-I6qp%pgi$3F7O$$D$zBg4!DX-=)kwWYC4vAQU>Vt!rZi~VG z)SmU}jxEZ4L*+OGpdbD4Y4tRO28FK3UF>5d)08K>aRa^uf)(}(v2PlkX*_ifc%tDy z>%H@&=zjPlms+1={^RiS=!MME76|bmhtdqpwlAqD$W8SJV@${nXZp?-9BOW&%NT2j zo8@0>*P$zO+YP2H5Vjw)hepS?CIZhM#u%bKJj6xFUS@E@+%qYu`MO-ev@$ovqi zNMjva7iqvA>NCqdjhvT;lsmsCyQ$^5L<#|FiuJy zUd7xE^Oa9JhC@h2=a`cPb_<@(-D3{2;=+9=gvn0^o}F~RXv^-vBpwft5WS?v>9;x{ z>I0^U++ugn89Q5F&vD+jrv6SR?X5wL$y_kiWS5Eqt07!A=BR82fbDzT(|~U-!i#V} zsW1r^caf8agW$7|rI_^!0rEn%mA{IB;v%5vd7DgU3ZOslhSeCmS&xoNx$k9G%#BUf zL%agHMk^@nryA|x*~OeH6pu;x$XN84{WzOE60p30o^JKFiFRbLT;T?6!2YB> z;P5K9{Mmy5lO*ae%CAYS$QGTq@STqw==%g;`EAteL^&k|#~;B0y|52(zMdRkANeeU z5#^dlT?3QOVA5}793&*)$dBkYG!-_pK27&D+%q7xemNSaInZ=x=xgiG6u_B2a3~<< zZ-f%rFxm0_{w}@>DdV8AJ!1pjfVzbAAKlO{C|IYg~ zg1c3}tpxDL^%Mg9Z0T7n=F<6bc70fhpRW$5S)J_BF(E8;FP_c?h~RQiBhqPlgnK5DQmBsE#w@98Y<5Su5!$W~TR zKH~bYz8X;53QxALG+ph{H#Ro5G|>P&1z;=m<#mc=Y(NA0!fIg-v%WDmk6zl`m{a01 zP7N;bvX*qND6c%)Sjg$k^gRhrHLlZuN|uZSr7_xYtw8D*i6`Ld7ao9)FDqcqa7icr`1p3d*cv2IC$4kznA zo@7#j^`q&E2W*j4w+07VC-N!^jV`+UBea=DtCi)xpCH^=cght>b!Xb73nn&~YT-J( zn}3P9kQLNVQOXuh~VfoZWYtGXvi!P1d}PZ$)g(63SD9AIK=6) z{lUs~~oYRrTxe!puK`;CIOO(9q(h4s2b#RC3wf@#Tll z_TdLr8-&LC3MZ=3G!=I}viP2l*@4W1hcdq{279y#@Lj-O98#2*R)t>`5_wcF0QvGy zp)8%1BGjOewN_Ld>F-?z@SVdPCg{q(^my3SYc*c@$6hq;KcYUGTk$xKQ$dNY%UEax z#pdE| zc{!tW0h&19!|>;9q2YtlPC5f4iY=^J7U0FEtJb(2#?e*GBeVc~1QI#X85tbsyDWBY z)tm#Y1Xt6^VQ?)|9(!wkL4mRa?9n#9b;*ucQdWkcLv4JAIx>lsc(G6nFJmIF?GM{_ z>lRm+RyA4K5=g!FAHsq9YO^r5ZwQ))TLF8PS1M%7tE#B;sbyA@aadD$%S7p=2q^H2 z07F!EZcRm9oh2a^X2Wx%+!keN!;-G-EN4yH{MM;tGJ*D^va;cN6JT-%rV3&$soA;M zou)QPfx&P=;o4l{Hwgxnnwl3&qv?Ym-$7+`RDQJX1%>7WXkGKHs$kqx>P5o# z2uu|>+Q~opx=&ub6SBPI2g>bhI(Z&=B@U-c=DIvB^`O5``QqH#>P%K6thwVCoTjj_ zSZ7^4psJ!_2Nt0!C>2J&mHNuziO0rJc_MWO2QWap5oP48ox5V)ZCr~0-knBOtuv<6YG3o%y#4C%}%g*sVa4 zF0ty$Ygqddbk!i&BXnB~CnJjkJUn`Zl_BBulg)N|=+mC$JjOsX5k)OVBy!Xf6EqMC zN8FStG-%dm_|0{I-b-P!Y3;xzOgd6&zb^tV*zBHo(&Xmm;p2L6wKv8zU@B8PAHW&w zc3D_CwVT)xA&X8YUo7F@mjA`OF3x zb(#JbdZfthwGkT?XahKh%cJA@t{Z_|X*3wy(~zD%c-gLY8rAx!q7mjSZrUi@v-?pg zm{JFl`f8LjM)M<^BGgb3TEFJlr%YRm=~C99uTy4E-k?@Koz?Di;qIo(G)QWvFl6r0dhyHZlvRNw<2Hn(|rZJ(o>95TL^8Bc|N8 ztT#uNYyk%E>+Uw+yb@vTb%c4W>gi|#s5!L)q07vcGmkWqzgK$@a(n>Qq?w2J>BZI_ z1@}q0&cc4ZbNBF!s{@L5_;&>U9iLG_eRg>CSgv4)@T#axVXNQXXhNBjVwyjhXtta+{0= zNkv(gVwHtAQ6B$1*~e!1)`;~cJoNU)ZjJv`#&S>GXLU5#diz=HX)p5BX6Q}Jc4r;T z3tA^S%kTNEL+ZMLtchjge)4iRKh033W)P)T2rfm?Woy6E)*TAfD=}E*M1wm0SC7!_&fWl^c@FGV8H0EPIIhCJIm&qYt!)N(DwN-vsT(rFpQ7F zUd%EM9so;d*0OQAOF;}DASN)^H582Ob9EEYS+qu=rl<~q-U|?3!1(Te6-NE*fG)xs54L6aWXY2!NHm*_D9qXE8RiE z{)Usz2%AcU8xQM3UPQnrm4-~6ua@7~a=tE3sKuRnh?r!#4TEF*bnW%aUPgS_pFQ@P zUf6{*d{iDhS?k2lm+|U4o;Xi1RbG}?uCm1k3j~)1l*KU`zjfyyx9K5h85ewInrD0U zjK$R*-ud~5sLcDB#)!dvS)!|lyl7tg5LcCm-c3YQzX!D+OUfoD@9_Tbd&DAGgh?CM`8}ut) zs0|;y{m0|J(_G}6I@;kJUI3QeGswWKJLr#W&@r%{cLl;YXe;j*4Gc2pCc|?=X7GUe zl>}~4%K5EQk$nk!U<4v>RSlwSA=@{uIiz6@l1am_j)kd7xT{5{wywM+1&UKSWjZMC zOLfv#eM#BEQ_Bn|uEj{ms9zF2hJ9YZALZQwL8q#1wLQl#tuL<3um2NUV%|3!A?;p0 z`c?sOHULHq;wBDIWY`*q=j!!Br(txik5e}R+@O(JT)s&!;rc`U;kuuz64-Xa)?5f~ zWq*IWtE&@iAeboevFa&3Hq17ZdkRM~s62QDS}(wdBW9}e3+#x~r8rFdf%V*pDsaq? z*Gel~ynbi;4VHvoTTiq2=#;;6{Edd!r}cuwiMGi)x38w?$IbO^d&=EB-6Hg~b{*A`{RWk>DyEZr8Or+ckz$ z$IlfLTZsP=tVD+`RY@)KUf0uhrXx_LwzqX6_T2 z-`>`N_HjQb%FolduPr3x$_^D*^Ri0Sbv)CQ za*d6bVeB74;Z&#+AvI`mc%4iC2;ggrn^2uYv9i7`WTa2oeT&|!r8yn!K zx`4ZQZA(ydveOM7OqopTXX^hVy}Dg`4K^#)Q0)s{s*+t54TcOsQNjTwzMdZ=@19=w zzbCZ*!MGuyS^$y={E${W6Y8gWY=yFk5K9&7rGq}KHA^6A8~HkMDD`BF=plEp6s>1= z+VP8s!oV|9_zK~a_r#m^E-!jRYm_)tmVS;rRz*LSwo;VT`axgg>&eGe5b*i?o-Dxf z4Y=!{Nid>Re zb6n36Il2y8c-H{V&)loe{=`q9jw?Rg2%l=R4zJQU z0d(8Z`TgWY@V;WpAJ+$qQe{_K+qS;&+B6k+RF9CfFq2KwzZt8$!scm%o&6Q=U|8*I zy~C|Qbi=ndEueXFf_31Q^w6)8wrMUsv7yf04s|pW7I$Nis#N;JT4VWjKa3?bh08fpGZLU$M^6inS znHSqHB;pK=y-0;RJ?!&$9e&hz%AIeuDDa}6@xmASUNUO*lLR{^@MQZJSE&Z_j-<$X zFey=v1pt9HVj0;-UamQYUq^P%wDn) z7b|QbQ0*7)L#O&vY$lHCPOp!Yy=r01Qeq{<9v>u4F>F+x!dRpA0WuWfuU`3;w=460 zMEqI-sc0`JB6m+ox2dAYYdf*A^aDF`M6FfYL(U0gs(+(mJ211m+l?oLoeio}KNvY|iU6S=HW;*9MzLQ(-4VG0@b8QL6Zp zd2>zm%!YK=HWuEZnUcz53|83g=0|SWMl&${0^xMzN{B#eu`wof6%Y*E`ou(~tieQ; zH1%uA-dJEgTU?^SoSNhnYAU8AB{S5^2MJ{BVjh0`B+gb>zk1}ZTivra)s6|H)w^G7 zR2;6HEb28^E4&QRCD=U$V*Mj8B79J!yv6>NP=FuVKTJ-}{6_TsC*8n25LC#A>(U z%VxgF3?}^6>Q1i$TOg9>zDY#@E(sTB^0Zq>Q@coJ8;M)n7Xm_9?SeBNU`$%xv|vH6 z)6a#|$GC=8I)PM?HHW%WZiCa-{0*SzG0OpX9~YRzn0jiiq_7!QS0=0*NIuRtHXt!2 z!L!;0uIxAo`{7Nkc=2;cE>JooLDcc<_BO6FPyZ6lP}T++JNZc1d~O~rCBG&TwCy-fpQ*fo+CS=y2-j>$mbFU|UP+ODz|0486E5qy`F_2gH?7R>}Hc zuZZ=*TJy+|#DcIekC!gb{LBX1xa(aJs^@US>+#@G!t!+Jq?wq5fajeQ_k1An`&j$h zUi>cDvK4`XjltGalB6uwfyH0RTnp%*(mZqcif0+Jf#MxYWtmstW5iOz+W6AVzg_?{ zaARhep96NV+0n!9{MPnvx1W(jpdYK}Ona_5lR>p$G}2$yB7hASbhr_XZpfHREKqq5 z4_DWxeI2@~pDBM#bh^ygsm(^7s78I(XS#7{vB4%xg9f6LT3ae=uUpuU>B}XE|AL0l zy6E!ivP2y$v$05}TbqTe`LM`oG+<|4(!nrMAa1FC){{ZR#`$)=Ox`gbUX(mp36kpi zJIPXouG*;B(-Fj!DQrt3q~()zJAFvz-7X4mS|bKu!N{Qy~tbzDuh}0>~X?c zszYqPo=R-39o%N2rB+(X=@y3VVhGt0SU_#v`lq9rh-;7ck&hh_RIo(E zcE1X&r#pF6A$B}iT$%Ivk^w4-Aw%YD7{2r?@%o4Q(2VWVw94&m>$M@JqrgIaT4Anfh|r1 z1G+@Zy0MXGJ^!@PbZF2?+vX+~bn_$pK@Y+iR1+D+DM>Zfou9@^yH^Lrcmr4bsXyKW z?3PKIEJ5NBxw&Jp`2Ho{>t4HmZJvHDzDse7L4RNs$Pj8g`U zn5{EsUjUm?gtaFLUTRIJo1I~WQ34&yKv0&H+*2#x1RQZt4e*k`i&HE!0(oH8Z)em$ zqWKQoS@q)Qlz0LVxowtwfK@#5*MY)WQJ|O#f?bUbmUq#`Q;3%)f?g0fue9I78+KD( z4(=11k+xJ_hvBYYD=(T~eQ)^sS_pUysAO^#_klq1TkEs4JK~*%%`ao_yekr7*Y<>xgi(xN;aznb-kQ>f#+O|7wBGBt*0HT}Pygeq}W(Mwn${VVg1 zBPi2#)ATAF;ctXbIs98v%-b?VqTWXBsGO)(;A0!#Jh%V0+!Ss(wY=uVAV}TnQc|vi zICg8Xo7ur>gX|4e783HI2CeI;H+2s26KB5A=7(fKmxrbV5)!>!gW!8S9Q&HSj=BLF z7U1gw!ek`NpL%RauW`i1LGD~yJa_0jNoSm?F4~RLwv48Q^(`8vpUAn}qt~9nTU3xcXGQ%p zP6Zz&e4Z1Dl9Ief^U;*(Gtu8*YYSKfyd7$gNw?T=e`@({x>=*o4_War`bpSM&Gj)a z9<8XQL@$7mscq|V5$Ha8SGY_b=&GFrP^M)dCb1FGQho_H^G15U}NHt^XTLl zP`=0%(cWLeA0w)P=(!%RbsAlr6lvw>t0tUNVnlpxAXPQ+Xtf z=E!OrEZY#n&=FKhtdL-Aehk2}K>{ozzm83?{+FgNvU3-rnxw)}lxsWMcZR<<2lUBO z#!dQI#jx@WLVt>}%vpj_(mGdx9iMBcYD&GInCQNN;mp8{kXT`0NT@mBbw0Rp)G@e@ z4V?RJu-8l9>B5*+5gQ5pHd;jcyj;Gx_Myf%>;bJ1s|*L{mtRH|_+O#Nq|HU$9&MCT z1k@Z$rx%RbXJeYO^d$lxpQ#oRMiAWGJf6-dPw`)DDt|^pg=C|lz6HD(<5N>UHMJMc zR{#X^Yp(c|Wd~T|{8l^3R78U!;Akz1K0mP4#&=fbc|OxYuGhI1;BPbB`fdll?xCpJ zDUkaY@Zfz2+0;NF)3KtTM|#BBkPYVPN?A&E-bWy26m(U=X5`osS5MEY)(bSWjDta3 z$3dQ(Tk}*Dw`m=QxbN*%TA^2rrO?L$-kk|0l`1c->`tc5%Nc87$8WxjgYTz_jnx7C z5ZFcpb-xnL==SN3*w<~i4g&ZQk_yTfjfb4+dlmE!M##X%W5w`PB(hiQehk>(f3pA4 zb`3jKh)ez;cHz9eVg2F%A?_`s>RPsKVG@$S#uEtc8a%kWy9bBh?(QU5a0~8kfrYyS zcXto&5ZvLboPG9r=eDoCU+>4AZ8nm{Vro^5(MPWuee4ELj>yVFySvoY)uDc5QWCpZ zIe$J}cJf5sA%OW^5Z&0+gp*fY|K0^l8q=@6=hgfhOFmBnc8~+jFy|W{&R|*JWD(FP z?_#4QH6iZfqty^Vib$JFX2pAgL9a7AxMhFcGPu=n&-*T2zH^IO=SO*%l3Gv06?^)y z{uSa?I&*bUO@m!*8y=mrX6x();{psKSVaDY4Jy$1Z zr*lur!}&!ycSAbQL6&quDA@%$HSRNu_z~fn_D{!?X<`sb+SrTr!e#o)kzdwIqF4DT zMKYLkPUpHiMosQb4}{KVzjbKw*dIR`W?R*)Tz(ddVf*+IXHcF1skQ(c5y8xgeospL8uW1VGo}lwx5zJrQr&s z>j-A{4%O+#U|Rc~2dzw<>JcexGi8Z?PgX)pSePrg%7dw>j2xe***=q7jS2gGQywL< znDU;}2ycEC8S;)2HnTEfx(p)iE;P-7|w+d^vqZ?7Of{Kn>w=xcL9su*kjf3tgN@YYlT@^dW8k3 z=<5#qpGb)|1*FYo@Z@w=aVYSi;E7x*5fKpyR6w&qQxE0j}6Q@f6gk2WVe6* z2xvWx1<9PeC2ibgZQS=XV=U~M7PkCBeAkXn)j|H3<{QX7?)&ei(dYPeZMrg5A!Ozz z1}xsWapvA9mZh}4NCe+PQwA8>Sk;)lhW2?V+=i(c;rz2LA~$SANWEC~e*V@uOVMUE zaZRM-rR#aqsOV=yVYJC&g2RnlY7%xJ$M*3IvRKvx|55JBqg(5{H^r{kdTlkpiXvNHRiw|7KDb=&QY40`q@ys;m9Z6F|^3w`!2T%hcgKY z?J6alT%3>hHumD%PCC?F!CI2pzpal;8eh5&qIiBpxUFAt)6Iu7 zP-1RHySciWL?^_Rs1_Grb)@W%o)BQJf2IwyfmOcdoKqkne8wpIBTU6%Ydv( zAj{jjotLh%({Jm&Dq%EOH#-*bt!tgdLry^*H!!g0XPGw-ms=y38Wzs3~J_YS(Ts~d-DK$A;+vjvx*yb`su?}Ha>^!Al-$ljarm-DqtIKdBW8~4SNbRVc3k>zV zu&m61i|U_AX~~O^%g2*HMUvw@NoMjX4Av3en*_>E8PpbTivNS~L8|OaXy|Ap>C`@KRzObdC35`qb zEGmj^=1s)f}zsTsBevS>Z_Le%7oZ>f0owq>d{nUJ=t6OzuCYk+_+l{7$g)ObU%Yspz-!=+K;R z@lvd|T0Z9~=yds^U(<9gJ9+9;_~QB`hA9Vz=I{Z%LP{_-n#0z7>S&1k{%5rd9F@C{uU%E0a=BV_b8Q=>XRx1~o5&9(CMsIDx5TEH zwnl9Ug=q3hRnfPJr$*(kJvRvVKEj6EB2XOKI>cLsB-fHJE^Y--u{<{39CEnD7ArVj zDlCPa>cv2SOV*9hZgJ%cL^&he#&Xi)#a@(?ij|>_O479l&VZvJUx)C*wi@Zu!KtPy z_v$Y1<^Pcx2U5|lux`IhNgI&(z1H;% z`{ld;xhnC+G17}s%@#w9c-{!9CR=*Z+7&h;(rxt!W#WRuuqxX{z6+FS$%204e&=qI zItTm)qvM1t#~vZ^esoG^Xr?SFJ%+Fp6^NUM-e@5H2RaBp+b(vf?g;r!$z6Ek3n^S z>fwC$aNpr2DBi;Fdg@(mStBx#R+xK6C6YGgeCfdDxJN`JOp2b-fX8B{7O6UiiC5aQ zD2=bnVboB#6PyZ#3{&*(duy3$w1j^Y&RXzf&qbUT{}Le*{ORxzz0AOnOHI`w z&Y=vf9uGv$jgNCH=y4~07!zmxAX!tn6<=;9`aRI_3I2%W6=Emk) zpH>VD7#=-b()b*bAqhD8udVt!mM#)x28V~GQQ+j&BWAY2ZgB6SUU6SPC63@9^)89^ z&_i}5>b`4ixqSfa{Hj6Twa#lgj!uI?)MKVsRUCPI-Kel4GU5517=8W8yXG~A40-7h z@7@e#YYt2XcuTtF=rotF^RZtN1i6rbZ<9=!eq zzQ#)Np{m2cz&8zTCv6$6j}xbs$)o)}ORK}_*efR)q1S%gXJ7s3v=g&h%+Z9r&=IF{nzm(HOq^!H*+*In{Qh(@^QZ> zk9t(hi_s=Lv(ssBw+tCE^*EhI<9GgYHdJ)-z;S8J%^nz#mc=j>F|y}0vNk(+6LKTr zeC5G~e7HaR5I3fcus7IL`(c;6u|D;@r?Z53Co(uIwZir+WH6=CX=sSvYLMifvk|7A zT9ylIch50wJ+Qf|W`cbcH73+IqbViIyl#+6~rgFNj;`r2vjwgx`nf>>sEvmD$ zm_;FNDY3VIh6>A6|K#5BBwF#N?e2u1M5nuCE7iMT2)JLGM`KNs*#E z+vVm_&^9N-+*obSf1pA|F8*6d&WLL5^BAVC7*O=s`q za4)Kr^{;S-vDw8&>wb%}&Ccr{ubUP2SE%2&zzaViglwb1Y+a@1K<)*34#FPZ%|{cx zaY2sq(*J-Gf|lT}qU7mbwj?a8FPeTh6dwVlXQJ1k&$zb&y~}@juzn7gVs2%e+iiTR zYxz}4OGiaRoz|s3p)1UDt4GqWkD~~U%~n48ymv@k_p0ni;9#WO=y^gB?f$^)Ul(YzvG3(58Ey1SIdHN*y7dHUG0S{iy#rI< zRO$PM=v&^VN9igGz+x@74Hl)m9k`Rn;0hfjN0@WFpmM?}KN{Nth-fYmhk+S0z$i{x z(?k<~<2vx_=Je<5*9UerRjKz+g}#mvBYIfPl4k>J-L)^FW3hl0zu1J-@;NNJ|Qbjmy7U8=Jv_#a-pI7LObZ{ z(V9XApHk^G7md_jh#-GlmR~Oaa~cguM%H z>)8=~@5~|3<#m9EGP;R!#)!e)8|rHZu?wW+`=^g-J6=ZT2d|TE^Y+}jmA?!x&O0Ba zy6lxsSh;Vuf$_BJ!~5m9n5eGsVbwy8m4LnNtPb_ynf-ilykzHuH?NXiCW+6D{qYK$ zj*q3_0|QCMovN13^aD@h1tB@_W7|T|s6E^k9V@_iEJW6(b9Oh35UiG0Txv=01R2kz z%t;rVEh^?UEaPjik~jFd0)eUqRAK5v%m7{xL|W(WywR(`#-E~(cWh@GeqS{l z&;xo2Plv{25G#R^uV=RosVC_I}tuRe#d!n{XhTCGYTbj09rDlRZO{`5wiRornUOi#{!ZhsP6eD+&(WnhU%K&rJpEYo!=s0T%E(o`iRJP=*|6n9S_s+mVRr`|f##Wf>7g6KobaD5|V;ng0| z0z;o#_mlL5@HbQQgB(AXztE%Lw>|kME2(KD>1v!Fj~#0yTc-XT^UksWQWf$FFSB-a z90yu<(FDJJDXUGYX#G-#l+q2vRg5s=`%8;TvjiG%+s3TQa%xiFEJRBrwWZZF*nL(- z72^3KfAfLP>ZCs$c&a3HMZIy22%P|rSIulRb%~|IhCs{ZrorI>;wpx zwLN*XRMjb!*M)|r1+?T*YDZRP7hfGuZu()INUbuZU^65kXrdluZi@c@00eZvj^ zyE8o}*|xfK8y&CC2XWgkNu77jXGB1s=&)NkO0i8(i9bKC@vIW8 zD0vs1xV86;tdl)PV(srqW6j_Dy=`GUlbWy(VP{$5#MhOIwJIzyu+TfyH!&-;&6GqH zY5}fW8hd7A`sE7ASdLhx*m;63*G?LGX-4~)+S9xyut!dt139VL_Q2lzUo=C=5iQ11 zBB7{YgMosWDGZ>OhOy4v_$g=jUPdF9IEr+o?F$qe=(xSk2BO)I0 z@Jb}zN+gAderI5r@-VTjQ&nS%ZBM*z|J}WaETytTUsQN^+xK1&Hne}rkSQg}Wqf)j zm9=q{zPC(b7=w*@2|=zID}8R^3*4zw9M^@Zgvk44WB$s#0Le5yT^VdnlTPo3w7%$iO_Sw@R4L$Fg?Y0r zK$ku<1M})|lPK<$qL)VUVa21)$}MAPeM{GNd^6iUgT`T#hHbg%BfG8bLLRRMTI4LIN$$

@LA6+E3N)xz3eNh)b38ztS-v`1ph~?-I*G(Ko0Wn z8UY^Tj(*$tFOJPCXQxd061?jKsTfHeb2_ASxpX!jpaKs{CQP)n-sI~jT`g(lBLAD zH=q~SIIP6UJW^Qj@R_b7}lqKc6{i*XMmw!ffN?Uab|k?wE#-l zCbIzboTiC|cxHO_Xx4~8*9{FPEhbG{?UPb(EZ0iRKsp~UT`d>7Y!g?8Jl~a+ z3JP+p%*^MeN1Z;a@<}O?2#RBgjr<}8zNgg7P?xsZDxW1{qvQx;Z}6meg0y#RXn>!R z9GkA%kub|bFx)+&Z_D-!9=JmN2bda&yOW*QNVH7n;~-|(I7}V+vbw%Ay6#Na6VC>yxtAUk4$h^hF;&l>R6zF5Q?0xC)R)Vw!{Ea=raCOXCYa zM21}e>WN`IALOV(2n+-^Iw!m#AvQcv^j>`J!Dg_%D-7dJ2Z)iMn@*|kMlY7urvTiQ zMfA(kv$h=TpX>*mDobsy4k2)OL-cfzwKIzbPjbJk(r#9Uv8sxX@wmc|d0p#)Hy2BQ zLmb%W^unVBQ79oj1;D#;Jg<(n9&Q~kR~}+8diwM}G9#7Vf8FQFM2z?mqExgo7teqZ ziUTJ}4KGPOz|I0V-Bx}=bT(77N?@8SX`}(sETd`PEk6)nvfv^~h?cj7zJ(S)ZF^gz zl^z%vm|hzX_{i9Bk){I=R4_C+V_GO|((B)rxni+~a@d^x258d!Jr-RXZ%O7RPQ(|f#DLV^_i92I$4=*||Acr-0&7Q1B%6Q8;?MF5K_jW;cJHGSAI5qfPfcC#Hv6n$>|uVDz~ zd!;v-u`?`#&~tgBuc!aP<;&buYAFo;z=&#{JJ3M^8(zP!eA|<@ClkD+Vii2RBu8p; zqx%;=k|y zY+Apd-ngJnj0QJlf9oIRRbX9}IE12$QOxHDTI8a6J!LI8*>Wb+`su$eMuCV(%F05) z%2L1qZ(&i{_VoExY(M|fpL0TG8sIFDe57N!1_&;>r2rEFk;F?|5VjzHWMInlF7eMC z3Zd|rH_@L77N{|Xy#rI)Th|{2v510y907P8>EtOjH#rOqB3qFglakaW-Ro~047_YG zxcF4Lz)V>e%mT~5cpE6r%=9#FV_957{lXyme;w<4S6;_hQ^zQRo9QP&{3jP9)9c2v z!rBLkZ~%+Ls?ohr{0fakXIcMdHs9a25w@xVtE4l2kcL z*ka?b@TUU&pdS`%SWAegIMG6M2n{O|-4~MY8W(1adoYpyExz@=tS~G#Dq2n^a`(Nj zZ#6@8-~IiK_my`GyGo>*fCQ3x1Ib zl>fRrmguPTI_Fg|DH9Xf16~vnAyhFk?H{V`-`iC8f>!|`V!&SLhy+#0``>pnC1T{1 zHuVasko~`Z3qtGv*L(cme?_~P`}NHwb;pBVdl%gYQ~p`4-Y3V}YBb~>e}KThHuJB& z5FgAG^6NY>?k*g!q1eEzvp+olYjF2?0_q}0b?qmtm-`P!IJ;&JOtC=;NJRSo9Gi*6 z&f-)x`>dd@O>sRfIBIirGgxMi7sX;1#!MDW!2rNVG z`}Sfm$8|?5s%!Udz@sogz4r%p@f9cM!m!rF($K55|1>X)UM^%(5+0gA1$05qmrSS| z3PSg6CjDgRyxo}p=|q|2fIt5a zy>H;D_U#=8v~O(?xGc(5?F_xucfaj_#4@2Wt&=A|zzo|#DPw`N*nP16A77^HEG)eF zzkZ3vHXz`D3-$ibnLo&|{OiNN-{O2a<1y5O|jF`Pf_F!#{7cumsZ~Uw0X68iyk9&ox47VS{g~`>h(p`Rcv=mUz z!@lN+`PgXlyZ%Z0$50pw93{8+V=&GdxQ#+RqHpO)fiya15-}`QJRpH_~@eB1zft>OUJ$)Sb~UsKS!hh$+ZWsHD19NnIq(& zCp#jQg2|-k!!$6d04_Yiv6wvS?(sWCW+zi*BxU*qDhF}2>-ZIkX#h0T_ftLX?Pe}D zp6ixnGJAm9;*YpP(2%N?zy(uwWImnmq<}xZEi(XrJXUduc%}1u?E?$4prHE-2nK%v ze=!gdK)|`EQRs%^J?Dc?iR(LqzPVTLe#K?8KN;QUR>k>w*1Qe3i9D{`u9}r#YC)j~l{aWr=~I@`QLB<7lY;3B`B;dLv!0oue$V4BTK_GG+Lq6c5S*UwRuSaD zS>yxwEtp8;C^SeAv9rEHj-qE@~m@9E$Jj$&rqEi*3ql$#Hg=3M&EW zE`)I+vC|6(62eCQ#D1LrCS^Uat7ysC%XQj}*z`E`@rR4w9qWk#epdFh{9 z7ixBIpI?mbE#uDJDRi*t&So6xrS=gbLdopJ=o};78P@lHe!RvrHUW+c-HNYc1Ah1H zLNzuh1qbKXN9aNv2LnKSaACxM+D400aOgmt5W7Os|HwQ=g*5HdoLxy}z93Sg`R7GE z>`Xx4jq0{N=2q3ghxT{%BT&i8iIr&jimCcsgSmtv4odacU`dm6Jo!F49DbYk@iug` zJ6b0cr{$WtZRF2~7`R5;_vbVorjh(k)LIGw@f&O%MP?*k%SZ49Sz|9V1J$JSAtjGd z_+Lz{#S|SX;UukSrKigMSfa(b9hOfViEPy%vd<~3ZvyzC@-RBh>zk)iiOolj`f@8C z8>_t-e#8&b7yZc3=fN$tHC<2(+P@Q&3kW=$@E_FNkEAniY}G#Ivhr$Y9MUc-#s9{fY%a(9)c4nI*fesq>~> zJ{1!RFVb+0imC0p?0#V!4~5OxWJVhgYydixlW`%W$@UKOZq#5Fzeo?oK4f55Ff+RK z_ zWAYpYHBRJOZvEtVY<}_ei!aV5Y7oLk<l@Mor-IaTEpxIMT#Ej|KX%Y9*D3S|eH zU(ASdoTv%=(`#C6RLWORf&Mw+=`E(k2KZj!4L%BIeH z$sd$k+Mi$G05@HB?B=}Yvn2tKN?Q8D7e!F(kY;OTp)h!Es%PYV*oD(L@Lei^P;87{ zJnT?>CA{pIN=>uek%polN(BY=8#isf<&^a8xV%gZ?bjokVH6b8g59#|zR=~CVlcM? z`sh@it;ls)F|T=pr~5*eJOKqXl(;tHRCyIiqXi2en@1`FRB8BAc9Ds0gn8KpsXgia z#|qH^YvBG&HGv0%ed$(V^bU$2CeWJnPehP8oH$>eAu%q;V; z-BQbq_rJT6r2J7>Q)=uk6<@uAf+eCf!(uz(CiI`0osD%)dAWz_@ppJeg%pfy%JrAT zDDSPlo)UyHNm>ZnOVvb7*CtZW@4ktXi&!5jQxO{*J6*VEsosl^;Z|K%M?f1d0j0z6 zV^|o%kyUA08m?yx&)1iaAn6!|n(}o(cVtHnERiKE96l7@d^a`Q#=F?Rh|-VC3`7xn zbf5}LDWbw>U$Tl2QQ>Z&fR8E&j&BTn)pAzw!GecMR>oN=m;zHDiB?gw zmzExsmry+@Qg{PAGybcII{V<^o`&pMdt@f=(ngr)$erVEQ^w0ZiWOti^GUj6;rihmT=Zz|kOPE`*{%BqkBb$7`3gN7Cb5z4h{M7z1xcrpS+`P($9y7FB`#iCxz=PNO4| z{NjghihQ*93GW0gt!P?!mhBdxeC>Ip2Ip$7e?E%6m7yg>51nMH0=ry#i=mO()7m|% zSS03ab^STNS)6h|Bw_X)!Yt9gBLH&Ls(%mIxY`3J9))R@nK>2faWkys}Gt@jdbBvkG+xfi4?0C8dnS za9`g5kv}`vJV|KISJ2s7Pgo{hE z9ZW1N&bfJn-P!Y@fD#L~qfBGbUUFgK?qUZoW+>MU@1-Z4wC<#{70nEvhcAX3?mXIf zV2ANo6wvc(eOjL^+^r{}7-3M*hR01+|#EF@&9xx)jwGe>{Dus77@m~Uig znlOX)5G0e z0mCUcme=DHJ406vwn#7%{;0h^TFBks8L0p6?Ga&*5dmvUWpeRbAd)0qr^PPe?)=!V z7myo2Q)1$yMN}||{rw%0QZM_JU&C9Nn!bX0{R(C+=JnWqh}7Wj9vIar8b5q8j&<G2!wGOG~9B20`sLG+6)q9#Po{XUI&Dtp^|7 zetz{2&z1G}@89F@kNzTjcB?Y3Z{irFyB*QY8K9Euf!>Yv;;!>4|hVdH67` zjto`k*ZX#y%jZ6})$V@1%IriIlh)+{hgN3TnkcO)!12j--@}K1*KT*tOckzMr)t$>`a}KvS`0*GNvzC+}HF2ag|8Z1k5%_T8Pk zxrUggrjU@3Ggh>CmrpB#U;Eh1L`+lh9bh0yN&P=EV=gRcuzNz`1sh^~APEelv>OH- zTwH2#BPU(TFe~$v%|6_ur<0bo$ZIiO-O;`={!@LBexad}Ox`03VX`){!mBGT9+>MZ zyiCW36Cp+_C~j?C>FbkM6@xJ$5f6!wv7{NSRMSQ?mUD7))MyfkP@{Reo(g~tk^>`b zf+dRt&Lf0G8{hp)yHH8YYcj|O-DZ~5orGsEO;nAd{ti^3I0YRYH zTR~AIyx&_plWv|*t$fHCygnlKWAv;kESB=mNlf{Pm^c#elSHLcMPb+M8>Uj5kkb+H z_(b+eS|2>``m&&R`9OM@?fMXzp!jvv2zC63Pu7PI=Q^*+Bk-R3tFLwW&?HykCFBg& zuOFS*898+>)_=>EjYOQt;4<0o`sUO1K^(OB--9KmCo>P?CU&9-c;z<-^Kfr8KC&>R za~*8hkp$6bwBnCLL&?W4EHrLeuukI9{!nAo703!T*0*HdsWN-KzhJk2pwWkgvcY}F zkr{$}XtlQoha}b>1B8Xz4sXw|T6&3-1R`mNWD1(u^d1uR>`%VyufH=H)fJU_uJ!V3 zgCra>ipi!fUJbVPtlFUFocCT0E(b!;X?(C#_+fz@Ic7`#OUI8uq4}aZui@6Bth80< z_XqBMyRON^`Ys?k#;GC=3rZU8$iM&><9w~hPSmwDw0S_!OXd$mhMzt5 zJ1#Y^sW?k{KSK9zR~rIS4b0MMD^)y|)YQ zFM0-QR0qd7faNg9vY;Lm&}*S+W0F<(!rj??pUUm9w-zhzu2h7$J6<#BUqOOKu_xfg z9pNe| z?fy^y!3}D%KYb$v^m@Bqo9FaK!t3dap~IRK=Tu%Fqsjd|`xM)j5xW;EwvShjg}dzL zoMWRNdu$HLN}(X#S?^>-#*y>}9U*W?q20_Z3|H5eXQ%cy#*ZAuZZ4Ldif$rgZLKSFePbgtjTQ;A ziqNQikJ$=*SQ;7{j-{=jj?VAC#Pt|Px=0f%vvFF|zWIKEtCl2nj<<3KK+PRMOB&jt zrodMmTnlVWjI_(T#UrTN_9Ymwq!t1S7miMs7mn{wH0&hiW4{%2=NSPY1Xgy&uHZ|l zg#IWkEh#M_@h_i0Oqj7=vv*Tquv4FR1Lh2vLLMF-4^2?RoUK9Or94>C?@OgGULE;1 zv`dE8`4q@h^I68qtjPoX;InkPcsn@jqjjBj_@4Xctdx{gY~y?6$gt>?zP}zROmsy@J$TeDOKKDh@H!+ZDB@BAzsq&gVA+As}S82M!ulGJ!nIZTj-yb)1 zF%+;`I!ON^M(an!V3X(LRcRA4+YgN;+FS<*-s4f9r{$Ito6}GNRW)VLUt_0L7-imd zrq+>V?GV)LJNNd8WglEVF9GvNSfV_m_W*SAIIQj0D+skct)n7nLAKA}Lx8G#;ofl; zJ;wnWbBuh55jER{Ixea~4k*OX$j$+5mWdgbYS}7SmGkI)nJ@6@QKxSEgJu}|p;?jP zsfs*bU4ha+YXQ_ecb+yt=O!X$YqP!VIa>5qBT~1=VIU1dd54#hs46cGTTxpLi*%Nt zQ5EZ_&2*hDH(EyT_PM<%Bv?+-`os5DjPER9IJG&ZCC^Mu9L|?M>yam0pp+&1S2zyqe?h-J+)J z43Gyo#sn;>hcmp}^HD{FOhz`B^RvLE^G8?y3brUvxV!xxc>v-MG9GgMS&$Ig?a z^Xm#0GoM;|BXB2*@kk|)zR7>!c0YUAJ5OM$vd58XrumeSp2ftMPOP3+`J-9?x!nLX zUlxH64mS3mfFsr{R_9LbHqM~zJ8BsDyI?x42Dh!dGyu&UT>`-J)zL+Vo2aK8F$811 z&r+9!XmH5S9?86me9t$2=zO{E^V3{RTrRUK&@Lx@@h#C6AKC5=7jhLk-wYi6m4Div zpyqizamjO!+)vI5)~fh=^Oy`Lf?{sdQ;Tc>Kp$;u3-*Bm8LB?_GI|RA3o?rt!ad@& z=reV_>kyO4)zu%V&EcBPBHZC|+Lq%F^MP;fBv)4}PFg+|8%b+rPvHvo!JM?PXlC;{ z%xf=}ZzYJyW$*T=_(zR*!YBl%@*&}K+3vCvkpd%n4amY;P_5)>*=P6e=-!@yLPJv{ z43EQb0mfz{6uS7vAC3s>y)a)Ik$++f`!4QLM;s3ItWDmtV|m@tY*SNs99crIjI?Bn z!!v`5yp*KG{XCGGjI2QxS@zZe^6{dzfl<-T2B`aRJ(%8BXeQ@^0q;tvRZVFy|4lGt!W_y4WNx}RIS|O;{{-E?d~BU+dx$w>-Ow&QNM!y=d8!= zdPB@(m=DK!4%w=WDGis1aBmmIo4ebwNX%C4oA*S9(}AVRa9xXoY02m2-fA_%?Iqm2 z8P=+${l>_Lvr?2<@^?BYf^^zVA|`gt+*-3@#YuO-c~mR_P}ud%WPnPlzHp4CEg`IfS!$cvZesZiUJ! zfuf_oz`!1{LMqr_WJ$@>KxJiSCMD&$cfQ;XlFN=~UUn6yK?jONDf`OuilB6Dt#+Gk zi<`@1NO#CJF$s0h!aF7<1NVG!ntbC4$w4Z~!Q7JZF^PcunH=-Cf9(4i3*Uftn>cBR zoOH09^zF{v9wIe{heu=0N|E@9xP;m5`pIlBA0%wf^_&`QrT&qSo6)#;TLrKw{!zd! zlTrIbna8QAsCZp;HNZp?J3m#=bSzb~=>Hk*G2dfqJ%^1O5h#I@tbZ=pb0|ivG^&PX zdf+fhkm!kSvAh`eRtuUYJcxO~Qv*2|LnZ9d(T7TaSGYajHF%D5H^{8IdOlUg;p`FT z9b$Q0_AGlKcg#Z6rdUV|IaaUfF=!GN>u-CGU_e>CrrIBqGczF}!O!leRCJNEUHaWk zwPjYt*C(?}tE%W3iHtVmaPx1D^(*RcsX6Sn*SC+_SX4+-Q54m*?0@zB1$Ye>5Dymb z?w5L8RJTPI9KLkB4FJU(@}E;5_W;7@`Bb@-nxI?nxP`1-lXJNAAhYPL-RfF4B`Nq7 zx+rC3?uHaUgZ>ireMX{UFPPLyM0j&~e?iK4*Ai@y`{~_NbA(@3)|a3Z_y|0H7*t#| ze%)920b->K88 z9ni7WaSc&WxtxsP@sElYg7IYD(!N7yrXzCO3M}Pl0IM#37bYIm9iMR7))zPS^p8Ic zjuto}?n!U)v8zl2$6|N2&LRjK*}Gb^>#F3+;Cd&?vj09Du2n&YN5)kfU! zv>W)~ajuMw5u(5*K;}Jnukj!=s{I3(ntL>|vvdL(vOcSF>vtKkDQ&=d6we%WQYd&_ zJ)hr-5&U{k)9*hkT7Pg*9NtyG;qK2?NLz>YJ<{Rx zJ?3Kp2+o^InEcv-I8ZjAx4YK8eDEkCzi3j~l-k|>1=l*Z4sr}oVTr&jY#E)bB`X&j zeFlEo#ZumZHJvPgi`3UgICaOoZndXGT%p}?h<_?-4?-}&SO)eWKI+1UjfF6D+^#+BBIuQ3i@wOCr<}k2TcPTqg%) zB{!Fs(!c}vPEJgu4sP*;%Vd@@0Xbmmna65_^Wsk-4T`emfHnwi)fD5)t*x!RxV~g& zU@b1$=piEowVvM!0x>>g&x+5hqba*Fw2n(9r#^9xMh8$W`JkUZU7X{aYiMd(?Eds7 zN`s~u8uk}d&@i4!I#*SiWcrI})@IqHYJQbIokFu*{o(OVF)2SR0}BH_oB6jkNL^Wk zvyQ-c9F9w2LOu7&td#1>gPoIaM%~x&qT({0YELf}m70d6{_=LcbZffgAi^8iFQO0!z@d`Di1J3d+-nQ=b|JSiBx%g-!NM6o~JA9kS#iz!aTx@&$B z&QBN?Ec3m1f#1Y55cw}ImpylY|0WY%v&2bNU{bArU{7%H3vj7JnHEUPgoz>HxZXXy z-v=o;Hd?LT6wLKXpybelR}1&Dc8YF)U;D`R&1zfnpoJ085O_Pm@w@KmpiQSTq_G&6 z+ueNqCK?VqO`dH6Ktj%~mO9d%`+IjQ^)}Cm3l3$WBo#i(t<#&8bn8X7vhS83J{y{r zU-;+(IXgbMNMe5wUu1$h*7h$bmG!xP*hI3d(ndAU?CtFjr80C64fzjHsn{1gKR;c6 zZD7*G-cj12TlMLw&tW6YsBp zj&D0EwMO?U)KqlyF2DL9iwpWuzwaZ7cNfxhM^fqP#Dt-}9s#%F*>+XHhR;Sq(s`N_ z^yvKyR>fT*wIz|wByfqGOH?!v`tFL`VU3E&UpBA00Nt;(O;3qJ+wdNmm&JnWi z(ULg~ikH500XX^G`fzwu#8fnp+bf%{WcwEo>v?G}!OC1o&tyiD{Zk?w^_k+I#G|YFnF?l#4M{E7MY7r}8wJa$r zEiL>-qg&-RJL6QJa5A_4E@sb6H-IFjR#|YO%BsoRhTGuLYsT}MuYa)x6omkcR)xjM zmo_9k3V*Fe_Y$9zW5sD1+0+I=TYdF=0AT9d@#GIB-;-}ddKi!;t;E&SHNp7lXYA~}n8z?nx9~@OU-@fc$4CiTCT5cIW(^QhaoP zIpa)w0#sDxMi?mJsMY3`k?ic?Al?#64N8*RMoc%dTri_sZ+6yW>U9 zO3KykEZ_0X(RLb4b3(YdrJrSV?g)36%^J#tsoag3PPeM=Pf#RJ2EyAy9f3d?0|aO zt?$|>pdU(Z8{(IP6tN->h-?Y`MFemGT*wPOFZcbL+$YzAEU`FoSqYW$#DIpPugh5k z{R&P-3;!QkZyi)+7xfLJNJ%#Wf=G9xw19wgcXxNUz>$LHlq`OnPySsU}_xC*C zyfZIz{(zYi7w6i0?e&Yb)?9y8JPHx7m{i0&W>@h^A$=-xtg>M9J1O|_$w?(9>FAO$ zY^~H?J6A_=XphA%KEA#ubTz}x3t?wtvvkvjp6{A71*s|^+(5!{GS@sX(UEfRuu$sd zw?pdJw|?%5a$VbIt-+WW^8s>u=UV!FoxI~>(!6j(+;80jCKlnB{gJ-rFV$g~j?J|A zHWqQvYQU1{crs}d67uyY8VjIub!v;>W6g-tgbKmyF(;A+qJVh`W*fX_UmZn^rHcZ7 z^ltq3$<-kv>$1Yd)G-v0D=W{r#+%VjaLRVVUg^{v5^#YRk55kDS&wTT?2O7VYPIBo zZFcZxq50`kp^6@&!z3@acjUl2n&7{UZ`5f|5hM1CHfew2Ef*5af_Oz;mE3w#-O+vO ze$F6B^?IFF`j?NH@qVhMOXjxeJfITAcq&n0$gxZOKHo$wiW>-*WVN(?8Rbusl-j|@ znz%|!Uf!u8Li~(IZ#NYv)*km^1qm$^^`PSAcnX*dWc2514xtd6F(m_d0WP= zQ!T{$9t~?Zf3)=vSaDnSPTRZFaqydG_yhe!kv`OZD)%cSF58dA#6$vHzh$ z(;zfBwrjn$7u7OOm^xKdU*t5tM%^OAJnSnj7Z0g5a}w)_jFYZ0r<^Pbbu>YaI=&TcErH zYj@E}|L=wIaO5Fn-1f%S^Gt%C4bLqN2J=2>Zj$GWa(Qs~7%yRu zRj>>Y8dXd5q2A#<65k=b#^l=*w$V*n-eI76v-&)gf*=}q=WGrSF` zJ_q~tAiHRD-I^K5z0kQQRKtiEjc+EJ$iDhFBDJ*@M#9h{XHNX5|ND8zq{ME0HT5Kg z={im+9PK`Z|@Nysvy#go0m0H6C)s&N84@_+Reg(W29$ zw0gG&l&ycDL|DUZY^~itMfX&szIiPGXfbdW5{vTuDsh#-4EYN8%;^mh(2Otq&|WQl zyPp~2>ujIt3h-x!zead&9GaH7ikVP)G}?(A6qMl--rrmX@1!Qv%Bu7NsA@`cXIN~T zKs&s{YuOjBzf=f@amE$j`r<$Jb@%HXxgcUW@)Z=zHdEXq*}47ge65oIG9L*>RuAyL zDRk5PQn8=ZBN_->cE1`dQinG-J4wKj(ZA;WJi4qV`YFok)gyU9lZRPB6E4+p>ux2Y z=kEIPv2t-eU1Ry5^@01@gv>z<$?r>FNW@}m#tqyfZe>;vuTc;K|(Wrl6S-z0GgU97*`{Ev&Iv&D1 zW)eI5^6s~ zI*A@nrt|Ia7u}D*M>#;K0m4{Y>y^IC##^ndykuZO3(1DA*reVK2=(i zta!2h;B=Y(afp%*hnDT*edoxl z64z21Z+m`b7&ufnKM_{eqwg?Oo)Y3u1-{*>Gfqx56LUlwD-z1yFX=zf3JXhn2!J<) z2G5Tq-%CzYK13z%_;^c96GuJCWolvyJ~W6zE&h1Uf$THg`9z`+{=WGf-MrUo*ZQyI zQNjg@p$w{G`R`saG}Lcp$3sdxvA@aL^=JJ`uIPa7_@m2m_fC*ZHbzNNvr_*le19Kb z(#2Nyh62M!k{RX4Z*$UXBsn>G3??D?{e+N^!WnxlC|}vXdS%Fmj<2O)eN-A7K1Y6a z9W!N<`LttFwoD4i%DSVFaw;c9#O$gvvx9RL?j?t>dfYQLFD}TG9x|Fb|JD~|Co8WW z7aJWHucNA^r2zDAeF@I`bXPe$vnmus(Nflrg@rVVG}ti`M^<;&xE{Xxs50vls&obn zaR@j#4(GQ&%Au{GI640UQ38Tyw?@IWCKt9wea)e)Sor;? zs7Szlf*z^A?wf9_Iu7M+vR^i6f?Yhic4sTMYFzZg{c)BwG-2lfZlai##+DV3jYJA} zAK}dGVCGR%K|Mws?zp~zERZ2qLY)>t389=?;7yh7REznrJ-6$FJ-xlT&F^)Easfs* zQE9e}d+cm05h|^Io~gs%FMN#(K}-sYSDxyclhu%@$|#2?!NSAvB}PH50jhuqe1HD{ z@S#8gL_r=i28Ok#@}da>&^~8@AfU|-Wooj!XMV6}j=H-5O`@oyLyU!CcQfv=_#?94 z(0ak^cEl&Yfi}IDoa~6#QNHiIt)|8>4*UEY7R}W|p^|`aF?nk%d2n=Bj$0abZLq0@ z*~i!*G#XmQmx41o2D2%(oIE4xq=v7z#L9|_9hR0hp2+t^v+L8rtd}L3ne9u{!7D7E zIi42=>%2F5p92PAvECYOPa#C4wY0ilJl7%*`dk~@!VsQ?Z#ZPL+i)1*gw_~eST8z` zS842|X>qHmJ5G(yM90T>rVEqkMIkC%nmggP)Ye*W1*9g({4Os3IXV(XRwF`YPIdau z?+wf&no^h~4(9PChkr;Vi}Cm3!eSgd71t|LU-mykbNda=euCbwOu7!a7FT8DxLVRy z%p9>?@$-7xW+F*RNCPC)z=e6AZjxJg*tt8l>9g}qFSDebILUaIv`I41e`^47sCb+QBoFb-w6kb38 z>$vQjvKZ$f4>ItMVZNw$on621b*j=qWCi4yuT?np5EQybih9-@hQfl+TuyQYpJi0e zhYaN9!*G=CiAl}&M6{Si6UugC76q+ZjmYiQ>6T~7ZO^MEkF@TGs<2+}>$bMj8pkmC` zjg!-(b}%|WS;5p5OOT4`Ga>6H+F+|&qX4Kqs#eI!&28ao@^EtYIe0L;kS3-(k^#Fv zbhfoiaRWXxXm<`k;%T%Y)_{BWdv`9d?WiFozrBe=+zssBDihHTLNa!R$YPfT z(j&NTC#L@(bH+3+^3>Z=Qc#lXIhfu6Gtf6`lp4e&#RcHyyfxD;EX>A!PXRqT51ut7 zMAbcHfaVpAdPOX~9@FhU`Xe9gs`uEkhVq5P@LAPO9@!s5mQN>K-~YBzV-3tW4T>%H zlUso`A%Dfn`c+N#7G*3%7JSa~9zb{>DXcqqClue@p>Kru(BTwS7KZHnjE;`O+0LM2 zP}_VkgMZ7WsQlc`swOCcM53syU18rTt?zD&QU16*92of?y)p=1V)0)XQP$S>wE|hv zhFtS&-`56K1nko%4wdD<6LgGm-P{@?%8oJk`}>CpF4{OfDG3(7%E_rIkyK;H)e{x} z?&xg7elz6OAtkUrERy^7DhnM?W>ON|UKsQaJG4w`40!b~MX*Sm-wprmg)KyrV}5&l z5=9l!n`=(*`#$TD-cK4Tsv0VkUEJB>4}o8K<^oaxfPD1(_}*$**SQ8(#l%Asa%Jnp z1^63qN2|xZ{cRlCn0vDKUO}1GT^>!dGn*ZDzgX+}cq6|h#kb$zy*z4)(3LO#S$#QF zun~y5H#)gsH@v;{I+UxLrz1gfe@^@N>(G1qfsfTalqC&`2x1~m)|Xd~y6ZKxs&qKo zW_eYCv^N-2MVM-})^t*kmz_U%Re7rg)%F*Nc#_7Gp#86zxvsgpFEwc&!MCaw(*796 zs9V-`rPL-zds4CTk*M8fCELp>m}VD56Pcr>Y(zJYVEngI$WE34tIUg!-y}vq4$(%^ zETC)~Q>fMoL}h+t-T#YEqic3YV02GxloT^opJ$i4Mxc%dOP>?&sx zSPa=XUY{EtoNUexuFVTbNkP(H+TR;=p&ZpVKf=_$WX2tQlUUiU>R$oj6~v1@!P^H{c21z_wb!lcn?W1zw$=MYGO z`>-HgJ?ZxuL_B+_VBnl2DO>i25B(_sFZ{K;_3UY`Oy!e4w9%G`3vCS>&y|i60LwPD93!ZIgXck7=i*E z=;}%MwpDfSG*b`LTU|q~0!4d*^!wx#BYPLqi;CWK#VBD+)CaeGiI2{D?Tqi*&-EO! zvaiE^i@~Db`uxgOFW`tkSvT0N0X-qBO6kLWLa31+H7&J%DJm<)%iro7?lGM79%mrA z8?`xPzq$Vb-xMMz*W+j1}P zj0mb)Zm(up8~@I!p$wrR?`js zPMi{2fSW1LmPFfBuPJN#RYV!(cyJE^=3nasP;ii6B&DkTT<+IRYxC~ievGOenqa}Z zM2tChcP}r1;oOopV;wC!$G)88J6m^Ir^iP3=rZYkeDw&gyQ748zU6VTyfHX??5T4} zLrtH?{hq|}X02*Exfwud=_iM+!P&?SIA{2w?NLk@6;Vr4URG`jhA5q?Q5~T^dN%jN z5XOWZKAB2~^HYqmMRW)?ft81aI( z0_jJ+yRSqv2&Cz+{lrFgHsZfnuVI=(yVi59Ik$_BYKK4P_ zN+4~M+P62>hneyP^}cXW-Tve3=roHQlUGzzsw?6neD1wDD)w2nV5X9Mj}S zfT#|qPb`(nBUVeJi=5uhF2DEv$5HiqNB5_d9R@@+kGXDdJ|AT24TEz|&vPEJ5-m+L zYz&`RoX@ZWPuyY1a=5wI{ux>&9oW^J^Qlg<93_$q*(@>-XBFQ zi-#(Z&qElEjJDo=#Iygs9uyK2iQXxNFO5>0vx7GK|KA~0_VE{J(;X_Bz6iXDgsaBp ztn*`k6aBv>MO8u@vEdeDjK=Q^H(@=nNVR2rtP8#KX!nRDuDkI{NqEp1__ZqDw8 z%81H8oa`zsuvpr1fz`A_>Q%M;pGfF$RIK^oy>%}Qj{dvZM_YRVjN2!Cp3+&fx3}c* zqvhjEgE2a+JWc8Vh95o1%n@_hk)d}S@%px}2^=+Jj53Cj$Tz|IHzrp(-KlX zJlkIC#aQ6BS6PRX(f0Nccd>sd;B~p#hj1GJ+dm=&#-EtNeSA&k%|lbs<Q>I=Q&u!d1BRF6ExFW$CY1-S$d0dp`IYhEbroISX6piD#YjD)r+rHzTq==XS@) zPkig((m>COXi%k4ogE4DGu8GcAwq|N!hFUHK8TV|ZS&jW!$#K}?9yejUh_Hvz`J#}TTQ`Zb2nuAZUas@D^iGS=HhCD_{J)Hf zBU3or@1>aP20)e=kxl^Z^`Q(Os@jT(-=}J4wLhiV)0aLpR6^$Ef>>baiNpt=Xy83& z@|WTzVX_ZYie1gL!{U~YpS<(qqm%4*1U5P+`y;T4qeXZz{1$C>#z)v11=Cc-02G(j6L zetPwWL{i*!87i2bur4iXVjee6x9bUuIwDU5q&qCM%aiE^OcT2EtCx@T^xbkt5zvUN z3YrI-FPygPI-w&vjc4m&C!P6aD5;-6IZveYj;{PJq(Xf^6~96hfP5S}i#O-7d!j_( z!s~Ix@R{u#Jagx{g~NMZaA<1&^*GTn1B&abHLE7nuv>>7Agbe*EdG4Dcp;=3l3u>| zrlYbPO0~$y_jsN_wn^V0ds)8q#yXJ=%Eb*k-`Ukm<;6eX^t!`8->HW2NpE_bxtR=< z(J5PQ=64!7_bFTMX`P;u)oIfwS=_Tih2uV05k>BnZJUw;_vHc#1VamDYb*+QIQwaFZP&mz^};>_Z$9B9u4weIpm5 zu}pXE-qjVNisf6%5<4m+kXzJWovplB_xAW;-eRKU6PHO_yM`J|IF`J+d_ zp2g=jCRdfq=k+4~Yau=jtfChgJqs(}CO@r4XG#VV1najM?5_g|zu4cAQ0)vzWqtU( zQ})t7$S&^ME3NZ*vdtTw@1yK|Ji)GUHnHyiaMROS-5@8YTv^N3S$k$!kanDcga7u2 zANRYpVV)1aIH9V|b}#Qi912fnbD8eZI(g(fD~<=y<%th=NV6KBG1bUif2zvEM4v!2 zN<}o^X2GjIfR_SV*kYf zqgu-@w>I7gc>ve_TXbeK(ctE_KU(rD0wmLTztrJ!upF~+lO6yO74D^gKt%6+P@0`i z`h5*dD2CO~A2u64o$whmC7#>cdwDo*&5f?bfRR3Kn^hLsMKMnWq{|8vKZW7DyU2<7 z>`KijDTK4%UoJy-^rfT*dEECqDr**0idWqBAAhARdgA|TH2&*JTbs}x!9KXXF?b6^ zE{7NGg?%s68(&QDISdTcO3Lz!RaPZ@nQw972E@e|IEw!~Nj~wzS+6)4t_?^PeR4~0 zb0q!5NZj=4!45fSE_2*8d6uqy+7{-AFo4nmndo!;zPZTEL2qmN66!(v5%UrnE`|2r za<0FuH+fKBlvUA`ci)>&MP9fmS>E(IycAx14;uF0OrhH#BeU24T_6kHk*+f`k^Lzn znZ{<(zZ5w7S^tCy__M)%kxXo{Jjra^Mno?kos_g9|+rAS!sT?z4Xzzq#f7#EEt@=9Z9vAfF}Kv z-xGw{7V+wY-&Q1WYG%f2*(owTEPtBhznoRBV2b?0L>I$J+k34!_aNBGp+LNo7neZ2J~X4Q>|@Z*!e7$)Oy>s8RLhs$&mOGU;GnEFi2ac&2)PEUo@^l zG~U@vZE?LSyDR#fM_YPJL&KofYVDziZqi*0mO@&aBlNpAH!A z8}h!JvFZEXBY6`5>s}N$^0##d{(9$Z@@3iQGRvfn!M$<5#>vSJVdaq@W<4|=MFf@> z7m+IRU~OBHt&d>;V6Wy;vVQQp??WA7+%+dr-!`IH3Yd*;& z{Fu}A@$dmPMXim?qrOy)Lq$ghch>W)t0-n+kF@EAnDfU@5lLE4R>LIEQi|9>PY}9` zO7F`EansAD@HbOMcVlHGCuA9up=}|MJ@#n#^$*-1&fgwRKOn~2*v6kI%PSIcoA<>c z7;xs%j>SdB#qm|z9Nd>YlsDDq|{JFuq2n4TTMZ`_2@$P8VQLX-;$HFKdn{GnjNbD1!&&EMAAI(STgnV z$Sgk;EsodoL|B=?8koYz&diZBd0?=Pal+@WEUQ{vm}P0U9Q9ZrEEq3sT}U&KU~(VL zt|9dk*~zIoHwT}*2!z4t`dzX4pfEg+)rlHz)%%RAg=>`Kl_IeE)RXqHe<_VuoS~+- zGEmV)M@R3kG;yaHRVb1Ex9i-T?7>%$k>@8fI+Vs`;s!v^2vI3ue&3$k16v&^GC^38 z_Y)#5e-jV`PmShVJ3MnW7#oXKajFR$6YK`T+dbg=yI}sisNgphbK($~dkz9#J;bIy zo=EVUvT}gQ%qxkDZ~_&^`E@V&hqY}jxPN|u&i`16=x87VXe%qH#YPihW5%D-!=U@g z*#;o*KL7|;azY7_mDF5^?1cg)#>mE^N3OG6E z^he-9$1J7TUs%*wZ;ws7bXhhq!GNnfSbc(b`!*VFgfcY{p;xa=j3Swb$7%wvA=4Qf~0D|FOh4N;~xwMV{c!5vOr-h5h6P1O0kS7m&099xY9 zk=1&?H2Z=mX6!%%lNGWM_**>tXr)1buBn6UEh1|SF}6q>zxOi*<);@7arOMX9-+Se zeul)G*`Wf=*Z!RV74H+^SF=r(&<)cKbH8oFw4x6KR{)HLuz|ujSU3u36EhQz&ehYi zlifeE5)~*T58Ib#_SXyigT<%;r2znPWhF}zK7K7kP{5iYJbh(x=jH->IT^w!19A~E z?)I_C>9mJ<-Cw(vFs$+P3H?uZ;Otl>`YR>Wm5C<72ucTHGoc?qSpX$DJ1GT2W_@)r zXhkEXl$#w~;opKBxYl-8ig+r~jgTwk!I9o-d$pY^noy@^Wwq4HDDbHHNl_`*L?(Ur z)L@tlKg{`$$7GV!a=Ts>eCALwVOr@rNRi;0Ke4ZT6c!P=2g=h$oK@e@0HCURy1>JZ z2cynL^=w^g@&I%*>9f>qaoxCnN^s(#Ck<6)*mZe+SH&3#L7$XWS!j0d4>)>Al%xpl zcE8@|)!xlIW2Ypq@!s-FN@}dt>q7@}AA3863?7(LTzhi5qB?aO|7VQRH}jP4bxuys z3`4q&g_YfXhYvY<*d8sAVouiwcy!i5`k}#QQdo*|GPnEefvl&CyfzHO=nJ=dP&*n_^v{y3n~ap9;uHYP0?{$L zP>)vz!F|01%J99gB$9Z%hkmcXMlnYGEqkkBW=74!j@AY-gS~GA>^JZfZw`ZY z3am)G3!O{Tjzzo^>Wt{u+=4OSXPZrU;iCds#ogtO;UpNK$NPnBqdTLsU zyMYt9>oG@_;h+Tlx;^Z2+(4QX?#uzM#pz@0ly!3y_NK?pDTW;-*SxyxqupdV+sUm0 zY!1iIHz>lQ3LD{hJ+@}^dmS3$=IzDQK0k?`SB5>jiJ?znVCp`fEkt^Q+;)2}DDU?N zAH$UykWgwpo_MspAw+hn5!>i}i`4n?L*T#EJdEkw=c{L-V?nzhG6HoZ4Mqjx#`1EGez4SekT7};Ejmh;pXcDY1zu^{ zE@>fA)r6W}msr36;x|2{$`>>Nl$4YQ>xL_BDs{`eO&zR-TFmxy ze?S$~lF#M(mPXrcm!Wq-izFA`)ydLP9m&&OqOC1BD3>2^Hm|rNj`_Xi1}FupYjV{_ zty2S$Hc7U{BqUDe>MLDP%RsFb#?cHO%fy5~++`ZCS5{s?BLpcU=t&sA%#JFMt5NKf~) z(xWhU-ErpM*d*9dwuLz>==Coret7^g$W?ve~3PE_|A3w=^Iav(kd1H(KsbWtEj7+B#() z7|U{g>l$X*=vmWvYF{lj`vIH#<6{S`h<`w1qi<_?a@9szSkz807;R*|>;(*3Q?pH8 zxo6}W3JL^x7-`8WvdY?}g@yX#<6YhTY2w%po|4`Ia$u^XoSc}OFw74@B}GN1vB^DC z{YBp+Sw`PY+<3-ik5>!YoNlT%2(5cb{%aDRWSU4GPs=`&DCwF0_txqy3AcCWPTZ|; z0(k1Z(@f6cDr`Yk4j_vNyqZRm4v!O*_V5=Cj8YRpx2Sem~-4CB>-x z(XRmgya+8zz^);?o`0H0H>gkrX*jc8Feo$!$9huIe)u&9PnBC8`By>*J1HhufAsiB z&=62tU0GX2w6|xaYp$?Yv4+{y!^v zqOp-nNeSBkwDG`&3%Y-wwYoNsIdRhS?IMNim6RK-I=dPHDrXK;q)7)$7b_1uJw zn5qBHU5{CZf55I|LW_c~hQZq00_S%J=-D<_EK7w2Y!~C~SRmOJg1a_r+4#W7I15Ci zZN8B3TH}}HCB2zf`Z+GBjNAl+xMf~1gsg-zl2Oq~=vesA4}F=#Cf%Uh$f}5AaK}K$ zVf@Y+ciHF z56x2F()43{xP|`73FB93(>(4l=LwqqYv*q_h zvp~RyQ8F$+pG1rUA~5jmf1V2(&ycx@AqxFZBbsVB#n@l9R;t>X^7uf@(9l?D?=;%) zTG`q()B}CH{`W@=sAu5!FIA^}RM!@c@{}GIM5Qxq_}H*-pw!lC@@<{ zpyC5)>IQ}O*X~%`+?fRa-d zgNH8p3LBSOz2mv9uN@Rr|E7f_20T4@*=MjM60qSQU%>Cduit>`O{fg=KWwB5bt&$| zScAGudSBFG_6rMvJPRIRSuo*4)kddRJRP?B5bt3BYBvS10KJx=p#Hm>FN09%|9x|V zufj`9Ld{}&q>|ZCdS0z+_xl@A06FO2^?yeLX&h8F-iM$6x)ahPe{OvJWkL1-eF79% zpHN>20=xCs7M`|`&FVI6SI;rUxXC41qoY$-#-R#*pskQWJo6+w$+O7&<*Dms4XWez1}2tualo_zRmK(I^Ln- zqhey48WgAS6uS&AJL*%G{kH76u6muEh;Lew!h^(?3^iu`qo`L@x%T?_``Ev0^OeGR z?ZmDGzR^ErSzF!ZK(a`P72(|k@dJYe>)&3!9NjzLS5qG%*!f1RwZ^7oai14z zXLX<=4WB>d9PIKDWjw-sp9wASoq4o_J{&^QqXPeC=D!|C|1G)y2?@4jBQnsds_U2rIeO+_qF@cCk@IRTL1a%)a(+59UFMo-2QH#SQScnjWkCoD+tP`bc`1w|GLcnTt(WxVdk?K5^5H#a=YjA5vG zG^p>>+~ED;>5qJIZ9Iv2J;KnyfHBI1ce;>QXs;LO*=wA(wy*)p8KA*KUdx=FHP`@c zH_`uRD}yn3oXkr+rb18|)0u^2WqHN2>T<|MH}7h@PE}cBomo%wgWah^(zEB zvII%90M7}gKqpAz2>!MiEd}Kfriq9&aYDEEK5ozR0D2i_)ahO;))#|4a^wsA?ahyZ zkVtUehW&zZe0NK3T0rfO24C$4YBnzRkKiDY{;5|@H}cFxX)K=MWq z@pQ1k`C%>j<{3V7CE!x0&7S1nWsaFnrC#(dd16Beei#erXQXOV{!V4Xlg1tBwSvf*BVA7o^BlDAoXgM(}7qTO8}!x#pI z{Bdy;<1^>BPE*t4&jq|HO0y7%Fd)x<#r^^q$n_4N%El$EI{DA`rlRa8~v z=nk67123()y0N{wyR)>mk`p_OoT6eZ@|9ESdrKwZaA!+hrd>y_T6#vxUOT>;8Yf8{ zh1(V}2{DkQhE^c5v)9mdnVXv4dW*9=Ee;2nNH1sOB=CHcHU{TG#^gb$xVWieQd}O_ zFY&Fvd0CZ}&CT*+astA_Y6&~$67ot)T5@XBe@r8PMF9uS_Fd$g@5-WeBSfkLFa|soF^FncPc*b4Y^h$uUUtxSphuNQ#97&a~{DvA*7`u~1DX#a_v)O_@~x(3D6 zXo89~R~tqD+5x^W0tF!@E~IJV%xN>HmOjpJK=A!-_Z6JAhD|h#t4VJmy-zQ zmDzp}PJ5-bJnv>A`XAe=^v#y-o(6t!C^tKJeqzE)hxXHHc{*kPTwL7jXF^aWeaxEo zXl7$Ey%~DGKK%gVo!&x(&A)g@>CLTo-vO#3)^0fT7W=v#^1*lPFCtS|X7jc(oj=_F{KmU+m;f81K>fhPXOq!>1idSWPYqlBMPVwoFID_?d~x# z+EG$cB6%yUEiDZi+RuGk=^4;e6BAn<1)nNJnl&*|B|bQ?=h&E)l@;Zi22fD%et>_- zzyN4&Q+-*y9z8MN$OXbaILACRXlW%_&z46(C2o6WRN~(Ta&jYic{XGxb~tVe4GTnX zB5y%?MNL0yzt~@~L9GphQ!~@QKJ8ah0m^g{d0>DRj08s$tgNgA{NY?+!CoSXo*WPp zkdXqyQ$Blun!F~Sw(~30pkN6eH`Dm2SBWaC7qhM0H}AJ*It+m2 zn3%`=dT+1}&c#MqMS-0E9jt$a-h(&TEQIM*XrU$^6fq4}H*4`WPxWnurKOyX>zdNm z%h7kNXoCF(dlJevYl-nZ%+yBh=g7UNht)N;tIr&)v?f;+Hs_OoZoRdT-JO78^&w_g ze9c%v<|6PjMwBk*9n?IhE4rnS7jHHElGp9wY5xVGv*!B6XKHe*&>Ml{jwR4q3aBaI z@v_w1W;TqWLg{gAuxDza-zx|YkawAd3Fk}6G8>k97B`pWw51KuB1fp}$vAZJ8^qlNU`$!|`C9>4`_`yK4>1tKs3zJ~q&&Z;V6cQ~zdHQPu_?9`ZIeliZJlThBAk5+Et$67PKhf~V>Ux7Rp+Tjf^^(v1 z>E0md8Ym(a8LJpkns#9<$-%+$#|K!P&w+Rgd|g!Zw7qOP z=HLQFdU`11R20|KCr#N1XfK{ynj#1}A~jqVAq;hoZ=lFRBFpAub*jgzJX1zV`1>yp z22;ca0h3?TY`_xEAJI8`x%}0>IfR*AlFMl^4_WhnVCQ_-IRsP0T($-5gJA@`KXs~5 ziQBA~ouEj_=;^m^9ywh6pG2iWz==THpKW>=k)j3Xm&*fTL{RvK>Gq~>lk?xl@I1rq zl%%BjN!;hAi`93`l*k-87ba5gxXMx8BEB}Wf!bxV=XdDXgJ%eWrW3c!h{jT8{ey3y zI-mkl;#Ox`C==E-w6vxNC#8TkK>~pF_6sN4Spfj?VJ8Qbzlpi0;6jCoHXJyAi>mxHTNz^d2L?vy zX&HC^J_GF}0b5i@CO60hd#e0P{9D4Z z{5-t{GRu6v*ojdK{%*{3F1l%_-3gB*TZoE^b1tu|tz?=61rV~4W!aNel@4vNfqEE} z{{1|*hd%*~29!~^H>hPe3Y8e>TQJ<&HZnAhsWY?3g}pxdJ2n=XmiV6F1^FRI2h?b? zjZkNPb+=3QI1l1WdYgvI^+%5p3|%$Tk1Zjh;%u^f_fAGA3TQV1dE|Z z8Ikaoo$Vv|FLB-aCG+*5mLTaC!aO$l>({S>f)aWf8nMB?brV6@?;J8-US1t_v5$To z_W`g{PEOne5*WDO#&KT&ZwP!qEG#U59s#bJ4sHR~#33LsE;c`cX95E<=sZtBS%tmD z_BnuonYyO@H{f5@C4{tYcN2}>SA&Xlc=Re%=WZtt4rS4eTWs~uj4^mf zWM05LD!*Pu`RETr+ZXL-X=<~2r;vd?=80!J+5E#4^`vM_zF_(7`!5W3z;dh82>!T;%f1 zQ*tm9XT0G<{I^PjGA+?(+2?ls$N@>g2&kKtU@y<5|4g2J`>iKBUWdt`-P}?`UA+)Q ziS?t7p6O6@L2x3qBXb_8FswGvuAx6{qT8_m!Xc-)xnrl#3=$dRfAo`jV=3hbB<1D# zQgU|G6dR=rjktB|zu-24)hc=0OZm;HV^ z?F~`O7kH;L+of7PP8WarCKPXkx5Dtq1XC_&i+ex?f6;6_JKdwzFH|x!D`&0O;V?2fvrNZ^t@pu9kbW;G{Rj;$22-5eE;$q2&(o|!a zcTeDd5_1htkX~fBRehe#s{VOv)9*@_!X#TBWGHWLZXre%Sk}G96Iv*In#wX#W7*S#4`a1BZX8t*fY1wC<&{{s03~ld(Y&e zlKe9!IS*P7!Avd=b#={w8kg=K%Iz&&{hcbuquPadncQ5H64kcsaOr%xg6mpkE-rNt zH1(pGL8CgLiybZk&(wSVa{u0L{R6ae{KmZZKa~_?6C{=8?CiceYKn=061BDr+xC#K z_)$WNkOHY=`$0*lFInHj#kgG$k!nk_vK|*!y@6+PUOkBozXo)-0Hi#_+Cnd5dxP5@ zkB|3OPYb5WbN9>P`rdrD1tlr87pN3C4;jdbtb zLS8=CXbUjbquMQ<$Ju{YPr=!Nph$JL0j!O{h}PMiCh+9$R?h`I#~f@-Aj2Wz^$L5( z1Dsv}&n`DwCz5GKt*=LiaeIL;8zkqwj~rdo{n`Qod_eoWdO!s^3n>IZ*&>ePM%!~J z7flcSAX0IIW3H0WFaX7Rkphb*!+UT;&#u-lSIpb_IB8ur>tTsld zyD|cqksdlTc%CRcxu=BKZCqAC=JO4SO&GD4AE3hPN_9A{eX5Eq%+UqGBDXs!^p9%q zR)7=hDF?V~-<$??`|Bm!P_cOeHniDj+={|N_wM`D`aFZqLg5kqZSD9Wh9Ix=pwIsf zYk4)LAAnS9o84xY2*Od;dT#|DkTH;loECC|juaZLrZ{IiV8s^24S~j)9g9(bAi5#u z^)mfrBK{Ut#YEC5F*Ze6UcD#tYflNc^fgDq2Le3&cDomf{Jf2OESb8K07Cv3D9g^7 zdHsd^twlRm7M6)rUQLkNPV8OTHu%m+NJ-J>MRXpkrQNX1fVLqVjBUE$dc~+y8qf|7 z2#9z+-s2NFt*=Uh6ZO#nN09J%6eROon%&p+S2+L!*SQjt1stFgw0ZBmJa@(;7vY>a z8jHd!DJ#=!HHCEw5Rd>5r{_}}pju8AtKZ7}8o8ia=S1!p%gr;Q(`oZpZAB;PV%37Z z;dF>lb#g*(3f9e*JICsY9Q0Fc2N9@cM|KTwf*?rcwQdH}bsX=kmMlVj`KOOVu7 z6Wjp36$$w0uu50FgO(Vw}Eqad}u1$qm9*U0Gzq^yJ=EsIqm( z8&pnCH4OrZ^*XmZm}KX}UYyThQ(43fZ+zR^1|?3t%h|t_lmPhCaDM;U$Eb&n+wt4lU_sOH_4%J&5u2rOECDH2i?i zjE=4Z6%TE$M;fB}Hz$)`dyns2(1JP0jB5G8oj^ zw;!~x0>`}Be2yJlM5@>ACAQ%1-=doh5!9VT*vJ%sle6;~*cf^VEnTd%b93)_FNc?= z&u*fGfKk}~?p_Nr2FkTSN$(yITH|%lhZ09edd|P!P_R3bRRGa!MO$m-&TGE9GqBxo zW_r8@m5|@T5w(}s_2SrohhV`*=)O#w3-_kkuDptfuYP4^r%bcK8`p#GRH#ehqK#jK z6k>0GQNcPsIo`u(W)P2H8uOhbpWL>(YArrsVt!(1x}V4cUZk7M_7fEigWJIk!UD%9 z*2LRhswE_idjm_u?;&O=P#~$nb6KO1{0m>C9Qx5hQ{GRvf^6CL-14AQA4KH|dF~YgBOw_bIEF83y zxEhj;=++B8f!`}g&PHcX_$;sZAUn>dwodlW)sqv7WhEt{5|ZvOYYtlPTQW2C7LJAF z<*T1tIJNNKA@zh@aoa{og5prnUK<=)AcM9SzdyKn1|F1T=ljcp>9FRuk7=+`N%QkH z8rQlq*nlPh;nj_^!LK8$ywB9Pvoz=T_b?Lwe$TUc+=p_Y)-f+dX19c$&dU9KTJHr7 zoX-MF8k6?_fC+L*cmNCldo~E4v_*v+N0%gj9i+9N1CGL!3Gu^bmIgbQXVitX!o+id z<~a&$ z;Pm~Wdq?!qJpeHX3~(C^y?_hb>GY@@QmYJa*Zumn*Xt2u3gUoyMn%tK`dUX4aIn2z z&S+MRiZo8=kctX~6{eNDi+Jk~5ZIfO;izN*IRWpB`CR6TU{$9upm;VQa{@yH5wjzd zjS*%Tho!6zwAa$zAcro7vA8*`+uxttt!%)V?vwq*Z2Va!4AVobgEVRu)O3`8R05tJ zO4MUE_BCGUni@7^spP)1%=~25woP4l618`UHG@0(c_v^+!biw4pJ%P!bXNM&{5W_e z#0IQ1l;hY?fz!p3ZVjxikjvaSd+qNi7nI9w04FP9M2EQA+m~+ zo?m25alr}ESQ8K+F?*uBfZ{y!F9QD70~RUVp`_@fxV0C~$jx1P;NR#mdOc61+h>L# z67heydke29qpp1zMM*&v0SQ4sK^jTPk&q6hJEWz%L8YanyIZ=Op}SKUQo6et=6BHN zdEfQ@0pI%8@0&Gi=`eH8-1j->-sjqTU;En6g^vo$GhSq05BBx|sezY<-`=CSqd)36 zb_%&PM-xY{+|x=1P)mh54-xY+oHFD&_t#7btf8};rn1ti)Q@js%QL@!cSG>I5hMaQ z7E7K}_Ul(}O{U4JKhTwBrbCl8{djD>xIswIPV%=F%~Kk*zpnq%!^qaP@&v?2k(vhp zU4KR2z}aK<{E%+X3$i&h@``i(M0`2A$YdN7>ayKciU#id(z1k_Uy`woSK~h zx$lF8;Y`aw$0v2kjMS0~bsJjIMHwBjjH(g9Ls&PlavDZ{U%$Gl4xx(J1fdgyIXNM( z{aP>lp=(q4WMk|XCgyj%6x@?N^Z42qy0^GIXQVBUxHu;9Uy*_|5K!#ZIG8G z=G!squ?bI!SB4VHJYZTFrQ6>-_qd_2^2;fZ7lvnYa=Q*gVjK1WfQFSR3##O9;HHqC4NFydZJl4@&2fGXA7EAhK3sq%F|a@cbxvG7#njBhPK z9VW;>1pf5-9}sFn_^snGy`i-z?Q;as8w{)t%z_bnTZtZLO*wqnXDyaDT?(X3YBoDP z&BYshTy61IDgJD(6|{Oh6MgF8v%EC3QG9aRWt{NdAvZXuo|R%8icY04rpx@?iJU}M z@EeKu&ph!AvnxQwjCMcC0r8s< z2K||MoQ(bC(DgIS8)T zWv}IUTpvm@SEUsdNmMu-G=v>a@i8LbsEL*JNr;P|>WSi^`h9o;1zP6K`=6h#%Ax8B zfr=AtCzF}}J7c%VwD1(i{P`?Squwu5Q$^1pDXAedg}nUSH@K|7j@!daR%H6Rx0;PK zjgBG=bs0uC2%Wi9=^vJSwdi%)f{6+ozq_J;_&g9Css(zZZde-f^050fPymNo z`~CoVk-h1vXx?7o^txe^2+_!%M@UGhU=t8QaodLf!FvgaMeoMlNIvE%&|#L}1*(Rt zGZpqwlJ{hkE$2khR@YWR?&A;b+RLgI6xJoD8ZL$GWtdy;Nz6Ir$l||YHF+lJ{=%N3A`sAwot9IqVm&xAf%|61@g|n87L(N^Y4nf zMw3?q4Q-52x_|&cA|Kon=0GNh#^-VCvXq$T^!>=uBjsR*8RXp-;|M;fskyWi_zp+0fJBX z$LTHLbkc7>d+W%ES~JGy)$>ZbIXW`>bynu;DuEKvsGFK9|EL$n%KIukw?-kdsp4+A z{bWqG0GS-LG*AYzd8~B%*3b#WXeYdfITm~E=mJA-<~Lh`x8dh4N=20ga=MD(dnPBA z*l)gUZygwdSSv|X`h!F*h2_m@mH99xhJF%R;nz>)aF&)j6o)I*(|<5P z(6y$YA`^?JYeimM%=kOq@=oA;J3PB>6h=z}e00!I=YD9}GqD#*^RvE*fH?pn5qhvM zzi0tR7gR8ZRm4>Mk59Vz{)1|Oum^+Q_NLuK!p~0xYxX%DJ1o6_24*yZ{F|XBHbyW2 z%pgps{$wOq0d$HH_U}fg5MEhyo&cm^h5^rmVMqHiGQ>5Sl*nHfy5(sNNO1q2Rs&1! zY;UO}Y=#tRh6CU)QV2@X6OgZNf5erpPw9qO!n{_A*GifdS3BF83=qj;iR}l$3sCCKyiZ}w`^K%-^0+{mWD@Xl8VFprfT{esZEmM#Dm#CXW!*T69oQ1G4ts?}~7vrlj!`5zfk$@e@!t(^+9( zGZ+FmNGT;A&vHFz6T&w{P2vP_pW2VM^>mBrz4d?PV5d60e`d&C{E$AP$L(b26a+i0 z@faYq4-M9g_au=k;N5}#h)Q50}|m85q{?0qMcGV`KA7t7Wd%%p{n%4-J9Q|0&Hm_dXgtZs@-+$u#oyKRO zJV+u7qk_Demkj!#5LSlDM4MlbxuIjEr=OaWmPSQK>E-Gm+Vy%dZzv%lakc?e3+gKB z8)Rmk@6YOVFIa1_)W1CIK}P!O1kxQ(L}nE};_UEw8-WH{ae!<<5^WE>Z&daClJ4i; zA9buQx^RUU;l6yV5fZ&>+pOh5gha(|# zX<_28M>o@Ls|2WzAU&*&_-e?Fg!*EyUm#BXc9IOi?^qc`GJ~30d{^IG+W;Z!$^##f5*2z3pCi6vB68h$ zzXV$sLk=KF^2!ZF@$BtfUe!}@LqK0yuWKx#VIJl;JI7N$fmk{aK1e6i7q**iFdzn! z6nvmiswVq=gf4U3D~mNjj{GkkoDbRWg$xCA+Pk}tv8xiQ1_uzOe|{QWG(EZD9;A{x zf<*}Kb``YHr;NoH{PZ6>f(1BMNNLf?h1}t5JXqKsy7Ospo6J+QEubo^#?j+x)E zQM_SG2bH+$lMJhI!rnd}+}dbfEh_Z}j}2*4;paft8k)9BRK!gxHofC*p-q`kb*k zBc)N8kGVI$TT@zY+zLGrPwBMpCbia>BvLCIIN(*ymH^YJaEgAzS{AIM#SsRVV_ zH3jcs*$*q$Yb;GHwD_~Ho}qj$VA#XV%YE6LAouB0)1{0TGP*Uz6IWH9C-6B5Hi z%yWO`{$#^_2mm|{Zl_Efb5%_y8!rnL8wK|(dW>pWAjejq0X?2QbjM$VoTmep6M%1> zt>d{0TOrbPJww{6;!lo)lHXnq1$YL1DPoU|4H=d4qWW}kg}V4H2*z3Pe7G8Q$WVdy zLprb~F4ndVR<;iB(F_dMMTEP2eMLPn&^q}PKy~-{@JIC~EBtva#Ses-Jx@Ey{Er4GH>hnTG2T{OkT41DRbaRhsT z5>w2>HEv2SVN4Ao@g}MFJgASj!evk?h3H)>8X72|ss8Oiur~5w*VU~QJ>~Q1Q!yt6 z`dhc1iKK%2>(wh2`0rBqP-Rj4*t0{)-r6;?ey3)D`D`wUGmi2aXu{ zxzm{+yGgG>@JV@)%>8&C(P0V&`fF-}#`Yj@4AQ`N$y}YYadc?tS5&l@B5dOmS0_6^ zoC^)~S2da(Q0d@`qB#iLRp0Yyc1UIQat^{BvBX-{sDG;sGHgGqr^Y{aom|wQmviA< ze`@D3e(F=~si#>Aa%_j?e_ncHprbpz-Wo+JBoIR86GCnTTruW2DL^8+1Kew!;P;KB z6TVW*jg3u!AkYUi>j$0aZGe;W+c&=)86tm=;=v5Zy%!7vog{^Mgah0fqb#FlBOy{} zbhtCKQH&d;ujVtk)fOUyU;E4Tor-Abzf<^YK zKSAWfhgzZSk&b2TXLa0aui|?F>85%3V2wX6P}zdd`4As=X6L9?X=j1y_$?MUphu?_ zOuNI)O?7JpvAzAtL9Pm*9C$=s#~1U(3vq`r*W%2lXy|F@ zRE=oAOVTLL7&j`fXZ9;IR7wfn3qKKln0Y=={)*iD=e+BJ^z__$o42*Jp1f)qVj(AD zI&M17Xi&;RAr4a1^XU9jJr8fMX<^6ea>aNtaOQiyFT?b9G&GZ_wMeX1 zZ+up?l@JpvW*{9M2)S0eJhkg=FpoDdivlPDG;!)4KCgufXwD-ItN!jZ8-n3L2%-wK84eGuNYj)gaFfIoh){pj?f6jyUh?P9WHR^I zN&T@}p~Fa*HepS;do2%l(%v#t(>|zTw-g8K7?X*@ zZd&*LBLRwajDtnKYM9CtH0x+n8B(NDS=I8Iij{6XU|lPs?)I6WR0=PDgYpJmh1b=7 zDD-UdPbcD$Ojn}&MGrwA9W7Qvho#p};_SMXhvXKh9P70iNnqOKMC|lxq5(m#s;cl_ zZTNw$TdU96th>W?FS4Cw=&Was+lbTBspDf({&^i=*3v_S^AgsjW~L}vO}S5?l(AT9 zxjg*mM`}CiOMQ(Y&ZOM@-}7mr8BKf$eAq z=atDWYK4>s-AB)oh)$8m#j=xtjoF*jQuTv3R$JA%V{O+pK4CZXR2}Fwaa*zm$rJ2M zEe1Ltf?PHJyd1=7Mc6eN3kV2&WQ>W5vGqtZ|E2JYXHUDv1h%&pF99mR3sp`n6o6bZKd?YGWW~6@@Uy!%5eo=^)HCtHRMjtEEl0As5nbgiH`Fzo2>3;n?S4 za!MI~<}S(OMTNlUal93@WjcHwDgeKd@Bxk4_Ge%v0YPe-@ZJP6j+kYWHbzJ3hhIzc z#*8g$cE91}PvYaTC)LU4jgv85%Fx25r$v0vM-g^QZRcVTb8t1Jo>x(gsPPtd;o;Hb z#ljTl^dJp+20u~^uHFi?j6s}oJ$!AqP=gF>~s+9xJ4Z3b~Z5znKUQmbn&i^@vE!j<|nsa_?B@rTyXEUNtVfJ zv3I5=xBH+4IX$yT5|CBb=iwIsLsGQ1(K9e$Vg_Q9@w|4|H#cWrk0(6RwCws3Ao-K$ zdv#^$<;JlkQ*C!-z_LZd-WK-4LsPnZLix^O&1zM&u=r@}^EO#kRaNfaupXu@M%A0_ zSsIKe)Q$Ocinjg6<7i6{4I1y*Q=(`X7vWOpB3 zIn+~ANr;P^Ui=2H+Ml78CFCv9UcT2^SXh`&;j(>B!)vEA2LF|ppFdRut8kd3BC{O2 z_vwi)pOA)TqFF&<{x1a(LIo`*BfD}Rp<~2ynNy%XGyZntpsOY#nE}NH0^g1*{%jlL z;7-OOeQw`Ve?M3xV|bgEgO;U1d1%qQU7x&tU7P8}Z>BP9g7p9bIS=ELwi-Rows-a| z9-8ci-WgtvLWq_(NWKE&Z5D=L~Y}8O!2&8 zXyZj=$0VgMRuI}g-tV@dk&%|!{({t!4Pe``5*)<5xBwgCvU*Ox8>8B60T5ANBCb@f zjx8C6AbPzIYG0(`0QXGb@q%ktxMd6`IS|rbgIuHlbA(OCGQ2RbQYg!g|gB2kN?W{sv76wvc8wPu^wuL=Z$SqH>_Ot``I$=>Bk#(5*+ zFhnN|D? zjGE3j`X}S!%WjR03SUlrR>q}a4w8JS^V~*fOs12S&Qq;|UfM2ahsq0T1ruNB=<>Ck zkG=Mz5!?sj4tjZ;BZu4=R6ydaFXq0~C z&<#P}npuraPo4x7RBm$;JbTu7I4XZNmZEmhEQF~&zG(-D7mp#Ae>_QDb#`*o$+t{xb?usW3(R=m9_e!dcI^O0RY~Jkc@-FZF0F<)58q1|w;^HK3>w?Bc&1FIf$-buBC8InN(7AO>f&z6nOD^AWkE_OZ zs3kjv=1o*FKeS|3Ajg3Fa7ewsOExJid9^ZCwmA7JfP}ogq1IMIBXQw+12b=8F)f4o z@e97`rTW=hOB40s0(u5*7l#%}%LZ*tc}j1PO^wtBJlk1g{Un|Gx_3Z!u=NfH6)NSA zZ30w2v1kF3UriS$8MKP_m;s5C|L_fYl8ob()a>RO%xMm+^+H=i&SZ}^_Azs_f_b%* z+7q!?jJIxm%N*inW(A1zURGJ&>-dwk?Vb%4_<|_G_WA_p0&_m!-MIq${(@vj`*Hzw zRFX8$gi!8F8lj>2b#x9X6hi)1yw?hli;7dL5_x-O)3!Wmp#YKH1h zeYXu|<%!_=A<{OEw(zCE$NZOnt_NTarnhs+Bdvm5hAYE<`$B+m<|qA z+U`e#XPVazAm$44k2RUXSw$AIsHSuEM@Vz?nvSzd8oJQ>D4D|P-oQ0a%RIMkw=`Vbv1Upk70}=7TksZ|I?4*`c z3${4$4k;Lm6p*XhG+jQyPpb)-E)C;~Lt{ER)~~Dk;~36me?w^hKtdM~w_V!k7qR2s{1GT@t#Fe6wR_o}Zr&bgXbAU1RHMIj87wSa#DqS8Z*RHUQQz0k7Lo3#d+M@8q1`&XCIkN>&?Q=dlDZpMC_)2z zsc&Z{q9kKr5G9iFc1D85W2ak|pz*MUTANb_avu7_>1gEI#n0#Fb}Bnsd2~i|FCGJl zGL4zjbL$Wln}pxvO3Ezi*QUVc^#|Vt*Y!e%o0su#4FnNQPZn-!Y`XT~?9x~`&!3Pw z0P~AQg}p;?d787|bUs25j0?H-;D0=Cdkyxncw(DuMc>DSHD5NvwI>U%_26r3+)BGI z@L4QQ4nSAeid{O5#H6O1jlxiVgvIt?WAHgV&gZeuImTR_UhxuL(9W3YrZQR6@e+Y( zNXnN6o*{a-5^FC@xTuyky~p--iU-_&pblk9T$`-8i>$%)=66{A`HL@0FrKYpVQ)Y9 z4Lrxw;1?S#L=hI#@_c2bV`gH!uV@`K(DeBtW!v4&I}!~STE8uB48-?3|x;`-#fFjUsOUNqf;#aF*|ek$ny zsSVZhV6rdR$Z4wsK??96nl+fXA-+p1+ifnVm3B`pr;MhC#(>GvF4<(~vsKOnH?hP1 zR=<_K7q58ENiO#5Q^a1(RG6LklLyYbBWc%+~y&*~xK-nE_Ld9WaWoId059tPEpx%mMwf@q41s>DO~ z%i%4b95On3+GuW-?HBl0%Q0kZwFwOB2g7r=mUUI-mPM1EHx=TV+b5Q49?p(w0Xvl< z8R9bHvOvGDk1n@Bj^^1K_SxAzz8eac^M(e9EeN6kzYQ?d0i}Oyi%6Z=l#?^2ne>)V z=>92joo2M<+nH2*6kzc8ZhUGkjypg-rW-)c?hU;4q7N6Th7S^&KfzrYYRpBb-2;JL zkPw%Y9_k+$D9$ZlVa0W*s%qk&f3l>TOlpBrB&@0FekG+LV(4Hk9*4v2%2Q8z+vM(Y zZf9?jclrh7$LyrsK3SK_<=Hvf?h-zI!h`G;eK79a@yr94a<@t zC7NHQi~97W5fn^w*^kyxO<9edvUUX#)twbfqlSs)ygO2<>HG%xa9lcY;z)YpPlj&KLbn6 zh?$iYbAO^7GdV>o8dyPH3yXQySGh)F^@B9P=~@y(0qDO(QL+-@iUo-Rzsqx~ss7k; zicp@rtkFL-CTA#;!_XYS1aO5T$(H+n2N~TMJK#`ccr4M566 zKVO=f=1bn5d92*>kkujvsUV=6co>GXgY`W5-}_A8z{EsMx@P5`Il_bR2yb#y36rd8 z9bkk|Qb<%(66hJZzPB|>M=$KbXth4eVti zVpMOxl$a&H=*iwRJCiriV3+sbjJI}Y_# z63($Vu<)oS9O{WyIl~0M-v`fkcN#5U_Yh;DIN(1>!_{^jmZQea;N;p+9>08f_ctrT zu)h*JQhoEtqNA*3@z-{%TTl_`$RNPL2gBD3Oc&?%%%;W|_Z3|L;Z6 zv%kyg-wVnQe~jWhK7>Bdn+Fav4QYD_wVk~9N1vH zR6t2}4Q%{Lj)!uB!CdKHU0xdA_j(l!2HOpKhG!J>6M?2ZtXm$W4PL_94_GpMu^|={ z6rvsyeUQgUPwk}(HLy>tDPMN>4D@I*t`o#O!O=)NQBQw24U3)G3nhNWh_`aW_0>W= zV-To+Gj zKlDv1Yxnbq5I^&{p2(19ekPGcLEr3pepo)5BKn!piWtQXnvg$$D$^&A&*z*_FW&Te zY;0kBIPUoMs^X}0-&xz+*$N~oSKN!e$DTcHzE~T02hCWx(g$*>3bZIbsPIPUEm?b> zZQksI`$^WsRTe|&&b`Es7dDocMo-n{@_r36(ZuxX-qR9!ngXUp+~=~q({4XrCwpwO zZkiFkUk`C4<;#Z{C?K@Ft{$9YfKnV4b=nuZ=Cu!EyWOL3S|7mJ6Lvh=$-V19bL*VN z_URQAA#;dBU*tCRBaW65{k51xnOsk8z1E-nB2AT9c#5G-#~CLo!hiHFUE%F%Egs2? zA!mYUqOQcIl9*q37@6#)%~gFQ(8A)BW7Spa9J)STPSxW8%F-DuW}mg?o{j!KgC=xc zn73N@eRXZ^Q^8l6hU`tXE>?r>)5|DvW>(=_TDyb=;;9@Oqkxs?= z+>sFaF7Cj6uvZ#Y$&7b+RysA6_)SK;Z8_ZuhuK+;z~jLJm1_+9uYUd{I$C z@fs?=lMsd9x2wl<^`Gt9ghPJS8P4}<^YRb6cSQ{<}p~OtRCLE?)JoU3+J6z|l4@uIja6^u|_jP3g z)#5o3;pab`-!KUnBzr!uFsIr+ysY!QIY|QGJVVmpS(j4h? z;Y^>%Q)nsc84=u!7tq88&p~p8R_ZzXW_Fte>q*;BcbOpRHk*t{`e4GRGB0 zDX%!CPZ0UaXw7ylA*8xVn>t3j!HDDD+ZSQvcSWNdyr1v{HF^R_;Pu`WiLDZf?4r1x zonsRt13k$pdoh_sMLjh1St-F51w}4;LBdeG4^zWF#wGrw_xjD?w0}h3f}W z!T+|_VYRKT_^j=74Dt%KIclicq;W<2gS*X>8~NW3TUh0={I^$NkpZ0@u16jG7oViE{``w#hPkf>tY3lKnyvS46Z<;&{_ht3o{r5Tgfi{Z zU-cn7l&5!Ubx3My*0v<*|Ig!fa@XRQ#f&5G@Cwq45YWsFJcQV%NJIrQ{z^=3Sp7SR z=b5W(@17E^-o5|6>mR^re#yYu6*m0l4p1O*Hc3y&6got~zc z_6-g7=ev&&lmA0I8#-DRGJiKnOMDiSd+yTZ%kiT_s>5sbvtm%AVO`w;o@@5=68d=2|| z_J7~d$^3U3e_uSO`8QC1U;Mv#$m{=rtqoq{vt+408jw09n}GZt%4o@fT5u^SsF8ce zvYN!^>6aP-vzoDa6`wL!?bm*%{x#iV! zo(M`2;hPwke3DnMMkutlu&y8f=-px?QWm|w*NB`G;$K!>Y2nPLR=BrP8<|lmZo=gQ z>^z38tupm>Q`+q^$|;CkeBt#J5WftI#guav6Sj&tMkD{xNBdYLC8pQw`T@^0Z08*j z(EV1N62m=Dez(4(lMfYBPzfc~vM3DplF8vo`sqNy|C)xixFY>mmCUk^>F)mK40XGu zioSojm*(<^FWYR^;)JCHEb?W?;OxdHDkqij2eBQN1o$(q&$pI#C#TkpA?#mr7OQZM(>O$hprAiHo)!yI;cVL15q_R=tH&_EZc`Z44D z2RjFKaGE+Q<(YcpP+t2Otl%{sx7F=@e|k9PVznvcq zKl+79=)3?gy-p*hW>+~LO28IVC2=LrH(DQ(d(#hs!DfXFJ~(m|!f3jB25C@m%lXKS z{CctA#(Z_GsJ^=ESW$j>Y*$)V@upZI%aCBwgbgrCY6{9)X8^9x;h(VGMta=A+qf`VxBY8Dn3`Ps&DX2xE@}1ta|Q4 zH05(lY}t+1dtvy|93+!f6WmBhIxqio>po99k1_u^H9i=}cH$c#w5mh|K%Lvc*aEob zt(m&4?K6JQ^Bok)4z2DB{Z+$48EWt6tf^BCae8oG&CF0XqTR_hqo!MTBWd%wmW^ZR z{yT1?QD(S-CdqDcGUxFslgHR*`&A?Lwa@AE324mC=BD?!b~Q<4FXoy|j7zVLZJGAz z-hJ}+xnlPuZ4D~#i-SC#nc=Ybo$=c7A&sod#0IoV#&%0% zD_Y2?pO}3I#5aiX3Aw?#N#$>mdn%Y-lsD)0(q-D$eS*F6kVZAX`bDk%Om|4|<(9c< zrf^_fTr8(;KY$+H!Ex9kDG2@5HFG~bvjTb;{!fRZ26J7LU0rk8La;>vw$0y->gxzi zz5<^^-}q+6)ut@TB)nf%d-E1do~3GNr4~ayToJzPQ)0PTZ|;ZTZvXIp*nk!+G6dQbiw6 zXa`=j5*m2{14D8gB){j->{YOznhA?0Iq`E+wp7Pdob-xx+P8c;$a+n?R#Hiz;2rJ6 z8oKtYB{*jH#fI(a#yxXPv1{dda$OTuzwx^0Rs3lFo3qEByIj`5i6Ma4cWIV)?eat& z+?HY46EP5gOAGj)>5gNgFYDPfG&F2PcF(kt5eT>yy_TU8^8xcu^K9UDo{T5ATEV~e z=X|@~ecN)Qy+j&&deDyJcv;)kW6tJ4K`ZLuA-Z^f%!s@|aXA(UwQ zeQ-vdSOw9NE^=O-PQLbjQ!K}Mj_1|tWdS%v6O(O664}8)bziJf24UKY_R?9u)M!y% z>Z^6J04#|vg4vQ&7OzgLbvsvZtSN30DcZZsWv)JpDfW{zO=z$i?}d>?;a~&>RbjSmb)Tmzi6oECOK8_zVdN_g{nx2`NL*!f^7}CT7YY zDW@^GABYN-&~y#bz~V$ts)MdZA+0YAR#jJ3d6V@H zKQdBMsMff0CgewLqVrS1-zt@CM8SoXil*neLP}swZuY88Wr7%dlE3T|LAR6R?8nIs zbEASZ-o`d;lmJ%I{l_+I9QL1tu1`! zigFZC;Vzl1q-PoT+&c-Ay97an7$pVkUSuRGjQ_0SIxhevvsyvA3)T?+gZ9)_a>}HV z8zjV++^HW0*KEgI>)x9e^e>Z`xU=D8>@OXIfI1WGW&+;oGH!`Dyij6z1DSNRuduZ7 zrM0)0`s`xV&db}aOiL0V{ly|#&$>q%rLbIi{WuqmdgW+=y=LQTi{BLmw|&x+V{5P0 ztkyHix35>Fcj{$w)XU);2CB~2ZJ)b!cYlI~3(t9qog^NKP>4-faJP=kF7Js;+($x+ z5d0Uf{koTG^mu;+?kc`|9$f?>Oc|X`1h#$x7LvLCj3f4Ty=d7_j}eU=te&9-KG;l| zhlY!#P<3gG9+tJe$e$IuSd8J!BeP{`T-yj+}#|wX%!G3UU(Gl-6uKZ3hPTsm*6Esv8 zlvJe4DZ*QGpx`U`Ek4#jbMm9I${OU75~DoXI5w@M{_ss{WkH?gMNfF}W9Yr#Jx{vW z>7uiY=jCnglFs`GhS-9Sq5ez1G@BHV8iaQ*;L<+INO1;&BD zr3vgOYBEdO^|B-Zji5=;=9*bbnEE(qe(S07h__%j&z_;1j+W<1ip>=J^0PHWgHM4% zEG9XH5S(lJ9uk$$f1KPo@1)>~dQ1!VYd?EAd0)y>f{=DCX8jgg`%@lBLH?Mr4~{Yx zc8W`B46FOOHe|`nM(K-#z$evJ0GG7g7g1PQSvNBPqR~S+nQ^2QK(l6KSNEKCC3|et zhRX$HxAqwluV<8@IqUdacAtj=oM;w?%Ra%|I49QT_swf{hG;4as%kviqqNG&-3LxzJ zTe@LsdxO|u0urhh`UXr6#~k}Xlxw=BD!{ztV+i4}tImEZ%2-Og=!28zG(WK-JJ$g31ulD+= zA8iV~2XPuxkbt*whIh&0;xwb-MMiez=mXy=Z=78zTEYmG$CGL5y``hi9}K5L`J6G$ z?rzYah$&xT(~cXJhuBkgssxD%)1W*l5C{S2J!LJ!lweaU4D`}W4R!7Au>M@vSv$V7 z(tjfnH8b79*geHJzR>qvX}&{uGgr-#jl;xP-yes9A?GD~eRF?5{qL! zepuVYE`-&!j;XSiUe{sUWl@gSD%`&&#Z^hjg}+w2ey!%oVMS_={4c-)U-kb<1m~YQ zde1YUR^$Kj?*$p9e)x~N_y3+!DnsH6GxWPmTeU3`H}SuX%KQ-6h@)jPkQ&4Q(a2ly z>M82tkLKrp!*Vke6z%&t2tL8T@wQa?k^j4sZ2vz<7^cF?O3PmU12{+iB92MQQkN$ojOCh$NYg%$Oon=1Z-|_GBCxGn2xJhPNOakt`&AabU zv(S!>X!pCEv@(%^5S{+3i;@&5V$?01XaGq`1nvT(x_jR%PGEy zNqp^nU}tYvR$f|NSyiThL2g!HOn%1&06>D4jmTVDGYKcXKu}RBkCw6cmxAK<(4vkh zUx8&w)WbaQzJLwnuX$)k*5`KCydE5t4H68p{dBaTW`hzy-MfeJONw^tX4u_8e$>#q zn{jVoxI##W%QGAP4}0$!4rkQ04ND|QL`Voi5J?aO(R&F(LiFCFw+zu4Wg>{^5xv(@ zgXm?Hgy_99YILLbGKOzUa^KJK{QJK5&-dq@V}2NO%~keZdzEvoy-t*ew!Wd|^(%jG zS`0gyv_s5x6pMzJVvhXAZEUPqQU^{>4Qa%0PoFg|(0^(HCeup*OqD($gW&=2 z&*kg+3z1EE*3b&TOja1`EZLOvrOi8STBnBg)C05!&)ShDl`%~Xn-Ije3D15fOj2sr zTKsYD6NZ}~IpJ}E!dxSM9tgv_=^#pI4&JN1Oil#se{bvgi)d0Acw-_TIm=!C=R>7? z<|9KRkTDn3p4xQM?jUC(E3B-ftS<$4BK0a#`%_*|4hyzMU#kG1Y^&ONvWp%2 z^OLuU;L{)V2JZIlyH^XT_rSD{Dv&E{Y8TrN(z?x+YYgJSv3s@GS(i}Bzaae#aVB&FG;sL)f&iI=a9OWoRoB{VF;?p;Dxc4pH>l-WpF6bf)86@c4ODUbd41q{ z9PYyt532C$3d&XGJK6*jfMhD>TI-*!SDwx73Yv!|i{#P#t;7JoNEHSn7y@cBvqD#x zFRE>NUx!GcTX04*x#y3D;xWBKB%qQ0@N&yUYRzDxBI_y*Y z0S`Qy^o#@o+JjdptFicgFH=NLXsZyveH4|KSu2x2e3SOU1_U`$hs%u$6g1~I!Ro`& zW$kALFsMM1{J8`0h5*{`dr6cdhRw#JMdePbg16Iw8P((FOZ>p~{Y0^=Rn(SLJQziD zIOD&pczI&f?@Ab0pTy!Gr|Y$!iA1d;852t%a9~+rgPWdEQ_VFvd#9!a*G?{(*G!br zlQ6k_D)NSpUw0Efeg$uUcBEf<{}BjqnFW}xyg>_nj1>fG(@&)l7AV^xa3|%FucF4N zjVJ{#nDe@tA`o5d^=SrFeH%$|~D_8hlU||D722dTjKbhV*5NP>00}t;* zj}(xVGS}uMm1X7Z;1%~H?)k%HAqF~oUm}p~`A_DlkKmdsVtB~PmnhVEq@}u%)%oG+ zSzrU<6VoNobms=`sow#>zEbhOw7t;zHs<Qg@w1Q9vc{vQ4X`cQ*HPjf>r=wni3xScbA7z*Tl%)20ti6hGy8! zTvRiRjn0FlG6+q@@%$P+T2{PvZ&4m5(aT(p7DyCL<9v!}yV7?L4ABTb2wJBH2sV2T zZAHC6r}!u#PmnJVwH3-4UiBrK2>T`CI4=oO@r{@BNY0rgAETnFt(}%vkY8F-_THfc zgxDaa;57LO=3s7@DMV^*?(N#*l1ZELCtEc^4SiK@C0%{E^5~eR4iv$^dSPF^u9MSm z<$&KQ)4@!&{!U2ufxw}QN&=xp!h9C!iNyFcS5=KU>L!C=LTJH_B)vfa8R_dU^i&cw z0Eai+xuWvG&V`4UH&Dh&OI#^UnT#YRKhJt9X#>EbMO75TJsU-Zxhzj}fdwopExx_& zY+XoCJ`68OB1<6!KJi{qCv)DAGtUS$>utzH8&8TZ%V)U`5&#-$NnJzR310a#FU*-{ zWVK}`A%4m`BgVYERb)GkP7kJe_h;iI|HQhZ#ifJN2ahjx&Idc0SDjkr2is{Y_@HGf zW)&vW5~FtDAf&~2?aK^!AY+56mmQj9VDv9D&s(O;+a2VcRo^Fk-LAgjoIbPSVI z_s^AIKK%gzbABXFQEyu7(R6M~d1zTq5CbyVr|vVW1^!0K2vY79vRN~VDFT_Kp;{4X?h1cmb-|aoe)iIVX)~9aJosPst zWlVq*#m02RGJtVBtfqrI?ZtHN@s5MEVZpdpSvw13k@A)h)EUUs5rEY;%kDs?FDHd6 zJ?RBs3Lv9YrKd{m#_fHkUz}SZpCKw?>4zF+@egaEV}lwMM%2k^sOOC&Kcv#6@zSky zIr)SkSToe$`8rdvB^& zWz~THG)xOth`MX>xkg=?iJY#ona`?r65y!kWMy=iN{+Az<| zBYUN%B%sU=5tahhc22?I+17+;)XlwOZ1+m%d}Q_RQ8Xyde=5Z^p{aFyi*kUuZ(@3+ zMn*3pUhh3hSKIkTfKVx1nK5Liddo!IElL}Y{mG-TrsfwgzYs)+WiM&8MqY|`EyvLu z_R()ouwVWL$YA#{KAUpmVim5m=r3siNo~{|HWFFY?kJE%ByMOkU6~EjBxEDhkOsYE zqiX0GIbj#8d-v3=lH@%Q&M`qgkO}>*wW0VCB>^V9*14}Swqr%ZtKw%upa!d&4;iJA zN0|KjO*bo%_!H9w&l3B+&Le|S*PlK4ss+L?1ei#G;yo{3^Cf!3F=Xp_@th17(OA;q z$8loLP|stusd%DLGz3OAj&H+vC}c21>)jP=yAeQ!@^#j>EA zA?5|ziwAzLq)7uJh8nzp2fEspFNaHMF5=_$-lnN`#c`EYkCQAdf#MysLQ2fJZjIN> z1$A}!%B5E-WeU;>%7)0cob07;#aP~T;;gI+8zABY(~i@Q54E9C3p4u7I-9fZ2>(?V zdIyjz6FGG&-S>FkzZ;???9O$P?&tufwOlZ)J&eAHf~5ZK=1|KFbwZ@uDPW_oD6a5; zPJ`^cLoWY-0}ABcZU^03tz|hgI&fa~Mnw7ThWIb9G*G&<1}hm9j28o6y5T5gvjZFX zsj9tJmkJuGM4` zRi^nG&@cc&!eO0IdZAfim;zTm={k!$m3z8bx0%PLVQWx2A4_C3x^s*uV%$BlhkSp# ztoCPB-U}3UPI}HstS~KdX;ycCRS!yin|7A$AWf8cg?2a!xU7onxinh_)vI6P;#_Ax z-2mBvVbP$2gdWEoc*!2nj5@m$&Wu6EX7r?8NzUC>-n-H+JrsK&x0G^TqIRi^_D}L4 zy*(-8zZqg^LtE+aX=^7F7n2yf6u-3m2kVVg0s?cC+1RTr&W4^PLWaPEU$$_ zcsz-9;qvXQUoM0Iw24VUfV_^;WpGR$cD!&ZAF`GR8M{3m;^Rq zpoim=lux`V>ARA6(tm!zQjaF>o#Zby&?H=S`I&+UrK}g;&~=07o%)L&$t>FWKpQ+R z%H>43f^Y30AT4bbFTW+}#Of)QoXpxQx>Q@No3*TQxz1A2^^dJnb%E4HxU91%wTl`Jfcc;gSoC zMZdUxo7+*9$WdUYQMi=S(bv?*=5RJK;G7J>HsLU*>jM`+D3{CNj`&ZQ2M_dqY+Gxv zIeVB}JGX^#q6oP&l5>0f9Ml}ux^-;hm={leY{g@Xf*_8|o*`0A=Dkm^Er4X^s=3`0 zka`xfU9!lnJ(-OQmcpEu?=5in0>?e-3q31OtJ3I@*TwQCf3JuA8oumOebvGUnaGUS zljsD3KgC$_Q7VcY7sEv<7miUwh#aDH`HVVrxoz?p=x$$^y}x{RdX%K_Oxq6pm}4*S zAiHSHb^A11)N4Np&TxQpdg-~ozptjIA$uJk-_JYf9Y2YRdhT6u;CoK-S!xb8uyl8C zRy!xEPLIT!Yi{VCv|zHChlj=SKw68zHf8$$){9C9d|s;0AbEVA9N{m`Q+K<~JC zWn-ucYZw1{)hwj{XJEWw4c8`}7psR!GRa9e)on^|?F<*>iS4oj#O2 z*tadge+GQbxjwvH4-w()pV>)zxdLB)r@jAPUJro|GWsIS!b%mHe2D<0kDEapQXu8j zc6cG`endsu+>iU-BXnBuRe9>ehk-_0Q?X+z!Sq?RtHSJcvA@WT->^?vK)LASe|;Tu z-wGrMfAsKSU1i416uPz1?aHQPeCrNI$St`Ga^Gy(Y_txu(dAmjH_4s_o2XIc*_?Bj zl8xY+4-N9^o+8E>L39ff;|e0RWLH`IdwI9ohl%d&Fv1t_iGTKbk>r5C2<+HI|6cK9 zacrD}btd1o9yWcDG0Y%ib!fPf>c(D1QDzS?%$;dtg7Mwr327BQFb`r0vvY3~qBNfP z9y_!K_-BWKom>o$=itA_t_VB6QS5IToj^9~>Z>QmN2TaIb;)z<8%d8#sd0QM6TC`h zVh$O<%RTbzZgT$pY>VK-+#$glLG+1Zl?Qi*Q+;1_VPz%eR~>ZDSQ zdyw~)tnLHIg5{$4=itPTI3l+oMoZy4xytD-N1-gL#&XXL-*rdqe_%}(IF}JK4{H&2 zu>D1be0HF&uH<$m{?7HCy^*%F(rYbCH$Nx`p`SP-UEl(|Kx>%bi5yGBmuR@Y5yWzu z_qzkQHMt^R`OO}r`na|lp*npLr;(d1Z(hOx^;IVtSuzDiB71&Di5YFx(#V;H2=@fU zr4(gMEM}ruIR-v^AosTqzFy^ziQtHoQL76tXi3r~xI-HE(065e+_-tEDdN-7$O8r2 zf$!7G4;gs)BhUm`-9ci1(B!hmM+(%=0=qIs+Mz^`m}ps;Xyx#Jzq2Kt97ZUJWgFIE zs4mb%r7~$xpB8$Zoo?vsBa5IWjT|I9yzvU=@LVZvZ_M$BQG1QVY~BeWaxws}-LR6|K>B z4U3EOKhglgEDsC{1MaYDar-1NUfPWRdNXd1Bo!a=0kA2?xcjc}TAnhvQ(CHp*M7SL zgO}vJApXkBzGoDI+d6_w8RvdUESqz(fNY$6)5;o6YF4o4$5S_5kFl=Yr>G}td_}*u zzKB|#1C2l}&5Rz^kr}8eSF9bsq?MHO_a4bXeM)%#^f)HOUzMr&I7V)!e&<7!Up0}@ ztIsoXxxP|D8xTJ!CzZTL<>g|{;Rl^gO)m>(E{9-13T3hwqPJe~^0kIeQ zG()cxU2;QWWMVjTfijaCBGf?0bG1XGXp$l1OU8-oPPkvuiQ__u&K9jmTa zO8SG+X>Q!bzDpDcjS-=hZyG8cFbM+-;#Zlnbk>kzit{QQ`81Q2 zMryp1RK};a07IvT8TyV zoX0M=w*T@uo!xvUD!>K}i60d}I+%eA}2KLB(UMbng@k#LVcODqQgr$n85@x(wGV_wHd`yH_XrWB9Ce*_+9@VDcdnJ9%IKD%8s-}rWZ35YS!v6 zZM=0yk&USGii3l?s%n-zzmAzttWg9F$GW^U-4#<`PA{W>?a%5@-T!6fbPVvPih z0uWRi5voi{N!^Z+O=jzpyB#JaJ=Cnq0_w|Ud&S)XjCGrAWsC=;QL*%NQb&YC&}Sh@ zi-?yx@mV5QlD6VB^*BW@+#fdNNWz$2z$l`=IpvGHSk8aVA9dcbA8sNY7GE5VkTu=Y z_i$9H+=%_$h|8ym-r&D)moccG zFU_1cz|Sm2EIZaTDl7AVpDY8C@NsScKc=VxNrsGsGUvsKO!a85O|IiMrrXFE_sK!K z4K}>nn`&HZZqmwhFga5bHl%&iF1x#*t;D@>&9Wtr?;A#w>;>IR2YB|CE_$!rAM*et?VEr{0>)|_H^3v*I`8+mn!)mNyGtf8s+BQ!jgt;9O zL4#+kzHfg;5GyY(E}ooH?E;j^Ej~AOrLi>Bc0P|!YE3v|<6+->-6WoQW4LrTP~Enn zcnY*U=iIxgQ|LR7gCiTD5>qHS>S&N-+8sL2vTsp7eK|o{$&AA6EQo1>P`%G}Bq{bP z6h1Zv7uB4?f(DrWv9Xn(xlM3KM{y*9>2P{fs%S_aEyF8L^_Sq9(-AQ!z4DVj{@l^X z+<=|WeVfpMbnUb=sD19cX`n%ganvd|susE=%rn-emioq5j^l0giE$3zb;BMtvFe4# zdrnnHDwpn#YzWi&x>_JV>wAez%41nLU$eiI!PL68HkM9F+ExSvBCH)FWEk zhIGW1QN{r7vaob}S+McADAQ|yZAq%KmlR^T%72~gLNI42HCYcXJ&~2g%L98Kq%JN| zW|2{oMx5!EKL^m4c~o$dGo#2N`T(7|Um)SWx^g<$!IF2cqlh7+#J9{Taw z2}DOX>B`z%q1ol1Qz1Iu;<-YUNl$(8Hc>2k0z(~4Ov|v!v=SNiDve65%3E`0*kZEJ z)Bb? zP{_G;B{p+^d(0$C=9vz{rL}nbhL1ar(+*&(&QpGhzX-pPt3@{dHO%01L`;+d<$b6G z9E_*JG&c|M%fN;%8M=DyQV`4q5#t=~Z!OpuTPi85C&xxR@CfXHIWC;G!S+E|ZSrw& zi9tCrM!LI+Zk_JOP%RzpS53dKIqxL$WokG?Z8!5*ik$o|0tU2f{Y;wlj0+w8!V_B% zm{^kq(shA5v~pxK6Ac+$ysm2=X(5G`jE_&(g`HmG-HAy&+5VWaG%4__Ap#Kschi>t z`IA#g=XR3D(qod}JM;(idX()J+F|^CkAbheNZLl29H0+yxhPOI_ulTio9pQ-jHOb6ikD}V zDi69@x>AECc7LofMYe+}P^KgK++uOR-_w78*P*PmhwUqkFb%?RX0?OKkEpS6a{mCA zp>jzyRq2N1?f@vHDyWJSPdhsvWwC3IQ~7k{n6%Vqd*l-DOdc`KiS%5P(G))?}CPFdw8=3{>z_X$e z=Dln_4AWL!i!`8Njq3K=>F;i8{vAPP*06J6;I})_!PGp1ke3Njj7TbPPB%A$R`XXZ zH6)-mLC`%a6$rZWobFc^{EjIS5p{1IqQ7?a%IxXIO~t<^OP+6BYev*^B_PNyrpRn4 z9<4XGw!yPJLi-qvGo)u~Xm)y-G01Q$P^WHwthvfWEvYUf*yKu*u=F7TIm{=lIU%_^ zseCcp!F)@Iv>w{J6WLW;LkZSYB72SP)S|*9J3UwGUT6lH(boY7`~~11vZyw`5Js?g zV!YYT$L^fWFeJ-6j}Xe=7qN8R3c>BI*d;@*K;3r>i-9s?=A~i%7Pq25XJO6l(D(z< zLQoYSJ`t&oybBbQVxem8#~N3!^VUNp%vC@K2e$Ru3cWXzv^XN$jI%I$=aJ#}ZdJzO zGpUs-FZ3gqh8p1J0MyQ1Q3-^$W0dr4%o=Pl?Z5hnF3dNnw@?=OqAYpbokfH(9cB$c z%k+)UXS-0Jn+~Q;F+5^o%IDGm1|*DfY#yc}%{MVowkw)CkGuaF5nqfp`4ug(+$AU1 zEhlFb!Pz54cYA9Pvxv5Faw?mr#uf|unyboWb>)-!I&skOs&J=4w)mx{mIH$U1UgsC z?fOf)D6sj3Q-utdvLSSnAhA7FQJ>;Y%vTjfT~%c>t9K^Wkkv)h%G!pNsp0f~2{-%R zLhvNmh+6x|?PXLWvazeF`6pP{fLV}pL|;WF7DyYT2t2jw*GR`eXE%FQLM2pWJK@-m zCw&?`32O@-LzZcbhAQw@+-d|4S5U*xS_kg36r@&h@juR%*06;u-;0P$j*l}pGaIOj za3@ZwtB|Lg{3hj8nwMUizN`C2S<1#x%gryS4#EbbA;Z-Ho}pFg%oz`7r6S(WLH^P8 z6KJ{P$-%br_W(yH@LSiy_7-divzq$EtX0W%lv1_0Hd`%0kD9MmzzI-oNH4hl4 zbE+v@i&;LbZg~vR*lY?GahtbyO4G-`p%IqF6(?+dkn0`ty2^hoP~(T5w&^KzUu91aGBo4v8e@X zZb(yy>oB01U~>1!PoWCIr)d1|32$Ms6=X~?vJ|D>MLzHoe@*xA{}%pu3qOde1d#kl zEnX8@12zp=2TrSR-RwL(y9f4=wrFrWpc-h1{NM3SVh?dY{QwchB?Ms7xokco?o_Ts zaJHFT>u`(sPq+<^d`>R3>FZ5C%)4EaWbaX@>w|Dp(SWGF=iOR!?yxH>r5@(}Qu9we zynPPtK#VI><@S^vvS!W8$FcRFZhiaKb?!z}8shSsTOT&%`Tyq|SMZznE1QuLE)SYg z=iM^}|5FRu)Fh!mP6A1?UkNpSUgp%1_TO)OBp>49ywu5lP%}^d71|(IGym_Ap1%@0 z@=eQ~VD#VQ!^yI_Jm5dys9v*wYk_hGX9@kCdyq6h^(Qz5S;Dlreui6pIrZKgac^W2 z;>ZW7TDPyc?D252M-Q|}as2Nyd{x}d|H^Dfz{}=CLfUQ?jIIWe)Y*$2;0QASr)ssF7|0^$GNWa zU_tD5k$Z!8_CyHv=X?c1X%|sJ^Au=<^kmNdx`)jMR+%1Pvss&4M1fq6rHvGX>j0;! z_7NvD10$XAy7-fFZ=UYXso}~}m&zUy=L()v6(-`Q`n0p18Emw?-3C^l%n(^(KL(p7 zj%Q%XFB^OGDM<&$FuZeQvzs|mg;GqX($FB5sIkB26i8h#it-how2GYP9rIZWu4-Wu?Zhp_Ze$?AaqB+jOEnxawYYo6wOXWm;qrM#EXqZ3YB6eY7ia0A z??A&$8C2u_L&%ZXrykeR=|!mhLxP^)p9m%EzN>^4*R_hIvmgW4YRD7L7Ex4S1Dxu0 zEbQZZ7e5vZT|#KY__FOfI+%I3gtkYs^i;3k>gqeMGw+RES(?iv^keT5yG{E5ix;E+ zekS)-BmZ4-z7J^~%#m!TZi6M|?l+cSaY^ufJ?k3F`x*@8RcGv=j4db;(ojD^&_*z6$#p1p_wnHqsdd`KK|J60Smep~ z5(taNJYwJXggo-aZgO52s(zf^v-+B^6$_hxb^PS0N2d@1k#-^5x-DNOj@A8?|1mFP zT`~h6TK;8nko@Tergh>(Zx%THg<~F?6G&Rxb1vG{0zRln)}}aPBC#9YK^&sBa=9Zn z@9EUe;GpU5g-`=)n+(uiQc^8Wl`cE$(?1eJ~CL5*ncl^=#O^5cIZ>m zzMZ*2{W)}QwaJ_7x&J4cS1+9HUpOmIb!WcsS$9K(1i!gP@wTb{yY2bV=?~VbgfKuC zSbi;)WqRZ$t@=BW2h<5??f9iR>7PJLs}v8kwX4ayI_r&h1=uxCY0-V3GBeGyGEGPH zSp6L8wf0ZDLl()A5BLF}%_eWORsH@{|J35|`x&70X8m1k$yKq=jxLtq*KaV8y|k+# zw2czj0+q{%AIFhh(lZxY4((rvE><(p2>uD3*th|+4;4Az(m6P@XHz`0tuR>UyNmn) zd6T>KpXGW5a#E)}6=8r}<1q!Rrz|i)RPsN;87nUbD52J!O52BTP(0+VK z&M10hMY@rjlzLB|lMdHBU|#3x6oGb*jUAgsb@W~3pE;|Rr6|+GGqg26nDdONaXHM~ zxni;C{?nMuh^vWzBjprNcZV#2ZO8%*9vJvdq%4y;U%6+4P za4u?lN}BgsYc_foDmaJLM=S%Mv)V6iBfgNY0+5?uM)-}SR}_vh`6tsR?|D;N zH_2NIsIagV)s?D2jpl{DDi7Ey8yJx)#bf_ zJ*&31J;x(2WSb)Xp7T-koXgr_a)`X@#Sm<#zP)0Ew2bJ-TfjGegpZ<#h>ecZdFuO= zx#$^75joLh`4nb1W_j!Egz=#Dr;7ge4$aeuox6`xM@={$a zq0i8E)G0q`+u7!01VUqaPV8c5jr-$yEIB3a36v}*x>>&@UZOQCj z$?i=?VK7~Qo0yCJH#a>w1%ouYzV74U0|8pk+W7cs&c5E})g?(c0u-50BStLFK7+0W!fEXLhLuGFEQx zmOFm3v9Zm)1!cHRV3^MFLlMww5!~L%Kbo{<2;5U1UfuL$=zsE|;uU<>a_XQt+vIfs zP!_ekH-2oBQ&P%NIYAWnKc?SBNhVC5i94N!zMjY1X!=fYYa>@4P@PSGGa?_djE;?G zp&Qz9z^V%1g3U<%VyLWa8m(R1Selo2$GR1}#6Nu-wqvFjqL^Kh&zcj$->b`vAsjQ=u^_QBuc=nPBpG~@ zcX?j{0%YKUpOHxd-Zlsh44~cv=;s>GMP}dG+aX5zjPl zJqvxTeQ$xnM2)Amy$5d%ZKRCBv*@6`(lzLNqwXHEPIKX4eUfpI(PrZ{XS(*m@Z6D` zQ!RTh=)QKr3vvyD4R70Q>xKjS-J3Qt_um{}ZszUNZE&4PbDtS6Qf4AC9WT3<8uFFQ z#4NjoJPvs?PAMjN9t;Jjl%)Kdx%zO(3>43CbP=5EPdK_lds_eB5Osy|??Hcj3~1oL zcouy6ko@Rx`sebS)a?I;p5VOy>n9z)Oh390_H1G2)n27UkNEmv?dg@a7A&APRnEnA z3N8LrEzq8<>hXGVEE&;~JVmeW@wqoc@3NpH{w1}*Z| zkb(^m`xBSCypistJ?LHFn<4Acy7U$`%0AHJbe<^?E73n=i!@v;&_JE#>SSGX^_&A z-QC8Z~Sa>J?zy`8VBDO(iWxmVOh6sVM}awzc`Px!KUXDMDp%Z};Y# zH^E5{s6P7C;5u)anPjf<>c22la&xN&g@com(>)bgPxu++Q$S}=CPo$q$M0MBhlH!7 zKLix-I1&D~vPU(ytwavps@7#S;gH40Co^ZDu?@qzD6HV&sX0>1Rqc3bO}DHX&0{F0JFq;WN-!e&2wLhSuq4qT)IWk*{LBaeR+bw{O$-KN+I z&o%%JLhp*#9(QH9bjM~wNr{UkG!srg*=cNU)0Xumaxj}Z z-_JN2)sFoVS$meP<{y@<`ZTIsy}E`^^mKN;owyaGp`aHl3K2P?ex78esOVX6)0x47ihoePJ400;|dbaG<>>)_a> zvqzy`Q=ivY#R0ke>~ue2KD(zDeX7gJsWPQS_e9_GFv_LGVHZ^>BmdfcC*(R!EgIEk z=z}WMYp_Ez7_g2exgEoE%J#czyc26%m9_PN4 zZ0YjE)B$&3%EN-`??SM+79~?|xL3HPex|kMz)34vbN@W9 zf@fri63K9t|M~k~q4N(!bF;sR$sVQq5>4w<4>Bm^l*@%-!Mej>1v6(S12ph5Kfe+z)RI!ohNfA=1Z03+UD5tHQHaGW2Ky&+;RiEMymlE6O)(eH10UR8@NDaFy(!c zS#x6uV0=Vu^%biaF8dMPgxmLjh0_ZA0Kw#P2}XPcgcLKfmOvaCiCYhE-gb1d(^fwm z(7|uOyff^jz9Mit?R(Bj!F3w4d!60gT~fEG$R7+%wX_KEvR!~v#Me3e9jOmEfY~Cp z;W!})ecPIKePMoeb^7Pe>56Z)LcB$@_PXM)E#rvNOG;2j4ZM26wP$;yK{*0t3LyBK z-Av+_zzdH~X1s2HNiIZ()D5gT8{Rd@bS#?vK{)N!+&uoeuEI7Y*+m`!IZ?>>(ka!) zD>gpzR%*ftp%Bs3v0-THxj~CVxPqS6ap2B|$FZBv<%#k>!+7}-rGNi!F*u@k)Lk#( z&21FPx3szMRoo0Y0{~FSE;_kRGz6&S8 z36v<+0Ov}@jf~1I zh~8Ior&XKE8*Hd_5uFJmt+lbzcE47^@6tO4ThWv>2_`by`k7IFl^^A1XErTTtxh66 z_@SQG>4Nsmbl5<-#wVe(b*n+`h=@Lv2DEnF+X!p8L90A4P`NS5Yuq78J6n|A zgRCI_B-gM$(CeHwF*%X67FjPBk@Y$_+2tYPa&uRI)YD?4KS{Ry>b_W^L=kMn-BH6T z5AIseTu_Bvz%#QbJqRS5eh+!jA1)v@O=%K zmwFvg#T=lfb`fVPtFO#NS15?90P3BSmBW}%k0p+*1buS6{Wjgz9R|@&N^QB_x8T)J0PEs(FlFUbFd z9u!zsY>x4+MzbTuUpOW4ld;l0V|gYOMxK7q#mim|Y0OE1p=I&x_%~$$xfJxcTpM_u ztsxXr6PG5!vbR7TNYbvZu?oHwwnb1LQ%8p^en>~2iIf!2ZE96ee#wFw6&;bi5%;EI zHe?w>d3g98?st+Hr&3-36l->`o{;_78;`kRz;pprA@GX7P+!UTM6kUxa1e$rR#^tk z>wSq{Okq4~C&6y;uU~m>XUJUW7zKHI8z+UnyH~DOhHN0=8B6+r$V~8%TEux*S4O)= zITtmH0xT^nnL1LIlxsPg1_8kV6BX#U;>t2M`t_idb0sSkL-73(&bEtBf$@=vHISp9 zv17jLlR~t=^aknW>rKm+SmU(#{X>&(3(^28O^s)4#igY?JGX0fy^DV8Uq0&x|Cur6#5)&6? zY-WX8LJ3vDQtTe%>S$9mvT?GA-^;tPyKZz2LDlN(458W;?c-Ovkwm<15v=p7NU zzSIb7C!L*9Ywh9{|2$m>%^6hcF9vq8(4~|$$>Y|%5iXV(ezD!QBICV4j?H}U3MRVE z{T&tq*;U|9D)<5a=HtbBmgxgNdl?zR`+d?bE9u^>Qu_LU04bb(r#yN~&~dJFOI;q} zw2rSJlFFi6w%@8Y5Ng}yIXTsRvS-SCKKWhE!Kd_hxb=vw+wXSV%{qs`vW5NXAY0hB zshM*qGyx1AMq0iQw7wdp>0Vpe$VC_N368_Qo17d9`*ZQ~xh*x)tnnz23$J&R(b&FH z608z{9d59}E)aSSK7C)Tdrt!c7dyT5>ulyjnf$~Z5>(6^M{SX-8>{EYy|CN&`?9A+ zRaKu|taQ>;Y=~%XArBIqZ_;kY^XwduQCk;YPEUIwMw$$EuIFVKzSKzNP6#`|bPz6e zQxPHa3sMBnW`A)fdDLD-xL6av?C$C~L8q`FM{ZLSQBd5S{QPckfFtwQ$+ql~(Ehhb z^&4tV;%cQyGIYLe9qmyf_ek$qclM2_j|yJT${Lmb(UB+9tyZZ}=QKZ}*Uo4u{uyWo z8dj~E<5FA1iSy>Y2Ym>0MT@Ou`OpJ}#0VA6LTzc9qBYN{QRto5c4#Q@f*h>=8miRQ z&=i4ed4&|N!qxlKeD=EsN71UP`tIMC-r8y?DJdyy(}N9tsu%C3sJs}>RvETdsDH@8 zmFo_tvr!Kts^dWV*e_C@DJq2p(~KOjKh3n0R!B}lkNBe)?}AdvVVD3vH>N#i>EfU> zMu6TLKq|d;=Re2sV|1M zZE7NnMxr(D7RlilS{fQcAX=K5&!Tw1X-{}XT+vb4S@I85n8P|cIwv-*eK?4dx z%IlKk)S@0We9!szO;^*N}|oE;N2A0J^>baRfE(6 zy^4TfCTYL;@Gvs)+78DbAV-i$O?sWzQE)q=UmwkIP!Swj?d<#`-dDP+n z9sZ2Y*-2b{+>9#I@M0Bn%zWWDv`fMp`n3tf@;V~x_WiMHUjG z1aea=h-r#ryUmz6Peadp;+x_*Ls?psK1`vFhM2cKdyZcy8of z>P;Jc>P;|#*30rNO5`A(g3FR7^TX@_-5Kzg%q<^48XT;@9N##lBANOfZ^*B#!7$z*+|0=?O*-F~B_h`mrdbruL8ATxe8G<0}iRG%dwJlI8^vxupw9Yq`UBb1z zRX}q0!9%+GQyVx5NpW$->E_<`D_->|2r$XGzK5r)`Goq3mB%aXrD`{u{Zn8KD1-tR z=S=y<=SWx7Z+Gh2W#?`+j<_zs=-t}s*^;>fZdDs@wI3??>^sq1Q*V6)HL3Cm#W6vf9tLk^ph>;^Wyw0w2?t(hiwUT)3 zCU0kAQrvZEtk}4oXrlBYzs7sfKH6PB`JBPfXm2GYKH)I?IU=Gds8=X&db*{hMw69H zoTRg8Fe67ZqcePycJ`kOp2hq$hgzzudAvA4OD`}~7{)FmBf7Wpp!;bk(wK146ZH4n zmSq?JJiNx!o9LAvs#QGWjv23MDmrmKyw6^CG40dpN96aiIgR#3ywZbm|D{RV>-P40 zH|K~WFdpUghnN*l7fUI-p=7V)ITSzY%|7)z6%vM>h)2MCf;k0Z=ep%8nkgI~lFxCg^ZA8I77aND$I zksTW?EN-`vGh$Bh!-g?9At50%ZU?KYM ztLlEU!yDGf2QKH-ef`(9&cSbd{{kN$qSZC`NY-j1uIi44*7?JIlM1>wB3efgA#do> z269_p7KVal`|XhS`1@OV_o@3xkj$@=l0DH2(8|@7ja6OFG^fd^ClFZkWY>D(Y9%PQ z9~*NWp&F+8bem!0`0;;g0jT-a)b$Hlh=J4gx-vM=Wq2tG4S3IhgYO~K8uB2W1s72c^mZk*O~WAm%-mCs&DWl~ZZ(zo0e{ZO}G-}IXUvHbxu zHQw-1hVh(?jI7gw!=Z=oX-b*cg-x0HU!f{{>X#1H$EF$_=IZnNY{AIx2s#SaVwc~# zPj01Ns$M6nCjkB7>iAwI8|{0|a-wUn4$4WzR__04|9eHrP=)itmviNG8Nd=7Fj_JG zp--P)+iL4g_k4XCHX2n}_l0^`qWGnSkOqmkWB#h_{?4dqGJe>*O18OjPF$C)Xp&WWe{k!m)#AkHB^fmW$a1QDolF&dqg7^IiWaXAVM?0_H+ zWmkV}PJK%_yL%05Qd0Eda5-g-7oYYcHYVm76rB~r08HfI$doy&?i<<8FENoa#s{HP z{0qZm$wN&8>8pjlTck|zl9INLp5|fZ4>vC{u(IOvVkSV)O%O{mTXXlTa;9Y6yj^Zw ze5~Wf06Pc0Jf7W4G!~DI)29XH((lhB?$XoKhaNg3c{yY5!r|I9qNh&7+=CUR5k46{ zp88(is;UrS-x|l=k!T-*vxo#2NLz_@QF*><`VzOxfyvA9>XUf>N&by!;!A8mYrL9J z^eEcaJmoh?bsXMA-Crnjm(OU~7~8*^pWwY&bFhW122R=~3g>-3^(Tnrb3ZrIhJNyQ z_ph0FcCpG^O&u(nLruM7WuK>H12`CMmL?Sx|0f82%?8HVRtsC^3x&b7iuc`UuTo66lZNIY8 zei9_Yoa->JudJNs4}V$8S7VqKxkCONEQ3_LlyA=g=Fx5fl8IP~rKROjSNTt_%f$!1 zfGg+DymPiM8?7sb`lLT9z}^NdAJ#hX`jq15yn_ekR+~`ATeu4g?~~F61jb%_CpJDR z5G98|%6goV3fQ-8TFfT0^MSx3+qR%7R%0L1qMMwRn`myBDVKH+SexjX;OCSB6qoxZ!t+Ul(UDPVVW#VEZI@@?G7xjn z`X=PdO4w^wL#GPs{C$ajzL+vdKtE9nq`a3Bd3cnCIVj}d9ca~duyO-NN* zE1uFIt)gPAV+}Zr5FZMuFbHS@&uqveNHxe<0SEgTOIOl^^~tG34?#Cq=l7N}#j-M; zVTm_qVEdCNvUcNc*H5lDx3oxsX7V;Yq1Y1G(ziFKiZon|jVY*6m8NK5xfJkeK~RE# z9a3$iB{D^!x~hZB*(6O&%%UQrnWC6zf5qOBXbJm`OR?YCF}wzpqaP7A`XE{4gtSSd zwg0BTiL1WcYhv*y^zg{27#$13lofSAR`kv7sjZD!mb>v@JXY{&E3(_j%nX{8KJqR# zBL^u3&(Fyj5)OB(L$&ZEdo4`=#JCrM(5*khJc`2FcD)H?RsrdltHt_haMZm~5P}Ep7r|LdaGjtT^M-#yCt9&+fO;!Z+NM8mF@r0AHxx zd?~KHTkj*y$x#Y$#LZaeiTH9#TN@?!Yo{)#m896nXTF-Jn7}C5w}V|rjj%Ac&MKu! zg`}r&OpSBg5vp{H^;9=kwPTTP%j5e(_RRSMkt*0IYZ7261n~FGn>XhRpXk+++#_~> zX|aD`sjoH8KjTqpw0gwsejWGl7HV?>o&jV$AON^%hN^ES52tW}|B%T-Qx8)wfML^5 zaV)I_47w9O06qQ#YlP6SUik;2CDLzQm}P^2@J^u#r6pT(*xv8WB)U(*#If*b0n`Q< z7M(cO&yEaOY|@BnaAvn)6*4XpJR>71Ufb%gdQ&O8X*rn%c|=~|{T3s9&LtzGJ;DaF zL2Q~)+w{07P+}V5h`YPHbIK~v!pI42g{0>T0qAUgK_t*mw7x+RnAE;h9ce;x?#y0- zm%4d*B!H|C-Cabvi+{csoiCo&)xp@-!T9;}>tna>iozrz15c}wwx4w;fdmsLLE(VN z)poXpRX+5nE0NXra{!{`qK}S9iSriM)9OCnrXeI2e;U~V+zarDA|@;@E;k?`fCT5uS|Xjc#jP)I zGwRr6*ZSB|+D=N=Jl~6ptt69TXD(Lg$<3A(TMc=J>;H)0j%hIjwYUx>a(!G>cgvEMHeC#l)TWY zf`V>A{Fqmi4D@siRy1OJ+Li63smi%M{ssaxUau!`K)A%Q7Ru-p(+e{$3=i+$^z`>5 zYPLT?!YW2KC`_%5G3~f}XNZ%5;49^BMM7(%1)b%T-%57N%7Py=$ZrY1SB1tr5q`W^ za3pemzScZ+4|UJ^@MC1Gh}*}Cr#<_xLDBuzW8d!;6LbU zw+^cMi~2^7N+T`ZiXhT0t#o&HcXxMpcS(15N;gP%m$cH&-M{C)&%F1K>zr|%!8vo7 zv%hyQ?b))qr1JxkZgs(XKJkoBBrfA8$#>65cm zCVf7syP?qm@5^%B&}itw8Y3;k^}hw4bVjeCTs9&GmaU@V?yweuu7`7P{ASx^DIPhJ zBG1frXt1I8LCQ&rr~*2&&U4`3;NFx5GhD}GG`Bvx*@M)s#X1*|(|@Keu6Y>Gkhr1|*)x4TWR zUf5di=*UEe)tb>CjTLg6+X!-7 zhidJY{#?4GmwI#U(;x5s-#%%%zJvmYYZ{e-O;g5agP7O*`8+AJsAqygOOi5aw7|u> z1Lo%Vav{ump?u|Pa-Z-XI`GGpnqm&8xsFMKbkfe?b%ao%Bhp3G z@wTm$S4lz1$mk%O3p+7Y^Z5@w55x6VGSAE7<9!s{BP8{iC3Xigg(m})oe?(bk4>z+wiXbx zn`)U49t9t-Zh6%UD58NFAkF*oE5_IJ)WWjhU;FsjpD&@uoR0fX^O=tUs9@P4InZB# zUa6haWo;nQj;i;#&M(cyJUZ7B>#qg%a8EP>kHM!=j_Y2ibq_75EMID!0krJ()!yik zu(_HWz^hvqM0~XM*E^7ruO^2$$OHaU{`hZMjnCP@_TZAHC_;|v^EH&oM5BM~4qOuI zBt=`CtR27JthEAVG^BX3tA6)lhb(ug%FZ`-4(gQK={o7B^Q)yx z+|(t6!0GWh_AkGrpWmB5rEGD)+-&{HMA$|YB~W?1OEmKF|ew%3|Z!2o?T|1%ru4=K)Q zf4aD^b9d(?mX7A>@UP;0Yx)l5jR^&x-@9xVmx%*$?*twqj{i*gqk^;QNqL+<{;~di zL@BC|tkOSaD%b2b{i~D|6~I)vxS86ds@1!Jx?4F-%VYNsd<0xxFOewTL&H=b_=dzs zlBi$p4|B2cGSg8%vUS;vX6Ghaw1n@1 zh9brTbr7@4lQ^E?@V z3WP;O9<7IKVQV&0Xo=~`+FG(( z088#2^5j^D0Oz4&B{?Ap3wtyWbpWg~ZXGc^*qxWa{4#?zGCn?FMvwPk8d{2@692yF9F=Z^4C_ z$q62Vvf}1Dx!FDN)7AE;+6z0N1_FtJ$oJ*NtCL)9-_9Z1oO}Rm zXa~Ud9T2%!SI+?ev$8sUTQVuvBR)Dkz!l zlbCL^A;OUY1qx_|%LB_71-G4vNlz`d8{*NE1{3Y9#(g<4d*Rv{_kHfu{-GVh6Z_a$ zoQq4$9Q+ipZI5RuHj`Jj8&9=f_txq87ATp`f43I8Cf&C>-JDgg=Fb0xmdLe+M?a2K ztvfh6{Z4wFh3EI+(j@je+=&AXSAD!IC@92>=F^Km-2SN#Q2m;t`nA|2Xk;KUW2VJa zC~>kn)kdn0{AWRZ#pPYlf!MYplRROqm|2$0=)xe``XR7~Wn0n5>?JFW~oyIyvAE7a7VttCys zh2E6zbq73%=kE5F>z7>Z4lD8NpOOS?%^O94viv)h2ID|5za)+XiYL*iGj=Tv)+G{g z!jYa>RHSzT1mm%ChbHa~6vVHH;bF?p9X0D#+_6)(kI0!O#}vr_g>)(>yVoa{m&JhP z!?(MKZMGuuf71&*er$}4cuaOf^3kgYTUzcMVKc_G?Kcy_d~i(iGnx3_Y*jOY zBV)m{#II^CtIduJ!0y-A+nGBTHMMiHYj@tiN^O#6wX*w0K}zWgjBaFN z?vmoDFwGs{zW(vAqmU+`ieFpYH3kMq-RrGHMIfDX%?2aS zia@>)(>kK5)xH0l4Ox#MRGi9^#N+%k$lE9qpO%oQoDmr^$IvSVOoOVN;`ZF>sdc5> zFX~9Z0!d9&*U_`sJv#Z_tX&j2R8cLWB}3L!)96rL*D__Vk$wICig20t?(eV-X@FP( zLK`ZRlZpw(B{M6l+e<9x+FHiO^A)z~@HxKIWEM}M`N1+BOn)E6YlmxB0xJuHR_h;V zLRkXQY@IIWeV)G$x#wkciQ>xAI7`vPcE8sZC_n7&Vafp!AP(WpZ{%TvS|IRx-Q>}U ze$umC?quU3J!}~~KVRT%J{!1*n+E>IQ@xp{me-G*u5)kiuN-43%xCA`U>?`J_4?fK zp!vjDNg6&tO$VLVx`4f2{fm7(n-1>X=sbw=ZlXmQOVcX*&wbIL1}8*CC1tYJQ*MW` z2`sMOk7Y;XQT-b5&3$Wnl?}V&b^kKW8nqO(G1XzyR%l#Kj!yEFpOGfSr@!9h(jAOt za1?8cbT}_0=jx*nl$mVDl!&KF>mt`5M82kOzenZ}u!ky^--F?r7r)2%236HCF{G_Bv z?~o^e53@aL+O6P(;}SFL{>)ed5AE{o*#0oh=jJ~^cC>UDAIEeAV}~Bk^s?G~`#ThL zhijx;PmV9?-sNcBPT*kgZm2F?8yx?99-KfR)yO1V_uBFz70}=t$xgpHU-V}DRMrZ+ zf^jS9xi!U!4>m(7Sb<0F@=Sl_ChMEmzMQG*HE^&nb~-y+!AUuv7|t7$IKHQJxY^#* zepw|Ch=E@M0KJ0E=6aM0{S)q8qBpu9p_cOt0N6(auaR{J6`j(B64#ZkG#< z1cl8C_oxMR-|z1fm6Qhhw1&qV+vtF z0xE1!a(1LjNC(~7X-PRLnpCv5w|9#xP~`dK?9|+he*E|eb{o~z>HByqU&5d~e7&ZY zo*XJKE_bkXI$CbY^UO0^v7(h^Pue&88drZ6QAT-q4-JhF*fT#+nAcR)lq@{6Az~sq zzOujfxuC{Yxwh6Ggb2>8D&M?lMy5BQJijh+jyAI<(IOLnY;MDHzI}^%baq(;Ibw@-By3ytuVsGsfkIzL!B1i)>wX*PltoM&baG-~j_aKMaIQqXt+v*gG=!ZhugjEw{`z2*Bj%~<YqPB_Q^%Uv+aQp8X+n=ifbuVAtyr7%o!V)FPJYj+Ub?+Oj_y&m1{gbKL zSFwthFj@Y`{=UyVUWf3IPG8+#3L1jhsv`f$z}h%J`c#b{1x5#0eV@tP9|M8!^yb&H zA<~M_an1zqw^iP68d_3?MdAK|p+A!f(-QKb;)lnEY5pQ2xbNLVBXU%hRB}{XM>l3K z4j|M8b?4&pzU4ze2*;J!Up@_nxjtp6mPWxe47{)_^|rb5Hzrb?o4c9c=iP~Sf&O%k zPC^Rn`<(QPxs%IWH_FP0JcwqC9(yDM?0#~?Bo z>{PYZdj1sZnY2%S*Af?pPA>G2LMWkQx`Q5$jNEA^@ zQ;Lb|MiFsw@kpH2$&nrfQCVLRG8rlmII;&}PVYPDgZ;hBuUixT#s{<;eXplyCMz>@ zqVk%O5)uh=TCoXnSRe4v2=IoF9X|T%7@M0KTAAe((a_KclE;dTeqpGACnD|yl~j-0 z{%W?LFkPwP41vev0na&#;_Fw&Vr^?!b@9)Ffy-(_a%RhAO~e@S@^3aT*@zia+s!YJ zu*Sb9(PV@cOls*}|B88Y_TaYcLS!h|Pd?g>@bBU#IwT4jB8k*6)eCF4%fQCK$M%U& zrW#*5AyIw_W$KVccw(dLb{oULo6=(+JCVnA=Xy7Cp)c06u_A3^H0dKN!|8Jq3FNn& zGK`P=BroANd+hX<8eZ0P)j6dEtn3Uhe>d=M_x(hFrdp7;NQAb?Si_?=ilx@3iGEM) z4@FU;M_LFDXT7mCHf?ymbC!q5=veQx;O7sO90ATcWJ<`PB3$)`x{sQ|qM<>jPnfe` z-z2p=%%?06KUYUx-)(ofuUR;$hO4PQALY>LVlp?mOx;q~Jugx*4C12f!6dB$V`+%@+%{I~@Kll5ne7l6dHEe|oQM8ViZ9HyIn^OMUs5H2cACv@=vPwX`wJiJhi0-~f{8*|u;%@) z=N@x&eA=ozr&oDeQJkaIYJ2-kK_})4{las(H~#Ks`pX`s^BM2;({8^m9_`1Et21+^ z-Q|>OX-`nfy|D;ck9*2K$@~uzl*ZFrF zD{MBuaiq?Zsj$Y0Sc|P6Bk#Ajr(&&FH*3(LZ@g#D_hV= z;#X~khlXl#v=}cvaEDSJLQwJ0F!0c3?p8(hgDW&6LA z`Ruy2Sb1O82x!e4p95p-sE}_*r6Z3`RYbh@g2|nRF){w&!hjQ);c`D+`ukwgZ-_lv zr+jXMIXE$}y;&(aEm!^>dOpdBANI!ecI$ZkQ&Fk}SR8+P=Ks3h;v}Y!zqQuiJ&%B4 z=f~I5KCJsOGA1S_B*ey%h2|S|X=P2q2UpN#g^{<1ueVdGkX1tX1|{8gQ^YN464tc zYjkcB05KmD5w|5cIy_=FUk<=9u59;sd;{U)@yMh%a-o`6($m9`(l)q(cMY z_2crIU7xpmD*lOGwbtVKN(piCdfH#lYjc-QHo~I)M8k5O&7gvZZRghf?fUd|Iaf?C z5L6(2zNpe#bS>()Q$#wpVk#IqjU^unGFlBryMAoh-cf{IFs?Bx2m}nb+zV3TmK&DC z(sdc^VQF#I2ML(zJ)T}VueV88TOl!qexak#R28JEDlYe*|Hft_WxB;6dbvPYprEjw zjdNu7Wn?r-?(23u6TRbCsv?_-;S~*f);UeBbVEl2>aW`gK zbYmHqzrIZ`92sTq+Fhq*t!0cWFvNtQ-ZQnN31MTR2{n=BmX@aRQNsEld`&bD zLima~Dx;{FnXFyyG}@V;z#PTJ!qQO1#S;`HoXJ@eeLRs~wejq9!aGrh41TcD{@mdO zMXgECzuDFE9>%|ioUEFtKT)y$#n9T?Ol=^I#@*h|dr7*NROj&l_MFS+ZPTPtuuz`& z)AsXq+s6F$DD3UQnP*d)&`S05cTj+K%-o;Wv@hCUo9Pkaq$1pX0pwVFI8UFyKR++ zwl1d?(XEXPQ|Y{F2#V^?ZL7X}Y;Jg_3J3_;pS&CF2)+a}O1oTPf18`(9mp!g*xS|< za^zsQr8Am}4)0wS-9rTk4fNS_iwmjQ-R1BEBRT&4-z>nT@zba4vPAVVX$$>8TZaag zb4_lqld)@Qtzv>E4y>8<@SR(V*K!@VB0g! zN<(_9Ro5Z6TRA1JT4!_p=mUS)X6z3j3+aXrskO7?Uia2*Dz&YyYjr$_XJr9N?U*== z5T(p?<(Sk=L`J#6E7hY#hhI)EZV`M%=9Icw zRm^3=wx&;&fS{tG@p$WHM>S&IvRMe;-`5q|c>FOjUR#8N7Zy-)nMwOxa+r!H%ll^Y zP&~TgIJv3bv50Q{*J_H1^9+}*jqpwxKM88mSaxh`=4JmVHpJ@Xw)vh9`R$ZoXsEPh z%7>bY9%~L8TxHu;0ve?WLSw_iXky9nA#RtOGqw#I$Y10TF83W=?^~0CgfMOj>$g6c z1yi+4d4!e{^q;aa+H-rhYm??%*b5R9<&`aXBg?+=KX6oObIEE3Hi#%|h_Ty0FLfMb zSI?DQyVJHe*J#T>V^3nz-t4C+?BI_Xa#9QiH@p{;aj~d6%Rpi1iot2Gcj}1h{4wRS zEFvvl9muJIug`}JfeIR|MP%=e{pYegFZweU?X&P$CfP(T2 zs|9aNLYxp8I^k(~1k$yjpdhftx*nl)9&MPw$RI*)`-dSbp1|TSRBCDIe-0hulj6wG zLk0T_Vy9|+Nw*}(>nbbHP72ZyE-u;~{N_n1NqHP}h#@-ksaeEby2Koui(jidoNLrA zb`%tqt(RTlzrh=tn~w0N>D!l%Ow|NCl^0AM zqD|SVF1Z>e*ulh8lF}BImKPW0%`NJ3Lf{bNjU{n8%|{l~mY4f>LBq;PPi|bP8D6SH z68La4Q;ZDLJ7x(RM;I$ATiRFzA9hjB)s$5<&`}9wWmF`{;}sJ=oLEvD7^=7nb#-*k z6r6wfKlerVM{33(Pgoh2$fV5pgH!yWm+3CM;fw+I<95%twu5bm2NHXjO-?|qlh$_C zi_b8w;#vwMphA?_WlYtQmseKCM#q|K6Lx+=waNoCNN6ai{QEcI5s-|r?SMC4rf(Xg7V zW{n9}c5zSmfz)6?#^PN*u)?%{`k;pJ`joHsSa<5)b=~nj{V(zwYUt!@t$;$t}{jKfea(e^%jxOXZz{!rNaVZz_YEXmwD?atd>Q9h5gFI9*(P z;)cPJLwIs-ArpF8p_7(9g-0hU9v+4{N zwtP0}A#m@JqeEUy4JX*^hYY5F|L~-UATrbI>;J_@^U3^iT_5C4AvW7%hPOm(f`ZUS zvPH!tU7kB)kesQ*<`^dZ3l-2t`9k4!_mnwoVCFot+Te=m-e7*@OjT59e8%g9rT#Jx zX6n0-;=S`Eyg|P>7+1L<9}eQ^9HqxH4L;G1j4hE55O!^w4IC|!WWmvA9eCZL`TDNs&8jp1nWTEXp)xd` zv5AWEQgmz#^OqJ-jcWojG9;XZyn;qK|}G7MYu=PBjviN7lHoOORh)tSu}CH)6p=DiU}EwY#qoWas?JMMv=OZht)4 zX%)n{_sA?vn_fqOI30EZ2SwJP3Fhh46J(b$clT&BRwvQMp}o~PU2nc(p$YIyUOx>#M|#cy_2tRk zAZD#>7QJUfo?Nf_XW{f_S=eP}8wjTV7AHYd-x=m4el9^H`a`65VMo z{RYV)p<{C%5V%qiB0<%}{$KFNY(yS3cw@bTiNr!4Wmv1Pmp@5#U&wG03!DV>Oai=I zIH0Mwi_T^Q)+CbH@hoWTNs2#z6)|xdAiXiay_9Ov0D8CnU7wj8-aDqt06Ie3G@Hyh)88 zn%}&Q@agHQmtH5GyY_vvFGVym=-#`Ho3~%V8CIa}eWlLeOzkBiB{VWC^JvV62kvX( zIz)ES5-?J>0HRY|wq;>nTK*+jTLnG2&$M|**}7pz$8tI!Hk91;?x;-Krq}5lNLISG zs(NNvSSPUh0|7sFg5kY(enuoft<>Enhd>)9z+4N8@s$^Rt_!~0`f}eS6N8zKb3`fd z;SI?ZP7@*bJQUo*rnjZ>b*-z$DGU(kA8+hD_x_ERKpq<1*mbr{d$FPvHg- zKvaOHQ`PjlZ&zWg778*?ojrIgDIq7P*L_mbhLQ#4vvAHstNsncd+6~Fb|-0%3Ttv- z#`fr3b-4es5a764&Ge<4MmV!#0ii)X2E;cy6ZN4q2)~QV^MB9G{Y!7+g>ZSF4l53c zijMypC7$7F%UQM_o6O*fGFKoaCH*-keju6A@ZpGX@R`ZTC0NsKb&!ylcE<;2!?4+j7E5Tbljsv{UnnD6C%AJ(O47 z9Pk{Imw8=q?EHp==2KZtwTh@L5hPB2rbZ?=CF~spnNJy_3rZ z^`pG(osVCx-n>OM!6~Y+gDdU2m+QY9@fTO8@+@6r(}03^Jt{vN`}zgaB2Kn-3_WnQ z-hJm2BX6ixf)wD2g}2*NjPShPexxCS8X!!nG4&UDJcfy zDx6bzb#|b?`|apW&m}D64tnr?#ezG|d!z3v)LK$2=@|)0nF#~(YOK>U8XV?hB+i33 zID9=lNOgi1#vI6*r`lUduzaZGYzY#13r{!c1#xmIM92OGfXm*?$|X2g@8P2T9_vF@ z&m8m{sekJ+$3x3g!&I*fAaXY6#?(0aH2S=l{^gW6s`(xTI3Eg>Pi z@UK9mYyyz|KfRk5?SFBo-qs1_d{P&6E9M8R4cEU{TQdugn7E`Gvz;^ZgDBRhkMdH6 zr{qZqs0Xi@au&#gquz-K5)oIQ$@*#zZDmaEV%~WAjB^Z5WflAecbwBd)7hRj9)*7H?NtJ6841wgaB0z}m=`pa#CiL&_gQ9`xFs9OcB~CUN zp(;+HyVwU?mm<&L`m!b2koGjqjaxXF1w@deMSIZAMi7P%+3)glUhe2FjG-G4CS_DK z2E>QZFwxEme^M87u&JtPcy3}5N-8>~M-KqhBF}FL2qq*%QsUs-k)E~Jb0;hWXjMiA z2j(SpogYwQ{3)6LW2P}XqvX2+{RNwuGx#ts^AanoxXJKycb3y)R z@}j*xY?+col?=WU8Lv0y!D6Xl#_1|&xz-%<@);r3R*7o=JDJUd>R(l8udx0)A zL6dMz4da2lUp4*GY9Ab1x(Ukv93;$k@_i$r3YRX&`lfjW<#B*%%F5jT?LX(T&n!ky z|402g^$tPnHp0lYNj#jvnDdOnL2Cul;6T{t-Ay7yj{WJC6&1ZLdmhj0%8H8AZ&vvY zfu_ONG~iNlcrP;a<9{@eyK=g6Ee>fZC78NIQ3)w-@9X)}&-U!K@V+w7P$ea}ADDMH zQwmT+1=OkRtv9{{jBNzZ?n`?8#_p&IV>?Vr$ENQ|Ro-XZG_;@v-h0sMNmfbjo3n?u zh|8?{daF^74n6BBS3t=(sJFU*e(`v^ zwWQ$S4Gj-h-%2}4tv8z@sxQzAb^Of(-0o-L@>t!KWBt^fu!CctK;RPhZp^TND7>v9 zw5q*6A)BzcGVBC&Tdp2Ll8D$S5xcqgWmkyYgnfg8nwZ~ln1q|sdbwh-OQvCYmE+d$ zwEg5De_?f5MJqQ)70p1*_?&XtX8`wx$v$EU4Y7U0!>~({NGd^!6|t#jc*E+Cug$t( z?4qV5G~Jh@zEZUV*b#Ro&2e_>SbI-g{;b8MwXcJlq$JJGbfz*8J+V=$&~jcmGA=HH zDsF41(NBWC{jQ!AgiF3uq+?5sGtTKOo&A?1XQ`!FaPI6<-!|D;hq!Y;9=pugG6U_;Bs5!MBUJt zD*@MSQHkfpo%cD}qVVt-pOl&{STGHdSmw$z zR#O&T%nTC)npOPgqAFN~eM`TJ9V^prE8-$Dmo3rzbwU=QonV%^>&E-zxWno^Q z;odj9nY%bF2s`Q0rzp-bGc=MQ=;eN#--iLhIb;*#!}DYa4+NDkSxrk89->tE5Vx%d z$;lrwz3?Xg7an#DX{wP|^lW~?8eL&0J+Y)f+tGOr47obIFu6EkXR2Jmdcz-$#mxry zt%M*DF+Mh_!zkM2{;aH#p^fu(l^)ER=o&V3BXBZqyjv;B4}f~DTcjCX@o#Jrl;MZ2 z3awY5O5E4=f7ERCN(O?3g2whK#-H^%Qn`uCh4UH#6lDsY8#6z?YJIKTk{=(3&%AQ{ zMh4mIbm%(T+fNSj+L}(;!vx!)AY!h8P1Yo=O~9i%I$zfx36RA*D)ssPM2-`&*GEDQTcS(Zo2$LH`#SPE_f#20*^!Z9Xh=Kx_(8BXYEpGW z<7UA&WJPOADW|;nmt9v(nlB8~+iq*U?|L+=IESW4=Nlc+(TIu#Qqxcxho(}Cm^WOJ zdF{=M>AHMT+9K~8xl_lh)ckwv&u;c9aa$mp_ypWP)-r|if`vZM6Em83C5?=VWiPdP zj7)n@{VRd|{d2wW!hP!22Ddn<_C@`Q1piVSi_n7leU_WY;IwQFWghIUiR>`=hs*MO z4`j_VB{5QlnR1r9T5xD)5~p!*B4RQoo>kdqJZu#ijzE@_+ig5 zzaU+;kYMG4!%9|?1dc^9{>D*Z4!~XnBf`TjvznA1ohAYj*%IGEhiBQF-Ps{|B-AnPeim z^>*!6o&DW^pQd_I|Eb&9H}Yrtm>hc`IllQHC@iUGkh3t^F<4Z!4%s}#$9ZO2U#;)X z=pXB&|Hy#_+1>`8PibGji&e&Zr09Wz-NMLKr(+=FCdc>tbwdRPBrlvtIoCh-&kjtj zKlG>Hst|clq&24$huXH42cp3R@N{T^5W!8yAjGFzr>~Fv8z*z0VM7fgB62vCPe5P7 zcAVUOl0=d1B<4Vnn@279@J+2RU@%(>i=x2n`^ecig8Q zPeO9)%Yl(jh*JCxQi;58u{3tZ{R#!)>7K$6Y6jyih3029Qiy9Nfm0vtfwEq*%N9bnL4c2WGTA_?jv+<~=U+#xmY=(Z`WK_9BWuDt5E$ zXt1WgEq*G?T>e_=QwkyU~GjVtbp4@Ov z{9d%+G8@dw!onI8vx&Zr8y$#&gHGfFJ9Mu&RMS*bK8fKAJ5k`9q)Gl1@;lJ0 z;gw%s-Q~aML9~4PBSm6*M(N;f&Sc~5e03H zh&_vOwr^=tRq#QSUTu8wpKKLaT^qV!_Yl)uKQeN;~Shipy$jwPItT&P-JQgE9NI<@s08Op|Sn!cyAcSP{U=F(Z>R02tDGt zlUaC~2JG{`(4veX^vQIy+P^iMxLNE>8LO6cZkwS&Cr(=4#~iEb$*4rv35y&ePj@ zW>avuG%h#Kru_Kxa`p7yd>57dp)T&v9bF#5%t|96CfZwM-28Ty<4(He4WtgEXS2G6 zfrYiYk-5xzCmyKUSmlIYMi!db;*ya@?nj$If~LYMEj5(1l5eg3@y=L7Z>B?5{jn`5 zHBPTrPb_})cZgKa8u!`j>GAF6qQJ&NQL6)g>q?E4ftaA7j*5k{gS3sDZV}Jegql%RCk)8|294W23}@Ji=KToD``AlJKm6pXbRZg_FYuf)x!=Z z{gI_E%Jc##Cqo8Gp0ukTvY&DWm6fcv{)waf^Tw7u*bXt;4}dB94iXzCg?GuyG9H&x zoYXV`i0MC(T)qF{fRj?D2%%?V*p5eH+C60ieJ{?{IWZyg9luiM(;aqbyP5$&wpl^0 z3|WAs*m+$}ib5NS@)>o1JR_hWJWPDvVqIQEVa)by z=fJvZ&CRF4v7rIx>Tvj)%7(PaPpm%!@@-l5-*8Y%55($?U79qFKZDbVUmHp`CU&yj zI56BjkDgdOXlGr%HrxFpKo5n20CRnG;q=$u=ZO|=H*ziHvgPXsc{{U*>lGAy<&&Pd zmFTRdxd@}ZIQA3_8#l@Ei)&6EkCzGa>GaHa=RNP&an;%v=SYTlw_x`@%-L6>j_+#7 zxt|VqjKnKZ;_bI?!Yda8wAS#m0#XKfa)97AB$uB@*@cIwYpJEK3OLu+zadIW^2{ZjhcdzUDFVeLWHgrFrhH2zo zIrI$dm=FWQd|7_fkS<)H4}t31>2m+p$6Ln4z@gmm*{eg>|KZ^d*7Ifm9R&Oi@y2w- zTsciGB{j_{CV;{XR;}L=38Cq3X{sw-+OYsH2@tuo#Q1RU9yAiz*T3s|L551~s{g%p zBXdPg^&jBOf-D^fv@|d?&Ci^Yf1(lm7)$x#KewbUdV^kg3qtQakUx;f#=@%7;jlg> zj}5ql*jn|oPipA7^(MyuW&w0ju!U)mSb)iSfnWx*p*AWYA6pv`a5wVV{vQVHt#g;# zC6CdW{q=sw=I;>SXE)%t0>hW0Aiv}B4#|F-!E`m`QCh2^)iLOyb%m9B!*!D^emGx_ z%K3!X>-5GSEPH)uT?T}PnWLjTA}%402onQ8JLRo~5d&d#OG&H~+j#vIsOHZ1`xUQ5 z=EY^(x{I*a&Vx-(A5q^53(8upRh+nY(3_XS$ca zec?ietb5%Mvz0fc=LzmuF#IcNBCb=KVOGFg(`-g$Z#Duq@n9=7M1znT&ycCh>6`}g zbiTz&9{(AAXJ3-=SC8o@qOKP;`B8{Zc#~EX?sWW`QL&(c;*6Hd%QxFY(gMFdK6y2{ zr;|=h4gT|xFgW7%Mw^MWPbth^4-fM2wo|FEe?5feX6LPzz32LrtrGt7w4NhtyOqTyv?cqDiPpf$K3ZYmZ9}`&3 zW@V+$b?d)8oCd#pxY@;f-Cs>ie-E}>g+TTE_Vwhj!(mDt8`w$8tgVIlC#0o>#f^fN z86zj=`a_oi?hA>kewese%>`9i490mSD0v*v?kL$}pgMWgr%gs84oXH`$(}p13ZdZt z*)~ChprfPz50DvL#f_&Yg@shEwmX9a!qe}{@iKU*oo6+J2Y_%5& zIqj-bQigo>59qz&z;ouo>R3BzeS50XW!6jUCW!<0Uag0`oT?tLvMUu zJl(`ZMb8$U3Dk>3#3%du(23xl2N)XlcSh#Yo%FA%t8!}FO6zKw`EN;bHB}j$Pn0*R zSD*AC-)!$?owTM#Tz7U2^6?I+DXCunV0t_b2g~}e0>5C{HLcodi&sARv==H7b45CP zwIQK*lbrdh?tSA-Gw-ltIQfOz1H<%8Tzu?TN-9{$Scm|y6I8UrKc#mhqNNIyml$pa zkv}XSjf4?4iR713M^o7>xe7|JyGY{AB`fcnsUa1#>pJ=fRZ$ixB_=8MeJ%(dB=fvM z?*2>wLB+u2an~uxYUHSz)&TRk#&KM?$j$Y;yX#)>p@cy$W%#?W|IT1P9huQ<31?_Y zIkYf(tc$@4pQ~zp1KiZlMIadn_5MVgn$&iWB6wY9XU zDD5b=p<%)(+l&6Q(6L$56aj+oUHX#ym4}JJdx!)suilf$E=g0~4?d|PoyP%S4!+ba zOgvMloXdUeDNCL>E-d)*5gqrvkfbF|qME9PTv|rj3y`Lp^AiL<9v(aKYc9Y~$tcgs zC<`L}AbC-^a%@<0Z}oUSLxlKtZ*9r)R?tY)@Eb6?o;NxI!?AMpj*iCP-admqmCMEF z(QirqZ>DX#O?91{?77PiVfFZ9Y7q&E*`6^RWsPMVl3)>%a?Q?PgS_9k7}FMR2&nAi zTbRh%*%$v(8V~vK%wriTzc`;?HW8=L_xb`*I0%Ifv8WiM%Vzi3-9U^CBnDCFK>NZQ z=W{TcVd2L7;6&%~@BTfw;<>4*B}X8>?qojj%7&4y$Tk4J<~cQ2~p z(5AEOUtJeLzE_lMRc|%dRhUW}Qwh4JGKvLS+l+yx$JLhY=ARuwSC0FN2{AD;&hAXn zlP`A^Yp3IQW&1!$V%-n&-9nNL)ZiGY!kYjWdRF`W0OLr@v)##Wtsa9jjE;C^qRjS- zVs=B|YZI<=W%I|Gj*YbJ43SMAnoC!EcijhhUm~$h&(zb;E6T-G7}8%Lia+t|Um}YC zl_%>HkUm$RT5N;|3^EAm%pK6P3xux_0v^|`U~QR=^~#a3utY^gTX<1xuCCqvy?4LB zaI37&t3Q0FwT;Z`bsjHM^2v}Ka1i%`GJ%&`If1;q#!S}NZ_2D@E2^qzX9g%W$WuaW z92*b5=z>%XKrnjNdO#yUi4MbMH}I*9`hNpap8xherL1^TC64xvoZEI@Cef>nUx4(Ut)O23{pM3RgQH>^rWp6sN z4I-nWP-`-W1*=Y_Z0ydWh^eb9hg-l!$vzH5Xz=&QYCv>MUeV0>@^pk}`fTymyOqv2 zqw52Fs-kG&oh^@>!`zaWMk$<+f7m4)O`-gcrSb3-!jV>|_^Ma6>nHWBx zVNDE;cs;%#zK~ZHmbM$LUm+6>fPsy|(wbbZXQloX1po;|H73<3d@;c3@Tb~?`A3$V#`tx%~7{!n=6|`M~vkf{gF=xd{Nn=p4m6hdLm>3k4 zVwkIcSuvQ;fN8`-)TNEj(eiBdx6NXhZ>iOa9eE`HaFdc0u+l^$44Eex{X%ol|6G`6W+e*Eo39JzWkBiO^**IRY6XbhyM8e)w~N;p~W* zEE*cFDJ6}hBR@A`YNKn`uKY08oYD^v0JsW0VMs*{LGE2uIrz7?Hrklg6F~t|;3!%@ zh}->K_9##TIMgwc#_(taP&phEKhxX=zg(g0p?AAa7k5gw-q-=1z?1+W%qL~iqP)lh zkZZEZ%UFHdrm*gmBwDMQ>w-g88kiYC!dYDTnJOMBZWJ{=C&%q@*46$tNev@A12=nY z?PzLQ!uz&$i5$j67;l@6u-LO*g`C_(yYzEu9+L#0#4qus$@+l~Do>fgm4ZnTN!(&P=o=ST`?|JQO9L49 z*4Pv`I{G#tZ^WCN#>RO%`dK>qgauX52ynKyA3+opGBG@mc&SXCf()6Kiz{YB)d2N? z;v*Ahk?zBR!5~$7eS(3s@*pty-4}x8w6=Dz6H+tP6%{Lrzf=|F?dg~YvXl1r!0Rj} zB!0lp2L7R6#}^S%v9Qq4i-#~Xyb3S%0MzWyJwRszlTPce%D3aD&F~~OBqXGi23BW> zhGe2=6L?n{#n>Z--a*jN&{i+lx^jkizcABv<|ZP%$gl*BER{E_(uY>oQA-eAX&hnv_$ccO zKhxf&NV@@~(|#%(d`<&&SqTLJEhRNg#UEAat%<3f;T@@QeBquG?+=ex2gh$F;!&!2 z?)p9(+N3(6hu$a&VF=&gE?2oz5YXd2Aom_*lZFf-{zIzqxqD}0TifPvvN z?1E;eNm5>>qlk)uvV|HrH%eHA&cr<4muDFJIf<97WU>DbUtbxNRoJfk5(+3Ff`Bw4 zDJ9(@CEcCU-Q7rccT0DdbR#X@-Q8X1@!MygnX_lknOVR5!Mv{Z#C>1)6;#8a;2K9f zoGCt&!NS@rHjGdQ@|U?mM@IIvY{Tg?FQHh+dX~1JO?Q5RZ*PXH<|jr0vRs0#FTl5&8O7sxJb8s|IV1qBE&Z|#)~>~1 zYr(XJ3w=CnY z8jpzLHJOd2->JpW%=pS-maTX`qCSM3Ik|*HKwQ^{(7Bc@!dfQagNusx#aNEPwb*fn z52!@-de5Q2P|NM&dZFe5<@(diQlBpA=;$yl#EWTgTk0c2=nSU&?W23&^9F>MG^Tl( zYnxS7Rml17x@*Uh`z)wm9j?f`d2!NH=!L<7HRIHO;Tr$pata4>|NCaP-d#h)crMD5 z7Jvmi@RbFHzv?`CI(un+)bGPAs+#g7um@Y~?-%3Lfa2&C1k^!d`R^)s4w;}I8{U-8xl19rI`` zUMfAcyS)kIwn{^v+`R0a8#w>-jvM-njKO$=f~rz~B+x^0+f!lT7P$k@eh@8|(FVcF}vi%BCo%Au-y!B)hgU zIT*UpaKTzLUJDmLMqXZ4QC(G%=smh#u0rhQAypVLNjJOma8On-A8d{1uBG!oYG6=s z5G4gQ+WYs4kqMyZ-wjr?avC06N0W<=mJ_3hEh;hrB5MF3fNcxIg(owXSK3t=r|`2w z8+h(8)z$`|+X(r)+%6}m-J}BvB@P6PD_%S^rlpXfQKqQ^LG-s(=>f|4vbp%g)cIR! z8S8#HCk|P9WDPn{ufPn(qXEn!0--bk>Nn)P3>iM0oPrVt!l27fjdc7o7a*Li53gSZ ze8|-*&2f-37G)MF zCyFb|DB9Uq9W4wBipzuPjq4qNL?B)#-~z}DOy;n5 z4^R<&Hz$EUZn>|U^zmlbS@T;2@&VTWS8)mrWJkB)4sbM6?P&ungLhp7$HS_>sai~uG z$xxiyCqt@veLcGVr0s`s*@%bQ5{|byJY;fe%KSgtT;EzzA~Fu(HvEjj1#NEObZ4(BUGr zs}#^O29KvaBw!mhc--`-+?3!roHuA6#1C)%Y1gA?Ib>JW{PxPrH2(%D$^g>{obImY z<1xBdv|owO_lj?l_lt0pMjY+ekAFTMZzyH7JVF5T&$`(+72x$!8qNJ{e5XLp_@deD z{Nn-pyV~H=_Yp;DN~qQSy(gP^&Y8a-5qkmTUcTf!bbL<6NVJ{XXDB9@z45?I@#ke7 zHd-=YR|Bvci$?>>xp8|MZ>%uhWAxsyofcB!6SOpVJWME^K>g?}=txnd=IyEyB>a?L zy^8tKJHLs;<_%jdVsE8f0&}U#`0|v~?Ily)-*=GD{+BBiChrS1DeJy6;Bz;u09g2V z(KFb&+r%;_hZtiho!(&+h12Veq?A&=$^>aDtL@enP=&3YtsVH%bxDLF-Ns^|;J${? zNL~ZEtU<&QJs^Z|yh~4Oazldvy^@k8k9Ykj=Rb zJ^JJTfP<|?=A!MJM)8)$hO!)RIiHTv2Gjioh^FY0LhrxAMe%GTlq}fqjLXm`EjD_T z6&9*gSvZQT6FmpT=J zhC%{QuwGz*Sy!et(#8wgt1ee&CJ$7J6xQtn$CcfTGsx-O;7)zPD{)fWou6oxXCkw0 zYBRL;b&RGn%SIE@+W(Ymf#5Plk2qZV@@Zu6((M#-1Db*hB}*7yMaiRX`$sB4l3WFasog2$pD2rz0@Td9 zUuY+t$hgx@Q?ZWpZ7lZb`$b&K*sqV7vKGQ zAZ>S6fHda!%j_$6JZf9CE)0cbT5?1s>l((%X{xlNeLXQER4Bqv6jz;Lt}7m}77I7D zX_^yyGVd~+iF5PB#6(7t*vs0Qij#w7H8eE%z-;* zkGcjS749#EtEWbgL^|uq9P3oNF*@5+xKYlp=m2vXN@BLStbi&S2v+t7l*ggN=FEye zQY&ru2Ur~X9ch&Nkj32bGwMLQij^>I>a?y&R3t0OUUd+Rn5)K`|7ubdT;<$sq*qs zB?j*5U}?E!EbrGkT3}zb3ag=sE_%RK5xaJ5Q80ac?-R7+Z!w=AWDFcL?wjI=mQ&L{ z4F&&%&Ym9+9Pq6P*lcHW1iAbr-0zNU&8cCo5TJkFUXXvGu*d)UzGeOs3Up3ph0*z; zJ{a4mH5%Xmu+h{Q903^i$XPKL*BdN{uV#yO3K*1*t!_{58LXo>uo@=vjg2+o6zu|Nqf%$M~I|1}p7FwM2?(5`C=p+yn zR8)HUzg(4`wvKSq$me9`%hefx&X$vvQ;?NIPBo<36cdTk$F2R+h3tNN4tNOcxnN}O@k{$IcaH)DH&cY5czJqLT>Op&V!FYE)`S9_0Ui@~s zcy89m0duxnWnK8-)vwbR!YU$Tz2l)ubr!1F7zVv*IR}^a_7wsK4i@?f=ozVi7Xlid zKap37;FO>)Q>tbU1>5(S$JsA80Qr!?@V@){IE)z;wfCM5?Vs#rFiwg2zhp1*lK)Y? z*nzLNx_a*^b#`ewT5dEF!Atlo94soTS6O2-mH*4Y!-v9mxKnZ{oXYJ#m*|UxU@2H^>Wh3*EE|{X)JLYXox^(Z)aIjEy@#9of zTimImllU26G?-4D_r^B>jwX1k%Iq90JRq;nJ}osVEr4XPap9t$l;TskMPt0Tda?Z* zt2PtJ>1uy5+?DA)>j-k4PID>t!TvrlI3H)IPD0EplVs7g7!BDztzLHj1>%KXP~3nc zhYX!H-ZT%OPzpQg@yVJvNNVcqD*!DtM{;rA*r^%N_*v7@Y5INKKRCa8qq64@|3lt{ zgz;f+#=-5$2gqi;bInVsnaefip<2d-fd6ZSFHj1V|6%#athnvO&P+4b-;wuL16x2B z+tj$6tF86=k=-rwp)t2Q>N_$hUJa$Ej9P>H+2ABKm#bVlK?V0fj8UH#bLhVBm&_p=z{ zqztT7*bv4kg~imP59ph3LPViGYX2aF+>2ySK=CT74hU=QBxsG^XqlRrK=G=HwRl>l zFQjeh<-lE9yYHQ0LENDE%&8<&RE}s@o+^BE3!`$A-p_Dm5fSP7ziV*X@QBI9+B;yZ zBiD(uG$8&&BUD-f?bW@>Xg@X0c@%Ud)f_s6%g<$QXNZ78ylMBw?ZBtk)YMqL%?eQr z_|;%wptpJ4>^*NJlr=MnzhwRejs(goiK=SJfQR>Z(vCnO62!GIT~zxxRa8;%QOwAs z=xj771+I<-9evWY{3LnIeV4{ z0^k?^CYC7S9OT_FW?A`{gJlUZA5Teh>S2E_ z!#lazMvQ+R^uxV030U=s@j(J%W$+Gw-8o&V9C#0PMs`Nl&!$H|H$5=K@)Z#R?fu#6 z6B*6VEy3RYWo!t5eYm(?Kq!%(-qz|bJ1Z$uCBu2i zW_{$U2nwotJg#Ba>$;WkdGzUm%FL{AIu0&%VOQnSz&$J(0)P|vYJW>7F(M-`9uN?b zyEJrvv(8)NmgbECF*Bo6Gs{ZzLt=*>^5M+e7Xf*BJ>ch>VEMR1@L_&ps?c7c$kb`} zboP2Z;=j271AmW~{x{Qn595C_%@39VQm~O{-@ilys-1)SBT8~AXoxob-JfM@>}JF7 zX{9W_yah?0e<5SroZigbErvR#A--bXI!kn=Zk zI>5N)uhv;f*JkP!N5m@+H)3j%pRXa%W3^hdMoc?x(HG>fkl8yMl>NxwatB@iD*{e4 z@_i(55Gg&9$0lG93o@AQy}Z3YZ+p8V6HwvP&-S9hJ#U}lHTd<{d!D^ZY%?@YnsL^7 zRcNnjuE7gsUg6D#C?80sw<;D|Lid-B^a)xOidB`Ljy6Y!zk7z8ZIU?PK`&M>)asSrCQt{c2elC|w=x9U>DFwwyBIi%9VhWBL~> zU!2sT#VS`?jdOvRAmb6ded2>7v*bsiYc0~kHk&j`I_B4M*3Zmk%O2wS6Ju!jKX1hf z^uK;Z-?8-R_))U#fPw^p%_`p5LklQmh1cYCy@(n=M<4~u*-kKcVk4>6h<-2vJq6ir zX#fXjcR*MS8WJY?E+YJuS8{T4#h(nw51j%s%FnMLUqpj}5FbMLl@WU35dQ zjBLBP-PhgAWYStRnu|b#hcWG{qhAW;K&IXAuFQT?1D10gMVIjX>N9D&w;u#+g$Tw=S@bN#FHW7v$l;1 zJ7o8ujx{y5Dw@y20}7I(uXi>&r}x^%o&T0+@U}n#05KK5rL}?}o>g=1{Jcr;$C~G; z#>F*K_jt&*g#kxS$W=@2d!68fs06=fSf2} z+7BWKbVX(L4qluhME`nj2&B7PpwVIr0oL0mY>SLm3jz-xWK~le8W>)lH@2pfVJzvO zxR=ECU$+(-KNs?f=^+r@_v^_5{R6~Iqq-m=1*r7D5i=@Te~J+XjaTBKOo!u+yM(zp zQ)zXPT3^QN>x%*7TZ~fIL?@60=;-ghDC8~$@8vhtgi6)T-V$pD*qkeO`~djhZ7gshTB4J3rjj9p&$oG((c*FWh!l@cm-8}2 zv5~?lDBbd@z&~;7U0lO0@v|fY{{c)Z0s=x3HrAuxO+%1hTtV#O%CrfPZSgP6)YM1* z->lZ%8PLbJ04k+E3?;P4t)*#$Bum(EF-9ho7q6ZD>)+UFCSRp_PPLb+*w)wA3kXmE zPtPl>Y)BYlqUm$OEb{aJ}TtU$FcYjzTczpu+LU-~KkDe=Mlfq==3aK_BeC9E%?TzMW$9QoEQQy-%K=%MRQmp<*S=ans|DyY2! zb=<&7oYFVsi03DYOn3Wf;PJ4%o_*J<{~li8;S*_brvAojh}w4O?vCDa?m~1@+5xu> z=?kZ)`Q=GDx3&~lGbGbf>LChd={eEC{Sn{orNLf*|MtF?5-NJ50z@u!#Dip^{0rfS!embA1a z76K)+`4ypf2e1>zgd(Fq>kg_)85T=01UzvevDljsil+J%j)mKY;Bm}7?oszpTk3cc zCrs=nMhnTxR>h>Dr}y=D#Pbqbv8J|1{()+7&B?RH8~^dm?zWC!y9K6JL#*O#eH zw?%xv__DEmm?TwXReJrHdW%>7j6Z)d$%l3V!GS*hvt#603y17Q6722BnR-#nmZ78ZcH!rzPK z);0Ih@8c+MXDoQ!Ud|{Xv#M$(8r%d3us@AHPb#e4axWsZZ&wtNvtt;fJ=Aa1O;CPR zL)Pr4rHYCsD%E1F=aWvj0kwZP!KBSht1>M-Hz{KO%8xu zvfTRI0r{b?&92{BK&oNRS5L2@*F<6X z3gG8Ohuf3yO(nHu5L_CPVP!WTB|nB1osySDR7pBacQOF24UI~R0$5|kDQG7 zll8gA_3LOf*t~U=o*q}8zK5mE9|I;-_ic1t$-kh1AH905HJOhLxVFED2HcpD#%$p@ zmp9~miuZ>=qJ&eK>>tZK!(n*AG$iBZfP+nEhMA5II$1qUz{9^Do!txjDiFSr+^Y8pfAYYC3EX{}O@b8aZg@-U- zgdve}QJVC&#Mtc;S#9})J!-VNsN>vjNOu??mZs(~`eCV$<_Qq`3p3h0;Jo(tO$w@e zs2)v-G-zpRj>tbdkb`})1+8n(1 zP-@9V+FjZJ?w~y+`ew|oGrt%Nk3T zesR{ZQ574MkPyTY)7#o)YC_>mZv=jHF5UlGOw%)i?UBi<2`cPY#&UGQ!D9G4o~{?m zy05cMIuo_sm~nyECKx&;ap*=vt30os*-qNbcWmqo7Hlz)Mwkl40epDp=}t}9=>x>= z@+>4w3Y)%taHu;_m^d#-zu9dg>`o1Lapj-~$MVw_*fp-sb{EfwZvp4sdLy6ukJIcMqW!K19k^Uo~ z$5?XTJ`N4N_BzY<|H0j=`T>X1)(6>T4=V{`wf{I}T93cGZzM8R_V(fG`XgJ-ng$NR z3PgRX;@4atZv8YO)(NJ7;Aw%c!Ko5)2NS@ zn_qD8ZeSoo5@7(I@@Z7I`>3GsT~edl-MBz*lC!aiX`A&2pU%M1gOqbq?^p-&&6Ngi z`LaiMIGwYzU9nPB*P2y&vyuKa)^*^|3%8n|k_+)M*@<5oG({ zKyR*`i2Woiy7g4AQDpU19rt%I?vOqfgq7p`LEQG(%=74YLhIbFuDBkEt@i}0Nsm~q z-ax=%mS8zsZQC{OlD!Hph>Ab6rjF?)s+`qlw3y?vvLOai%kH~~5LjOUs+9`Eu3+c+ z7bWNchEh$6;c4EGiHjp1%>vO=4DA5~)f^XiZeQ#D0A?_bG^>sFm zjEkGA)xWA@S|~C#fe=#BlTt;mOl|D0!2mCZ#W%sYkl8ASy>t5oy6SuzY%cD#Hu7)R zNGftuTiC5sZ>H9kDopHgyZIKv$o>RQJp0m;4}Z?#94vu9 zj8KHz!;kHpvAz)vxSUQP2S36F%o*xp%tKZy?8r9(0SrnRiV8E3_Q2<*$*I_dg%LlI z0Be3OD9BUU+U4GwQD%NpdJ<+#fOTzppwD93vO+`dR z7TtV#30dt?jj}n-GL;S^a^Oanq1$gVPr-3=eX)N$*{{g2ArUN)HMTeR`xOtWoZA!s z#oG|@HL|cIicbG@xO?+uFT0<=g0!_)d_vn+QA;$Eo;11)-fa+|LcQK6sSZzT5zSVD zKq~daSDhRjxvU;_>?M%GH0gBCi8v@36wvjvV3oANB9=wA0+!6+kynUqbhNIjWt^Cq zP$V3zo;yrkgnn5)1;?=VL~VMw;)>?dU)vetNTMB4Jvw7f=+H)WkDN4=baUFb$ zQ%XG?>cNHVxxU!8#dzOcxp?G(WIu2Q!?8`Hk&qe<7>E-FD6Ry{#g>Ot zwrx1X@88vI&lZir+doYfe#yeV{8$3e z&@w*2h%6G=Aq3@KSD(w+ z^!VfPlYlTn6La}|t+RrUf_ViluufJ;qH|XFF$}~QNw2cHj%~~Uemq7)f#0$&8+ON0 z#Q1Md_vfaH4QDPhRL1De{xe5Z5ETu+XP-w-GXqLwwR}^Qz}FP{V|<-iU|uUGIzUW9 z0)8#P@PR804i;%qst~y3#Yg{uk)yu6LVy225jo|Ya#BhPmKZa8-Tvs%;cf!~F#oe- z$AgDe!NLkoE^1A$%l#eHSspF{s69}bRf=(%8gf4FogvMssm;ym)vv~H2PXYgB`+@C zjr=nQ(~~AI>h-^uM82Lass|^1N$7w^QlZ?f1=SZC?Mh`xCNJsWzySQ03>gK%+;|Xb z`i)pV(YNKi2yg?YHrOJ{WL`O#_4aHJC;UpcfLKQrR@3!2W z);-zZ$A(nYGFH(u@Iv2-ePqjgh`P8*qEggT7d4XCG2~z%M)M8qCL^|7a)*bMP!HFa zXf;JqSp(%nnIB{PPfLU#>S_{_9u!td}OZfpU7n52jDMbKRpMZ&CTH} zZ8osaza8+Qw1K@-*ho7tkit3`8@?J3gX3=fnJ9(FO>Xi8^JtDyoRAT-y5fqILHww-DZCQL*XAT4tC{j1-uFcl+kX7LQ<99GFPJ z_WKS!4skK!1B5)fKg3Y#{4HWgutb3pv}o{V#}lWa6jm{>5D*9=g07=ua`J1jsmUxZ za?0@b5@%Iy9jsUKeR@3UCrZkVu_cL1mm)cNK|2Xk6Ehg;0z&AE5U-rB_1DvuoD^`1kg2EzkTn@Lx zKOLxiQ?Gxm$>h_h7n@23=o*^;*1jU@LUx&L>*^g{Z^6dG>$^dKh-33q(G@9Q28JnTJuk(1r9WE!p+pDGZ=34P`Ye&stY zkB)Jl=HzD9ocM{sfjvEE6@9^~Kx;)K^tfDoMg@IAwz(BgH>uF0B0YI@bh&;#6_QtX zc^ExwR-yv~d+3N1(K*pXDnhEFq2}^3idMEY*GQqzUU0PB^}2${f_;nTe|3mNpx^~A z|J0L%Xc-VqH5=aFvFPV=4$%Z)ilI~mLxC7 zI{%w71yb;rl_{JZWahlbL9%Pc#J=$%r>FDZ-er;xD}Rh%a8$bVNON-x^$h-zyy%OB ziKDBsyulyL{Y@u=y7xI50d4iA3!iz#~O)+nzN>o&(`)ct*=P7rEWb^_91{I6jxF z?dq(`+Er+9*l7Q-r5~P~%ol#@d}{Sdib4nDKV$)E4-r6)y|AEK$5#&w)|`TZ#FSet zT-m}#ZrC1XqCO8;naFiPej0s&(za&16I5=LdY9k3*lAF$<#zjg527rL%t(A(zeW;M ztQlF3)#p#_cHqkr|1AF$q3Qng$b|^_Rk19zDbNs^Gm3fw*R~gIS;>uF5~YQZIX*dRX#6xZXFUlPo?M-slveDjz|PP%Q&cLjsKjWHR;3}wFC)i9clg)rMK>%bFOQUtt+b*bQ3#~#6iu2k zbbxo$@(`H&aC1ySMtLyTg~u@9e74OW^MP%GEbQrVsx7*spset>)741tR7i@d_A-f) zqhkV3fvvNCdSq;}mb7x0UQuCzvRX2eLDGPjq{Z@R$>MkwJw`|_FRA0-iLr?(i%>z< zG|i~UR8@6ZRk<#C1x5PZ%msN3Ra0ac8Q30!>PnaCuSxqKm~>&D;yq$f;J*oriLi0Z zf~R&n!2>|?sDLqmirVHZ!XF)5L(fu4&+^UZTPOh0GKM?nyWhnJ_n4iF5maDyMsRx&qZ%*V zZ+Q^V3typ?T>fg$XLEfQz6$sFB?G##vq>7n|Ldn=ppc-axe6V2E_rp?E<(c)Qx_ zMzX-Dp|;XIuu3af%yQhR^-CfY5pWU>O>~IA{p-Uy-S23axzDDb$edr5VirrsdaU`W zYPMSU@oX_R*2Kz3l$8OW4ZSKiYmq|FALQls;pAN+ANoI|nE$-d0uG%&)AQ0Dy3}}p zGiVY*FP{0R^Vi;{NYS25 zx6qL@LaWI^wR^RuL%vo_H_bDk|`C) zW2ckW$B)(+IMQScVJRpilrOno@4I8oxPJY?!}8?nI+VerBB&zb#57=B1K5HsPp2f5 zWC6@NG2qr7@`$2Rf8VzUe3^nW7> zjtJwJUR-YLwYlOO$m#vl4j|a7sU)83yPmI88?^r_F13HY{?k9 zm-LfVFt;QXKPJ-u)gc81rB=n-?(Zkc#WXYLljwLNIu(GJgt_BQ=BG!MjiOXf&0qD? zn)Pp^DXS`|eDlfb{2`xjxTSrsOBzz^aKq+~XAigFqMkIi8`rFwirk0|LV^IR@-lcJ zr^am2Xm4-9@|9pN=DRWrJx11VyGF8hwG=9Bevs7-f}bDQ0_zaqyauW--a>%G_6ZuK zpePL2OGc%?{T$dMY`dS_ctw`FBF`}G_p}~#k*$nsJptb2Vw>Zri>0BwLg`#HoLcZ{ zr%q+7+go_#u-RJuu-(pIV|&UBl#~EN;!!f2O^SIKoQrb7n}UweVlowD5Ii0N9rZ()kXD_AtGGJf zlFR+=J)ED=;e0dpyDk;W3?dXy*qiibhl+Y;hB7jBeG{SWqTvVQ%Quk7gc!W>6||sY z4et+Zo}3!$cm(P^DZmpyEk=qM6>}(AN3S5F+Z!;+Y(SynoyNhio9XCug5jL*$b(1} z)c}G;h7MDuv|3@>r_|oN{7B83|Z?6n#)5 zyDi)_H?TUGy-H$iq6db^^JxLf^V9P>N9T8L2J-?$)@z3p;2XHN!PWBd3!GqbgL?%y z>wbP4UQ51iWwNEsn?2U{3bwD8Y#Z~#808mMv*)EGP)}YT`iC2R!WPr~b~b8u9G?;9 z_v>^>f9sMYC1PVf9I8cKX_O#8yMM`nf+jb;;qmqw&C*QE;3}L!s{C+z;=R$j9C&x0 z_skHdM#uSPw>O%|rCH~XVLO->t(#V?OiV*@3!}}Al~RIxJi`{VO(77g`~8WH1bk?( zDx$fU9%C6_!exgzSz~bFKzS`M;bD$`)Yk3{yOfsa<8nJ*?o!f%cy+BEayx8v_#Iuv zgCZXJ)b40!?QniBc!=S%HEXKZ-v$`N<8ERwGSIRbx`I2&3rA;?_D%Dn%lm^6eETyU zUSPpjm=N*s00Ht|#yWAzJKo=AN3zast{Z3aqAS0A;TpHYy$sy&jy2VJ1%-g40R%!| zR8U@21gwu2Zt$aGV|R~aBBsvs@Ze6VZP4{|TQ#|Wg}|JdgO1JqLqsbhyE_urBTnZf0)Ce%notjmO5v zf1%C>g3;Fjf>R30I$&;eKBZP>$Z2d`n69xzMJ=8y91Q>_?)zJ91In>Vo^6^1tG&se zjUfw)_*q%-!96;qixp@y>j15Gadlug%d)1MlKAkjor!BObbjBQ^tevi zFXuB$WFo`Fw9>QVV{CyxZE! zg5|PjSy|!xX!GSE5uNI zsPUsSRQ*z-uOs;Fvey3PXSE^;4CMaGj=NF!{ZHO@9F$2j$K)Dug(t5i441@7 zQvehBz}1L{Mfi06s993e_TL-es@mXo^DiTSm_*Pf+CsbNW!shzl4G#>cQ+0+rkCo3 z^Q$;#+qGg)KX#sV6N&o9J3l<+_z4gGS#&=ot{by7o7H~sM|Y#4J2#Y=t8nM3``vtU zYV3+biDb{Tj5w<0dAmJ&bw67gaxrZ`=%nvq+4ONJ_0)1 zK9%E&%fqTF?pub9mW65HseVqQ@(!2d;r_`GB0Tzs)9t)-n@Mb6v6U&HfZV&${)OV0 z?FTPG1P~4mp8DFRFg$s3ToXL(V+PJ0d`3C)^4+ZQ-DLdM=3k6CiaR+2BT~3T;}i*x zGR(`Lw!Srn^uRw33Lo<@%jZwZe1d}g!@sK|Q}12aZ|B6_s#`@8KU%LGfNe?n*sSFg6&Z=*5h;DSHp$+# zT3SB}j94>TTt@SI0CJR?TJq`*YmgCe=AC=t+HS<=vymO#lUlZVqQu78Ezw~7wQOcf ziez20BXoCK;JybF|291m7dLW#Me<1ofk+hcc3;ks^PeHrcS%jXuSF1E zCAEXgs3m*66yUIfw^Xm7CI$o9y$l8>Qx}^0mC8Hii{0aSz5Y#gCEFZwAoR)2>PG17 zUR}R>&ZOvzVff)T2aIaD!Z1LRD+H8SzH82?0au@E=J;CY<`j7-V~RpxlBK=3TJfiy z9l6Ojvw$~U#+4S*{Uof~X71Uh9fLu&;&i^)M!@~B9Y{N!8JIRmW(%s{e_TriNr+c- z>S1^o<)*36Q}fGhZL5iTL7_G@B#7Age08Kz6&Dg28@W3+Wxw}`<(7SL*n^{-oRw!| z_rk16jSK<9Tr<_`?%`b9Bj_t#z~0j8FXa?Sp(Ynnn5tD%AB9Ten6pr=_Eidjf#?BB zC3W$zMA+XFKO?wM-0}L5YM+u&@g*@m=ih(@ve;s$cX$dVEPeD*FUrRV8K`^#_{0ZT zqA2DufkwypjbXXYKAM=9cC{0p22kyz;y+X5rly#Fy_X5UKoHfw$ zW$UVqo=H|e_+fsAaaU1E!oZU({faoE(Rz0xVS)P|#(fF&?v3pBjSdfMAmgwsohwUH zMsNZ;Yff+bgOwHzG4Z6lZ>uBOvtwuu`^k&mVFv|BmaGsh|uLyuk zro1eQ(N`ATM~L`TM?ul^DU+Vl3kKVdNz|icHto#JyKZ4EH7PYbG*`a!(d{V}r4Su+ z11~n^kU)`6Ce2~~m@M!7YMVFgs-N}0u3M5(kRUO?A5nTZz7#IAEfbA&l7>P5xIA(oHva1fDIvCe42m7Vj8`dAMG0 zV$UomGk~lSd{6|&X*jGE-+-5wJrGZuRoZQ@Ghz1ehqXBZ%OCy8a~wzfwG3cL6(HEV zdQF~;{lQTwA3(qW1Aplq+0wnsQgJJKg-Aluqsg_H(e@1Cv~aQp-oVzBF<3FZ2obk` z08}H=*t6|0Tpe2S)Pv0gk$aJ!E2ONgv!d zi6$oCdi}H}^ETd+wb2u2YIendG@V*`9jo{OD50CPZ3R+}0h9j~p^9VU&3g6j{|u;gG&;canqcF{Xm=udyh6cdvf^TSn>R18_Di5ngJL zYydC}1SKQ#ajdUgX_S=JP>t19l!YZfMCk-JKybPxIYI08_AaGwbfuw*y-bx|Nlk8} z-tM^kKwRO|lc~9}{~lRIyI`~1)!WVMm}>%4LSTw9bkj*lK>-lH&m@AK%#12j4%Q+R4hRu6fuCu~T=)N$ z9X*8qc#Q>VU$a7ukwc4-1F(q%AqpDDW7homuC?qN0gddxE{WkoJhpP>g7=*@fH^VGMi&e|4OY5E^BrVKD3? zT3_)%4!(cD3EjcOD3BIm7T#;m9V{kOc0bjq_(NnZ;db9+S+Uk^Yoc z-pq^FfItSAuNwe%q9RZE&QgCxiCH(typAl-F|UF!V>Q@6U}92axU&g3Rr{WX6gt%ak#4a6&Yyb2_+KDu6`B;%qCLzLd31R-%!yg!&_TAUh(ig0xh2Czj*X4gPAc4b zsRdr&C}NCGL!EzlrPPCQ!KkKJ4^)=&iVI9I_b$^ZhSFM6;!!bCNg9@q&T9bY)89)j zodwz<4_x*NDY=EL(p)x+$w=#Le+aL&i1A`X&gx%y%M8m`em%8vEcQp+EiAMInCn{g ze(d1Gmf7W+JjH2OS~+#_Kf_njzlY^+1Z}F{4kPofSN_G@H_oOD|M{d}zHAqrVb_5D zBQzZk(C!F03ScY?2(Qc1vhwYSk4qYOO4LaU&3pjq5f)xQjyhVZtNX4r<5E4vg$eQ( zIOcVF>44{Z6rHv9dpZxOPq!ff^`k5b3R@ubm!&ouSYZLbGBmrX>3MakIFJ?9w>g)& z;437`&w8t}q_~@)h$BhOh{)HKeF`rGrgSp$gfs4x@v*TWD2WF(W^kZDw2gurZ9PIr z(Z9)6RzYK;9Z|aJ>a~7qd-$92&S68r2Z^U3-Z zoLF(~nfc#Em7i3!WhR6DOrp6z9~1y~0|JUeLgdH*uVm={uE+Ew?~9E(E(gF1bdSeq zG@2I50YJuI!HBAWzEx%+sj-xbF_Lr-1b7pe<=P8RzwchX`(M<(g;&&3+cr9i3aErq z(gFg~(hW*WH%JRemvn>D-QC^YLrKTb-AH$LpN-G+zUw>RTIUZqvu3%R(fQ4u{oDKA zcU;$Xr!*yVx$08c(b7pUu;ka^*eKZ{J(!FK({^}^JD}UYq~|EV(L40T2H z>)l*lA8(=}x;#nC}!)6yjG$3W)U=%*ObA_&QT%5m}r} zPNbS4t;=3DS5w82t!n7g<>55eqS4lL5rJXpw8>lr9=rsSuU=YaH-0FSBV7F3tr`7> zFHtC(1Qbua4aS)zJ_^@fyput7suxJh&v)G0w*qmQsDGvXwXckaV=m=nMuvLr5@;{S zUg`hfEJ#Vwo#FG82oIko2q%n$oQU4(B z4{5DvZ&6UHCsWHve!Z{$JMQUS82SWIme@YU8mr5wsIZtWh_;O4D|PJLP1|*(Pcul2 z9S@{1JFK^kKXQRD4|@t8g79e_;YdkIyFv)^Q$H`94gTDXkkY)3&+mTIXw~sFp-sK{ zO#%<(aQL|=*=xjGT(=ceQ^g!HEuTNJ)Y*rL;$*M-2f;Rx`rmD0v1x!yfbkFCxW=R+ zS4Hx!mLKjFG71tjIaDtfx|HNYyO`j!#=IL7qauGQRD+bjoUdm{ z`SD9&LY+6tUh`Inn^8RXn3^ShnbPF*(6=t>AzVYZJ+AQA`U_17>SWA1);*RRpK!o$ zf|s5az-s{jqG!C)b15?6nVF#`rg(ya4C`{n77q)6nNWgF8oIV7CJJ-!O8$(h&BS2# zx@5tnJ;Xgy1_r}fA}vV|U)nn{m2TFMxF!af_2l_E0J;X+4SPOqFV3Q{GI=c5%`0uLv$ryYiBaTTMY2) z5c31AHiDJYb8A0*Wa1mz0{~Orw_aOXpj}7lGlHl}T98LCS%h8CPK*@UPor;IP+A8 zMP6xASC{$k1Q|Uunwa&_umKUGZ=-tpU&7Y$7~D<6m&41D&!R%rwM|S3UFBWUQfS=h z2%ayzaFGDOK5hbv&TQ6KloE%;m5-!{N=LU~&$?0^(+@RX2=tpedQVw(+d1F0Yy_s@ z6El4~mbpYw%?HN~|K}1&#S>hl%o%wX%akMDL0gRF_MMr%g!BSepq|>fDB!svE&&7F zta*^~KRmy3-5iVhfqPD^QExT4RI%vM8m2r$-%SkYhoFpu1d{h%lD5EWhgh7gd(hGb z|HgA(0ljj+_Gr4IQK0Y7iis#krjjiEhZXRWD!L;aeftRegeExcxghO@Gua+tG zFRJKT4gGAHap%}NMKqO%)(Lxf14P?dkjOGEs(zp=kV@DnoJzl+VYqBf6-0^$?%acj z_!IZPL)}p0?oL1|h>uNt`G|DpD=$+njIH%ljL`Te*@qmR%xWbIDs-(UZdaHVTQlkq z1fXoxH_%cuhU=I1>)xG*0G#yC+(EJg;I~B7kiJh4To9om09Jd^T%9%mkuf|%S3y(> zr~qUR#XtZ?O!O<9`{ooOHKv%zCH9=BdQ!T#vDwTqYEeK?6uTY=Ar66poE-n#zhjjE z!4-V-OHV|=iI{`-2;gO0e|-nNTfpHRi;W9lpAI+9ZkM;}goDL5)rA7YMltUL{u2xE zN7Kdv-bP7r-nRxv*iDG$^Tt8oxT;@XOawkJxZ<(1EbV#=Dk?E0iv>5X_Up@MK*aN< zn-AVqGdt@6H!Uq~5i9=>(v_(#F=Z7Eje*XQ#Ebl_I>WsH_`8X_9AYy8e_3oqM8BTz){LsR@Ja(I<84vOjPb^}` z=E`G>w?E$@2=ga78OTGMHs9Cm^LE_&7pgXWmgC2lcts;@H0D!jaB#}D_`C4(e3Byr zdB$Ej^76y?E_J8c>6NWnDbJ*Vrn9MpjpATFOCI|zl2t&PP1(7#eGP=HsK1ox`Z9P` z3)Bnbq|3&ZtyjUDcU-C;eVvo*qsdI-d*6;=Hq=+6_=ZkhJ?T&oC)>w>G8kD+XD+bM z#H6`NqY2*I|HwqtgR6wOkSOefF!G+cnGyA^IgpclTSXqE(PS8m2XfJ%lJWF2smIYl z(2cg?^N0@r0*pKkK?%Oi@A1)_YGr@-*F6W%C72^D_xtXqDJf`9ULm}REUL{sO>je9 zq%n2&oygK>Wu^W?lhb&1PyBH<0_))LL<%%?kd)MDISz8lp;cCumUp(lxQ?+KJeVuB z2Cct=(#<=bTvkZK!)ZU=Cv?Po_21P0ar}_!8R=&#^^PXEg{Mqdk~GSGT%?Ij9@ri% zn`r9-9G;SbQcRRggdAhK&Ik*#*JM)8|XjUirhvuul9_-C8 z+(D-AL~y$rz%({fW0J!VAvpt@jkY_#fJ7+)N4=(bNErVO>sPXLc|E{4UPx*Fu1G=X za;v=(;r;rMfsvm0tshvFbc^;n#kemnF5VvJ=jCMVtqu3zRM0^{XThM9YLMoYf^$(P z3y6viyV?rS)Nm#8`@jxpGoY%DNGU*lH`4nEh+sVoKiIxJ+%krA_YI6oKXiN~=Q!`m zGPVQZ=*-&5%_*PjxC&--SAD4fyqKMhI?Q&0FGJ9x z9R*^4Mz1xNnyj9*^Ya=jA=LJ+g_F9dK=5djSbXUXe#-UVtijDM+&dbgqA=*yC0w*^ zDKjJE$O5K_bk`6q z-snqlUK)?(=A+qy`?G>&j`7UryVKo0T0{@d(4<^BQAb}8XUL(!N3#LsqD5fF>a2v( zbUaogts!hyThN36{uQKTM_Vg78O2uOxkRW%|8!YKpf&dkZt{WqfE^YgWPItG&}tTe zoep3+=Ee|If2DA^jkhcaDG3L-wS*?xW(^$CJzGXugI)YBate`y8J=W^q|pCFKUl4} z=&pkwtQB^U9buTqEeDk-KLGd9>|NWjJK%AA0;yJ92Xf6qPp?QmQqnryZ}S5YeULV5 zOb_qS1DFh{3s3~8kxcPlqiTZpG?}$S7i)HL#tPaE*wf_(ikBEN&pup*3< zcXJh8B^FaM8$8~eC=?=fN%u6UG0EnneWalmv=+YmM_9n%42wyKaRUV^X9qorBrPM8 z_qTn2dpj!7!eu|Hi+z{r0i<^Ol5mmmgh^@)4LAC8F=OP%qY^b+{M{a)=@??4O*E1N z?Uy60-J!!=q}c}#frvr=980vw~># ztgi@H)2uRzBL+xUtpArUIbh63Bl=8z`M@u4~(!48sGULGibWow=hVu~prpYmt}#g9UJ# zPIdwqH4rjPHPB#A%C@}WE%k=`Jc!A{$oF0Vp4DKWr{C?QCCBX`ndVdwhBcs4tBn@W zpB+8APBUNnW1a!4>zXh$C2l%Ivw?=1dZBEKSC|#hR3bV)ld4b>>dS<}9i4mn#yx8b)fOt+9CW6Q9thHlTT&rBIwH`jM z%8n-~D7bd}1J~A6W4^5E%#BZ0^t1HTDq@Dc;#j63S{EL>%^TJ7;>BvWjm{|DV>eL- zmIRmk-Xc~)9-RkU8>{lNWu~0^ddBMdeq9D$ z=X=}Lu@2D}{1OV!c{#m4-=@tuoF4!26N8kR?kyADaXZkhVWB^!H9oO23oL`<9c%JR7^l*s<0AwVw2u>>fU{9jB4#|oFHz8csL4t0#e=W0zqFZSCuocBe6 zOO7>|v_9Mz5K_{7Xq5YrDv2Aqv6&}!!0S>pI|W+4ev3~4B^?EOBI5H$M{$^x?irFz z)u@jhXefJ|;C9%Mh3jZ*M+zZgBIY2a(E%(spuE+}-S-|feW<=ZqcBjOG$X!$-TC+x z;moA^-xWjSYJ(Zyl^DV&kjCHi`u#*o@CEzD=jaMq43v9LB84Gt&x+OGF_(6^YZUV} zDQOC~^(h}@m~W2Ikk4_bI@uf)ew9g!R(j^y(RhU}-L`5ZksZ9XF*cmOzh^}0pRA70 zbM5`~&!l_zK1sAeE(HwMGbSnQUn6O1l=%!)hGjOlCWc!D9)ypjAcshSIBX7kfb!E9 zbcXnhPfxPNr-TS}HH2vC|0v8Zyu7M^mpp=uUnb%O(kUa_D^bM_@OpGF=1FXl@%Ig_UE95!9T@tkZA&9ar?!RFYv9`)>iZL z^Zmto!(~T660`K(?tXEwGpYDH#$C{LZc2L{Aamagm+`}&(ldSz}$SY++fN!no(Q zQ9Sp|y)`}D0mj0{K2=}O0cyLTy+X7MEJAMN&Z8|{Ze-7t+5v+d@=34#J*Wyj_M<;U zd5zC%x&$y^n0jO%!dC%tVe)`i=ZiJuvxLHrwqC(A#nd7snJ+*PqCbThdz2ltb1hk{ zmXeW4^GF65_<_qfiHy-Q9vdsmKc1k9GbHbKiRN&6i=W*Q$LhkBf(Z|aj6?@o%VjlV z#l=bL93)iR=ZLba?^QyLYY4S(eged$am=9bQ$rRI;HcUXw%SQXMWIJUVIY2nKzKAc zfciiiK%)XC$A}gSC?|oPG!S>+oEce3=tlumx=;OIVjCKg3O;w?#s_kmq@;pGHLCw# zjFFU!v44z_%mpXV`U3FPKtuL!0MQs>7;)2ETf-RI#CZ0!8`iUBU<^n_OZ1Y@(z5dz zbdGDWBf9)P(w4~!iTX2{cV_0H3Bf<}(O+8yEih%q6!| zYgOiU5+Iu9>*Oa|U=R+l7sYz~AKr%SDc?4GaR!LNp1fymzU*Um)r{ zy{XxyTi)PO$1@@P`m-d?8@I>aBB~R#>J<*Gw)W|%rMao42+iBmZR&xG%mA_ofIt9R zuH#J&u)chJeq-tsUqwam7AhQ;di!G;-qUcOTuxIpS9*!4)jNZRMqjCR!}V|b`~#P~ z9d^ds8jpfC0IvwZFkA38T(8!pXmThhLlds|&^li`U7kiAN>i0HFf=@@udi*auf4*4 z!I%@+#?Mvly5S!HP?Fs}Fak~&?0{2DJWOmn%<1l^hbH_T7PANVfG{~VS;xKeHs~}k zqak+{5(-tWR^u$p0$z*`w;}+l2ZWZjlF0i z_ZKM*t(9y57V1&YYHLy~M*3{I1fW7={j+BFm{{k!+@Z9_=T!|+qbkT;+&a+4Y%7jLgtq!D^7q=Wa?`RKOwMD&;Jgg6 zfUp;#`pnhoUX7INcf_8!tg&NJA)5-X_no$Wztfr~G6Fd5@EXsDV|W8;zG>L5{vn`R z4-5;7ehag6xy1wu(r^%FjE-M4dU~FpeX)e$>l0!mxg4%on5a2O7y~q9dUUJhL#+l0 z_Rp&oK`88S)ro7>)a#R(nVD4OC44~(JdQEO4hjkluB*X1Dpo>XCpm>X*siXgcBN%c zVsN$M*?w~XzQ$zjT|@kONO6i1ig?%~;Bn(F3wxn&bxk+t-u$$TF~(~Zo7d3*OQ}nW z$MqLD8Ga(h2Fyc0hOTit)dKbFzEY2cgFUsM^tSC_l;+3s#^15)L%YuSXbgJ zrI>4gRRC^MqwB*}?^!)izPE9uTNOqk{Pd&seslG2uH>kWp>o@@cvJ(ilvX;1i zYs<{di%-Bt+FY3~3IO$>)M%@_-`y}0LF8qWjLbl&=^`R^quCYSmwTt6*Qe0~Nx$32 zo8E-%tX6+E`@YpKpGgmKXws7)A(q`|kW(h3X={Tv4 zL%#bVF?!qAP2DP(HrKVmiCLFgA28S(I+?Cl{#>f`B1}@uR%1tjS6^ z0Mnm;lyC1X=d^>@ukS~IWg<=%ke;8JQ?$RWHGXMz8uT5YD4KrBKuWDAaT6mVWhRqbYKXLqjN{qg z{3+PvKNcOI<-Ob=siD8cKGFc4`>0NAAQ9b%=SN{#%6 z8$6bdM*w7LF4Lg~xM+^n2fy2=L;#iPWA-`7$u09KDrN_WZLH5zLATCRcmx2^W0+n| z-xy5PCK2E5mrOm~riZ>W4K_Vxw-2GsE{(qg+&@>kP*PHsc>+HH2wFU*SA8dPDv}%q zY4M-DZ!TS(o;4SFznA=qwXPb=iX4du)H7{7McnGb|pWb5)v964aBDY@buAre2IVz zgV6yFH*lk;zo{aWb%Hi5&)h{{Oh!2X=JYut-La+R@5G4cQ3ImyGQb#+lKuz}e+z&g zz6!t>l~0D5Wej8i0R|P~p1-|aKEvjZvyHyA##BP~``J76 zx5=8y0$I^il$5}M6clky>9_u2$nL> z6>BnxSxNN(Q$o5A${<8iv7Qjg@OWXVP#u553+L3Clz3i7TW zg@DaU_wKpuKldol@FH<#0vcZufvoymiz5F7s)kS_tL zD6m-h;uRHzxPR_10_XE&ToJgQpc4z+jpH@EHw&4k$4BNRG%YiYmir9-r@ve6`DKx? zhg55c$r-8iTTca@3HNy|y)QKX&D%0YIVvR;HgJZWfh@J=2O+ilAc_x6T-#$_#=sBx z%`o*%)l9ijD$Z!^uSMdnmTftD{?AIh^kH|I*RNy`~QUL_lEx zZAmXmOOrf0-t-3S_e!1ULF@|Wfp;H#YzMcvVWFsxHsmM=fLDO*1$aLDA)S+pzfkc_ z=E>Fh=i9HIg6+$OByg2*-k<&PJf(YKh6Sv9wA7R`?_n1^^P`}35jCBrvi$AV&P8e{ z0z|89+6g&n+d#jlFtgifCB07W$=?9;&${AVkj*V46v942;_=5+_@YyY#$ygZ4Aw1j zak9O&s+A)tqr>p6$8c*<)8)??97t-4d>E-dgY8U2%P3S?_baLe=m6~^Yb`>Z4CJ>0 zewBfJlLtIrp28po1i&m2`vSodMC2+U!xHS)YiJEXlLYGOfXdyidj*;;Km3u_*hzGh z0qM4O1d#Ak?|2J$Z6Es%_pbxf$4D}p+;<6zSS$&Rjw`g_d$e^wX^Zg6p^+dG)05Jy zfVdAVDv)6Bp+c?h0h)xam+cPJMY*}3GMi%>i=rMg*1ttca{GD*f}}*Q*9if6FoZzr zSvr#TkMB_96Uq>fd0H2AOQhJOIJoh0&vEh zz0{g)SxFdi#iVdpweJgi>}#=bsZIdM4=7^nPQ}4uM=LOk5RtR@L9eaZ3Y(y8Z;tz` z%%3DU=uP42PidLd!bXb%!~htle@EhorHJ-|pz#kB_kYMnJt`MTzfG)% zi02hKx_TSV*T?vf;swV)pbw9dJYWmF6a#TLZ6agUeoY`DH`AT|LyycSuJ)1tgumdV_)F#dzZ4Zj3iyx!M%z^_9za&64aOg?@OwNJ zS$^O6O8og=VcffhK1>8h|H-Yl6u3e}?2mH-PJ1p#jv4S2LH0ctP*7k_F0Q8!xO_!K z{06X`Ee7PjLt!EdNCK~z)-4ne(7JN8P@N<$W4TTFREL%;1xD?U?g-Rm0U#f%FPT`RJ77Z+O#`_x z{jC@i2_MS8GrajZN&SL%wZ;p@`7s-sFeU)i2m|G&1G-=rrv})?lH>;|{dFpz*>&%%Z$$sRHTtxznt;~OKK1-+3 zn)-qzm6NRx9Fb7tszsqM--k<&DscJ~0V~sP>cbZ$h5Nw;mo8zMbrd`|+fgSXMDR&O zL?Tqj1N_X7zdzb`RK!#yIXXtCI9AIITITAGu7ECg#%E$h-w~9NrkLbFJ&I zBxF3l%gA6ducFTu7O4p=1bS@~QIks|pvg+6fVg$!7IXXQp4F*+;+au8O`(z$E31;~ zJ9A9)_~3w;@GJonaZS(-GF{i?Ve`^ctTQ^8fSzezya9FDnQyrfh1cn%)d4h}J3c#F z_k99+e7GC8gSLA-59j(6E`Pgwjxv1WXdjNSc!780yqS7$ozzGi3MbdlAl{F-zIxhq zX=uRou;ka!@RZ``xpS+j7-7taQJR@#Zb_tEffuSEY1YUCWdZsO6$KSt%-ebBi}-u1 z`QP;_Rb(?y;Y`*@8y`oyOhvdgyh0f<;(fkproGj=Q!-onL)ca_8I>5A#*-y| zmsTefJXCHr7QM~C%5J=xw&MTB3z;g5ZF2EO_V?pVc<(t?X*!6{s=3J6e#l*OeeGRy zf48_hck2_gAk;if|CIfcAg!H&Pjk8-raVP{0s1)2ZRC0o+uMF!>wPn`Y+i$OcR{1$ zjpj8Yl7x|EiP5xL23Sf?mx2bu>eiB(cAiiDY&pupZL{4y4R;KQr*y%1jNAxqPG*tD zL%lc|E(_cFM8{+ZK6BzcXbt@MIN$EbAyB!#42V*A`9u8!{W6M(BM`+q35{@TRK0`#dM`i zx1-y8;Nax^D&uoq(5Rf%)U=S#e1#Sf5oiOIG($Wil)89FWihUv0Ruw~I9K52HqzsF^51dpN-LKTq%=NkpOiGFL@0&ECE3A7$g%p8 zcwA3Qi+`z8VvK{f(Wc(NC5!?v>8H#PseE(h(r#bXsAOVoP!80FyeBZ2?Xue*^`@+(~BGs)J|CyzhV7&gd zxKs_Z?vOj<2AU`8#xk8E1^W7Y{J|pKcK4ASpRd)C-#Bv>!M?Yq=jhnj2aF0036qGb zgb6y1<2f$@qV}s-ei9NAWCR;rRyBb$GmcpSer|G-CEm^T-qc?ZNzND?0tNGJmK+6Gx(X znM^DS$%<-cn9Tk&8<3&tH{HK5p1-&h^8>jJpr4@Ap`A{`Y`5X2d^Fo3)}y3urK6J} zDQSN&L`1g*-d}giV(IG>3ApgL zt++Y;=-P+t)YF0WQ(bP~q(t)^v)&0x@!B>sG3!bc9hd2AEC3s$?$TJ9TUPe-qNxq# z{_G&@$J(za>f&S5#9#sxP2)X0+8XWVsIwx%M>Sk___N9->p$!0tTgl7FSqAakSUJj z%F;CK{laK2rlFx>RZgu|ygG`KPW16{df5DU&hX#)IS;a1{o{Xq5wMLhLqfs`V$xk0 zVf4ePs>p?wJ~_GJR^e6zmIkk?sIZXd-AY}TrpwMExa*)uU2=N@&c+%`01SWgy^)KX zdwY8aM4BMQy@K`@49at&f@kJh;MDcH?B?O@2#e2`Qt)h+koxALyo4VL&)L1k?(iI3 zhtHRQXg+$u;biB)*`#pbztfG`mr!MVkS3=;@QTRxfHN^Pmg@m(820A7WgQnYx7Epn zI`%hqK3|859Mc&nY)TX;+uffm5=Oh!IsbKcEu(j7YvXfiXvD)$qNJiii;%6}pHq^Z z;bL5Hv7`BEYpYEMWQ-mkT8jv)f1nx;_a#z|hiLBADpk?}e(XqcAWrS};y z7)lGbpD7c=;%Efj14FU<)a~y~u&_2z9!?Y_B#Kb^@(RVi_77191^Kz2MClKzi$vb< zEG?N?kpBBMMx#J&zrN(*&Tm6QLZpG(vzzYYEx7MCy|-)osyCh&3~#!_a>utTHqZu= zr9LTY^`u{kVfO1k014`UW4w4ABUrX>*XN^W2{kEfHY_Am!ZPafLt3Z(R2kkVS7$zh zuHPDe&S#shVMM)vod&&5VKW=EVa$1_+B5e}4(!ARv(a-8Vae4stdCJP>d;HC3*;nj zmyuOJ9ejK|N-7N5uXHFV1rzNo->M7=2=-GRZi>v^)aoqHgp%^cp>{jhmmU}5$)EHV zTQWL%_TR_wo?OEDgd}k|-SbHljcCRFy307IH}2_P@|doBoYyx*Z`wBIAM)lpG|BY$ ze}cEBh~vs)bgenAI@W3)R$fuuO@)73@e!!7;iRilNdghgmZcF$*4an|6$Au?)F=%S z%#*Y^-CWw&{9HS5IlY<*n)O5_@kYK~)hw~olGv?{FgA|rl1%RDffW=Lz1m#Hn&7n7 zdN7D9%nG?=VrI_D*4NWrLq|ua$T!wySX$a&XjHWjx7L-`iWwLmOu0S1S*rLd>o|!; zrzXdFT)cv{aBZt&#mnVp9@<<3(Q1z7TI4d_0!bz*-06pSqZ|K8axuT5%`%)&F4K-5<3E%7F!Z4$sSgCVx{bwzlT=)eC2Q5AX*Z@YHc!#G{%&>6kY zZLu0L*c>t9LP7$zcsc;Z{O3^}m^ZGa6r9Y}XnEH8-2)!SZTb3tumIxkI4L=`aA7Y&PzEO(i_*CCe@>%haq)3Cq(4t@3C%V@2Dg}DHxI?0#Qj(9(@?CF4{BpvUA zfS5+;fA<&Ua~JT*>fd4y&khde6c)BB|0>Jy=B_| zJsR_d$V8wVr6Kc8NofO!gRksiEO1(H{(bW0*OSw?5J+k2tmbrqnQUHUhxPNJuL?@& zm>Xn0Ne!Oj1elyj$iNW07ZHWRIg!M`NJv_VNm@4n)ByrfQbPlKJs7IMU+mFo!h|arBxcAu>$N2U0r7_&Z%FVxMrq1DU>a z7RDb@Tu_k1uXQm%`oVDI%M<*QAQ>|$d=4gPP2|uV`*U~XOj%|HwJXf20;fM!6So@k zP=gV{AGa!mrTQ~BwdYN4L5GO1Bi{DL`)8A3tUTGYe#tN>=2zV4GlbiD>EvWIaz%<^ zjR2{a*Ekb?pX20Chp#QJs|>`S71a5C{_L-3FR`ZAZ|s&)8l2jdYZDY8!l5YFtz5aWZJN#D?1`grRfUy;F=szoxpd8WfT zo#*fe4kyEr&t!Aobwa08w6eOGjJsP#wg(R;cj^ zOwmjy^vUK_#_h4h4fDg^_JLK=nY)LF8e^K8v!$Y*27SiUQn5IPj-#G$vNMCTZ7pA1 z!GF**VkT;jA#A^7>6a(TH| zu5CR1ZvTd(CjFOCrnsN;g0UWFceVug9)amK?o>%e{&yv%nYL>$!x5iBvZMaDpW$bI zo+h3r_4_7&nbach!{$|vxZjvGjygkn{rbF<|w|idrn3Z~W<&NVYM$vHu!M<_1H5Oq0#VwG{>ME6+C3@zm~gE#a@1@>}_TFGo~lmDuw62HmW9ieoNh z6lB0P!lzo*M<(Bt5VB|t^u^roZ0sFvS=n*0TUs#w=yq?q2}zcn7=K0NF9ur~(A*s^ zDrinY;fAZ1QB!HsK3iv2wKcR%u05D=m07}9AdFKfs;;1QJxMrol2jWTz~=mDWYAoi zt1*x51}PN&rj!z=iDyT@cR0=4kK%OEaAeny>(*~8=Ah7@-t3Z|w6e10x-#}gz5dek z8U4zqA&Mvk$`Q+#llwd}M;nXUCHIqPGv?B@6N!r#ZFMe-h8vzjA^403CGii6O6b)V z3Isgfgv;_1^7%NCG~YZDl(6d3>PLdkdZ9DD(SIA7<`{UMtFU7Wzt6p#{9f`T@hh9>rPkRvA4`qsL^R?noFhmQB`GyFs^ zG93Gh=5J3w`@F-Mze(ycZ|Y3x=CS2j>-bUd?zOv%)6Q^5hBis@8i9e4(Okvy)orC* zeBu)wtP4V?i1lWw#pW!p0_fUw)Qgi4lWR7{BMY&KPY3r>NDEs7wf2Xb>r$%a-osXe zlhDySawbpjsW1Fy4z^jSmUR_5eJu)PD_@o-zV=fzJOr6pf|5+Fj1@U&LsN%)x))yrp zU8fA>(=pG78@2R*2Nr+6o4s#HOuecPkHIv> zDpm|u4Ao0fs?C?FjN97>Xn%Klyv&TVk`||1!@rk}BCTWTC6^R`H0Z%`JjdVIGh5lnOEUUK669%w=3mSt>-?oT=IpusI)_*qVKbP@(zJDd5$7zRMqo z$$8Ss%S%B^gT>cy)tVu*)M#Sc{bZ9mhfxw0+%O(+!?Z@5ZhxLNk>sezmYWNduqL}z z$}#0vHt<~!r3mNA`R!|rWUyC5hAV9vV_ziSCyN3Yq9x`X~SLLmigaLS1s#MiT^D641CYg>!vpW z7kJDpJaHdr8Skibn*5CRobR{CCg~q8un@tV@cBw!82JpSS}9A*$&I$?+QI-O~aIQ$VMfI87L3vWavXl51`VDZ(~L1SyP1`NJ% zqOe1plfXAd6X!g7-tGeh`A`*6dk4oMT&;iJ=(!I$@LiKURElV6N^*1Mfk*D(7bR+Mv!mOe=4Rujvg?nYCm&QZkY`?WJw88;^Xzs-@A}#-v~`7VxY2gD0C%fWzuw>#M9UBWwXw(T0lrvXJ@AsVvJO>goub}Kh4Jxjn%Dv zmRz?*Ts^7%`HK3osD#Ehs+?0zw~O1SyJ|%3rmPyQ@1Y^t;y0Ul~KFi4oR(CYXVdcnF)IasHRFy7t>bs zad2?6@Z2;Ht{BPHS#9PN%qglj3_EtMG+{|~7KUvTjYl!EOzdFu>PZkV&W@-L&k#)} z>+_#f8^Wqil-@9UOB6Wv$sZp{HFeu}c13J^MJH2ym@wP}bZwm?aRk}W7ql8 zyCFYD1%DigGjA0CZ)|od2g&mi&Y+tbq(4Uz3W4l?&B&) zH6CKj#Cj7ynVP!*=4JZT*AlKf_m-aTEPGQA&5JArC2AtZ1qu^MTnQG(CVT=11i+Y{ z?wl{^FOp~JZmzQCje{ixp0gMD4ohWDjg4FqN=_^1E>w!bb3~cAIB~MF=%6D=(Iu`Pj6eG4W1}YT~xZVdF{LXNPe(8AMX~i4rqLG z@^4|#GjE?jt}dK&&TB}A3e}7A>*$Dad`V|~+h<&#Wn^|VY_K4yt0IgPqPRz5%C%o* z6k@iwBw6urFLNRkWg9%eSs)RJDQ0RMfLB(t%Taqf8N`7fr3zG;+}zZltKQRcU5G@F zJ9f8T(}wb!ipEu23GOI^6@eK2bcF~PV5~1w-4bSMWL~b8R zHAxscpXbH)cOmo~xMq%*L}Ny0`9Yi$CeuZlPqw6^Riz{2<1!xBVZiWIrwkOj{1zGP zbvel>0sVAs^6i-R|IJ>*4jI~_$P|KJdJ4 z<8pm@=U+ZJcEkgT578}%+v1XkHq-{m+9GrV1D%lFp}Ml6Jgw{~qr?^D_TMoxO-wCv z+YxO&E}pS%5Lo}ZBL;a;^l|2#0rkECDpWpQ-3^zMQ7AcKt@?!=;#ZiGX*h`c?@543 zn_wOqAFnGeZmMix;HF~bq5tMMUV)rdzCZ5fk{qG=@s5>?N%u1K9ie#%z}7|RqscN_ zd7}0hek*CXxKgF#G1D#%1n*AVH3=gu9qe!e;{Xog^lTO)imJ6RJVBfO2baUebgGzB zew_^X^8jl*J1R;FPU~fHGuL4(-okh`y;ucSDEbR4i_nXpO?(2Jk&eYlaKI{li;0VH zTkiAPj#sHtxf_jNbZOwq3B>)b&0RF_tM{74LoG2$7ztl1-CcCt`qAuhv#ZNCObQuVE*@ui-Vn&Jq4L8HplHm z>w=#T!?jO_UC2W4%^F3k^JA#%<#yQYcTXp{eY{XzNF0OS%`JA#78Ir*+k>=%V>IgW z@>-=`jLXBZ+y(9}Mst;qT741{O}T``W@rhqDK<;C6EIF4fn}@?yN8++rbXrk;)A^s zlCW*GiP5RN?~^S$vW}hYGD+G_hBv!6niu%&;|;>KwXcJMkPk4Hm-ZD43Sk2@wV{!+ z((*<}CpI^U1AY6^%i-mdytlG1-}LB~zwyx4z80#juZxb1BILH=A>k~+Zwh+~(Sm<; zj%WM|3Sj(pbeMdZy1HU^r0d|YvHz^8vvxtiL!s^u(=AsuPvmy%6o7|^?;lac>bp1U zz}b?$3Rr7C3wX_Vj@F@&V9UYI&SQC@y{Dzs5ipxVmq{UC$_|pw1)b*-)S>~Eu1159w^68~vq8r~W z?wN)*buBDj!^%XlTIzX)i}T}9%RW6@eU)wBMm-p<5>^cPL9^-RAJ_Q4gp!o2?Pi`B zvL7mCj$*DdeyGcIHS=sbfP9Zj7d`iB2OO-RBNer$a{Z5*md}ztiHeU4YYMtiDo*C~ zH`r^G?Dl7~69BO#*zU%-23Rb~_5vN@%&yp(1&ksb%RMWW$}yIzs{hxD+Spp+vAs-f z6qZCJ0^YYCb*R<7b~*a?&=z-d0<$@E(eBw#tDtDzZsKm+FkRwY?RMfd#( zPYz>6{Thr=NlXS$%B}`0Svmji*-0z==!Dty9qih4;$G+_#2r6$^f_dRA~RF}Ztu(u z_vYfR7s+;F0fZb6cN4dt`|tzSsbjeXZ)DY=vM!hV9rQ9we7<0NZ|SGK$Q$1#->+BD zM@FhsbKQ`zs(;QbC1!a#=8z=ym6rm8Z?4IpxMh)nlbsznN7QV&0k(QzqbBo{XP(kb z4vsY;$ADG@=h+UU3FfSL>`_Y6y6I^94Tk%-aS_({j;}PyRh8AC7l&Ud?*1kPRMKsE zp&}zAU(8wFeJZ-iRAeykY~RTa9&MJO^R%$BaWL7U8A0(kHj09mN~A9%IAKYz0usDp z+q!k)=c{lvJK~?s^Qsg2e`gGf-Yu53W!zo7(Q&GLj$&u3uc5C}jmRN`N8}k&v%y`X zMwhihmxUe^y=c2)!$=-7>}n38rcn<&+VBwvuMQCXSKF$Tg@aCbMvGr3chxF#0;5I> zSPzU$}a-ZYveGLHYg5vxxc=-Y`N3*2SI3eP;%JZM(eCs2tK zueR7{(ixO~3UTL6vfX+L*+k#o#$_l!3B04Crmd@~-q3y`oY5G&LIzXXZl@H`j~Y>G??Yd3&^8{2h z44$zeq^@7|VP;Cny~igedWOKWsKd3g%S zBr$^xqDdZoeq+|mh+u3qZoQU0?S-iwG)x=<9AcLuZmryUN=gj!0gcd+QS#t@wOQkf zmKEkIQ6kTRfD<@7+L2%;RU0?Mn$b9i`E*2v4gs&7XK_r?)Ul-IKkc4CEczcKj#9y# z6|}5&PZ@r3E|0eu!=U)j)eORu@?I@$v;(#+;BZHNq$wHtwtBEbLpq@ur&C1wHay;( zH3XlhdwBTpHVcIE@q~;qnkmmP zj~m=3gD@EL=-F$(AX}!l+6Tilx%iZZoWhxJ_+PMba$>6i;0ezMmYKf`^T_{Y-lHNSA|TRBKzax1HAIx&LN5VAuOal%6TXe*ywCgN{Q3T!nR8FZ8AtE7 z@4eQ#_O-6HwkrSm+XwIP{o4xWBZlUMVs1PD$@)>o>stO9sV^}JocE=n2xa*-y$aoZ(Rfa_;N zO7Riqm$NK*#0KZI@Ykw-+g}9smb+9|n7i zl2`2r)9KdShC6q}EOL|CQg1aSUNRp0P!E7xPPRsd`5aTGsU_nZph+n3>+FHk&QxWc`&*|jq zSUj@a$jo$Hy5sgU8a{>K137}^CTWfI6kR#w@&4v@-?rg=LpLog6Gn$hM+$btDecke zeO{0^GF`ZK;r5rlV(@!+$yWm7bGRK6yvpm?Lc(Y_?t#JrVF2I|oB{O^nB$5lR!_Ie z(HOKW@GS!*MGQiC8T@-e|Gn=MsHd%9#XiVD2ZB$jBSBy(B=CG()e3v%<^0Dg23uX3 z6LR4d@BH73ZA6bM9IXu9u@Uql*S6==Ej$*raxXoWFt!dYUzsYvbu22h3q0n9#l6{@ zj7(y$jCH z!MLlwW~rY8oQMdBnqByTqU2;1C(b^PHz^vs(K}-B*d7i*ICR?ac>5?cz94wl0*%YV zK!EGN{{w)xYL-+IfRfN5P?`>vrU$E53a*FHE*na>3PFgw_1u9?&BdF!b{2+CI?v{M z)D}V#MRq$;*rw2%-pwghXzKJb7uXn7fZ%DU6)m~?WU!x=0&~Z~k>~g(Av3NrE-!H7 zF!}lh2(P_`GX9qwWAn_P+YgbzxRfcGnPrxIb5p9#05r`lY!P5mva`1KKHi!G$mz)k zPG1QwG1tOC3Y((x3Lo-P4?f-OV^RCR1Gx?cQg85kvm~G)yUAgX6Na!S2;w z7yr&mw~iZF?si*h-loK+PqL}$kGhg)GS})n5=SG~zjqO2(i<9jqzn`WyGbjw$~5K2 zKb;k0uQNd7od*wWdm6IdU22-2t|{)<0lZfD3Fu^h#2HQJW%8!z=sL!)61pKK$+#Bf zouj;#7-bVt6OTp6+g8rK*D;@_nxRs=@b+33fTt6Pl0dgut%36Mej4B3!1+5l%zIB@}o4pq9TI zK)$zBQo&Is;3lXgb2dOaQtsSk+FlzYg;!@?v%%mDU-kn!Sxrrrt_BF&X)kTtaAZ@i z5l9c}wr3%HCmbzb=2MSnkq>>HzSd5MMq{lo2S$%=3pnHA)YUa(V-!+rRbO6C^bq)~ z1wa&)VkaF~xS^Xd_sPuY*wswSXj0f$hbTtVzrNMygdHS{dg2$&QtSgN#$BaTZCt~i z?OZ+-9UF0=TKSq3{0YTMCC^sCJ`nPyTuQ4Hwb6{GC_-|ZjNpu$hWiAZHg7BiTo&DB z4F5+MDE2?kPcGB-u<_li?6cw0(xL_=^b)}XFhnnzCeE6G7%ZB}`*VDHdS!a$-kBYc zwo_0L(b86Ja{;OPj8*_wzA>XE6ZkeX15>7pFXp|HXEHm7aEun?9(ix*y^Na^cIfz~ zmE>XHGiIsURTUIn#;iS5`&N|y?r4LjDsOzWE9FIJd;4$RldgMca$y9+-aSO7-W6@8tPFsjpw z*-lJAJlVfLzj+yvul}=6YtVaIp|*%eD!ROSlR2jUs!6$l)E1-!>1{%0x<>_9ZUH`Qeu*F6L_`4HCMUNhX1{XOIQuq5_4YfUVFk6) zqDwopm&Rq}ZVI{TdOpqYDG$OWbdi6=4UKUehBne*qqg(LIug7KW`LIs23~eJ9wS~O zAE>s#Nz48Xzrj# z6=TskI?1sSMJkcnFE$g|_-j`q8N__ilPREZSwmBU>~_LUdKpT(;s4GP(}3VJzh=tl zc3|o(Pwb`X?u|aM%!3kxOx{}}x6_?$3IX}pHs^TzsjN<8^~7%JsB00tFZQuo)5YkF zh0ac{1E1a7gnxpAbL?P-sB6Gbv`)kE9}rk6T@g)Iy$iKw_I=Qwg#IaPaKt&f6Z#peuML(lpp$u zNn`*NQruD2x*=G{sU8F|YA&IS!xAHNJ<9L0rf4Bs?Acr#!Q4}sN>)8cE!2#N)LNdn z{IDnYI*q)~k2q#eNQp;gvC{6fnpVr$_BsPgGQZAuZ8^&olKavsYHaTo`)U@bcE_z& z3?5&%^P+>=s-^Y_1OxzY6SO$|bG)qY-NnLJqi$Ayb^Dg^`JDo3Rja1ZTtWFImKmAF z+RQZi>J*<4$wAu>9OTyR7wLWnta8AnC7(V#e;cfFy7Zra;mK6KIpDFo@s;G*az;i# z%joA0h;(%98F{&qJ`z0u)VeB}zZZ1=H8jtcPRNY0I_1Nvgw3{`T}en{Ov3zZ%VGQ} zlFAE8Q6xro8J`%7Y_W;DrUv=NtBiKX=BuC16As=93kzS1jG_+>lJwrp(+A4zZ%IYw z?CAQKm#5PiS(%tJC5gSFfH115sgmpaF7w`BiAL}Vt1G|)*)+nS;e%C~=YEbN;Pey3 zT-||`rXo5(Za`hb03NCKIVOR@jcBZxuj&+NaXFZEpd29sx~*Kf#vsC14skDJIA!2H zu5XZYl^-zKXG0p{LQaoag<$N-#Agp&S0>}fI zQA|g5hc5tccj?DK<9?BsU6GZfr2T>-*U{DtK#7l4mP!g??mzQbPMh2j9~=KD*LLw{ zaJV7=1k$d-U!FyiR-TtBSi=42)1^D&qHD23651e10-gv+lYwP|=ndM1D5P5fbVR?q zFiJzuGyG|!df_wQNRTu36`*UFF34=JULNJ@1xY27P8%CO^!&QTp@cCDy?bR%MqGCh za6C%+CXXgQ?8&omAqY#gG*n>o#WgHWM^nWGDDa{T&ZaM%BlsZf^d8no{h1HX11PP7 zDCy*Fs7sm?%VWcv62#)a?%N?-zA=nfyDp#Z85cw;%XPN2bgAVIOdQp&elMz&HF+Pf zdtt-g4Up;p`Zh3U+zb+(s4?HYUDR`Z>{m=*EcbW165iKW;s{{*vF~WD??=1-;0bqZoh-<@-mPNO{ zDo(H9w6L&V`Tj^r+oA<{VL{sou6GYuazXAi&n!$;r?84`YH@MB7qWJE6riE{rX%49 zOwckXjc`?7@C)XEi@ENB^IUlR{yBp9F2AXeBxJ~`^mPwcIvo0Dr=vkW@0hsPJNcC) zS5t!T-rh*SswaSo!K@mQDf7((@r}!p*D7t7D?i&lW-3@76^c4|_8t{4@v^V$^pg)M zUy8Y+jx*GZEUb~E)CT4QqIiy*)=d`27X8jwWzKP%qumv6W#az#rz^6X6|k?Yx&HU# z&CFxX+XUCdqU}9EnhaXLGN`kRTiOU^UUPGE!+tKXNSX;PqwAeIJv#lf2~eXeMVU4) zjnvR=za0(GvhV0 z?}vTE;@ub)hiNe$eYL}jErBnqBcyzj*Li%+iz_`Xcfbn% z&xwLl8^!&TR0*&eZwI#UJJpt$?Y@mb%h)Q?iFS0ZOa(xr!Q7h?7>SqSlJ59e;B@JQ zu4_%KXI4;&3X6dAW~1oL_V+z2q$DZnhu89{8IPBn`J?-Y&W!a#%~&vQCyBtry1Sb6 z@^0f|k`KPa1~TYD`U&R^<{8#pF;39L*w!~Jm|P$T>NU!t>NH9dTrMnzt_#(clc8TN zC}VQAzR%$+^hLA<0utdGbZ#fs4*W_0Hcq>C$m5HG#NHx3DF_F@Ax z&rT_sGOhT^P_XE`6PZBj`zCb#o0h(xgvdU=M-tu&tOseeFMB2U-!FE1)x zGZ%X{G*P7D*g{{<(={>$-S;yeY81`=^QR%kTR1} zQ&V%))s%KM1d8M4(Gv6Bg3BO1gP8{zH_r65=twPAD=VO81~b{u?;lV6IajB$X}lpw ze}hweRb)*gQ;JmPR#pI`LWN;_dWaBW>K_u+uT9WYSJgF0b#`;2AYaE0F?TdJRuR0o zmQ!0t2!z-U#FTOU8R&V-^??&2lE$b;(nzKu>7+(#OXy zw8DJo(Q(~nMEZqc@lne^D+rUrmNKWx8q#>9vxGc^AEYuuRb5w8OKWWbOHXxebq#AG zYS3f*3m$@1l*i~?7&V6VKhg)Z#14jla_ecY>C>WkB$M4b>t$eIj=`c#}~6 zfQ1dV?Oze!Z*{vt#_lK|C>cEW&}r&};(KH9U`>0wAs!`3d#xu(;uOiFq`z`moF(X1 zrFzxeNa3DJ0(2xtQ8ev=(rct$(zisQbtEDEtH8m5tu0Tsd{y-%ev)j6Ty@z$=`nk~3QSt@sfnB;U)jt24@L zQ{>J~%`Z={El34`3z`64NS!tE(w&FjQ6sI{WsF*b1JMytWS{{v{gHPq91dUXg}(5S zk}}l*x&9q{HzzlzbSV-Ni`)N8n?wUrhJTL#8+6z;q=+k{%vb`$7I1NDdWSkqJIGEx=9vP=9F^xXAPTs|R+6cMW+gl{nlm59u<)HKGBP9bL0wNR{6B%CE8rD8Hs_BysaqP^XpE#=2Wp*0u z-)P5pRRz0%hL~P^BgcI}iR#$0iNX{a@oKPY0F94t9oSQ_eGaI#ltMxrg(?Iu;<-|xHAVXLHn{{mO_)j6ccn1H@deD$!!&^Ve=g=v?SV`1;b1#hkN>Pl2rGiDM zEm}Bs!~a<{)%Irx4Bp{ohbS@; zQQ<^lqv)J@qJ6g10L10kj?~RJtXkczl4p>UlQnMuQRvsv9q2rmJrGr>51Hux3hxsa zvRG-kO|OweR}jCm+6;lp*lunHm9GrD{dDQGPHT4DBD1So%Rctmv9(sXI)llPrl8A8 z8mUmeA^!J@7WqmvE!i5ZFrAWT%uYQ=u{^PPsX4O?SaF-)#r&F!R_4AH#`zkjg^9ir zk53uAy97i`#d0c|Dmst}qmx4gBcN0V z&~>i=(~QFcOc%kzB0N4R)7Hk8bkHn{^u|Ai`WecZg2;-fh|WBdY6_l8P6anP3A*Jj zPfi-`c!KN`1qDS^PP|-tPC+()XR|b$`()762g?we;A1O=!12Oje}=UZw5issh6>=i zTR~Ca6&g(Y$~6H#fz@Fz_wjkSc^7k3l-9Z1W8?Z~w}TQyoqAI|cnu5;bc?^pn#cg* z#FD~Zxoe*i_Kc%*%d6t74;a8XtZ9qE2x~Ql9*2D(sYo!+?9w0p=CZ57c#V~#aJU>h zSybtii85ekWr+)cYWGcQvl6_ob!C*< zda9NtpTlEvF`GZ$^`@{xMrE49Ie_rHE0l*LHd0p+IB6imxIVbxXw|eF3^qp;OAOa~ zk(;-*v9mlP?mpk|KsK)sp4iC>X1Z&m6@gM>YY_p;b8g=MGdI4T2I0zOAqQ+{vA?Jz z@0{ba`{fLm7Y{7a(gx}Uhz+X;_38b3Sn9=C;` zC`MriX;J2f-;dYVB+E8r!}=DuIO}$02}fsF>vRqSZ_)=wtJzhpS^5}2wY<=y@wC2o@fz757DRxzUo{g(0zsuY;<~^a- zlp^%xLtes>7X#ri(XUyzSb*TJtm0t#BR%5cQr2S2I*d3Y>lehCX_gSI%vZTM8@sn( zAyMl+=h+8x#7}*Eeo`Xy^xTK2qSbg6qH96r)nr$*#QtV;)Kdq*loPixe$5LLJ1wBX z*nt1z-FHHCu%$X?;>L~)L_i=%;c9MfE^IwkE4onMjy+{$I7=6dawu|LIZ|%Gc64>j zU!H8$yF(kS$5eFE-Pg0VloJC5$4dheeR1)OT6CYL&)r_8qa_mdDU}H`a8N6I(f`jY z8^w!2ZY@mqs2#+DO#=i?G?rq8DaE(Y8h5V)TMi}Sf`eqWLYv7M#lHGqvt`^eQwmw4 zduiWOK#;-nQZh>VG)tjCO?3H@GKiL(h6k^7rOS){0VF;(mk;ZJdsd!*IZk3Pni(<( z6Z~QcLfyfL)&#H0q`#nqm^)v=eHbB$kyznVt}Cs%Djtf^HdJvLc)5Rc>|=B%13v~` zW3;Fr9xb(519PA=8R&z59oA|KWU~o9UJ>^men$o+71HEZMspgplttaqW4@*a&c~TOvGjf+gj}Ih~nsuFL9iUl-g#(%VK6woI^i!sTsolV8hW0 ztUT`D=Ph;wmE*>F1MzMv&?!iNEal%PH4?oks^xNRfS*REN+dRH3V|+GI(v5i3*ACT z0#88@((bWJ>pbpo&>Ec#VF$LDf%Bs7h&40T009`F)2mk_EzFxBwPSC@%~o&{2dM7a z{V8gVk5+I6Ev;y3qUUiZnHkj36A3-;Supg1EVRb-Z|B)*<@rv(5p|Iu4kD549D-tS z^x~A?b&wVL27^dK*5YVQT@8sFY70X)Be4LDK=wK*>mJ^qSf;vuu-g=%#rVrc61@ranCkbwx4E@MLaK68M z>jcP7=J3w^jcy_@n8{`rWq@sfT9ppN}UpoI70Qc*X+;^_#&cU%Ie zZ;JWzb&&i`a-R(05Zz4@Im~qXc{5ZLw$d+5z?0KoE%6=^1+v3lnBPCupBN+_i0~Zw znNBQ47js5& zC`&qq>SK;N#&Ck2H5&!ZV4dyq5w85a_jm6o^$2aBE7_O7wXu>eRP`IJi%Lb9N1}lgIQib)#Fnf_#xW8{*sEMGz4t z^-}<x-!O=Idvk zgOmIO6Nf@d&)-gfCtzR%VffCUDrypI88BN;h-dTVv3T1^(d#I|7}#iv2k91gM+;AX zQZ_YBm612_!8BW*7}%U8Q0)l0vnwILCoE9~=Q|C5B_71egd8tvu$`q4U+CcwT2H($ zSe`=+ARn;l_)w4VdVj+z!n!rsX?qK7#{w$g%bd0+uH1WfL!uI^6$CFB8@8pfG}-@*?bRRB0M8R2j@ zSkR{7^Dsa;4t#*o{B_XvxBeq+Ri-PcmO-$Pg>Em8j1S*omT9&%I9=kn>9PrA^9qG( zOyuertuM+(1M1HOUrK=*Q7a-Nqcm=?gGMBdxx}a)Pq}uA!V;Ksl4e24*m^nkrSm9_ z7sqE;Jjk!JYN+LELXAtcqv{vT-i6&rCMv0@WIxpI(C$z@*TZ`+M(XyTXBo~=5-5wC zzkS69NbI);%?<``F|3kCfrP0-Vc#UL6mH-18a-Hj@W(J>YBHI81{T%HBB7nP%>m z@0GJxP6|1GRUddj3%sohH8?C10I>yqmmdP$|^4H61@g5y(~9+}pPi;Je! zsIv|fVk>C%`J9b97=wA-Y(?nNPT)GUhUPv#Qn`bcc|QS!SG1{-J}qCC`r0gSaNDBj zmW<3#RBPFKE3ZjWmTr2FySa-ZN5GRm@BU2@Zf#Wtyx&3W8UOp>-f5*{4XS?OejZ+s z`}e{j+IdMM975lus?|G%QD{9@39N6euXaJmJ!j%VK4)%54UM+CQte(4Yfy9cZxNM|YV~V{XkUnt9F5?E?eP zU_YusN;BB!UQf#%+=H(r;o-=_P~YKsW7S+OD0J4*Bk5ardp3>8K3E|<+zB(u>~eVU zckYh%w`OQ=?c&d$gtUI7`S=q#({bWYW(rPp-NNVkquWIUg}fIk2C2slvd4=OxG{Ow z$mIaY;5dGrtjX-O7g*{Z^PQXByewsc55kE%j~%_a54y5RJeLHCCn`&t5e`Tw2y7qf zPj7DlXpo|sP44?@bk&a(%WG{9!zI)g^&B+r+!Qmj*A<|W`aw!T*AJ20S`X|aWkj`c zv16>V{->h<5oXCs3`A~tLhVrs>$5Ts=;30}oC_W=5(~1r$G%8a?ik3KKg^u?O$p1H^uUThweT7lL;s1+Ra;!E3w@e4Yyrvze*=hre2YVESap<^q2_ z*+*SIoAHgxEmZAMfzte}!(F_5&cJdeaU&!!x;RT6=)$MVoR~uB3m|!4vu2w$=99ar+0IkE816v!k-w~~L7s`koW^|swIQ{>i;TZ3)PEW5-<$fYxx zVpsi8wVRIxt?q&yFDNc<_bG=e*YcT6edK~^J1y?T-eoqaBP{LGT_4Y-68OtV#9Rxq zb!!Z7qP-#N$japVWAPbT>7fbMy1pxm;bG=H>{p6_hY)jT?nW(-+aR%R-# zX{(^0QePajOeKL68Gp=b%h}~*uB3zXh*qjr&GyD4xBy30k=0@!sGSh;r2x4a7E2ZQ z6eUqUUB?XRg=K%M-2Ta>wNd}K5ybejU9#Cd{AS29RiZujGtrI3)|gDjy4?KC;^I>E z!aIHaX%;V8+{@0c>0Y@yIr`&bFwaQr@RmrR8gq!P$0g;=^JEwIc2c!YA9=@(PKDgx zGMV988TNY0Y`_$4H{xN`-gO%1-lu09=fOFZ;`{9>N8#5j|9UC(D#GUFcGWwn=B`C# zM7ZAPaJ}vyMO{i4lj;bypjRbqDOP4}1UCCh$0MC6pKf|h#4a z>%eVYVYaL%s26PgXYhc4f@LT0_xK)Q5b zbLJ!#CP{{VB)8iYrsB+KxUbL7&KcP{o49JL5;g_c38?A4c@#$tV0cdBu7d9wW`lIL zzabaR#=;RAjQp(iLS08g4Ja{R-VQ^LR11lo+1vm3?B4#NXT?kcU`dNMs2v^2a&n;H z*I-=ahbWD%FxR$!^lGhpkc%ZEE{q;HsI=1BlpI~oXpo!*{NwvGXUSI?M?x+?dkc2X z(7Gw&$|Q*VR-G|8KG0YZJ@l^U=gpOLaiZGf{*a{M?$AYe!%M%ELEqg|jA4K0Mp-rV zmdvGh%n*0;tu6wZWy@UAXDu?bX4ZI~-RFgQLlYI{7_oc2LiMN7JRsByqCmyEEty z4ZV(Yqx7DFCLdC$+6>0V%l}Rt$eOfB{khPh z9b8HibDgkJ71aY*dLfm%l34}|PAao*oZ+$^u4d*H)!OrT&=n*Fz5LF|9~6=eLU#X$_q z|0hmqd&^EjlG@J3C7u)+tETJC$xBes<)Qfto8FRA0cVeeJ z?}f>w-=QY?{-6zNMHF+{MyWNoS>?1+{2Hw8SWSgk?(B)z57!J|o$3C0lHa|X70q8) z^>Tc5PrLZ{m7SBPk}nr>)O9vu67xtr$-!84t2A?hHg72Y&fp)hH~C*6dQg(x^@AZR zzt6yF8bG3B!qYJl@v)kQY9{s-$MHT}DLpf-oL|jPMD_HW<8^ShU0lCxtx0dcLzqgz z&Q6zVBJ8&XWTK1hFL!{cz73+%T({sx&H&1xlfz1D>iwbA5% zFY1L#(y*%>9u57tDE0pLJl~qaP3MR|$TS>$o=7rI4RZMwCAFKq#c@{now1jxEx+kg zBirkrXfsEYy4vqcyD-{`(4ZCb zF+tUAb(&}XvCsSm@Gg;f7nfiP&FfL%v6Ng@B?aAJf|S%=Ua9J*!=AtDixNIfk*LOg zT~AL}!0kgVfYiy|vJoE|H9Hx5>4R2i^A;Fw#$&()h8vq$RuT4xS{_MszGeued<5P+ z<{}vDO3fqRqZYSA*YdSt8x-ORduHx~s~`QSyVgWuoVP`DGm`84#Sw6W_wU{XpTmla zq=csXKago0-{ne{H5*YTX72iA+>sh zU8s18_|B(rxbw(6tStk7!MI(%dUg&Ln~TM&k!_YYCx)S|b%m_P@@EQ5K0kjzO|qC> z92B&=H!THksQM`*DFEe+wO2GfvEvR3DQ`Jk4EjRA{`0$7)^!xYE_)JOEz50hlg5>1kXHrZqk`!1L>|ETE39;q)=|;W8hsEgiXN$YjypU!dr)0n#vgrE&Q#yb)6C(ehhnf_etvX#dv~|} zA)6FSQ?Ve?THmbNxBB@#U2c~q4ijctv9`8e9@bkASZ^(Mwf0T(^l5=T?<9!3kJNBlka-+3IC;$`hqU+Jdm z7xdHIWV$kO|jRJy7`WddA9`-P$2j1d&O zG%F?~q~Ns4SCoxS6iNrtssQ1t z;NnGk)X$Jv1#{oHvUx+*O((U@t&h2^V3-SYM=AUQT$Hd__dcuV&H&oVV+=rrU5wC)&v$2$QZ64&VJ6MVGw7`VA^ff9_TEOUBj;c+w@ zL+`{^ytt@9f=EVq<#ICfX)!Up8)tsO0lwwU>GK_T#7(E1)bfD! zp?hUN+U&_hR-*|sX{DF5S;mPOHYBj)uF}c3<50-K3~?%_;<~Xi{w2qduG^O->O#8` zMu#N7&UuYE2zUF;gfwjaY+98?ZKEnkwE!oh}&+&53cY>G zg6MLxBA$N_BHULjy#2Y?OWd*u!rTB4fCTupg33iwM^&;-L&8i&d-{EjgygR&M?rAC zp(LS$WsY{e#52zZ`?q_mAxi1`{V-Wff9Z_%c5MHwpeSge7+a zH706{-94Nfo`SE}jy!hc8TpRjYnjtaQHCp3Z8zn+-5|uxl<P^NyJ<}6(k=_AE_O^CGbpIMYRXrCd#fQWee4j=tcBCF zuE*?^gdXoHw)?=)zLLJPOG(J+rOtbU{`lH6k}DI49iBHB`<2`P)5lN`nNd~O-`SI?bdT8p@6!&2mOoA6(R5hs2uT@i zZ!F7KUZu>s?Yp|PAPDxYqq9UE-;O1e;e$^Q&Mqw)xa!BsDXmM(9%zJqeo-Y(M& zBywMQD|?~S@^d>(xb`@wMq_BGP|`tN(udHRnyD6E=;?0hf`?=(J3sL+Eh{cADYvyf z6&m;cfU?4RF(H9FS2c>|Dr9BMc>O+3$$X}~E<{jVtyE$rs#v8?RuuU)m>Bd!yU|9~ z%rF|*hEZaSE(Lh^XnA=Ide5uB z2Nm6yz6zdl3c6MrKJYFgATRQRk%G(lN@4&HhsX}e9PG749%1*hh(p)^LTC9mYPO(cOH7r*@ zRtORckYg3v?1|hKi=QVS0@RSuJDty+i9gG|5-cftn@7%^BBmf`LF;~5In8B;DXX>5 z2C6sVQ7Mq~yA914F9HRcMZ?8Kl400sz-z>V6YV@+1JFYA)29H8k#BB0mbtxvJrrQM zDtx~&ApFYUP4TQPJ|Az*CjW#L0S>jEkOb}uRZZ5&U|qtHfTkt@n!wc2yT)UKLmL_9 z9|z6HR9E)|voph;0Gxe%>6PMU#E}?WtqLu=S6R7y&r`U@`P)~;ji8jfjW*i~4_l6~ zKqO1jK?T6&BbjW`K(wN{)#W1Xqv;>%XiEK2T^;-m9*e^YIPN-N5HVwdi7GqW*qm|9p=x=trnY7=a4kp<2M->;yg z4C+o}zsXAsWJI!9jyc&;B$S)K}GTx@~`FzURQZDn3l87^pV}xJr{pguFEiN{Y zBn;rrh7L`Jg%7=*C-}38V=e?1-gT45d;as(+gq=wsmC`O0}ed!!+BNGTo!6aDT~h? z;yJ0RwSM_cj}ty&G3%m)WNu}#-s5cbC=IRs>MFW1ojYNlR_U|DE3x{Yh>55Uy$6F> z%4*Y<21zA`d>e~yr7?Qy+`a|h^&cJtDozI8$2Ajbm;SXgk3Nof+JgBKF%L$DzYbC4 z&l{OQ9}>J>=ijR}>k3W1{ywQm>UhPeI0Ly;JFi@Y3IBmEU=(Tqpka(PxhJS&G16+G z%7Q?(LwU~AVJe?gl+$N?45DADet2pbv0;vP)024lu`7lHFskMYT)M<1FA;v${D+dM zSq~f9YoJ(g&+D*(i+-eL(6Ut#6xGfYbnZ!3bZeNNP3 zWe1`&hCAQ7a{~lq-tD@Ep`(We7x~217CzBL72i5XiHh;*%PKl120Xp9gk}61Xh~CX z;+hKcet~sYc-LjuQQ*N7SZy_F#DBbReeDBT8*z+duS5$fRAp@k)Q-Cdd>%J zu_vUC-Mrbn!~Jkq33$Mvp{?s2H~5tn!aN>dy5qUOyrwh$qHx4EqFrs%314Dcp6o?) z<;wNR)tA+J7N1*Hx3(_cPn@{!bbtz%8M-OHHyVaKX2GL@H;xmj%#E>8I-r37*7v<#}8-tKCi0X%*HZ}S1B-cJ}^Fto{n4&?3!%Akt| zi>3-+1vcog4nUZgA$B|jPHr*w=~g|BJtS`Tdolr2)L+GPBma9E0A!<3g zcx}(G0SiuB)X0)0Nw@v8Y;wlrH zI9PW6KV#x-RxX+TmbTY|b)?=?Qk++_0~T6a^%CAA|APGr_@LzE7vRFLv@1jHwuFt0 z*u(UDR4oMrXx#rD?rN*{8M55I!|R#ObHnnRO*?_M_8wF|NMJm5NA-?@HYy(M^#+mz zH?jh>^p5n+7E@FGV-_>ZDDq?GW*$8O^|KN~lsE63d)trBn5pm;pjAeVHc^d`z0YdSF#LI7++&{8~ZjS7=3eJuR!M1O zSP!Q-J-V5k@eb0G3V-)U!77<-M~*`RerwmJX6RnN$^&xHTO_hiAFKDY5A_Tx^{Ld| zKTyXmsPf8n#k^!0)z&S)Dg4m(WKqid49jcFLN^y*xN2uv0=;@;MkqhO+yvwW1e!T42& zoBluew9j%^&=U36)2aXRbhf{Gsx`I!cw&Ah^&^^=Fb-Ji@GZwtht2x>WZ(e#St85I zdX`uqt7u(Ar^cFlbMsk(N8@7b_piVQ{D6l6Q^tCN;y|cITTgd+`#vxdYAsFLg!i1u zvDqkeauUqwYk)ae!zmq1k)5xZG0u$Daf_FJm#i}sAvag`iO&zq0g8^87U;6*e7p%A zX(tilXnYJ$rvxxKz3LoSogA*|T$c zmGUgJB$MF6w;pPnFXU8Lm(W-?d1AMJcoRls&?? zoJ!}`>~{Ggm$}X8Jmk{n+zYO~Jja;F!IApUTW7;yR9mIKZg1Qcbwu=%el3W?!0a3|By604Eph)(WX~n@zgIbVc>t_C2UHWKlSc8H!~?Rri0CpVAG3DYR!5cKNxq{ zFbrPOgVXPb1D<3BAD-U9!ZL8Zj5#Ul*_g3!<#Q}}5UwZov7x3(h@{*hH{omuj6fJ- z5WO0om;+Z;*V(9#qG;;GWS=IbAivO~to0z0(RO5gFLWY7p)PC78v#;xmIl< z{~L~VrRgA6ZdISRzSW3mNo(3WKR1n}H#C(iVTF4Z{5w?VfN8T?F zhcm{AW2MH7HHcBy!33b?Jo;ITGl~Nhw?r;AiTIQa@A7ObUzZfDe!f*3q7%R zh!7>)*#~QN?{KC)7}H+@01vkzWuhg{yM`~z?&DF&RMCCkW)s$$=3m5BhLMrK(Z~@p zpKVKD+MmlF+a^_he|LI%363r9>(^gNBC@d8_Je*{=Z^#8c~}rigULkYbvZ4(tr?Q7t|d_e1gtw#q)51WACu2k<03<~#82#O9BQH_dvR_1 z#T5s*72W%%g9+*(|K=+8U!H&@?x)hv(j|vM{VbVnzW1X;OTQ*d_$eMJwZsVQsFVV^ z{0&s1;u%|58%ZkxN4T;_r^)jlHrJSiE%6U|X1LWuo*B79_aXOXsAiq}s6z5n-G-VJ zJY|JX!)3m%TDm|F+i20(XhG@aM^zB{7LwLlD`KR^&2^b(% z167ZFJ-KW?@$s}a6_MI{vh?VQFi;TQ8e*fDi|U!`1MHK9jiAr3<(}7o9z7&AxpoVH zOIv@wYpJLnk4*jm#Y*C2fG;UN%&DrZfqu~MqMT#f&Z1E)0C~e@S1OA!sw+EZbL&k? zuE$+TeXz#bbAQ$1MOtK^=HmT!CW8kk!?zS$QRy{HYfA$$G7}=o1avT+;QK!a$h(NG znd=Zj__672?(f2#UDg!dmSdA*5kR9cS&rNnLuc&P|7btYwo%`E>&tX%B)QFJq@l-G zS+FU5d6txTcn!n6HOe4H=5#(d+^{I(HzSmmj_|k>i?heURrNgCf<=jdh8TS*DdBj% zQMUwP%Kz(UBSPJxOuoN7tURvrMrZVo!ScU%i{!$^t^A7muj3u?JCOU=1wFtc!>oXW zkVlbvnL*mqZVTXteGi;$amBgmVU4Az#>~~?RSM%=LD==ZB5cO?~)!CH;K07hv)iK_tN0m- zY+s-sk~p0#7mul_OKCCsNuhS6MQwB2*WONYN}a(FaY(decrHOjHO^Y`f2KFtG= z4e}Qx6mFBEsn8=FWLd5e6?~TSp6l!k8fM2H&P*u)anh!|5+4(4B8iOw1`bG+*2$yA`fV6Nqm#~9JSKn` z26zH!kp(~ma_^M}HcGodd!p;sT;K1N0d>nITw7SKuWP?aLA6#zRoPE`mLqTDO5lkH z4YK0jr;>j?&iH`2@#9S;CH=*)|LHRhRm9vMqZ)F4VTV0Oi=@9@M9xn33mQJS4-bE6 zCavvEnz5$m%)Nf9tFgCFd5&i_)>Gb6Vj=A2DE~U%nkI%Axy9l@YyNb2dwi``nkZ~Q zG)jxZIod%U_w8b1Tw&pz1c;=S`A)_U>ZSp^n1!{5$34H;@HsMm;LR*c%P5%ybWwk=zMWvIv2TD2S0l2I>{p|DT%3IO#wELpvIyWt zwDxCXlTr&$Bjch@eMVL{!rqW#JFc(FFI4>}7O zOlIf3E^5 zu}l0PA?kC-(>JeQkA|Y+uw=`{D??+|mDJ=ERaKRz-p5fXQsYTTxZ=%0ODq6)ZGjqE z%$F!h-}=+>+QwMS-re0?*H@6Xrn8u&}69{{8Ek$w>#LNefieM+Yg*8Vh7!PjQXZRh1L%IC8TJDjL#T zO3J@zMi0ut7n|LhH8rGQlN8h+NtLKsoaN=kJ)lZrU!>gKr9L-0LSV?QTnc(ny? z2?#hkl8ROe3U*+e&hk;sKZ-MxHS^+Rg5n(2Ns&O<#r4ZPwYjwNTb&9mXn~O+@PvU! zMErrWuqsE57Neen9RLVIumON_3fqFaD84XNh9?zA~DkPR$AN>{J+jvggB(LaBh$Ll93`1#Klqs)KJQ~UGZx&uuAe(rz%Uz_Cr z$V-548Rh@~_baLzT(z^q-~^VW_jo&4*;+@6lC3>~oH=#VLuKE`V8VpYuecQR6##AG zvRfwxXyj-vuvAtk)YMd@m}%`JzZSb@`(82>1tQtt;_9nA5{Zisu%||g(;;vg?H%!d zMiLAqWc5lzg`9C$=+-2MXS zjwIJTy_Lanax?`_D?|f3L~pU*Vsir#H5o@`rur;Gyq+(CRvt?gfhI_A`pcO%Lz6jb zVPSN&bV*_eJq^`_feD zzI^>|4hMCcNGt&I$x(S}`f87X@43&>?}1tkt}po2+6;#{*tE_UQ$(6s7f*Zz@*Z~j zry<)9ET`?I%9y|Tx$;s?iMZT`b!ja2XKA**y_0}++qQ_c@XOHWC?|2VGji8n4)3X{ zz?dv?@!sJq1%-*G+nKP`*6X6{my8IIsv$hQNds)0lBW9)0G&2Hxuo zM;6}~{I>T+#l;3PiC5Q&L)S2;0(t5J6Yw-*k{)?k6hJB94u_~9SkfM^Gvy${H`^~E zV!XP_elcQ9HU7{wmDW97e?NVlrEp%8>HAP-tzN!okilL@9W2OgBYN+6f4K%aP?&>D zJ;7iXS-T1H^E~A4;4LR9H$@)oKr~csJ5cuhYN=ZAbTjm}yPzOm_cf#;d%#dyX<`ES zcRPD4qjeI%^!GZ)fs6`EJUf0~2sk8F`|6D$3jcR3e5FM>YKFCdoEyJsYUmR7rF=_( zPKEq?DDdJ74J55t&&FB!%W`>0@NOu9&^VdKA9HC;zdc`v?9!x$W052qYw2 zrZP7(IFm)t`|(r2&J}x(0i-p)QwX-bI7lDTEp2q`Y(`6WJ3n9Bl|)@!5*ySBda)4r zudMG6YbUJx%xkv%=t^-c9z$}R7Cjtv1h4pQ6(>uJ$s%-OkF z7Y+^_B_({(AX4DviXDfdw|$7sINevG_7f5K-=I&zCDr;ixmcHYw_(Q-x1hf-c1vDs ze~|+DHz2>iBkvZYmOp_EKLf*HnCHH@|L#Gn3iEANsC4IOEi0tY6nSa`%w_Ik(92gj z0gouwSI!PA|9m(W)5HovgjA`^S%kW@1HjjR;UYE6)wX@M<<7g59!e1*Iv$)WNB`VM zc5XUc3yQtAJ<;cA7TybR`_%Ya#?h&z$v(AY6$Vo``&&y;QPH6~8}KXn%On0TK4k#e zH3r=A{Us#mNbG>!HsW1%|dPGK})f z_N9Ba{$3j5&ycja-uE}~vAKzsnWr{`Sro1HI>jJo?|%_z^3b0GD1c)d2LNsuf#hU& z=>gQzjLMZFVrFh@A8YUJ1=qkYopn&Ho3n^Ezhe~6dLERqoW2ZJZaw-nC>Jadq>b%e ze~Pv4Lwt7r;$eS)b0bZ}tbpbx;T$(1BJ>pDEjQq$eTk7l1;9q-%`1o<&SoXk1Ri*c zmIkMK(BGW%7Uo=E<=6BWVwdpqifo|f0<-s%y0dG8X-%U^`B4hfI}t7K#T?=3cN%f_+cEG!fNb>lSTMn&_DwpOx0eqwRI?yjj~w68&pA5Eay)jW*n z7SPdvG9X6AIi;tG76+B1R1u*1hAHr+@?j~|aIU*^VhT{^&nA5RK3Teje0*|YZCzbmu_MNoEte09J_>6$f4o^;&f*%DKNTXy5acQPL}Ow~ z9~XNh|8^?}M1=r!-q+tUS8M7ifUo-K zXCYOX@o~4?~w^Aef4C+2wP;QpI=ss>Q zL_Ns<_=Kl@EzkQ>GG1B`$&2IR^w&Km!B4|ewYNG1Fyb7}U_{Wp2bomIcZ`&BxY7`gp(JL&CX=IzaoqX_(<>h-hF=R~et3Zkt0HOWL!L~Q7B`+=1w z=J~U<80O`qrBtU9^=_N5qjKYT@~~Hu(g&Jui%~=^kucx=9glTHns&j zy9oR$tU47jovnOEx=f?5#j}Z#CAy?W7qyx{Nn2{QVlOHX)M12l?yI|P0g%=%gZ+uG z4>{<@!uwiGPaqH;bI*-becptPv9Ok0etnH1EzUGGT(-Vh((`~ ziG-C@@wRu5HU<^>ovl!u2CBRTFq`7%0)fpf57tVNzWg1(ETZ-Ko$VhKQS@@KxQM-mBbln zhf@p;-Soy!g5|3y-JU=OkghsBZsTNqt}tqYig z&alnH*)y-#5WB|>n-GW(F>+$2jPaf64sDEw?NTNtsLT8IsOoJ6tsyXM$UZq&i>BVo%$TSUL}PaQySplH3qFJn>|YO)g8`Yr z7O|+4r6r|_e!YcbDD#W8H!lUrH`fb3?*WvnGVsS*JGcM$fF>k^l`TVMq>hDtBZ8{_p+nMHG* zh@YC9UojXAWrDt&EElXvNxdq%Spiu4BCNO3G>qV5oy-?~f6gQB^h2!NK3ZzpqW8eZ zD+p#exVSmw5F>+F3?SR}+2nZA1iPz+T-P$dZh#spVq`k46#naVec)zx_CLKE0wIDt z-J_~5uy49e-8b+Xozl4$g(O|R5hH^lE$wiNaA>}QN0WIb4D$-GzO-}=$KQ!PlX=r; zS}tLDL6c8-!USpWS~=KE@8(m{z@h5FhHUr(e(jU&`_~E39=yO%ckO|M;T8|j_?jZF7 z!@;-E()-tyT+h&>O#d@*CQ^s3cgwLTJI+qbpb6WBG`ET@bL()#8*oCPh$Gvx05dKU zfxj$TR5;*#QkpNuMhp#sk}Ka}lD+!-@5lUTX2nIQ6e;(5@SJYLfvMEF&EM;0K)|Vg z!dx~a)BojYit%Re4BLlT^iR7yZtZ3R4ssR_wT*WL4V>p=5E3XaOy2s=#_7Iuz8`QB ze6~0Mzp>km?4zYuumo1JG9WY!SklO-+H%i)`R!V(mSfzQ3+V@S!BKdz2Zj2gti zLhq^-b5>z|vg#xc&|{!Bx>GSmEDq(2u%*drdn-tENAnu`KeYhaa@QLH_NR;P=NChZ zTF2*<*k60qYZrA)&5N4>7EBCq$8!g1Prs_yCj|TuQs5Tt=-=fv zGGqZE^ZjE99H@p>VMufH9S0!bp8J%U-PrFVwu27$Vky;Z5d;7RLKiINc9=>-XK%*sO|G|A?tE@d9~-^FndNGkC>XrFBT z0*!w10O(wgQ3QR86B8^otgNA-7raoyuY?nRA86}VM-pZ^`O$ct_TxR=fA^wQsS6CY zy4QIVrsyFL?>lic8r3kw#j;E6t2_T~XZz#4*IhbAz;|9r7DM$oX>H6F&S1Kj#z&KM@lWLtQn-Ix?8-Muh+=6p6&}V@1tp z#@ulR4V!4DLc=lQ>qL~H9*vG749c~+oAX74OEbmB40G7qn=;u?T^RQ%JxTC$*8|fYKZW3t3ii^%Sjgwpyv}k+X zs`1DU=g#6D6@e-*FX*|^zW0|$-fTbcPd9sYK|Jwtka}Z(WoLiIW#UxS+$@Wg)pd19 zdY@19UU9(h2k<`kY?~X2!lAc-H;!?2d`_*?=f%%Y2nC>b^WEDZU1J^ezX#}vDYH`y zZMj&M?uuW_*{O;1{0RWxzcBiHCEoeP++`k0nDLV>WYAMJ;2>l!wM%-u@u@O&Hh3+h z@YTV&vn4B z4>i4u;TQ}Qeb$ktLVyDiiNfG*-t{g#c)PcLyJ&t%wyDXR6b%aRe1U^Ywb8%IEoy+B zCab4Rs##9jU37F3N`Alpdx0hV4ENC|bz{#f<{mwP%WINYEohF122b0<>KvRXq+gPu zpy6~gg*>$QAK?tk+fzNUGWLKmT+`yGhB z>95C78OPh7qxcZ_tV^}tm6MZDtp^S*lI;TYC@x%-3jR*gzUYsp=qVtaT5odb&>&YlCoFx+|WS5c3=b>LCLV=^`Ok2!Nc(EBJm$8vBjJ zjlV?Ul)0QG=vObgxvDNKal?Fz-mnu=dq3Y4G7sSO4Q->j&+H4)pk<2Pg_|4|vLB@u zuB@m8F?;Hn3-Z@R#R1D=bVN+xuWh( zxvrIY6IzrAbN9W+(NNT^gtwJPjck7VCQ(dF2jozJ1Q)mcijlfH9_~Z)wimrMJJpxw z4_?dc;JX|V;%;`J=lY6(D|3;M>8sU?^pHsjQE_MZs%2#4nEmK0e2#O_9>2NBxPFXe z!S;q8w$i9IF2Zx!;l7w+LF9fu*6?Gl+L7Ir*LXj`JdDo@KKitBf@)+F!v@X6^d9Y3;!{%k-T{IcQ+XxYIJDD$$D4IS1a9UG4Gk$bXX9# zv%Y+>9LIN8Zk&68`Wf+J?5+|>=M)l6-qbrQM+u}Sdr(~51kg3!x;XQlPJaV}9(YZ& zI|;LnysGuOhdREc7Sk2#Xny{-hTQP0W&%F2GPKCQQ+&IZ?zR%Y6`1UIXu`uhVEa34 zP^YgQ;g6XAVX=~FIT#$rwlR11XS)#n1wdmDh^!!xG4!aX!qz8$I0i17ROPumet+(k zE_K>_3Z&mEqcRuxd>H}mr8RfCc-wOdW{A)w31-Xjoc6&J-iKaHb z+L@T_s4c&<(p^L2T3xwWJ1h{n-+lTUYwt1eJkfNjS0 zufO6w|K`Bm^$WT))epof7P`9BqEytNBgdKFZGHg=Z?XJvMtN~c(7K404$PbdDj`7Q zAx0S}4;2Cvp1b|Pxot^m9n8Zf4$MKJ17HS`5D=B8rdrD!&K==D83D7_=JzH-7as^q zFwcE1#TOh!eXAOm>U#b;UV|RlC%Jgw)^YKpYnPq~k}E&jD3dL)kTUj0By>#hc;!W{ z>toXap4%#m)r;>%{Xik(a={viCx|q#bM&pB0JW+X_%`6!zr-AW2C{ z0i7Au1|x^1MICGmF(4l)W6BDK9;5Q3J^v;Sv>a|pFX=6peLSz-ZK3=TZUf8shF?i)(q_9d8VlIkeaO+0y{*g}Oxwi}|Pu7?TC4 z=J;fF^-IUA+?Mp?2t#OpPXnl#0Rk@-owf!tt9{`O<&QpH4tYYMh3WIi9Xv^$8s z3}R8_j)euD`sr$FEp@~x-8mvF9iX7+cq2001I^7f9?$=@2RUB?#GAK*T?E{OO%&0bmFhNU*I zwAlYiV0IsUzo_{2>sLgVS#Xge!H?rns!J7enVq^*yH(6st!W2uEGeB49<>e5uJp7> z(Xj1#9K#5pF2MEv_qD{frB-a4SFi$Wl~b1A)>b?g%l}dyEC}1=katqVkF4kWLD_z! z(c}>}U{I?tM`Vh9Y63yX=|Ws^=MJ*7M;VytjaBX>iRaK~j(A_1CqC4NwYbM2-+fZR zP8MPy^^u8{Y&6bd5l625n?DnHLnsiewp{d-%<|N#4p)9nk}j8@I7J^55sT>Mw^r5j z6JfC$j9eG|om-d~&WaY=+diV`X%8Z0JCF=`q+VD5oHPDaLh(~x)VfDVG+|GL5t zU%s~C3^c+0A!>Is2WxMO6BPKu?=gI|EAtE9qZ}+% zX2TA7T)MOwAGP&-sncJ|g}n>EbNn<-AP2+ZyQ?w*onY;=}zQDc{gA|wV(?PLtgEhNoe zFL@F6&2$HoZMk_*kJ^N^Q z>u6W45|nJ&A#KZ6UX&_djGf6XQFH??J@K%gLVy5$Wo!`>3+vIjjiY^)eOdKujIq3j zh1EX6M_Vk9HRy8MCi(9@jqKyLlome5@t*oUnt-&vqLj$avK4#PEa$u8f($Zlrug)n z018Im=x7BA{mzb#Ait{UtTOW;78Xe-iD-4>f^QY?NY)glkg*$HudVyNE;Xw_dhUda zf5fa3osu=eJ1c9sR02i?^PVr+uUNVmNCOq5RvgR6UfJ>1uZY6ZpSbL98yzc-w#X5f zF$T8h8;chr@2ijtJ$+_vt<;Ym`Tttra7zNa;z^D-Z?`ali}spgf42oa_Wh;D<{lRp z8T~p%wLr|**vROnqLG10%se|^>aChd0a~1$9j9-vUQM;w!1@oNO3Hg%ZHd|viprQO)21&&)AM#rG}p070SWJqHY0~r7gur0lkrIx!vDSf5Uz9hLy`u6O^jUV zi@t%BP48W0F)NGAT}+TtiDk>a_j@U;uEoG@t4YA-+azlnJLF~6Z3>Z{ispwxYY5y0 zvgO{qq<|Euf6(e^W5if_EGVu_$)!x$h^uC7uesNIb{jY} zOScB`s)!UYR(%S2{W=zT2*eeV*w+<}S}GdUdUS~c^q{O_RAA*8nO{*gI4`}TL0D5U z+qYOTJnr<|=JMPI$qU?@tW{6%>6FcB2;}d?+!0RclKthy*27(?2&t9>IjMeDI-7~7y>F0$gpL0Mn-S21PH zwhy`}*n-kflG>o`nYe}9x()>rQy@Go){efL(`<@)!5S<0j+wi`R=tFnfz3=iRo?TE zICBIc(9izi!I_y`yc6&4BF|pQ4*Xx1TQt{d>+1SVO4ukIH5rtPv{_u^ej>>o`^oBo z5Mg9wG~WdG%)nWLPadx5b!!r0YeK)w$mM(GGW3X2(-Sa4c3!VU$=RO1P*;Or&Ct9{ zFUr$4%bN;?HFGBi3##a87&t(4Yqg{l>kAt(BPjl)1550SLNVmMSt_t?CoC~Th}hDx z3!<0*w!opMAIbLSW9FJ9wqADaNV%Hs?v&fMu)klk!}Vx(YvQg;J=6*mHk-8%0@$n{ z3>F~Oh)Tfi5bL-c?!iO#`Z(D~=qvnpzIom#zy*W)qp=Xe3vx`Z0kvEbCxwWY?3b`3R8r>51WxsrH8Zb$03rvkR zOBd09Qj$>7y}zP^Is)V1Up8Y#@{aj`9ARo;#YtcE_#bfX?cH4He?nX9PS4A-TMhLU zR8v#yn;Y{N-CV;mTX02yZ-)4Ou;E1c@S&W}G)_9{^ac9Dn#Ct=7lVsP=dta8QKgI7`6i*){n!9A1&Tk2H>jgpV)Kt(G&!=R_2D@ z;yUpCH7?7p@0Gz6K4P)&ge3!ewE_bpq^#i7aj}sw=Zok8*E=K#`p9wrl*Lko1J(BA zcO29}OmqUzi_zkk5X0?Vlc*u`R?B z3W#Gkr0J zxmki?dr9=P+0M46U7JG-8-xLvSo+zXsAifMb2WdGxFJCjlF7AhzM-+gIBb-f1X5V! zov^Mh@Z9~fs*bjt4En(^yiKVWC}D$ry+~fSB7@wnSe#X!3{Tt%IFktz!)?wTAlp5r zJ9WzzgAQ75-R{lvzDrGfPJ3oOjXlf0us1uq%*>;ix|j?LZl8M8AZ1R2uL4L?7ajF3wn)rP2W zdd?OOAAw;P*on9Ohg|T>gUM25Hae;5goIh}UU8N%kbBp0%#f4i0v*J^z?$%(DqAAV zx4`{w3e!tCJoQQ8xj?GlDL;iIq7Y|ez#Zqm=c8#PrLtYu__@m>_BrLtX)OpFWE)!w)|+V77n$cxKixV{8d1wA4&mqDn5 z4B{12-|M|^%EB@Z(Y_cj7e;i!bHq2x8oMOmT-P==o_IZtcYOogeeats<5Z3&+fDD) zM>4*_s36c6$)W0f>xH8C!k(d20KdaNa& z`CxKlplfvRsx|I&d*-#k9hxK_U(><{I~fQ%6_w>VTwoXw@Q5w>oLu0)##kq0p|ddWHUD7; z8*7_tk~p|V3Vyf7yg3~#p~;gB;tlYG>P>kn zA3*CI;uuXx0%JzO_+hlw>Fd&Bs|=6(alj=hm5{*i(zv0WY+JQiAX~XaGo#^R!(JXuJ6iF}et0>0?)bmUN%>3y;?GE8oB zZo!tv`L(;pxIrh$DLIsTYVTpo;h7({7h9^D41hx4@%$SY`B)uok$)#Vq) zr`tZsc4~hF3icJYrxN|tU^oeo1FOjJ$uIY=&ZdeH*p7a%Mi?k8 z6Pk&jZ%^4TKnRRflh0S6Lt$Yvbk!$+`mX)db@N#5b2w2T{B9`7uejRT!h>Xac*CQu z(gB$Rax}C})9hlGE<1ZqX7zX2SgtevqxIFi>Z&u*l^zHN_4Qw#*~89Gpe=gBf(ke1 ziLca~FG1|+#uVTWkS$z9OXZp?SiU0(}z z{BA~)u4|k%H8s_CB%wPV8~-9tfr13F{_wrrkC^YB8LR8<?U0QF!YxAt;WtA>m$=Mf~Tx54Rr@uZWf%riVT zO;l9YsszjDrVYgWKMFl#Bcm1M6|qy~m4pqJVkf{TK85GcpKBqG&po8tYcD?Mf{t4W zaq+$xGY1K0LT)!{GP3;fMV@BFi_M+Js@VpdUx$KAlYhN=P|oUWyny$%1OrmF5?5E` zd(PVK7LJKQA2kSK-B5b^MDP1BWp~VtUS9y>9(fc);kl5qdZi*(;3o|Y4P|xt&ze~9 z{SdFmxc~eDDj@>5>p2BJWPCy)mLdKTBwn@N@qh=#<*p{HDyfIb{G_M8vWtzOf7yx< z!OB0$f)|Yas$iWo5DsuropFX(E%Aw-8L?qtU{o4@pQ~GEfV`(EZE|V%+{r@UpH2p- z`p4q@^nh+0{_u_QVG~RYIWP&KzrPR{4^LlTKRO0$-;9=)_Pct*jk<;$BIH~Om&x}R zaqQ7MN*Yp-e7i`WSelz_nrrevWb+lU)|DQYfV@{wm|SmtO;K@zdGzgxO@&K?no=wU zaFQ8&PAX1j9v)wW^jZAhrDtU5S(+M~>(96tU(Q7kB;lCfZ@T%gQpcG#$xtmlOkiGgMC@So-AZc(z~PhXj51yrwA1%d5($P?J_N=*=)* zL!fxfnn;lP^MaN6BNt^W_`%B#1%=nev{LR34Y;^FH4P)eoVQYcZ0)eS14bv+&Bza8V&?LGUj%!fT3L&PX+1=A4mE#lX zY(@4^R#6booS>$nQc+DfJKKLJj)c(}jGW-?>|b4dUQo~%fY{*S-t^<;c~GLIlOBG= z*vtr`h!Z#^P}!b`oG5&VU)*H!^sg*rcw#~g)#~&j(>HMF#6O$-^WYIWHoCfsf+J%e|%BEk2w_#I)~9%JLM{B{Vx?&kknJTP13vk9cVV3Ar!u z(4r{=6?k+Y6AVJHA!w77bnk2SqWpYn8GJsyy*&N_K3d5LY#^oiuP#6+Zfnq`rJWhZ z5(eZax6ZUreU#vk*c}gisra6L;SmIhtDn@<(jxzMW(Le#fWC0Ki|zc%qE(ok&1+_L zn40q)Lxq)w>+q_#pFykHkfpM0Tvj=m#eEMPa{rF7*Z#neBJ#REfi%0im@k$w({NdC zrs#gK?sc2N-!BFcitg9>Z%OS2l|%^KeCLPN8@R!Bbte~4Sws} zX@<)nw7PT~c-o0??HZS=aE^^Flsn(9KB zgBpeTiB5~nNkum~*)P`H!Sc!}t!^R(Uz@l+VBfa9wPFWMYD@dOWwercT?rtP-_M7V zXK9*JD{W8MEI4{<7-T-_eY&AtTUOh zR~LF}X|0$N2eu!3UP+!6hXe-?Bo3e$e=aO;wA!}da^Md%98B|!nqvd+2MG%L&gZZq zCubupv0|s{yq-?Fsa!JiMQiD(n&obCfl}VE*ahQCEe(jt3JGQ7(7zyLJ)C#tG~X!L z{FP42@9_~#-vQ>jM(0$^jx)1T4Uh0Go9o@|hE@lJ7~~a`O-qY`-N;mn)1q3H9*2X? z-3f&N3^cG!L2Qm^wN>1LZ=u#OG0`}}k|w~aTr1#1dIo4Tstq`a~gW=mCJ95O*=C^{HZ$xhaWAA zeE1#GgDv5unNVc$ytBuEfFosa!`#)`8z7Jl`2@z;IXf$!-Y}~X==42-c(}WX;xOal z793ch9xjSJY4i2k9Lo*Gc8_($c}0n1sQ&f&8aOq&!YD?(F>>rBgAN!gD;?c^!l_ zg{DTpQZJ)h>!6oE?uRxn-%q7`23J*8DYUjMfRJW|Oi% zr~%>&JKs0ZHK$-;P*zhqzUiEE)cF8u`&tS1@5YvPc2U-JOVlI4?cw6$K0|7Pq@@v0 z%}c$D>QrlXaFnw_Jf3R%1ZJS>&`YzPoY-`duQNacpiMSmL0L|>lb->EqB_=nf}JZz z6^c69A-@GL`+4Wl;&y&Y#de4g+&A(;qg+HosAR%zMJllEKGe`JC>ETGpJv%+v;WW_meDMO4O3Vc6aXgkb z6<4|-BQbQhJN&lE0|cgOkNxqpZfSpVt2L`5UIt+(e}v7wyLtajG4`%H=h%vnmo^++knUtkEMy| z?OXq>#4vaj73n1rxBDTmo(SB}37KrXs$<^9>x^EdLu6G{1l$^@MinSynMKZ$-eJ6r zu{!GzdRJUjRA)YBL)jfB@YKoCKAF>4(AfC8xKZZ1kAf_RfYT;LlxtSqe6jAkG66WB zcq|!9$5C;?hhAYRO?!(2nIPLI*xT&cLV%Q&v(8oOvQT-esKh7Fzm?NSQWlNMsoqPg!^t@tsnk#8HV-4)pGhaMzwWYia z$2^)tuuk2X?Q~vk=xox85_!2VL2WoOQRVu|Cr1+hPVChOpBr|K*7I=Ypkph{zYRyf zv$>9g^Vj~?=#PD*5<7TWl9i>cW3e{Xy}7CH8Lh1FsBPwytRkASmx&3rzCIRkA|TRoahYv$7A^1Y8&z-;q+ia;(MKoUui;ZVnp~Zj@aKB&-x>b# zyQA{dk6n+AA59q<8Eb1sW0fLP6Q+)1Z@>S}h$Ow&`0L8pa!4dd2Hb1dy=CFcuuCQ} zv4V=ojOv(~;a|-(plQ0m~%gd({tc>xrkUE!kt+a64vyRSDxsv+Pc~#6DsXMf+vETDOzvuG| z9RY@G2D;CG@BZThKsY%YCWE8AhX8^~hX|*+f5lF1#07M3;Y+4-^Rkv5~kT?aV#7)fK0bD$IQZ<$2ybVBd?T&#y`nC%RR?0AZR$L7%<+1!%Ez!PEOl9iipWk zg}G{b#Z%VH%TqckI!d4N^-Y{>K9>()jg~H|$3}lUb#KHdP~SIhW<6_9K?>|5&wBi6 z__Wqd07S?8UhT{C_4j92!4hkc!C(c})5qS%mGTmg&Z^-iveA+&;$wV%vOt#@%b?cq z45DIo;Td;ewpbR;=BpmXR24aTvbX29`00|RYIZI9C1hAunP9jSUjL#N@>kP|_W$to zm0?vzOWPY!>5`IekXAstyOCB>KtQ^C1B!%5cb9ZGn?~vG?(Xj9Tb}csmuvq)xbTU! zX3g9&vy@3oe_2~M?P6P6T{WI06-U7{Nj;jBL~tUj(db&s!<5W*^CyEHvP^aiy*w{x zV31v(t*Wx5eCPUEDFLtM8KhVC7Zjc?pzUUB+?bM?iW&G)Zevqn00Vr!h~GaD3_ia5 z4f*M_(!+BK-pQZ?&oCV|9MGG_K0(|)N$D|PA$)jB~Z;0y^OQNFb@xV6l9t$A4 zc0PozA9NP39;;c`81@7g^I@#12a_M<{H`#c?>8n!G0=RTB9#5lb zrr(Uz5~W^%zs>xOi`-3dzXJNf;y0jasj#QzonO7?{D+7>m;g!%A)QH+26!A4P2txsMRJ5WPpPUTWt=c3R8nsEK@2)Vle7!(3nQ(Mj&JoUe7zl@UtqF+UZ9x09=XeVH=p3~_b!vqi7dRz`L8EDTB~0|`jZ7EwX|eEKY%FNEO-c-u>`>;-g7Kg zgHjV2S?4V7itH{q_-8Pkqn(SU2L@M3$0h%eVRlHM+=hd5d<$HuNARkz{T3uZ66*KkB{=H2^RzzE-Cl zMknz(KtV#9uBmj$h!ixSIMltVeYbz3J^J_0{?3|KfwyqWA%GcN&8Vasl$AB=oi~>^ zmtto>CB$i}ojl~>^zP>yh3(Hd+IG2Dbf;-g>2NpOEc|-0Tfi_7RLe7G^JujbGvXY<9IJp*er}ot!SF zYy^_Iod{N1i9ogskyk(s6xZ`o)~%c#OiK$NenTmeAzd|bxUL}0eYNpPB0nfMDm)o5 zBK|MuW%fqir7R2NwCKZPHI&~(3koy^I)ZY1Ny|Pk7!3W&L4te6NU?trudSJ2pN`O&AaWHr(DQYc)==1Y>eq&>$ z(X`LpNKBwv-TUApm)APE12&aj5)TONpe{~b;0Nt>`nCe6+Byuj{Pi38lhqRmKv`$+ zRJ8yzu2vp9WA+o}Y4$$*Jrh;^f+s|Z7rkCX>^}g;ujuHkpAG}@@|$gRf9JUjG!Yhm zqELghwVpRo+;;p9BNdHowB{(RGAGRQ)%CACaMzelHQ z*W0tI+Oy8D+tuthNg%`Kdm5#44_k-TqJmo#h^Hwcx$FQ^QXQ)}Jqy|eh%zBzPnc^+ zNXVC(-2%dySb+&B>2ISt+V`=Wv;$&yRVp%B#789m9&8g1Lih__YTX>FT1Bry@ zTB7=qB>T*)`VANmke=N880ystyn7cS4Nx3p7^ahRg};LW7zi)0Kr%P0xT5$yIvU6I z2nWqEa?o*0{W^~#WrGn7wUJDv!}E57&D zp{wWq-f(dP#E(+3RygvObl1~heSfsWvC4+rY(JSo+|Evi3075XvBi}V5;gUSMr0%< zg@OJ))`H^VQk$>OugQZWC_g#9Kfq2=lk~5L+PS+%L`C7^<5RWsv>vdt(woii7*1;B zw8BQsUxF0)yKDyP?%djOKFazl31EZOcu%&#%`pK2{43Sb9va2V$={Z8qN1(i-Q2T&oN zD*UT{__wQs^_q}MEl;H<=SvB{$H#y47FrrJErl0`8k(DESSRs6GqcDKBi?k`s276x z`btVl&>g?HvR>1Kcd}x>Xl`nP`}X&9Ge%gQ)7D8L*L>ZsC8ufM<&$om?T4SLC_saR z@wKX6)wFcM19SY|LearoqWdhLo}QlWRSv(yFA3l9Q8;p5TC?5$81cmW97%J{_Q+A% z0F(FpOS1s1?)fl1uru2+15{Ht_MQ(KI_Iyq_riBZE0Fala;uX0bQ8IX{Dgy11?9Jx zEI50#+8a{KFwpo}riQBteX5YiY`7L^SH2H^#=(!H$u)VUhm8^>yUFwsX4*w)N(VRO zw4uOB= z=K%D!T}!;A`^av*mA>&@|KwIC z9VbkknjL1CzXqebtL7e961HR!^QBSFD?4 zh=|+D_nk})Tr3{^D218jiApp-*VPx|f)v#QJwKm#Y*7TLz$6z(iB`kj`AN#{z-q;C z;%|l;-e62OM2g&-Ym={efGC%3jBG&(qVP>VMl9Kq97UX4m{@tO^XwZyp`$u6UVIxJ zwQuFb0t5{&!+I6YS^%Hzd$c1ayr5|;<*$}m-cXdx`#O&U{eVG!L{~*~A;6lzj(&p^ z!Y3*nV6Fh|Q>MB-QP8Bc!36lh!CXUs(R_A`B{SnwdL{tr@(URH`g&w#RRX`pLBgZ^ z(YV|9T1p(S>?jk3cX#UqSjQ6K4)I~b65&TLvD?BnT{JX+u#R{IpDnHOJhH-GnKL>( zIyOoZ0zolR%u>mpK1;*Fu@5IpWcEOpYHx@C1Bl8l)V(kg;31h3`{ZWaJ=dKxWk1>P z))J_~uxxkS$b+qoEXIIoOw$1p>*U)jLI_Y2w?IiI2j+^gU9vR8O7hJ%=h4T~YB*!sZqma~nqWcSUnk_p$hyioEi;n(kf*^R`?2MDV<@0a;I>$sR`Oh2 zpe5Ci@;|l|L){C+$MlsQ9chD-q?XsRlL~eJWPXMqqJ)8R`Z=T#BaA{LB8L+YF@TW) z&pt0+{N)dsY!zt-VsP48SX3(U7dT)8)+iRQlQP!adV`9Q>S3Z_WH zJ#VR}yt4AnQ)nuN>$%}l$&zadg&c>I)?E%0&@U}LaSP{h189xP|L*2u$Hds6)uK;& zxN_+WZPdj?YQm_Bitn$^-0=6g6R6+_|LI>}0E8f>Z@W*jihipAMUynvXm8ZFDk}WJ znLZfYpYiF>!esjUmr7olT`a<0Uw+BTB5{SrsK3_66$vj5`XGpT!Q9>;jj8_MQUllbN@V2)zj6{U`m+@Ew(T-E7IXC%&vAO^XMecsGbbq zm<_1gg0k4HZ*2TrEl$K7ab$Ti7HDwbphB7GqOmsep|r+A#uo!y2|q8`YYAfGnvW>H}2J|(jg!>1Z$&QPPMMH~<`p0{yo&=$I zkWqGfQX$Wcxq(dVD+z76AFvcxTdZ|He?$Z{2#aD3v#0&DZ@+(!mKMy+YV88f8WJ9x zIyPAWauxA-;!_80ZIIv-zw0k8&aQAk*6Ir%8tV>geFZ`k4qgmI*x9-HSeDJrSr%Rl zAC&a1{bok@WP+vn`7geCE9#%@tx?g?^zfN)tmFM;GvC<2kHC9H=<_=%+=R=Kmx`8C zMoz9NuND}-qkr; z6=#FLe>?;+H}0kr*H!NjR_YQ~l9Ew|`}YFG9Szqadwz0bccQtuUO5)FfD9w>zEq5o zfuUfwz5`&#K&l^j9yq4S%eBgJNLZpG87Vn}BB&Et-i;hu8d~(GnNs^`p1aWner0X*xqC*R)p?&@F z=O`@Yc`6!OS`#bd+ZT(Pd^?y_-57q2jZ9b}{;wIW7v0($VxtKvD$aV{d`da!_}LD{ z=tU;KgSY$gC9n9crz=zbaIJUIFn|-n-6|^20?T*@Dm{^Pk}ZFVYq>kGsq9XyqFSOZxXe3~Do*(o*y z(h>D-?)Osd0$)W_e#`m{fMoUtV&XTA?G%XxlM4}MN}+#HTj%Y(J{r|s9Ht^YoPYOy zU~~S=n>H)5;p$nSFUSC@vp>^ME%@=J{tNq_Id$Ir`>zRh^Yil|!lI(PKy}nEYE0u7 zDurSZW4%bWKhCOCe->C{Xkz|`j!rBK$utSdg$SvI9*Afc8oz(>#_+9HvSEiwrOkrd zeN0-2K!Zc>t^;>+-_#*2nwgU&LGz^;X4a3d7Usu~xGd{BbtJ?5yS}zZ<^f8&&mpEB z7B)(n`v*ht^%{3IJ>9(uihU3Wn$vqAwB96Ot2~QT84IjRsrG9w`8H&lVK9QLt`7NZ zVQi32cYUX2C-dnz>B>Oa4h)b6!cYkbH+TUORt!^pRu_isj>M3psoO;wT1=;zB2Row;4ROI$IFRSfhz4&7Aj>!_7@RpDe zW#wg9>6znVW7&qs4p$Cl+PZ!y5tvkb1e4ugejClXk;>osgaLok;&FR%zH)XP9~1R- z-jM`68(>x|;ACVNB$P$;q_oF+1`C23qWlT#lY}v|uz?Plx}`bc^f&9C z27&Dv;2w2V48$_%0aUimiz{nS>c}B@=^1~Eskva`112aZwao5{D`1`Q@P3r)qJH)V zKs5h_n16np?40eN=wRWeM28GMxbzhURaUDrmNz8G)B24fu^PseE)wz9#>M8ipQ&%L z&eJLc`J?X+z-kI}D{Eq3uIL`L?=S)$qz9w9F}cAu2)aea)3}7slgYaWwsn|)QSiBM zxmajg<8ufut~%Edf8yKNJKODDOVclTQm?(!8f**o$;iHkIb4xG@1V0^*heK8mjlWQ zTQN{nf?t36Eo`d#II2+g0%4LL7hIN+q?jEGyCM1U;ep605tsT*mL<8Kfdhm z4!uO?S>gl(r7d9l2oNjgkh{3h_d~Y3EqhmF0#u{eQW$zS?B1`{Ig=jtas+(we?}p| z-s9Vwf322M)NS%sn6&FoRar3 z)I0r0$VPLq1(3}3Y&i0!Y;#3|LtV(LQ;4YC6vmZt@ zYabrV@GJZ4V7{x=MQ_aU>cT$2IW@TR%E`&SZ|P`s$96vUHbt`8fqb^inyMRd!F+x* z4unHLS6SpGB@?gztZ$l=o%8#?{yCW3AZRrEdeKrIZ*P5vamq#zM~{f-sKH;uW2d(X zY0bat{6_CQxwzuX%J@wX7oZtl{-`^TW8U-D9v{&DcAOlO;&Qh*MuU=j!lKjodA2ca zYHSCU_eZ(o#-w`R);Ulv9K#!XiJSn=7))CiIvXlcSVVv3rovP5%VjZqTmJ=Mf zepUrl78eHTp22xxzQb6}<2&Tli2Cv7{v4C=UbIyjse>p~|Q)zs6&gM4Ii z-oDFgL4Qa}(#*@weyo-l5ev7$Cn*jScONY_b|_hL`>6bB<6uWc`BO}I@}q&8C6D~z zN)$dMJt!$EsglI`!W7;w*@g(%Ykw=k(mRxD{Pj)C6?2ATvENLs{Mv zdlo8Yh5B9^`Zo-p8r|h2Cj)Ls*g$1!&o(i^!TSRa0=A}{_uS=OfgdoWSzD|rB{#UQ zk4v1EMrUOe{CZVfQ{u0ot*2aVyFvCeS1EEn3kCw#MVASDDs(wDchDk>`P-&gMt%m4{P4}Ai7qEOLLzq%1{ zL66gyTuB1}|5>WKSO%NGvr%V29&Ka8n23k~r~rXZuKa`^$UXkB_taB?#O{mnmg$#G z#gsbp4^O0up_4$n2GD1vA9_;aol)>drHq@1OoDmdVO$hgzPos7@w~>jtl%0CESUi= za*87R?1x^|)ia>W^{*C!aS%MIjs*t17V+XpB-r4mWv=nKc1I_e4m#dQiqd}TdATdt z)R;_cwZ35tQ~~;pP$_gO;2#Bgxqy1E6#t`od;mZVrr0*sn&VQK$tA3M@dB^4gqtl0z{9 zl%fMWC^25p(}O97v%p$ICsHbNV6gsHNYHp;y#4dU>5R4uDeyD?C^s4!L5VwZ20iw_ zrqhAMtFXAQw>K#G7oGpSnX2Sz?7`gJ)_Ou=dvKlw7rfzWwa}GP<=@Mf924hj48mn*xXd-LG`Z zFme*{J0pr*nxJI=&d$m>Yq&r2X`N3VJ2`Je6~D;0Cb>lXCl>}XTcm$y8BoSMbK(Jc zGz60nZ-8$7i^&#loL8;v7+a-WALF>Xkoz+aB0L3ZofONP!Sh+5jA)ad3P!5gp`JTxFeR=W?1(}a;~21cY)k_FK?w)j; z_<%8(vL(3c{0Czf-13!`x1U4@+@GqGHo9QaeobbxP=}iwp^R1OpFu;(Z0_x!g${B8 z*wF)`lJw67**161xK<6XlfOfC`2$0X@=I02BMywW$iM3kwP2l|26ew)4!zVU=4Dyy z_&XTs7;{E_JQ3+P=Iiktvm4+-aYfUk=A3SBJcgGvCvhDnbTp>N&248HrMXQ&eFs#V zz^8EPNkv7~+tJhII&%Oa!r*At4Rv!@E6_c~V?Dj-{(*UKf?MYn-HjqKI!P}n++2|(kCW$TP}`o znZ1nYf$c$e&b#_8*P{=eIvuIu^P7?7_q}ztH;3H@WWn}8(MUzh;(X;r4)jK-Ko4@C zB*B{bf!gcoV2iXlwH3CyY4{WoP9>c&GX8;mc#5J53K^ap9UhywwR~7!n7B9+;woCI zYAI@b1^Fe{>l+|uVHT~Z*aI8~micNJ`a>OLefQ7dZ=FCUK3lQo?W?P(;k;Of;`Wpx_f2UXoAT}L{!B>u2Pz^)XKsg9u6}re)X0dDx%lK&XUQEFn4rQ^xL^#yH3x2 z{4C_Hc-(Nt6_(35WgQp&MOw72POhlGPO#Yx&9T-kQ8wZtoEZ4_J^=XVi_()#*Y8hJ zNI*JAO-;?OOQJ+NFUhTL^dH+ge$#*@XHvh7VV3u0cgdmN;X1Ju3j3N6?2*&ncB$f* z>?*c!VV1E8Fo&e@)qCt-ulR1H$H&J5CtTW`*WT`-IJ!I;XRE=gr}2AO>x7OBAGK&`E`;hfZ3p)S4p8?@>O}X9TnoOA_YPPe5vWITkLjhp7`c$ zbO>SIT~XW86n420r2)Wc(_IA^cFq`FY;3Ba@ySUc40J0gc%763{sj{4fk_zbD#b!d zlO^q7Z>=PqoQ^s-CcCF1EjLY|4nZA&nj*H&6+G@=D~qcknZiK%Cgi1G5A6xoU@R}^ zx450ALU`JaE~%+3=XQby1f+|AE*K~MfeT>siT+Z}kN)?6xwWwg#>qa9 zRNDnImZ7v6y2i*M;jMq`H6uJGg0|QeF_YLn#@F|HSAX+x@v-oaZ7{$jK@kH~rgOP* znuXdc>ee{88`_axZVI;ZIzli;QUs5}Ak(X%l=mWHO&!BEfMZ?jLKXCc#*KEhU9-xW z-n!QH_ViS#4bS&~1xg?Q9*#tw7!Dokf?IeCl}_7i{)q&{A9(#KS>sGP3B@5dLQ*w6 ze6t*^zcWheElU1nPA*X<*jcUZ%@zh zmo$lui))j6EyDXuLez&V-bIWmC0yCfENcvC4qiYib3@=E3TW>atDpHL50`y4RJ`{| z_!@dP6Grd+7Z#dZxEAQ33AgO|1y%oENu~P}hsP-(umWk|@0{t_hj)VWro|}_PV9>6 zK>nI^Rfpl$QmXaU%79^eA*|l*U^D}JbBc?DnmSjalbV}v`se5-aPI-PY~XpPzPE-N zCZ@L2#_A_vi}Dp;T{Rj@?6|GJhx2Oh=-3*He|;*53^_#!IztHpew?|Hp1S{bG$?z$ z06=F~QA&LbB{1p5@kL0@LuWQs*dg;a*YjeBW*j;1N3)$Fq3T2{P(3)9gxXwb0LjB! zo6G0Te!wH9|FY(<=Oqv-|B6(QkT7c>Zl?0T+_2@YbMpfl_7^W`-dAaXdXS2WhLx@` zH`mGl3~&7HD@7*K%nV}ZUU~zA97@z$04yilw5y#qCPq^wALqR~E<3vz+hSm7jY;8J zWMaW-uvw<5Jki4Lrg(qd=fkWQ4%*%;;}awI4={dNvUsQMzQu&?(yS%qruK5 znw*4L%Nv(BBT6ry-!$(WZOttW!4xEN6bausjsA?Cba!#OINa6x@R1ld<)en)C;g@L zyf*+MUD>siYG=jT$@C3~dV_!3Yh0Q5)yZO(GfG86!(?_HYm7FmyLT{1 zNxHD8AZNrJc$<8Dl2c1!>+&<)Q3!{+R~{tMkGfWDfe;d4cgR7qU#r|90M!kS&8e@B z$$v8`0T8@Gjs8zHu9Co@0qzEPJc2p^EjTeTQdPmR5hODhR37n|0Z&v}arWCNQns%b z$bKf&VnTw?iT9a$KJtlR48hc(ljr4tSH~TFyFws-WJ+PR+@OriL#H}g@zb{=(&rg6 z58|jnRrR===X>0^SVnrUb0+t*#kJ8>WEWEpd;}uM@GFsS9_a2f$obxc!e&8R}gLKia&%M|nUdN^Y zU}<_ORXflWk_*uSTORlIA;Zi){J@mB=!vop~Of;<8G7>qRClt$)Q zd^sSNt)fe873laT%%vLGPj(3jA+Gg2aNh!g;2{8fUth-!cav7KIAlRzS3~zX&f}(B z8S7ARg-_nEP0e3>r7CXeV5y2I!m4CSF@xA*_4|*hVtYUdx{I^C`uk$f6w5`9{)5Lm z@h|evQk}ZF&?vUjE8hbZ{(nE~j1>HoER<}i(s#GDMgjj+3pcL6G18VYldi~s)sAmY z8=EXdQ0QQsYH>O(3e`Dsf3Y z(Xm3vIhbaF)9(|h7X@`W{m-v%9SCF=3Mb{Y^ptuU_FA-8r!1v7<$vqkbXv@FI$LSY zkoEgemW0~^u+x_FOK6^9+vzi;Dlw&Zi;C302&Exn%5<}%FE8xUXVw`#zkH1jiCV(FfY1IxDDhD3q8#k>dUqM z7p=sm#C0EZb=BqNgU8MvU^OMp%^4B*MTD=3Bg2ywGd^sd;)n>+y;+B1gtN(uJ3dPp zByZh|phTl6bhcY_Ck6Iy0)Zl-e*%=%sL|T@wI_Epg_gg(agRB;f%*Q;K>CQeL{Fz- zIl#7lRz;?7b5pa8s^B2|fkqozxzW+l;nmPi$%Ay5%@oMN({7J=Pq+lw=>*ufmV&t8 z{*3n5284AD!!=zvR(VxnhYhSmzj?ctH4PayKByqOe74_YHUOJ*6&=jCundoh>@3F- z!StnW$~zXd#e@VOpI2~kbpGq7aP!E0w=d&t78sd4KV7m=rLI;E%zaU(f=rwHH%X+V zp$ne;z|0D?@?dXRn12Oy>yVzWp!;d0BYruw2lW#HR7gTAD|NB$d^FX+U*m5Tu<+8&4 z$)oTRisl+DMaBWEr>I{c8fT-hz%0bkT}|gVSnj|B$NvrZAQ>3xA>TkS&j9nPtp7Ir z{0cByTG~rN%Gd!Dc<=*4vrMsnRZN7^_2=@L-(QTp%l6>jooGQ28EQ9izrP^pcHAKkwa^)PY*}j)y!aTq_vF`D^?Du_tC7t-~=G;`%L5j)MPMAHq z4|GL>Pz>WG=-Z9IPI2(@b8eZTD|{@g#B)VTUK_w{*C~PMI-!fuPC>Dti5il zTlg-%f9y#Y;oA&oZiG8Bs`ez^O}~ZA6Xv6rWAa5#Ms_fnjgaqB?z`yV)wkN(+J?%u zF?<`asrYBxIw*Kq>M|nG$?oLO9G&`sv$r6JCfnqrcKqp$-kJnpKmV_VN=rjnz?%VJ zz@T4$6(brE84V~AAidD0qo(`OLw&i*_;;dbaH0pGGeiVYn7B!=@o@3e;(Nc8|Llrj zgx#MZ|3W|1d0I<$o%}URKqHa20YyM{z6gJe;E)FUZY5ex6N>>|#9etMQdLTM@ZV}n z{W;LH`9SoV2=_IS-@k{$dwx*`LTqyg52 zVLurd6lga7A^jy}x%Xs2XdL(IF1xb*IVJ>gzUpkxa(&VQAKoZaYk;%Kbg8Gg$!FEj zmRtff9eCEu)2-?AqlrfU8E_Rn+W?8aCRkgctKOhAWzl@T39u_ZPWqC%KNESfB%FKxH z4IxY)*h2e&1C|4;sSYXw?8J)X%r+n-fr8gp8USp-LEO!M$Z{#(UNBs?{oS*{^vlMp zzCdzbD5%ZwVr&(lCW?n&GI>0RP(-`k$OyPRRf5EXKYp3dJ*q<&pn22XhekBrl^Zm7 z{RuQpDTI)7Q}1_S*@xSt>=NG5+!aP#svx$Ue2|842T2{KAo^7D97U-?8nIkFS$)B_}bvsR@{mG5y zN6-D%J}$5rI={hcfxl7tX@n0Y18p(P&GO(NNdxF!XM!vgFCfYLMPIVAyZN6?kDAw# z-aJ7R0+q zVT1*bp{y365&=UgUPPMcp}Ps8RO-0Bg)12rmn%>I>KJ2`BNNe|XZWmSd~SN~75M>4 znPN0sIl8ETWE97zLG%v3pXJddr-|^b| z(wS3)MAk!4Mi(89gsWV$#oNT2J3~{bf*bMyL50_}_ewu6?9WTcSLsW?=V4E$Q7^oU z<%W7_2_eVn;iEilctZH~1j>b3^DULoMGIGi9Pvol!gIB4&^43YTQ_-8 zt?D5REKW)?sKqE1WB?sb{%2#ezfxHB!Ft?(+LklX_Jn53Qn}|XxUg4qp-&x&W;U3} zkXg3JKG3XHaR&btl!o@U5=j;D#j!0ZoH?7$4g8m>4u{h>Cum90gY--+t48~$OQf8I zlh!{2<-Sg~H`xrO%f_*CRcTp5e{<}QRjx^DG1P9krHxSKu7u~-M@TYbi#l5Bd-Yd^ ze#ebPveey~l~<70OE|4k12VVM{uweEu>f^7U^-;sKA&*GTMR8W%T&WQTGPxCq>52W zMN8$m%dGXKzL~i9@gdVFeKMRG>Xekcs;BxAw3({1vfbkV9(oK*RAl1gDbZPI3U8@O zK7;P^>BR95DJ8j$KM-%hU&B};f?m5AJAUvSmMb(z4~1)n10@m_BYx z|G7tOo>wRi>`!=$8?k2i!=uB)9}We7=jKjjSQQp%a2wh#^=Y;}%~q%)+TEY~WvuDF zo`y*U)D$NZAB2PiS2)-ku0@sTwYaUf*R`+XP_*xr$`{;BhLX-co!7}VT#{jl67xBE zO-HfTSgvfWe@;t@B6{U4z$&DVc(Qp;ug`MF!$+rynBi@c)3~>gd6uhR{Dz64cd&=# zqMxz&A#rYIcAF@gv~P1L)!@t5kCd{Csc3?3oYRAo^-^)@#n2=!XPYniWx&pEcOpYK zs54#f;H|7&j@h~}VJw_>2*t^ua=(KK+A+%`uBMG-M(%RN2AX|d|Xcv^E_<~gVF61ToGJ|VHnddJ~$)p#ev#zBcu&d=Ye z`C<1A{j#P{qZnUw9UsT~(dB5<`^wzn2aq5ehq0$43L))V86MGKJJv0II{bgYh^rn}8m~gXEQuU5cjLo2Urigy2 z59LT&BXkV-{{3%5n2-$dw}9dOk+s_wRW#^Mcxka(Q6nK)Ze>5fuT~_avKAIzzPH0BJzsFMd0?GD$283? zgE}2|FBZ>+chhrKC1M(ms)iA%2|ywvA|g_Xoy=!b3=)z_4#1hWrY?>n8F@GzHJ(njpp){bf1D{}2G2;Bw}xoGCAXna?m!Uw z@@1vIhuwB&6W2rK*1lhO);b3jm6)kiKzgW^B-+7Je_K;o*_)R14_Y4$47|d8IPIl{ z^vQ-o=A01%^g@D^`Um=0xqK62MBiEMj&&)fv^d^qWn-1qsGyQed+hp_qv9cPybTh( zqj>jD0&%2LDmS9-77KF#j>mERJcqf#4^vXoKM22BqIS5fbT{{$qqUDp4 zzIT=z+|p+1Q9m@JKoh9cX+y~b?C!UJudW^^MJU6JqEtsuPf>z{gDs1g1d=!}FHShv z)(f9=2>*>M5tr&hqDXYD#R@Tpe}$jsxZ5v5;TwS!AkOhI%l2F1$BVutZ|i9m-{UKb z$DOW9L_|`iU3%iLx;oHRhoKf-_o9!@d37$ThR=i04odJO z+{i=(gts@YH#e>)1`x=yffzDEUfun@Pk+B9g}5^(3BfTrg`dL`E)E{}LZU!LMeQGu zM?#!H9)C_xj0V0XCOKHcj`YO@4^s@`vgDFkG4#^@remuUOMO9v5`mY2g<>a6VJ_P~ z4Ybn%?ZU?~@U{7rOyiP}PO`TsvTQ#XSxj2bHax{+wl0<4B$X-{PeD>93d#d+#Z)l<%RP!(tp#2D%vODC@OH++lRm6!R`tx=?NDdG!7&l$_YWZ z+);!$Ap{c7b9r0NF0!N>I}Fj{>eu5s-GyA?nQpBkQ!w=p_zTP?*}{DbJ-a-A{jRDM zvo(B#_GDT6t##|KCca{3rq1V1bbdb)f!7VK1&hYFtfsx9Nsvly#6D9LtE_<=H62#(Ataa_q=>dU`&~@k=K5#q<)?3l7h6AoCLacDFYh z^3#HglEG6(U;k;qEGhtXTLLT@5Vrl23M-Qe+3g^Xz5Ot}0yXg0541J37?=(;=js>Hd9ud_3EGdoBEK{262KsR7C#Ov)*Y*n|?-dU|@+*OXQE_Gl!#Ei0Gw z%yKqf7_N5e^z?L$ovpZ-X<%UB&|bYm)>u+f!@wY;x>`NO^i(Y$pp85;Gd#4kot&H+ z)g<{$fZkwd*ETS)HqIU%8*6BR%g+F7b-W7I&#x{!TY{v9hl`63Y6o8n zT<%|=M#S!>pl&e9!Q87yiFrrNe~ZJ@{rSZXR+bvW+N;RXQT8!5wzD%79K6lY%No4_ z{@1=(f0|+=BB~F=tUNABqBb9}ML#kyxjP=cb#I3(;ptUtO&r8u4`^Qj1JjB|yVZ@1 zhsPXC%4{ep&!1ek%ETgSK|!;#nkLW+r_uQ#c3e*d(ICBasn#y@PtBvJyOY^mXKBGz zfm*?WN7x~s?fw4EFu0S^C0DT@oW&2MJ2Pg}%YqYg-0O;d$Vp5QmDM`YZ6c90)`vC(|+;Ss4?I*Bw+Ug`6JS7GDa{(qxxB zcv5%wh|5cN9%te_kH{RU3rB{ zw|?4~DJv~)wuy z32yLx+9)%P*w`wNk%PbkwJsTCh9Db5GxJMJxwKR$5=)^dO;7<&mUkepQQA&ONZ_8Ht8Zwi zIzC~7(FDrSi~ek^pt4m~jyMnI1&J3*bw;PT`g{8P9Z+5KvL&f6d`M4Ee@n$KhnPWk z4&qgn3ZK(-b#n52@;v{)ry5Zl5;MuD)liKpyeHaLLdNaNcIymvb`DKR36V0Ig7a)J z6Kvpfzc0e1-GOZ&mhynADhfFMB>(XaKl zf`$Ig&E85xpiGY_qBhZ9tkXlO)H-N9uoA{%k`uTp)3zC&D@TK9N)& z0F2G{YwP(a$|L1wI|=ThLz9yQeqpAt^goHo^OHIfh5>{t2 zzhUC{TD{K>w}2U~f}B%M);B+}ys4?!f{IQ^zx3N=^KS8sxq2}KG9bOaEXe-$t*wib zor@C%4rF(%XYB8p&@HV1mE1#5(SEYG3-NsU-f;9*YW(qdcVba)S=sg+nb$&<8?3_8 zg!PlG>}<`!9n5M;2v~R%#?r=y%X9Mbt{cd_=4y>K?aC|W)Mk`T}8dcyMdHg^W6Hq|la?S4?q9Xc_-==sjXM%$HXI~}!m zjbjJ6eXx*E*+6Rs4W)H)$qpv{o4bfiQwGu*Vv>DYC1Q5a)XSW2NJ66H@w`tT_~oKU zF*dmQ_)y74w`Ot~36^xSRUR7FmX(B_)Zb~RmzGZ5MWo~B3eI$Qcqgk1O8TU}Gd=BZCh4M<2mqCYir6`;`ZOm(Y?l}yu3mTOsFoTH_iLJjj=^oGkm zn_7Pnn7A>0zr{9CDY(A5*>v0aEI)tjn$AC;&i?UwFYTZK z9f^Y13Gwko9TjYHtNgof&g_(0jy(YXYc-1A7>~Zr7G0##EjaC4-4 zwP>lEMGAwqgN>J0#kkl!FJAZ^4cx7QHxtt=m(4KK@N`|iwuR?)?XGY*C?&p_deYe% z9bNT{bs;p+`|LpCY-@0TrpbtAuEqGuRKIAv@ictc0u+??l)0n@lG?jEd|Fu@ze*gK z;fL$5ue0Jz`-|gKV)hq3&Nal(`D~gUub#fHAA6~QK5L9yU_(#TJVx*pVUW$j3OYiG z1=#O?&}^LkjRa?AY4Zb++l%a~>M4Whh`~7dRBv7BKQQgG+zz6f9A1lRTPUkKF{%2} z!|~j$(9=lcblq77IHz_^k6@J6a`cR7K#9e@kv$_Pl(M$B_G#x?pO;Rq8<`)bA845r zLlO=LaS0Pghb*svQtZ9WUrm)!Q6;6L9Xcs~b}wIkm@E`J8p-O->vek1)i?Q?h{!^3 z!bf7M=TmYh$+H|mE)BvckyoC4hlMuYJYJ_@v#jqdr)fN0fc4Ix7d)VQ9FyE(XXgy{ zI>in<-_Ba(5Ttl_v%I33RI{I73zzU|=E$o|t@&gsS2He7-?g`6M#McHgov?SAuj6~ zX5Tjq45_f=y);ee>TLJU;)^VQHuegzwIo65<&Vfj4ihKBVEXlhKfoB zpawxMt#)$ot{f>Sjw!B!+u#4CDxJW|bock1oNTkC=<#!~>*_6ru$Zmi4!;Gh7h8mm zf{ONu!IDp1;JbxcmXwuA%gG64OukqjLYr&3-!#p)G&g5U*T1?vxF;drkM8JAg8c?D zAK!~g7P0(k8HtJtM9>zM9U-j-uRfQ)sa^~|7t66G3lA*hn}1}ELGhrNpLladk=Iy-3(7rf=&=~-Kri)VsbKHZ9_*Da1d3=Af` z-x`aK=8*I20K)PIXcG={5U`#62vAHF^pHUWpcV)JM} z&M@NgKCiKNDTFeCGJV`Hs6{hh7EGZ@^E7e1bAHr_{RRG&N^0o-bS`H(eK4kv4`0^iG)IsdSj40>2J3Q@?|~Jr$*q53Amc%W^G$^d}zYnn{6rr?1${ zTN^tDR#s+~meDZ@o)_-zmnjI@7O_P|$yZyR<+oiOxOAzgbx_^KuFk`d)Yt4;ASD9V zAtu%~-`EItJxGq|6m_}1R-q|{|BJ?L8v!wU8*uGRjOB?3aUe*kzW9|0c zH<@kav*Tx1w4`0-1})Z9s72pNfu9(W5#3+sNdeHbuc->Uadf(j5f(zo?J(<2h%`I< zvFY*ObcH)IjUoUpZEScyZYZTOyWf$%$lA;bTFFf#K3N^7d#e0fGkMf_W2mS1o#XT1 zgaub=fE4GIwaVea60{{JJ^cg2kf)1#X==)Zel@Jp&HIuO;0{Ev{mIFf@Mu54`1heH zi@);;yIsFe*8JuC${4wtZ<-z7mVDD-7QA0pNAS>gEC1vC2s|$zn@?SK1P4n5q}-j4 zJ|3dH@vX%quCeUk&9^i%sq+;YZaC#ZO&Ng{Q0Q{p_|4-F1VK1|eD-TR~NUd*D}-R{$vRvQp? z3BCJ&)qMq5lwH*JppSqep$G_wh$tOON+T#KAT6yRUD6%WsFb9%$N&S<3^fcOT|*Du z4MTSge1|9A_5OtSJB!7dS=__i=RW7`v-fpf`|LOPcWBd|7z-t*JjgUHQ{f#`sTiwD z(#wk>XRYqOaAua?Fy+?F0ruP0ur$q+S7$j^650;QcI{z4^U zWj`n@drI^li4}M`EX4@EI_`}cuqSUF9i1(Wn6??(?~XA6t4)eSQBSbK<3mTKf3VVz zi=(2|_qJz$lFdOx_x#y$&np*=xLf@E@uEmS=PeOIFd5ZyzmHTlk%NI zJ{756mJrZ=iS0nw*~D~oD;|EOsL|Sqd{N{RvCJ3|-BDb86CHkcexPQxmA3Y$w~qmr z_r#|vy!H!!#!jUBiCm@HeeTmfd#wjfG~*RRb6x1fynj85gD(X-z_TduYiJO=nDSnp zc*p*g9y3^s8kAKs!-yTOPyMM~CqcT6Pvw$F8l>um*}X z>CaQKx|yvop-%Q~NiDyL@x1o)s={1TyFnkW%&9Pdg9s62L5&BUeZX$zeZJ}fvof*R zMVoG2U9!5mdayQHHDTiF!Mi$Lp;N^cUVY((5P5Erqc)`xSE}qXFSY8KEQx!-mFM1O zOl`?(O(=K^!sH_$g%5!cM@UL5b*2dLlsN)}1rpY!8{I+FcaJ*hQEXb$c0y375~}M> zFwxr6*iO`$``p5fDlt^$b4k`IZuI(gAGgclV(5pUAoatNl;eG-sv3{2VvjqKV^o(- z&j59yBq2nEIvv+vUwo7LA}e=Rw?G%cT=Cs| zu{oPH9QN4bgWM|3VcxJ!Tqhx(!EJ>7L6(~L;cA~kQsuDcc=$shRwV~6r+!j(>w0i6 z3p^VD?d2`pTck+k;>tZAd6Diz4EwaFp;krNX78{E^;;S(#pGL*E!tQ}e=XDRbILP@ zj5#f=jQ;-J|HnHmZU2?}atkoiRz}va*h`R6_QFbLb#ygcFZdS0R~C-wr1(11EoGRn z<6BNb)6V`-V|AlvxMdc7RYugX8n0oxgO{9~fx5B8`J5tRQ<-aaUedQQj2j*wl>4Ge zOk{aomUKfVX3ZN0eof@wr{A<`8)gIkpk{JuetOaV_m`;}q>#5gG7I7rPoSALBjp~m z-?y?+iI-ytTAv+kPeq0eDLG{9t~MWaSXGtQdmSG_dC~q~(LtI$c=6%H?@cWFMT=sW zd{|sUfWslk&uQR&p*t!UK;n;g_QkOG`cUtKWB9tRWrGpFLXvfs1XO`8y2!JU%m|VbD<4NN=xHzjk`R|O&eFgT1}f-&Sf$PGQDMOT&jnl64;(b zSrEu9Z)>|x_!YfTN#XhY_dbb*Hg^U4U=Mepr^F7=J zar6*)Q;{8wLHp<$fPW#-3+S@4Zy?Y4e$CE8DE?)=GL=f+NN=n>vR*})#d-m{h zA&+V+9uqGwZd!43Yw-Rh5036|B6!+@>Qp@NW_T** zhlYobc5E0F^u4fB3s(d9PKTL=#!-TyE-^{d`#W{jNnJ&D@Y3a6-w3*`Et1dQ_qUcR zKcUgjHs&93wfgE#3DH%Es#mXDZ_#Dgv|>%4(m}o4J!_KCpaveRrj1M>xo9etN$nkp zGV6KPp%^4)4Fw<`7QL?VywI%cQQFblz~hr{6~dcMSG0!qol8m-DtPatJdYq;>&ur^ zN0niI{F0Ob7e=oHK*fiPo3^lVb%_GRQ~kz=tych(Y=xv z+p;)Q9g2+p@@1N5X2V{OS~eMe-39z5?xRA{_hREjcZ1?Mln_`TqXlvu?Zf zF$Fg>RAg_zLC?~ZLLTJTIK1mdngNMMzXtvVVs5KNC~2Td;)WG(x|sr)P};$esJ(vC6-evk;{95YG_!@PJy=U3GO1n%DEjU6LhU zUcR&ENjzbg`ex+!7&SrNOw-a9_+-sOiLE7^w1>ysbD79UPA+!Zu8PqRI*2#V!oM<< zGmEXbDBW^?0|LXP{FUZ3_REj$5&EeQaq_jxZf+A=VK`Bq8nz(la8U&@^W;AsM4bFpnX_Ydw3f}OuWpH+b~ET@1v!}6C!c?TL*H5ceh z@|cbpx?kQB*m*rzL~}JF>s(EF^s|hlOMh;gzY3qVN-L$=ch&w{gIedFsb6xjxnqHg zL1rZ{uX|jS&Wp>e@UC(#3_B(H=v8Qa_^=7w1mHg%e!gJ$%4?s+fo$FdL!<;W6`-DJ zT)sZg5S2L23W*?`V+;WQjGa!}lV~UaJN}n1wyiB^C$4v%LtJEQ8DO&Grhtsi-6j8_ zCEzsf+mInN?_wIO*P!)o&{!YLlZldz-`)mnUGkTKD$tG|&d6muw1)Vag7@h0{lsy* zcn;MM)!H*3ObT`^Ii}S*I=W#mxV+T?>2O*4y`(05X&e*V!Eu(2es~vs_Q=Fo(~a|Y zw{-v@9cnMmvBY8*Qfu+_0lYk{NOyPl&(_%1Fv|X?IeJ*Nu<{9o4YMd` z*2K7(`jkTd)YK)O(q!+V;^O{Kjg$5D=^O?dLd#IJ7o5FJfc70&A_WEA{e!Vd6&ooR zyMLWPuiWMv9zsbrdU=#~!wn5@_O1ot;sRo$fAG83Xq8L2qH(YWDBlk$g=tN#I5IFgH8m9=(hUt0BO@|YKf(OXt{q&)VF~m7IlG+Jb+Ma- zzTAAlRF{|%?Xl~)4Nyu$=qG1qR{;Eh_@VciG6`hHCbI31*Dp1eWwvv}ql;g_qcu+J zv7DU*oHr;CDzEi`?g|DDMzG$w{^{sjhe<&}dx%b8pt`;xB~_gMoV%O*K4t*EM4di$ zBGsJ*63B->7<}t2h+;{ZP;aaS#@k%v(LqaBms+MVKKxraE`V7)mvDF!$J%o{#+Lc) z?S7YkDEXaROG9Q0LBF5hU*NaYSHUiN3?2{n^=)E$rp0JI-OjrGvwL(wc&a`PFI|@G zP9j9Xs^ES=Icv1DbTC@HP zRk^O$3&Rtcp{7Puw%Y)eO*t|}A)j8ioFJEKKbtxG9n8}fR5T=S6*ur%2)V$XIotNt zXf>x5>0eJ};V3Ba1KjAe+g>y9s3)q|nq+*m@Qghb-!Gar-lHbI+jP>H6ex9HY}P|f}Qk45QV3}=r6WH_lM zP++oQ=TvoYVEjWGaRBg<2w7O*LCFKs(bVMQ&JR3`i^cDWcU@K|HsO@fYvp%YE#+a# z&NR4T%1msTM;ha_Ogh$|Jp}h{5Pa<8aC7E>kWhund-ewspGJ0~mlv11&WMtWO5tu! zW3N(jr3qbRwyt~zJXR`SjoPvPQ4?n;1MB=WB_hf}kLFx4g0Ek=xJ!v22r{E$N=>Jq z!(K)J%|g*1uUtJ@XGLC}*ffJjrY{B~WHg3ut+I`QQ6OG=gUA*e=zl6Hwb5fsOe5;W zI#;fy|IF4T!iI)b?>7jtCCR0iHkY{E<`UXTTNjoobcqPWTxNHRK`l?ch)8{;9rqfR zxbA$jv&xehB92-45`H#+mNsNP?I44{GdMC!GY&7P6z?RTGT`)fIiGJNuNB%GwWANc z56S=Jz@E~qR+|8mv(n87dP%&{hqRsgrop1_G=4v#B}`7!p0%1N#eRWV$M*F^LAm)< zSdy~J%hZqR2=KsDQ=zEKM%z~wQyA)sj2ITOG`IAih0u~*5h}Bpx%IcsU7-aBxG!SL zr^3E`LDzf?7$X-P>RjgnlvxDP%E@68r`7p9;)a81hh((%Low1v1u2t(?_&G8*@Oo{l?>ub9tEt+Cyp16=rKYGd_1A? z8O}S1q%{jByU;H<=1yjw!P2v{ zmjPj^UzoCg;Bl~i-jb)f@XO`hJJ+fknG&JByW7UQj%McOiGp?=1Jq^}&SjOB#IIgo z&pFT>uW*zaFDm+kQt;u3@q)8L`F!886Y>5EI^xtjX)emi83Cx4X2DcUDy2Y~fx-|t z^ZB0hK~T^QKQtdedvrF2OZPnij*@d6u;4-Q;pF~9>Ld~}pi??T0R-d+SGlccATk&`2jb~O7(CP0Q3w-JK{FQp1rfkhZJ ze%HAi&7MVX>RnqjxA(rVHk#zi(uqIZl^eRKg_mE~$8)k^jR&YP^h0}Cl8gE&2Rm-& z)c$&(vQ1p=QUyw#{fhFg`Yp@gOZ6}+(@zCMrup9*f;lR1D2P zH~_Ho+9dY<-F+ZrqPZ)UUF-b-XXdBGW4r@vOt@EnCQ(pO(669<`bUtpXyByT$H$0W zoL1!8>6KqS#uP5P0Du3<$d^dyF5n7)Kqa>w6uoYn{4?eJ{<2>&_-B0KK0HvUNNt(O*5&y`Q4uG*dA2ggm#`6 z8G2QDn2zPrux}=As_{VBErzq82%S}5q}av2hoYhl^s86J`fx+WV?vA_3n2IzpMc|* z2R*M#5pjyVwVHPNG@MGW;s(SBYYvhYhdp&?0O3?~Vqhay6}FN|vtRC?td7fi$7eI@ zb-r+7aj6VQG#V+BL#v@mhgLonx4j_|!0Hd5BZ-WSv{Qv>RnWE*wwMaDhdS`F=W(Ez z)b8G`h0X{0YOD}(n$QP>tf2IG=N7a=2;pdEWDI-609hOTs+t;e3ve-oby&ORo!>_^ zFLp*LSabDyVwPJ31>%Szv5j92(m3>#(SyGzDYd{FI9iXXU~#tiA-H;znwk|>V}D>^ z;EAm6$4r&&ccCr$+5j9JpA|klJRIGz)vc$!)BpU*{us3_=lA7W@70r?-H{jC#4=t( z>0xlFn*QN@^^LE!VIgeDbuhI+P+iM|^(STI1cH!-X2@U@U?u_UBP;DsY0EZJx|J{( zSe*ev7_z*JZcxE5ziD2?$Do zs9>$|^fV?*r=-MT=(1ShYgO`;h){g0VxAh{==u1M#BL^KWXx^F7h*jI6j7`&hYd5k zVaLo29JBZ)s-17nC$-ygdR_71w z#rHbC%^VgjAWk_QKV^@INQ;Zri$9dQ^w=>cB{hPbcO%lUrC4#O*g&yeD4-r=Q|(M2 zbcb)~bgJ*6Tr4|i>eX4$W78EEMTW?wOVQKN%lvl-rdGO(>^DIyA%^t~6c#hi4#&}k zX+V4hB_(`M@3}uccOL!A{Uy!#T@uJap*3|~aRlM%{MLpCRWo`b%_Kgo#5_Yn;l?axHgXHL=wm#i_6QC#ax!IrYaTg=7^JPYe8dN968{6%l6+?#ZIqlae7yHTDwcd zy-ga!@x?_Ule&H!^w$I$-^fTyu2 zWvkd$2OA>>t;yhh*enMmfQp;rTi$ES7<9h;S<5B|0|NsPW%1j-P7$HC2l#tn;OKrl zV@{pRcD;0X+6&IGc83$W;{j3w+L5|opUx!{6NX|t*|S4h0`{cHmXVPa>}a8kIb*)O zoFaBJcp3znK*uHM-NpY zyF=+P(;FJlbPte_ee}2&7sF!e`3qODBO;5Qj8GhLdbnA=RUZmK{hM?gy0<+n<}Cdm zZY&Nrbb?!Uelr0uN8rcZq60~wg&*(>!OHZ~ah0R_GmNcu8d%Z;(H=t!bo6UDjkZ#X z+Q}2dl1Wy?;nq)4n~(9`8w*mJ=v!tRNwKlObv?|an0nY-=n-5$6nE<@VDB2Gd)nBc zxA*@6_T>_~0Dpn!kcIO0kpq#+|CLQL<%jwbu^tRLkfU2C`BdxesK4&hQyp+k%8m~{G{$7RB zURFL<2O!#D8uC1U-rdnr_3&Y_>E&}bhE(OoGavvogKFbI)?^ma^;3jBvM+6uG;06) ztKMs;L%!#u@uG~vwXa{kxNHch6qJ6g(2N!+)pwtbJ>n_M4pb^7oA%*3!#?eARssqK zAnVEO9F}WGnhpX9<>XxeRy+x&MbAQyq4r<3%$8m8)I#3}}ZJszO#Z``BcNUM9> zE)1m)KJu|XbpX1#gDkzIIKb(b&dl2`YJ3LMR!a*?KdkjG0p*Q>KAaTvK0$Hra(k|S zP6W7W9<*W3|Av9AK@LDrR$IKF`vLmUo`?ri(}RN_fj;6M1#2J)P~HUSRVg zH=45^Zi4xNJ=qx;PTM=|EHy=bv^A&|sQLkGPw4aaK6dsdZ1uKUpE5)5zAl||t<@S} zi7+oO9|E!W+5RsrBPrqn$Ilw~c3pv!yvNonM|g z=kM>cR59#Z&z_txmQK4ml?3TCYu}bZ-ZVoSqLk8*qDCt3g?v1Os3j$@gWo=4TR@Lh zb+`-#l8h$+$Y!A_m-zWU&MG}i(8;RwBAzqV%RMIO&rPpH{uE)mt|J3fp2cTrynWrX zy?(Hm?o+Ez#Ww(QcT7HZV&yp(SMt%+H{xj@^si#eW*{E_21-`$Vb-1kkZpwl1U&L)q74xG3UoE>ds&2HrD zz~3Ep_5)@PsGA2q&2Bf%OKPwiSb0%2ULLNEx?wpDzTF=i9vsZyxJA++6XV!SaT0!HfrTm} z!p&J&-__OB-ZxeHBe)`78d^cnDJ$Ot)Eo9HDsmdm;lNn}TCF6=v^_obpsDy~VdZht z^Z9J~MJ2IBGyoMFFP<@GW{cIW%fp0$rraCnn}N2^X_uD{%l@|IK}$$0RhKu2cmk$F zs-3Z*GM0&=Vk!)Fg=8D?Q<(F_QWazZQR}hc$i_&({(cuIo(&|Y#fYy41e`iL7>j9Y zGBRrW6Ky#Q_m?jL#|%wMasV8ao4c?3@#ji1J0UI>KDy@uj384;gn)p1SW1!I3rRz> zv1yBb!kC>ZK=C~xCPR&dhGt#o)Xjs>QaAYE3IgrJ=9}q#dbZ!ctivy}2xbKHUe8#VNpe^07%BSYH87v5^`}_xKEt?!2Eo9gw z-_L}IW(;?;kvOPpBcSkNbduc~q?_4OGLcux%F0R;p-qg50SeibD}`+J>j4J1WdIc= znS;POs77J}#JNDX3x!^Z(*eoi@6T5^3FMYQ^D6ba860!L5-UeQ;BeI|Phq`e8Xntm zV2h(wRSrN&#|^7Q9y`m8E!AiO!jp+dKzGW;!XQe=EF`29a@WCu+msP+nmiR?Foa9j zg$Rbxf-$7;@wqY2+fke|bC3}lK(4F65v$!+Y52N{X(r+#V|cI|lAa`K{=Ir5M5 z_X9SPBBq?}D~TDcZs!NNgId0HAcuk{9QV^D*t77;4~VTWg|Tn}!7+8gn+hnv7N4BD zc96;q$K{8qr|G|}CXGD9tn4`0j}8xiRT6p|Ot0dw#Q?my36vbjf){eK8h|No9c+IB zKS%>aVPWCc;_B?;>Q8_5`I+}eE54~Nb*Ts#4SRJR8R3lnGd{jGs(Va+YX@nYV-Bx6 z=I*axh^Z$K-I(Ii(!Pn4ENunh39xD>F<4$MuHTJrXL>eCsj2!vsX!I<7y_ASWW5J$ z_7lPdmVk76cF4?BJF4>%+ta^Vt8?l;KKAYos5|K!1O(Suc(G${{^q3Mix<76gpl6e z$hMcrk`gwW5PC-bcryrOjp1DCqlBTodg0R2qR;khJWvJpH!9QPk~|0h0(TvV+v$H9 zX6^#>4>AnWXFr(n4(2gDEC_7}LT_LUfM^?tybZ5Ga`Fi%8f!0X6pjiut^^*?vz18^Q30;19*nZ-9HKnl`f60dh=9>(WQ60eKD1Xi_6pc^oief zh*@MM{Iff+{U3;^+6@4&v;TCt z&8CeFd{$Q4@K-@f17RB29BYnDcr4cxnZVn-BH)gHvyp+sB=5s>s6@!NjEuIRi;gcCjyk&p`=4~rn5z*BZ$(CWpf`&DMw zKTb2$7wWuy#QflDOgnn((^Z2ZbOE)swSj?JKpcy^zcu^bRwpmk)!ki&b;o-4wG0lV z`eIxf|7{QM;p$JdT+!3>;F|MuKq*9cF8LTq-@avqv9*TD$jK5S+>Q1Qk?qXpTgVeJ z)b&Cf$U_t0E$jJ>Io-672K|9NWN0*SmLMzH6U_m9s&;VJ8@2m4RUkx3)@?=@tF z;jX~{yb=5YM4|sS-|)v#*Z+Hy{!3=zzt@lzrvIxS$q9;X1F7wCouTcZO82M2<|anD z8`P{f`pjgw>>cp0K_KGT5P5{dfra%SH+aEUMcK#r5Qwmqz5U+i86QvbtMJ&x6xFHY zrjMff^+IX-4h^B46#dwEaVkf1-R>*s3Zp&>WqG_0PELCK5He8n#;@ zcrmBQxts*oYkt+tOiW!}EFz|s>vE5!sT27T4rJJk+yl=xUk{gaZS0xJ$#J^!z!M6JdOGGDJ8rbN#J@eEB(hyxztG z{E%*ofk9Q4%;kl1xAkW@I8>RyV^~B~bP}V-h30UeqJIrlhmBL}5jyR4rs*u&pF!%* z;FN?cQKY>kRl=+8BL_NX;mLCjxMnFTuv2HWyTF~vv7R$;Lc*IH?$y$?RW)Ywv}+rZ zI4dRJCX#z9Y$9nz$I7oihnXroc(ADJ`cY+0$;0I?DTKH9qV1&sp>5t<$31J=H;zM* z^@{dxlTp`sJg!kagv!CUx14qx!EOtTA4weU`?qN}_9sbcL(H)})e;-Y!Y;C2RPnxX zsIB3+`&}tQ6@LQIO-&>DQ3Fw`6q&(vnhiaAw;XI2tW4!^(e$>Mhp9N7O0-dkT~Lw6 zGl>`(L0vE5%L;EER?BJ~1^+cI6B6CgbGSDA%(fz!eRVmEEzoQ%^{|`(Wl}wi=Ac zfau3rYvIk+)m_btyk-5tVcSQjX<|L1s@Q}_0o(RA`y}11K4wl)42sXIthA!KXH8?v3M-O4%T zwh2pi$Auq#awsk~fiR_W%-#L|IKznefe^b(Q&fb(XOMW-;9FC15wAzlBfIG3K*5`U zmFsCwJ;zd*kB+V>5m`^Sx1uM9^Pe01B`&={M*fQ7X(~}2Lr(;|!KVvBEIU}y5{?E| z!fxxB4V}R>^SPBl_-pwsse-VLA4k`GNhC5&OsC$2OuNlmeN@Rf-TNizlEP>9VyDke z4r)JSo7X*gThpgRVZa3B`&!Ped^E9=2my)Xu?Z`d#T5~{Hx+tLdXV9NyPeiABqP|4 zEhlRu-_teu`qrP!GYAKer6eSucC24cR2P?&t@O4`mxdxVRX3f|Yxci&ROw(^}4*6OV}kg@1XsJw24olu&DIHH9JbNR`{j_MgTL32&9 zlG3I5kJIS!?$o=cl(4(55YN-DdnGT9`f|GUBJ`2Jz5O}4n2RTn+K?Qk-hs149LW&> zO^`I@dQ@z;m?&~Xdg6O)>-FAeZ>V`{=#1?i1$?qEN)kP3FVvW#6y)S)HJy%9?dtfP zxcAAy&A6ZR$c38feZ+(I*zFLqI_tMe%aGnU8=7%ZfwjJ1H;d$;F?#9mn(vO6izV}j z*Jpnb2{SPquJLDCDwVvbe)NQ8?yC>qkukmu+rYF3Kh*7+pR>=gUYTG<(L^-9hW_2>3rA9syA*HWrasru@9C-p?}|s zSjA@bJ@Rf-qTVkry{ZXUhvz>i&dN|ZKbC$Bb>K@A_`}PRUTQ`5uVbvf-q&oUazAl!n1FuvL)w-SMEh3$zw7 zKj;GRL&G0a)%!O$x3v7}?--Nlo1uOk)do}Ze)+*f!`Z^!#9of7&$nx0Vh|ZsN+N-$ zQ%o)T#OYCBs?A@Jw>l;*Eha6_68K<)%fNxa4UK-1kWZA|@Wt%}u9RS#Lk>9eT#);koCf@lYaq@a1&TZV+j^Cs*#Gw$;&=}vCjHlFUO&eB@3pw}|NW2n z;xdWw2S0)kLO>^S;2>H1uYY>EQt80oc=K8>BodtctZiwv#S}k3pmGy96<5vIBjA7h zDIOjWNpNsi;%|%71HTHsWN5@1FPZTR1O!*Dm-oYSp!@aubO%EG0Jv9gEib~7+7@Xj zx=H@=)zU2o8QaV!GnsL49_oVUnm%-!@V21)Nt@|wa1_Je|HJW%3T(o2DL}1U^@BHyQBr~6-uCB_k#gseWl84oB zf?8=TehGhH3eh9EOSW@x$iCYsi^L-#{rB1ikb{l>L%$e2;G-?Y@y}3BxPXwE_IxMr z{FBABv^CkShHJvTz-ttzIk2A8NM)ABe~!&IHn5!rF?c8lTR_Y40XA~J-Dtk<{MAD> zl;VdIwTP)#xURjc045;00m(@+gzlozV#&333f?UmKkxU^nTaQfgKYb`e|o(3Qvbtj z_@1e^NR5)>EI!@(y*P{YfcFr{wLIBZFV!rwdoqNG{uvyE%5OE%vklYP9lOQU3U#J2 z-h*4aJA01K&-wWn-u`_%;s}Y1Uk!#Ce)KlLqH`YI)q0RlM+XMWCGQ<J?Zij`RW16CpLhZ;}b^hr|FNaW|F<#G7L^+0Jq^q-QB)}I!NDO9Mr6y35H+Ah# z`1w(Pbo~3veKo2^Mg+BW#ulNAIp~; zqs3RlRR72aW4_K5_Vy$^)dh2$Atf4C{>)(v0;*@bsMQySIWLtd{rRv-tYeb D4_GeN literal 0 HcmV?d00001 diff --git a/docs/en/.readthedocs.yaml b/docs/en/.readthedocs.yaml new file mode 100644 index 0000000..c534012 --- /dev/null +++ b/docs/en/.readthedocs.yaml @@ -0,0 +1,17 @@ +# .readthedocs.yaml +# Read the Docs configuration file +# See https://docs.readthedocs.io/en/stable/config-file/v2.html for details + +version: 2 + +build: + os: ubuntu-22.04 + tools: + python: "3.11" + +sphinx: + configuration: docs/en/conf.py + +python: + install: + - requirements: docs/requirements.txt diff --git a/docs/en/API_Reference/index.md b/docs/en/API_Reference/index.md new file mode 100644 index 0000000..6faaf4a --- /dev/null +++ b/docs/en/API_Reference/index.md @@ -0,0 +1,171 @@ +# API reference + +The main functions and classes are available through `import entropack as ep`. +For complete examples, see [General tensor compression](../Usage/Tensor-compression.md) and +[Compressed Linear usage](../Usage/Linear-layers.md). Configuration parameters are listed in +[Compression configuration](../Usage/Configuration.md). + +## Tensor encoding and decoding + +### compress + +```text +compress(tensor: torch.Tensor, config: CompressionConfig) -> CompressedTensor +``` + +Compresses `tensor` using the scheme selected by `config`. Returns a `CompressedTensor` +that decompresses to the same shape and dtype as the input. + +Input requirements depend on the scheme: +DFloat11 accepts BF16 tensors, Tile-ANS supports multiple dtypes, and lattice quantization requires +a nonempty two-dimensional tensor with finite values. + +| Parameter | Meaning | +| --- | --- | +| `tensor` | PyTorch tensor to compress | +| `config` | Required. Selects a scheme through `DFloat11Config`, `TileANSConfig`, or `LatticeRANSConfig` | + +Some encoding failures emit a warning explaining the failure and return an uncompressed +container with `compress_method == "raw"`. Invalid configurations and backend dispatch +failures raise errors. + +### decompress + +```text +decompress(compressed: CompressedTensor, config: CompressionConfig) -> torch.Tensor +``` + +Returns a tensor with the input's original shape and dtype. Lossless schemes restore its values +bit for bit; lossy schemes return an approximate reconstruction. The result is on the original +input device unless the container has been moved, for example with `compressed.to(device)`. + +| Parameter | Meaning | +| --- | --- | +| `compressed` | Container returned by `compress` or restored from a checkpoint | +| `config` | Configuration for the corresponding scheme; decode settings control reconstruction | + +Changing encoding parameters such as `target_bpp` at decode time does not alter the stored data +or requantize the tensor. + +## CompressedTensor + +Holds the compressed representation of one tensor. Usually returned by `compress` or restored +from a checkpoint with `from_state_dict`. + +### Common properties + +| Property | Type | Meaning | +| --- | --- | --- | +| `shape` | `tuple[int, ...]` | Original tensor shape | +| `dtype` | `torch.dtype` | Reconstructed tensor dtype | +| `compress_method` | `str` | Scheme actually used by the container | +| `lossless` | `bool` | Whether the scheme is lossless | +| `actual_bpp` | `float` | Stored bits per element, including metadata | + +`actual_bpp = 8 * storage_nbytes() / math.prod(shape)`. +This measures the compressed representation, not checkpoint file size or runtime memory use. + +### Common methods + +| Method | Returns | Meaning | +| --- | --- | --- | +| `to(device)` | `CompressedTensor` | Returns a new container on the specified device without changing its dtype or values; accepts only a device argument | +| `storage_nbytes(include_header=True)` | `int` | Total compressed size in bytes; `include_header=False` excludes the container header | +| `state_dict(prefix="")` | `dict[str, torch.Tensor]` | Exports the compressed tensor for saving | +| `CompressedTensor.from_state_dict(state, prefix="")` | `CompressedTensor` | Restores the container from that dictionary without recompression | + +Use the same `prefix` when saving and restoring. The dictionary can be saved with `torch.save` +and loaded with `torch.load(..., weights_only=True)`. Set `map_location` to choose the device +on which it will be restored. + +## CompressedLinear + +A linear layer that uses reconstructed weights for each forward call. Weights are stored in +compressed form; the bias is not compressed. +Requires a CUDA GPU and the matching CuPy package. + +### Creating a layer + +```text +CompressedLinear(in_features, out_features, bias=True, *, + config=None, device=None, dtype=torch.bfloat16) +CompressedLinear.from_linear(linear, **kwargs) -> CompressedLinear +``` + +| Constructor parameter | Meaning | +| --- | --- | +| `in_features` / `out_features` | Input and output feature counts | +| `bias` | Whether to include a bias | +| `config` | Weight compression configuration. The default `None` selects DFloat11 for BF16 and Tile-ANS for other supported dtypes | +| `device` | Bias device when constructing a layer directly | +| `dtype` | Dtype used when compressing weights and initializing the bias | + +`from_linear` returns a new layer with compressed source weights and a copy of the bias. +The source weights must already be loaded and cannot be on the `meta` device. +A typical call is `ep.CompressedLinear.from_linear(linear, config=config)`. +Pass `config` or `dtype` through `kwargs`; `dtype` defaults to the source weight dtype, +and the device is taken from the source layer. + +Calling the constructor directly creates a layer without weight data. Call `compress_weight` +or load a checkpoint before running inference. + +### Common methods and properties + +| Interface | Returns | Meaning | +| --- | --- | --- | +| `compress_weight(weight)` | `None` | Compresses and replaces the weights; requires shape `(out_features, in_features)` | +| `dequantize(device=None)` | `torch.Tensor` | Returns dense weights with `container_dtype`, on the layer's device unless `device` is specified | +| `forward(x)` | `torch.Tensor` | Applies the layer to `x` of shape `(..., in_features)` and returns shape `(..., out_features)` | +| `compressed_weight` | `CompressedTensor` | Compressed weight container held by the layer | +| `container_dtype` | `torch.dtype` | Dtype of the weights or quantized codes in the compressed container | +| `stored_nbytes` | `int` | Weight storage bytes, including metadata and low-precision quantization scales, excluding bias | +| `compressed_bits` | `float` | `8 * stored_nbytes / (in_features * out_features)` | + +Invoke the forward operation as `layer(x)`. The dense `.weight` is `None`; use `dequantize()` +when numerical weights are needed. + +Use standard `state_dict()` / `load_state_dict()` calls to save and restore layer state. +Before loading, construct layers with matching classes, shapes, container dtypes, and compression +schemes. The Config object itself is not stored in the checkpoint. `.to(device)` moves the layer; +model dtype conversion does not re-encode its compressed weights. + +## CompressedFP8Linear and CompressedINT8Linear + +Linear layers with FP8 or INT8 weights and activations. They share the constructor arguments, +`from_linear`, storage properties, and checkpoint interfaces of `CompressedLinear`. +For input shape `(..., in_features)`, the output has shape `(..., out_features)` and the input's +dtype and device. + +| Class | Weight and activation format | CUDA GPU requirement | +| --- | --- | --- | +| `CompressedFP8Linear` | FP8 E4M3FN | SM8.9 or later | +| `CompressedINT8Linear` | INT8 | SM8.0 or later | + +The layer class determines the code format. The constructor's `dtype` argument does not change +the FP8 or INT8 format. + +With `config=None`, quantized codes are stored directly. Passing `LatticeRANSConfig` applies +additional lossy compression with `1 <= target_bpp < 8`. `stored_nbytes` includes the +per-row quantization scales needed to reconstruct weights. + +| Method | Returns | Meaning | +| --- | --- | --- | +| `codes(device=None)` | FP8 or INT8 tensor | Restores quantized codes without applying row scales | +| `dequantize(device=None)` | FP32 tensor | Multiplies restored codes by their row scales to obtain numerical weights | + +Both methods return tensors on the layer's device unless `device` is specified. +Lossy compression may change the codes from their initial quantized values. + +## Config classes + +Configs control tensor encoding and decoding as well as weight storage in Compressed Linear. +The following classes inherit from `CompressionConfig`: + +| Class | Purpose | +| --- | --- | +| `DFloat11Config` | Lossless BF16 compression | +| `TileANSConfig` | Lossless tiled ANS compression for multiple dtypes | +| `LatticeRANSConfig` | Lossy lattice quantization with bitrate controlled by `target_bpp` | + +See [Compression configuration](../Usage/Configuration.md) for scheme selection, defaults, +and parameter ranges. diff --git a/docs/en/Makefile b/docs/en/Makefile new file mode 100644 index 0000000..4ae5e53 --- /dev/null +++ b/docs/en/Makefile @@ -0,0 +1,14 @@ +# Minimal makefile for Sphinx documentation + +SPHINXOPTS ?= +SPHINXBUILD ?= sphinx-build +SOURCEDIR = . +BUILDDIR = ../_build/en + +help: + @$(SPHINXBUILD) -M help "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O) + +.PHONY: help Makefile + +%: Makefile + @$(SPHINXBUILD) -M $@ "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O) diff --git a/docs/en/Principles/DFloat11.md b/docs/en/Principles/DFloat11.md new file mode 100644 index 0000000..49cb49b --- /dev/null +++ b/docs/en/Principles/DFloat11.md @@ -0,0 +1,52 @@ +# DFloat11 + +DFloat11 compresses BF16 tensors without changing their bits. It exploits the fact that +the exponent values in many tensors are concentrated in a small part of the available +range. Frequent exponents can then be represented with fewer bits, while the sign and +fraction remain unchanged. + +## Encoding + +A BF16 value contains one sign bit, eight exponent bits, and seven fraction bits. +The encoder separates each value into an exponent and a byte containing its sign and +fraction. These bytes are stored directly. The exponent stream is compressed using a +Huffman code constructed from the tensor's exponent frequencies: common exponents receive +short codes and uncommon exponents receive longer ones. + +Huffman codes have variable lengths, so a decoder cannot start at an arbitrary bit and +immediately identify the next symbol. EntroPack records entry positions and symbol counts +for coding regions, allowing different regions to be decoded in parallel. The entry +information and Huffman tables add metadata to the compressed representation. + +## Decoding and storage + +Decoding recovers the exponent sequence through the Huffman tables and combines each +exponent with its stored sign and fraction. Reassembling these fields restores the +original BF16 bits, and the stored shape determines how they form the output tensor. +No numerical quantization or rounding is involved. + +The achieved size depends on the exponent distribution and decoding metadata. A concentrated +distribution offers more compression than a broad one, and metadata has a larger relative +cost for small tensors. `DFloat11Config` does not specify a target bitrate. The name +DFloat11 does not imply that every tensor is stored at exactly 11 bits per element. + +## Usage + +The example compresses a BF16 tensor and verifies that decompression preserves its bits. +See [Config](../Usage/Configuration.md) for coding-region parameters. + +```python +import torch +import entropack as ep + +tensor = (torch.randn(256, 256, device="cuda") * 0.02).to(torch.bfloat16) +config = ep.DFloat11Config() + +compressed = ep.compress(tensor, config) +restored = ep.decompress(compressed, config) + +assert restored.shape == tensor.shape +assert restored.dtype == tensor.dtype +assert torch.equal(restored.view(torch.uint8), tensor.view(torch.uint8)) +print(f"Stored: {compressed.actual_bpp:.2f} bits per element") +``` diff --git a/docs/en/Principles/Lattice-rANS.md b/docs/en/Principles/Lattice-rANS.md new file mode 100644 index 0000000..c9dd9d9 --- /dev/null +++ b/docs/en/Principles/Lattice-rANS.md @@ -0,0 +1,76 @@ +# EntroPack lattice compression + +EntroPack's lattice scheme combines lossy vector quantization with lossless entropy coding +to compress two-dimensional tensors. `LatticeRANSConfig` accepts target bitrates from 1 to +11 bits per element, including non-integer values. Decompression retains the input dtype, +while the target parameter controls storage rate. + +![EntroPack encoding and decoding pipeline](../../assets/entropack-pipeline.png) + +EntroPack's encoding and decoding pipeline, illustrated with a weight matrix. The upper panel +shows rate search and encoding; the lower panel shows fused GPU decoding and reconstruction. + +## Lattice quantization and integer fields + +Rows can differ substantially in numerical scale. The encoder first normalizes each row +by its root mean square, then groups the normalized values into eight-dimensional vectors. +Each vector is approximated by its nearest point on a scaled E8 lattice, a regular +arrangement of points in eight dimensions. A shared quantization scale controls the spacing +between these points. Finer spacing generally reduces reconstruction error but requires +more bits to describe the selected points. + +E8 has integer-coordinate and half-integer-coordinate subsets, called cosets. EntroPack +represents each point by its coset and eight invertible integer fields, using the lattice's +parity constraint to compact the final coordinate. The probability model conditions each +coordinate field on the coset, capturing differences between the two subsets. Frequently +occurring field values can then be encoded with fewer bits on average. The fields can +later be inverted arithmetically without a reconstruction codebook. + +## Rate selection and refinement + +The encoder searches for a quantization scale using sampled rows. For each candidate scale, +it estimates the coded field size and the metadata needed for decoding. This avoids +repeatedly producing a full compressed stream during the search. Once the scale is selected, +the encoder quantizes the full tensor and fits a reconstruction scale for each row by +least squares. + +Optional per-row rate–distortion refinement compares several resolutions for each row and +allocates them under an estimated storage budget. It alternates candidate selection with +updates to the shared probability model. This adds encoding work and is disabled by default. + +## Encoding and reconstruction + +The selected fields are encoded with rANS in independently decodable tiles. +The representation also stores the probability tables, row scales, and tile metadata. +Decoding recovers the fields, reconstructs lattice points, and applies the row scales in +a fused GPU operation. The result has the input's shape and dtype. + +Quantization and conversion back to the output dtype determine reconstruction error. +Entropy coding itself preserves the selected fields exactly. The achieved bitrate can +differ from the requested target because scale selection uses a size estimate. The +`actual_bpp` property reports the actual stored bytes, including metadata, divided by the +element count and multiplied by eight. + +## Usage + +The example targets 3.5 bits per element and measures relative L2 error against the input. +Search and refinement settings are described in [Config](../Usage/Configuration.md). + +```python +import torch +import entropack as ep + +tensor = (torch.randn(256, 256, device="cuda") * 0.02).to(torch.bfloat16) +config = ep.LatticeRANSConfig(target_bpp=3.5) + +compressed = ep.compress(tensor, config) +restored = ep.decompress(compressed, config) + +reference = tensor.float() +relative_l2 = (restored.float() - reference).norm() / reference.norm() +assert restored.shape == tensor.shape +assert restored.dtype == tensor.dtype +print(f"Target: {config.target_bpp:.2f} bits per element") +print(f"Stored: {compressed.actual_bpp:.2f} bits per element") +print(f"Relative L2 error: {100 * relative_l2.item():.2f}%") +``` diff --git a/docs/en/Principles/Tile-ANS.md b/docs/en/Principles/Tile-ANS.md new file mode 100644 index 0000000..b42872e --- /dev/null +++ b/docs/en/Principles/Tile-ANS.md @@ -0,0 +1,57 @@ +# Tile-ANS + +Tile-ANS compresses tensors losslessly by encoding their storage bytes. It supports +floating-point and integer tensors, including BF16, FP16, FP32, FP8, and INT8. Because it +works on bit representations rather than numerical approximations, decompression restores +the original values exactly. + +## Byte streams and probability tables + +Different byte positions within a numerical format often have different distributions. +Tile-ANS therefore groups bytes by their position within each element. A two-byte format +produces two streams, while a four-byte format produces four. Each stream collects the +corresponding byte from every tensor element. + +The encoder counts byte frequencies separately for these streams and builds a probability +table for each. A skewed distribution can be encoded compactly because common bytes receive +shorter representations on average. When a stream offers little benefit after accounting +for coding overhead, it is stored directly. A single tensor can therefore contain both +entropy-coded streams and directly stored streams. + +## Tiled encoding and decoding + +Each stream is divided into independently decodable tiles. Entropy-coded tiles use range +asymmetric numeral systems (rANS), which encode symbols through reversible integer-state +updates. Multiple interleaved states allow symbols within a tile to be decoded in parallel, +while separate tiles provide additional parallel work. All tiles of a stream share its +probability table, avoiding a separate table for every tile. + +The decoder uses the same probability tables to reverse the state updates and recover +each coded byte stream. It then combines the decoded and directly stored streams, placing +their bytes back into the original positions within the tensor elements. + +No quantization is performed. Storage depends on the byte distributions and metadata, +so lossless compression does not provide a chosen target bitrate or guarantee a smaller +representation for every input. Larger tiles reduce metadata per element, while smaller +tiles expose more independent decoding tasks. + +## Usage + +The example compresses an FP16 tensor and checks its original bits after decompression. +Tile and probability-table settings are described in [Config](../Usage/Configuration.md). + +```python +import torch +import entropack as ep + +tensor = (torch.randn(256, 256, device="cuda") * 0.02).to(torch.float16) +config = ep.TileANSConfig() + +compressed = ep.compress(tensor, config) +restored = ep.decompress(compressed, config) + +assert restored.shape == tensor.shape +assert restored.dtype == tensor.dtype +assert torch.equal(restored.view(torch.uint8), tensor.view(torch.uint8)) +print(f"Stored: {compressed.actual_bpp:.2f} bits per element") +``` diff --git a/docs/en/Usage/Configuration.md b/docs/en/Usage/Configuration.md new file mode 100644 index 0000000..56b0fe9 --- /dev/null +++ b/docs/en/Usage/Configuration.md @@ -0,0 +1,100 @@ +# Compression configuration + +A Config selects the compression scheme and its encoding and decoding parameters. +`execution_backend` defaults to `"auto"`, which selects the execution backend automatically. + +## Choose a configuration + +| Config | Compression | Input requirements | +| --- | --- | --- | +| `DFloat11Config()` | Lossless BF16 compression | BF16 tensors | +| `TileANSConfig()` | Lossless compression for multiple dtypes | Supported tensor dtypes | +| `LatticeRANSConfig(target_bpp=...)` | Lossy compression at a target bitrate | Nonempty, finite, two-dimensional tensors | + +Both lossless schemes reproduce every input bit. Their compressed size depends on the tensor's +data distribution. `TileANSConfig` also supports BF16, so either lossless scheme can be used for +that dtype. `LatticeRANSConfig` accepts targets from 1 to 11 bits per element, including +non-integer values. Its actual stored rate is available through `CompressedTensor.actual_bpp`. + +## Supported tensor dtypes + +| Tensor dtype | `DFloat11Config` | `TileANSConfig` | `LatticeRANSConfig` | +|---|---|---|---| +| `float32` | — | lossless | lossy | +| `float16` | — | lossless | lossy | +| `bfloat16` | lossless | lossless | lossy | +| `float8_e4m3fn` | — | lossless | lossy | +| `float8_e4m3fnuz` | — | lossless | lossy | +| `float8_e5m2` | — | lossless | lossy | +| `float8_e5m2fnuz` | — | lossless | lossy | +| `int64` | — | lossless | lossy | +| `int32` | — | lossless | lossy | +| `int16` | — | lossless | lossy | +| `int8` | — | lossless | lossy | +| `uint64` | — | lossless | lossy | +| `uint32` | — | lossless | lossy | +| `uint16` | — | lossless | lossy | +| `uint8` | — | lossless | lossy | +| `bool` | — | lossless | lossy | + +The lossless schemes accept tensors of different shapes, while lattice quantization requires +two-dimensional input. BF16, FP16, and FP8 refer to the corresponding PyTorch dtypes above. +Packed four-bit formats, FP64, and complex dtypes are not supported. + +## Parameter conventions + +Changes to encode settings affect subsequent compression, not existing compressed data. +Decode settings take effect when restoring a tensor. Most settings can retain their defaults. For lossy compression, `target_bpp` controls the +storage rate and a positive `row_rdo_iterations` enables per-row rate–distortion optimization (RDO). + +## CompressionConfig + +These execution settings apply to all three configuration classes. + +| Field | Type | Default | Stage | Meaning | +| --- | --- | --- | --- | --- | +| `execution_backend` | `str` or `None` | `"auto"` | Both | `"auto"` or `None` prefers CUDA and falls back to the PyTorch implementation if unavailable. `"cuda"` requires CUDA. `"eager"` selects the PyTorch fallback for tensor encoding and decoding. | + +## DFloat11Config + +Lossless BF16 compression. + +| Field | Type | Default | Stage | Meaning | +| --- | --- | --- | --- | --- | +| `execution_backend` | `str` or `None` | `"auto"` | Both | Backend selection as above | +| `bytes_per_thread` | Positive `int` or `None` | `16` | Encode | Encoded bytes processed per thread. Affects compression ratio and decoding parallelism. | +| `threads_per_block` | Positive `int` or `None` | `128` | Encode | Threads per block during encoding | + +## TileANSConfig + +Lossless compression of the supported tensor dtypes. + +| Field | Type | Default | Stage | Meaning | +| --- | --- | --- | --- | --- | +| `execution_backend` | `str` or `None` | `"auto"` | Both | Backend selection as above | +| `tile_elements` | `int` in [0, 2^31 − 1] | `0` | Encode | Elements per compressed tile. Affects compression ratio and decoding parallelism. `0` selects automatically. | +| `probability_bits` | `0`, `9`, `10`, `11`, `12` | `0` | Encode | Probability-table precision. `0` selects automatically. | +| `raw_lane_threshold` | `float` in [0, 8] | `7.9` | Encode | Threshold for storing hard-to-compress data directly, measured in estimated encoded bits per input byte. | +| `threads_per_block` | Positive `int` or `None` | `None` | Both | GPU block width. `None` selects automatically. | + +## LatticeRANSConfig + +Lossy compression of finite, two-dimensional tensors. + +| Field | Type | Default | Stage | Meaning | +| --- | --- | --- | --- | --- | +| `execution_backend` | `str` or `None` | `"auto"` | Both | Backend selection as above | +| `target_bpp` | `float` in [1, 11] | `4.0` | Encode | Target bits per input element. Non-integer targets are supported. Inspect `actual_bpp` for the stored rate. | +| `prob_bits` | `int` in [9, 15], `0`, or `None` | `None` | Encode | Probability-table precision. `None` or `0` selects automatically. | +| `tile_elements` | Positive `int` or `None` | `None` | Encode | Elements per compressed tile. Affects compression ratio and decoding parallelism. `None` selects automatically. | +| `row_rdo_iterations` | `int` in [0, 8] | `0` | Encode | Per-row rate–distortion refinement sweeps. `0` disables refinement. More sweeps increase compression time. | +| `row_rdo_candidates` | Positive `int` | `5` | Encode | Number of candidate quantizations per row for RDO. More candidates increase compression time. | +| `scale_search_iterations` | Positive `int` | `12` | Encode | Number of quantization-scale search iterations | +| `scale_search_max_vectors` | Positive `int` | `262144` | Encode | Sample limit for rate search, in vectors of eight elements | +| `threads_per_block` | Positive `int` or `None` | `None` | Decode | GPU block width. `None` selects automatically. | +| `l2_prefetch` | `bool` | `True` | Decode | Enable GPU L2 cache prefetching during decoding | + +## Compression principles + +The encoding and decoding processes are described in [DFloat11](../Principles/DFloat11.md), +[Tile-ANS](../Principles/Tile-ANS.md), and [Lattice-rANS](../Principles/Lattice-rANS.md). diff --git a/docs/en/Usage/Linear-layers.md b/docs/en/Usage/Linear-layers.md new file mode 100644 index 0000000..916684e --- /dev/null +++ b/docs/en/Usage/Linear-layers.md @@ -0,0 +1,163 @@ +# Compressed Linear usage + +Compressed Linear replaces a PyTorch linear layer with one that uses compressed weights. +Call `layer(x)` as usual: the input's last dimension changes from `in_features` to +`out_features`, and the other dimensions stay the same. A [Config](Configuration.md) +selects the compression scheme and its parameters. + +Compressed Linear requires a CUDA GPU and the matching CuPy package. See +[Quick start](Quick-start.md) for installation. + +`CompressedLinear` accepts `DFloat11Config`, `TileANSConfig`, or `LatticeRANSConfig`. +The selected scheme must support the weight dtype. + +## Replace an existing layer + +`from_linear` compresses an existing layer's weights and returns a new layer with a copy +of the original bias. For pretrained models, load the checkpoint before calling this method: + +```python +import torch +import entropack as ep + +linear = torch.nn.Linear(256, 256, dtype=torch.bfloat16, device="cuda") +config = ep.LatticeRANSConfig(target_bpp=4.0) +layer = ep.CompressedLinear.from_linear(linear, config=config) +x = torch.randn(8, 256, dtype=torch.bfloat16, device="cuda") + +with torch.inference_mode(): + output = layer(x) +print(output.shape, f"{layer.compressed_bits:.2f} bits per weight") +``` + +Assign the returned layer to the corresponding module attribute to use it in the model. + +## Replace several layers in a model + +This example replaces ordinary linear layers recursively and keeps the output layer in its +original format. Names in `skip` are module paths, as reported by `named_modules()`. + +```python +import torch +import entropack as ep + +model = torch.nn.Sequential( + torch.nn.Linear(256, 256), + torch.nn.GELU(), + torch.nn.Linear(256, 64), +).to(device="cuda", dtype=torch.bfloat16).eval() +config = ep.LatticeRANSConfig(target_bpp=4.0) + + +def compress_linears(module, config, skip=(), prefix=""): + for name, child in list(module.named_children()): + path = f"{prefix}.{name}" if prefix else name + if path in skip: + continue + if type(child) is torch.nn.Linear: + replacement = ep.CompressedLinear.from_linear(child, config=config) + setattr(module, name, replacement.train(child.training)) + else: + compress_linears(child, config, skip, path) + + +compress_linears(model, config, skip={"2"}) +x = torch.randn(8, 256, device="cuda", dtype=torch.bfloat16) +with torch.inference_mode(): + output = model(x) +print(output.shape, type(model[0]).__name__, type(model[2]).__name__) +``` + +The example selects standard `torch.nn.Linear` layers. Custom linear classes or shared +weights may need model-specific handling. Omit `skip` to compress every ordinary linear +layer. If `CompressedLinear` cannot compress a layer's weights, replacement raises an +error; use `skip` to keep that layer in its original form. + +## Combine compression with FP8 or INT8 computation + +| Layer | Weight format | Computation | +| --- | --- | --- | +| `CompressedLinear` | Input weight dtype | Standard linear operation, using the activation dtype | +| `CompressedFP8Linear` | FP8 E4M3FN codes | FP8 weights and activations, requires a CUDA GPU with SM8.9 or later | +| `CompressedINT8Linear` | INT8 codes | INT8 weights and activations, requires a CUDA GPU with SM8.0 or later | + +For `CompressedFP8Linear` and `CompressedINT8Linear`, use `config=None` for FP8 or INT8 +quantization alone, or pass `LatticeRANSConfig(target_bpp=...)` to apply further lossy +compression to the quantized weights. The target must be at least 1 bpp and less than 8 bpp. + +```python +import torch +import entropack as ep + +linear = torch.nn.Linear(256, 256, dtype=torch.bfloat16, device="cuda") +layer = ep.CompressedINT8Linear.from_linear( + linear, config=ep.LatticeRANSConfig(target_bpp=4.0) +) +x = torch.randn(32, 256, dtype=torch.bfloat16, device="cuda") +with torch.inference_mode(): + output = layer(x) +print(output.shape, layer.container_dtype, f"{layer.compressed_bits:.2f} bits per weight") +``` + +Use `CompressedFP8Linear` in the same pattern for FP8, on supported hardware. +To inspect the weights, `codes()` returns their FP8 or INT8 values, and `dequantize()` +returns their floating-point values after dequantization. + +## Measure storage + +`stored_nbytes` reports the compressed weight size, including metadata and FP8 or INT8 quantization scales. +`compressed_bits` is `8 * stored_nbytes / (in_features * out_features)`. +For multiple layers, sum stored bytes and weight elements before computing the ratio. +Biases are separate from this weight-storage measure. + +This measures weight storage, not peak inference memory. + +## Save and load a model + +Save the model's `state_dict`, then construct a model with the same architecture and +Compressed Linear classes before loading it. The following example compresses two layers +and restores their saved weights into a fresh model: + +```python +from pathlib import Path + +import torch +import entropack as ep + +config = ep.LatticeRANSConfig(target_bpp=4.0) +model = torch.nn.Sequential( + torch.nn.Linear(256, 256), + torch.nn.GELU(), + torch.nn.Linear(256, 64), +).to(device="cuda", dtype=torch.bfloat16) +for index in (0, 2): + model[index] = ep.CompressedLinear.from_linear(model[index], config=config) +model.eval() + +x = torch.randn(8, 256, device="cuda", dtype=torch.bfloat16) +with torch.inference_mode(): + expected = model(x) + +path = Path("compressed_model.pt") +torch.save(model.state_dict(), path) + +restored = torch.nn.Sequential( + ep.CompressedLinear(256, 256, config=config, device="cuda", dtype=torch.bfloat16), + torch.nn.GELU(), + ep.CompressedLinear(256, 64, config=config, device="cuda", dtype=torch.bfloat16), +).eval() +state = torch.load(path, map_location="cuda", weights_only=True) +restored.load_state_dict(state) +with torch.inference_mode(): + actual = restored(x) + +assert torch.allclose(actual, expected) +print(actual.shape) +``` + +The example saves `compressed_model.pt` in the current directory. Change the path as needed. +To load an existing checkpoint, construct the `restored` model, then call `torch.load` and +`load_state_dict`. +Keep the model architecture, layer names and classes, compression configurations, weight +dtypes, and library version with the checkpoint. Ordinary `torch.nn.Linear` layers +cannot load Compressed Linear checkpoints directly. diff --git a/docs/en/Usage/Quick-start.md b/docs/en/Usage/Quick-start.md new file mode 100644 index 0000000..e049e3f --- /dev/null +++ b/docs/en/Usage/Quick-start.md @@ -0,0 +1,84 @@ +# Quick start + +This guide covers installation, tensor compression and decompression, and basic Compressed Linear usage. + +## Installation + +Python 3.10 or later is required. First install a CUDA-enabled build of PyTorch 2.10 or later +for your environment. + +### Install from source (recommended) + +```bash +git clone https://github.com/modelscope/entropack.git +cd entropack +pip install -e ".[cuda13]" +``` + +### Install from PyPI + +PyPI releases may lag behind source updates. Install from source for the latest features. + +```bash +pip install "entropack[cuda13]" +``` + +Both installation methods above use CUDA 13 and include the matching CuPy package. +For CUDA 12, replace `cuda13` with `cuda12` in either command. If a compatible CuPy is +already installed, use `pip install -e .` for source installation or `pip install entropack` for PyPI. + +Select PyTorch's CUDA variant when installing PyTorch. +Optional Triton kernels for INT8 computation can be enabled with `pip install triton`. +See [Compressed Linear usage](Linear-layers.md) for additional FP8 and INT8 hardware requirements. +The first call may be slower while CUDA kernels compile. Measure performance after warm-up. + +## Direct tensor compression + +This example selects lossy compression with `LatticeRANSConfig(target_bpp=3.5)`, compresses +a 2D BF16 tensor at a target of 3.5 bits per element, and restores its original shape and dtype. + +```python +import torch +import entropack as ep + +tensor = (torch.randn(256, 256, device="cuda") * 0.02).to(torch.bfloat16) +config = ep.LatticeRANSConfig(target_bpp=3.5) + +compressed = ep.compress(tensor, config) +restored = ep.decompress(compressed, config) + +print(f"Target: {config.target_bpp:.2f} bits per element") +print(f"Stored: {compressed.actual_bpp:.2f} bits per element") +print(restored.shape, restored.dtype) +``` + +`target_bpp` is measured in bits per element (bpp) and accepts integer or non-integer +values from 1 to 11. `actual_bpp` includes metadata and can differ from the target, +especially for small tensors. This scheme requires a nonempty 2D input without NaN or infinite values. + +See [Tensor compression](Tensor-compression.md) for reconstruction error, +device transfers, and saving and loading. [Compression configuration](Configuration.md) +covers lossless schemes and the complete parameter reference. + +## Use Compressed Linear + +`CompressedLinear.from_linear` compresses an existing layer's weights and returns a new +layer with a copy of the original bias: + +```python +import torch +import entropack as ep + +linear = torch.nn.Linear(256, 256, dtype=torch.bfloat16, device="cuda") +config = ep.LatticeRANSConfig(target_bpp=4.0) +layer = ep.CompressedLinear.from_linear(linear, config=config) +x = torch.randn(8, 256, dtype=torch.bfloat16, device="cuda") + +with torch.inference_mode(): + output = layer(x) +print(output.shape, f"{layer.compressed_bits:.2f} bits per weight") +``` + +For pretrained models, load the checkpoint before replacing the corresponding layers. +[Compressed Linear usage](Linear-layers.md) +covers replacing multiple layers, low-precision computation, and saving and loading compressed checkpoints. diff --git a/docs/en/Usage/Tensor-compression.md b/docs/en/Usage/Tensor-compression.md new file mode 100644 index 0000000..21a0a05 --- /dev/null +++ b/docs/en/Usage/Tensor-compression.md @@ -0,0 +1,116 @@ +# Tensor compression + +Use `compress(tensor, config)` to compress a weight or other tensor into a +`CompressedTensor`, then `decompress(compressed, config)` to restore it. +A lossless scheme preserves every input bit; a lossy scheme returns an approximation. + +## Compress and restore + +By default, the decompressed tensor has the same shape, `dtype`, and `device` as the input. +To select a different dtype or device, convert the input with `tensor.to(dtype=..., device=...)` +before compression. + +This example compresses a tensor at a 4 bpp target and measures the relative L2 error +after decompression. Encoding and decoding use a configuration from the same scheme: + +```python +import torch +import entropack as ep + +tensor = (torch.randn(128, 256, device="cuda") * 0.02).to(torch.bfloat16) +config = ep.LatticeRANSConfig(target_bpp=4.0) +compressed = ep.compress(tensor, config) +restored = ep.decompress(compressed, config) +relative_error = (restored.float() - tensor.float()).norm() / tensor.float().norm() + +print(f"Stored: {compressed.actual_bpp:.2f} bits per element") +print(f"Relative L2 error: {100 * relative_error:.2f}%") +``` + +To change the bitrate, compress the source tensor again with a new `target_bpp`. + +## Inspect stored size + +```python +import torch +import entropack as ep + +tensor = (torch.randn(128, 256, device="cuda") * 0.02).to(torch.bfloat16) +config = ep.LatticeRANSConfig(target_bpp=4.0) +compressed = ep.compress(tensor, config) + +print(compressed.shape, compressed.dtype, compressed.compress_method) +print(f"Stored: {compressed.storage_nbytes()} bytes") +print(f"Rate: {compressed.actual_bpp:.2f} bits per element") +``` + +`storage_nbytes()` reports the compressed size in bytes, including metadata needed for decompression. +`actual_bpp` is `8 * storage_nbytes() / tensor.numel()`. These measure the compressed +result, not the size of a checkpoint file or peak runtime memory. + +The achieved bitrate can differ from `target_bpp`, particularly for small tensors. +Compare the stored rate and reconstruction error when selecting a target. + +`CompressedTensor` also exposes `shape`, `dtype`, `compress_method`, and `lossless`. +The [API reference](../API_Reference/index.md) describes its remaining properties. + +## Move a compressed tensor + +`compressed.to(device)` returns a compressed tensor on the requested device. This example +moves it to CPU for storage, then back to the GPU for decompression: + +```python +import torch +import entropack as ep + +tensor = (torch.randn(128, 256, device="cuda") * 0.02).to(torch.bfloat16) +config = ep.DFloat11Config() +compressed = ep.compress(tensor, config) +cpu_copy = compressed.to("cpu") +gpu_copy = cpu_copy.to("cuda") +restored = ep.decompress(gpu_copy, config) +print(restored.device, restored.dtype) +``` + +Decompressing the object returned by `compressed.to(device)` restores the tensor on that device. + +## Save and load + +Save a compressed tensor's `state_dict()`, load it with +`torch.load(..., weights_only=True)`, and restore the `CompressedTensor` with +`CompressedTensor.from_state_dict()`: + +```python +from pathlib import Path + +import torch +import entropack as ep + +tensor = (torch.randn(128, 256, device="cuda") * 0.02).to(torch.bfloat16) +config = ep.DFloat11Config() +compressed = ep.compress(tensor, config) + +path = Path("compressed_tensor.pt") +torch.save(compressed.state_dict(), path) +state = torch.load(path, map_location="cuda", weights_only=True) +loaded = ep.CompressedTensor.from_state_dict(state) + +restored = ep.decompress(loaded, config) +assert torch.equal(restored.view(torch.uint8), tensor.view(torch.uint8)) +``` + +The example saves `compressed_tensor.pt` in the current directory. Change the path as needed. +`map_location` selects the device for loading. Keep the scheme configuration and library version +with the checkpoint. + +## Input requirements + +The lattice scheme requires a nonempty, finite, two-dimensional tensor. DFloat11 accepts +BF16, and Tile-ANS accepts the dtypes listed in [Config](Configuration.md). Lossless +schemes accept higher-dimensional tensors directly. For lattice compression, reshape +higher-dimensional data to 2D before compression and restore its outer shape after +decompression. + +If compression emits a warning, check `compress_method`: a value of `"raw"` means the +tensor was stored without compression. Invalid configurations and unsupported backend +selections raise errors. diff --git a/docs/en/conf.py b/docs/en/conf.py new file mode 100644 index 0000000..f7094c0 --- /dev/null +++ b/docs/en/conf.py @@ -0,0 +1,43 @@ +# Configuration file for the Sphinx documentation builder. + +import tomllib +from pathlib import Path + +# -- Project information ----------------------------------------------------- + +project = "entropack" +copyright = "2026, EntroPack Authors" +author = "EntroPack Authors" +html_theme = "sphinx_rtd_theme" +language = "en" + + +def get_version() -> str: + pyproject = Path(__file__).resolve().parents[2] / "pyproject.toml" + with pyproject.open("rb") as handle: + return tomllib.load(handle)["project"]["version"] + + +version = get_version() +release = version + +# -- General configuration --------------------------------------------------- + +extensions = [ + "sphinx_markdown_tables", + "sphinx_copybutton", + "sphinx_rtd_theme", + "sphinx.ext.mathjax", + "myst_parser", +] + +source_suffix = [".rst", ".md"] +root_doc = "index" +exclude_patterns = ["build", "_build"] + +# -- Extension configuration ------------------------------------------------- + +copybutton_prompt_text = r">>> |\.\.\. " +copybutton_prompt_is_regexp = True +intersphinx_mapping = {"https://docs.python.org/": None} +myst_enable_extensions = ["amsmath", "dollarmath", "colon_fence"] diff --git a/docs/en/index.rst b/docs/en/index.rst new file mode 100644 index 0000000..fa7d536 --- /dev/null +++ b/docs/en/index.rst @@ -0,0 +1,27 @@ +EntroPack Documentation +============================================== + +General-purpose tensor compression for PyTorch, with lossless and adjustable lossy modes. + +.. toctree:: + :maxdepth: 2 + :caption: Usage + + Usage/Quick-start + Usage/Configuration + Usage/Tensor-compression + Usage/Linear-layers + +.. toctree:: + :maxdepth: 2 + :caption: API reference + + API_Reference/index + +.. toctree:: + :maxdepth: 2 + :caption: Compression principles + + Principles/DFloat11 + Principles/Tile-ANS + Principles/Lattice-rANS diff --git a/docs/requirements.txt b/docs/requirements.txt new file mode 100644 index 0000000..bfd20f4 --- /dev/null +++ b/docs/requirements.txt @@ -0,0 +1,7 @@ +docutils>=0.16.0 +myst_parser +sphinx>=5.3.0 +sphinx-copybutton +sphinx-rtd-theme +sphinx_markdown_tables +pymdown-extensions diff --git a/docs/zh/.readthedocs.yaml b/docs/zh/.readthedocs.yaml new file mode 100644 index 0000000..fbb7868 --- /dev/null +++ b/docs/zh/.readthedocs.yaml @@ -0,0 +1,17 @@ +# .readthedocs.yaml +# Read the Docs configuration file +# See https://docs.readthedocs.io/en/stable/config-file/v2.html for details + +version: 2 + +build: + os: ubuntu-22.04 + tools: + python: "3.11" + +sphinx: + configuration: docs/zh/conf.py + +python: + install: + - requirements: docs/requirements.txt diff --git a/docs/zh/API_Reference/index.md b/docs/zh/API_Reference/index.md new file mode 100644 index 0000000..99b5b3d --- /dev/null +++ b/docs/zh/API_Reference/index.md @@ -0,0 +1,151 @@ +# API 参考 + +常用函数与类均可通过 `import entropack as ep` 访问。 +完整用例见[通用张量压缩](../Usage/Tensor-compression.md)和 +[Compressed Linear 使用指南](../Usage/Linear-layers.md),配置参数见[压缩配置](../Usage/Configuration.md)。 + +## 张量编解码 + +### compress + +```text +compress(tensor: torch.Tensor, config: CompressionConfig) -> CompressedTensor +``` + +按 `config` 指定的方案压缩 `tensor`,返回 `CompressedTensor`。 +解压后的张量与输入具有相同的形状和数据类型。 + +输入要求由方案决定:DFloat11 接受 BF16 张量,Tile-ANS 支持多种数据类型,格量化要求非空、有限值组成的二维张量。 + +| 参数 | 含义 | +| --- | --- | +| `tensor` | 待压缩的 PyTorch 张量 | +| `config` | 必传,使用 `DFloat11Config`、`TileANSConfig` 或 `LatticeRANSConfig` 选择方案 | + +部分编码失败会发出说明原因的警告,并返回 `compress_method == "raw"` 的未压缩容器。 +无效配置或后端选择失败会直接报错。 + +### decompress + +```text +decompress(compressed: CompressedTensor, config: CompressionConfig) -> torch.Tensor +``` + +返回与原始输入形状和数据类型相同的张量。无损方案逐位恢复输入值,有损方案返回近似重建。 +结果默认位于原始输入设备;通过 `compressed.to(device)` 等方式迁移容器后,结果位于迁移后的设备。 + +| 参数 | 含义 | +| --- | --- | +| `compressed` | `compress` 生成或从检查点加载的容器 | +| `config` | 对应压缩方案的配置,解码参数控制恢复过程 | + +解码时修改 `target_bpp` 等编码参数不会改变已保存的数据或重新量化张量。 + +## CompressedTensor + +保存一个张量的压缩表示。通常由 `compress` 返回,或由 `from_state_dict` 从检查点恢复。 + +### 常用属性 + +| 属性 | 类型 | 含义 | +| --- | --- | --- | +| `shape` | `tuple[int, ...]` | 原始张量的形状 | +| `dtype` | `torch.dtype` | 解压后的数据类型 | +| `compress_method` | `str` | 容器实际使用的压缩方案 | +| `lossless` | `bool` | 该方案是否无损 | +| `actual_bpp` | `float` | 每元素实际存储比特数,包含元数据 | + +`actual_bpp = 8 * storage_nbytes() / math.prod(shape)`。 +这项指标衡量压缩表示的大小,不等于检查点文件大小或运行时显存占用。 + +### 常用方法 + +| 方法 | 返回值 | 含义 | +| --- | --- | --- | +| `to(device)` | `CompressedTensor` | 返回位于指定设备的新容器,不改变数据类型或数值;仅接受设备参数 | +| `storage_nbytes(include_header=True)` | `int` | 压缩结果的总字节数,`include_header=False` 时不计容器头部 | +| `state_dict(prefix="")` | `dict[str, torch.Tensor]` | 将压缩张量导出为可保存的字典 | +| `CompressedTensor.from_state_dict(state, prefix="")` | `CompressedTensor` | 从上述字典恢复容器,不重新压缩 | + +保存与加载的 `prefix` 必须一致。可用 `torch.save` 保存字典,并用 +`torch.load(..., weights_only=True)` 加载,通过 `map_location` 指定恢复后的设备。 + +## CompressedLinear + +每次前向调用使用重建权重执行线性运算的层。权重以压缩形式保存,偏置不压缩。 +运行需要 CUDA GPU 和对应版本的 CuPy。 + +### 创建层 + +```text +CompressedLinear(in_features, out_features, bias=True, *, + config=None, device=None, dtype=torch.bfloat16) +CompressedLinear.from_linear(linear, **kwargs) -> CompressedLinear +``` + +| 构造参数 | 含义 | +| --- | --- | +| `in_features` / `out_features` | 输入与输出特征数 | +| `bias` | 是否包含偏置 | +| `config` | 权重压缩配置。默认 `None` 为 BF16 选择 DFloat11,为其他支持的数据类型选择 Tile-ANS | +| `device` | 直接构造时偏置所在的设备 | +| `dtype` | 压缩权重和初始化偏置时使用的数据类型 | + +`from_linear` 返回一个新层,压缩源层的权重并复制偏置。源权重必须已加载,不能位于 `meta` 设备。 +常用调用为 `ep.CompressedLinear.from_linear(linear, config=config)`。 +`kwargs` 可指定 `config` 或 `dtype`,其中 `dtype` 默认沿用源权重类型,设备自动沿用源层。 + +直接调用构造函数会创建尚无权重数据的层,需要再调用 `compress_weight` 或加载检查点后才能推理。 + +### 常用方法与属性 + +| 接口 | 返回值 | 含义 | +| --- | --- | --- | +| `compress_weight(weight)` | `None` | 压缩并替换层内权重,要求形状为 `(out_features, in_features)` | +| `dequantize(device=None)` | `torch.Tensor` | 返回数据类型为 `container_dtype` 的稠密权重;未指定 `device` 时位于层所在设备 | +| `forward(x)` | `torch.Tensor` | 对形状为 `(..., in_features)` 的输入 `x` 执行线性运算,返回形状为 `(..., out_features)` 的张量 | +| `compressed_weight` | `CompressedTensor` | 层持有的压缩权重容器 | +| `container_dtype` | `torch.dtype` | 压缩容器中权重或量化码的数据类型 | +| `stored_nbytes` | `int` | 权重存储字节数,含元数据和低精度层的量化尺度,不含偏置 | +| `compressed_bits` | `float` | `8 * stored_nbytes / (in_features * out_features)` | + +通过 `layer(x)` 调用前向运算。层的稠密 `.weight` 为 `None`,需要数值权重时使用 `dequantize()`。 + +使用标准 `state_dict()` / `load_state_dict()` 保存与恢复层状态。 +加载前须创建相同层类型、形状、容器数据类型和压缩方案的层,Config 对象本身不会保存在检查点中。 +`.to(device)` 可迁移层,模型的数据类型转换不会重新编码已压缩的权重。 + +## CompressedFP8Linear 与 CompressedINT8Linear + +权重和激活均使用 FP8 或 INT8 的线性层,沿用 `CompressedLinear` 的构造参数、 +`from_linear`、存储属性和检查点接口。输入形状为 `(..., in_features)` 时, +输出形状为 `(..., out_features)`,数据类型和设备与输入一致。 + +| 类 | 权重与激活格式 | CUDA GPU 要求 | +| --- | --- | --- | +| `CompressedFP8Linear` | FP8 E4M3FN | SM8.9 及以上 | +| `CompressedINT8Linear` | INT8 | SM8.0 及以上 | + +量化码格式由层类决定,构造参数 `dtype` 不改变 FP8 或 INT8 格式。 + +`config=None` 时直接保存量化码。指定 `LatticeRANSConfig` 时进一步进行有损压缩, +目标码率须满足 `1 <= target_bpp < 8`。`stored_nbytes` 包含重建权重所需的逐行量化尺度。 + +| 方法 | 返回值 | 含义 | +| --- | --- | --- | +| `codes(device=None)` | FP8 或 INT8 张量 | 恢复量化码,尚未乘回行尺度 | +| `dequantize(device=None)` | FP32 张量 | 将恢复的量化码乘回行尺度,得到数值权重 | + +两种方法未指定 `device` 时,返回张量均位于层所在设备。有损压缩后的量化码可能与初始量化结果不同。 + +## Config 类 + +Config 同时用于张量编解码和 Compressed Linear 的权重存储。以下三个类继承自 `CompressionConfig`: + +| 类 | 用途 | +| --- | --- | +| `DFloat11Config` | BF16 无损压缩 | +| `TileANSConfig` | 多种数据类型的分块 ANS 无损压缩 | +| `LatticeRANSConfig` | 以 `target_bpp` 控制码率的格量化有损压缩 | + +方案选择、默认值和参数范围见[压缩配置](../Usage/Configuration.md)。 diff --git a/docs/zh/Makefile b/docs/zh/Makefile new file mode 100644 index 0000000..45d227d --- /dev/null +++ b/docs/zh/Makefile @@ -0,0 +1,14 @@ +# Minimal makefile for Sphinx documentation + +SPHINXOPTS ?= +SPHINXBUILD ?= sphinx-build +SOURCEDIR = . +BUILDDIR = ../_build/zh + +help: + @$(SPHINXBUILD) -M help "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O) + +.PHONY: help Makefile + +%: Makefile + @$(SPHINXBUILD) -M $@ "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O) diff --git a/docs/zh/Principles/DFloat11.md b/docs/zh/Principles/DFloat11.md new file mode 100644 index 0000000..efe0dff --- /dev/null +++ b/docs/zh/Principles/DFloat11.md @@ -0,0 +1,45 @@ +# DFloat11 + +DFloat11 对 BF16 张量进行无损压缩,完整保留原始位表示。许多张量的指数值集中在较小的范围内, +因此可以用较短的编码表示常见指数,同时原样保存符号位和尾数部分。 + +## 编码 + +一个 BF16 数值包含 1 位符号、8 位指数和 7 位尾数。编码器将它拆成两部分: +指数单独组成符号序列,符号位和尾数则合并成一个字节直接存储。 +编码器统计当前张量的指数频率,再构建 Huffman 编码表。常见指数使用较短的编码, +不常见指数使用较长的编码,从而减少整个指数序列占用的空间。 + +Huffman 编码长度不固定,因此解码器无法从任意一位直接识别下一个符号。 +EntroPack 为编码区域保存起始位置和符号数量,使不同区域可以并行解码。 +这些入口信息和 Huffman 表构成压缩表示中的元数据开销。 + +## 解码与存储大小 + +解码器通过 Huffman 表恢复指数序列,再将每个指数与对应的符号位、尾数重新组合, +得到原始 BF16 位表示,并按照保存的形状组织为输出张量。整个过程不涉及数值量化或舍入。 + +实际存储大小取决于指数分布和解码所需的元数据。指数越集中,通常越容易压缩。 +对于较小的张量,元数据占比也会更高。`DFloat11Config` 不设置目标码率, +DFloat11 这一名称也不意味着所有张量都恰好以每元素 11 bit 存储。 + +## 使用示例 + +以下示例压缩一个 BF16 张量,并检查解压后的位表示是否与输入一致。 +编码区域相关参数见 [Config](../Usage/Configuration.md)。 + +```python +import torch +import entropack as ep + +tensor = (torch.randn(256, 256, device="cuda") * 0.02).to(torch.bfloat16) +config = ep.DFloat11Config() + +compressed = ep.compress(tensor, config) +restored = ep.decompress(compressed, config) + +assert restored.shape == tensor.shape +assert restored.dtype == tensor.dtype +assert torch.equal(restored.view(torch.uint8), tensor.view(torch.uint8)) +print(f"Stored: {compressed.actual_bpp:.2f} bits per element") +``` diff --git a/docs/zh/Principles/Lattice-rANS.md b/docs/zh/Principles/Lattice-rANS.md new file mode 100644 index 0000000..adc81aa --- /dev/null +++ b/docs/zh/Principles/Lattice-rANS.md @@ -0,0 +1,64 @@ +# EntroPack 格量化压缩 + +EntroPack 的格量化方案结合有损向量量化与无损熵编码,压缩二维张量。 +`LatticeRANSConfig` 支持每元素 1 至 11 bit 的目标,包括非整数码率。 +解压后仍保留输入的数据类型,存储码率则通过目标参数调节。 + +![EntroPack 编码与解码流程](../../assets/entropack-pipeline.png) + +以权重矩阵为例的 EntroPack 编解码流程。上半部分为码率搜索与编码,下半部分为融合 GPU 解码与重建。 + +## 格量化与整数字段 + +不同张量行的数值尺度可能相差较大。编码器首先用每行的均方根归一化该行, +再将归一化后的数值每八个组成一个向量。E8 格是八维空间中按规则排列的一组点, +编码器用缩放后的格中最近的点近似每个向量。共享的量化尺度控制格点之间的间距。 +间距越小,通常重建误差越小,但描述所选格点需要的比特也越多。 + +E8 包含整数坐标与半整数坐标两类格点,对应两个陪集。 +EntroPack 用陪集标记和八个可逆整数字段表示格点,并利用奇偶约束压缩最后一个坐标的表示。 +概率模型根据陪集分别统计各坐标字段的分布,以捕捉两类格点的差异。 +频繁出现的字段值平均可用更少的比特表示。解码时,这些字段可通过算术运算还原为格点,无需重建码本。 + +## 码率选择与精度优化 + +编码器在采样行上搜索量化尺度,对每个候选尺度估计字段的编码大小及解码所需的元数据, +无需在搜索过程中反复生成完整压缩码流。选定尺度后,再量化整个张量, +并通过最小二乘拟合每行的重建尺度。 + +可选的逐行率失真优化会为每行比较多个量化精度,在估计的存储预算内分配候选。 +这一过程交替进行候选选择和共享概率模型更新,需要额外的编码计算,默认关闭。 + +## 编码与重建 + +选定的字段由 rANS 熵编码为可独立解码的 tile。 +压缩表示还保存概率表、行尺度和 tile 定位信息。 +解码时,GPU 在融合操作中恢复字段、重建格点并应用行尺度,输出与输入形状和 dtype 相同的张量。 + +重建误差来自量化以及转换回输出 dtype 时的舍入,熵编码本身完整保留选定的字段。 +由于尺度搜索使用大小估计,实际码率可能与目标有差别。 +`actual_bpp` 按包含元数据的实际存储字节数计算,即字节数乘以八,再除以张量元素数。 + +## 使用示例 + +以下示例以每元素 3.5 bit 为目标压缩张量,并以原始张量为参考计算相对 L2 误差。 +搜索与精度优化参数见 [Config](../Usage/Configuration.md)。 + +```python +import torch +import entropack as ep + +tensor = (torch.randn(256, 256, device="cuda") * 0.02).to(torch.bfloat16) +config = ep.LatticeRANSConfig(target_bpp=3.5) + +compressed = ep.compress(tensor, config) +restored = ep.decompress(compressed, config) + +reference = tensor.float() +relative_l2 = (restored.float() - reference).norm() / reference.norm() +assert restored.shape == tensor.shape +assert restored.dtype == tensor.dtype +print(f"Target: {config.target_bpp:.2f} bits per element") +print(f"Stored: {compressed.actual_bpp:.2f} bits per element") +print(f"Relative L2 error: {100 * relative_l2.item():.2f}%") +``` diff --git a/docs/zh/Principles/Tile-ANS.md b/docs/zh/Principles/Tile-ANS.md new file mode 100644 index 0000000..0929666 --- /dev/null +++ b/docs/zh/Principles/Tile-ANS.md @@ -0,0 +1,48 @@ +# Tile-ANS + +Tile-ANS 通过编码张量的存储字节实现无损压缩,支持 BF16、FP16、FP32、FP8、INT8 等 +浮点和整数类型。它处理的是数值的位表示,不对数值进行近似,因此解压后能够完整恢复原始数据。 + +## 字节流与概率表 + +同一种数值格式中,不同字节位置的分布往往不同。Tile-ANS 按字节在元素内部的位置, +将张量拆成多个字节流。例如,两字节格式对应两个流,四字节格式对应四个流。 +每个流收集所有元素在对应位置上的字节。 + +编码器分别统计这些流的字节频率,为每个流构建概率表。 +当分布较集中时,常见字节平均使用更短的表示,从而减少存储空间。 +对于计入编码开销后压缩收益仍较小的流,编码器直接存储原始字节。 +因此,同一个张量中可以同时存在熵编码流和直接存储的流。 + +## 分块编码与解码 + +每个流进一步划分为可独立解码的 tile。需要熵编码的 tile 使用范围非对称数字系统 rANS, +通过可逆的整数状态更新编码符号。一个 tile 内部交错使用多个编码状态,支持并行恢复符号。 +不同 tile 之间也可以独立解码。同一字节流的所有 tile 共享概率表,无需逐 tile 存储一份表。 + +解码器使用相同的概率表逆转状态更新,恢复各个经过熵编码的字节流, +再将恢复出的字节流与直接存储的字节流合并,把字节放回元素内的原始位置,重建张量。 + +该过程不涉及量化。实际大小由字节分布和元数据共同决定,因此不能指定一个有损压缩式的目标码率, +也不保证每个输入都能缩小。较大的 tile 可以降低每元素的元数据开销,较小的 tile 则提供更多独立解码任务。 + +## 使用示例 + +以下示例压缩一个 FP16 张量,并检查解压后的位表示是否与输入一致。 +分块大小和概率表参数见 [Config](../Usage/Configuration.md)。 + +```python +import torch +import entropack as ep + +tensor = (torch.randn(256, 256, device="cuda") * 0.02).to(torch.float16) +config = ep.TileANSConfig() + +compressed = ep.compress(tensor, config) +restored = ep.decompress(compressed, config) + +assert restored.shape == tensor.shape +assert restored.dtype == tensor.dtype +assert torch.equal(restored.view(torch.uint8), tensor.view(torch.uint8)) +print(f"Stored: {compressed.actual_bpp:.2f} bits per element") +``` diff --git a/docs/zh/Usage/Configuration.md b/docs/zh/Usage/Configuration.md new file mode 100644 index 0000000..fe14d54 --- /dev/null +++ b/docs/zh/Usage/Configuration.md @@ -0,0 +1,97 @@ +# 压缩配置 + +Config 选择压缩方案并设置编解码参数。`execution_backend` 默认为 `"auto"`,自动选择计算后端。 + +## 选择配置 + +| Config | 压缩方式 | 输入要求 | +| --- | --- | --- | +| `DFloat11Config()` | BF16 无损压缩 | BF16 张量 | +| `TileANSConfig()` | 多种数据类型的无损压缩 | 支持的数据类型 | +| `LatticeRANSConfig(target_bpp=...)` | 按目标码率进行有损压缩 | 非空、有限值组成的二维张量 | + +两个无损方案均逐位恢复输入,压缩后的大小取决于张量的数据分布。`TileANSConfig` 也支持 BF16, +因此 BF16 张量可以选择其中任一无损方案。`LatticeRANSConfig` 接受每元素 1–11 bit 的目标码率, +支持非整数值,实际存储码率可通过 `CompressedTensor.actual_bpp` 查看。 + +## 支持的张量数据类型 + +| 数据类型 | `DFloat11Config` | `TileANSConfig` | `LatticeRANSConfig` | +|---|---|---|---| +| `float32` | — | 无损 | 有损 | +| `float16` | — | 无损 | 有损 | +| `bfloat16` | 无损 | 无损 | 有损 | +| `float8_e4m3fn` | — | 无损 | 有损 | +| `float8_e4m3fnuz` | — | 无损 | 有损 | +| `float8_e5m2` | — | 无损 | 有损 | +| `float8_e5m2fnuz` | — | 无损 | 有损 | +| `int64` | — | 无损 | 有损 | +| `int32` | — | 无损 | 有损 | +| `int16` | — | 无损 | 有损 | +| `int8` | — | 无损 | 有损 | +| `uint64` | — | 无损 | 有损 | +| `uint32` | — | 无损 | 有损 | +| `uint16` | — | 无损 | 有损 | +| `uint8` | — | 无损 | 有损 | +| `bool` | — | 无损 | 有损 | + +无损方案接受多种形状的张量,格量化要求二维输入。文中的 BF16、FP16、FP8 对应上表列出的 +PyTorch 数据类型。打包的四比特格式、FP64 和复数类型不在支持范围内。 + +## 参数说明 + +修改编码参数只影响后续压缩,不会改变已有压缩结果。解码参数在恢复张量时生效。 +多数设置可保留默认值。 +有损压缩的存储码率由 `target_bpp` 控制,`row_rdo_iterations` 设为正数时启用逐行率失真优化(RDO)。 + +## CompressionConfig + +以下执行参数适用于三个配置类。 + +| 字段 | 类型 | 默认值 | 阶段 | 含义 | +| --- | --- | --- | --- | --- | +| `execution_backend` | `str` 或 `None` | `"auto"` | 编解码 | `"auto"` 或 `None` 优先选择 CUDA,不可用时回退到 PyTorch 实现;`"cuda"` 强制使用 CUDA。`"eager"` 为张量编解码的 PyTorch 后备实现。 | + +## DFloat11Config + +用于 BF16 无损压缩。 + +| 字段 | 类型 | 默认值 | 阶段 | 含义 | +| --- | --- | --- | --- | --- | +| `execution_backend` | `str` 或 `None` | `"auto"` | 编解码 | 后端选择,含义同上 | +| `bytes_per_thread` | 正整数或 `None` | `16` | 编码 | 每个线程处理的编码字节数,影响压缩率和解码并行度 | +| `threads_per_block` | 正整数或 `None` | `128` | 编码 | 编码时每个线程块的线程数 | + +## TileANSConfig + +用于支持的数据类型的无损压缩。 + +| 字段 | 类型 | 默认值 | 阶段 | 含义 | +| --- | --- | --- | --- | --- | +| `execution_backend` | `str` 或 `None` | `"auto"` | 编解码 | 后端选择,含义同上 | +| `tile_elements` | [0, 2^31 − 1] 内整数 | `0` | 编码 | 每个压缩块的元素数,影响压缩率和解码并行度。`0` 自动选择。 | +| `probability_bits` | `0`、`9`、`10`、`11`、`12` | `0` | 编码 | 概率表精度,`0` 自动选择 | +| `raw_lane_threshold` | [0, 8] 内浮点数 | `7.9` | 编码 | 决定何时直接存储难以压缩的数据,单位为每字节的预计编码比特数 | +| `threads_per_block` | 正整数或 `None` | `None` | 编解码 | GPU 线程块宽度,`None` 自动选择 | + +## LatticeRANSConfig + +用于有限值组成的二维张量的有损压缩。 + +| 字段 | 类型 | 默认值 | 阶段 | 含义 | +| --- | --- | --- | --- | --- | +| `execution_backend` | `str` 或 `None` | `"auto"` | 编解码 | 后端选择,含义同上 | +| `target_bpp` | [1, 11] 内浮点数 | `4.0` | 编码 | 每个输入元素的目标比特数,支持非整数。实际码率通过 `actual_bpp` 查看。 | +| `prob_bits` | [9, 15] 内整数、`0` 或 `None` | `None` | 编码 | 概率表精度,`None` 或 `0` 自动选择 | +| `tile_elements` | 正整数或 `None` | `None` | 编码 | 每个压缩块的元素数,影响压缩率和解码并行度。`None` 自动选择。 | +| `row_rdo_iterations` | [0, 8] 内整数 | `0` | 编码 | 逐行率失真优化的轮数,`0` 关闭。更多轮次会增加压缩耗时。 | +| `row_rdo_candidates` | 正整数 | `5` | 编码 | RDO 为每行比较的候选量化结果数,更多候选会增加压缩耗时 | +| `scale_search_iterations` | 正整数 | `12` | 编码 | 量化尺度搜索的迭代次数 | +| `scale_search_max_vectors` | 正整数 | `262144` | 编码 | 码率搜索的采样上限,每个向量包含八个元素 | +| `threads_per_block` | 正整数或 `None` | `None` | 解码 | GPU 线程块宽度,`None` 自动选择 | +| `l2_prefetch` | `bool` | `True` | 解码 | 解码时启用 GPU L2 缓存预取 | + +## 压缩原理 + +各方案的编解码过程分别见 [DFloat11](../Principles/DFloat11.md)、 +[Tile-ANS](../Principles/Tile-ANS.md) 和 [Lattice-rANS](../Principles/Lattice-rANS.md)。 diff --git a/docs/zh/Usage/Linear-layers.md b/docs/zh/Usage/Linear-layers.md new file mode 100644 index 0000000..c6c0f78 --- /dev/null +++ b/docs/zh/Usage/Linear-layers.md @@ -0,0 +1,155 @@ +# Compressed Linear 使用指南 + +Compressed Linear 将 PyTorch 线性层替换为使用压缩权重的层。 +仍通过 `layer(x)` 调用:输入的最后一维从 `in_features` 变为 `out_features`, +其余维度不变。[Config](Configuration.md) 指定压缩方案和参数。 + +Compressed Linear 需要 CUDA GPU 和对应版本的 CuPy,安装方式见[快速上手](Quick-start.md)。 + +`CompressedLinear` 可使用 `DFloat11Config`、`TileANSConfig` 或 `LatticeRANSConfig`, +所选方案需支持权重的数据类型。 + +## 替换一个已有层 + +`from_linear` 压缩现有层的权重并返回一个新层,偏置保持原样复制。 +对于预训练模型,应先加载检查点,再调用该方法: + +```python +import torch +import entropack as ep + +linear = torch.nn.Linear(256, 256, dtype=torch.bfloat16, device="cuda") +config = ep.LatticeRANSConfig(target_bpp=4.0) +layer = ep.CompressedLinear.from_linear(linear, config=config) +x = torch.randn(8, 256, dtype=torch.bfloat16, device="cuda") + +with torch.inference_mode(): + output = layer(x) +print(output.shape, f"{layer.compressed_bits:.2f} bits per weight") +``` + +将返回的层赋给模型中的对应属性后,即可使用压缩版本。 + +## 替换模型中的多个层 + +下面的例子递归替换普通线性层,并让输出层保留原来的格式。 +`skip` 使用 `named_modules()` 中的模块路径。 + +```python +import torch +import entropack as ep + +model = torch.nn.Sequential( + torch.nn.Linear(256, 256), + torch.nn.GELU(), + torch.nn.Linear(256, 64), +).to(device="cuda", dtype=torch.bfloat16).eval() +config = ep.LatticeRANSConfig(target_bpp=4.0) + + +def compress_linears(module, config, skip=(), prefix=""): + for name, child in list(module.named_children()): + path = f"{prefix}.{name}" if prefix else name + if path in skip: + continue + if type(child) is torch.nn.Linear: + replacement = ep.CompressedLinear.from_linear(child, config=config) + setattr(module, name, replacement.train(child.training)) + else: + compress_linears(child, config, skip, path) + + +compress_linears(model, config, skip={"2"}) +x = torch.randn(8, 256, device="cuda", dtype=torch.bfloat16) +with torch.inference_mode(): + output = model(x) +print(output.shape, type(model[0]).__name__, type(model[2]).__name__) +``` + +示例仅选择标准 `torch.nn.Linear`,自定义线性层或共享权重需要结合模型处理。 +省略 `skip` 即可压缩所有普通线性层。如果 `CompressedLinear` 无法压缩某层的权重, +替换时会报错;可通过 `skip` 让该层保留原始格式。 + +## 结合 FP8 或 INT8 计算 + +| 层 | 权重格式 | 计算方式 | +| --- | --- | --- | +| `CompressedLinear` | 输入权重的数据类型 | 使用激活数据类型进行普通线性运算 | +| `CompressedFP8Linear` | FP8 E4M3FN 量化码 | FP8 权重和激活,需 CUDA GPU(SM8.9 及以上) | +| `CompressedINT8Linear` | INT8 量化码 | INT8 权重和激活,需 CUDA GPU(SM8.0 及以上) | + +对于 `CompressedFP8Linear` 和 `CompressedINT8Linear`,`config=None` 仅做 FP8 或 INT8 量化, +传入 `LatticeRANSConfig(target_bpp=...)` 则会对量化后的权重进一步进行有损压缩。 +目标码率需大于等于 1 bpp 且低于 8 bpp。 + +```python +import torch +import entropack as ep + +linear = torch.nn.Linear(256, 256, dtype=torch.bfloat16, device="cuda") +layer = ep.CompressedINT8Linear.from_linear( + linear, config=ep.LatticeRANSConfig(target_bpp=4.0) +) +x = torch.randn(32, 256, dtype=torch.bfloat16, device="cuda") +with torch.inference_mode(): + output = layer(x) +print(output.shape, layer.container_dtype, f"{layer.compressed_bits:.2f} bits per weight") +``` + +在支持的硬件上,FP8 可以按同样方式使用 `CompressedFP8Linear`。 +需要查看权重时,`codes()` 返回 FP8 或 INT8 数值,`dequantize()` 返回反量化后的浮点数值。 + +## 统计存储 + +`stored_nbytes` 统计压缩权重的总字节数,包含元数据及 FP8、INT8 的量化尺度。 +`compressed_bits` 等于 `8 * stored_nbytes / (in_features * out_features)`。 +统计多个层时,应先分别累加字节数与权重元素数,再计算比例。偏置不计入这项权重存储指标。 + +这项指标衡量权重存储大小,不代表推理时的峰值显存。 + +## 保存与加载模型 + +保存模型的 `state_dict` 后,先构造具有相同结构、使用相同 Compressed Linear 类的模型,再加载状态。 +以下示例压缩两个层,并将保存的权重加载到一个新模型中: + +```python +from pathlib import Path + +import torch +import entropack as ep + +config = ep.LatticeRANSConfig(target_bpp=4.0) +model = torch.nn.Sequential( + torch.nn.Linear(256, 256), + torch.nn.GELU(), + torch.nn.Linear(256, 64), +).to(device="cuda", dtype=torch.bfloat16) +for index in (0, 2): + model[index] = ep.CompressedLinear.from_linear(model[index], config=config) +model.eval() + +x = torch.randn(8, 256, device="cuda", dtype=torch.bfloat16) +with torch.inference_mode(): + expected = model(x) + +path = Path("compressed_model.pt") +torch.save(model.state_dict(), path) + +restored = torch.nn.Sequential( + ep.CompressedLinear(256, 256, config=config, device="cuda", dtype=torch.bfloat16), + torch.nn.GELU(), + ep.CompressedLinear(256, 64, config=config, device="cuda", dtype=torch.bfloat16), +).eval() +state = torch.load(path, map_location="cuda", weights_only=True) +restored.load_state_dict(state) +with torch.inference_mode(): + actual = restored(x) + +assert torch.allclose(actual, expected) +print(actual.shape) +``` + +示例将检查点保存到当前目录的 `compressed_model.pt`,可按需修改路径。 +加载已有检查点时,构造 `restored` 模型,再调用 `torch.load` 和 `load_state_dict`。 +应随检查点保留模型结构、层名及类型、压缩配置、权重数据类型和库版本。 +普通 `torch.nn.Linear` 无法直接加载 Compressed Linear 的检查点。 diff --git a/docs/zh/Usage/Quick-start.md b/docs/zh/Usage/Quick-start.md new file mode 100644 index 0000000..2fda2c4 --- /dev/null +++ b/docs/zh/Usage/Quick-start.md @@ -0,0 +1,81 @@ +# 快速上手 + +本页介绍安装、张量编解码和 Compressed Linear 的基本用法。 + +## 安装 + +需要 Python 3.10 及以上,并先安装与环境匹配的 CUDA 版 PyTorch 2.10 及以上。 + +### 源码安装(推荐) + +```bash +git clone https://github.com/modelscope/entropack.git +cd entropack +pip install -e ".[cuda13]" +``` + +### 从 PyPI 安装 + +PyPI 版本更新可能有所延迟,如需最新功能,推荐从源码安装。 + +```bash +pip install "entropack[cuda13]" +``` + +上述两种安装方式均以 CUDA 13 为例,并包含对应版本的 CuPy。使用 CUDA 12 时, +将命令中的 `cuda13` 改为 `cuda12`。如果已安装匹配的 CuPy,源码安装和 PyPI 安装 +可分别使用 `pip install -e .` 和 `pip install entropack`。 + +PyTorch 的 CUDA 版本需在安装 PyTorch 时选定。 +INT8 计算可通过 `pip install triton` 启用可选的 Triton 内核。 +FP8 和 INT8 的额外硬件要求见 [Compressed Linear 使用指南](Linear-layers.md)。 +首次调用需要编译 CUDA 内核,可能比后续调用更慢,性能计时应在预热后进行。 + +## 直接压缩张量 + +以下示例使用 `LatticeRANSConfig(target_bpp=3.5)` 选择有损压缩,将二维 BF16 张量 +压缩到每元素 3.5 bit 的目标码率,再恢复为原来的形状和数据类型。 + +```python +import torch +import entropack as ep + +tensor = (torch.randn(256, 256, device="cuda") * 0.02).to(torch.bfloat16) +config = ep.LatticeRANSConfig(target_bpp=3.5) + +compressed = ep.compress(tensor, config) +restored = ep.decompress(compressed, config) + +print(f"Target: {config.target_bpp:.2f} bits per element") +print(f"Stored: {compressed.actual_bpp:.2f} bits per element") +print(restored.shape, restored.dtype) +``` + +`target_bpp` 的单位为每元素比特数(bpp),可设为 1–11 范围内的整数或非整数值。 +`actual_bpp` 返回包含元数据的实际存储码率,可能与目标不同,尤其在张量较小时。 +该方案要求输入为非空的二维张量,且不含 NaN 或无穷值。 + +重建误差、设备迁移和保存加载见[通用张量压缩](Tensor-compression.md)。 +无损方案及完整参数见[压缩配置](Configuration.md)。 + +## 使用 Compressed Linear + +`CompressedLinear.from_linear` 压缩现有层的权重并返回一个新层,偏置保持原样复制: + +```python +import torch +import entropack as ep + +linear = torch.nn.Linear(256, 256, dtype=torch.bfloat16, device="cuda") +config = ep.LatticeRANSConfig(target_bpp=4.0) +layer = ep.CompressedLinear.from_linear(linear, config=config) +x = torch.randn(8, 256, dtype=torch.bfloat16, device="cuda") + +with torch.inference_mode(): + output = layer(x) +print(output.shape, f"{layer.compressed_bits:.2f} bits per weight") +``` + +对于预训练模型,应先加载检查点,再将模型中的对应层替换为新层。 +[Compressed Linear 使用指南](Linear-layers.md)介绍多个层的替换、 +低精度计算和压缩检查点的保存加载。 diff --git a/docs/zh/Usage/Tensor-compression.md b/docs/zh/Usage/Tensor-compression.md new file mode 100644 index 0000000..ed24044 --- /dev/null +++ b/docs/zh/Usage/Tensor-compression.md @@ -0,0 +1,110 @@ +# 通用张量压缩 + +使用 `compress(tensor, config)` 将权重或其他张量压缩为 `CompressedTensor`, +再通过 `decompress(compressed, config)` 恢复张量。 +无损方案逐位还原输入,有损方案返回近似结果。 + +## 压缩与恢复 + +解压后的张量默认与输入张量具有相同的形状、`dtype` 和 `device`。 +如需指定数据类型或设备,在压缩前用 `tensor.to(dtype=..., device=...)` 转换输入即可。 + +以下示例使用 4 bpp 的目标码率压缩张量,并计算解压后的相对 L2 误差。 +压缩与解压使用同一方案的配置: + +```python +import torch +import entropack as ep + +tensor = (torch.randn(128, 256, device="cuda") * 0.02).to(torch.bfloat16) +config = ep.LatticeRANSConfig(target_bpp=4.0) +compressed = ep.compress(tensor, config) +restored = ep.decompress(compressed, config) +relative_error = (restored.float() - tensor.float()).norm() / tensor.float().norm() + +print(f"Stored: {compressed.actual_bpp:.2f} bits per element") +print(f"Relative L2 error: {100 * relative_error:.2f}%") +``` + +需要改变码率时,应使用新的 `target_bpp` 重新压缩源张量。 + +## 查看实际存储 + +```python +import torch +import entropack as ep + +tensor = (torch.randn(128, 256, device="cuda") * 0.02).to(torch.bfloat16) +config = ep.LatticeRANSConfig(target_bpp=4.0) +compressed = ep.compress(tensor, config) + +print(compressed.shape, compressed.dtype, compressed.compress_method) +print(f"Stored: {compressed.storage_nbytes()} bytes") +print(f"Rate: {compressed.actual_bpp:.2f} bits per element") +``` + +`storage_nbytes()` 统计压缩结果的总字节数,包含恢复张量所需的元数据, +`actual_bpp` 等于 `8 * storage_nbytes() / tensor.numel()`。 +它们衡量压缩结果的大小,不代表检查点文件大小或运行时峰值显存。 + +实际码率可能与 `target_bpp` 有所不同,小张量尤其如此。 +选择目标码率时,应同时比较实际存储和重建误差。 + +`CompressedTensor` 还提供 `shape`、`dtype`、`compress_method` 和 `lossless`。 +其他属性见 [API 参考](../API_Reference/index.md)。 + +## 移动压缩张量 + +`compressed.to(device)` 返回位于指定设备的压缩张量。以下示例先将其转存到 CPU, +再移回 GPU 解压: + +```python +import torch +import entropack as ep + +tensor = (torch.randn(128, 256, device="cuda") * 0.02).to(torch.bfloat16) +config = ep.DFloat11Config() +compressed = ep.compress(tensor, config) +cpu_copy = compressed.to("cpu") +gpu_copy = cpu_copy.to("cuda") +restored = ep.decompress(gpu_copy, config) +print(restored.device, restored.dtype) +``` + +对 `compressed.to(device)` 返回的压缩张量解压,得到的张量也位于该设备上。 + +## 保存与加载 + +保存压缩张量的 `state_dict()`,用 `torch.load(..., weights_only=True)` 加载后, +通过 `CompressedTensor.from_state_dict()` 恢复 `CompressedTensor`: + +```python +from pathlib import Path + +import torch +import entropack as ep + +tensor = (torch.randn(128, 256, device="cuda") * 0.02).to(torch.bfloat16) +config = ep.DFloat11Config() +compressed = ep.compress(tensor, config) + +path = Path("compressed_tensor.pt") +torch.save(compressed.state_dict(), path) +state = torch.load(path, map_location="cuda", weights_only=True) +loaded = ep.CompressedTensor.from_state_dict(state) + +restored = ep.decompress(loaded, config) +assert torch.equal(restored.view(torch.uint8), tensor.view(torch.uint8)) +``` + +示例将检查点保存到当前目录的 `compressed_tensor.pt`,可按需修改路径。 +`map_location` 指定加载设备。建议随检查点保留对应的配置和库版本。 + +## 输入要求 + +格量化方案要求非空、有限值组成的二维张量。DFloat11 接受 BF16, +Tile-ANS 接受 [Config](Configuration.md) 中列出的数据类型。无损方案可直接接受高维张量。 +使用格量化压缩高维数据时,需先转换为二维布局,并在解压后还原原始形状。 + +压缩时若出现警告,可检查 `compress_method`:值为 `"raw"` 表示张量未经压缩就被保存。 +无效配置或不支持的后端选择会直接报错。 diff --git a/docs/zh/conf.py b/docs/zh/conf.py new file mode 100644 index 0000000..3ec7ef8 --- /dev/null +++ b/docs/zh/conf.py @@ -0,0 +1,43 @@ +# Configuration file for the Sphinx documentation builder. + +import tomllib +from pathlib import Path + +# -- Project information ----------------------------------------------------- + +project = "entropack" +copyright = "2026, EntroPack Authors" +author = "EntroPack Authors" +html_theme = "sphinx_rtd_theme" +language = "zh_CN" + + +def get_version() -> str: + pyproject = Path(__file__).resolve().parents[2] / "pyproject.toml" + with pyproject.open("rb") as handle: + return tomllib.load(handle)["project"]["version"] + + +version = get_version() +release = version + +# -- General configuration --------------------------------------------------- + +extensions = [ + "sphinx_markdown_tables", + "sphinx_copybutton", + "sphinx_rtd_theme", + "sphinx.ext.mathjax", + "myst_parser", +] + +source_suffix = [".rst", ".md"] +root_doc = "index" +exclude_patterns = ["build", "_build"] + +# -- Extension configuration ------------------------------------------------- + +copybutton_prompt_text = r">>> |\.\.\. " +copybutton_prompt_is_regexp = True +intersphinx_mapping = {"https://docs.python.org/": None} +myst_enable_extensions = ["amsmath", "dollarmath", "colon_fence"] diff --git a/docs/zh/index.rst b/docs/zh/index.rst new file mode 100644 index 0000000..50db26e --- /dev/null +++ b/docs/zh/index.rst @@ -0,0 +1,27 @@ +EntroPack 文档 +======================== + +面向 PyTorch 的通用张量压缩,支持无损压缩与目标码率可调的有损压缩。 + +.. toctree:: + :maxdepth: 2 + :caption: 使用文档 + + Usage/Quick-start + Usage/Configuration + Usage/Tensor-compression + Usage/Linear-layers + +.. toctree:: + :maxdepth: 2 + :caption: API 参考 + + API_Reference/index + +.. toctree:: + :maxdepth: 2 + :caption: 压缩原理 + + Principles/DFloat11 + Principles/Tile-ANS + Principles/Lattice-rANS diff --git a/entropack/__init__.py b/entropack/__init__.py new file mode 100644 index 0000000..203b22b --- /dev/null +++ b/entropack/__init__.py @@ -0,0 +1,15 @@ +from importlib import metadata + +try: + __version__ = metadata.version("entropack") +except metadata.PackageNotFoundError: + __version__ = "0+unknown" + +from .compression import CompressedTensor, compress, decompress +from .linear import CompressedFP8Linear, CompressedINT8Linear, CompressedLinear +from .schemes import CompressionConfig, DFloat11Config, LatticeRANSConfig, RawConfig, TileANSConfig + +__all__ = [ + "CompressedFP8Linear", "CompressedINT8Linear", "CompressedLinear", "CompressedTensor", "CompressionConfig", + "DFloat11Config", "LatticeRANSConfig", "RawConfig", "TileANSConfig", "__version__", "compress", "decompress", +] diff --git a/entropack/backends/__init__.py b/entropack/backends/__init__.py new file mode 100644 index 0000000..137e6a9 --- /dev/null +++ b/entropack/backends/__init__.py @@ -0,0 +1,19 @@ +from collections.abc import Callable +from dataclasses import dataclass + +import torch + +from . import cuda + + +@dataclass(frozen=True) +class Backend: + name: str + priority: int = 0 + probe: Callable[[], str | None] | None = None + device: Callable[[], torch.device] | None = None + + +BACKENDS: tuple[Backend, ...] = ( + Backend("cuda", priority=100, probe=cuda.probe, device=cuda.runs_on), Backend("eager", priority=0), +) diff --git a/entropack/backends/cuda/__init__.py b/entropack/backends/cuda/__init__.py new file mode 100644 index 0000000..452877a --- /dev/null +++ b/entropack/backends/cuda/__init__.py @@ -0,0 +1,16 @@ +import torch + + +def probe() -> str | None: + if not torch.cuda.is_available(): + return "torch.cuda.is_available() is False (no NVIDIA device / driver)" + try: + import cupy + except ImportError as e: + return (f"cupy is not installed ({e}); install a matching build, e.g. " + "`pip install cupy-cuda13x` for CUDA 13 or `pip install cupy-cuda12x` for CUDA 12") + return None + + +def runs_on() -> torch.device: + return torch.device("cuda", torch.cuda.current_device()) diff --git a/entropack/backends/cuda/device.py b/entropack/backends/cuda/device.py new file mode 100644 index 0000000..33997a0 --- /dev/null +++ b/entropack/backends/cuda/device.py @@ -0,0 +1,96 @@ +from dataclasses import dataclass + +import cupy +import torch + +from .kernels import device_index + +_FALLBACK_WARP_SIZE = 32 + +_DEFAULT_GRID_WAVES = 8 + +__all__ = ["DeviceCaps", "caps", "resolve_threads", "validate_threads_per_block"] + + +def _shared_optin(index: int, properties) -> int: + optin = getattr(properties, "shared_memory_per_block_optin", 0) + if optin: + return int(optin) + with cupy.cuda.Device(index): + attributes = cupy.cuda.Device(index).attributes + return int(attributes["MaxSharedMemoryPerBlockOptin"]) + + +@dataclass(frozen=True) +class DeviceCaps: + index: int + name: str + compute_capability: tuple[int, int] + sm_count: int + warp_size: int + max_threads_per_block: int + threads_per_sm: int + regs_per_sm: int + shared_per_block: int + shared_optin: int + shared_per_sm: int + l2_bytes: int + + def blocks_per_sm(self, threads_per_block: int, shared_per_block: int = 0, regs_per_thread: int = 0) -> int: + blocks = self.threads_per_sm // threads_per_block + if shared_per_block > 0: + blocks = min(blocks, self.shared_per_sm // shared_per_block) + if regs_per_thread > 0: + blocks = min(blocks, self.regs_per_sm // (regs_per_thread * threads_per_block)) + return max(1, blocks) + + def resident_blocks(self, threads_per_block: int, shared_per_block: int = 0, regs_per_thread: int = 0) -> int: + return self.sm_count * self.blocks_per_sm(threads_per_block, shared_per_block, regs_per_thread) + + def grid(self, wanted: int, threads_per_block: int, shared_per_block: int = 0, waves: int = _DEFAULT_GRID_WAVES) -> int: + limit = self.resident_blocks(threads_per_block, shared_per_block) * waves + return max(1, min(wanted, limit)) + + def shared_limit(self, static_slack: int = 0) -> int: + return max(0, min(self.shared_optin, self.shared_per_sm - static_slack)) + + def threads_per_block(self, wanted: int) -> int: + usable = min(wanted, self.max_threads_per_block) + usable -= usable % self.warp_size + return max(self.warp_size, usable) + + +_caps_cache: dict[int, DeviceCaps] = {} + + +def caps(device=None) -> DeviceCaps: + index = device_index(device) + cached = _caps_cache.get(index) + if cached is not None: + return cached + properties = torch.cuda.get_device_properties(index) + queried = DeviceCaps( + index=index, name=properties.name, compute_capability=(properties.major, properties.minor), + sm_count=properties.multi_processor_count, warp_size=getattr(properties, "warp_size", 0) or _FALLBACK_WARP_SIZE, + max_threads_per_block=properties.max_threads_per_block, threads_per_sm=properties.max_threads_per_multi_processor, + regs_per_sm=properties.regs_per_multiprocessor, shared_per_block=properties.shared_memory_per_block, + shared_optin=_shared_optin(index, properties), shared_per_sm=properties.shared_memory_per_multiprocessor, + l2_bytes=getattr(properties, "L2_cache_size", 0), + ) + _caps_cache[index] = queried + return queried + + +def validate_threads_per_block(caps: DeviceCaps, threads_per_block: int) -> None: + if threads_per_block % caps.warp_size or threads_per_block > caps.max_threads_per_block: + raise ValueError( + f"threads_per_block={threads_per_block} is not launchable on {caps.name}: it must be " + f"a multiple of the {caps.warp_size}-thread warp size and at most {caps.max_threads_per_block}" + ) + + +def resolve_threads(caps: DeviceCaps, requested: int | None, default: int) -> int: + if requested is None: + return caps.threads_per_block(default) + validate_threads_per_block(caps, requested) + return requested diff --git a/entropack/backends/cuda/kernels.py b/entropack/backends/cuda/kernels.py new file mode 100644 index 0000000..6c37a72 --- /dev/null +++ b/entropack/backends/cuda/kernels.py @@ -0,0 +1,70 @@ +from collections.abc import Callable, Hashable, Sequence +from pathlib import Path + +import cupy +import numpy as np +import torch + +_modules: dict[tuple, object] = {} +_kernels: dict[tuple, object] = {} +_streams: dict[tuple, object] = {} + +__all__ = ["KernelLibrary", "device_index", "ensure_dynamic_shared", "external_stream", "pointer"] + + +def ensure_dynamic_shared(kernel, shared_bytes: int) -> None: + if shared_bytes > kernel.max_dynamic_shared_size_bytes: + kernel.max_dynamic_shared_size_bytes = shared_bytes + + +class KernelLibrary: + def __init__( + self, key: str, source: Path, defines: Callable[[int, Hashable], Sequence[str]], + includes: Sequence[Path] = (), kernel_names: Sequence[str] | None = None, + ): + self.key = key + self.source = source + self.defines = defines + self.includes = tuple(includes) + self.kernel_names = None if kernel_names is None else frozenset(kernel_names) + + def kernel(self, device_index: int, variant: Hashable, name: str): + if self.kernel_names is not None and name not in self.kernel_names: + raise KeyError(name) + module_key = (self.key, device_index, variant) + module = _modules.get(module_key) + if module is None: + options = ["--std=c++17"] + options += [f"-D{definition}" for definition in self.defines(device_index, variant)] + options += [f"-I{include}" for include in self.includes] + with cupy.cuda.Device(device_index): + module = cupy.RawModule(code=self.source.read_text(), options=tuple(options)) + _modules[module_key] = module + kernel_key = (self.key, device_index, variant, name) + kernel = _kernels.get(kernel_key) + if kernel is None: + kernel = module.get_function(name) + _kernels[kernel_key] = kernel + return kernel + + +def external_stream(stream: torch.cuda.Stream): + key = (device_index(stream.device), int(stream.cuda_stream)) + wrapper = _streams.get(key) + if wrapper is None: + factory = getattr(cupy.cuda.Stream, "from_external", None) + wrapper = factory(stream) if factory else cupy.cuda.ExternalStream(key[1]) + _streams[key] = wrapper + return wrapper + + +def pointer(tensor: torch.Tensor) -> np.uint64: + return np.uint64(tensor.data_ptr()) + + +def device_index(target) -> int: + if isinstance(target, int): + return target + device = target.device if isinstance(target, torch.Tensor) else target + index = getattr(device, "index", None) + return torch.cuda.current_device() if index is None else index diff --git a/entropack/compression/__init__.py b/entropack/compression/__init__.py new file mode 100644 index 0000000..2affb50 --- /dev/null +++ b/entropack/compression/__init__.py @@ -0,0 +1,4 @@ +from .api import CompressionFallbackWarning, compress, decompress +from .compressed_tensor import CompressedTensor + +__all__ = ["CompressedTensor", "CompressionFallbackWarning", "compress", "decompress"] diff --git a/entropack/compression/api.py b/entropack/compression/api.py new file mode 100644 index 0000000..42a459f --- /dev/null +++ b/entropack/compression/api.py @@ -0,0 +1,73 @@ +import logging +import warnings + +import torch + +from ..registry import DispatchError, get_scheme, name_for_config +from ..schemes import CompressionConfig, RawConfig +from ..schemes.config import validate_config +from .compressed_tensor import CompressedTensor + +logger = logging.getLogger("entropack") + + +class CompressionFallbackWarning(RuntimeWarning): + """The category :func:`compress` warns under when it stored a tensor uncompressed.""" + + +def compress(tensor: torch.Tensor, config: CompressionConfig) -> CompressedTensor: + """Compress a tensor using the selected scheme. + + Args: + tensor: input values. The container preserves the input shape, dtype, and device. + config: configuration selecting the compression scheme and execution backend. + + Returns: + A :class:`CompressedTensor` containing the encoded data and metadata. + + Encoding failures can return an uncompressed ``raw`` container with a + :class:`CompressionFallbackWarning`. Its header records the requested scheme and + failure reason. Invalid configurations and backend dispatch failures raise errors.""" + scheme = get_scheme(name_for_config(config)) + validate_config(config) + try: + packed = scheme.encode(tensor, config) + except DispatchError: + raise + except Exception as error: + logger.warning( + "%s could not encode a %s %s tensor, storing it uncompressed: %s", + scheme.name, tuple(tensor.shape), tensor.dtype, error, + ) + return _compress_raw(tensor, scheme.name, error) + return CompressedTensor( + header={"compress_method": scheme.name}, buffers=packed, shape=tuple(tensor.shape), dtype=tensor.dtype, + ) + + +def _compress_raw(tensor: torch.Tensor, requested: str, error: Exception) -> CompressedTensor: + reason = f"{type(error).__name__}: {error}" + warnings.warn( + f"entropack stored a {tuple(tensor.shape)} {tensor.dtype} tensor uncompressed, at " + f"{tensor.element_size() * 8} bits per element: compress_method={requested!r} failed with {reason}", + CompressionFallbackWarning, stacklevel=3, + ) + return CompressedTensor( + header={"compress_method": "raw", "requested": requested, "reason": reason}, + buffers=get_scheme("raw").encode(tensor, RawConfig()), shape=tuple(tensor.shape), dtype=tensor.dtype, + ) + + +def decompress(compressed: CompressedTensor, config: CompressionConfig) -> torch.Tensor: + """Restore a tensor from its compressed representation. + + Args: + compressed: container to decode. Its header identifies the compression scheme. + config: configuration for the same scheme. Decode settings select the backend and + execution options. Encode settings such as the target bitrate do not recompress + the stored data. + + Returns: + A tensor with the container's shape and dtype, on the device holding its buffers.""" + validate_config(config) + return compressed.scheme.decode(compressed.buffers, shape=compressed.shape, dtype=compressed.dtype, config=config) diff --git a/entropack/compression/compressed_tensor.py b/entropack/compression/compressed_tensor.py new file mode 100644 index 0000000..b706262 --- /dev/null +++ b/entropack/compression/compressed_tensor.py @@ -0,0 +1,164 @@ +import copy +import json +import math +from collections.abc import Mapping +from dataclasses import dataclass +from typing import Any + +import torch + +from ..registry import get_scheme +from ..schemes import Scheme + + +def _parse_dtype(name: str) -> torch.dtype: + if not name: + raise TypeError("CompressedTensor serialized dtype must be a non-empty string") + dtype = getattr(torch, name, None) + if not isinstance(dtype, torch.dtype): + raise ValueError(f"Unsupported serialized torch dtype '{name}'") + return dtype + + +@dataclass +class CompressedTensor: + """Compressed data and metadata for one tensor. + + The container records the scheme, input shape, and dtype. Use + :func:`entropack.decompress` with the matching scheme's configuration to restore values. + Use :meth:`state_dict` and :meth:`from_state_dict` for tensor-only checkpoint entries + compatible with ``torch.load(..., weights_only=True)``. + + Construction checks the scheme, supported dtype, and buffer names.""" + + #: ``{"compress_method": }`` for a coded container, plus ``requested`` and ``reason`` + #: when a codec refused the tensor and it was stored verbatim. + header: dict[str, Any] + #: The scheme's buffers, named exactly as its ``buffer_names`` declares. + buffers: dict[str, torch.Tensor] + #: Shape of the tensor that comes back, before any padding the encode applied. + shape: tuple[int, ...] + #: Its dtype. No scheme changes it: a container holds the format it was given. + dtype: torch.dtype + + def __post_init__(self): + self.header = copy.deepcopy(self.header) + self.buffers = dict(self.buffers) + self.shape = tuple(self.shape) + self.validate() + + @property + def compress_method(self) -> str: + """The scheme name the header carries.""" + return self.header["compress_method"] + + @property + def scheme(self) -> Scheme: + """The codec named by the header.""" + return get_scheme(self.compress_method) + + @property + def lossless(self) -> bool: + """Whether the scheme that wrote this container reconstructs its input exactly.""" + return self.scheme.lossless + + @property + def actual_bpp(self) -> float: + """Bits per element of :attr:`shape`, serialized header included.""" + return self.storage_nbytes() * 8 / math.prod(self.shape) + + def validate(self) -> None: + """Check the scheme, supported dtype, and required buffer names. + + This check does not inspect buffer contents.""" + scheme = self.scheme + if not scheme.supports(self.dtype): + raise ValueError(f"'{scheme.name}' does not support format {self.dtype}") + if set(self.buffers) != set(scheme.buffer_names): + raise ValueError( + f"CompressedTensor buffers for {self.compress_method} must be " + f"{list(scheme.buffer_names)}, got {sorted(self.buffers)}" + ) + + def to(self, device: str | torch.device) -> "CompressedTensor": + """A copy of this container with every buffer on ``device``.""" + target = torch.device(device) + return type(self)( + header=self.header, buffers={name: value.to(device=target) for name, value in self.buffers.items()}, + shape=self.shape, dtype=self.dtype, + ) + + def to_dict(self) -> dict[str, Any]: + """A plain-dict view of the four fields, for a caller that serializes them itself.""" + self.validate() + return {"header": copy.deepcopy(self.header), "buffers": dict(self.buffers), "shape": self.shape, "dtype": self.dtype} + + @classmethod + def from_dict(cls, data: Mapping[str, Any]) -> "CompressedTensor": + """Rebuild the container :meth:`to_dict` produced.""" + required = {"header", "buffers", "shape", "dtype"} + missing = required - data.keys() + if missing: + raise ValueError(f"CompressedTensor data is missing fields: {sorted(missing)}") + return cls(header=data["header"], buffers=data["buffers"], shape=data["shape"], dtype=data["dtype"]) + + def _serialized_header_tensor(self) -> torch.Tensor: + metadata = { + "compress_method": self.compress_method, "dtype": str(self.dtype).removeprefix("torch."), + "shape": list(self.shape), "buffer_names": list(self.buffers), "header": self.header, + } + encoded = json.dumps(metadata, sort_keys=True, separators=(",", ":"), allow_nan=False).encode("utf-8") + return torch.frombuffer(bytearray(encoded), dtype=torch.uint8) + + def storage_nbytes(self, include_header: bool = True) -> int: + """Bytes the buffers occupy, plus the serialized header unless ``include_header`` is false.""" + self.validate() + total = sum(value.numel() * value.element_size() for value in self.buffers.values()) + if include_header: + header = self._serialized_header_tensor() + total += header.numel() * header.element_size() + return total + + def state_dict(self, prefix: str = "") -> dict[str, torch.Tensor]: + """The container as flat tensors under ``prefix``: one 1D uint8 header, then one entry per buffer.""" + self.validate() + state = {f"{prefix}header": self._serialized_header_tensor()} + state.update({f"{prefix}buffers.{name}": value for name, value in self.buffers.items()}) + return state + + @classmethod + def from_state_dict(cls, state: Mapping[str, torch.Tensor], prefix: str = "") -> "CompressedTensor": + """Restore a container from the tensor entries produced by :meth:`state_dict`. + + Metadata is stored as JSON bytes in a uint8 tensor.""" + header_key = f"{prefix}header" + if header_key not in state: + raise ValueError(f"CompressedTensor state is missing '{header_key}'") + header_tensor = state[header_key] + if header_tensor.dtype != torch.uint8 or header_tensor.ndim != 1: + raise TypeError("CompressedTensor serialized header must be a 1D uint8 tensor") + try: + metadata = json.loads(bytes(header_tensor.detach().cpu().tolist()).decode("utf-8")) + except (UnicodeDecodeError, json.JSONDecodeError) as error: + raise ValueError("CompressedTensor serialized header is invalid") from error + if not isinstance(metadata, dict): + raise ValueError("CompressedTensor serialized header must contain a JSON object") + + required = {"compress_method", "dtype", "shape", "buffer_names", "header"} + missing = required - metadata.keys() + if missing: + raise ValueError(f"CompressedTensor serialized header is missing fields: {sorted(missing)}") + buffer_names = metadata["buffer_names"] + if not isinstance(buffer_names, list) or any(not isinstance(name, str) or not name for name in buffer_names): + raise TypeError("CompressedTensor buffer_names must be a list of strings") + buffers = {} + for name in buffer_names: + key = f"{prefix}buffers.{name}" + if key not in state: + raise ValueError(f"CompressedTensor state is missing '{key}'") + buffers[name] = state[key] + + return cls( + header=metadata["header"], buffers=buffers, shape=tuple(metadata["shape"]), + dtype=_parse_dtype(metadata["dtype"]), + ) diff --git a/entropack/linear/__init__.py b/entropack/linear/__init__.py new file mode 100644 index 0000000..30df85e --- /dev/null +++ b/entropack/linear/__init__.py @@ -0,0 +1,5 @@ +from .linear import CompressedFP8Linear, CompressedINT8Linear, CompressedLinear, QuantizedLinear + +__all__ = [ + "CompressedFP8Linear", "CompressedINT8Linear", "CompressedLinear", "QuantizedLinear", +] diff --git a/entropack/linear/linear.py b/entropack/linear/linear.py new file mode 100644 index 0000000..18e645a --- /dev/null +++ b/entropack/linear/linear.py @@ -0,0 +1,499 @@ +import copy +import dataclasses +from numbers import Real +from typing import Any, ClassVar + +import torch +from torch.nn import functional as F + +from ..compression import CompressedTensor, compress, decompress +from ..registry import default_config as _default_config, get_scheme, name_for_config, require_dtype +from ..schemes import LatticeRANSConfig, RawConfig +from .utils import capability_of, load_quant_kernels, pad, round_up + +_EPS = torch.finfo(torch.float32).eps +_BACKEND = "cuda" + + +class CompressedLinear(torch.nn.Linear): + """A linear layer with compressed weights, reconstructed during each forward call. + + The layer registers compressed buffers for ``state_dict``, device transfers, and + ``deepcopy``. Its dense ``weight`` parameter is ``None``. Each forward reconstructs a + temporary weight, casts it to the activation dtype, and applies ``F.linear``. + CUDA and CuPy are required. + + Args: + in_features: number of input features. + out_features: number of output features. + bias: whether to keep a bias. The bias is not compressed. + config: compression configuration. ``None`` selects a lossless scheme by dtype. + device: device for the bias. Compressed buffers retain the source weight's device. + dtype: container dtype, which must be supported by the selected scheme.""" + + state_buffer_names: ClassVar[tuple[str, ...]] = () + state_prefix: ClassVar[str] = "_entropack." + accepts_raw_container: ClassVar[bool] = False + + def __init__( + self, in_features: int, out_features: int, bias: bool = True, *, config=None, + device: str | torch.device | None = None, dtype: torch.dtype = torch.bfloat16, + ): + with torch.device("meta"): + super().__init__(in_features, out_features, bias=False, dtype=dtype) + self.weight = None + if bias: + self.bias = torch.nn.Parameter(torch.zeros(out_features, dtype=dtype, device=device), requires_grad=False) + self.config = config + self._container_dtype = require_dtype(dtype) + if config is not None: + scheme = get_scheme(name_for_config(config)) + if not scheme.supports(self.container_dtype): + raise ValueError(f"'{scheme.name}' does not support format {self.container_dtype}") + self._compressed = None + for name in self.state_buffer_names: + self.register_buffer(name, None, persistent=False) + + @property + def scheme_name(self) -> str: + """The scheme the stored container uses, else the one the config names, else ``"auto"``.""" + if self._compressed is not None: + return self._compressed.compress_method + return "auto" if self.config is None else name_for_config(self.config) + + @property + def _encode_config(self): + if self.config is None: + return _default_config(self.container_dtype, execution_backend=_BACKEND) + return dataclasses.replace(self.config, execution_backend=_BACKEND) + + @property + def _decode_config(self): + if self.config is None: + return self._compressed.scheme.make_config({"execution_backend": _BACKEND}) + return dataclasses.replace(self.config, execution_backend=_BACKEND) + + @property + def container_dtype(self) -> torch.dtype: + """The format the weight is stored in, which the scheme has to serve.""" + return self._container_dtype + + @property + def buffer_names(self) -> tuple[str, ...]: + """The stored container's buffer names; empty while the layer holds no weight.""" + return () if self._compressed is None else tuple(self._compressed.scheme.buffer_names) + + @property + def qweight(self) -> torch.Tensor: + """One tensor standing in for the weight, for a caller that must move or measure it.""" + for name in self.buffer_names: + buffer = self._buffers.get(name) + if buffer is not None: + return buffer + return self.bias + + @property + def compressed_weight(self) -> CompressedTensor: + """The stored container.""" + if self._compressed is None: + raise RuntimeError(f"{type(self).__name__} has no compressed weight; load one or call compress_weight") + return self._compressed + + @property + def stored_nbytes(self) -> int: + """Bytes this layer's weight occupies: the container, its serialized header, and any W8A8 scale.""" + total = self.compressed_weight.storage_nbytes() + for name in self.state_buffer_names: + buffer = self._buffers.get(name) + if buffer is not None: + total += buffer.numel() * buffer.element_size() + return total + + @property + def compressed_bits(self) -> float: + """Bits per element of the source weight's shape, which is what a network rate is built from.""" + rows, cols = self.compressed_weight.shape + return self.stored_nbytes * 8 / (rows * cols) + + @property + def _held_buffers(self) -> tuple[str, ...]: + return self.buffer_names + self.state_buffer_names + + def _holds(self, method: str) -> bool: + return self.config is None or method == self.scheme_name or ( + self.accepts_raw_container and method == "raw" + ) + + def compress_weight(self, weight: torch.Tensor) -> None: + """Compress ``weight`` and store it, replacing whatever the layer held.""" + self._prepare(weight.detach()) + + def _prepare(self, weight: torch.Tensor) -> None: + self.set_compressed(self._container_for(weight)) + + def _container_for(self, tensor: torch.Tensor) -> CompressedTensor: + compressed = compress(tensor.to(self.container_dtype), self._encode_config) + if compressed.compress_method == "raw" and not self.accepts_raw_container: + raise RuntimeError( + f"{self.scheme_name} cannot store a {tuple(tensor.shape)} {self.container_dtype} weight for " + f"{type(self).__name__}: {compressed.header.get('reason', 'asked to store the weight verbatim')}" + ) + return compressed + + def set_compressed(self, compressed: CompressedTensor) -> None: + """Adopt a container built elsewhere, such as one read from a checkpoint. + + It has to hold this layer's shape and container format and use the scheme this layer's config + names, so a container written for another layer cannot be loaded into this one by accident. + """ + if not isinstance(compressed, CompressedTensor): + raise TypeError(f"expected an entropack CompressedTensor, got {type(compressed).__name__}") + expected = (self.out_features, self.in_features) + if compressed.shape != expected: + raise ValueError(f"compressed weight shape {compressed.shape} does not match this Linear's {expected}") + if not self._holds(compressed.compress_method): + raise ValueError( + f"compressed weight uses '{compressed.compress_method}', this Linear is '{self.scheme_name}'" + ) + if compressed.dtype != self.container_dtype: + raise ValueError( + f"compressed weight holds {compressed.dtype}, this Linear is configured for {self.container_dtype}" + ) + names = tuple(compressed.scheme.buffer_names) + self._compressed = CompressedTensor( + header=compressed.header, buffers={name: compressed.buffers[name] for name in names}, + shape=compressed.shape, dtype=compressed.dtype, + ) + for name in names: + self.register_buffer(name, self._compressed.buffers[name], persistent=False) + + def _container_on(self, device: str | torch.device | None) -> CompressedTensor: + compressed = self.compressed_weight + if device is None or compressed.buffers[self.buffer_names[0]].device == torch.device(device): + return compressed + return compressed.to(device) + + def _reconstruct(self, device: str | torch.device | None = None) -> torch.Tensor: + return decompress(self._container_on(device), self._decode_config) + + def dequantize(self, device: str | torch.device | None = None) -> torch.Tensor: + """The dense weight, reconstructed on ``device``, or on the container's device by default.""" + return self._reconstruct(device) + + @classmethod + def from_linear(cls, linear: torch.nn.Linear, **kwargs) -> "CompressedLinear": + """Compress an existing layer's weight into a new layer of this class. + + Args: + linear: the source layer, with its weight materialized -- so this runs after a + ``load_state_dict``, not on a meta-device skeleton. + **kwargs: passed to the constructor; ``dtype`` defaults to the source weight's. + + Returns: + A layer of this class holding the compressed weight and, when the source had one, a copy + of its bias. + """ + weight = linear.weight + if weight is None or weight.device.type == "meta": + raise ValueError("cannot compress a Linear whose weight is not materialized") + kwargs.setdefault("dtype", weight.dtype) + out = cls( + linear.in_features, linear.out_features, bias=linear.bias is not None, device=weight.device, **kwargs, + ) + out.compress_weight(weight.data) + if linear.bias is not None: + out.bias = torch.nn.Parameter(linear.bias.data.clone(), requires_grad=False) + return out + + @property + def device(self) -> torch.device: + """The device the layer's buffers are on, or ``meta`` while it holds none.""" + for name in self._held_buffers: + buffer = self._buffers.get(name) + if buffer is not None: + return buffer.device + return self.bias.device if self.bias is not None else torch.device("meta") + + def _bias_on(self, device: torch.device) -> torch.Tensor | None: + return None if self.bias is None else self.bias.to(device) + + def forward(self, x: torch.Tensor) -> torch.Tensor: + """Reconstruct the weight on ``x``'s device and apply the layer to ``x``.""" + return F.linear(x, self.dequantize(x.device).to(x.dtype), self._bias_on(x.device)) + + def extra_repr(self) -> str: + parts = [f"scheme={self.scheme_name}", f"container={str(self.container_dtype).removeprefix('torch.')}"] + if self._compressed is not None: + parts.append(f"bits={self.compressed_bits:.3f}") + if self.config is not None: + parts.append(f"config={type(self.config).__name__}") + return ", ".join(parts) + + def _sync_compressed(self) -> None: + if self._compressed is not None: + self._compressed.buffers = {name: self._buffers[name] for name in self.buffer_names} + + def _apply(self, fn, recurse=True): + held = {name: self._buffers.pop(name) for name in self._held_buffers if name in self._buffers} + try: + super()._apply(fn, recurse=recurse) + finally: + for name, buffer in held.items(): + if buffer is None: + self._buffers[name] = None + continue + moved = fn(buffer) + self._buffers[name] = moved if moved.dtype == buffer.dtype else buffer.to(device=moved.device) + self._sync_compressed() + return self + + def __deepcopy__(self, memo: dict[int, Any]) -> "CompressedLinear": + clone = type(self).__new__(type(self)) + memo[id(self)] = clone + for key, value in self.__dict__.items(): + clone.__dict__[key] = copy.deepcopy(value, memo) + clone._sync_compressed() + return clone + + def _save_to_state_dict(self, destination: dict, prefix: str, keep_vars: bool) -> None: + super()._save_to_state_dict(destination, prefix, keep_vars) + if self._compressed is None: + return + container_prefix = prefix + self.state_prefix + written = self._compressed.state_dict(container_prefix) + for name in self.state_buffer_names: + buffer = self._buffers[name] + if buffer is not None: + written[container_prefix + name] = buffer + for key, value in written.items(): + destination[key] = value if keep_vars else value.detach() + + def _load_from_state_dict( + self, state_dict: dict, prefix: str, local_metadata, strict, missing_keys, unexpected_keys, error_msgs, + ) -> None: + container_prefix = prefix + self.state_prefix + header_key = container_prefix + "header" + if header_key in state_dict: + try: + self.set_compressed(CompressedTensor.from_state_dict(state_dict, prefix=container_prefix)) + for name in self.state_buffer_names: + key = container_prefix + name + if key not in state_dict: + raise ValueError(f"compressed state is missing '{key}'") + self._buffers[name] = state_dict[key] + except (TypeError, ValueError) as error: + error_msgs.append(f"{prefix[:-1]}: {error}") + for key in [key for key in state_dict if key.startswith(container_prefix)]: + state_dict.pop(key) + elif strict: + missing_keys.append(header_key) + super()._load_from_state_dict( + state_dict, prefix, local_metadata, strict, missing_keys, unexpected_keys, error_msgs, + ) + + +class _W8A8LinearFunction(torch.autograd.Function): + + @staticmethod + def forward(ctx, x, layer): + ctx.layer = layer + ctx.x_shape = x.shape + return layer._forward8(x.detach()) + + @staticmethod + @torch.autograd.function.once_differentiable + def backward(ctx, grad_output): + grad = grad_output.reshape(-1, grad_output.shape[-1]) + weight = ctx.layer.dequantize(grad.device).to(grad.dtype) + return torch.mm(grad, weight).reshape(ctx.x_shape), None + + +class QuantizedLinear(CompressedLinear): + + code_dtype: ClassVar[torch.dtype] + code_max: ClassVar[float] + rounds_to_integer: ClassVar[bool] + code_alignment: ClassVar[int] + min_tokens: ClassVar[int] + min_capability: ClassVar[tuple[int, int]] + state_buffer_names = ("weight_scale",) + accepts_raw_container = True + + def __init__( + self, in_features: int, out_features: int, bias: bool = True, *, config=None, + device: str | torch.device | None = None, dtype: torch.dtype = torch.bfloat16, + ): + super().__init__(in_features, out_features, bias=bias, config=self._coded(config), device=device, dtype=dtype) + + @property + def container_dtype(self) -> torch.dtype: + return self.code_dtype + + @classmethod + def _coded(cls, config): + if config is None: + return RawConfig() + if isinstance(config, RawConfig): + return config + if not isinstance(config, LatticeRANSConfig): + raise TypeError( + f"{cls.__name__} stores {cls.code_dtype} codes: either uncoded (RawConfig) or lattice-coded " + f"(LatticeRANSConfig), got {type(config).__name__}" + ) + rate = config.target_bpp + if isinstance(rate, bool) or not isinstance(rate, Real) or not 0.0 < float(rate) < 8.0: + raise ValueError( + f"{cls.__name__} stores {cls.code_dtype} codes, 8 bits wide, so a target_bpp " + f"only pays below 8; got {rate!r}. Pass RawConfig to store the codes uncoded." + ) + return config + + def _quantize_rows(self, tensor: torch.Tensor) -> tuple[torch.Tensor, torch.Tensor]: + flat = tensor.reshape(-1, tensor.shape[-1]).float() + scale = (flat.abs().amax(dim=1) / self.code_max).clamp(min=_EPS) + scaled = flat / scale.unsqueeze(1) + if self.rounds_to_integer: + scaled = scaled.round() + return scaled.clamp(-self.code_max, self.code_max).to(self.code_dtype), scale + + def _prepare(self, weight: torch.Tensor) -> None: + codes, scale = self._quantize_rows(weight) + self.set_compressed(self._container_for(codes)) + self.weight_scale = scale + + def codes(self, device: str | torch.device | None = None) -> torch.Tensor: + """The stored codes as a dense tensor. The weight is these times their per-row scale.""" + return self._reconstruct(device) + + def dequantize(self, device: str | torch.device | None = None) -> torch.Tensor: + """The dense weight: :meth:`codes` times the per-row scale the quantization fitted.""" + codes = self.codes(device) + return codes.float() * self.weight_scale.to(codes.device).unsqueeze(1) + + def _require_8bit_gemm(self, device: torch.device) -> None: + if device.type != "cuda": + raise RuntimeError( + f"{type(self).__name__} needs a CUDA device with an 8-bit tensor core, compute capability " + f"{self.min_capability} or later; got {device}." + ) + capability = capability_of(device) + if capability < self.min_capability: + raise RuntimeError( + f"{type(self).__name__} needs compute capability {self.min_capability} or later and " + f"{torch.cuda.get_device_name(device)} reports {capability}. This format has no path on " + "older hardware." + ) + + def _quantize_activation(self, flat: torch.Tensor) -> tuple[torch.Tensor, torch.Tensor]: + kernels = load_quant_kernels() + if kernels is None: + return self._quantize_rows(flat) + return kernels.quantize_rows(flat, self.code_dtype, self.code_max, self.rounds_to_integer) + + def _gemm_shapes(self, tokens: int) -> tuple[int, int, int]: + return (max(tokens, self.min_tokens), round_up(self.out_features, self.code_alignment), + round_up(self.in_features, self.code_alignment)) + + def forward(self, x: torch.Tensor) -> torch.Tensor: + """Apply the low-precision linear operation. + + The layer recovers its FP8 or INT8 codes, quantizes activations per row, and runs the + corresponding matrix multiplication. Input gradients use the reconstructed numerical + weight. The compressed base weights remain frozen.""" + self._require_8bit_gemm(x.device) + if x.requires_grad: + return _W8A8LinearFunction.apply(x, self) + return self._forward8(x) + + def _forward8(self, x: torch.Tensor) -> torch.Tensor: + flat = x.reshape(-1, x.shape[-1]) + activation, scale = self._quantize_activation(flat.detach()) + codes = self.codes(x.device) + weight_scale = self.weight_scale.to(codes.device) + tokens = activation.shape[0] + rows, outs, cols = self._gemm_shapes(tokens) + padded_activation = pad(pad(activation, 0, rows), 1, cols) + padded_codes = pad(pad(codes, 0, outs), 1, cols) + out = self._gemm(padded_activation, padded_codes, pad(scale, 0, rows), + pad(weight_scale, 0, outs), x.dtype) + return out[:tokens, : self.out_features].reshape(*x.shape[:-1], self.out_features) + + def _bias_padded(self, device: torch.device, columns: int) -> torch.Tensor | None: + bias = self._bias_on(device) + return None if bias is None else pad(bias, 0, columns) + + def _epilogue( + self, product: torch.Tensor, activation_scale: torch.Tensor, weight_scale: torch.Tensor, + out_dtype: torch.dtype, + ) -> torch.Tensor: + out = product.to(torch.float32) + out.mul_(activation_scale.unsqueeze(1)) + bias = self._bias_padded(out.device, out.shape[1]) + if bias is None: + return out.mul_(weight_scale).to(out_dtype) + return torch.addcmul(bias.to(torch.float32), out, weight_scale).to(out_dtype) + + def _gemm( + self, activation: torch.Tensor, codes: torch.Tensor, activation_scale: torch.Tensor, + weight_scale: torch.Tensor, out_dtype: torch.dtype, + ) -> torch.Tensor: + raise NotImplementedError + + +class CompressedFP8Linear(QuantizedLinear): + """A linear layer using FP8 E4M3FN weights and activations. + + Requires CUDA compute capability 8.9 or later and uses ``torch._scaled_mm``. + With no compression configuration, quantized codes are stored directly. + :class:`~entropack.LatticeRANSConfig` additionally compresses them at targets from 1 up + to, but excluding, 8 bits per element. Inference decodes the FP8 codes before matrix + multiplication. Stored size also includes metadata and per-row weight scales.""" + + code_dtype = torch.float8_e4m3fn + code_max = float(torch.finfo(torch.float8_e4m3fn).max) + rounds_to_integer = False + code_alignment = 16 + min_tokens = 1 + min_capability = (8, 9) + + def _gemm(self, activation, codes, activation_scale, weight_scale, out_dtype): + bias = self._bias_padded(codes.device, weight_scale.shape[0]) + fused = bias is None or out_dtype is not torch.float32 + out = torch._scaled_mm( + activation, codes.t(), scale_a=activation_scale.unsqueeze(1), scale_b=weight_scale.unsqueeze(0), + bias=bias if fused else None, out_dtype=out_dtype, + ) + return out if fused else out.add_(bias) + + +class CompressedINT8Linear(QuantizedLinear): + """A linear layer using symmetric INT8 weights and activations. + + Requires CUDA compute capability 8.0 or later. Matrix multiplication uses Triton when + available and ``torch._int_mm`` otherwise. With no compression configuration, + quantized codes are stored directly. :class:`~entropack.LatticeRANSConfig` additionally compresses + them at targets from 1 up to, but excluding, 8 bits per element. Stored size also + includes metadata and per-row weight scales.""" + + code_dtype = torch.int8 + code_max = 127.0 + rounds_to_integer = True + code_alignment = 8 + min_tokens = 17 + min_capability = (8, 0) + + def _gemm_shapes(self, tokens): + if load_quant_kernels() is not None: + return tokens, self.out_features, self.in_features + return super()._gemm_shapes(tokens) + + def _gemm(self, activation, codes, activation_scale, weight_scale, out_dtype): + bias = self._bias_padded(codes.device, weight_scale.shape[0]) + kernels = load_quant_kernels() + if kernels is not None: + return kernels.int8_gemm(activation, codes, activation_scale, weight_scale, bias, out_dtype) + return self._epilogue(torch._int_mm(activation, codes.t()), activation_scale, weight_scale, out_dtype) + +__all__ = [ + "CompressedFP8Linear", "CompressedINT8Linear", "CompressedLinear", "QuantizedLinear", +] diff --git a/entropack/linear/quant_kernels.py b/entropack/linear/quant_kernels.py new file mode 100644 index 0000000..a668c8c --- /dev/null +++ b/entropack/linear/quant_kernels.py @@ -0,0 +1,108 @@ + +import torch +import triton +import triton.language as tl +import triton.language.extra.libdevice as libdevice + +_INT8_GEMM_CONFIGS = [ + triton.Config({'BLOCK_M': 64, 'BLOCK_N': 64, 'BLOCK_K': 64, 'GROUP_M': 8}, num_warps=4, num_stages=3), + triton.Config({'BLOCK_M': 64, 'BLOCK_N': 64, 'BLOCK_K': 32, 'GROUP_M': 8}, num_warps=4, num_stages=4), + triton.Config({'BLOCK_M': 64, 'BLOCK_N': 128, 'BLOCK_K': 64, 'GROUP_M': 8}, num_warps=4, num_stages=4), + triton.Config({'BLOCK_M': 128, 'BLOCK_N': 64, 'BLOCK_K': 64, 'GROUP_M': 8}, num_warps=4, num_stages=4), + triton.Config({'BLOCK_M': 128, 'BLOCK_N': 128, 'BLOCK_K': 64, 'GROUP_M': 8}, num_warps=8, num_stages=3), +] + +@triton.autotune(configs=_INT8_GEMM_CONFIGS, key=['M', 'N', 'K']) +@triton.jit +def _int8_gemm_kernel( + A, B, A_SCALE, B_SCALE, BIAS, C, M, N, K, stride_am, stride_bn, + HAS_BIAS: tl.constexpr, BLOCK_M: tl.constexpr, BLOCK_N: tl.constexpr, BLOCK_K: tl.constexpr, + GROUP_M: tl.constexpr, +): + pid = tl.program_id(0) + grid_m = tl.cdiv(M, BLOCK_M) + grid_n = tl.cdiv(N, BLOCK_N) + width = GROUP_M * grid_n + group_id = pid // width + group_size = tl.minimum(grid_m - group_id * GROUP_M, GROUP_M) + pid_m = group_id * GROUP_M + (pid % group_size) + pid_n = (pid % width) // group_size + + rm = pid_m * BLOCK_M + tl.arange(0, BLOCK_M) + rn = pid_n * BLOCK_N + tl.arange(0, BLOCK_N) + rk = tl.arange(0, BLOCK_K) + mask_m = rm < M + mask_n = rn < N + + a_ptr = A + rm[:, None] * stride_am + rk[None, :] + b_ptr = B + rn[:, None] * stride_bn + rk[None, :] + acc = tl.zeros((BLOCK_M, BLOCK_N), dtype=tl.int32) + for k in range(0, K, BLOCK_K): + mask_k = (k + rk) < K + a = tl.load(a_ptr, mask=mask_m[:, None] & mask_k[None, :], other=0) + b = tl.load(b_ptr, mask=mask_n[:, None] & mask_k[None, :], other=0) + acc += tl.dot(a, tl.trans(b)) + a_ptr += BLOCK_K + b_ptr += BLOCK_K + + out = acc.to(tl.float32) + out = out * tl.load(A_SCALE + rm, mask=mask_m, other=0.0)[:, None] + out = out * tl.load(B_SCALE + rn, mask=mask_n, other=0.0)[None, :] + if HAS_BIAS: + out += tl.load(BIAS + rn, mask=mask_n, other=0.0).to(tl.float32)[None, :] + tl.store(C + rm[:, None] * N + rn[None, :], out.to(C.dtype.element_ty), + mask=mask_m[:, None] & mask_n[None, :]) + +@triton.jit +def _quantize_rows_kernel( + X, OUT, SCALE, K, stride_xm, CODE_MAX: tl.constexpr, ROUNDS: tl.constexpr, EPS: tl.constexpr, + BLOCK_K: tl.constexpr, +): + row = tl.program_id(0) + base = X + row * stride_xm + amax = tl.zeros((), dtype=tl.float32) + for k in range(0, K, BLOCK_K): + rk = k + tl.arange(0, BLOCK_K) + x = tl.load(base + rk, mask=rk < K, other=0.0).to(tl.float32) + amax = tl.maximum(amax, tl.max(tl.abs(x))) + scale = tl.maximum(amax / CODE_MAX, EPS) + + out_base = OUT + row * K + for k in range(0, K, BLOCK_K): + rk = k + tl.arange(0, BLOCK_K) + x = tl.load(base + rk, mask=rk < K, other=0.0).to(tl.float32) + scaled = x / scale + if ROUNDS: + scaled = libdevice.rint(scaled) + scaled = tl.minimum(tl.maximum(scaled, -CODE_MAX), CODE_MAX) + tl.store(out_base + rk, scaled.to(OUT.dtype.element_ty), mask=rk < K) + tl.store(SCALE + row, scale) + +def int8_gemm( + activation: torch.Tensor, codes: torch.Tensor, activation_scale: torch.Tensor, + weight_scale: torch.Tensor, bias: torch.Tensor | None, out_dtype: torch.dtype, +) -> torch.Tensor: + tokens, inner = activation.shape + outer = codes.shape[0] + out = torch.empty(tokens, outer, dtype=out_dtype, device=activation.device) + grid = lambda meta: (triton.cdiv(tokens, meta['BLOCK_M']) * triton.cdiv(outer, meta['BLOCK_N']),) + _int8_gemm_kernel[grid]( + activation, codes, activation_scale, weight_scale, bias, out, tokens, outer, inner, + activation.stride(0), codes.stride(0), bias is not None, + ) + return out + +def quantize_rows( + tensor: torch.Tensor, code_dtype: torch.dtype, code_max: float, rounds_to_integer: bool, +) -> tuple[torch.Tensor, torch.Tensor]: + flat = tensor.reshape(-1, tensor.shape[-1]) + rows, inner = flat.shape + codes = torch.empty(rows, inner, dtype=code_dtype, device=flat.device) + scale = torch.empty(rows, dtype=torch.float32, device=flat.device) + _quantize_rows_kernel[(rows,)]( + flat, codes, scale, inner, flat.stride(0), code_max, rounds_to_integer, + torch.finfo(torch.float32).eps, BLOCK_K=1024, num_warps=8, + ) + return codes, scale + +__all__ = ["int8_gemm", "quantize_rows"] diff --git a/entropack/linear/utils.py b/entropack/linear/utils.py new file mode 100644 index 0000000..8eb5218 --- /dev/null +++ b/entropack/linear/utils.py @@ -0,0 +1,41 @@ +import torch + +_kernels = None +_looked = False +_device_capabilities: dict[int, tuple[int, int]] = {} + + +def round_up(value: int, multiple: int) -> int: + return -(-value // multiple) * multiple + + +def pad(tensor: torch.Tensor, dim: int, size: int) -> torch.Tensor: + shortfall = size - tensor.shape[dim] + if shortfall <= 0: + return tensor + shape = list(tensor.shape) + shape[dim] = shortfall + return torch.cat([tensor, tensor.new_zeros(shape)], dim=dim) + + +def load_quant_kernels(): + global _kernels, _looked + if not _looked: + _looked = True + try: + from . import quant_kernels + _kernels = quant_kernels + except ImportError: + _kernels = None + return _kernels + + +def capability_of(device: torch.device) -> tuple[int, int]: + index = device.index if device.index is not None else torch.cuda.current_device() + capability = _device_capabilities.get(index) + if capability is None: + capability = torch.cuda.get_device_capability(index) + _device_capabilities[index] = capability + return capability + +__all__ = ["capability_of", "load_quant_kernels", "pad", "round_up"] diff --git a/entropack/registry.py b/entropack/registry.py new file mode 100644 index 0000000..87ba763 --- /dev/null +++ b/entropack/registry.py @@ -0,0 +1,121 @@ +import logging +import threading +from typing import Any + +import torch + +from .backends import BACKENDS, Backend + +logger = logging.getLogger("entropack.registry") + +_backends: dict[str, Backend] = {backend.name: backend for backend in BACKENDS} +_schemes: dict[str, Any] = {} +_reasons: dict[str, str | None] = {} +_lock = threading.Lock() + + +_CONTAINER_DTYPES = ( + torch.float32, torch.float16, torch.bfloat16, + torch.float8_e4m3fn, torch.float8_e4m3fnuz, torch.float8_e5m2, torch.float8_e5m2fnuz, + torch.int64, torch.int32, torch.int16, torch.int8, torch.uint64, torch.uint32, torch.uint16, torch.uint8, torch.bool, +) + + +class DispatchError(RuntimeError): + """``reasons`` maps every backend considered to why it was rejected, so no fallback is silent.""" + + def __init__(self, scheme: str, reasons: dict[str, str]): + self.scheme = scheme + self.reasons = dict(reasons) + detail = "; ".join(f"{name}: {reason}" for name, reason in self.reasons.items()) + super().__init__(f"No backend can handle '{scheme}'" + (f": {detail}" if detail else "")) + + +def register_scheme(scheme) -> Any: + with _lock: + _schemes[scheme.name] = scheme + return scheme + + +def get_scheme(name: str) -> Any: + try: + return _schemes[name] + except KeyError: + raise KeyError(f"Unknown scheme '{name}'; available: {sorted(_schemes)}") from None + + +def all_schemes() -> list: + return sorted(_schemes.values(), key=lambda scheme: (-scheme.priority, scheme.name)) + + +def name_for_config(config) -> str: + for cls in type(config).__mro__: + for scheme in _schemes.values(): + if scheme.config_cls is cls: + return scheme.name + raise TypeError(f"{type(config).__name__} is not the config of any registered scheme") + + +def require_dtype(dtype: torch.dtype) -> torch.dtype: + if dtype not in _CONTAINER_DTYPES: + raise ValueError(f"Unsupported tensor dtype {dtype}; supported: {sorted(_CONTAINER_DTYPES, key=str)}") + return dtype + + +def default_config(dtype, **overrides): + dtype = require_dtype(dtype) + scheme = next(scheme for scheme in all_schemes() if scheme.supports(dtype)) + return scheme.make_config({**scheme.options_for(dtype), **overrides}) + + +def _ordered() -> list[Backend]: + return sorted(_backends.values(), key=lambda backend: backend.priority, reverse=True) + + +def reason(backend: str) -> str | None: + spec = _backends.get(backend) + if spec is None: + return "not registered" + if backend not in _reasons: + with _lock: + if backend not in _reasons: + _reasons[backend] = spec.probe() if spec.probe is not None else None + return _reasons[backend] + + +def _rejection(name: str, scheme, dtype: torch.dtype | None) -> str | None: + if name not in _backends: + return "not registered" + if name not in scheme.lanes: + return f"scheme '{scheme.name}' declares no '{name}' lane" + unavailable = reason(name) + if unavailable is not None: + return unavailable + if dtype is not None and not scheme.supports(dtype): + return f"dtype {dtype} is not supported" + return None + + +def select(scheme, tensor: torch.Tensor | None = None, backend: str | None = None, + *, gate_dtype: bool = True) -> str: + dtype = tensor.dtype if (gate_dtype and tensor is not None) else None + if backend is not None: + rejected = _rejection(backend, scheme, dtype) + if rejected is not None: + raise DispatchError(scheme.name, {backend: rejected}) + logger.debug("Backend %s selected for %s", backend, scheme.name) + return backend + + failures: dict[str, str] = {} + for spec in _ordered(): + rejected = _rejection(spec.name, scheme, dtype) + if rejected is None: + logger.debug("Backend %s selected for %s", spec.name, scheme.name) + return spec.name + failures[spec.name] = rejected + raise DispatchError(scheme.name, failures) + + +def backend_device(backend: str | None) -> torch.device | None: + spec = _backends.get(backend) if backend is not None else None + return spec.device() if spec is not None and spec.device is not None else None diff --git a/entropack/schemes/__init__.py b/entropack/schemes/__init__.py new file mode 100644 index 0000000..a888689 --- /dev/null +++ b/entropack/schemes/__init__.py @@ -0,0 +1,12 @@ +from ..registry import all_schemes +from .base import Scheme +from .config import CompressionConfig, RawConfig +from .dfloat11 import DFloat11Config +from .lattice_rans import LatticeRANSConfig +from .tile_ans import TileANSConfig +from . import dfloat11, lattice_rans, tile_ans + +__all__ = [ + "CompressionConfig", "DFloat11Config", "LatticeRANSConfig", "RawConfig", "Scheme", "TileANSConfig", + "all_schemes", +] diff --git a/entropack/schemes/base.py b/entropack/schemes/base.py new file mode 100644 index 0000000..1628d55 --- /dev/null +++ b/entropack/schemes/base.py @@ -0,0 +1,124 @@ +import importlib +from abc import ABC, abstractmethod +from dataclasses import fields +from typing import Any, get_type_hints + +import torch + +from .config import CompressionConfig, RawConfig, validate_config +from ..registry import DispatchError, backend_device, register_scheme, select + +__all__ = ["RawScheme", "Scheme", "buffers_fingerprint", "cached_parse", "packed_buffers", "register_scheme"] + +_lane_modules: dict[tuple[type, str], Any] = {} +_config_classes: dict[type, type] = {} + + +def buffers_fingerprint(buffers: dict, shape, dtype) -> Any: + return ( + tuple(shape) if shape is not None else None, + dtype, + tuple( + ( + name, tensor.data_ptr(), tensor._version, tuple(tensor.shape), tuple(tensor.stride()), tensor.dtype, + tensor.device, + ) + for name, tensor in sorted(buffers.items()) + ), + ) + + +def packed_buffers(packed: dict, kind: type) -> Any: + return kind(**{name: packed[name] for name in kind._fields}) + + +def cached_parse(layout: torch.Tensor, parse, attribute: str): + cached = getattr(layout, attribute, None) + if cached is None or cached[0] != layout._version: + cached = (layout._version, parse(layout)) + setattr(layout, attribute, cached) + return cached[1] + + +class Scheme(ABC): + name: str + buffer_names: tuple[str, ...] + lossless: bool = True + priority: int = 0 + dtypes: tuple[torch.dtype, ...] | None = None + lanes: dict[str, str] = {} + + def supports(self, dtype: torch.dtype) -> bool: + return self.dtypes is None or dtype in self.dtypes + + def options_for(self, dtype: torch.dtype) -> dict[str, Any]: + return {} + + def lane(self, backend: str) -> Any: + key = (type(self), backend) + module = _lane_modules.get(key) + if module is None: + if backend not in self.lanes: + raise DispatchError(self.name, {backend: f"scheme '{self.name}' declares no '{backend}' lane"}) + try: + module = importlib.import_module(f".{self.lanes[backend]}", package=type(self).__module__) + except Exception as error: + raise DispatchError(self.name, {backend: f"lane module failed to import: {error!r}"}) from error + _lane_modules[key] = module + return module + + def lane_for(self, tensor: torch.Tensor | None = None, backend: str | None = None, *, gate_dtype: bool = True): + name = select(self, tensor, None if backend in (None, "auto") else backend, gate_dtype=gate_dtype) + return self.lane(name), backend_device(name) + + @abstractmethod + def encode(self, weight: Any, config: CompressionConfig) -> dict: + ... + + @abstractmethod + def decode(self, packed: dict, *, shape: tuple[int, ...], dtype: Any, config: CompressionConfig) -> Any: + ... + + @property + def config_cls(self) -> type: + """The config class this scheme's ``encode`` annotation promises.""" + cached = _config_classes.get(type(self)) + if cached is None: + cached = get_type_hints(type(self).encode)["config"] + _config_classes[type(self)] = cached + return cached + + def make_config(self, options: dict) -> Any: + """Build this scheme's config from option names and values, rejecting both unknown and bad ones.""" + unknown = set(options) - {field.name for field in fields(self.config_cls)} + if unknown: + raise TypeError(f"Unknown {self.name} options: {sorted(unknown)}") + config = self.config_cls(**options) + validate_config(config) + return config + + def validate_buffers(self, buffers: dict, shape: tuple[int, ...], dtype: Any) -> None: + ... + + +class RawScheme(Scheme): + name = "raw" + buffer_names = ("data",) + priority = -1 + dtypes = None + lanes = {} + + def encode(self, weight, config: RawConfig) -> dict: + data = weight.detach().contiguous().clone() + return {"data": data} + + def validate_buffers(self, buffers, shape, dtype) -> None: + data = buffers["data"] + if data.dtype != dtype or tuple(data.shape) != tuple(shape): + raise ValueError(f"raw buffer must be {tuple(shape)} {dtype}, got {tuple(data.shape)} {data.dtype}") + + def decode(self, packed: dict, *, shape, dtype, config: RawConfig) -> torch.Tensor: + return packed["data"] + + +register_scheme(RawScheme()) diff --git a/entropack/schemes/checks.py b/entropack/schemes/checks.py new file mode 100644 index 0000000..69cc553 --- /dev/null +++ b/entropack/schemes/checks.py @@ -0,0 +1,33 @@ +from collections.abc import Iterable +from numbers import Real + +import torch + +__all__ = ["prepare_weight"] + +_NO_MINMAX_KERNEL = frozenset({ + torch.float8_e4m3fn, torch.float8_e4m3fnuz, torch.float8_e5m2, torch.float8_e5m2fnuz, +}) + + +def _all_finite(weight: torch.Tensor) -> bool: + if not weight.dtype.is_floating_point: + return True + probe = weight.float() if weight.dtype in _NO_MINMAX_KERNEL else weight + limit = float(torch.finfo(weight.dtype).max) + return float(probe.amin()) >= -limit and float(probe.amax()) <= limit + + +def prepare_weight( + weight: torch.Tensor, *, scheme: str, ndim: int | None = None, dtypes: Iterable[torch.dtype] | None = None, + require_finite: bool = False, +) -> torch.Tensor: + if dtypes is not None and weight.dtype not in frozenset(dtypes): + raise ValueError(f"{scheme} does not support dtype {weight.dtype}") + if ndim is not None and weight.ndim != ndim: + raise ValueError(f"{scheme} encode requires exactly {ndim}D input, got {weight.ndim}D") + if weight.numel() == 0: + raise ValueError(f"{scheme} does not support empty tensors") + if require_finite and not _all_finite(weight): + raise ValueError(f"{scheme} input must contain only finite values") + return weight.contiguous() diff --git a/entropack/schemes/config.py b/entropack/schemes/config.py new file mode 100644 index 0000000..ac24252 --- /dev/null +++ b/entropack/schemes/config.py @@ -0,0 +1,111 @@ +from dataclasses import dataclass, fields +from numbers import Real +from typing import Annotated, get_args, get_origin, get_type_hints + +__all__ = ["CompressionConfig", "OneOf", "Range", "RawConfig", "config_fields", "validate_config"] + + +@dataclass +class CompressionConfig: + """Common execution settings for compression schemes. + + Choose a concrete configuration class to select a scheme. Its fields control compression + or decompression as documented for that scheme.""" + + #: Execution backend: ``"auto"`` or ``None`` selects automatically, or use ``"cuda"`` or ``"eager"``. + execution_backend: str | None = "auto" + + +@dataclass +class RawConfig(CompressionConfig): + """Store tensor values without entropy coding. + + Used explicitly for uncoded FP8 or INT8 weights and as the tensor API's fallback when + encoding fails. Storage still includes container metadata.""" + + +class Range: + """An inclusive numeric bound. ``message`` overrides the generated text.""" + + def __init__(self, low=None, high=None, *, message: str | None = None): + self.low = low + self.high = high + self.message = message + + def holds(self, value) -> bool: + return (self.low is None or value >= self.low) and (self.high is None or value <= self.high) + + def __repr__(self) -> str: + return f"Range({self.low!r}, {self.high!r})" + + def describe(self, name: str) -> str: + if self.message is not None: + return self.message + if self.low is None: + return f"{name} must be <= {self.high}" + if self.high is None: + return f"{name} must be >= {self.low}" + return f"{name} must be in [{self.low}, {self.high}]" + + +class OneOf: + """A closed set of values. ``silent`` holds the ones that ask for automatic selection.""" + + def __init__(self, choices, *, silent=(), message: str | None = None): + self.choices = tuple(choices) + self.silent = frozenset(silent) + self.message = message + + def holds(self, value) -> bool: + return value in self.silent or value in self.choices + + def __repr__(self) -> str: + return f"OneOf({list(self.choices)!r})" + + def describe(self, name: str) -> str: + return self.message or f"{name} must be one of {self.choices}" + + +_hints: dict[type, dict] = {} + + +def config_fields(cls: type) -> dict: + cached = _hints.get(cls) + if cached is None: + hints = get_type_hints(cls, include_extras=True) + cached = {field.name: hints[field.name] for field in fields(cls) if field.name != "execution_backend"} + _hints[cls] = cached + return cached + + +def _check(name: str, value, hint) -> None: + if get_origin(hint) is Annotated: + hint, *markers = get_args(hint) + else: + markers = [] + union = get_args(hint) if get_origin(hint) is not None else (hint,) + kinds = tuple(kind for kind in union if kind is not type(None)) + optional = " or None" if len(kinds) < len(union) else "" + + if value is None: + if not optional: + raise TypeError(f"{name} must be {'an integer' if kinds[0] is int else 'a real number'}") + return + + if kinds[0] is int: + holds, expected = isinstance(value, int) and not isinstance(value, bool), "an integer" + elif kinds[0] is bool: + holds, expected = isinstance(value, bool), "a bool" + else: + holds, expected = isinstance(value, Real) and not isinstance(value, bool), "a real number" + if not holds: + raise TypeError(f"{name} must be {expected}{optional}") + + for marker in markers: + if not marker.holds(value): + raise ValueError(marker.describe(name)) + + +def validate_config(config) -> None: + for name, hint in config_fields(type(config)).items(): + _check(name, getattr(config, name), hint) diff --git a/entropack/schemes/dfloat11/__init__.py b/entropack/schemes/dfloat11/__init__.py new file mode 100644 index 0000000..9c744c1 --- /dev/null +++ b/entropack/schemes/dfloat11/__init__.py @@ -0,0 +1,52 @@ +import torch + +from ..base import Scheme, packed_buffers, register_scheme +from ..checks import prepare_weight +from .eager import get_32bit_codec, get_luts +from .format import PACKED_KEYS, DFloat11Buffers, DFloat11Config, validate_packed + + +class DFloat11Scheme(Scheme): + name = "dfloat11" + buffer_names = PACKED_KEYS + priority = 200 + dtypes = (torch.bfloat16,) + lanes = {"eager": "eager", "cuda": "cuda"} + + def encode(self, weight: torch.Tensor, config: DFloat11Config) -> dict: + weight = prepare_weight(weight, scheme=self.name, dtypes=self.dtypes) + device = weight.device + flat = weight.reshape(-1) + + lane, run_on = self.lane_for(flat, config.execution_backend) + if run_on is not None and device != run_on: + flat = flat.to(run_on) + counter = lane.exponent_counter(flat, config.threads_per_block) + codec, _counter, table = get_32bit_codec(counter) + luts = get_luts(table) + + buffers = lane.encode( + weight=flat, codec=codec, luts=luts, bytes_per_thread=config.bytes_per_thread, + threads_per_block=config.threads_per_block, + ) + + return {key: value.to(device) for key, value in buffers._asdict().items()} + + def validate_buffers(self, buffers, shape, dtype): + validate_packed(buffers, shape) + + def decode(self, packed: dict, *, shape: tuple[int, ...], dtype: torch.dtype, + config: DFloat11Config) -> torch.Tensor: + buffers = packed_buffers(packed, DFloat11Buffers) + source = buffers.layout.device + lane, lane_device = self.lane_for(buffers.layout, config.execution_backend, gate_dtype=False) + if lane_device is not None and source != lane_device: + buffers = DFloat11Buffers._make(value.to(lane_device) for value in buffers) + flat = lane.decode(buffers) + out = flat.reshape(shape) + return out if out.device == source else out.to(source) + + +register_scheme(DFloat11Scheme()) + +__all__ = ["DFloat11Config", "DFloat11Scheme"] diff --git a/entropack/schemes/dfloat11/cuda.py b/entropack/schemes/dfloat11/cuda.py new file mode 100644 index 0000000..7c7c391 --- /dev/null +++ b/entropack/schemes/dfloat11/cuda.py @@ -0,0 +1,223 @@ +from pathlib import Path + +import cupy +import numpy as np +import torch + +from ...backends.cuda import device as _device_caps +from ...backends.cuda.kernels import KernelLibrary +from ...backends.cuda.kernels import device_index as _device_index +from ...backends.cuda.kernels import ensure_dynamic_shared as _ensure_dynamic_shared +from ...backends.cuda.kernels import external_stream as _external_stream +from ...backends.cuda.kernels import pointer as _pointer +from .format import BLOCK_SIZE, MAX_RESIDENT_BLOCKS_PER_SM, DFloat11Buffers, make_layout, max_block_elems, parse_layout_cached + +_CUDA_PATH = Path(__file__).parent / "dfloat11.cu" +_DECODE_KERNEL_NAME = "dfloat11_decode_kernel" +_ENCODE_KERNEL_NAMES = ( + "dfloat11_exponent_histogram_kernel", "dfloat11_split_len_kernel", "dfloat11_pack_kernel", "dfloat11_thread_meta_kernel", + "dfloat11_output_positions_kernel", +) + + +def _compile_defines(device_index: int, threads_per_block: int) -> tuple[str, ...]: + caps = _device_caps.caps(device_index) + min_blocks = max(1, min(MAX_RESIDENT_BLOCKS_PER_SM, caps.threads_per_sm // threads_per_block)) + return (f"DFLOAT11_THREADS_PER_BLOCK={threads_per_block}", f"DFLOAT11_MIN_BLOCKS_PER_SM={min_blocks}") + + +_LIBRARY = KernelLibrary( + key="dfloat11", source=_CUDA_PATH, defines=_compile_defines, kernel_names=(_DECODE_KERNEL_NAME, *_ENCODE_KERNEL_NAMES), +) +_kernel = _LIBRARY.kernel + + +def _encode_kernels(device_index: int, threads_per_block: int) -> dict: + return {name: _kernel(device_index, threads_per_block, name) for name in _ENCODE_KERNEL_NAMES} + + +def _shared_budget(device: torch.device) -> int: + return _device_caps.caps(device).shared_limit() + + +def _kernel_tensor(t: torch.Tensor, stream: torch.cuda.Stream) -> torch.Tensor: + contiguous = t.contiguous() + if contiguous is not t: + contiguous.record_stream(stream) + return contiguous + + +def _cupy_view(t: torch.Tensor): + """Not cached: the DLPack capsule keeps the source tensor alive, so a cache would pin every tensor ever viewed for the life + of the process. + """ + return cupy.from_dlpack(t) + + +def decode(buffers: DFloat11Buffers) -> torch.Tensor: + encoded_exponent, sign_mantissa, luts = buffers.encoded_exponent, buffers.sign_mantissa, buffers.luts + output_positions, thread_meta, layout = buffers.output_positions, buffers.thread_meta, buffers.layout + bytes_per_thread, threads_per_block, max_block_elems = parse_layout_cached(layout) + num_luts = int(luts.shape[0]) + n_bytes = int(encoded_exponent.numel()) + n_elements = int(sign_mantissa.numel()) + + n_threads = (n_bytes + bytes_per_thread - 1) // bytes_per_thread + blocks = (n_threads + threads_per_block - 1) // threads_per_block + + budget = _shared_budget(sign_mantissa.device) + fixed_bytes = threads_per_block * 4 + num_luts * 256 + stage_enc_bytes = threads_per_block * bytes_per_thread + 8 + stage_elems = max_block_elems + if fixed_bytes + stage_enc_bytes + stage_elems + 4 > budget: + stage_elems = 0 + if fixed_bytes + stage_enc_bytes > budget: + stage_enc_bytes = 0 + shared_bytes = fixed_bytes + stage_enc_bytes + stage_elems + (4 if stage_elems else 0) + + out = torch.empty(n_elements, dtype=torch.bfloat16, device=sign_mantissa.device) + if blocks == 0: + return out + + device_index = _device_index(sign_mantissa) + with cupy.cuda.Device(device_index): + kernel = _kernel(device_index, threads_per_block, _DECODE_KERNEL_NAME) + _ensure_dynamic_shared(kernel, shared_bytes) + + torch_stream = torch.cuda.current_stream(sign_mantissa.device) + luts_c = _kernel_tensor(luts, torch_stream) + encoded_c = _kernel_tensor(encoded_exponent, torch_stream) + sign_mantissa_c = _kernel_tensor(sign_mantissa, torch_stream) + output_positions_c = _kernel_tensor(output_positions, torch_stream) + thread_meta_c = _kernel_tensor(thread_meta, torch_stream) + args = ( + _pointer(luts_c), _pointer(encoded_c), _pointer(sign_mantissa_c), _pointer(output_positions_c), + _pointer(thread_meta_c), _pointer(out), np.int32(num_luts), np.int64(n_bytes), np.int64(n_elements), + np.int32(bytes_per_thread), np.int32(stage_enc_bytes), np.int32(stage_elems), + np.int32(1 if (sign_mantissa_c.data_ptr() & 3) == 0 else 0), + ) + with _external_stream(torch_stream): + kernel((blocks,), (threads_per_block,), args, shared_mem=shared_bytes) + return out + + +def exponent_counter(weight: torch.Tensor, threads_per_block: int) -> dict[int, int]: + flat = weight.reshape(-1) + n_elements = int(flat.numel()) + histogram = torch.zeros(256, dtype=torch.int64, device=flat.device) + if n_elements == 0: + return {} + + caps = _device_caps.caps(flat.device) + threads = caps.threads_per_block(BLOCK_SIZE) + blocks = caps.grid((n_elements + threads - 1) // threads, threads) + device_index = _device_index(flat) + torch_stream = torch.cuda.current_stream(flat.device) + stream_ctx = _external_stream(torch_stream) + with cupy.cuda.Device(device_index), stream_ctx: + _kernel(device_index, threads_per_block, "dfloat11_exponent_histogram_kernel")( + (blocks,), (threads,), (_pointer(flat), _pointer(histogram), np.int64(n_elements)), + ) + counts = histogram.cpu().tolist() + return {i: int(count) for i, count in enumerate(counts) if count > 0} + + +def _code_table(codec, device): + code_len = torch.zeros(256, dtype=torch.int32) + code_val = torch.zeros(256, dtype=torch.int32) + for k, (bits, val) in codec._table.items(): + if isinstance(k, int): + code_len[k] = bits + code_val[k] = val + eof_len, eof_val = codec._table[codec._eof] + return (code_len.to(device).contiguous(), code_val.to(device).contiguous(), int(eof_len), int(eof_val)) + + +def encode( + *, weight: torch.Tensor, codec, luts: torch.Tensor, bytes_per_thread: int, threads_per_block: int, +) -> DFloat11Buffers: + device = weight.device + device_index = _device_index(device) + flat = weight.reshape(-1) + n_elements = int(flat.numel()) + if not 5 <= bytes_per_thread <= 255: + raise ValueError( + "dfloat11 bytes_per_thread must be in [5, 255]: the region must exceed " + "the 32-bit maximum code length and its worst-case symbol count must fit " "the 11-bit thread_meta field" + ) + if n_elements >= 1 << 32: + raise ValueError("dfloat11 requires fewer than 2^32 elements per tensor") + code_len_gpu, code_val_gpu, eof_len, eof_val = _code_table(codec, device) + + kernels = _encode_kernels(device_index, threads_per_block) + threads = _device_caps.caps(device).threads_per_block(BLOCK_SIZE) + + exponent = torch.empty(n_elements, dtype=torch.uint8, device=device) + sign_mantissa = torch.empty(n_elements, dtype=torch.uint8, device=device) + len_scratch = torch.empty(n_elements, dtype=torch.uint8, device=device) + pref = torch.empty(n_elements + 1, dtype=torch.int64, device=device) + + blocks_split = (n_elements + threads - 1) // threads + + torch_stream = torch.cuda.current_stream(flat.device) + stream_ctx = _external_stream(torch_stream) + with cupy.cuda.Device(device_index), stream_ctx: + kernels["dfloat11_split_len_kernel"]( + (blocks_split,), + (threads,), + ( + _pointer(flat), _pointer(code_len_gpu), _pointer(exponent), _pointer(sign_mantissa), _pointer(len_scratch), + np.int64(n_elements), + ), + ) + + pref_cp = _cupy_view(pref) + pref_cp[0] = 0 + if n_elements > 0: + cupy.cumsum(_cupy_view(len_scratch), dtype=cupy.int64, out=pref_cp[1:]) + + total_bits = int(pref[n_elements].item()) + + n_bytes = (total_bits + 7) // 8 + region_bits = bytes_per_thread * 8 + block_bits = region_bits * threads_per_block + bytes_per_block = bytes_per_thread * threads_per_block + num_blocks = (n_bytes + bytes_per_block - 1) // bytes_per_block + n_regions = threads_per_block * num_blocks + + encoded = torch.zeros(n_bytes, dtype=torch.uint8, device=device) + thread_meta = torch.empty(n_regions, dtype=torch.uint16, device=device) + output_positions = torch.empty(num_blocks + 1, dtype=torch.uint32, device=device) + + with cupy.cuda.Device(device_index), stream_ctx: + pack_chunks = (n_bytes + 3) // 4 + kernels["dfloat11_pack_kernel"]( + ((pack_chunks + threads - 1) // threads,), + (threads,), + ( + _pointer(pref), _pointer(exponent), _pointer(code_len_gpu), _pointer(code_val_gpu), _pointer(encoded), + np.int64(n_elements), np.int64(n_bytes), np.int64(total_bits), np.int32(eof_len), np.uint32(eof_val), + ), + ) + kernels["dfloat11_thread_meta_kernel"]( + ((n_regions + threads - 1) // threads,), + (threads,), + ( + _pointer(pref), _pointer(thread_meta), np.int64(n_elements), np.int64(total_bits), np.int64(region_bits), + np.int64(n_regions), + ), + ) + kernels["dfloat11_output_positions_kernel"]( + ((num_blocks + 1 + threads - 1) // threads,), (threads,), + (_pointer(pref), _pointer(output_positions), np.int64(n_elements), np.int64(block_bits), np.int64(num_blocks)), + ) + + op_host = torch.empty(num_blocks + 1, dtype=torch.uint32, pin_memory=True) + op_host.copy_(output_positions, non_blocking=True) + torch_stream.synchronize() + + layout = make_layout(bytes_per_thread, threads_per_block, max_block_elems(op_host)) + return DFloat11Buffers( + encoded_exponent=encoded, sign_mantissa=sign_mantissa, luts=luts, output_positions=output_positions, + thread_meta=thread_meta, layout=layout, + ) diff --git a/entropack/schemes/dfloat11/dfloat11.cu b/entropack/schemes/dfloat11/dfloat11.cu new file mode 100644 index 0000000..9bafe8d --- /dev/null +++ b/entropack/schemes/dfloat11/dfloat11.cu @@ -0,0 +1,740 @@ +// EntroPack -- DFloat11 lossless bf16 compression: CUDA kernels. +// +// Decode kernel: +// dfloat11_decode_kernel one thread per bitstream region: walks the multi-level Huffman LUTs to recover the +// exponents, stages them in shared memory, and writes the reconstructed bf16 back +// coalesced. __launch_bounds__ come from -D macros the host derives from the queried +// device. +// +// Encode kernels, in pipeline order: +// dfloat11_exponent_histogram_kernel 256-bin bf16 exponent counts, accumulated in shared memory per block. +// dfloat11_split_len_kernel splits bf16 into exponent and sign/mantissa bytes and looks up each exponent's +// Huffman code length in the same pass. +// dfloat11_pack_kernel writes the MSB-first bitstream, one thread per short output chunk. +// dfloat11_thread_meta_kernel per-region gap and symbol count, as the checkpoint's uint16 thread_meta. +// dfloat11_output_positions_kernel per-block starting element index. +// +// The device helpers below serve both directions: MSB-first bit windows, the multi-level LUT walk, and a block-wide +// exclusive scan. Python orchestration lives in cuda.py; format.py is authoritative for the buffer layout and for the +// decode granularity recorded in each checkpoint. NVRTC has no system include path, so libcudacxx supplies the +// uint8_t/uint32_t/int64_t typedefs that would. +#include + +constexpr int kMetaCountBits = 11; +constexpr uint32_t kMetaCountMask = (1u << kMetaCountBits) - 1u; + +#ifndef DFLOAT11_THREADS_PER_BLOCK +#define DFLOAT11_THREADS_PER_BLOCK 128 +#endif +#ifndef DFLOAT11_MIN_BLOCKS_PER_SM +#define DFLOAT11_MIN_BLOCKS_PER_SM 1 +#endif + + +__device__ __forceinline__ uint32_t read_byte_msb( + const uint8_t* __restrict__ data, int64_t n_bytes, int64_t bit_pos) { + const int64_t byte_idx = bit_pos >> 3; + const uint32_t shift = static_cast(bit_pos & 7); + const uint32_t hi = (byte_idx < n_bytes) ? data[byte_idx] : 0u; + if (shift == 0u) { + return hi; + } + const uint32_t lo = (byte_idx + 1 < n_bytes) ? data[byte_idx + 1] : 0u; + return ((hi << shift) | (lo >> (8u - shift))) & 0xFFu; +} + +__device__ __forceinline__ uint32_t decode_symbol( + const uint8_t* __restrict__ luts, + int num_luts, // == num_levels + 1 + const uint8_t* __restrict__ encoded, + int64_t n_bytes, + int64_t bit_pos, + uint32_t* code_len) { + const int num_levels = num_luts - 1; + const uint32_t ptr_min = + (num_levels > 1) ? static_cast(256 - (num_levels - 1)) : 256u; + const uint8_t* lens_row = luts + static_cast(num_levels) * 256; + + int level = 0; + int hop = 0; + for (;;) { + const uint32_t byte = read_byte_msb(encoded, n_bytes, bit_pos + hop * 8); + const uint32_t entry = luts[static_cast(level) * 256 + byte]; + if (num_levels > 1 && entry >= ptr_min) { + level = 256 - static_cast(entry); // pointer -> child level + ++hop; + } else { + *code_len = lens_row[entry]; // leaf symbol + return entry; + } + } +} + +__device__ __forceinline__ uint16_t make_bf16_bits(uint32_t exponent, uint32_t sign_mantissa) { + return static_cast( + ((sign_mantissa & 0x80u) << 8) | (exponent << 7) | (sign_mantissa & 0x7Fu)); +} + +__device__ __forceinline__ uint64_t pack_bf16x4(uint32_t e32, uint32_t s32) { + const uint32_t lo = ((e32 & 0x01010101u) << 7) | (s32 & 0x7F7F7F7Fu); + const uint32_t hi = (s32 & 0x80808080u) | ((e32 >> 1) & 0x7F7F7F7Fu); + uint32_t w0, w1; + asm("prmt.b32 %0, %1, %2, 0x5140;" : "=r"(w0) : "r"(lo), "r"(hi)); + asm("prmt.b32 %0, %1, %2, 0x7362;" : "=r"(w1) : "r"(lo), "r"(hi)); + return static_cast(w0) | (static_cast(w1) << 32); +} + +// A sliding MSB-first bit window held in registers. `decode_symbol` re-reads the stream for every symbol, and again for each +// LUT hop because a codeword is not byte-aligned. Walking a contiguous run instead keeps the next bits in `buf` and refills a +// byte at a time, so the hot loop touches the stream once per byte consumed. +// +// The `BitWindowLocal` variant stages the block's slice in shared memory, zero-padded and sized for the worst-case lookahead, +// so the generic refill's bounds check is never taken and is dropped, and addressing shrinks to 32 bit. +struct BitWindow { + const uint8_t* __restrict__ data; + int64_t n_bytes; + int64_t byte_pos; // next byte to pull into `buf` + uint64_t buf; + int n_valid; +}; + +__device__ __forceinline__ void bit_window_refill(BitWindow* w) { + while (w->n_valid <= 32) { + const uint32_t byte = (w->byte_pos < w->n_bytes) ? w->data[w->byte_pos] : 0u; + w->buf |= static_cast(byte) << (56 - w->n_valid); + w->n_valid += 8; + ++w->byte_pos; + } +} + +__device__ __forceinline__ BitWindow bit_window_open( + const uint8_t* __restrict__ data, int64_t n_bytes, int64_t bit_pos) { + BitWindow w; + w.data = data; + w.n_bytes = n_bytes; + w.byte_pos = bit_pos >> 3; + w.buf = 0; + w.n_valid = 0; + bit_window_refill(&w); + const int skip = static_cast(bit_pos & 7); // land on the first bit of the codeword + w.buf <<= skip; + w.n_valid -= skip; + bit_window_refill(&w); + return w; +} + +struct LutWalk { + const uint8_t* luts; // staged decode rows + const uint8_t* lens_row; // per-symbol code length row + uint32_t ptr_min; // entries >= ptr_min are pointers, below are leaf symbols + int num_levels; +}; + +__device__ __forceinline__ uint32_t bit_window_decode( + BitWindow* w, + const uint8_t* __restrict__ luts, + int num_luts) { // == num_levels + 1 + const int num_levels = num_luts - 1; + const uint32_t ptr_min = + (num_levels > 1) ? static_cast(256 - (num_levels - 1)) : 256u; + const uint8_t* lens_row = luts + static_cast(num_levels) * 256; + + int level = 0; + int shift = 56; // peek the byte `hop` bytes into the window without consuming it + for (;;) { + const uint32_t byte = static_cast((w->buf >> shift) & 0xFFu); + const uint32_t entry = luts[static_cast(level) * 256 + byte]; + if (num_levels > 1 && entry >= ptr_min) { + level = 256 - static_cast(entry); // pointer -> child level + shift -= 8; + } else { + const int code_len = lens_row[entry]; // length of the whole codeword + w->buf <<= code_len; + w->n_valid -= code_len; + bit_window_refill(w); + return entry; + } + } +} + +struct BitWindowLocal { + const uint8_t* data; + int byte_pos; + uint64_t buf; + int n_valid; +}; + +__device__ __forceinline__ void bwl_refill(BitWindowLocal* w) { + while (w->n_valid <= 32) { + w->buf |= static_cast(w->data[w->byte_pos]) << (56 - w->n_valid); + w->n_valid += 8; + ++w->byte_pos; + } +} + +__device__ __forceinline__ BitWindowLocal bwl_open(const uint8_t* data, int bit_pos) { + BitWindowLocal w; + w.data = data; + w.byte_pos = bit_pos >> 3; + w.buf = 0; + w.n_valid = 0; + bwl_refill(&w); + const int skip = bit_pos & 7; + w.buf <<= skip; + w.n_valid -= skip; + bwl_refill(&w); + return w; +} + +__device__ __forceinline__ uint32_t bwl_decode(BitWindowLocal* w, const LutWalk* ctx) { + int level = 0; + int shift = 56; + for (;;) { + const uint32_t byte = static_cast((w->buf >> shift) & 0xFFu); + const uint32_t entry = ctx->luts[level * 256 + byte]; + if (ctx->num_levels > 1 && entry >= ctx->ptr_min) { + level = 256 - static_cast(entry); + shift -= 8; + } else { + const int code_len = ctx->lens_row[entry]; + w->buf <<= code_len; + w->n_valid -= code_len; + bwl_refill(w); + return entry; + } + } +} + +__device__ __forceinline__ uint32_t bswap32(uint32_t x) { + uint32_t r; + asm("prmt.b32 %0, %1, 0, 0x0123;" : "=r"(r) : "r"(x)); + return r; +} + +// Register-resident bitstream for bytes_per_thread == 16: a thread's whole span, its region plus the worst-case lookahead, is +// preloaded into three big-endian u64 words, so the walk issues no stream loads, only funnel shifts. +struct RegStream { + uint64_t w0, w1, w2; + int bit; +}; + +__device__ __forceinline__ RegStream reg_stream_open(const uint8_t* span16, int gap_bits) { + const uint4 q = *reinterpret_cast(span16); + const uint2 r = *reinterpret_cast(span16 + 16); + RegStream s; + s.w0 = (static_cast(bswap32(q.x)) << 32) | bswap32(q.y); + s.w1 = (static_cast(bswap32(q.z)) << 32) | bswap32(q.w); + s.w2 = (static_cast(bswap32(r.x)) << 32) | bswap32(r.y); + s.bit = gap_bits; + return s; +} + +__device__ __forceinline__ uint32_t rs_peek32(const RegStream* s) { + const int b = s->bit; + const uint64_t v = (s->w0 << b) | (b ? (s->w1 >> (64 - b)) : 0ull); + return static_cast(v >> 32); +} + +__device__ __forceinline__ uint32_t rs_decode(RegStream* s, const LutWalk* ctx) { + uint32_t peek = rs_peek32(s); + int lvl_off = 0; + for (;;) { + const uint32_t byte = peek >> 24; + const uint32_t entry = ctx->luts[lvl_off + byte]; + if (entry >= ctx->ptr_min) { // pointer -> child level (ptr_min == 256: never true) + lvl_off = (256u - entry) << 8; + peek <<= 8; + } else { + const int code_len = ctx->lens_row[entry]; + s->bit += code_len; + if (s->bit >= 64) { // rotate the word window; bit + code_len < 95 < 128, once is enough + s->w0 = s->w1; + s->w1 = s->w2; + s->w2 = 0ull; + s->bit -= 64; + } + return entry; + } + } +} + +// Warp-shuffle scan per warp, then a scan of the warp totals: two barriers instead of a full Hillis-Steele sweep over the +// block. +__device__ __forceinline__ int32_t block_exclusive_scan(int32_t* s_scan, int32_t value) { + if ((blockDim.x & 31u) == 0u && blockDim.x >= 32u) { + const unsigned full = 0xFFFFFFFFu; + const int lane = static_cast(threadIdx.x) & 31; + const int warp_id = static_cast(threadIdx.x) >> 5; + const int nwarps = static_cast(blockDim.x) >> 5; + + int32_t v = value; // inclusive scan within the warp +#pragma unroll + for (int off = 1; off < 32; off <<= 1) { + const int32_t n = __shfl_up_sync(full, v, off); + if (lane >= off) v += n; + } + if (lane == 31) { + s_scan[warp_id] = v; + } + __syncthreads(); + + if (warp_id == 0) { + const unsigned wmask = (nwarps == 32) ? full : ((1u << nwarps) - 1u); + int32_t w = (lane < nwarps) ? s_scan[lane] : 0; +#pragma unroll + for (int off = 1; off < 32; off <<= 1) { + const int32_t n = __shfl_up_sync(wmask, w, off); + if (lane >= off && lane < nwarps) w += n; + } + if (lane < nwarps) { + s_scan[lane] = w; + } + } + __syncthreads(); + + const int32_t prefix = (warp_id > 0) ? s_scan[warp_id - 1] : 0; + return prefix + (v - value); + } + + s_scan[threadIdx.x] = value; + __syncthreads(); + for (unsigned offset = 1; offset < blockDim.x; offset <<= 1) { + const int32_t addend = (threadIdx.x >= offset) ? s_scan[threadIdx.x - offset] : 0; + __syncthreads(); // every read of round `offset` completes before any write + if (threadIdx.x >= offset) { + s_scan[threadIdx.x] += addend; + } + __syncthreads(); + } + return s_scan[threadIdx.x] - value; // inclusive -> exclusive +} + + +extern "C" __global__ void __launch_bounds__(DFLOAT11_THREADS_PER_BLOCK, DFLOAT11_MIN_BLOCKS_PER_SM) dfloat11_decode_kernel( + const uint8_t* __restrict__ luts, + const uint8_t* __restrict__ encoded, + const uint8_t* __restrict__ sign_mantissa, + const uint32_t* __restrict__ output_positions, + const uint16_t* __restrict__ thread_meta, + uint16_t* __restrict__ out_bits, // bf16 bit patterns + int num_luts, + int64_t n_bytes, + int64_t n_elements, + int bytes_per_thread, + int stage_enc_bytes, // 0 = read the bitstream straight from global memory + int stage_elems, // 0 = scatter to out_bits directly instead of staging + int sm_u32_aligned) { // 1 = sign_mantissa pointer is 4-byte aligned (vector write-back) + extern __shared__ __align__(16) uint8_t s_raw[]; + int32_t* const s_scan = reinterpret_cast(s_raw); + uint8_t* const s_luts = s_raw + blockDim.x * sizeof(int32_t); + uint8_t* const s_enc = s_luts + static_cast(num_luts) * 256; + + const int64_t out_base = static_cast(output_positions[blockIdx.x]); + uint8_t* const s_exp = s_enc + stage_enc_bytes + (out_base & 3); + + const int lut_bytes = num_luts * 256; + for (int i = threadIdx.x; i < lut_bytes; i += blockDim.x) { + s_luts[i] = luts[i]; + } + + const int64_t block_byte_base = + static_cast(blockIdx.x) * blockDim.x * bytes_per_thread; + const bool direct_ok = (bytes_per_thread == 16) && (stage_enc_bytes > 0) && + (block_byte_base + stage_enc_bytes <= n_bytes) && + ((reinterpret_cast(encoded + block_byte_base) & 15u) == 0u); + if (stage_enc_bytes > 0 && !direct_ok && + ((reinterpret_cast(encoded + block_byte_base) & 15u) == 0u) && + blockDim.x * 16 >= stage_enc_bytes) { + const uint4* src4 = reinterpret_cast(encoded + block_byte_base); + uint4* dst4 = reinterpret_cast(s_enc); + const int n4 = stage_enc_bytes >> 4; // floor; the tail is handled below + if (static_cast(threadIdx.x) < n4) { + const int64_t src = block_byte_base + (static_cast(threadIdx.x) << 4); + uint4 v; + if (src + 16 <= n_bytes) { + v = src4[threadIdx.x]; + } else { // partial sector: rebuild byte-exactly like the scalar path (zero padding) + uint8_t tmp[16]; +#pragma unroll + for (int k = 0; k < 16; ++k) { + tmp[k] = (src + k < n_bytes) ? encoded[src + k] : 0u; + } + v.x = tmp[0] | (tmp[1] << 8) | (tmp[2] << 16) | (static_cast(tmp[3]) << 24); + v.y = tmp[4] | (tmp[5] << 8) | (tmp[6] << 16) | (static_cast(tmp[7]) << 24); + v.z = tmp[8] | (tmp[9] << 8) | (tmp[10] << 16) | (static_cast(tmp[11]) << 24); + v.w = tmp[12] | (tmp[13] << 8) | (tmp[14] << 16) | (static_cast(tmp[15]) << 24); + } + dst4[threadIdx.x] = v; + } + for (int i = (n4 << 4) + threadIdx.x; i < stage_enc_bytes; i += blockDim.x) { + const int64_t src = block_byte_base + i; + s_enc[i] = (src < n_bytes) ? encoded[src] : 0u; + } + } else if (stage_enc_bytes > 0 && !direct_ok) { + for (int i = threadIdx.x; i < stage_enc_bytes; i += blockDim.x) { + const int64_t src = block_byte_base + i; + s_enc[i] = (src < n_bytes) ? encoded[src] : 0u; + } + } + + const int64_t gt = static_cast(blockIdx.x) * blockDim.x + threadIdx.x; + const uint32_t meta = thread_meta[gt]; + const int64_t start_bit = + gt * static_cast(bytes_per_thread) * 8 + (meta >> kMetaCountBits); + int32_t count = static_cast(meta & kMetaCountMask); + + __syncthreads(); // s_luts / s_enc are now visible to the whole block + + const int32_t local_base = block_exclusive_scan(s_scan, count); + + if (stage_enc_bytes > 0 && stage_elems > 0) { + if (local_base + count > stage_elems) { // only reachable on a corrupt buffer + count = (local_base < stage_elems) ? (stage_elems - local_base) : 0; + } + + LutWalk ctx; + ctx.luts = s_luts; + ctx.num_levels = num_luts - 1; + ctx.ptr_min = + (ctx.num_levels > 1) ? static_cast(256 - (ctx.num_levels - 1)) : 256u; + ctx.lens_row = s_luts + ctx.num_levels * 256; + + const int cursor = static_cast(start_bit - block_byte_base * 8); + if (bytes_per_thread == 16) { + const uint8_t* const span = + direct_ok ? (encoded + block_byte_base) : s_enc; + RegStream rs = reg_stream_open(span + (static_cast(threadIdx.x) << 4), + static_cast(meta >> kMetaCountBits)); + for (int32_t i = 0; i < count; ++i) { + s_exp[local_base + i] = static_cast(rs_decode(&rs, &ctx)); + } + } else { + BitWindowLocal window = bwl_open(s_enc, cursor); + for (int32_t i = 0; i < count; ++i) { + s_exp[local_base + i] = static_cast(bwl_decode(&window, &ctx)); + } + } + __syncthreads(); + + const int64_t block_elems = + static_cast(output_positions[blockIdx.x + 1]) - out_base; + + if (sm_u32_aligned && block_elems >= 8) { + const int64_t p = (8 - (out_base & 7)) & 7; // first 8-aligned element index + int64_t j = threadIdx.x; + for (; j < block_elems && j < p; j += blockDim.x) { + const int64_t gidx = out_base + j; + if (gidx < n_elements) { + out_bits[gidx] = make_bf16_bits(s_exp[j], sign_mantissa[gidx]); + } + } + const int64_t nvec = (block_elems - p) >> 3; + const bool sm_u64_aligned = ((reinterpret_cast(sign_mantissa) & 7u) == 0u); + const bool in_bounds = out_base + block_elems <= n_elements; + for (int64_t c = threadIdx.x; c < nvec; c += blockDim.x) { + const int64_t v = p + (c << 3); + const int64_t gidx = out_base + v; + const uint32_t e0123 = *reinterpret_cast(s_exp + v); + const uint32_t e4567 = *reinterpret_cast(s_exp + v + 4); + uint32_t s0123, s4567; + if (sm_u64_aligned) { + const uint2 s8 = *reinterpret_cast(sign_mantissa + gidx); + s0123 = s8.x; + s4567 = s8.y; + } else { + s0123 = *reinterpret_cast(sign_mantissa + gidx); + s4567 = *reinterpret_cast(sign_mantissa + gidx + 4); + } + const uint64_t lo4 = pack_bf16x4(e0123, s0123); + const uint64_t hi4 = pack_bf16x4(e4567, s4567); + if (in_bounds) { + uint4 o; + o.x = static_cast(lo4); + o.y = static_cast(lo4 >> 32); + o.z = static_cast(hi4); + o.w = static_cast(hi4 >> 32); + *reinterpret_cast(out_bits + gidx) = o; + } else { // ragged final chunk of the final block +#pragma unroll + for (int k = 0; k < 8; ++k) { + if (gidx + k < n_elements) { + out_bits[gidx + k] = make_bf16_bits(s_exp[v + k], sign_mantissa[gidx + k]); + } + } + } + } + const int64_t tail = (block_elems - p) & 7; + const int64_t vtail = p + (nvec << 3); + if (threadIdx.x < tail) { + const int64_t gidx = out_base + vtail + threadIdx.x; + if (gidx < n_elements) { + out_bits[gidx] = make_bf16_bits(s_exp[vtail + threadIdx.x], sign_mantissa[gidx]); + } + } + return; + } + + for (int64_t j = threadIdx.x; j < block_elems; j += blockDim.x) { + const int64_t gidx = out_base + j; + if (gidx < n_elements) { + out_bits[gidx] = make_bf16_bits(s_exp[j], sign_mantissa[gidx]); + } + } + return; + } + + const uint8_t* const stream = (stage_enc_bytes > 0) ? s_enc : encoded; + const int64_t stream_bytes = (stage_enc_bytes > 0) ? stage_enc_bytes : n_bytes; + const int64_t cursor = (stage_enc_bytes > 0) ? (start_bit - block_byte_base * 8) : start_bit; + + if (stage_elems > 0) { + if (local_base + count > stage_elems) { // only reachable on a corrupt buffer + count = (local_base < stage_elems) ? (stage_elems - local_base) : 0; + } + BitWindow window = bit_window_open(stream, stream_bytes, cursor); + for (int32_t i = 0; i < count; ++i) { + s_exp[local_base + i] = + static_cast(bit_window_decode(&window, s_luts, num_luts)); + } + __syncthreads(); + + const int64_t block_elems = + static_cast(output_positions[blockIdx.x + 1]) - out_base; + for (int64_t j = threadIdx.x; j < block_elems; j += blockDim.x) { + const int64_t gidx = out_base + j; + if (gidx < n_elements) { + out_bits[gidx] = make_bf16_bits(s_exp[j], sign_mantissa[gidx]); + } + } + return; + } + + const int64_t base = out_base + local_base; + BitWindow window = bit_window_open(stream, stream_bytes, cursor); + for (int32_t i = 0; i < count; ++i) { + const uint32_t exponent = bit_window_decode(&window, s_luts, num_luts); + const int64_t gidx = base + i; + if (gidx < n_elements) { + out_bits[gidx] = make_bf16_bits(exponent, sign_mantissa[gidx]); + } + } +} + + +// Each block accumulates in shared uint32 bins and then contributes at most one global uint64 atomic per bin, so the encode +// never materializes a per-element exponent array, which would be larger than the weight being encoded. +extern "C" __global__ void dfloat11_exponent_histogram_kernel( + const uint16_t* __restrict__ bf16_bits, + uint64_t* __restrict__ histogram, + int64_t n_elements) { + __shared__ uint32_t bins[256]; + if (threadIdx.x < 256) { + bins[threadIdx.x] = 0; + } + __syncthreads(); + + int64_t idx = static_cast(blockIdx.x) * blockDim.x + threadIdx.x; + const int64_t stride = static_cast(gridDim.x) * blockDim.x; + for (; idx < n_elements; idx += stride) { + const uint32_t exponent = (bf16_bits[idx] >> 7) & 0xFFu; + atomicAdd(&bins[exponent], 1u); + } + __syncthreads(); + + if (threadIdx.x < 256 && bins[threadIdx.x] != 0) { + atomicAdd(reinterpret_cast(histogram + threadIdx.x), + static_cast(bins[threadIdx.x])); + } +} + + +extern "C" __global__ void dfloat11_split_len_kernel( + const uint16_t* __restrict__ bf16_bits, + const int32_t* __restrict__ code_len, // 256 entries + uint8_t* __restrict__ exponent, + uint8_t* __restrict__ sign_mantissa, + uint8_t* __restrict__ len_out, + int64_t n_elements) { + const int64_t idx = static_cast(blockIdx.x) * blockDim.x + threadIdx.x; + if (idx >= n_elements) { + return; + } + const uint32_t bits = bf16_bits[idx]; + const uint32_t exp = (bits >> 7) & 0xFFu; + exponent[idx] = static_cast(exp); + sign_mantissa[idx] = + static_cast(((bits >> 8) & 0x80u) | (bits & 0x7Fu)); + len_out[idx] = static_cast(code_len[exp]); +} + + +__device__ __forceinline__ int64_t lower_bound_i64( + const int64_t* __restrict__ a, int64_t m, int64_t key) { + int64_t lo = 0; + int64_t hi = m; + while (lo < hi) { + const int64_t mid = (lo + hi) >> 1; + if (a[mid] < key) { + lo = mid + 1; + } else { + hi = mid; + } + } + return lo; +} + + +__device__ __forceinline__ uint32_t byte_contrib(uint64_t vL, int64_t byte_start, int64_t p) { + const int shift = static_cast(byte_start - p); // in [-7, 31] for overlapping codes + const int sh = 24 - shift; + if (sh >= 0) { + return static_cast((vL >> sh) & 0xFFu); + } + return static_cast((vL << (-sh)) & 0xFFu); +} + + +// Consecutive output bytes walk the same monotone prefix array, so a thread pays one lower_bound per chunk and then advances +// the symbol cursor linearly. The chunk is wide enough that binary searches amortize over several output bytes and narrow +// enough that the cursor advance stays short and the thread count stays high. +constexpr int kPackBytesPerThread = 4; + +extern "C" __global__ void dfloat11_pack_kernel( + const int64_t* __restrict__ pref, // n_elements + 1 exclusive prefix sums of code lengths + const uint8_t* __restrict__ exponent, + const int32_t* __restrict__ code_len, // 256 + const int32_t* __restrict__ code_val, // 256 + uint8_t* __restrict__ encoded, + int64_t n_elements, + int64_t n_bytes, + int64_t total_bits, + int eof_len, + uint32_t eof_val) { + const int64_t chunk = static_cast(blockIdx.x) * blockDim.x + threadIdx.x; + const int64_t first_byte = chunk * kPackBytesPerThread; + if (first_byte >= n_bytes) { + return; + } + const int byte_count = static_cast( + (n_bytes - first_byte < kPackBytesPerThread) ? (n_bytes - first_byte) + : kPackBytesPerThread); + + const int64_t first_bit = first_byte * 8; + int64_t i_start = lower_bound_i64(pref, n_elements + 1, first_bit + 1) - 1; + if (i_start < 0) { + i_start = 0; + } + + uint64_t packed = 0; +#pragma unroll + for (int k = 0; k < kPackBytesPerThread; ++k) { + if (k >= byte_count) { + break; + } + const int64_t j = first_byte + k; + const int64_t byte_start = j * 8; + const int64_t byte_end = byte_start + 8; + + while (i_start < n_elements) { + const uint32_t sym = exponent[i_start]; + const int64_t end = pref[i_start] + code_len[sym]; + if (end > byte_start) { + break; + } + ++i_start; + } + + uint32_t acc = 0; + for (int64_t i = i_start; i < n_elements && pref[i] < byte_end; ++i) { + const int64_t p = pref[i]; + const uint32_t sym = exponent[i]; + const int b = code_len[sym]; + const int64_t end = p + b; + if (end <= byte_start) { + continue; + } + const uint64_t vL = + static_cast(static_cast(code_val[sym])) << (32 - b); + acc |= byte_contrib(vL, byte_start, p); + } + + if (j == n_bytes - 1) { + const int64_t r = total_bits - byte_start; + if (r > 0 && r < 8) { + const uint64_t eL = static_cast(eof_val) << (32 - eof_len); + acc |= byte_contrib(eL, byte_start, total_bits); + } + } + packed |= static_cast(acc) << (k * 8); + } + + if (byte_count == kPackBytesPerThread) { + if constexpr (kPackBytesPerThread == 8) { + *reinterpret_cast(encoded + first_byte) = packed; + } else if constexpr (kPackBytesPerThread == 4) { + *reinterpret_cast(encoded + first_byte) = static_cast(packed); + } else { + *reinterpret_cast(encoded + first_byte) = static_cast(packed); + } + } else { +#pragma unroll + for (int k = 0; k < kPackBytesPerThread; ++k) { + if (k < byte_count) { + encoded[first_byte + k] = static_cast(packed >> (k * 8)); + } + } + } +} + + +extern "C" __global__ void dfloat11_thread_meta_kernel( + const int64_t* __restrict__ pref, + uint16_t* __restrict__ thread_meta, + int64_t n_elements, + int64_t total_bits, + int64_t region_bits, + int64_t n_regions) { + const int64_t t = static_cast(blockIdx.x) * blockDim.x + threadIdx.x; + if (t >= n_regions) { + return; + } + const int64_t key_lo = t * region_bits; + if (key_lo > total_bits) { + thread_meta[t] = 0; + return; + } + const int64_t key_hi = key_lo + region_bits; + int64_t lb_lo = lower_bound_i64(pref, n_elements + 1, key_lo); + if (lb_lo > n_elements) { + lb_lo = n_elements; + } + int64_t lb_hi = (key_hi > total_bits) + ? n_elements + : lower_bound_i64(pref, n_elements + 1, key_hi); + if (lb_hi > n_elements) { + lb_hi = n_elements; + } + const uint32_t count = static_cast(lb_hi - lb_lo); + const uint32_t gap = (count > 0) ? static_cast(pref[lb_lo] - key_lo) : 0u; + thread_meta[t] = static_cast((gap << kMetaCountBits) | count); +} + + +extern "C" __global__ void dfloat11_output_positions_kernel( + const int64_t* __restrict__ pref, + uint32_t* __restrict__ output_positions, + int64_t n_elements, + int64_t block_bits, + int64_t num_blocks) { + const int64_t blk = static_cast(blockIdx.x) * blockDim.x + threadIdx.x; + if (blk > num_blocks) { + return; + } + if (blk == num_blocks) { + output_positions[blk] = static_cast(n_elements); + return; + } + const int64_t key = blk * block_bits; + const int64_t lb = lower_bound_i64(pref, n_elements + 1, key); + output_positions[blk] = static_cast((lb <= n_elements) ? lb : n_elements); +} diff --git a/entropack/schemes/dfloat11/eager.py b/entropack/schemes/dfloat11/eager.py new file mode 100644 index 0000000..7002261 --- /dev/null +++ b/entropack/schemes/dfloat11/eager.py @@ -0,0 +1,239 @@ +from copy import copy + +import numpy as np +import torch + +from .format import ( + DFloat11Buffers, make_layout, max_block_elems, pack_thread_meta, + reconstruct_bf16_bits, +) + + +def _huffman_codec(): + from dahuffman import HuffmanCodec + + return HuffmanCodec + + +def exponent_counter(weight: torch.Tensor, threads_per_block: int | None = None) -> dict: + W = weight.reshape(-1).view(torch.int16) + exponent_8bits = ((W >> 7) & 0xFF).to(torch.int64) + counts = torch.bincount(exponent_8bits, minlength=256).cpu().tolist() + return {i: int(c) for i, c in enumerate(counts) if c > 0} + + +def get_32bit_codec(counter: dict): + HuffmanCodec = _huffman_codec() + codec = HuffmanCodec.from_frequencies(counter) + table = codec.get_code_table() + max_len = 0 + for _, (length, _) in table.items(): + max_len = max(max_len, length) + + compressed_codec = codec + compressed_counter = counter + + min_k = 2 + freq = np.array(list(counter.values())) + while max_len > 32: + min_indices = np.argpartition(freq, min_k)[:min_k] + min_k += 1 + min_keys = np.array(list(counter.keys()))[min_indices] + + compressed_counter = copy(counter) + for k in min_keys: + compressed_counter[k] = 1 + compressed_codec = HuffmanCodec.from_frequencies(compressed_counter) + table = compressed_codec.get_code_table() + max_len = 0 + for _, (length, _) in table.items(): + max_len = max(max_len, length) + + return compressed_codec, compressed_counter, table + + +def get_luts(table: dict) -> torch.Tensor: + prefixes = [""] + + for key, (bits, val) in table.items(): + if isinstance(key, int): + prefix = bin(val)[2:].rjust(bits, "0")[: ((bits - 1) // 8 * 8)] + if prefix not in prefixes: + prefixes.append(prefix) + + prefixes.sort(key=len) + + luts = np.zeros((len(prefixes), 256), dtype=np.uint8) + + for pi, p in enumerate(prefixes): + bytes_dict = {} + pl = len(p) // 8 + for key, (bits, val) in table.items(): + if isinstance(key, int): + bin_val = bin(val)[2:].rjust(bits, "0") + + if bin_val.startswith(p): + if (bits - 1) // 8 == pl: + dict_key = int(bin_val[(pl * 8) :].ljust(8, "0"), 2) + dict_value = key + else: + dict_key = int(bin_val[(pl * 8) : (pl * 8 + 8)], 2) + dict_value = 256 - prefixes.index(bin_val[: (pl * 8 + 8)]) + + if dict_key in bytes_dict and bytes_dict[dict_key] != dict_value: + raise ValueError(f"Key {dict_key} already exists in {bytes_dict}") + else: + bytes_dict[dict_key] = dict_value + + curr_val = 0 + for i in range(256): + if i in bytes_dict: + curr_val = bytes_dict[i] + luts[pi, i] = curr_val + + lens = np.zeros((1, 256), dtype=np.uint8) + for key, (bits, _val) in table.items(): + if isinstance(key, int): + lens[-1, key] = bits + + return torch.from_numpy(np.concatenate((luts, lens), axis=0)) + + +def _encode_bitstream(data, codec, bytes_per_thread: int, threads_per_block: int): + encoded = [] + + gaps = [] + counts = [] + output_positions = [] + + region_bits = 8 * bytes_per_thread + block_bits = region_bits * threads_per_block + + buffer = 0 + size = 0 + total_size = 0 + element_count = 0 + for s in data: + if total_size // region_bits + 1 > len(gaps): + gaps.append(total_size - total_size // region_bits * region_bits) + counts.append(0) + + if total_size // block_bits + 1 > len(output_positions): + output_positions.append(element_count) + + counts[-1] += 1 + + b, v = codec._table[s] + buffer = (buffer << b) + v + size += b + total_size += b + element_count += 1 + while size >= 8: + byte = buffer >> (size - 8) + encoded.append(byte) + buffer = buffer - (byte << (size - 8)) + size -= 8 + + if size > 0: + if total_size // region_bits + 1 > len(gaps): + gaps.append(0) + counts.append(0) + + if total_size // block_bits + 1 > len(output_positions): + output_positions.append(element_count) + + b, v = codec._table[codec._eof] + buffer = (buffer << b) + v + size += b + if size >= 8: + byte = buffer >> (size - 8) + else: + byte = buffer << (8 - size) + encoded.append(byte) + + output_positions.append(len(data)) + + blocks_per_grid = int(np.ceil(len(encoded) / (threads_per_block * bytes_per_thread))) + n_regions = threads_per_block * blocks_per_grid + gaps.extend([0] * (n_regions - len(gaps))) + counts.extend([0] * (n_regions - len(counts))) + + return ( + np.frombuffer(bytes(encoded), dtype=np.uint8).copy(), np.array(gaps, dtype=np.int64), np.array(counts, dtype=np.int64), + np.array(output_positions, dtype=np.uint32), + ) + + +def encode_weights(weights, codec, bytes_per_thread: int, threads_per_block: int): + W_combined = torch.cat(weights).view(torch.int16) + + exponent_8bits = ((W_combined >> 7) & 0xFF).to(torch.uint8) + other_8bits = ((W_combined >> 8) & 0x80 | (W_combined & 0x7F)).to(torch.uint8) + + encoded, gaps, counts, output_positions = _encode_bitstream( + exponent_8bits.tolist(), codec, bytes_per_thread, threads_per_block + ) + + return ( + torch.from_numpy(encoded), other_8bits, torch.from_numpy(output_positions), pack_thread_meta(gaps, counts), + make_layout(bytes_per_thread, threads_per_block, max_block_elems(output_positions)), + ) + + +def _decode_exponents(luts: np.ndarray, encoded: np.ndarray, n_elements: int) -> np.ndarray: + lut = luts.astype(np.int64) + lens_row = lut[-1] + num_levels = lut.shape[0] - 1 + ptr_min = 256 - (num_levels - 1) if num_levels > 1 else 256 + + bits = np.unpackbits(encoded.astype(np.uint8)) + n_bits = bits.size + + def read_byte(offset: int) -> int: + if offset + 8 <= n_bits: + seg = bits[offset : offset + 8] + else: + seg = np.zeros(8, np.uint8) + avail = n_bits - offset + if avail > 0: + seg[:avail] = bits[offset:] + return int(np.packbits(seg)[0]) + + out = np.empty(n_elements, np.int64) + cursor = 0 + for i in range(n_elements): + level = 0 + hop = 0 + while True: + entry = lut[level][read_byte(cursor + hop * 8)] + if num_levels > 1 and entry >= ptr_min: + level = 256 - entry + hop += 1 + else: + out[i] = entry + cursor += int(lens_row[entry]) + break + return out + + +def decode(buffers: DFloat11Buffers) -> torch.Tensor: + encoded_exponent, sign_mantissa, luts = buffers.encoded_exponent, buffers.sign_mantissa, buffers.luts + n_elements = sign_mantissa.numel() + exponents = _decode_exponents(luts.detach().cpu().numpy(), encoded_exponent.detach().cpu().numpy(), n_elements) + sm = sign_mantissa.detach().cpu().numpy().astype(np.uint8) + bf16_bits = reconstruct_bf16_bits(exponents, sm) + flat = torch.from_numpy(bf16_bits.view(np.int16)).view(torch.bfloat16) + return flat.to(sign_mantissa.device) + + +def encode( + *, weight: torch.Tensor, codec, luts: torch.Tensor, bytes_per_thread: int, threads_per_block: int, +) -> DFloat11Buffers: + flat = weight.reshape(-1).cpu() + encoded_exponent, sign_mantissa, output_positions, thread_meta, layout = encode_weights( + [flat], codec, bytes_per_thread, threads_per_block + ) + return DFloat11Buffers( + encoded_exponent=encoded_exponent, sign_mantissa=sign_mantissa, luts=luts, + output_positions=output_positions, thread_meta=thread_meta, layout=layout, + ) diff --git a/entropack/schemes/dfloat11/format.py b/entropack/schemes/dfloat11/format.py new file mode 100644 index 0000000..db472c3 --- /dev/null +++ b/entropack/schemes/dfloat11/format.py @@ -0,0 +1,146 @@ +import math +from dataclasses import dataclass +from typing import Annotated, NamedTuple + +import numpy as np +import torch + +from ..config import CompressionConfig, Range +from ..base import cached_parse + +BLOCK_SIZE = 256 +MAX_RESIDENT_BLOCKS_PER_SM = 32 + + +@dataclass +class DFloat11Config(CompressionConfig): + """Lossless BF16 compression settings. + + The coding-region settings are stored with the compressed tensor. Decoding reads them + from the representation rather than from the configuration.""" + + #: Bitstream bytes per coding region. Controls metadata overhead and decode parallelism. + bytes_per_thread: Annotated[int | None, Range(1, None)] = 16 + #: Threads per block in the encoded layout. + threads_per_block: Annotated[int | None, Range(1, None)] = 128 + + +META_COUNT_BITS = 11 +META_COUNT_MASK = (1 << META_COUNT_BITS) - 1 + + +class DFloat11Buffers(NamedTuple): + encoded_exponent: torch.Tensor + sign_mantissa: torch.Tensor + luts: torch.Tensor + output_positions: torch.Tensor + thread_meta: torch.Tensor + layout: torch.Tensor + + +PACKED_KEYS = DFloat11Buffers._fields + + +def reconstruct_bf16_bits(exponent: np.ndarray, sign_mantissa: np.ndarray) -> np.ndarray: + exp = exponent.astype(np.uint16) + sm = sign_mantissa.astype(np.uint16) + return (((sm & 0x80) << 8) | (exp << 7) | (sm & 0x7F)).astype(np.uint16) + + +def pack_thread_meta(gaps, counts) -> torch.Tensor: + g = np.asarray(gaps, dtype=np.int64) + c = np.asarray(counts, dtype=np.int64) + if g.size != c.size: + raise ValueError(f"gaps/counts length mismatch: {g.size} vs {c.size}") + if g.size and int(g.max()) > 31: + raise ValueError(f"gap {int(g.max())} exceeds 5 bits; max Huffman code length must be < 32") + if c.size and int(c.max()) > META_COUNT_MASK: + raise ValueError(f"per-thread symbol count {int(c.max())} exceeds {META_COUNT_BITS} bits; lower BYTES_PER_THREAD") + return torch.from_numpy(((g << META_COUNT_BITS) | c).astype(np.uint16)) + + +def make_layout(bytes_per_thread: int, threads_per_block: int, max_block_elems: int): + return torch.tensor([bytes_per_thread, threads_per_block, max_block_elems], dtype=torch.int32) + + +def parse_layout(layout) -> tuple[int, int, int]: + if layout.numel() != 3: + raise ValueError(f"dfloat11 layout must contain 3 int32 values, got {layout.numel()}") + bpt, tpb, max_block_elems = (int(v) for v in layout.detach().cpu().tolist()) + if bpt <= 0 or tpb <= 0 or max_block_elems < 0: + raise ValueError(f"invalid dfloat11 layout values: bpt={bpt}, tpb={tpb}, max_block_elems={max_block_elems}") + return bpt, tpb, max_block_elems + + +def parse_layout_cached(layout) -> tuple[int, int, int]: + return cached_parse(layout, parse_layout, "_dfloat11_layout") + + +def max_block_elems(output_positions) -> int: + op = np.asarray(output_positions, dtype=np.int64) + if op.size < 2: + return int(op[0]) if op.size else 0 + return int(np.diff(op).max()) + + +def validate_packed(buffers: dict, shape=None) -> None: + missing = [key for key in PACKED_KEYS if key not in buffers] + if missing: + raise ValueError(f"dfloat11 packed data is missing buffers: {missing}") + if not all(isinstance(buffers[key], torch.Tensor) for key in PACKED_KEYS): + raise TypeError("dfloat11 packed buffers must be torch.Tensor values") + + expected = { + "encoded_exponent": (torch.uint8, 1), "sign_mantissa": (torch.uint8, 1), + "luts": (torch.uint8, 2), "output_positions": (torch.uint32, 1), + "thread_meta": (torch.uint16, 1), "layout": (torch.int32, 1), + } + for key, (dtype, ndim) in expected.items(): + tensor = buffers[key] + if tensor.dtype != dtype or tensor.ndim != ndim: + raise ValueError( + f"dfloat11 buffer '{key}' must be {ndim}D {dtype}, got shape={tuple(tensor.shape)}, dtype={tensor.dtype}" + ) + + devices = {buffers[key].device for key in PACKED_KEYS} + if len(devices) != 1: + raise ValueError(f"dfloat11 packed buffers must share one device, got {devices}") + if buffers["luts"].shape[0] < 2 or buffers["luts"].shape[1] != 256: + raise ValueError(f"dfloat11 luts must have shape (num_levels + 1, 256), got {tuple(buffers['luts'].shape)}") + + bpt, tpb, max_elems = parse_layout_cached(buffers["layout"]) + n_elements = buffers["sign_mantissa"].numel() + if n_elements == 0: + raise ValueError("dfloat11 does not support empty tensors") + if shape is not None: + normalized_shape = tuple(shape) + if any(not isinstance(dim, int) or dim < 0 for dim in normalized_shape): + raise ValueError(f"invalid dfloat11 tensor shape: {normalized_shape}") + if math.prod(normalized_shape) != n_elements: + raise ValueError( + f"dfloat11 shape {normalized_shape} has {math.prod(normalized_shape)} " + f"elements but sign_mantissa has {n_elements}" + ) + + n_bytes = buffers["encoded_exponent"].numel() + blocks = (n_bytes + bpt * tpb - 1) // (bpt * tpb) + if buffers["thread_meta"].numel() != blocks * tpb: + raise ValueError("dfloat11 thread_meta length does not match layout/bitstream") + if buffers["output_positions"].numel() != blocks + 1: + raise ValueError("dfloat11 output_positions length does not match layout/bitstream") + + positions = buffers["output_positions"].detach().cpu().to(torch.int64) + if positions[0] != 0 or positions[-1] != n_elements: + raise ValueError("dfloat11 output_positions endpoints are invalid") + differences = positions[1:] - positions[:-1] + if (differences < 0).any() or (differences.max() if differences.numel() else 0) != max_elems: + raise ValueError("dfloat11 output_positions are not monotone or mismatch layout") + + metadata = buffers["thread_meta"].detach().cpu().to(torch.int64) + gaps = metadata >> META_COUNT_BITS + counts = metadata & META_COUNT_MASK + if (gaps > 31).any() or counts.sum() != n_elements: + raise ValueError("dfloat11 thread_meta fields are invalid") + lengths = buffers["luts"][-1].detach().cpu() + if int(lengths.max()) > 32: + raise ValueError("dfloat11 Huffman code length exceeds 32 bits") diff --git a/entropack/schemes/lattice_rans/__init__.py b/entropack/schemes/lattice_rans/__init__.py new file mode 100644 index 0000000..0d5228b --- /dev/null +++ b/entropack/schemes/lattice_rans/__init__.py @@ -0,0 +1,60 @@ +import torch + +from ..base import Scheme, packed_buffers, register_scheme +from ..checks import prepare_weight +from .format import ( + LATTICE_DIM, PACKED_KEYS, SUPPORTED_DTYPES, LatticeBuffers, LatticeRANSConfig, recommended_tile_elements, + validate_packed, +) + +__all__ = ["LatticeRANSScheme", "LatticeRANSConfig"] + + +class LatticeRANSScheme(Scheme): + name = "lattice_rans" + buffer_names = PACKED_KEYS + lossless = False + priority = 0 + dtypes = SUPPORTED_DTYPES + lanes = {"eager": "eager", "cuda": "cuda"} + + def encode(self, weight: torch.Tensor, config: LatticeRANSConfig) -> dict: + weight = prepare_weight(weight, scheme=self.name, ndim=2, dtypes=SUPPORTED_DTYPES, require_finite=True) + shape, device = tuple(weight.shape), weight.device + pad = -shape[1] % LATTICE_DIM + if pad: + weight = torch.cat([weight, weight.new_zeros(shape[0], pad)], dim=1) + lane, run_on = self.lane_for(weight, config.execution_backend) + if run_on is not None and device != run_on: + weight = weight.to(run_on) + tile_elements = ( + recommended_tile_elements(float(config.target_bpp)) if config.tile_elements is None + else int(config.tile_elements) + ) + packed = lane.encode( + weight=weight, target_bpp=float(config.target_bpp), + prob_bits=None if config.prob_bits in (None, 0) else int(config.prob_bits), + tile_elements=tile_elements, row_rdo_iterations=config.row_rdo_iterations, + row_rdo_candidates=config.row_rdo_candidates, scale_search_iterations=config.scale_search_iterations, + scale_search_max_vectors=config.scale_search_max_vectors, + ) + return {key: value.to(device) for key, value in packed._asdict().items()} + + def validate_buffers(self, buffers, shape, dtype): + validate_packed(buffers, tuple(shape), dtype) + + def decode(self, packed: dict, *, shape: tuple[int, ...], dtype: torch.dtype, + config: LatticeRANSConfig) -> torch.Tensor: + buffers = packed_buffers(packed, LatticeBuffers) + source = buffers.layout.device + lane, lane_device = self.lane_for(buffers.layout, config.execution_backend, gate_dtype=False) + if lane_device is not None and source != lane_device: + buffers = LatticeBuffers._make(value.to(lane_device) for value in buffers) + out = lane.decode(buffers, shape=shape, dtype=dtype, threads_per_block=config.threads_per_block, + l2_prefetch=config.l2_prefetch) + if shape and tuple(out.shape) != shape: + out = out[: shape[0], : shape[1]] + return out if out.device == source else out.to(source) + + +register_scheme(LatticeRANSScheme()) diff --git a/entropack/schemes/lattice_rans/cuda.py b/entropack/schemes/lattice_rans/cuda.py new file mode 100644 index 0000000..e6acad4 --- /dev/null +++ b/entropack/schemes/lattice_rans/cuda.py @@ -0,0 +1,670 @@ +from dataclasses import dataclass, replace +from pathlib import Path + +import cupy +import numpy as np +import torch + +from ...backends.cuda import device as _device_caps +from ...backends.cuda.kernels import KernelLibrary +from ...backends.cuda.kernels import device_index as _device_index +from ...backends.cuda.kernels import ensure_dynamic_shared as _ensure_dynamic_shared +from ...backends.cuda.kernels import external_stream as _external_stream +from ...backends.cuda.kernels import pointer as _pointer +from ..tile_ans.format import NUM_STATES +from . import rans, rdo +from . import eager as _eager +from .format import ( + ALPHABET_COARSEN_MARGIN, BITS_PER_BYTE, BLOCK_SIZE, + LATTICE_DIM, META_ALPHABET, META_FREQ_OFFSET, META_N_SYMBOLS, META_SYM_MIN, MIN_SHARED_BLOCKS_PER_SM, + MODEL_LAYOUT_BYTES, NUM_COORD_STREAMS, NUM_STREAMS_FULL, SHARED_STAGING_HEADROOM, STATIC_SHARED, STREAM_META_WIDTH, + LatticeBuffers, check_row_scales_finite, decode_geometry, make_layout, report_alphabet_clamp, resolve_prob_bits, + snap_to_container, subsample_index, vector_tile_elements, +) + +_CUDA_PATH = Path(__file__).parent / "lattice_rans.cu" +_TILE_INCLUDE_PATH = Path(__file__).parent.parent / "tile_ans" +_KERNEL_NAMES = ( + "e8_quantize_fields_kernel", + "e8_quantize_fields_f32_kernel", + "e8_refit_scales_kernel", + "e8_refit_scales_f32_kernel", + "e8_minmax_fields_kernel", + "e8_histogram_kernel", + "e8_rans_encode_vector_kernel", + "e8_compact_kernel", + "e8_decode_vector_shlut8pf_kernel", + "e8_decode_vector_shlut8pf_g_kernel", + "e8_decode_vector_packed32pf_kernel", + "e8_decode_vector_packed32pf_g_kernel", + "e8_decode_vector_packed32_kernel", + "e8_decode_vector_packed32_g_kernel", + "e8_decode_vector_fused_kernel", + "e8_decode_vector_fused_g_kernel", +) +_LIBRARY = KernelLibrary( + key="lattice_rans", source=_CUDA_PATH, + defines=lambda _device, probability_bits: (f"TILE_ANS_PROB_BITS={probability_bits}",), includes=(_TILE_INCLUDE_PATH,), + kernel_names=_KERNEL_NAMES, +) +_kernel = _LIBRARY.kernel + +_MIN_RMS = _eager.MIN_RMS +_INT_MAX = 0x7FFFFFFF +_INT_MIN = -0x7FFFFFFF - 1 +_MM_CMAX = 2 * (NUM_STREAMS_FULL - 1) +_MM_LEN = _MM_CMAX + 1 +_MINMAX_TEMPLATE = np.array( + [(_INT_MAX if (t & 1) == 0 and t != _MM_CMAX else _INT_MIN) for t in range(_MM_LEN)], dtype=np.int32, +) + +_ENCODE_KERNELS = { + torch.bfloat16: ("e8_quantize_fields_kernel", "e8_refit_scales_kernel"), + torch.float32: ("e8_quantize_fields_f32_kernel", "e8_refit_scales_f32_kernel"), +} +_STORE_KINDS = { + torch.float32: 0, torch.float16: 1, torch.float8_e4m3fn: 2, torch.float8_e5m2: 3, torch.int8: 4, torch.int16: 5, + torch.int32: 6, torch.int64: 7, torch.uint8: 8, torch.uint16: 9, torch.uint32: 10, torch.uint64: 11, torch.bool: 12, +} +_SCRATCH_DTYPES = frozenset({torch.float8_e4m3fnuz, torch.float8_e5m2fnuz}) +_GENERIC_DECODE_KERNEL = { + "e8_decode_vector_shlut8pf_kernel": "e8_decode_vector_shlut8pf_g_kernel", + "e8_decode_vector_packed32pf_kernel": "e8_decode_vector_packed32pf_g_kernel", + "e8_decode_vector_packed32_kernel": "e8_decode_vector_packed32_g_kernel", + "e8_decode_vector_fused_kernel": "e8_decode_vector_fused_g_kernel", +} + + +def _grid(caps, total: int, threads: int = BLOCK_SIZE) -> int: + return caps.grid(-(-total // threads), threads) + + +@dataclass(frozen=True) +class _QuantizeRequest: + weight: torch.Tensor + rms: torch.Tensor + scale: float + prob_bits: int + rows: int + cols: int + device: torch.device + device_index: int + + @classmethod + def of(cls, weight: torch.Tensor, rms: torch.Tensor, scale: float, prob_bits: int): + rows, cols = weight.shape + return cls( + weight=weight, rms=rms, scale=scale, prob_bits=prob_bits, rows=rows, cols=cols, device=weight.device, + device_index=_device_index(weight), + ) + + +@dataclass(frozen=True) +class _QuantizeResult: + counts: list + sizes: list + sym_min: list + alphabets: list + fields: torch.Tensor + c_arr: torch.Tensor + minmax: torch.Tensor + scales: torch.Tensor | None = None + row_sse: torch.Tensor | None = None + + +def _quantize_pass(request: _QuantizeRequest) -> _QuantizeResult: + weight, rms = request.weight, request.rms + rows, cols = request.rows, request.cols + device, device_index, prob_bits = request.device, request.device_index, request.prob_bits + V = rows * (cols // LATTICE_DIM) + fields = torch.empty(V * LATTICE_DIM, dtype=torch.int32, device=device) + c_arr = torch.empty(V, dtype=torch.int32, device=device) + minmax = torch.from_numpy(_MINMAX_TEMPLATE.copy()).to(device) + torch_stream = torch.cuda.current_stream(device) + with cupy.cuda.Device(device_index), _external_stream(torch_stream): + _kernel(device_index, prob_bits, _ENCODE_KERNELS[weight.dtype][0])( + (_grid(_device_caps.caps(device_index), V),), + (BLOCK_SIZE,), + ( + _pointer(weight), _pointer(rms), np.float32(request.scale), np.int32(rows), np.int32(cols), + np.int32(cols // LATTICE_DIM), _pointer(fields), _pointer(c_arr), _pointer(minmax), + ), + ) + minmax_np = minmax.cpu().numpy() + counts, sizes, sym_min, alphabets = _counts_from_minmax( + fields, c_arr, minmax, minmax_np, V, device, device_index, prob_bits + ) + return _QuantizeResult( + counts=counts, sizes=sizes, sym_min=sym_min, alphabets=alphabets, fields=fields, c_arr=c_arr, minmax=minmax, + ) + + +def _refit_row_scales(request: _QuantizeRequest, result: _QuantizeResult): + weight, rms = request.weight, request.rms + rows, cols = request.rows, request.cols + device_index, prob_bits = request.device_index, request.prob_bits + scales = torch.empty(rows, dtype=torch.float32, device=weight.device) + row_sse = torch.empty(rows, dtype=torch.float32, device=weight.device) + torch_stream = torch.cuda.current_stream(weight.device) + with cupy.cuda.Device(device_index), _external_stream(torch_stream): + _kernel(device_index, prob_bits, _ENCODE_KERNELS[weight.dtype][1])( + (rows,), + (BLOCK_SIZE,), + ( + _pointer(weight), _pointer(result.fields), _pointer(result.c_arr), _pointer(rms), np.float32(request.scale), + _pointer(scales), _pointer(row_sse), np.int32(rows), np.int32(cols), np.int32(cols // LATTICE_DIM), + ), + ) + return scales, row_sse + + +def _summarize_fields(request: _QuantizeRequest, fields, c_arr) -> _QuantizeResult: + rows, cols = request.rows, request.cols + device, device_index, prob_bits = request.device, request.device_index, request.prob_bits + V = rows * (cols // LATTICE_DIM) + minmax = torch.from_numpy(_MINMAX_TEMPLATE.copy()).to(device) + torch_stream = torch.cuda.current_stream(device) + with cupy.cuda.Device(device_index), _external_stream(torch_stream): + _kernel(device_index, prob_bits, "e8_minmax_fields_kernel")( + (_grid(_device_caps.caps(device_index), V),), (BLOCK_SIZE,), + (_pointer(fields), _pointer(c_arr), np.int64(V), _pointer(minmax)), + ) + minmax_np = minmax.cpu().numpy() + counts, sizes, sym_min, alphabets = _counts_from_minmax( + fields, c_arr, minmax, minmax_np, V, device, device_index, prob_bits + ) + return _QuantizeResult( + counts=counts, sizes=sizes, sym_min=sym_min, alphabets=alphabets, fields=fields, c_arr=c_arr, minmax=minmax, + ) + + +def _counts_from_minmax(fields, c_arr, minmax, minmax_np, V, device, device_index, prob_bits): + alphabets = [int(minmax_np[_MM_CMAX]) + 1] + sym_min = [0] + for stream in range(NUM_COORD_STREAMS): + lo, hi = int(minmax_np[stream * 2]), int(minmax_np[stream * 2 + 1]) + live = lo != _INT_MAX + alphabets.append(hi - lo + 1 if live else 0) + sym_min.append(lo if live else 0) + + bin_off = np.zeros(NUM_STREAMS_FULL, dtype=np.int32) + cursor = 2 + for stream in range(1, NUM_STREAMS_FULL): + bin_off[stream] = cursor + cursor += alphabets[stream] + total_bins = cursor + + bins = torch.zeros(total_bins, dtype=torch.int32, device=device) + bin_off_gpu = torch.from_numpy(bin_off).to(device) + caps = _device_caps.caps(device_index) + staging_bins = (caps.shared_per_block - SHARED_STAGING_HEADROOM) // 4 + shared_bins = total_bins if total_bins <= staging_bins else 0 + torch_stream = torch.cuda.current_stream(device) + with cupy.cuda.Device(device_index), _external_stream(torch_stream): + _kernel(device_index, prob_bits, "e8_histogram_kernel")( + (_grid(caps, V),), + (BLOCK_SIZE,), + ( + _pointer(fields), _pointer(c_arr), _pointer(minmax), _pointer(bin_off_gpu), _pointer(bins), np.int64(V), + np.int32(shared_bins), + ), + shared_mem=shared_bins * 4, + ) + bins_np = bins.cpu().numpy().astype(np.int64) + n0, n1 = int(bins_np[0]), int(bins_np[1]) + sizes = [V] + [n0 if stream % 2 == 0 else n1 for stream in range(NUM_COORD_STREAMS)] + counts = [bins_np[: alphabets[0]].copy()] + for stream in range(1, NUM_STREAMS_FULL): + offset = int(bin_off[stream]) + counts.append(bins_np[offset : offset + alphabets[stream]].copy()) + return counts, sizes, sym_min, alphabets + + +def _analytic_total_bytes(counts, sizes, rows, prob_bits, tile_elements): + return rans.coded_bytes(counts, sizes, prob_bits, tile_elements) + rows * 4 + MODEL_LAYOUT_BYTES + + +def _optimize_rows( + request: _QuantizeRequest, baseline: _QuantizeResult, iterations: int, candidate_count: int, tile_elements: int, +) -> _QuantizeResult: + rows, device, prob_bits = request.rows, request.device, request.prob_bits + ratios = rdo.ratio_ladder(candidate_count) + baseline_index = ratios.index(1.0) + candidates = [] + for index, ratio in enumerate(ratios): + if index == baseline_index: + candidate = baseline + else: + scaled = replace(request, scale=request.scale * ratio) + summary = _quantize_pass(scaled) + scales, row_sse = _refit_row_scales(scaled, summary) + candidate = replace(summary, scales=scales, row_sse=row_sse) + candidates.append(rdo.compact_candidate(candidate)) + + summary, fitted_scales = rdo.optimize_rows( + candidates, rows=rows, cols=request.cols, baseline_index=baseline_index, iterations=iterations, device=device, + summarize=lambda fields, c_arr: _summarize_fields(request, fields, c_arr), + total_bytes=lambda counts, sizes: _analytic_total_bytes(counts, sizes, rows, prob_bits, tile_elements), + ) + return replace(summary, scales=fitted_scales) + + +def _bisect_scale_cuda(work, rms, prob_bits, target_bpp, tile_elements, n_iter=34, max_vectors=262144): + rows, cols = work.shape + device = work.device + vecs_per_row = cols // LATTICE_DIM + sub_rows = max(1, min(rows, max_vectors // vecs_per_row)) + index = subsample_index(rows, cols, sub_rows, device) + w_sub = work if index is None else work.index_select(0, index) + rms_sub = rms if index is None else rms.index_select(0, index) + N_sub = sub_rows * cols + + lo, hi = 0.001, 8.0 + for _ in range(n_iter): + mid = (lo * hi) ** 0.5 + summary = _quantize_pass(_QuantizeRequest.of(w_sub, rms_sub, mid, prob_bits)) + total = _analytic_total_bytes(summary.counts, summary.sizes, sub_rows, prob_bits, tile_elements) + if BITS_PER_BYTE * total / N_sub > target_bpp: + lo = mid + else: + hi = mid + return (lo * hi) ** 0.5 + + +@dataclass(frozen=True) +class _EncodeOptions: + target_bpp: float + prob_bits: int + auto_prob_bits: bool + tile_elements: int + row_rdo_iterations: int + row_rdo_candidates: int + scale_search_iterations: int + scale_search_max_vectors: int + table_size: int + + +def _resolve_options( + target_bpp, prob_bits, tile_elements, row_rdo_iterations, row_rdo_candidates, scale_search_iterations, + scale_search_max_vectors, +) -> _EncodeOptions: + resolved_prob_bits, auto_prob_bits = resolve_prob_bits(prob_bits, target_bpp) + return _EncodeOptions( + target_bpp=float(target_bpp), + prob_bits=resolved_prob_bits, + auto_prob_bits=auto_prob_bits, + tile_elements=int(tile_elements), + row_rdo_iterations=int(row_rdo_iterations), + row_rdo_candidates=int(row_rdo_candidates), + scale_search_iterations=int(scale_search_iterations), + scale_search_max_vectors=int(scale_search_max_vectors), + table_size=1 << resolved_prob_bits, + ) + + +def _resolve_scale(work, rms, options: _EncodeOptions): + """The table never shrinks below its starting precision: a coarser grid normalizes the distributions less exactly, so + shrinking costs rate, while the decode it would buy is not certain -- the decode is priced per tile and per symbol, and + the table size only decides which representation still stages in shared memory. + """ + prob_bits = options.prob_bits + if float(work.abs().amax().item()) == 0.0: + request = _QuantizeRequest.of(work, rms, 1.0, prob_bits) + return options, request, _quantize_pass(request) + clamped = False + while True: + escalate = False + scale = _bisect_scale_cuda( + work, rms, prob_bits, options.target_bpp, options.tile_elements, n_iter=options.scale_search_iterations, + max_vectors=options.scale_search_max_vectors, + ) + while True: + request = _QuantizeRequest.of(work, rms, scale, prob_bits) + summary = _quantize_pass(request) + alpha = max(summary.alphabets) if summary.alphabets else 0 + if alpha <= options.table_size: + break + if options.auto_prob_bits and prob_bits < 15: + prob_bits += 1 + options = replace(options, prob_bits=prob_bits, table_size=1 << prob_bits) + escalate = True + break + clamped = True + scale *= alpha / options.table_size * ALPHABET_COARSEN_MARGIN + if not escalate: + if clamped: + report_alphabet_clamp(scale, alpha, options.table_size) + return options, request, summary + + +def _build_codec_tables(summary: _QuantizeResult, options: _EncodeOptions, device): + """Only alphabet-sized frequencies are stored, never a full ``table_size`` LUT; the decoder rebuilds its slot->symbol + table from them, so a per-layer table stays small. + """ + n_streams = NUM_STREAMS_FULL + table_size = options.table_size + freq_parts = [] + cdf_parts = [] + freq_cursor = 0 + meta = np.zeros((n_streams, STREAM_META_WIDTH), dtype=np.int64) + for table in range(n_streams): + n_s = int(summary.sizes[table]) + alphabet = int(summary.alphabets[table]) if n_s else 0 + if alphabet > table_size: + raise ValueError( + f"E8 coordinate alphabet {alphabet} exceeds rANS table_size {table_size}; " + "raise prob_bits or coarsen the lattice scale" + ) + if alphabet: + frequency = rans.normalize_freq(summary.counts[table], table_size).astype(np.uint16) + cdf = np.zeros(alphabet, dtype=np.uint16) + if alphabet > 1: + cdf[1:] = np.cumsum(frequency.astype(np.int64))[:-1].astype(np.uint16) + freq_parts.append(frequency) + cdf_parts.append(cdf) + meta[table, META_N_SYMBOLS] = n_s + meta[table, META_SYM_MIN] = summary.sym_min[table] + meta[table, META_FREQ_OFFSET] = freq_cursor + meta[table, META_ALPHABET] = alphabet + freq_cursor += alphabet + + freq_tables_np = np.concatenate(freq_parts) if freq_parts else np.empty(0, dtype=np.uint16) + cdfs_np = np.concatenate(cdf_parts) if cdf_parts else np.empty(0, dtype=np.uint16) + return ( + torch.from_numpy(freq_tables_np).to(device), torch.from_numpy(cdfs_np).to(device), + torch.from_numpy(meta.astype(np.int32)).to(device), + ) + + +def _encode_streams( + summary: _QuantizeResult, freq_tables, cdfs_gpu, stream_meta, total_vectors: int, options: _EncodeOptions, device, + device_index: int, +): + tile_vectors = vector_tile_elements(options.tile_elements) + total_tiles = max(1, (total_vectors + tile_vectors - 1) // tile_vectors) + states = torch.empty((total_tiles, NUM_STATES), dtype=torch.uint32, device=device) + word_counts = torch.empty(total_tiles, dtype=torch.uint32, device=device) + scratch_stride = tile_vectors * 9 + scratch = torch.empty(total_tiles * scratch_stride, dtype=torch.uint16, device=device) + warps_per_block = BLOCK_SIZE // _device_caps.caps(device_index).warp_size + blocks_x = max(1, -(-total_tiles // warps_per_block)) + torch_stream = torch.cuda.current_stream(device) + with cupy.cuda.Device(device_index), _external_stream(torch_stream): + _kernel(device_index, options.prob_bits, "e8_rans_encode_vector_kernel")( + (blocks_x,), + (BLOCK_SIZE,), + ( + _pointer(summary.fields), _pointer(summary.c_arr), _pointer(freq_tables), _pointer(cdfs_gpu), + _pointer(stream_meta), _pointer(scratch), _pointer(word_counts), _pointer(states), np.int64(total_vectors), + np.int32(tile_vectors), np.int32(total_tiles), + ), + ) + counts_cp = cupy.from_dlpack(word_counts) + offsets64 = torch.empty(total_tiles + 1, dtype=torch.int64, device=device) + offsets_cp = cupy.from_dlpack(offsets64) + offsets_cp[0] = 0 + cupy.cumsum(counts_cp, dtype=cupy.int64, out=offsets_cp[1:]) + total_words = int(offsets64[-1].item()) + if total_words >= 1 << 32: + raise ValueError("lattice_rans payload exceeds uint32 offset capacity") + offsets = offsets64.to(torch.uint32) + payload = torch.empty(total_words, dtype=torch.uint16, device=device) + with cupy.cuda.Device(device_index), _external_stream(torch_stream): + _kernel(device_index, options.prob_bits, "e8_compact_kernel")( + (total_tiles,), (BLOCK_SIZE,), + (_pointer(scratch), _pointer(offsets), _pointer(payload), np.int32(scratch_stride), np.int32(total_tiles)), + ) + return payload, offsets, states + + +def encode( + weight, *, target_bpp, prob_bits, tile_elements, row_rdo_iterations, row_rdo_candidates, scale_search_iterations, + scale_search_max_vectors, +): + options = _resolve_options(target_bpp, prob_bits, tile_elements, row_rdo_iterations, + row_rdo_candidates, scale_search_iterations, scale_search_max_vectors) + + weight = weight.contiguous() + device = weight.device + device_index = _device_index(weight) + rows, cols = weight.shape + V = rows * (cols // LATTICE_DIM) + + if weight.dtype == torch.bfloat16: + work = weight + xf = weight.float() + else: + work = weight.float().contiguous() + xf = work + rms = xf.square().mean(dim=1, keepdim=True).sqrt().clamp_min(_MIN_RMS) + check_row_scales_finite(rms, weight.dtype) + # For a bf16 container this fp32 view only served the row RMS above. Releasing it before the scale search keeps two + # full-tensor copies from being alive at once. + del xf + + options, request, summary = _resolve_scale(work, rms, options) + scales, row_sse = _refit_row_scales(request, summary) + summary = replace(summary, scales=scales, row_sse=row_sse) + if options.row_rdo_iterations > 0 and float(work.abs().amax().item()) != 0.0: + summary = _optimize_rows(request, summary, options.row_rdo_iterations, + options.row_rdo_candidates, options.tile_elements) + + freq_tables, cdfs_gpu, stream_meta = _build_codec_tables(summary, options, device) + payload, offsets, states = _encode_streams( + summary, freq_tables, cdfs_gpu, stream_meta, V, options, device, device_index) + fitted_scales = summary.scales + del summary + + layout = make_layout(cols, options.prob_bits, options.tile_elements).to(device) + return LatticeBuffers( + payload=payload, offsets=offsets, states=states, stream_meta=stream_meta, freq_tables=freq_tables, + scales=fitted_scales.contiguous(), layout=layout, + ) + + +def _shared_lut_usable(shared_bytes: int | None, caps, threads: int) -> bool: + """A table that consumes most of an SM's shared memory leaves one CTA resident, and this decode hides renormalization + latency with warp count, so it would run slower reading the table from shared memory than from global. The staged table + has to leave room for ``MIN_SHARED_BLOCKS_PER_SM`` resident blocks. + """ + if shared_bytes is None: + return False + staged = shared_bytes + STATIC_SHARED + if staged > caps.shared_limit(STATIC_SHARED): + return False + return caps.blocks_per_sm(threads, staged) >= MIN_SHARED_BLOCKS_PER_SM + + +def _stream_slices(meta_np) -> list[tuple[int, int, int]]: + return [ + (stream, int(row[META_FREQ_OFFSET]), int(row[META_ALPHABET])) for stream, row in enumerate(meta_np) + if int(row[META_ALPHABET]) + ] + + +def _cdf_and_frequency(freq_np, offset: int, alphabet: int): + frequency = freq_np[offset : offset + alphabet].astype(np.int64) + cdf = np.zeros(alphabet, dtype=np.int64) + cdf[1:] = np.cumsum(frequency)[:-1] + return cdf, frequency + + +def _pack_bits(freq_np, slices, table_size: int, n_streams: int) -> np.ndarray | None: + pack_bits = np.zeros(n_streams, dtype=np.int32) + for stream, offset, alphabet in slices: + _cdf, frequency = _cdf_and_frequency(freq_np, offset, alphabet) + stored = np.where(frequency == table_size, 0, frequency) + symbol_bits = max(1, (alphabet - 1).bit_length()) + frequency_bits = max(1, int(stored.max()).bit_length()) + delta_bits = max(1, (int(frequency.max()) - 1).bit_length()) + if symbol_bits + frequency_bits + delta_bits > 32: + return None + pack_bits[stream] = symbol_bits | (frequency_bits << 8) + return pack_bits + + +def _slot_fields(freq_np, offset: int, alphabet: int, table_size: int): + cdf, frequency = _cdf_and_frequency(freq_np, offset, alphabet) + stored = np.where(frequency == table_size, 0, frequency) + return (np.repeat(np.arange(alphabet), frequency), np.repeat(stored, frequency), np.repeat(cdf, frequency)) + + +def _shared_tables(freq_np, slices, table_size: int, n_streams: int): + symbols = np.zeros(n_streams * table_size, dtype=np.uint8) + begin_frequency = np.zeros(int(freq_np.size), dtype=np.uint32) + for stream, offset, alphabet in slices: + cdf, frequency = _cdf_and_frequency(freq_np, offset, alphabet) + begin_frequency[offset : offset + alphabet] = (cdf | (frequency << 16)).astype(np.uint32) + base = stream * table_size + symbols[base : base + table_size] = np.repeat(np.arange(alphabet, dtype=np.uint8), frequency) + return symbols, begin_frequency + + +def _packed_tables(freq_np, slices, table_size: int, n_streams: int, pack_bits): + packed = np.zeros((n_streams, table_size), dtype=np.uint32) + for stream, offset, alphabet in slices: + symbol, frequency, begin = _slot_fields(freq_np, offset, alphabet, table_size) + symbol_bits = int(pack_bits[stream] & 0xFF) + frequency_bits = int((pack_bits[stream] >> 8) & 0xFF) + packed[stream] = ( + symbol.astype(np.uint32) | (frequency.astype(np.uint32) << np.uint32(symbol_bits)) + | ((np.arange(table_size, dtype=np.int64) - begin).astype(np.uint32) << np.uint32(symbol_bits + frequency_bits)) + ) + return packed + + +def _wide_tables(freq_np, slices, table_size: int, n_streams: int): + luts = np.zeros((n_streams, table_size), dtype=np.uint64) + for stream, offset, alphabet in slices: + symbol, frequency, begin = _slot_fields(freq_np, offset, alphabet, table_size) + luts[stream] = ( + (begin.astype(np.uint64) << np.uint64(32)) | (frequency.astype(np.uint64) << np.uint64(16)) + | symbol.astype(np.uint64) + ) + return luts + + +@dataclass(frozen=True) +class _DecodePlan: + representation: str + coset_frequency0: int + error: torch.Tensor + vector_tiles: int + tile_vectors: int + sym_u8: torch.Tensor | None = None + fb_lut: torch.Tensor | None = None + shared_bytes: int = 0 + packed_luts: torch.Tensor | None = None + pack_bits: torch.Tensor | None = None + decode_luts: torch.Tensor | None = None + + +def _decode_plan(layout, stream_meta, freq_tables, info, caps, threads: int) -> _DecodePlan: + device = freq_tables.device + fingerprint = (stream_meta.data_ptr(), stream_meta._version, freq_tables.data_ptr(), freq_tables._version, device, threads) + cached = getattr(layout, "_lattice_rans_decode_plan", None) + if cached is not None and cached[0] == fingerprint: + return cached[1] + + meta_np = stream_meta.detach().cpu().numpy().astype(np.int64) + freq_np = freq_tables.detach().cpu().numpy().astype(np.uint16) + prob_bits = info["prob_bits"] + table_size = 1 << prob_bits + n_streams = info["n_streams"] + slices = _stream_slices(meta_np) + alphabets = meta_np[:, META_ALPHABET] + max_alpha = int(alphabets.max()) if alphabets.size else 0 + alphabet_sum = int(alphabets.sum()) + + shared_bytes = n_streams * table_size + alphabet_sum * 4 if max_alpha < 256 else None + fields: dict = {} + if _shared_lut_usable(shared_bytes, caps, threads): + representation = "shared" + symbols, begin_frequency = _shared_tables(freq_np, slices, table_size, n_streams) + fields = { + "sym_u8": torch.from_numpy(symbols).to(device), "fb_lut": torch.from_numpy(begin_frequency).to(device), + "shared_bytes": int(symbols.nbytes + begin_frequency.nbytes), + } + else: + pack_bits = _pack_bits(freq_np, slices, table_size, n_streams) + if pack_bits is not None: + representation = "packed" + packed = _packed_tables(freq_np, slices, table_size, n_streams, pack_bits) + fields = {"packed_luts": torch.from_numpy(packed).to(device), "pack_bits": torch.from_numpy(pack_bits).to(device)} + else: + representation = "wide" + luts = _wide_tables(freq_np, slices, table_size, n_streams) + fields = {"decode_luts": torch.from_numpy(luts).to(device)} + + vectors = info["rows"] * (info["cols"] // LATTICE_DIM) + tile_vectors = vector_tile_elements(info["tile_elements"]) + plan = _DecodePlan( + representation=representation, coset_frequency0=int(freq_np[0]), error=torch.zeros(1, dtype=torch.int32, device=device), + vector_tiles=max(1, (vectors + tile_vectors - 1) // tile_vectors), tile_vectors=tile_vectors, **fields, + ) + layout._lattice_rans_decode_plan = (fingerprint, plan) + return plan + + +def _decode_launch(plan, buffers: LatticeBuffers, output, rows, cols, l2_prefetch): + vecs_per_row = cols // LATTICE_DIM + total_vectors = rows * vecs_per_row + native_bf16 = output.dtype == torch.bfloat16 + store_kind = 0 if native_bf16 else _STORE_KINDS[output.dtype] + if plan.representation == "shared": + kernel_name = "e8_decode_vector_shlut8pf_kernel" + shared = plan.shared_bytes + args = ( + _pointer(buffers.payload), _pointer(buffers.offsets), _pointer(buffers.states), _pointer(plan.sym_u8), + _pointer(plan.fb_lut), np.uint32(plan.coset_frequency0), _pointer(buffers.stream_meta), + _pointer(buffers.scales), _pointer(output), np.int32(vecs_per_row), np.int64(total_vectors), + np.int32(plan.tile_vectors), np.int32(plan.vector_tiles), np.int32(plan.fb_lut.numel()), _pointer(plan.error), + ) + elif plan.representation == "packed": + kernel_name = "e8_decode_vector_packed32pf_kernel" if l2_prefetch else "e8_decode_vector_packed32_kernel" + shared = 0 + args = ( + _pointer(buffers.payload), _pointer(buffers.offsets), _pointer(buffers.states), _pointer(plan.packed_luts), + _pointer(plan.pack_bits), np.uint32(plan.coset_frequency0), _pointer(buffers.stream_meta), + _pointer(buffers.scales), _pointer(output), np.int32(vecs_per_row), np.int64(total_vectors), + np.int32(plan.tile_vectors), np.int32(plan.vector_tiles), _pointer(plan.error), + ) + else: + kernel_name = "e8_decode_vector_fused_kernel" + shared = 0 + args = ( + _pointer(buffers.payload), _pointer(buffers.offsets), _pointer(buffers.states), _pointer(plan.decode_luts), + _pointer(buffers.stream_meta), _pointer(buffers.scales), _pointer(output), np.int32(vecs_per_row), + np.int64(total_vectors), np.int32(plan.tile_vectors), np.int32(plan.vector_tiles), _pointer(plan.error), + ) + if not native_bf16: + kernel_name = _GENERIC_DECODE_KERNEL[kernel_name] + args += (np.int32(store_kind),) + return kernel_name, shared, args + + +def decode(buffers: LatticeBuffers, *, shape, dtype, threads_per_block, l2_prefetch): + layout = buffers.layout + info = decode_geometry(buffers._asdict(), shape) + rows, cols = info["rows"], info["cols"] + prob_bits = info["prob_bits"] + + payload = buffers.payload + device_index = _device_index(payload) + caps = _device_caps.caps(device_index) + threads = _device_caps.resolve_threads(caps, threads_per_block, BLOCK_SIZE) + plan = _decode_plan(layout, buffers.stream_meta, buffers.freq_tables, info, caps, threads) + scratch = dtype in _SCRATCH_DTYPES + out_dtype = torch.float32 if scratch else dtype + output = torch.empty((rows, cols), dtype=out_dtype, device=payload.device) + torch_stream = torch.cuda.current_stream(payload.device) + + kernel_name, shared, args = _decode_launch(plan, buffers, output, rows, cols, l2_prefetch) + launch_blocks = max(1, -(-plan.vector_tiles // (threads // caps.warp_size))) + with cupy.cuda.Device(device_index), _external_stream(torch_stream): + kernel = _kernel(device_index, prob_bits, kernel_name) + _ensure_dynamic_shared(kernel, shared) + kernel((launch_blocks,), (threads,), args, shared_mem=shared) + if scratch: + output = snap_to_container(output, dtype) + return output diff --git a/entropack/schemes/lattice_rans/eager.py b/entropack/schemes/lattice_rans/eager.py new file mode 100644 index 0000000..f1cdaf8 --- /dev/null +++ b/entropack/schemes/lattice_rans/eager.py @@ -0,0 +1,405 @@ +import math +from dataclasses import dataclass + +import numpy as np +import torch + +from . import rans, rdo +from .format import ( + ALPHABET_COARSEN_MARGIN, BITS_PER_BYTE, LATTICE_DIM, + META_ALPHABET, META_FREQ_OFFSET, META_N_SYMBOLS, META_SYM_MIN, MODEL_LAYOUT_BYTES, MODEL_STREAM_META_BYTES, + NUM_COORD_FIELDS, STREAM_META_WIDTH, LatticeBuffers, check_row_scales_finite, decode_geometry, make_layout, + report_alphabet_clamp, resolve_prob_bits, snap_to_container, subsample_index, vector_tile_elements, +) + +MIN_RMS = 1.0e-12 + + +def _nearest_dn(Y: torch.Tensor) -> torch.Tensor: + f = torch.round(Y) + resid = Y - f + par = torch.remainder(f.sum(1), 2.0) + idx = resid.abs().argmax(1, keepdim=True) + sgn = torch.sign(resid.gather(1, idx)) + sgn = torch.where(sgn == 0, torch.ones_like(sgn), sgn) + onehot = torch.zeros_like(f) + onehot.scatter_(1, idx, 1.0) + return f + onehot * sgn * (par != 0).float().unsqueeze(1) + + +def nearest_e8(Y: torch.Tensor) -> torch.Tensor: + """The coset comparison runs in float64 so exact ties, common on bf16 and half-integer grids, resolve as the CUDA kernel + does rather than following fp32 summation order. + """ + c0 = _nearest_dn(Y) + c1 = _nearest_dn(Y - 0.5) + 0.5 + yd = Y.to(torch.float64) + d0 = (yd - c0.to(torch.float64)).square().sum(1, keepdim=True) + d1 = (yd - c1.to(torch.float64)).square().sum(1, keepdim=True) + # ``+ 0.0`` folds -0.0 to +0.0: torch.round of a small negative coordinate is -0.0, and the CUDA kernel rounds in integer + # coordinates so it can never emit one. + return torch.where(d0 <= d1, c0, c1) + 0.0 + + +def point_to_fields(p: torch.Tensor): + doubled = torch.round(2 * p) + c = (doubled.remainder(2.0) != 0).any(dim=1).to(torch.int64) + z = torch.round(p - 0.5 * c.unsqueeze(1)).to(torch.int64) + z0_6 = z[:, :7] + par = z0_6.sum(1).remainder(2) + m = torch.div(z[:, 7] - par, 2, rounding_mode="floor") + return c, z0_6, m + + +def fields_to_point(c: torch.Tensor, z0_6: torch.Tensor, m: torch.Tensor) -> torch.Tensor: + par = z0_6.sum(1).remainder(2) + z7 = 2 * m + par + z = torch.cat([z0_6, z7.unsqueeze(1)], dim=1) + return z.to(torch.float32) + 0.5 * c.to(torch.float32).unsqueeze(1) + + +def _coord_streams(c_np, z0_6_np, m_np): + idx = [np.flatnonzero(c_np == 0), np.flatnonzero(c_np == 1)] + fields = [z0_6_np[:, f] for f in range(NUM_COORD_FIELDS - 1)] + [m_np] + streams = [(c_np.astype(np.int64), 0)] + for f in range(NUM_COORD_FIELDS): + for k in (0, 1): + vals = fields[f][idx[k]].astype(np.int64) + if vals.size == 0: + streams.append((vals, 0)) + else: + mn = int(vals.min()) + streams.append((vals - mn, mn)) + return streams + + +def _field_matrix(z0_6_np, m_np): + return np.column_stack([z0_6_np[:, field] for field in range(NUM_COORD_FIELDS - 1)] + [m_np]) + + +def _stream_stats(streams): + counts, sizes, sym_min, alphabets = [], [], [], [] + for symbols, minimum in streams: + alphabet = int(symbols.max()) + 1 if symbols.size else 0 + counts.append(np.bincount(symbols, minlength=alphabet).astype(np.int64)) + sizes.append(int(symbols.size)) + sym_min.append(int(minimum)) + alphabets.append(alphabet) + return counts, sizes, sym_min, alphabets + + +def _rms_and_X(x: torch.Tensor): + rms = x.square().mean(dim=1, keepdim=True).sqrt().clamp_min(MIN_RMS) + X = (x / rms).reshape(-1, LATTICE_DIM).contiguous() + return rms, X + + +def _quantize_at(X, scale): + c, z0_6, m = point_to_fields(nearest_e8(X / scale)) + return c.cpu().numpy(), z0_6.cpu().numpy(), m.cpu().numpy() + + +def _analytic_bytes_from_counts(counts, sizes, rows, prob_bits, tile_elements): + """The cuda model leaves the stream metadata and the fixed slack out, which moves the scale the bisection settles on, so the + two byte models cannot be unified. + """ + coded = rans.coded_bytes(counts, sizes, prob_bits, tile_elements) + return coded + rows * 4 + MODEL_LAYOUT_BYTES + MODEL_STREAM_META_BYTES + 384 + + +def _analytic_total_bytes(c_np, z0_6_np, m_np, rows, prob_bits, tile_elements): + counts, sizes, _sym_min, _alphabets = _stream_stats(_coord_streams(c_np, z0_6_np, m_np)) + return _analytic_bytes_from_counts(counts, sizes, rows, prob_bits, tile_elements) + + +def _bisect_scale(X, rows, cols, target_bpp, prob_bits, tile_elements, n_iter=34, max_vectors=262144): + vecs_per_row = cols // LATTICE_DIM + sub_rows = max(1, min(rows, max_vectors // vecs_per_row)) + index = subsample_index(rows, cols, sub_rows, X.device) + X_sub = X if index is None else X.reshape(rows, cols).index_select(0, index).reshape(-1, LATTICE_DIM) + N_sub = sub_rows * cols + + def total_at(s: float) -> float: + p = nearest_e8(X_sub / s) + c, z0_6, m = point_to_fields(p) + return _analytic_total_bytes( + c.cpu().numpy(), z0_6.cpu().numpy(), m.cpu().numpy(), sub_rows, prob_bits, tile_elements + ) + + lo, hi = 0.001, 8.0 + for _ in range(n_iter): + mid = math.sqrt(lo * hi) + if BITS_PER_BYTE * total_at(mid) / N_sub > target_bpp: + lo = mid + else: + hi = mid + return math.sqrt(lo * hi) + + +def _max_stream_alpha(quantized) -> int: + alpha = 0 + for symbols, _minimum in _coord_streams(*quantized): + if symbols.size: + alpha = max(alpha, int(symbols.max()) + 1) + return alpha + + +def _resolve_scale(x, X, rows, cols, target_bpp, prob_bits, tile_elements, iterations, max_vectors): + """Auto prob_bits grows a bit at a time, only when the alphabet the chosen scale produces overflows the table, up to the + ceiling; that growth is what lets the real rate track ``target_bpp`` at the high end. It never shrinks below its start: + a coarser grid normalizes the distributions less exactly, so shrinking costs rate. + + Past the ceiling, growing further is self-defeating: a wider table prices the same scale cheaper, so the bisection + refines the scale and widens the alphabet again. The scale is coarsened by the actual overflow instead, and only the + alphabet is re-measured; re-bisecting would undo it. + """ + prob_bits, auto_prob_bits = resolve_prob_bits(prob_bits, target_bpp) + if float(x.abs().amax().item()) == 0.0: + return 1.0, _quantize_at(X, 1.0), prob_bits, True + + table_size = 1 << prob_bits + clamped = False + while True: + escalate = False + scale = _bisect_scale( + X, rows, cols, target_bpp, prob_bits, tile_elements, n_iter=iterations, max_vectors=max_vectors + ) + while True: + quantized = _quantize_at(X, scale) + alpha = _max_stream_alpha(quantized) + if alpha <= table_size: + break + if auto_prob_bits and prob_bits < 15: + prob_bits += 1 + table_size = 1 << prob_bits + escalate = True + break + clamped = True + scale *= alpha / table_size * ALPHABET_COARSEN_MARGIN + if not escalate: + if clamped: + report_alphabet_clamp(scale, alpha, table_size) + return scale, quantized, prob_bits, False + + +@dataclass(frozen=True) +class _Candidate: + counts: list + sizes: list + sym_min: list + alphabets: list + fields: torch.Tensor + c_arr: torch.Tensor + scales: torch.Tensor | None = None + row_sse: torch.Tensor | None = None + + +def _refit_row_scales(x, rms, quantized, scale, device): + """Fitting the scale to the points the quantizer actually chose lowers distortion at no rate cost, so it runs on every + encode, not only under the row-RDO pass. + """ + rows, cols = x.shape + c_np, z0_6_np, m_np = quantized + c = torch.from_numpy(c_np).to(device=device, dtype=torch.int64) + z0_6 = torch.from_numpy(z0_6_np).to(device=device, dtype=torch.int64) + m = torch.from_numpy(m_np).to(device=device, dtype=torch.int64) + levels = fields_to_point(c, z0_6, m).reshape(rows, cols) + numerator = (x * levels).sum(1) + denominator = levels.square().sum(1) + scales = torch.where(denominator > 0, (numerator / denominator).clamp_min(MIN_RMS), scale * rms.squeeze(1)) + energy = x.square().sum(1) + row_sse = torch.where(denominator > 0, (energy - numerator * numerator / denominator).clamp_min(0.0), energy) + return scales, row_sse + + +def _candidate_of(quantized, scales, row_sse, device): + c_np, z_np, m_np = quantized + counts, sizes, sym_min, alphabets = _stream_stats(_coord_streams(c_np, z_np, m_np)) + return _Candidate( + counts, sizes, sym_min, alphabets, torch.from_numpy(_field_matrix(z_np, m_np).reshape(-1)).to(device), + torch.from_numpy(c_np).to(device), scales, row_sse, + ) + + +def _candidate_at(x, rms, X, scale, device): + quantized = _quantize_at(X, scale) + scales, row_sse = _refit_row_scales(x, rms, quantized, scale, device) + return _candidate_of(quantized, scales, row_sse, device) + + +def _summarize_fields(fields, c_arr): + field_matrix = fields.cpu().numpy().reshape(-1, LATTICE_DIM) + c_np = c_arr.cpu().numpy() + counts, sizes, sym_min, alphabets = _stream_stats( + _coord_streams(c_np, field_matrix[:, : NUM_COORD_FIELDS - 1], field_matrix[:, NUM_COORD_FIELDS - 1]) + ) + return _Candidate(counts, sizes, sym_min, alphabets, fields, c_arr) + + +def _optimize_rows(x, rms, X, baseline, scale, prob_bits, tile_elements, iterations, candidate_count): + rows, cols = x.shape + device = x.device + ratios = rdo.ratio_ladder(candidate_count) + baseline_index = ratios.index(1.0) + candidates = [] + for index, ratio in enumerate(ratios): + candidate = baseline if index == baseline_index else _candidate_at(x, rms, X, scale * ratio, device) + candidates.append(rdo.compact_candidate(candidate)) + summary, scales = rdo.optimize_rows( + candidates, rows=rows, cols=cols, baseline_index=baseline_index, iterations=iterations, device=device, + summarize=_summarize_fields, + total_bytes=lambda counts, sizes: _analytic_bytes_from_counts(counts, sizes, rows, prob_bits, tile_elements), + ) + chosen = summary.fields.cpu().numpy().reshape(-1, LATTICE_DIM) + return summary.c_arr.cpu().numpy(), chosen[:, : NUM_COORD_FIELDS - 1], chosen[:, NUM_COORD_FIELDS - 1], scales + + +def _encode_streams(c_np, z_np, m_np, prob_bits, tile_elements): + streams = _coord_streams(c_np, z_np, m_np) + n_streams = len(streams) + + table_size = 1 << prob_bits + meta = np.zeros((n_streams, STREAM_META_WIDTH), dtype=np.int64) + frequencies = [] + freq_parts = [] + freq_cursor = 0 + for table, (symbols, sym_min) in enumerate(streams): + n = symbols.size + if n == 0: + frequency = np.empty(0, dtype=np.uint16) + alphabet = 0 + else: + alphabet = int(symbols.max()) + 1 + counts = np.bincount(symbols, minlength=alphabet).astype(np.int64) + frequency = rans.normalize_freq(counts, table_size).astype(np.uint16) + meta[table, META_N_SYMBOLS] = n + meta[table, META_SYM_MIN] = sym_min + meta[table, META_FREQ_OFFSET] = freq_cursor + meta[table, META_ALPHABET] = alphabet + frequencies.append(frequency) + if alphabet: + freq_parts.append(frequency) + freq_cursor += alphabet + + field_symbols = _field_matrix(z_np, m_np).astype(np.int64) + for field in range(NUM_COORD_FIELDS): + table0 = 1 + 2 * field + table1 = table0 + 1 + field_symbols[c_np == 0, field] -= int(meta[table0, META_SYM_MIN]) + field_symbols[c_np == 1, field] -= int(meta[table1, META_SYM_MIN]) + tile_vectors = vector_tile_elements(tile_elements) + payload, states, offsets = rans.encode_vector_stream(c_np, field_symbols, frequencies, prob_bits, tile_vectors) + freq_tables = np.concatenate(freq_parts) if freq_parts else np.empty(0, dtype=np.uint16) + return payload, states, offsets, meta, freq_tables + + +def _encode_impl( + weight, target_bpp, prob_bits, tile_elements, row_rdo_iterations, row_rdo_candidates, scale_search_iterations, + scale_search_max_vectors, +): + if weight.shape[1] % LATTICE_DIM != 0: + raise ValueError( + "lattice_rans quantizes whole E8 vectors, so the column count must be a multiple of " + f"{LATTICE_DIM}; got cols={weight.shape[1]}" + ) + + device = weight.device + x = weight.float().contiguous() + rows, cols = x.shape + + rms, X = _rms_and_X(x) + check_row_scales_finite(rms, weight.dtype) + + s, quantized, prob_bits, all_zero = _resolve_scale( + x, X, rows, cols, target_bpp, prob_bits, tile_elements, scale_search_iterations, scale_search_max_vectors, + ) + c_np, z_np, m_np = quantized + + scales, row_sse = _refit_row_scales(x, rms, quantized, s, device) + if row_rdo_iterations > 0 and not all_zero: + baseline = _candidate_of(quantized, scales, row_sse, device) + c_np, z_np, m_np, scales = _optimize_rows( + x, rms, X, baseline, s, prob_bits, tile_elements, row_rdo_iterations, row_rdo_candidates + ) + + payload, states, offsets, meta, freq_tables = _encode_streams(c_np, z_np, m_np, prob_bits, tile_elements) + + layout = make_layout(cols, prob_bits, tile_elements) + return LatticeBuffers( + payload=torch.from_numpy(payload.astype(np.uint16)).to(device), + offsets=torch.from_numpy(offsets.astype(np.uint32)).to(device), + states=torch.from_numpy(states.astype(np.uint32)).to(device), + stream_meta=torch.from_numpy(meta.astype(np.int32)).to(device), + freq_tables=torch.from_numpy(freq_tables.astype(np.uint16)).to(device), scales=scales.contiguous(), + layout=layout.to(device), + ) + + +def _host_views(buffers: LatticeBuffers): + return ( + buffers.payload.detach().cpu().numpy().astype(np.uint16), + buffers.offsets.detach().cpu().numpy().astype(np.int64), + buffers.states.detach().cpu().numpy().astype(np.uint32), + buffers.stream_meta.detach().cpu().numpy().astype(np.int64), + buffers.freq_tables.detach().cpu().numpy().astype(np.uint16), + buffers.scales.detach().cpu(), + ) + + +def _stream_frequencies(freq_tables: np.ndarray, meta: np.ndarray) -> list: + return [ + freq_tables[int(row[META_FREQ_OFFSET]) : int(row[META_FREQ_OFFSET] + row[META_ALPHABET])] for row in meta + ] + + +def _restore_field_minima(c_np: np.ndarray, shifted_fields: np.ndarray, meta: np.ndarray) -> np.ndarray: + field_values = shifted_fields.copy() + for field in range(NUM_COORD_FIELDS): + table0 = 1 + 2 * field + table1 = table0 + 1 + field_values[:, field] += np.where(c_np == 0, int(meta[table0, META_SYM_MIN]), int(meta[table1, META_SYM_MIN])) + return field_values + + +def _reconstruct(c_np: np.ndarray, field_values: np.ndarray, scales: torch.Tensor, rows: int, cols: int): + c = torch.from_numpy(c_np).to(torch.int64) + z0_6 = torch.from_numpy(field_values[:, :7]).to(torch.int64) + m = torch.from_numpy(field_values[:, 7]).to(torch.int64) + points = fields_to_point(c, z0_6, m) + scale_vec = scales.repeat_interleave(cols // LATTICE_DIM) + return (points * scale_vec.unsqueeze(1)).reshape(rows, cols) + + +def _decode_impl(buffers: LatticeBuffers, shape, dtype): + info = decode_geometry(buffers._asdict(), shape) + rows, cols = info["rows"], info["cols"] + prob_bits = info["prob_bits"] + tile_vectors = vector_tile_elements(info["tile_elements"]) + payload, offsets, states, meta, freq_tables, scales = _host_views(buffers) + + frequencies = _stream_frequencies(freq_tables, meta) + V = rows * (cols // LATTICE_DIM) + c_np, shifted_fields = rans.decode_vector_stream(payload, offsets, states, frequencies, prob_bits, tile_vectors, V) + + field_values = _restore_field_minima(c_np, shifted_fields, meta) + xhat = _reconstruct(c_np, field_values, scales, rows, cols) + + result = snap_to_container(xhat, dtype) + if buffers.layout.device.type != "cpu": + result = result.to(buffers.layout.device) + return result + + +def encode( + weight, *, target_bpp, prob_bits, tile_elements, row_rdo_iterations, row_rdo_candidates, scale_search_iterations, + scale_search_max_vectors, +): + pb = None if prob_bits in (None, 0) else int(prob_bits) + return _encode_impl( + weight, float(target_bpp), pb, int(tile_elements), int(row_rdo_iterations), int(row_rdo_candidates), + int(scale_search_iterations), int(scale_search_max_vectors), + ) + + +def decode(buffers: LatticeBuffers, *, shape, dtype, threads_per_block, l2_prefetch): + return _decode_impl(buffers, shape, dtype) diff --git a/entropack/schemes/lattice_rans/format.py b/entropack/schemes/lattice_rans/format.py new file mode 100644 index 0000000..2e21e90 --- /dev/null +++ b/entropack/schemes/lattice_rans/format.py @@ -0,0 +1,309 @@ +import logging +from dataclasses import dataclass +from typing import Annotated, NamedTuple + +import torch + +from ..config import CompressionConfig, OneOf, Range +from ..base import buffers_fingerprint, cached_parse +from ..tile_ans.format import NUM_STATES + +logger = logging.getLogger("entropack.lattice_rans") + +BLOCK_SIZE = 256 +SHARED_STAGING_HEADROOM = 8 * 1024 +STATIC_SHARED = 256 +MIN_SHARED_BLOCKS_PER_SM = 2 + +SUPPORTED_PROB_BITS = (9, 10, 11, 12, 13, 14, 15) +#: The margin only saves a re-measure and cannot change the outcome, because every step is re-checked against the real +#: histogram. Both lanes read this one definition so they settle on the same scale. +ALPHABET_COARSEN_MARGIN = 1.05 +SUBSAMPLE_SEED_MULTIPLIER = 1000003 + + +@dataclass +class LatticeRANSConfig(CompressionConfig): + """Target-rate compression settings for finite, two-dimensional tensors. + + The encoder selects a quantization scale using sampled storage estimates, fits row + reconstruction scales, and optionally applies per-row rate–distortion refinement. + Decoding restores the input shape and dtype. Tile size and probability precision are + stored with the compressed representation.""" + + #: Requested bits per input element, from 1 to 11 including non-integer values. Read actual_bpp for the stored rate. + target_bpp: Annotated[float, Range(1.0, 11.0)] = 4.0 + #: Probability-table precision. None or zero selects automatically. + prob_bits: Annotated[int | None, OneOf(SUPPORTED_PROB_BITS, silent=(0,))] = None + #: Elements per tile. Larger tiles reduce per-tile metadata and the number of independent decode tasks. None selects by target rate. + tile_elements: Annotated[int | None, Range(1, None)] = None + #: Per-row rate–distortion allocation sweeps using estimated coding costs. Zero disables this refinement. + row_rdo_iterations: Annotated[int, Range(0, 8)] = 0 + #: Number of candidate quantization scales per row for rate–distortion refinement. + row_rdo_candidates: Annotated[int, Range(1, None)] = 5 + #: Number of bisection steps in quantization-scale selection. + scale_search_iterations: Annotated[int, Range(1, None)] = 12 + #: Sampling budget for scale search, measured in eight-value vectors. + scale_search_max_vectors: Annotated[int, Range(1, None)] = 262144 + #: GPU decode block width. None selects a device-dependent value. + threads_per_block: Annotated[int | None, Range(1, None)] = None + #: Prefetch encoded payload into the GPU L2 cache during decoding. + l2_prefetch: bool = True + + +LATTICE_DIM = 8 +NUM_COORD_FIELDS = 8 +NUM_COORD_STREAMS = 2 * NUM_COORD_FIELDS +NUM_STREAMS_FULL = 1 + NUM_COORD_STREAMS + +SUPPORTED_DTYPES = ( + torch.float32, torch.float16, torch.bfloat16, + torch.float8_e4m3fn, torch.float8_e4m3fnuz, torch.float8_e5m2, torch.float8_e5m2fnuz, + torch.int64, torch.int32, torch.int16, torch.int8, torch.uint64, torch.uint32, torch.uint16, torch.uint8, torch.bool, +) +_INTEGER_DTYPES = frozenset({ + torch.int64, torch.int32, torch.int16, torch.int8, torch.uint64, torch.uint32, torch.uint16, torch.uint8, +}) +_EXPECTED_DTYPES = { + "payload": torch.uint16, "offsets": torch.uint32, "states": torch.uint32, "stream_meta": torch.int32, + "freq_tables": torch.uint16, "scales": torch.float32, "layout": torch.int64, +} + +STREAM_META_WIDTH = 4 +META_N_SYMBOLS = 0 +META_SYM_MIN = 1 +META_FREQ_OFFSET = 2 +META_ALPHABET = 3 + +LAYOUT_LEN = 3 + +BITS_PER_BYTE = 8 +#: Not ``LAYOUT_LEN * 8``: tying it to the format would let a layout change move the scale the bisection converges on, so +#: encoded bytes would shift for reasons unrelated to the rate. +MODEL_LAYOUT_BYTES = 64 +#: Pinned the same way and for the same reason, so narrowing ``stream_meta`` changed no encoded byte. +MODEL_STREAM_META_BYTES = 408 + + +class LatticeBuffers(NamedTuple): + payload: torch.Tensor + offsets: torch.Tensor + states: torch.Tensor + stream_meta: torch.Tensor + freq_tables: torch.Tensor + scales: torch.Tensor + layout: torch.Tensor + + +PACKED_KEYS = LatticeBuffers._fields + + +def recommended_tile_elements(target_bpp: float) -> int: + if target_bpp <= 2.0: + return 32768 + if target_bpp <= 4.0: + return 16384 + if target_bpp <= 7.0: + return 8192 + return 4096 + + +def resolve_prob_bits(prob_bits: int | None, target_bpp: float) -> tuple[int, bool]: + if prob_bits in (None, 0): + return (11 if float(target_bpp) <= 7.0 else 12), True + return int(prob_bits), False + + +def report_alphabet_clamp(scale: float, alphabet: int, table_size: int) -> None: + logger.debug( + "lattice_rans coarsened the lattice scale to %.6g so a coordinate alphabet of %d fits the " + "%d-entry rANS table; actual_bpp for this tensor will fall below target_bpp", scale, alphabet, table_size, + ) + + +def subsample_index(rows: int, cols: int, sub_rows: int, device: torch.device) -> torch.Tensor | None: + """A seeded permutation rather than a prefix or a fixed stride: neither is unbiased against the row order a weight happens + to come in. + """ + if sub_rows >= rows: + return None + generator = torch.Generator().manual_seed(rows * SUBSAMPLE_SEED_MULTIPLIER + cols) + return torch.randperm(rows, generator=generator)[:sub_rows].to(device) + + +def _integer_high_bound(dtype: torch.dtype, like: torch.Tensor) -> float: + edge = torch.tensor(float(torch.iinfo(dtype).max) + 1.0, dtype=like.dtype) + below = torch.nextafter(edge, torch.full_like(edge, float("-inf"))) + return float(below) + + +def check_row_scales_finite(rms: torch.Tensor, dtype: torch.dtype) -> None: + if not bool(torch.isfinite(rms).all()): + raise ValueError( + f"lattice_rans normalizes rows in fp32, and the row RMS of this {dtype} tensor overflowed. " + "The limit is sum(weight**2) per row below ~3.4e38, i.e. |values| below " + "sqrt(3.4e38 / cols) -- about 3e17 for a 4k-wide row, 1.6e18 for a 128-wide one. " + "Rescale the source, or use tile_ans, which codes storage bytes and has no numeric " "range limit." + ) + + +def snap_to_container(values: torch.Tensor, dtype: torch.dtype) -> torch.Tensor: + """Saturation is required: torch's float->``float8_e4m3fn`` cast yields NaN above the format's largest finite value and + float->``float16`` yields inf above its own, and the lattice overshoots the source range regularly. A NaN in a decoded + weight destroys inference. + + Integers round half-to-even via ``torch.round``; the CUDA lane matches that with ``rintf``, since ``llroundf`` rounds + half away from zero and would disagree on exact ties. + """ + if dtype == torch.float32: + return values + if dtype == torch.bool: + return values.round() != 0 + if dtype in _INTEGER_DTYPES: + return values.round().clamp(float(torch.iinfo(dtype).min), _integer_high_bound(dtype, values)).to(dtype) + limit = float(torch.finfo(dtype).max) + return values.clamp(-limit, limit).to(dtype) + + +def vector_tile_elements(symbol_tile_elements: int) -> int: + return max(32, symbol_tile_elements // 9) + + +def make_layout(cols: int, prob_bits: int, tile_elements: int) -> torch.Tensor: + return torch.tensor([cols, prob_bits, tile_elements], dtype=torch.int64) + + +def parse_layout(layout: torch.Tensor) -> dict: + if layout.dtype != torch.int64 or layout.ndim != 1 or layout.numel() != LAYOUT_LEN: + raise ValueError(f"lattice_rans layout must be int64[{LAYOUT_LEN}]") + cols, prob_bits, tile_elements = (int(x) for x in layout.detach().cpu().tolist()) + if cols <= 0 or cols % LATTICE_DIM != 0: + raise ValueError("lattice_rans vector-plane format requires positive cols divisible by 8") + if prob_bits not in SUPPORTED_PROB_BITS: + raise ValueError(f"lattice_rans prob_bits {prob_bits} unsupported") + if tile_elements <= 0: + raise ValueError("lattice_rans tile_elements must be positive") + return {"cols": cols, "prob_bits": prob_bits, "tile_elements": tile_elements} + + +def _check_container(buffers: dict, shape: tuple, dtype: torch.dtype) -> None: + missing = [k for k in PACKED_KEYS if k not in buffers] + if missing: + raise ValueError(f"lattice_rans packed data is missing buffers: {missing}") + if set(buffers) != set(PACKED_KEYS): + extra = sorted(set(buffers) - set(PACKED_KEYS)) + raise ValueError(f"lattice_rans packed data has unexpected buffers: {extra}") + if not all(isinstance(buffers[k], torch.Tensor) for k in PACKED_KEYS): + raise TypeError("lattice_rans packed buffers must be torch.Tensor values") + if dtype not in SUPPORTED_DTYPES: + raise ValueError(f"lattice_rans does not support output dtype {dtype}") + if len(shape) != 2 or any(not isinstance(d, int) or d <= 0 for d in shape): + raise ValueError(f"lattice_rans shape must be non-empty 2D, got {shape}") + + +def _check_buffers(buffers: dict) -> None: + devices = {buffers[k].device for k in PACKED_KEYS} + if len(devices) != 1: + raise ValueError(f"lattice_rans packed buffers must share one device, got {devices}") + for k in PACKED_KEYS: + if not buffers[k].is_contiguous(): + raise ValueError(f"lattice_rans buffer '{k}' must be contiguous") + for k, dt in _EXPECTED_DTYPES.items(): + if buffers[k].dtype != dt: + raise ValueError(f"lattice_rans buffer '{k}' must be {dt}, got {buffers[k].dtype}") + + +def _check_scales(buffers: dict, info: dict) -> None: + scales = buffers["scales"] + if scales.ndim != 1 or scales.numel() != info["rows"]: + raise ValueError(f"lattice_rans scales must be 1D with one entry per row, got {tuple(scales.shape)}") + if not bool((torch.isfinite(scales) & (scales > 0)).all().item()): + raise ValueError("lattice_rans scales must be finite and positive") + + +def _check_tile_geometry(buffers: dict, info: dict) -> None: + n_streams, total_tiles = info["n_streams"], info["total_tiles"] + meta = buffers["stream_meta"] + if meta.ndim != 2 or tuple(meta.shape) != (n_streams, STREAM_META_WIDTH): + raise ValueError(f"lattice_rans stream_meta must be ({n_streams},{STREAM_META_WIDTH}), got {tuple(meta.shape)}") + if tuple(buffers["states"].shape) != (total_tiles, NUM_STATES): + raise ValueError("lattice_rans states shape does not match total_tiles") + if buffers["offsets"].numel() != total_tiles + 1: + raise ValueError("lattice_rans offsets length must be total_tiles+1") + offsets = buffers["offsets"].to(torch.int64) + if int(offsets[0].item()) != 0 or int(offsets[-1].item()) != buffers["payload"].numel(): + raise ValueError("lattice_rans payload offsets endpoints are invalid") + if total_tiles > 0 and bool((offsets[1:] < offsets[:-1]).any().item()): + raise ValueError("lattice_rans payload offsets must be monotone") + + +def _check_stream_metadata(buffers: dict, info: dict) -> None: + n_streams = info["n_streams"] + rows, cols = info["rows"], info["cols"] + table_size = 1 << info["prob_bits"] + meta_cpu = buffers["stream_meta"].detach().cpu().to(torch.int64) + freq_cpu = buffers["freq_tables"].detach().cpu().to(torch.int64) + freq_total = freq_cpu.numel() + running_freq = 0 + symbol_counts = [] + for table in range(n_streams): + row = meta_cpu[table] + n_sym = int(row[META_N_SYMBOLS]) + freq_off = int(row[META_FREQ_OFFSET]) + alphabet = int(row[META_ALPHABET]) + if n_sym < 0 or alphabet < 0 or freq_off < 0: + raise ValueError("lattice_rans table metadata has a negative field") + if freq_off != running_freq or freq_off + alphabet > freq_total: + raise ValueError("lattice_rans frequency-table ranges must be contiguous") + if (n_sym == 0) != (alphabet == 0): + raise ValueError("lattice_rans empty table metadata is inconsistent") + if alphabet: + if alphabet > table_size: + raise ValueError("lattice_rans table alphabet exceeds probability precision") + if int(freq_cpu[freq_off : freq_off + alphabet].sum().item()) != table_size: + raise ValueError("lattice_rans normalized frequencies have an invalid sum") + running_freq += alphabet + symbol_counts.append(n_sym) + if running_freq != freq_total: + raise ValueError("lattice_rans frequency table has trailing entries") + vectors = rows * (cols // LATTICE_DIM) + if symbol_counts[0] != vectors or int(meta_cpu[0, META_SYM_MIN]) != 0: + raise ValueError("lattice_rans coset table metadata is invalid") + n0, n1 = symbol_counts[1], symbol_counts[2] + if n0 + n1 != vectors: + raise ValueError("lattice_rans conditioned table counts do not cover all vectors") + for field in range(NUM_COORD_FIELDS): + if symbol_counts[1 + 2 * field] != n0 or symbol_counts[2 + 2 * field] != n1: + raise ValueError("lattice_rans conditioned table counts disagree across fields") + + +def decode_geometry(buffers: dict, shape: tuple) -> dict: + stored = cached_parse(buffers["layout"], parse_layout, "_lattice_rans_layout") + rows = int(shape[0]) + tile_vectors = vector_tile_elements(stored["tile_elements"]) + vectors = rows * (stored["cols"] // LATTICE_DIM) + return { + **stored, "rows": rows, "n_streams": NUM_STREAMS_FULL, + "total_tiles": (vectors + tile_vectors - 1) // tile_vectors, + } + + +def validate_packed(buffers: dict, shape: tuple, dtype: torch.dtype) -> dict: + _check_container(buffers, shape, dtype) + + cached = getattr(buffers["layout"], "_lattice_rans_validation_cache", None) + fp = buffers_fingerprint(buffers, shape, dtype) + if cached is not None and cached[0] == fp: + return cached[1] + + _check_buffers(buffers) + info = decode_geometry(buffers, shape) + rows, cols = tuple(shape) + if not 0 <= info["cols"] - cols < LATTICE_DIM: + raise ValueError(f"lattice_rans layout columns {info['cols']} do not match {(rows, cols)}") + _check_scales(buffers, info) + _check_tile_geometry(buffers, info) + _check_stream_metadata(buffers, info) + + buffers["layout"]._lattice_rans_validation_cache = (fp, info) + return info diff --git a/entropack/schemes/lattice_rans/lattice_rans.cu b/entropack/schemes/lattice_rans/lattice_rans.cu new file mode 100644 index 0000000..8f6d96d --- /dev/null +++ b/entropack/schemes/lattice_rans/lattice_rans.cu @@ -0,0 +1,1140 @@ +// EntroPack lattice_rans CUDA codec: E8-lattice vector quantization + coset-conditioned rANS. +// +// Encode kernels: +// e8_quantize_fields_kernel/_f32 nearest-E8 (Conway-Sloane: the two cosets D8 and D8+g, each reduced by a parity fix on +// the max-residual coordinate) fused with the point->fields split (coset c, parity-reduced +// coordinates z0..z6, m) and a block-reduced per-stream min/max pass. Bit-exact with +// eager.nearest_e8 / point_to_fields: rintf == torch.round (half-to-even), the squared +// distances use torch's sum(dim=1) tree order, __fmul_rn/__fadd_rn block FMA contraction, +// argmax keeps the lowest index on ties, and the coset tie rule is d0 <= d1. The two +// variants differ only in how the weight is read: bf16 native, fp32 for every other +// container. +// e8_refit_scales_kernel/_f32 least-squares refit of each row scale against the quantized points. +// e8_minmax_fields_kernel per-stream symbol min/max; sizes the alphabets. +// e8_histogram_kernel exact per-stream symbol counts. Bins are staged in shared memory when they fit the +// device budget, and fall back to global atomics otherwise. +// e8_rans_encode_vector_kernel tiled 32-state interleaved rANS encode, one warp per tile, bit-exact with +// tile_ans/encode_cpu._encode_stream: the state machine of device.cuh, 16-bit words, +// ballot-coalesced renormalization emission. +// e8_compact_kernel gathers the per-tile scratch words into one payload. +// +// Decode kernels: one warp decodes one tile and reconstructs straight into the container, with no symbol scratch and no +// prefix sum. The variants differ only in how the slot->symbol table is held, and all produce identical output, so the host +// picks one from the stored alphabet widths and the queried device limits: +// e8_decode_vector_shlut8pf_* slot->symbol (uint8) and begin|freq tables staged in shared memory. Used when every +// alphabet is below 256 and the staged table still leaves the SM its resident-block +// target. +// e8_decode_vector_packed32pf_* symbol|freq|delta packed into 32 bits in global memory, so no shared memory is needed +// and occupancy is unrestricted, plus an L2 prefetch of the descending renorm stream. +// e8_decode_vector_packed32_* the same without the prefetch, reached by decoding with l2_prefetch=False. +// e8_decode_vector_fused_* 64-bit table in global memory; the fallback for alphabets too wide to pack into 32 bits. +// A "_g" suffix marks the generic-container twin, which takes a store kind and writes any supported container instead of +// bf16. + +#include +#include +#include +#include +#include "device.cuh" + +namespace { + +constexpr int kCMax = 32; +constexpr int kMinMaxLen = kCMax + 1; +constexpr int kIntMax = 0x7FFFFFFF; +constexpr int kIntMin = -0x7FFFFFFF - 1; + +constexpr int kMetaStride = 4; +constexpr int kMetaSymMin = 1; +constexpr int kMetaFreqOff = 2; + +__device__ __forceinline__ void set_error(int* error, int code) { + atomicCAS(error, 0, code); +} + +__device__ __forceinline__ int div_i64_i32(int64_t a, int b) { + return static_cast(a / b); +} + +__device__ __forceinline__ void nearest_dn_int( + const float* __restrict__ y, int* __restrict__ out) { + float f[8]; + float r[8]; + int parity = 0; +#pragma unroll + for (int j = 0; j < 8; ++j) { + f[j] = rintf(y[j]); // torch.round == round-half-to-even == rintf + r[j] = __fsub_rn(y[j], f[j]); + parity += static_cast(f[j]); // exact: f is an integer float within int32 range + } + int idx = 0; + float best = -1.0f; +#pragma unroll + for (int j = 0; j < 8; ++j) { + const float a = fabsf(r[j]); + if (a > best) { // strict > keeps the lowest index on ties (torch argmax) + best = a; + idx = j; + } + } +#pragma unroll + for (int j = 0; j < 8; ++j) { + out[j] = static_cast(f[j]); + } + if ((parity & 1) != 0) { + out[idx] += (r[idx] < 0.0f) ? -1 : 1; // torch.sign(0) is replaced by +1 in eager + } +} + +// Accumulated in float64, in torch's sum(dim=1) tree order, so exact ties resolve identically to eager.nearest_e8: +// the parenthesization is the point and must not be flattened. +__device__ __forceinline__ double dist2_f64( + const float* __restrict__ y, const int* __restrict__ z, float offset) { + double s[8]; +#pragma unroll + for (int j = 0; j < 8; ++j) { + const double c = static_cast(z[j]) + static_cast(offset); + const double r = static_cast(y[j]) - c; + s[j] = r * r; + } + return ((s[0] + s[4]) + (s[2] + s[6])) + ((s[1] + s[5]) + (s[3] + s[7])); +} + +__device__ __forceinline__ void block_minmax_init(int* shared) { + for (int t = threadIdx.x; t < kMinMaxLen; t += blockDim.x) { + shared[t] = ((t & 1) == 0 && t != kCMax) ? kIntMax : kIntMin; + } + __syncthreads(); +} + +__device__ __forceinline__ void block_minmax_flush(int* shared, int* __restrict__ minmax) { + __syncthreads(); + for (int t = threadIdx.x; t < kMinMaxLen; t += blockDim.x) { + const int v = shared[t]; + if ((t & 1) == 0 && t != kCMax) { + if (v != kIntMax) atomicMin(&minmax[t], v); + } else { + if (v != kIntMin) atomicMax(&minmax[t], v); + } + } +} + +} // namespace + +__device__ __forceinline__ void e8_load8(const __nv_bfloat16* w, float* xf) { + const uint4 packed = *reinterpret_cast(w); + const unsigned words[4] = {packed.x, packed.y, packed.z, packed.w}; +#pragma unroll + for (int j = 0; j < 8; ++j) { + const unsigned bits = (words[j >> 1] >> ((j & 1) * 16)) & 0xFFFFu; + xf[j] = __bfloat162float(*reinterpret_cast(&bits)); + } +} + +__device__ __forceinline__ void e8_load8(const float* w, float* xf) { + const float4 lo = *reinterpret_cast(w); + const float4 hi = *reinterpret_cast(w + 4); + xf[0] = lo.x; xf[1] = lo.y; xf[2] = lo.z; xf[3] = lo.w; + xf[4] = hi.x; xf[5] = hi.y; xf[6] = hi.z; xf[7] = hi.w; +} + +__device__ __forceinline__ float e8_load_scalar(const __nv_bfloat16* p) { + return __bfloat162float(*p); +} + +__device__ __forceinline__ float e8_load_scalar(const float* p) { + return *p; +} + +template +__device__ __forceinline__ void e8_quantize_fields_body( + const InT* __restrict__ weight, + const float* __restrict__ rms, + float s, + int rows, + int cols, + int vecs_per_row, + int* __restrict__ fields, // [V,8] int32: z0..z6, m + int* __restrict__ c_arr, // [V] int32 + int* __restrict__ minmax) { // kMinMaxLen int32, pre-initialized to kIntMax/kIntMin + __shared__ int sm[kMinMaxLen]; + block_minmax_init(sm); + + const int64_t V = static_cast(rows) * vecs_per_row; + const int64_t stride = static_cast(gridDim.x) * blockDim.x; + + for (int64_t idx = static_cast(blockIdx.x) * blockDim.x + threadIdx.x; + idx < V; idx += stride) { + const int row = div_i64_i32(idx, vecs_per_row); + const int col8 = static_cast(idx - static_cast(row) * vecs_per_row); + const float rmsv = rms[row]; + const InT* w = weight + static_cast(row) * cols + col8 * 8; + float xf[8]; + e8_load8(w, xf); + float y[8]; +#pragma unroll + for (int j = 0; j < 8; ++j) { + y[j] = __fdiv_rn(__fdiv_rn(xf[j], rmsv), s); + } + + int z0[8]; + int z1[8]; + nearest_dn_int(y, z0); + float ym[8]; +#pragma unroll + for (int j = 0; j < 8; ++j) { + ym[j] = __fsub_rn(y[j], 0.5f); + } + nearest_dn_int(ym, z1); + const double d0 = dist2_f64(y, z0, 0.0f); + const double d1 = dist2_f64(y, z1, 0.5f); + const bool pick0 = d0 <= d1; // tie picks the D8 coset, exactly as eager + const int* z = pick0 ? z0 : z1; + const int c = pick0 ? 0 : 1; + + int zv[8]; + int par = 0; +#pragma unroll + for (int j = 0; j < 7; ++j) { + zv[j] = z[j]; + par += zv[j]; + } + par &= 1; // two's-complement &1 == torch.remainder(., 2) + zv[7] = (z[7] - par) >> 1; // m = floor((z7 - par)/2) + + int* fout = fields + idx * 8; +#pragma unroll + for (int j = 0; j < 8; ++j) { + fout[j] = zv[j]; + } + c_arr[idx] = c; + + atomicMax(&sm[kCMax], c); +#pragma unroll + for (int j = 0; j < 8; ++j) { + const int st = j * 2 + c; // coord stream index (field f, coset k) + atomicMin(&sm[st * 2], zv[j]); + atomicMax(&sm[st * 2 + 1], zv[j]); + } + } + block_minmax_flush(sm, minmax); +} + +#define E8_QUANTIZE_WRAPPER(NAME, INT) \ + extern "C" __global__ void NAME( \ + const INT* __restrict__ weight, \ + const float* __restrict__ rms, \ + float s, \ + int rows, \ + int cols, \ + int vecs_per_row, \ + int* __restrict__ fields, \ + int* __restrict__ c_arr, \ + int* __restrict__ minmax) { \ + e8_quantize_fields_body( \ + weight, rms, s, rows, cols, vecs_per_row, fields, c_arr, minmax); \ + } + +E8_QUANTIZE_WRAPPER(e8_quantize_fields_kernel, __nv_bfloat16) +E8_QUANTIZE_WRAPPER(e8_quantize_fields_f32_kernel, float) + +#undef E8_QUANTIZE_WRAPPER + +template +__device__ __forceinline__ void e8_refit_scales_body( + const InT* __restrict__ weight, + const int* __restrict__ fields, // [V,8]: z0..z6, m + const int* __restrict__ c_arr, // [V] + const float* __restrict__ rms, + float initial_s, + float* __restrict__ scales, + float* __restrict__ row_sse, + int rows, + int cols, + int vecs_per_row) { + const int row = blockIdx.x; + if (row >= rows) return; + + float numerator = 0.0f; + float denominator = 0.0f; + float energy = 0.0f; + for (int element = threadIdx.x; element < cols; element += blockDim.x) { + const int vector_in_row = element >> 3; + const int coordinate = element & 7; + const int64_t vector = static_cast(row) * vecs_per_row + vector_in_row; + const int* vector_fields = fields + vector * 8; + const int c = c_arr[vector]; + int z; + if (coordinate < 7) { + z = vector_fields[coordinate]; + } else { + int parity = 0; +#pragma unroll + for (int j = 0; j < 7; ++j) parity += vector_fields[j]; + z = 2 * vector_fields[7] + (parity & 1); + } + const float point = __fadd_rn(static_cast(z), c ? 0.5f : 0.0f); + const float value = e8_load_scalar( + weight + static_cast(row) * cols + element); + numerator = __fadd_rn(numerator, __fmul_rn(value, point)); + denominator = __fadd_rn(denominator, __fmul_rn(point, point)); + energy = __fadd_rn(energy, __fmul_rn(value, value)); + } + + __shared__ float numerator_shared[256]; + __shared__ float denominator_shared[256]; + __shared__ float energy_shared[256]; + numerator_shared[threadIdx.x] = numerator; + denominator_shared[threadIdx.x] = denominator; + energy_shared[threadIdx.x] = energy; + __syncthreads(); + for (int offset = blockDim.x >> 1; offset > 0; offset >>= 1) { + if (threadIdx.x < offset) { + numerator_shared[threadIdx.x] += numerator_shared[threadIdx.x + offset]; + denominator_shared[threadIdx.x] += denominator_shared[threadIdx.x + offset]; + energy_shared[threadIdx.x] += energy_shared[threadIdx.x + offset]; + } + __syncthreads(); + } + if (threadIdx.x == 0) { + const float fallback = __fmul_rn(initial_s, rms[row]); + const float fitted = denominator_shared[0] > 0.0f + ? __fdiv_rn(numerator_shared[0], denominator_shared[0]) + : fallback; + scales[row] = (isfinite(fitted) && fitted > 0.0f) ? fitted : fallback; + row_sse[row] = denominator_shared[0] > 0.0f + ? fmaxf(0.0f, energy_shared[0] - numerator_shared[0] * numerator_shared[0] / + denominator_shared[0]) + : energy_shared[0]; + } +} + +#define E8_REFIT_WRAPPER(NAME, INT) \ + extern "C" __global__ void NAME( \ + const INT* __restrict__ weight, \ + const int* __restrict__ fields, \ + const int* __restrict__ c_arr, \ + const float* __restrict__ rms, \ + float initial_s, \ + float* __restrict__ scales, \ + float* __restrict__ row_sse, \ + int rows, \ + int cols, \ + int vecs_per_row) { \ + e8_refit_scales_body( \ + weight, fields, c_arr, rms, initial_s, scales, row_sse, \ + rows, cols, vecs_per_row); \ + } + +E8_REFIT_WRAPPER(e8_refit_scales_kernel, __nv_bfloat16) +E8_REFIT_WRAPPER(e8_refit_scales_f32_kernel, float) + +#undef E8_REFIT_WRAPPER + +extern "C" __global__ void e8_minmax_fields_kernel( + const int* __restrict__ fields, + const int* __restrict__ c_arr, + int64_t V, + int* __restrict__ minmax) { + __shared__ int sm[kMinMaxLen]; + block_minmax_init(sm); + const int64_t stride = static_cast(gridDim.x) * blockDim.x; + for (int64_t idx = static_cast(blockIdx.x) * blockDim.x + threadIdx.x; + idx < V; idx += stride) { + const int c = c_arr[idx]; + atomicMax(&sm[kCMax], c); + const int* vector_fields = fields + idx * 8; +#pragma unroll + for (int j = 0; j < 8; ++j) { + const int stream = j * 2 + c; + const int value = vector_fields[j]; + atomicMin(&sm[stream * 2], value); + atomicMax(&sm[stream * 2 + 1], value); + } + } + block_minmax_flush(sm, minmax); +} + +extern "C" __global__ void e8_histogram_kernel( + const int* __restrict__ fields, + const int* __restrict__ c_arr, + const int* __restrict__ minmax, + const int* __restrict__ bin_off, // [17]: coset (2 bins), then the 16 coord streams + int* __restrict__ bins, + int64_t V, + int shared_bins) { // >0: stage in dynamic shared memory of this many ints + extern __shared__ int sbins[]; + int* target; + if (shared_bins > 0) { + for (int t = threadIdx.x; t < shared_bins; t += blockDim.x) sbins[t] = 0; + __syncthreads(); + target = sbins; + } else { + target = bins; + } + + const int64_t stride = static_cast(gridDim.x) * blockDim.x; + for (int64_t idx = static_cast(blockIdx.x) * blockDim.x + threadIdx.x; + idx < V; idx += stride) { + const int c = c_arr[idx]; + atomicAdd(&target[bin_off[0] + c], 1); + const int* f = fields + idx * 8; +#pragma unroll + for (int j = 0; j < 8; ++j) { + const int st = j * 2 + c; + atomicAdd(&target[bin_off[1 + st] + (f[j] - minmax[st * 2])], 1); + } + } + if (shared_bins > 0) { + __syncthreads(); + for (int t = threadIdx.x; t < shared_bins; t += blockDim.x) { + if (sbins[t]) atomicAdd(&bins[t], sbins[t]); + } + } +} + +extern "C" __global__ void e8_rans_encode_vector_kernel( + const int* __restrict__ fields, + const int* __restrict__ c_arr, + const uint16_t* __restrict__ freq_tables, + const uint16_t* __restrict__ cdfs, + const int* __restrict__ table_meta, + uint16_t* __restrict__ scratch, + uint32_t* __restrict__ word_counts, + uint32_t* __restrict__ final_states, + int64_t num_vectors, + int tile_vectors, + int num_tiles) { + const int warp_in_block = threadIdx.x >> 5; + const int warps_per_block = blockDim.x >> 5; + const int tile = blockIdx.x * warps_per_block + warp_in_block; + if (tile >= num_tiles) return; + const int lane = threadIdx.x & 31; + const int64_t tile_begin = static_cast(tile) * tile_vectors; + const int tile_count = static_cast( + num_vectors - tile_begin < tile_vectors ? num_vectors - tile_begin : tile_vectors); + uint16_t* tile_scratch = scratch + static_cast(tile) * tile_vectors * 9; + uint32_t state = kStateMin; + uint32_t word_count = 0; + constexpr uint32_t state_check_mul = 1u << (31 - kProbBits); + + for (int base = 0; base < tile_count; base += kNumStates) { + const bool valid = base + lane < tile_count; + const int64_t vector = tile_begin + base + lane; + const int c = valid ? c_arr[vector] : 0; +#pragma unroll + for (int field = 7; field >= 0; --field) { + const int table = 1 + 2 * field + c; + const uint32_t symbol = valid ? static_cast( + fields[vector * 8 + field] - table_meta[table * kMetaStride + kMetaSymMin]) : 0u; + const int freq_off = valid ? table_meta[table * kMetaStride + kMetaFreqOff] : 0; + const uint32_t frequency = valid ? freq_tables[freq_off + symbol] : 1u; + const bool emit = valid && state >= frequency * state_check_mul; + const uint32_t vote = __ballot_sync(0xffffffffu, emit); + const uint32_t prefix = __popc(vote & lane_mask_lt()); + if (emit) { + tile_scratch[word_count + prefix] = static_cast(state); + state >>= 16; + } + word_count += __popc(vote); + if (valid) { + state = (state / frequency) * kTableSize + (state % frequency) + + cdfs[freq_off + symbol]; + } + } + const uint32_t symbol = static_cast(c); + const uint32_t frequency = valid ? freq_tables[symbol] : 1u; + const bool emit = valid && state >= frequency * state_check_mul; + const uint32_t vote = __ballot_sync(0xffffffffu, emit); + const uint32_t prefix = __popc(vote & lane_mask_lt()); + if (emit) { + tile_scratch[word_count + prefix] = static_cast(state); + state >>= 16; + } + word_count += __popc(vote); + if (valid) { + state = (state / frequency) * kTableSize + (state % frequency) + cdfs[symbol]; + } + } + final_states[tile * kNumStates + lane] = state; + if (lane == 0) word_counts[tile] = word_count; +} + +extern "C" __global__ void e8_compact_kernel( + const uint16_t* __restrict__ scratch, + const uint32_t* __restrict__ offsets, + uint16_t* __restrict__ payload, + int tile_elements, + int num_tiles) { + const int tile = blockIdx.x; + if (tile >= num_tiles) return; + const uint32_t begin = offsets[tile]; + const uint32_t count = offsets[tile + 1] - begin; + const uint16_t* source = scratch + static_cast(tile) * tile_elements; + for (uint32_t i = threadIdx.x; i < count; i += blockDim.x) { + payload[begin + i] = source[i]; + } +} + +__device__ __forceinline__ uint32_t e8_rans_decode_symbol( + uint32_t& state, const uint64_t* __restrict__ table) { + const uint32_t slot = state & kStateMask; + const uint64_t entry = __ldg(table + slot); + const uint32_t symbol = static_cast(entry & 0xffffu); + uint32_t frequency = static_cast((entry >> 16) & 0xffffu); + if (frequency == 0) frequency = kTableSize; + const uint32_t cdf = static_cast(entry >> 32); + state = frequency * (state >> kProbBits) + (slot - cdf); + return symbol; +} + +__device__ __forceinline__ uint32_t e8_decode_coset( + uint32_t& state, uint32_t frequency0) { + const uint32_t slot = state & kStateMask; + const uint32_t symbol = slot >= frequency0; + const uint32_t begin = symbol ? frequency0 : 0u; + const uint32_t frequency = symbol ? kTableSize - frequency0 : frequency0; + state = frequency * (state >> kProbBits) + (slot - begin); + return symbol; +} + +__device__ __forceinline__ uint32_t e8_decode_packed32( + uint32_t& state, + const uint32_t* __restrict__ tables, + const int* __restrict__ pack_bits, + int table) { + const uint32_t slot = state & kStateMask; + const uint32_t entry = __ldg(tables + static_cast(table) * kTableSize + slot); + const int bits = pack_bits[table]; + const int symbol_bits = bits & 0xff; + const int frequency_bits = (bits >> 8) & 0xff; + const uint32_t symbol_mask = (1u << symbol_bits) - 1u; + const uint32_t frequency_mask = (1u << frequency_bits) - 1u; + const uint32_t symbol = entry & symbol_mask; + uint32_t frequency = (entry >> symbol_bits) & frequency_mask; + if (frequency == 0) frequency = kTableSize; + const uint32_t delta = entry >> (symbol_bits + frequency_bits); + state = frequency * (state >> kProbBits) + delta; + return symbol; +} + +enum : int { + kStoreF32 = 0, + kStoreF16 = 1, + kStoreF8E4M3 = 2, + kStoreF8E5M2 = 3, + kStoreI8 = 4, + kStoreI16 = 5, + kStoreI32 = 6, + kStoreI64 = 7, + kStoreU8 = 8, + kStoreU16 = 9, + kStoreU32 = 10, + kStoreU64 = 11, + kStoreBool = 12, +}; + +template struct e8_int_range; + +#define E8_INT_RANGE(TYPE, LOW, HIGH_EXCLUSIVE) \ + template <> struct e8_int_range { \ + static constexpr float kLow = LOW; \ + static constexpr float kHighExclusive = HIGH_EXCLUSIVE; \ + } + +E8_INT_RANGE(int8_t, -128.0f, 128.0f); +E8_INT_RANGE(int16_t, -32768.0f, 32768.0f); +E8_INT_RANGE(int32_t, -2147483648.0f, 2147483648.0f); +E8_INT_RANGE(int64_t, -9223372036854775808.0f, 9223372036854775808.0f); +E8_INT_RANGE(uint8_t, 0.0f, 256.0f); +E8_INT_RANGE(uint16_t, 0.0f, 65536.0f); +E8_INT_RANGE(uint32_t, 0.0f, 4294967296.0f); +E8_INT_RANGE(uint64_t, 0.0f, 18446744073709551616.0f); + +#undef E8_INT_RANGE + +template +__device__ __forceinline__ T e8_snap_integer(float value) { + const float high = nextafterf(e8_int_range::kHighExclusive, 0.0f); + const float clamped = fminf(fmaxf(rintf(value), e8_int_range::kLow), high); + return static_cast(clamped); +} + +struct BF16Store { + using pointer = __nv_bfloat16*; + + __device__ static void put8_words( + pointer out, int64_t vector, const int* value, float chalf, float scale, int) { + uint4 packed; + unsigned* words = reinterpret_cast(&packed); +#pragma unroll + for (int pair = 0; pair < 4; ++pair) { + const float p0 = __fadd_rn(static_cast(value[2 * pair]), chalf); + const float p1 = __fadd_rn(static_cast(value[2 * pair + 1]), chalf); + const __nv_bfloat16 lo = __float2bfloat16_rn(__fmul_rn(p0, scale)); + const __nv_bfloat16 hi = __float2bfloat16_rn(__fmul_rn(p1, scale)); + const unsigned lo_bits = reinterpret_cast(&lo)[0]; + const unsigned hi_bits = reinterpret_cast(&hi)[0]; + words[pair] = lo_bits | (hi_bits << 16); + } + *reinterpret_cast(out + vector * 8) = packed; + } + + __device__ static void put8_pair( + pointer out, int64_t vector, const int* value, float chalf, float scale, int) { + uint4 packed; + unsigned* words = reinterpret_cast(&packed); +#pragma unroll + for (int pair = 0; pair < 4; ++pair) { + const float p0 = __fadd_rn(static_cast(value[2 * pair]), chalf); + const float p1 = __fadd_rn(static_cast(value[2 * pair + 1]), chalf); + const __nv_bfloat162 pair_bf = __floats2bfloat162_rn( + __fmul_rn(p0, scale), __fmul_rn(p1, scale)); + words[pair] = *reinterpret_cast(&pair_bf); + } + *reinterpret_cast(out + vector * 8) = packed; + } +}; + +template struct e8_conv; + +template <> struct e8_conv { + using type = uint32_t; + __device__ static uint32_t from(float v) { return __float_as_uint(v); } +}; + +template <> struct e8_conv { + using type = uint16_t; + __device__ static uint16_t from(float v) { + return __half_as_ushort(__float2half_rn(fminf(fmaxf(v, -65504.0f), 65504.0f))); + } +}; + +#define E8_CONV_FP8(K, LIMIT, INTERP) \ + template <> struct e8_conv { \ + using type = __nv_fp8_storage_t; \ + __device__ static __nv_fp8_storage_t from(float v) { \ + return __nv_cvt_float_to_fp8( \ + fminf(fmaxf(v, -LIMIT), LIMIT), __NV_SATFINITE, INTERP); \ + } \ + } + +E8_CONV_FP8(kStoreF8E4M3, 448.0f, __NV_E4M3); +E8_CONV_FP8(kStoreF8E5M2, 57344.0f, __NV_E5M2); + +#undef E8_CONV_FP8 + +#define E8_CONV_INT(K, TYPE) \ + template <> struct e8_conv { \ + using type = TYPE; \ + __device__ static TYPE from(float v) { \ + return e8_snap_integer(v); \ + } \ + } + +E8_CONV_INT(kStoreI8, int8_t); +E8_CONV_INT(kStoreI16, int16_t); +E8_CONV_INT(kStoreI32, int32_t); +E8_CONV_INT(kStoreI64, int64_t); +E8_CONV_INT(kStoreU8, uint8_t); +E8_CONV_INT(kStoreU16, uint16_t); +E8_CONV_INT(kStoreU32, uint32_t); +E8_CONV_INT(kStoreU64, uint64_t); + +#undef E8_CONV_INT + +template <> struct e8_conv { + using type = bool; + __device__ static bool from(float v) { return rintf(v) != 0.0f; } +}; + +// Store count, not conversion cost, dominates here, so the eight converted values are packed into words and written with as +// few wide stores as the container allows. Only 8-byte containers store scalar: one vector of them is wider than a uint4. +template +__device__ __forceinline__ void e8_store8(void* out, int64_t vector, const float* v) { + using T = typename Conv::type; + constexpr int kBytes = 8 * static_cast(sizeof(T)); + char* base = static_cast(out) + vector * kBytes; + if constexpr (sizeof(T) == 8) { +#pragma unroll + for (int j = 0; j < 8; ++j) reinterpret_cast(base)[j] = Conv::from(v[j]); + } else { + constexpr int kWords = kBytes / 4; + constexpr int kPerWord = 4 / static_cast(sizeof(T)); + unsigned words[kWords]; +#pragma unroll + for (int word = 0; word < kWords; ++word) { + unsigned packed = 0; +#pragma unroll + for (int slot = 0; slot < kPerWord; ++slot) { + const unsigned bits = static_cast( + sizeof(T) == 1 ? static_cast(Conv::from(v[word * kPerWord + slot])) + : sizeof(T) == 2 ? static_cast(Conv::from(v[word * kPerWord + slot])) + : static_cast(Conv::from(v[word * kPerWord + slot]))); + packed |= bits << (8 * static_cast(sizeof(T)) * slot); + } + words[word] = packed; + } + if constexpr (kWords == 2) { + *reinterpret_cast(base) = make_uint2(words[0], words[1]); + } else { + *reinterpret_cast(base) = make_uint4(words[0], words[1], words[2], words[3]); + if constexpr (kWords == 8) { + *reinterpret_cast(base + 16) = + make_uint4(words[4], words[5], words[6], words[7]); + } + } + } +} + +struct GenericStore { + using pointer = void*; + + __device__ static void put8( + pointer out, int64_t vector, const int* value, float chalf, float scale, int kind) { + float v[8]; +#pragma unroll + for (int j = 0; j < 8; ++j) { + v[j] = __fmul_rn(__fadd_rn(static_cast(value[j]), chalf), scale); + } + switch (kind) { + case kStoreF32: e8_store8>(out, vector, v); break; + case kStoreF16: e8_store8>(out, vector, v); break; + case kStoreF8E4M3: e8_store8>(out, vector, v); break; + case kStoreF8E5M2: e8_store8>(out, vector, v); break; + case kStoreI8: e8_store8>(out, vector, v); break; + case kStoreI16: e8_store8>(out, vector, v); break; + case kStoreI32: e8_store8>(out, vector, v); break; + case kStoreI64: e8_store8>(out, vector, v); break; + case kStoreU8: e8_store8>(out, vector, v); break; + case kStoreU16: e8_store8>(out, vector, v); break; + case kStoreU32: e8_store8>(out, vector, v); break; + case kStoreU64: e8_store8>(out, vector, v); break; + default: e8_store8>(out, vector, v); break; + } + } + + __device__ static void put8_words( + pointer out, int64_t vector, const int* value, float chalf, float scale, int kind) { + put8(out, vector, value, chalf, scale, kind); + } + + __device__ static void put8_pair( + pointer out, int64_t vector, const int* value, float chalf, float scale, int kind) { + put8(out, vector, value, chalf, scale, kind); + } +}; + +template +__device__ __forceinline__ void e8_decode_vector_fused_body( + const uint16_t* __restrict__ payload, + const uint32_t* __restrict__ offsets, + const uint32_t* __restrict__ states, + const uint64_t* __restrict__ decode_luts, + const int* __restrict__ table_meta, + const float* __restrict__ scales, + typename Store::pointer __restrict__ output, + int store_kind, + int vecs_per_row, + int64_t num_vectors, + int tile_elements, + int num_tiles, + int* __restrict__ error) { + const int warp_in_block = threadIdx.x >> 5; + const int warps_per_block = blockDim.x >> 5; + const int tile = blockIdx.x * warps_per_block + warp_in_block; + if (tile >= num_tiles) return; + const int lane = threadIdx.x & 31; + const int64_t tile_begin = static_cast(tile) * tile_elements; + const int tile_count = static_cast( + num_vectors - tile_begin < tile_elements ? num_vectors - tile_begin : tile_elements); + + const uint16_t* input_begin = payload + offsets[tile]; + const uint16_t* input = payload + offsets[tile + 1]; + uint32_t state = states[tile * kNumStates + lane]; + + const int remainder = tile_count & (kNumStates - 1); + int output_offset = tile_count - remainder; + bool first = true; + while (first || output_offset > 0) { + int valid_lanes; + if (first && remainder) { + valid_lanes = remainder; + } else { + if (first) output_offset = tile_count; + output_offset -= kNumStates; + valid_lanes = kNumStates; + } + first = false; + const bool valid = lane < valid_lanes; + int c = 0; + int value[8]; + if (valid) c = static_cast(e8_rans_decode_symbol(state, decode_luts)); + if (!rans_renormalize_checked(valid, state, input, input_begin)) set_error(error, 2); +#pragma unroll + for (int field = 0; field < 8; ++field) { + if (valid) { + const int table = 1 + 2 * field + c; + const uint64_t* lut = decode_luts + static_cast(table) * kTableSize; + value[field] = static_cast(e8_rans_decode_symbol(state, lut)) + + table_meta[table * kMetaStride + kMetaSymMin]; + } + if (!rans_renormalize_checked(valid, state, input, input_begin)) set_error(error, 2); + } + if (valid) { + int parity = 0; +#pragma unroll + for (int field = 0; field < 7; ++field) parity += value[field]; + value[7] = 2 * value[7] + (parity & 1); + const int64_t vector = tile_begin + output_offset + lane; + const int row = div_i64_i32(vector, vecs_per_row); + Store::put8_words(output, vector, value, c ? 0.5f : 0.0f, scales[row], store_kind); + } + if (output_offset == 0) break; + } + if (input != input_begin || state != kStateMin) set_error(error, 3); +} + +#define E8_FUSED_WRAPPER(NAME) \ + extern "C" __global__ void NAME( \ + const uint16_t* __restrict__ payload, \ + const uint32_t* __restrict__ offsets, \ + const uint32_t* __restrict__ states, \ + const uint64_t* __restrict__ decode_luts, \ + const int* __restrict__ table_meta, \ + const float* __restrict__ scales, \ + __nv_bfloat16* __restrict__ output, \ + int vecs_per_row, \ + int64_t num_vectors, \ + int tile_elements, \ + int num_tiles, \ + int* __restrict__ error) { \ + e8_decode_vector_fused_body( \ + payload, offsets, states, decode_luts, table_meta, scales, \ + output, 0, vecs_per_row, num_vectors, tile_elements, num_tiles, \ + error); \ + } + +#define E8_FUSED_GENERIC_WRAPPER(NAME) \ + extern "C" __global__ void NAME( \ + const uint16_t* __restrict__ payload, \ + const uint32_t* __restrict__ offsets, \ + const uint32_t* __restrict__ states, \ + const uint64_t* __restrict__ decode_luts, \ + const int* __restrict__ table_meta, \ + const float* __restrict__ scales, \ + void* __restrict__ output, \ + int vecs_per_row, \ + int64_t num_vectors, \ + int tile_elements, \ + int num_tiles, \ + int* __restrict__ error, \ + int store_kind) { \ + e8_decode_vector_fused_body( \ + payload, offsets, states, decode_luts, table_meta, scales, \ + output, store_kind, vecs_per_row, num_vectors, tile_elements, \ + num_tiles, error); \ + } + +E8_FUSED_WRAPPER(e8_decode_vector_fused_kernel) +E8_FUSED_GENERIC_WRAPPER(e8_decode_vector_fused_g_kernel) + +#undef E8_FUSED_WRAPPER +#undef E8_FUSED_GENERIC_WRAPPER + +template +__device__ __forceinline__ void e8_decode_packed32_body( + const uint16_t* __restrict__ payload, + const uint32_t* __restrict__ offsets, + const uint32_t* __restrict__ states, + const uint32_t* __restrict__ decode_luts, + const int* __restrict__ pack_bits, + uint32_t coset_frequency0, + const int* __restrict__ table_meta, + const float* __restrict__ scales, + typename Store::pointer __restrict__ output, + int store_kind, + int vecs_per_row, + int64_t num_vectors, + int tile_vectors, + int num_tiles, + int* __restrict__ error) { + __shared__ int pack_bits_shared[17]; + __shared__ int sym_min_shared[17]; + if (threadIdx.x < 17) { + pack_bits_shared[threadIdx.x] = pack_bits[threadIdx.x]; + sym_min_shared[threadIdx.x] = table_meta[threadIdx.x * kMetaStride + kMetaSymMin]; + } + __syncthreads(); + const int warp_in_block = threadIdx.x >> 5; + const int warps_per_block = blockDim.x >> 5; + const int tile = blockIdx.x * warps_per_block + warp_in_block; + if (tile >= num_tiles) return; + const int lane = threadIdx.x & 31; + const int64_t tile_begin = static_cast(tile) * tile_vectors; + const int tile_count = static_cast( + num_vectors - tile_begin < tile_vectors ? num_vectors - tile_begin : tile_vectors); + const uint16_t* input_begin = payload + offsets[tile]; + const uint16_t* input = payload + offsets[tile + 1]; + uint32_t state = states[tile * kNumStates + lane]; + const int remainder = tile_count & (kNumStates - 1); + int output_offset = tile_count - remainder; + bool first = true; + while (first || output_offset > 0) { + int valid_lanes; + if (first && remainder) { + valid_lanes = remainder; + } else { + if (first) output_offset = tile_count; + output_offset -= kNumStates; + valid_lanes = kNumStates; + } + first = false; + const bool valid = lane < valid_lanes; + if constexpr (PREF) { + // The renorm rate grows with the coded rate, so at high coded rates the cold-payload latency matters more. + if (lane < 4) { + const char* p = reinterpret_cast(input) - 128 * (lane + 1); + if (p >= reinterpret_cast(payload)) { + asm volatile("prefetch.global.L2 [%0];" ::"l"(p)); + } + } + } + int c = 0; + int value[8]; + if (valid) c = static_cast(e8_decode_coset(state, coset_frequency0)); + if (!rans_renormalize_checked(valid, state, input, input_begin)) set_error(error, 2); +#pragma unroll + for (int field = 0; field < 8; ++field) { + if (valid) { + const int table = 1 + 2 * field + c; + value[field] = static_cast( + e8_decode_packed32(state, decode_luts, pack_bits_shared, table)) + + sym_min_shared[table]; + } + if (!rans_renormalize_checked(valid, state, input, input_begin)) set_error(error, 2); + } + if (valid) { + int parity = 0; +#pragma unroll + for (int field = 0; field < 7; ++field) parity += value[field]; + value[7] = 2 * value[7] + (parity & 1); + const int64_t vector = tile_begin + output_offset + lane; + const int c_row = div_i64_i32(vector, vecs_per_row); + Store::put8_words(output, vector, value, c ? 0.5f : 0.0f, scales[c_row], store_kind); + } + if (output_offset == 0) break; + } + if (input != input_begin || state != kStateMin) set_error(error, 3); +} + +#define E8_PACKED32_WRAPPER(NAME, PREF) \ + extern "C" __global__ void NAME( \ + const uint16_t* __restrict__ payload, \ + const uint32_t* __restrict__ offsets, \ + const uint32_t* __restrict__ states, \ + const uint32_t* __restrict__ decode_luts, \ + const int* __restrict__ pack_bits, \ + uint32_t coset_frequency0, \ + const int* __restrict__ table_meta, \ + const float* __restrict__ scales, \ + __nv_bfloat16* __restrict__ output, \ + int vecs_per_row, \ + int64_t num_vectors, \ + int tile_vectors, \ + int num_tiles, \ + int* __restrict__ error) { \ + e8_decode_packed32_body( \ + payload, offsets, states, decode_luts, pack_bits, coset_frequency0, \ + table_meta, scales, output, 0, vecs_per_row, num_vectors, \ + tile_vectors, num_tiles, error); \ + } + +#define E8_PACKED32_GENERIC_WRAPPER(NAME, PREF) \ + extern "C" __global__ void NAME( \ + const uint16_t* __restrict__ payload, \ + const uint32_t* __restrict__ offsets, \ + const uint32_t* __restrict__ states, \ + const uint32_t* __restrict__ decode_luts, \ + const int* __restrict__ pack_bits, \ + uint32_t coset_frequency0, \ + const int* __restrict__ table_meta, \ + const float* __restrict__ scales, \ + void* __restrict__ output, \ + int vecs_per_row, \ + int64_t num_vectors, \ + int tile_vectors, \ + int num_tiles, \ + int* __restrict__ error, \ + int store_kind) { \ + e8_decode_packed32_body( \ + payload, offsets, states, decode_luts, pack_bits, coset_frequency0, \ + table_meta, scales, output, store_kind, vecs_per_row, \ + num_vectors, tile_vectors, num_tiles, error); \ + } + +E8_PACKED32_WRAPPER(e8_decode_vector_packed32_kernel, false) +E8_PACKED32_WRAPPER(e8_decode_vector_packed32pf_kernel, true) +E8_PACKED32_GENERIC_WRAPPER(e8_decode_vector_packed32_g_kernel, false) +E8_PACKED32_GENERIC_WRAPPER(e8_decode_vector_packed32pf_g_kernel, true) + +#undef E8_PACKED32_WRAPPER +#undef E8_PACKED32_GENERIC_WRAPPER + +// The global-LUT path gathers a 32-lane random table through L1, which a table larger than the cache does not stay in; staging +// two compact tables in dynamic shared memory is faster while the SM still holds enough resident CTAs. +template +__device__ __forceinline__ void e8_decode_shlut_body( + const uint16_t* __restrict__ payload, + const uint32_t* __restrict__ offsets, + const uint32_t* __restrict__ states, + const uint8_t* __restrict__ sym_lut, // [17 * kTableSize] + const uint32_t* __restrict__ fb_lut, // [fb_entries] begin | freq<<16 + uint32_t coset_frequency0, + const int* __restrict__ table_meta, // [17, kMetaStride]: see kMetaFreqOff, kMetaSymMin + const float* __restrict__ scales, + typename Store::pointer __restrict__ output, + int store_kind, + int vecs_per_row, + int64_t num_vectors, + int tile_vectors, + int num_tiles, + int fb_entries, + int* __restrict__ error) { + extern __shared__ unsigned char shmem_raw[]; + uint8_t* sym_sh = reinterpret_cast(shmem_raw); + uint32_t* fb_sh = reinterpret_cast(sym_sh + 17 * kTableSize); + __shared__ int meta_sh[34]; // [0,17): freq_off, [17,34): sym_min + { + const int n4 = 17 * kTableSize / 16; // uint8 slots, staged as uint4 + const uint4* src4 = reinterpret_cast(sym_lut); + uint4* dst4 = reinterpret_cast(sym_sh); + for (int i = threadIdx.x; i < n4; i += blockDim.x) dst4[i] = src4[i]; + for (int i = threadIdx.x; i < fb_entries; i += blockDim.x) fb_sh[i] = fb_lut[i]; + if (threadIdx.x < 17) { + meta_sh[threadIdx.x] = table_meta[threadIdx.x * kMetaStride + kMetaFreqOff]; + meta_sh[threadIdx.x + 17] = table_meta[threadIdx.x * kMetaStride + kMetaSymMin]; + } + } + __syncthreads(); + + const int warp_in_block = threadIdx.x >> 5; + const int warps_per_block = blockDim.x >> 5; + const int tile = blockIdx.x * warps_per_block + warp_in_block; + if (tile >= num_tiles) return; + const int lane = threadIdx.x & 31; + const uint32_t mask_ge = lane_mask_ge(); + const int64_t tile_begin = static_cast(tile) * tile_vectors; + const int tile_count = static_cast( + num_vectors - tile_begin < tile_vectors ? num_vectors - tile_begin : tile_vectors); + const uint16_t* input_begin = payload + offsets[tile]; + const uint16_t* input = payload + offsets[tile + 1]; + uint32_t state = states[tile * kNumStates + lane]; + const int remainder = tile_count & (kNumStates - 1); + int output_offset = tile_count - remainder; + bool first = true; + while (first || output_offset > 0) { + int valid_lanes; + if (first && remainder) { + valid_lanes = remainder; + } else { + if (first) output_offset = tile_count; + output_offset -= kNumStates; + valid_lanes = kNumStates; + } + first = false; + const bool valid = lane < valid_lanes; + if (lane < 4) { + const char* p = reinterpret_cast(input) - 128 * (lane + 1); + if (p >= reinterpret_cast(payload)) { + asm volatile("prefetch.global.L2 [%0];" ::"l"(p)); + } + } + int c = 0; + int value[8]; + if (valid) c = static_cast(e8_decode_coset(state, coset_frequency0)); + rans_renormalize_unchecked(valid, state, input, mask_ge); + const int cTS = c * kTableSize; +#pragma unroll + for (int field = 0; field < 8; ++field) { + if (valid) { + const uint32_t slot = state & kStateMask; + const uint32_t sym = static_cast( + sym_sh[(2 * field + 1) * kTableSize + cTS + slot]); + const int table = (2 * field + 1) + c; + const uint32_t fb = fb_sh[meta_sh[table] + sym]; + const uint32_t begin = fb & 0xFFFFu; + const uint32_t freq = fb >> 16; + state = freq * (state >> kProbBits) + (slot - begin); + value[field] = static_cast(sym) + meta_sh[table + 17]; + } + rans_renormalize_unchecked(valid, state, input, mask_ge); + } + if (valid) { + int parity = 0; +#pragma unroll + for (int field = 0; field < 7; ++field) parity += value[field]; + value[7] = 2 * value[7] + (parity & 1); + const int64_t vector = tile_begin + output_offset + lane; + const int c_row = div_i64_i32(vector, vecs_per_row); + const float chalf = c ? 0.5f : 0.0f; + Store::put8_pair(output, vector, value, chalf, scales[c_row], store_kind); + } + if (output_offset == 0) break; + } + if (input != input_begin || state != kStateMin) set_error(error, 3); +} + +#define E8_SHLUT_WRAPPER(NAME) \ + extern "C" __global__ void NAME( \ + const uint16_t* __restrict__ payload, \ + const uint32_t* __restrict__ offsets, \ + const uint32_t* __restrict__ states, \ + const uint8_t* __restrict__ sym_lut, \ + const uint32_t* __restrict__ fb_lut, \ + uint32_t coset_frequency0, \ + const int* __restrict__ table_meta, \ + const float* __restrict__ scales, \ + __nv_bfloat16* __restrict__ output, \ + int vecs_per_row, \ + int64_t num_vectors, \ + int tile_vectors, \ + int num_tiles, \ + int fb_entries, \ + int* __restrict__ error) { \ + e8_decode_shlut_body( \ + payload, offsets, states, sym_lut, fb_lut, coset_frequency0, table_meta, \ + scales, output, 0, vecs_per_row, num_vectors, tile_vectors, \ + num_tiles, fb_entries, error); \ + } + +#define E8_SHLUT_GENERIC_WRAPPER(NAME) \ + extern "C" __global__ void NAME( \ + const uint16_t* __restrict__ payload, \ + const uint32_t* __restrict__ offsets, \ + const uint32_t* __restrict__ states, \ + const uint8_t* __restrict__ sym_lut, \ + const uint32_t* __restrict__ fb_lut, \ + uint32_t coset_frequency0, \ + const int* __restrict__ table_meta, \ + const float* __restrict__ scales, \ + void* __restrict__ output, \ + int vecs_per_row, \ + int64_t num_vectors, \ + int tile_vectors, \ + int num_tiles, \ + int fb_entries, \ + int* __restrict__ error, \ + int store_kind) { \ + e8_decode_shlut_body( \ + payload, offsets, states, sym_lut, fb_lut, coset_frequency0, table_meta, \ + scales, output, store_kind, vecs_per_row, num_vectors, \ + tile_vectors, num_tiles, fb_entries, error); \ + } + +E8_SHLUT_WRAPPER(e8_decode_vector_shlut8pf_kernel) +E8_SHLUT_GENERIC_WRAPPER(e8_decode_vector_shlut8pf_g_kernel) + +#undef E8_SHLUT_WRAPPER +#undef E8_SHLUT_GENERIC_WRAPPER + diff --git a/entropack/schemes/lattice_rans/rans.py b/entropack/schemes/lattice_rans/rans.py new file mode 100644 index 0000000..b13de7f --- /dev/null +++ b/entropack/schemes/lattice_rans/rans.py @@ -0,0 +1,209 @@ +import numpy as np + +from ..tile_ans.eager import normalize_counts, quantized_cross_entropy +from ..tile_ans.format import NUM_STATES, STATE_MIN +from .format import BITS_PER_BYTE, NUM_COORD_FIELDS, vector_tile_elements + +__all__ = [ + "build_codec_tables", "coded_bytes", "decode_vector_stream", "encode_vector_stream", "normalize_freq", + "vector_stream_analytic_bytes", +] + + +def normalize_freq(counts: np.ndarray, table_size: int) -> np.ndarray: + counts = np.asarray(counts, dtype=np.int64) + if counts.size == 0 or int(counts.sum()) == 0: + raise ValueError("cannot build an rANS table from an empty symbol stream") + if counts.size > table_size: + raise ValueError( + f"E8 coordinate alphabet {counts.size} exceeds rANS table_size {table_size}; " + "raise prob_bits or coarsen the lattice scale" + ) + return normalize_counts(counts, table_size) + + +def build_codec_tables(freq: np.ndarray, probability_bits: int): + freq = np.asarray(freq, dtype=np.int64) + table_size = 1 << probability_bits + if freq.size > table_size: + raise ValueError(f"alphabet {freq.size} exceeds table_size {table_size}") + if int(freq.sum()) != table_size: + raise ValueError("normalized frequencies must sum to table_size") + cdf = np.zeros(freq.size, dtype=np.int64) + cdf[1:] = np.cumsum(freq)[:-1] + lut = np.zeros(table_size, dtype=np.uint64) + running = 0 + for symbol in range(freq.size): + value = int(freq[symbol]) + if value == 0: + continue + packed_freq = 0 if value == table_size else value + entry = (np.uint64(running) << np.uint64(32)) | (np.uint64(packed_freq) << np.uint64(16)) | np.uint64(symbol) + lut[running : running + value] = entry + running += value + if running != table_size: + raise AssertionError(f"frequency sum is {running}, expected {table_size}") + return cdf, lut + + +def _decode_group(states, luts, table_ids, words, pointer, output, valid_lanes, probability_bits): + table_size = 1 << probability_bits + reads = [] + for lane in range(valid_lanes): + state = int(states[lane]) + slot = state & (table_size - 1) + entry = int(luts[0 if table_ids is None else int(table_ids[lane])][slot]) + symbol = entry & 0xFFFF + frequency = (entry >> 16) & 0xFFFF + if frequency == 0: + frequency = table_size + cdf = entry >> 32 + output[lane] = symbol + state = frequency * (state >> probability_bits) + (slot - cdf) + states[lane] = state + if state < STATE_MIN: + reads.append(lane) + first_word = pointer - len(reads) + if first_word < 0: + raise ValueError("lattice_rans rANS payload is truncated") + for index, lane in enumerate(reads): + states[lane] = (int(states[lane]) << 16) | int(words[first_word + index]) + return first_word + + +def _decode_tile_group(luts, states, words, pointer, coset_out, field_out, valid_lanes, probability_bits): + pointer = _decode_group(states, luts, None, words, pointer, coset_out, valid_lanes, probability_bits) + for field in range(NUM_COORD_FIELDS): + pointer = _decode_group( + states, [luts[1 + 2 * field], luts[2 + 2 * field]], coset_out, words, pointer, field_out[:, field], valid_lanes, + probability_bits, + ) + return pointer + + +def encode_vector_stream( + cosets: np.ndarray, field_symbols: np.ndarray, frequencies: list[np.ndarray], probability_bits: int, tile_vectors: int, +): + cosets = np.asarray(cosets, dtype=np.int64) + field_symbols = np.asarray(field_symbols, dtype=np.int64) + if field_symbols.shape != (cosets.size, NUM_COORD_FIELDS): + raise ValueError(f"field_symbols must have shape [num_vectors, {NUM_COORD_FIELDS}]") + cdfs = [] + for frequency in frequencies: + frequency = np.asarray(frequency, dtype=np.int64) + if frequency.size: + cdfs.append(build_codec_tables(frequency, probability_bits)[0]) + else: + cdfs.append(np.empty(0, dtype=np.int64)) + table_size = 1 << probability_bits + state_check_shift = 31 - probability_bits + num_tiles = max(1, (cosets.size + tile_vectors - 1) // tile_vectors) + states = np.empty((num_tiles, NUM_STATES), dtype=np.uint32) + parts = [] + offsets = np.zeros(num_tiles + 1, dtype=np.int64) + total = 0 + for tile in range(num_tiles): + begin = tile * tile_vectors + end = min(begin + tile_vectors, cosets.size) + tile_states = np.full(NUM_STATES, STATE_MIN, dtype=np.uint32) + words = [] + for base in range(begin, end, NUM_STATES): + limit = min(NUM_STATES, end - base) + for field in range(7, -1, -1): + for lane in range(limit): + index = base + lane + table = 1 + 2 * field + int(cosets[index]) + symbol = int(field_symbols[index, field]) + frequency = int(frequencies[table][symbol]) + state = int(tile_states[lane]) + if state >= (frequency << state_check_shift): + words.append(state & 0xFFFF) + state >>= 16 + tile_states[lane] = (state // frequency) * table_size + (state % frequency) + int(cdfs[table][symbol]) + for lane in range(limit): + symbol = int(cosets[base + lane]) + frequency = int(frequencies[0][symbol]) + state = int(tile_states[lane]) + if state >= (frequency << state_check_shift): + words.append(state & 0xFFFF) + state >>= 16 + tile_states[lane] = (state // frequency) * table_size + (state % frequency) + int(cdfs[0][symbol]) + tile_words = np.asarray(words, dtype=np.uint16) + states[tile] = tile_states + parts.append(tile_words) + total += tile_words.size + offsets[tile + 1] = total + payload = np.concatenate(parts) if parts else np.empty(0, dtype=np.uint16) + return payload, states, offsets + + +def decode_vector_stream( + words: np.ndarray, offsets: np.ndarray, states: np.ndarray, frequencies: list[np.ndarray], probability_bits: int, + tile_vectors: int, num_vectors: int, +): + luts = [ + np.empty(0, dtype=np.uint64) if np.asarray(frequency).size == 0 + else build_codec_tables(np.asarray(frequency, dtype=np.int64), probability_bits)[1] for frequency in frequencies + ] + cosets = np.empty(num_vectors, dtype=np.int64) + fields = np.empty((num_vectors, NUM_COORD_FIELDS), dtype=np.int64) + num_tiles = max(1, (num_vectors + tile_vectors - 1) // tile_vectors) + for tile in range(num_tiles): + begin = tile * tile_vectors + tile_count = min(tile_vectors, num_vectors - begin) + tile_words = np.asarray(words[offsets[tile] : offsets[tile + 1]], dtype=np.uint16) + tile_states = np.asarray(states[tile], dtype=np.uint32).copy() + pointer = tile_words.size + remainder = tile_count % NUM_STATES + offset = tile_count - remainder + if remainder: + pointer = _decode_tile_group( + luts, tile_states, tile_words, pointer, cosets[begin + offset : begin + tile_count], + fields[begin + offset : begin + tile_count], remainder, probability_bits, + ) + while offset > 0: + offset -= NUM_STATES + pointer = _decode_tile_group( + luts, tile_states, tile_words, pointer, cosets[begin + offset : begin + offset + NUM_STATES], + fields[begin + offset : begin + offset + NUM_STATES], NUM_STATES, probability_bits, + ) + if pointer != 0: + raise ValueError("lattice_rans vector rANS stream contains unread payload words") + return cosets, fields + + +def coded_bytes(counts, sizes, prob_bits: int, tile_elements: int) -> float: + table_size = 1 << prob_bits + counts_by_table = [] + frequencies = [] + overflow_bits = 0.0 + empty_counts = np.empty(0, dtype=np.int64) + empty_freqs = np.empty(0, dtype=np.uint16) + for n_symbols, count in zip(sizes, counts, strict=True): + if n_symbols and count.size <= table_size: + counts_by_table.append(count) + frequencies.append(normalize_freq(count, table_size)) + else: + overflow_bits += n_symbols * prob_bits + counts_by_table.append(empty_counts) + frequencies.append(empty_freqs) + total = vector_stream_analytic_bytes( + counts_by_table, frequencies, prob_bits, vector_tile_elements(tile_elements), sizes[0] + ) + return total + overflow_bits / BITS_PER_BYTE + + +def vector_stream_analytic_bytes( + counts_by_table: list[np.ndarray], frequencies: list[np.ndarray], probability_bits: int, tile_vectors: int, + num_vectors: int, +) -> float: + bits = 0.0 + frequency_bytes = 0 + for counts, frequency in zip(counts_by_table, frequencies, strict=True): + if counts.size: + bits += quantized_cross_entropy(counts.astype(np.int64), frequency.astype(np.int64), probability_bits) + frequency_bytes += frequency.size * 2 + num_tiles = max(1, (num_vectors + tile_vectors - 1) // tile_vectors) + state_residual_bits = num_tiles * NUM_STATES * 16 + payload_words = int(np.ceil(max(0.0, bits - state_residual_bits) / 16.0)) + return payload_words * 2 + num_tiles * NUM_STATES * 4 + (num_tiles + 1) * 4 + frequency_bytes diff --git a/entropack/schemes/lattice_rans/rdo.py b/entropack/schemes/lattice_rans/rdo.py new file mode 100644 index 0000000..b31463a --- /dev/null +++ b/entropack/schemes/lattice_rans/rdo.py @@ -0,0 +1,162 @@ +from dataclasses import replace +from typing import NamedTuple + +import numpy as np +import torch + +from .format import BITS_PER_BYTE, LATTICE_DIM, NUM_COORD_FIELDS + +RATIO_SPAN = (0.70, 1.45) + + +def ratio_ladder(count): + """The scale ratios one refinement pass prices: 1.0, then the geometric midpoint of the widest + log-gap inside RATIO_SPAN, one per further candidate. A K-point ladder is therefore a subset of + every wider one, so widening can only help a row.""" + pts = [1.0] + while len(pts) < count: + bounds = [RATIO_SPAN[0], *pts, RATIO_SPAN[1]] + i = max(range(1, len(bounds)), key=lambda j: bounds[j] / bounds[j - 1]) + pts.insert(i - 1, (bounds[i] * bounds[i - 1]) ** 0.5) + return tuple(pts) + + +class RateTable(NamedTuple): + """Both index tensors are int32, which keeps the per-iteration index temporaries :func:`row_rates` builds small against the + table they index. + """ + + values: torch.Tensor + offsets: torch.Tensor + minima: torch.Tensor + + +def stream_ranges(candidates, n_streams): + minima = [0] * n_streams + maxima = [0] * n_streams + for stream in range(n_streams): + alive = [ + (candidate.sym_min[stream], candidate.sym_min[stream] + candidate.alphabets[stream] - 1) for candidate in candidates + if candidate.alphabets[stream] > 0 + ] + if alive: + minima[stream] = min(item[0] for item in alive) + maxima[stream] = max(item[1] for item in alive) + return minima, maxima + + +def rate_costs(counts, sizes, sym_min, minima, maxima, device, alpha=0.5): + """The streams are concatenated on the host and uploaded as one table; uploading them separately costs one host-to-device + copy per stream and dominates this function. The arithmetic itself is small and runs slower on the device than here. + """ + tables = [] + offsets = [0] + for stream, count in enumerate(counts): + width = maxima[stream] - minima[stream] + 1 + expanded = np.zeros(width, dtype=np.float64) + if count.size: + offset = sym_min[stream] - minima[stream] + expanded[offset : offset + count.size] = count + probability = (expanded + alpha) / (sizes[stream] + alpha * width) + tables.append(-np.log2(probability).astype(np.float32)) + offsets.append(offsets[-1] + width) + return RateTable( + torch.from_numpy(np.concatenate(tables)).to(device), torch.tensor(offsets, dtype=torch.int32, device=device), + torch.tensor(minima, dtype=torch.int32, device=device), + ) + + +def row_rates(rows, cols, candidate, table): + """Gather rather than mask: a boolean mask per stream needs ``nonzero`` to bring the hit count back to the host, and those + per-stream synchronizations leave small layers host-bound. Transposing the field matrix first turns the coordinate reads + from strided into contiguous ones, which matters more the less of the matrix the device cache holds. The accumulation + order is untouched, so the sum is bit-identical. + """ + vectors_per_row = cols // LATTICE_DIM + fields = candidate.fields.reshape(-1, LATTICE_DIM).t().contiguous().to(torch.int32) + coset = candidate.c_arr.to(torch.int32) + rate = table.values[table.offsets[0] + coset] + for field in range(NUM_COORD_FIELDS): + stream = 1 + 2 * field + coset + rate = rate + table.values[table.offsets[stream] + (fields[field] - table.minima[stream])] + return rate.reshape(rows, vectors_per_row).sum(1) + + +def compact_candidate(candidate): + live = [index for index in range(1, len(candidate.sym_min)) if candidate.alphabets[index] > 0] + lows = [candidate.sym_min[index] for index in live] + highs = [candidate.sym_min[index] + candidate.alphabets[index] - 1 for index in live] + low = min(lows, default=0) + high = max(highs, default=0) + if low >= -128 and high <= 127: + field_dtype = torch.int8 + elif low >= -32768 and high <= 32767: + field_dtype = torch.int16 + else: + field_dtype = torch.int32 + return replace(candidate, fields=candidate.fields.to(field_dtype), c_arr=candidate.c_arr.to(torch.uint8)) + + +def select_candidate_rows(rows, cols, candidates, choice): + vectors_per_row = cols // LATTICE_DIM + device = candidates[0].fields.device + fields = torch.empty(candidates[0].fields.numel(), dtype=torch.int32, device=device) + c_arr = torch.empty(candidates[0].c_arr.numel(), dtype=torch.int32, device=device) + fields_rows = fields.reshape(rows, vectors_per_row, LATTICE_DIM) + c_rows = c_arr.reshape(rows, vectors_per_row) + for index, candidate in enumerate(candidates): + selected_rows = torch.nonzero(choice == index, as_tuple=False).flatten() + if selected_rows.numel() == 0: + continue + source = candidate.fields.reshape(rows, vectors_per_row, LATTICE_DIM)[selected_rows] + fields_rows[selected_rows] = source.to(torch.int32) + c_rows[selected_rows] = candidate.c_arr.reshape(rows, vectors_per_row)[selected_rows].to(torch.int32) + scales = torch.stack([candidate.scales for candidate in candidates], dim=1) + row = torch.arange(rows, device=choice.device) + return fields, c_arr, scales[row, choice] + + +def choose_rate_tradeoff(distortion, rates, row, desired_rate): + zero_choice = distortion.argmin(1) + if rates[row, zero_choice].sum() <= desired_rate: + return zero_choice + choice = zero_choice + low = 0.0 + high = max(float(distortion.mean().item()) * 1.0e-4, 1.0e-12) + for _ in range(50): + trial = (distortion + high * rates).argmin(1) + if rates[row, trial].sum() <= desired_rate: + break + high *= 2.0 + for _ in range(20): + middle = high * 0.5 if low == 0.0 else (low * high) ** 0.5 + trial = (distortion + middle * rates).argmin(1) + if rates[row, trial].sum() > desired_rate: + low = middle + else: + high = middle + choice = trial + return choice + + +def optimize_rows(candidates, *, rows, cols, baseline_index, iterations, device, summarize, total_bytes): + minima, maxima = stream_ranges(candidates, len(candidates[0].counts)) + distortion = torch.stack([candidate.row_sse for candidate in candidates], dim=1) + choice = torch.full((rows,), baseline_index, dtype=torch.int64, device=device) + row = torch.arange(rows, device=device) + fields, c_arr, fitted_scales = select_candidate_rows(rows, cols, candidates, choice) + summary = summarize(fields, c_arr) + baseline = candidates[baseline_index] + target_bytes = total_bytes(baseline.counts, baseline.sizes) + + for _ in range(iterations): + table = rate_costs(summary.counts, summary.sizes, summary.sym_min, minima, maxima, device) + rates = torch.stack([row_rates(rows, cols, candidate, table) for candidate in candidates], dim=1) + current_rate = rates[row, choice].sum() + current_bytes = total_bytes(summary.counts, summary.sizes) + desired_rate = current_rate + BITS_PER_BYTE * (target_bytes - current_bytes) + choice = choose_rate_tradeoff(distortion, rates, row, desired_rate) + fields, c_arr, fitted_scales = select_candidate_rows(rows, cols, candidates, choice) + summary = summarize(fields, c_arr) + + return summary, fitted_scales diff --git a/entropack/schemes/tile_ans/__init__.py b/entropack/schemes/tile_ans/__init__.py new file mode 100644 index 0000000..a50af43 --- /dev/null +++ b/entropack/schemes/tile_ans/__init__.py @@ -0,0 +1,55 @@ +import torch + +from ..base import Scheme, packed_buffers, register_scheme +from ..checks import prepare_weight +from .format import OPTIONS_BY_DTYPE, PACKED_KEYS, TileBuffers, TileANSConfig, validate_packed + + +class TileANSScheme(Scheme): + name = "tile_ans" + buffer_names = PACKED_KEYS + priority = 100 + dtypes = tuple(OPTIONS_BY_DTYPE) + lanes = {"eager": "eager", "cuda": "cuda"} + + def options_for(self, dtype): + tile_elements, probability_bits = OPTIONS_BY_DTYPE[dtype] + return { + "tile_elements": tile_elements, "probability_bits": probability_bits, + "raw_lane_threshold": TileANSConfig().raw_lane_threshold, + } + + def encode(self, weight: torch.Tensor, config: TileANSConfig) -> dict: + weight = prepare_weight(weight, scheme=self.name) + tile_elements = config.tile_elements + if tile_elements == 0: + storage_bytes = weight.numel() * weight.element_size() + tile_elements = 4096 if storage_bytes <= 32 * 1024 * 1024 else 8192 + device = weight.device + lane, run_on = self.lane_for(weight, config.execution_backend) + if run_on is not None and device != run_on: + weight = weight.to(run_on) + buffers = lane.encode( + weight=weight, tile_elements=tile_elements, probability_bits=config.probability_bits, + raw_lane_threshold=config.raw_lane_threshold, threads_per_block=config.threads_per_block, + ) + return {key: value.to(device) for key, value in buffers._asdict().items()} + + def validate_buffers(self, buffers, shape, dtype): + validate_packed(buffers, shape, dtype) + + def decode(self, packed: dict, *, shape: tuple[int, ...], dtype: torch.dtype, + config: TileANSConfig) -> torch.Tensor: + buffers = packed_buffers(packed, TileBuffers) + source = buffers.layout.device + lane, lane_device = self.lane_for(buffers.layout, config.execution_backend, gate_dtype=False) + if lane_device is not None and source != lane_device: + buffers = TileBuffers._make(value.to(lane_device) for value in buffers) + flat = lane.decode(buffers, dtype=dtype, threads_per_block=config.threads_per_block) + out = flat.reshape(shape) + return out if out.device == source else out.to(source) + + +register_scheme(TileANSScheme()) + +__all__ = ["TileANSConfig", "TileANSScheme"] diff --git a/entropack/schemes/tile_ans/cuda.py b/entropack/schemes/tile_ans/cuda.py new file mode 100644 index 0000000..25d42dd --- /dev/null +++ b/entropack/schemes/tile_ans/cuda.py @@ -0,0 +1,170 @@ +from pathlib import Path + +import cupy +import numpy as np +import torch + +from ...backends.cuda import device as _device_caps +from ...backends.cuda.kernels import KernelLibrary +from ...backends.cuda.kernels import device_index as _device_index +from ...backends.cuda.kernels import external_stream as _external_stream +from ...backends.cuda.kernels import pointer as _pointer +from .eager import build_tables_from_counts, lane_modes_from_counts, select_coding_options +from .format import ( + AUTO_PROB_BITS, BLOCK_SIZE, ENCODE_TABLE_SHARED_BYTES, HISTOGRAM_MAX_WARPS, HISTOGRAM_MIN_WARPS, + HISTOGRAM_WARP_BUDGET, LANE_ANS, LANE_RAW, NUM_STATES, TileBuffers, make_layout, num_streams, + num_tiles, parse_layout_cached, +) + +_CUDA_PATH = Path(__file__).parent / "tile_ans.cu" +_KERNEL_NAMES = ( + "tile_ans_histogram_kernel", "tile_ans_encode_kernel", "tile_ans_compact_kernel", "tile_ans_decode_raw0_ans1_kernel", + "tile_ans_decode_raw3_ans1_kernel", "tile_ans_decode_all_raw_kernel", "tile_ans_decode_kernel", +) +_LIBRARY = KernelLibrary( + key="tile_ans", source=_CUDA_PATH, defines=lambda _device, probability_bits: (f"TILE_ANS_PROB_BITS={probability_bits}",), + includes=(_CUDA_PATH.parent,), kernel_names=_KERNEL_NAMES, +) +_kernel = _LIBRARY.kernel + + +def _lane_histogram(contiguous, caps, device_index, torch_stream, probability_bits): + num_elements = contiguous.numel() + num_lanes = contiguous.element_size() + histograms = torch.zeros((num_lanes, 256), dtype=torch.int64, device=contiguous.device) + histogram_warps = min( + HISTOGRAM_MAX_WARPS, max(HISTOGRAM_MIN_WARPS, HISTOGRAM_WARP_BUDGET // num_lanes), + ) + histogram_threads = histogram_warps * caps.warp_size + histogram_blocks = caps.grid(-(-num_elements // histogram_threads), histogram_threads) + histogram_shared = histogram_warps * num_lanes * 256 * 4 + with cupy.cuda.Device(device_index), _external_stream(torch_stream): + _kernel(device_index, probability_bits, "tile_ans_histogram_kernel")( + (histogram_blocks,), (histogram_threads,), + (_pointer(contiguous), _pointer(histograms), np.int64(num_elements), np.int32(num_lanes)), + shared_mem=histogram_shared, + ) + return histograms.cpu().numpy() + + +def _lane_tables(counts, probability_bits, raw_lane_threshold, tile_elements, device): + if probability_bits == AUTO_PROB_BITS: + probability_bits, frequencies, cdfs, decode_tables, lane_modes = select_coding_options( + counts, raw_lane_threshold, tile_elements + ) + else: + frequencies, cdfs, decode_tables = build_tables_from_counts(counts, probability_bits) + lane_modes = lane_modes_from_counts(counts, raw_lane_threshold, tile_elements) + return ( + probability_bits, int(np.count_nonzero(lane_modes == LANE_ANS)), torch.from_numpy(frequencies).to(device), + torch.from_numpy(cdfs).to(device), torch.from_numpy(decode_tables).to(device), torch.from_numpy(lane_modes).to(device), + ) + + +def encode( + *, weight: torch.Tensor, tile_elements: int, probability_bits: int, raw_lane_threshold: float, + threads_per_block: int | None, +) -> TileBuffers: + contiguous = weight.contiguous() + num_elements = contiguous.numel() + num_lanes = contiguous.element_size() + streams = num_streams(num_elements, num_lanes, tile_elements) + device = contiguous.device + device_index = _device_index(contiguous) + torch_stream = torch.cuda.current_stream(device) + caps = _device_caps.caps(device) + + counts = _lane_histogram(contiguous, caps, device_index, torch_stream, probability_bits) + ( + probability_bits, num_ans_lanes, frequencies_gpu, cdfs_gpu, decode_tables_gpu, lane_modes_gpu, + ) = _lane_tables(counts, probability_bits, raw_lane_threshold, tile_elements, device) + + tiles = num_tiles(num_elements, tile_elements) + states = torch.empty((tiles * num_ans_lanes, NUM_STATES), dtype=torch.uint32, device=device) + word_counts = torch.empty(streams, dtype=torch.uint32, device=device) + scratch = torch.empty(streams * tile_elements, dtype=torch.uint16, device=device) + threads = _device_caps.resolve_threads(caps, threads_per_block, BLOCK_SIZE) + warps = threads // caps.warp_size + blocks = -(-tiles // warps) + args = ( + _pointer(contiguous), _pointer(frequencies_gpu), _pointer(cdfs_gpu), _pointer(lane_modes_gpu), _pointer(scratch), + _pointer(word_counts), _pointer(states), np.int64(num_elements), np.int32(tile_elements), np.int32(num_lanes), + np.int32(tiles), + ) + + with cupy.cuda.Device(device_index), _external_stream(torch_stream): + _kernel(device_index, probability_bits, "tile_ans_encode_kernel")( + (blocks, num_lanes), (threads,), args, shared_mem=ENCODE_TABLE_SHARED_BYTES, + ) + counts_cp = cupy.from_dlpack(word_counts) + offsets64 = torch.empty(streams + 1, dtype=torch.int64, device=device) + offsets_cp = cupy.from_dlpack(offsets64) + offsets_cp[0] = 0 + cupy.cumsum(counts_cp, dtype=cupy.int64, out=offsets_cp[1:]) + total_words = int(offsets64[-1].item()) + if total_words >= 1 << 32: + raise ValueError("tile_ans payload exceeds uint32 offset capacity") + + offsets = offsets64.to(torch.uint32) + payload = torch.empty(total_words, dtype=torch.uint16, device=device) + with cupy.cuda.Device(device_index), _external_stream(torch_stream): + _kernel(device_index, probability_bits, "tile_ans_compact_kernel")( + (streams,), (threads,), + (_pointer(scratch), _pointer(offsets), _pointer(payload), np.int32(tile_elements), np.int32(streams)), + ) + layout = make_layout(num_elements, num_lanes, tile_elements, probability_bits).to(device) + return TileBuffers( + payload=payload, offsets=offsets, states=states, decode_tables=decode_tables_gpu, lane_modes=lane_modes_gpu, + layout=layout, + ) + + +def decode(buffers: TileBuffers, *, dtype: torch.dtype, threads_per_block: int | None) -> torch.Tensor: + payload, offsets, states = buffers.payload, buffers.offsets, buffers.states + decode_tables, lane_modes, layout = buffers.decode_tables, buffers.lane_modes, buffers.layout + tile_elements, probability_bits, num_lanes, num_elements = parse_layout_cached(layout) + output = torch.empty(num_elements * num_lanes, dtype=torch.uint8, device=payload.device) + device_index = _device_index(payload) + torch_stream = torch.cuda.current_stream(payload.device) + caps = _device_caps.caps(payload.device) + + tiles = num_tiles(num_elements, tile_elements) + lane_mode_values = getattr(layout, "_tile_ans_lane_modes", None) + if lane_mode_values is None: + lane_mode_values = tuple(int(value) for value in lane_modes.detach().cpu().tolist()) + layout._tile_ans_lane_modes = lane_mode_values + + all_raw_writer = all(mode == LANE_RAW for mode in lane_mode_values) + paired_writer = num_lanes == 2 and lane_mode_values == (LANE_RAW, LANE_ANS) + quad_writer = num_lanes == 4 and lane_mode_values == (LANE_RAW, LANE_RAW, LANE_RAW, LANE_ANS) + table_shared = (1 << probability_bits) * 4 + grid_lanes = 1 + block_owns_tile = False + if all_raw_writer: + kernel_name, shared_bytes = "tile_ans_decode_all_raw_kernel", 0 + block_owns_tile = True + elif paired_writer: + # bf16 splits into a raw low byte and an ANS-coded high byte, so the coded lane's table is small enough to stage in + # shared memory: renormalization then drops its bounds check and the lane mask is hoisted out of the symbol loop. + kernel_name = "tile_ans_decode_raw0_ans1_kernel" + shared_bytes = table_shared + elif quad_writer: + kernel_name, shared_bytes = "tile_ans_decode_raw3_ans1_kernel", 0 + else: + kernel_name, shared_bytes = "tile_ans_decode_kernel", table_shared + grid_lanes = num_lanes + + kernel = _kernel(device_index, probability_bits, kernel_name) + args = ( + _pointer(payload), _pointer(offsets), _pointer(states), _pointer(decode_tables), _pointer(lane_modes), _pointer(output), + np.int64(num_elements), np.int32(tile_elements), np.int32(num_lanes), np.int32(tiles), + ) + + def launch_decode(threads: int) -> None: + blocks = tiles if block_owns_tile else -(-tiles // (threads // caps.warp_size)) + grid = (blocks, grid_lanes) if grid_lanes > 1 else (blocks,) + with cupy.cuda.Device(device_index), _external_stream(torch_stream): + kernel(grid, (threads,), args, shared_mem=shared_bytes) + + launch_decode(_device_caps.resolve_threads(caps, threads_per_block, BLOCK_SIZE)) + return output.view(dtype) diff --git a/entropack/schemes/tile_ans/device.cuh b/entropack/schemes/tile_ans/device.cuh new file mode 100644 index 0000000..eb267b6 --- /dev/null +++ b/entropack/schemes/tile_ans/device.cuh @@ -0,0 +1,90 @@ +#pragma once + +#include + +#ifndef TILE_ANS_PROB_BITS +#define TILE_ANS_PROB_BITS 12 +#endif + +constexpr int kProbBits = TILE_ANS_PROB_BITS; +constexpr uint32_t kTableSize = 1u << kProbBits; +constexpr uint32_t kStateMask = kTableSize - 1u; +constexpr uint32_t kStateMin = 1u << 15; +constexpr int kNumStates = 32; + +__device__ __forceinline__ uint32_t lane_mask_ge() { + uint32_t mask; + asm("mov.u32 %0, %%lanemask_ge;" : "=r"(mask)); + return mask; +} + +__device__ __forceinline__ uint32_t lane_mask_lt() { + uint32_t mask; + asm("mov.u32 %0, %%lanemask_lt;" : "=r"(mask)); + return mask; +} + +__device__ __forceinline__ uint32_t rans_decode_symbol( + uint32_t& state, + const uint32_t* __restrict__ table) { + const uint32_t slot = state & kStateMask; + const uint32_t entry = table[slot]; + const uint32_t symbol = entry & 0xFFu; + uint32_t frequency = (entry >> 8) & 0xFFFu; + if (frequency == 0) { + frequency = kTableSize; + } + const uint32_t cdf = entry >> 20; + state = frequency * (state >> kProbBits) + (slot - cdf); + return symbol; +} + +__device__ __forceinline__ bool rans_renormalize_checked( + bool valid, + uint32_t& state, + const uint16_t*& input, + const uint16_t* __restrict__ input_begin) { + const bool read = valid && state < kStateMin; + const uint32_t vote = __ballot_sync(0xFFFFFFFFu, read); + const uint32_t prefix = __popc(vote & lane_mask_ge()); + bool valid_read = true; + if (read) { + const uint16_t* address = input - prefix; + valid_read = address >= input_begin; + const uint32_t word = valid_read ? *address : 0u; + state = (state << 16) | word; + } + input -= __popc(vote); + return valid_read; +} + +__device__ __forceinline__ void rans_renormalize_unchecked( + bool valid, + uint32_t& state, + const uint16_t*& input, + uint32_t mask_ge) { + const bool read = valid && state < kStateMin; + const uint32_t vote = __ballot_sync(0xFFFFFFFFu, read); + if (read) { + const uint32_t prefix = __popc(vote & mask_ge); + state = (state << 16) | input[-static_cast(prefix)]; + } + input -= __popc(vote); +} + +__device__ __forceinline__ void decode_group( + bool valid, + uint32_t& state, + const uint32_t* __restrict__ table, + const uint16_t*& input, + const uint16_t* __restrict__ input_begin, + uint8_t* __restrict__ output, + int64_t output_index, + int64_t num_lanes, + int byte_lane) { + if (valid) { + output[output_index * num_lanes + byte_lane] = + static_cast(rans_decode_symbol(state, table)); + } + (void)rans_renormalize_checked(valid, state, input, input_begin); +} diff --git a/entropack/schemes/tile_ans/eager.py b/entropack/schemes/tile_ans/eager.py new file mode 100644 index 0000000..98e700b --- /dev/null +++ b/entropack/schemes/tile_ans/eager.py @@ -0,0 +1,307 @@ +import heapq + +import numpy as np +import torch + +from .format import ( + AUTO_PROB_BITS, LANE_ANS, LANE_RAW, NUM_STATES, STATE_MIN, SUPPORTED_PROB_BITS, TileBuffers, + make_layout, num_tiles, parse_layout_cached, +) + +_STATE_BITS = 31 +_RENORM_BITS = 16 +_AUTO_RATIO_TOLERANCE = 0.0015 + + +def normalize_counts(counts: np.ndarray, table_size: int) -> np.ndarray: + counts = np.asarray(counts, dtype=np.int64) + present = np.flatnonzero(counts) + if present.size == 0: + raise ValueError("cannot build an rANS codebook from empty input") + + target = counts.astype(np.float64) * (table_size / int(counts.sum())) + frequencies = np.floor(target).astype(np.int64) + frequencies[present] = np.maximum(frequencies[present], 1) + + difference = table_size - int(frequencies.sum()) + if difference > 0: + residual = target - frequencies + queue = [(-float(residual[symbol]), int(symbol)) for symbol in present] + heapq.heapify(queue) + for _ in range(difference): + negative_residual, symbol = heapq.heappop(queue) + frequencies[symbol] += 1 + heapq.heappush(queue, (negative_residual + 1.0, symbol)) + elif difference < 0: + residual = target - frequencies + queue = [(float(residual[symbol]), int(symbol)) for symbol in present if frequencies[symbol] > 1] + if not queue: + raise ValueError("unable to normalize rANS frequencies") + heapq.heapify(queue) + for step in range(-difference): + symbol_residual, symbol = heapq.heappop(queue) + frequencies[symbol] -= 1 + if frequencies[symbol] > 1: + heapq.heappush(queue, (symbol_residual + 1.0, symbol)) + elif not queue and step + 1 < -difference: + raise ValueError("unable to normalize rANS frequencies") + return frequencies.astype(np.uint16) + + +def _build_tables_from_frequencies(frequencies: np.ndarray, probability_bits: int): + frequencies = np.asarray(frequencies, dtype=np.uint16) + table_size = 1 << probability_bits + num_lanes = frequencies.shape[0] + cdfs = np.zeros((num_lanes, 256), dtype=np.uint16) + decode_tables = np.empty((num_lanes, table_size), dtype=np.uint32) + + for lane, freq in enumerate(frequencies): + cdf = np.zeros(256, dtype=np.uint16) + running = 0 + for symbol in range(256): + cdf[symbol] = running + value = int(freq[symbol]) + if value: + packed_frequency = 0 if value == 4096 else value + decode_tables[lane, running : running + value] = (running << 20) | (packed_frequency << 8) | symbol + running += value + if running != table_size: + raise AssertionError(f"normalized frequency sum is {running}, expected {table_size}") + cdfs[lane] = cdf + return frequencies, cdfs, decode_tables + + +def build_tables_from_counts(counts_by_lane: np.ndarray, probability_bits: int): + counts_by_lane = np.asarray(counts_by_lane, dtype=np.int64) + table_size = 1 << probability_bits + frequencies = np.stack([normalize_counts(counts, table_size) for counts in counts_by_lane]) + return _build_tables_from_frequencies(frequencies, probability_bits) + + +def quantized_cross_entropy(counts: np.ndarray, frequencies: np.ndarray, probability_bits: int) -> float: + present = counts > 0 + return float( + np.sum(counts[present].astype(np.float64) * (probability_bits - np.log2(frequencies[present].astype(np.float64)))) + ) + + +def lane_modes_from_counts( + counts_by_lane: np.ndarray, raw_lane_threshold: float, tile_elements: int, frequencies: np.ndarray | None = None, + probability_bits: int | None = None, +) -> np.ndarray: + if not 0.0 <= raw_lane_threshold <= 8.0: + raise ValueError("raw_lane_threshold must be in [0, 8]") + effective_threshold = min(raw_lane_threshold, 8.0 - (NUM_STATES * 32) / tile_elements) + counts_by_lane = np.asarray(counts_by_lane, dtype=np.int64) + modes = np.empty(counts_by_lane.shape[0], dtype=np.uint8) + for lane, counts in enumerate(counts_by_lane): + if frequencies is None: + probabilities = counts[counts > 0].astype(np.float64) / int(counts.sum()) + bits_per_symbol = -np.sum(probabilities * np.log2(probabilities)) + else: + if probability_bits is None: + raise ValueError("probability_bits is required with normalized frequencies") + bits_per_symbol = quantized_cross_entropy(counts, frequencies[lane], probability_bits) / int(counts.sum()) + modes[lane] = LANE_RAW if bits_per_symbol >= effective_threshold else LANE_ANS + return modes + + +def select_coding_options(counts_by_lane: np.ndarray, raw_lane_threshold: float, tile_elements: int): + counts_by_lane = np.asarray(counts_by_lane, dtype=np.int64) + num_elements = int(counts_by_lane[0].sum()) + num_lanes = counts_by_lane.shape[0] + tiles = num_tiles(num_elements, tile_elements) + full_tiles, tail = divmod(num_elements, tile_elements) + raw_words = full_tiles * ((tile_elements + 1) // 2) + (tail + 1) // 2 + original_bytes = num_elements * num_lanes + # Quantizing a table can only add cost, so a lane above the threshold stays above it at every precision; when all of them + # are, only the coarsest table needs pricing. + plain = lane_modes_from_counts(counts_by_lane, raw_lane_threshold, tile_elements) + probability_variants = SUPPORTED_PROB_BITS[:1] if np.all(plain == LANE_RAW) else SUPPORTED_PROB_BITS + + candidates = [] + for probability_bits in probability_variants: + table_size = 1 << probability_bits + frequencies = np.stack([normalize_counts(counts, table_size) for counts in counts_by_lane]) + modes = lane_modes_from_counts(counts_by_lane, raw_lane_threshold, tile_elements, frequencies, probability_bits) + estimated_bytes = (tiles * num_lanes + 1) * 4 + num_lanes * table_size * 4 + num_lanes + 5 * 8 + for lane, mode in enumerate(modes): + if mode == LANE_RAW: + estimated_bytes += raw_words * 2 + else: + cross_entropy_bits = quantized_cross_entropy(counts_by_lane[lane], frequencies[lane], probability_bits) + estimated_bytes += cross_entropy_bits / 8 + estimated_bytes += tiles * NUM_STATES * 4 + candidates.append((estimated_bytes, probability_bits, frequencies, modes)) + + # Among the tables within tolerance of the cheapest, the smallest: a coarser grid costs a little rate and saves a + # proportionally larger decode table. + best_bytes = min(value for value, *_ in candidates) + tolerance = original_bytes * _AUTO_RATIO_TOLERANCE + _, probability_bits, frequencies, modes = min( + (candidate for candidate in candidates if candidate[0] <= best_bytes + tolerance), key=lambda candidate: candidate[1], + ) + frequencies, cdfs, decode_tables = _build_tables_from_frequencies(frequencies, probability_bits) + return probability_bits, frequencies, cdfs, decode_tables, modes + + +def _encode_raw_stream(symbols: np.ndarray) -> np.ndarray: + words = np.zeros((symbols.size + 1) // 2, dtype=np.uint16) + words |= symbols[0::2].astype(np.uint16) + if symbols.size > 1: + words[: symbols[1::2].size] |= symbols[1::2].astype(np.uint16) << 8 + return words + + +def _encode_stream(symbols: np.ndarray, frequencies: np.ndarray, cdfs: np.ndarray, probability_bits: int): + states = np.full(NUM_STATES, STATE_MIN, dtype=np.uint32) + words: list[int] = [] + table_size = 1 << probability_bits + state_check_shift = _STATE_BITS - probability_bits + + for base in range(0, symbols.size, NUM_STATES): + limit = min(NUM_STATES, symbols.size - base) + for lane in range(limit): + symbol = int(symbols[base + lane]) + frequency = int(frequencies[symbol]) + state = int(states[lane]) + if state >= (frequency << state_check_shift): + words.append(state & 0xFFFF) + state >>= _RENORM_BITS + state = (state // frequency) * table_size + (state % frequency) + int(cdfs[symbol]) + states[lane] = state + return states, np.asarray(words, dtype=np.uint16) + + +def encode( + *, weight: torch.Tensor, tile_elements: int, probability_bits: int, raw_lane_threshold: float, **_ignored, +) -> TileBuffers: + if weight.numel() == 0: + raise ValueError("tile_ans does not support empty tensors") + contiguous = weight.detach().contiguous() + num_elements = contiguous.numel() + num_lanes = contiguous.element_size() + raw = contiguous.reshape(-1).view(torch.uint8).cpu().numpy().copy() + lane_bytes = raw.reshape(num_elements, num_lanes) + counts = np.stack([np.bincount(lane_bytes[:, lane], minlength=256) for lane in range(num_lanes)]) + if probability_bits == AUTO_PROB_BITS: + probability_bits, frequencies, cdfs, decode_tables, lane_modes = select_coding_options( + counts, raw_lane_threshold, tile_elements + ) + else: + frequencies, cdfs, decode_tables = build_tables_from_counts(counts, probability_bits) + lane_modes = lane_modes_from_counts(counts, raw_lane_threshold, tile_elements) + + tiles = num_tiles(num_elements, tile_elements) + num_streams = tiles * num_lanes + num_ans_lanes = int((lane_modes == LANE_ANS).sum()) + states = np.empty((tiles * num_ans_lanes, NUM_STATES), dtype=np.uint32) + offsets = np.empty(num_streams + 1, dtype=np.uint32) + offsets[0] = 0 + payload_parts = [] + + stream = 0 + ans_stream = 0 + payload_words = 0 + for lane in range(num_lanes): + for tile in range(tiles): + begin = tile * tile_elements + end = min(begin + tile_elements, num_elements) + if lane_modes[lane] == LANE_RAW: + words = _encode_raw_stream(lane_bytes[begin:end, lane]) + else: + stream_states, words = _encode_stream( + lane_bytes[begin:end, lane], frequencies[lane], cdfs[lane], probability_bits, + ) + states[ans_stream] = stream_states + ans_stream += 1 + if payload_words + words.size >= 1 << 32: + raise ValueError("tile_ans payload exceeds uint32 offset capacity") + payload_parts.append(words) + payload_words += words.size + offsets[stream + 1] = payload_words + stream += 1 + + payload = np.concatenate(payload_parts) if payload_parts else np.empty(0, dtype=np.uint16) + return TileBuffers( + payload=torch.from_numpy(payload), offsets=torch.from_numpy(offsets), states=torch.from_numpy(states), + decode_tables=torch.from_numpy(decode_tables), lane_modes=torch.from_numpy(lane_modes), + layout=make_layout(num_elements, num_lanes, tile_elements, probability_bits), + ) + + +def _decode_group(states, table, words, pointer, output, base, valid_lanes, probability_bits): + table_size = 1 << probability_bits + reads = [] + for lane in range(valid_lanes): + state = int(states[lane]) + slot = state & (table_size - 1) + entry = int(table[slot]) + symbol = entry & 0xFF + frequency = (entry >> 8) & 0xFFF + if frequency == 0: + frequency = table_size + cdf = entry >> 20 + output[base + lane] = symbol + state = frequency * (state >> probability_bits) + (slot - cdf) + states[lane] = state + if state < STATE_MIN: + reads.append(lane) + + first_word = pointer - len(reads) + if first_word < 0: + raise ValueError("tile_ans payload is truncated") + for index, lane in enumerate(reads): + states[lane] = (int(states[lane]) << _RENORM_BITS) | int(words[first_word + index]) + return first_word + + +def decode(buffers: TileBuffers, *, dtype: torch.dtype, **_ignored) -> torch.Tensor: + payload = buffers.payload.detach().cpu().numpy() + offsets = buffers.offsets.detach().cpu().numpy().astype(np.int64) + states = buffers.states.detach().cpu().numpy().astype(np.uint32) + tables = buffers.decode_tables.detach().cpu().numpy().astype(np.uint32) + lane_modes = buffers.lane_modes.detach().cpu().numpy().astype(np.uint8) + tile_elements, probability_bits, num_lanes, num_elements = parse_layout_cached(buffers.layout) + + output = np.empty(num_elements * num_lanes, dtype=np.uint8) + tiles = num_tiles(num_elements, tile_elements) + stream = 0 + ans_stream = 0 + for byte_lane in range(num_lanes): + for tile in range(tiles): + tile_begin = tile * tile_elements + tile_count = min(tile_elements, num_elements - tile_begin) + begin = int(offsets[stream]) + pointer = int(offsets[stream + 1]) + words = payload[begin:pointer] + if lane_modes[byte_lane] == LANE_RAW: + expected_words = (tile_count + 1) // 2 + if words.size != expected_words: + raise ValueError("tile_ans raw lane payload length is invalid") + lane_output = np.empty(tile_count, dtype=np.uint8) + lane_output[0::2] = (words & 0xFF).astype(np.uint8) + if tile_count > 1: + lane_output[1::2] = (words[: tile_count // 2] >> 8).astype(np.uint8) + else: + remainder = tile_count % NUM_STATES + stream_states = states[ans_stream].copy() + ans_stream += 1 + pointer -= begin + lane_output = np.empty(tile_count, dtype=np.uint8) + offset = tile_count - remainder + if remainder: + pointer = _decode_group( + stream_states, tables[byte_lane], words, pointer, lane_output, offset, remainder, probability_bits + ) + while offset > 0: + offset -= NUM_STATES + pointer = _decode_group( + stream_states, tables[byte_lane], words, pointer, lane_output, offset, NUM_STATES, probability_bits + ) + if pointer != 0: + raise ValueError("tile_ans payload contains unread words") + output[tile_begin * num_lanes + byte_lane : (tile_begin + tile_count) * num_lanes : num_lanes] = lane_output + stream += 1 + + return torch.from_numpy(output).view(dtype).to(buffers.payload.device) diff --git a/entropack/schemes/tile_ans/format.py b/entropack/schemes/tile_ans/format.py new file mode 100644 index 0000000..0704b2d --- /dev/null +++ b/entropack/schemes/tile_ans/format.py @@ -0,0 +1,164 @@ +import math +from dataclasses import dataclass +from typing import Annotated, NamedTuple + +import torch + +from ..config import CompressionConfig, OneOf, Range +from ..base import cached_parse + +BLOCK_SIZE = 256 +HISTOGRAM_WARP_BUDGET = 16 +HISTOGRAM_MIN_WARPS = 2 +HISTOGRAM_MAX_WARPS = 8 +ENCODE_TABLE_SHARED_BYTES = 2 * 256 * 2 + +AUTO_PROB_BITS = 0 +AUTO_TILE_ELEMENTS = 0 +SUPPORTED_PROB_BITS = (9, 10, 11, 12) +NUM_STATES = 32 +STATE_MIN = 1 << 15 +LANE_ANS = 0 +LANE_RAW = 1 + +TILE_ELEMENTS_LIMIT = 1 << 31 +RAW_LANE_THRESHOLD_LIMIT = 8.0 +TILE_ELEMENTS_MESSAGE = ( + "tile_elements must be positive, or 0 for automatic selection, and fit the CUDA int32 launch ABI" +) + + +@dataclass +class TileANSConfig(CompressionConfig): + """Lossless compression settings for the supported tensor dtypes. + + Tile size and probability precision are stored with the compressed tensor. A zero value + requests automatic selection. Block width controls GPU execution.""" + + #: Elements per independent tile. Zero selects 4096, or 8192 for tensors larger than 32 MiB. + tile_elements: Annotated[int, Range(0, TILE_ELEMENTS_LIMIT - 1, message=TILE_ELEMENTS_MESSAGE)] = AUTO_TILE_ELEMENTS + #: Probability-table precision. Zero selects using the tensor symbol histogram. + probability_bits: Annotated[int, OneOf(SUPPORTED_PROB_BITS, silent=(AUTO_PROB_BITS,))] = AUTO_PROB_BITS + #: Byte streams at or above this estimated cost in bits per symbol are stored directly. + raw_lane_threshold: Annotated[float, Range(0.0, RAW_LANE_THRESHOLD_LIMIT)] = 7.9 + #: GPU block width for encoding and decoding. None selects a device-dependent value. + threads_per_block: Annotated[int | None, Range(1, None)] = None + + +OPTIONS_BY_DTYPE = { + torch.float32: (8192, 10), torch.float16: (8192, 11), torch.bfloat16: (0, 11), + torch.float8_e4m3fn: (8192, 0), torch.float8_e4m3fnuz: (8192, 0), + torch.float8_e5m2: (8192, 0), torch.float8_e5m2fnuz: (8192, 0), + torch.int64: (8192, 10), torch.int32: (16384, 10), torch.int16: (8192, 9), + torch.int8: (8192, 0), torch.uint64: (8192, 9), torch.uint32: (8192, 10), + torch.uint16: (8192, 9), torch.uint8: (8192, 0), torch.bool: (8192, 9), +} + + +class TileBuffers(NamedTuple): + payload: torch.Tensor + offsets: torch.Tensor + states: torch.Tensor + decode_tables: torch.Tensor + lane_modes: torch.Tensor + layout: torch.Tensor + + +PACKED_KEYS = TileBuffers._fields + + +def num_tiles(num_elements: int, tile_elements: int) -> int: + return -(-num_elements // tile_elements) + + +def num_streams(num_elements: int, num_lanes: int, tile_elements: int) -> int: + return num_tiles(num_elements, tile_elements) * num_lanes + + +def make_layout(num_elements: int, num_lanes: int, tile_elements: int, probability_bits: int): + if probability_bits not in SUPPORTED_PROB_BITS: + raise ValueError(f"probability_bits must be one of {SUPPORTED_PROB_BITS}") + return torch.tensor([tile_elements, probability_bits, num_lanes, num_elements], dtype=torch.int64) + + +def parse_layout(layout: torch.Tensor) -> tuple[int, int, int, int]: + if layout.dtype != torch.int64 or layout.ndim != 1 or layout.numel() != 4: + raise ValueError("tile_ans layout must be int64[4]") + tile_elements, prob_bits, num_lanes, num_elements = (int(value) for value in layout.detach().cpu().tolist()) + if tile_elements <= 0 or prob_bits not in SUPPORTED_PROB_BITS or num_lanes <= 0 or num_elements <= 0: + raise ValueError( + "invalid tile_ans layout values: " f"tile_elements={tile_elements}, prob_bits={prob_bits}, " + f"num_lanes={num_lanes}, num_elements={num_elements}" + ) + return tile_elements, prob_bits, num_lanes, num_elements + + +def parse_layout_cached(layout: torch.Tensor) -> tuple[int, int, int, int]: + return cached_parse(layout, parse_layout, "_tile_ans_layout") + + +def validate_packed(buffers: dict[str, torch.Tensor], shape, dtype: torch.dtype) -> None: + missing = [key for key in PACKED_KEYS if key not in buffers] + if missing: + raise ValueError(f"tile_ans packed data is missing buffers: {missing}") + if not all(isinstance(buffers[key], torch.Tensor) for key in PACKED_KEYS): + raise TypeError("tile_ans packed buffers must be torch.Tensor values") + + expected_layout = { + "payload": (torch.uint16, 1), "offsets": (torch.uint32, 1), "states": (torch.uint32, 2), + "decode_tables": (torch.uint32, 2), "lane_modes": (torch.uint8, 1), "layout": (torch.int64, 1), + } + for key, (expected_dtype, ndim) in expected_layout.items(): + tensor = buffers[key] + if not tensor.is_contiguous(): + raise ValueError(f"tile_ans buffer '{key}' must be contiguous") + if tensor.dtype != expected_dtype or tensor.ndim != ndim: + raise ValueError( + f"tile_ans buffer '{key}' must be {ndim}D {expected_dtype}, " + f"got shape={tuple(tensor.shape)}, dtype={tensor.dtype}" + ) + + devices = {buffers[key].device for key in PACKED_KEYS} + if len(devices) != 1: + raise ValueError(f"tile_ans packed buffers must share one device, got {devices}") + + tile_elements, prob_bits, num_lanes, num_elements = parse_layout(buffers["layout"]) + element_size = torch.empty((), dtype=dtype).element_size() + if num_lanes != element_size: + raise ValueError(f"tile_ans layout has {num_lanes} byte lanes but dtype {dtype} uses {element_size} bytes") + normalized_shape = tuple(shape) + if any(not isinstance(dim, int) or dim < 0 for dim in normalized_shape): + raise ValueError(f"invalid tile_ans tensor shape: {normalized_shape}") + if math.prod(normalized_shape) != num_elements: + raise ValueError(f"tile_ans shape {normalized_shape} does not match {num_elements} elements") + + streams = num_streams(num_elements, num_lanes, tile_elements) + tiles = num_tiles(num_elements, tile_elements) + if tuple(buffers["decode_tables"].shape) != (num_lanes, 1 << prob_bits): + raise ValueError("tile_ans decode_tables shape does not match layout") + if buffers["lane_modes"].numel() != num_lanes: + raise ValueError("tile_ans lane_modes length does not match layout") + lane_modes = buffers["lane_modes"].detach().cpu() + if ((lane_modes != LANE_ANS) & (lane_modes != LANE_RAW)).any(): + raise ValueError("tile_ans lane_modes contains an unknown codec mode") + ans_lanes = int((lane_modes == LANE_ANS).sum()) + if tuple(buffers["states"].shape) != (tiles * ans_lanes, NUM_STATES): + raise ValueError("tile_ans states shape does not match ANS lanes/layout") + if buffers["offsets"].numel() != streams + 1: + raise ValueError("tile_ans offsets length does not match layout") + + offsets = buffers["offsets"].detach().cpu().to(torch.int64) + if offsets[0] != 0 or offsets[-1] != buffers["payload"].numel(): + raise ValueError("tile_ans offsets endpoints are invalid") + if (offsets[1:] < offsets[:-1]).any(): + raise ValueError("tile_ans offsets must be monotone") + + for byte_lane, mode in enumerate(lane_modes.tolist()): + if mode != LANE_RAW: + continue + first_stream = byte_lane * tiles + sizes = offsets[first_stream + 1 : first_stream + tiles + 1] - offsets[first_stream : first_stream + tiles] + raw_words = torch.full_like(sizes, (tile_elements + 1) // 2) + raw_words[-1] = (num_elements - (tiles - 1) * tile_elements + 1) // 2 + if not torch.equal(sizes, raw_words): + raise ValueError("tile_ans raw lane payload length is invalid") diff --git a/entropack/schemes/tile_ans/tile_ans.cu b/entropack/schemes/tile_ans/tile_ans.cu new file mode 100644 index 0000000..400a5ca --- /dev/null +++ b/entropack/schemes/tile_ans/tile_ans.cu @@ -0,0 +1,469 @@ +#include "device.cuh" + +// Encode kernels: +// tile_ans_histogram_kernel per-byte-lane 256-bin symbol counts. Every warp keeps a private histogram in shared +// memory and the block folds them, so a block contributes at most 256 global atomics per +// lane. +// tile_ans_encode_kernel one warp per (tile, byte lane) stream: 32 interleaved rANS states, 16-bit +// renormalization, words written to that stream's scratch region. A lane marked raw is +// packed two bytes per word instead of being coded. +// tile_ans_compact_kernel gathers the per-stream scratch words into one payload. +// +// Decode kernels, selected by the byte-lane pattern stored in the checkpoint. All produce identical output and differ only +// in how many lanes one warp reassembles at once: +// tile_ans_decode_raw0_ans1_kernel two lanes, raw then coded: the bf16 case, where one warp rebuilds both bytes of every +// element and stages the coded lane's table in shared memory. +// tile_ans_decode_raw3_ans1_kernel four lanes, the first three raw: the fp32 case. +// tile_ans_decode_all_raw_kernel every lane raw, i.e. nothing was compressible: one block per tile, a plain unpack. +// tile_ans_decode_kernel the general case: the grid carries the byte lane, coded lanes are decoded by rANS +// through a shared-memory table and raw lanes are unpacked. +// All four share one argument list, so the host builds one tuple and picks a name; the raw-only kernels ignore the state and +// table arguments. +// +// The rANS state machine, the renormalization variants and the shared decode helpers live in device.cuh, which the +// lattice_rans lane compiles against as well. + + +extern "C" __global__ void tile_ans_histogram_kernel( + const uint8_t* __restrict__ input, + uint64_t* __restrict__ histograms, + int64_t num_elements, + int num_lanes) { + extern __shared__ uint32_t warp_bins[]; + const int warp = threadIdx.x >> 5; + const int warps_per_block = blockDim.x >> 5; + const int bins_per_warp = num_lanes * 256; + const int total_bins = warps_per_block * bins_per_warp; + for (int index = threadIdx.x; index < total_bins; index += blockDim.x) { + warp_bins[index] = 0; + } + __syncthreads(); + + uint32_t* bins = warp_bins + warp * bins_per_warp; + int64_t element = static_cast(blockIdx.x) * blockDim.x + threadIdx.x; + const int64_t stride = static_cast(gridDim.x) * blockDim.x; + for (; element < num_elements; element += stride) { + const uint8_t* value = input + element * num_lanes; +#pragma unroll + for (int byte_lane = 0; byte_lane < num_lanes; ++byte_lane) { + atomicAdd(&bins[byte_lane * 256 + value[byte_lane]], 1u); + } + } + __syncthreads(); + for (int index = threadIdx.x; index < bins_per_warp; index += blockDim.x) { + uint32_t sum = 0; +#pragma unroll + for (int source_warp = 0; source_warp < warps_per_block; ++source_warp) { + sum += warp_bins[source_warp * bins_per_warp + index]; + } + if (sum) { + atomicAdd( + reinterpret_cast(histograms + index), + static_cast(sum)); + } + } +} + +extern "C" __global__ void tile_ans_encode_kernel( + const uint8_t* __restrict__ input, + const uint16_t* __restrict__ frequencies, + const uint16_t* __restrict__ cdfs, + const uint8_t* __restrict__ lane_modes, + uint16_t* __restrict__ scratch, + uint32_t* __restrict__ word_counts, + uint32_t* __restrict__ final_states, + int64_t num_elements, + int tile_elements, + int num_lanes, + int num_tiles) { + const int warp_in_block = threadIdx.x >> 5; + const int lane = threadIdx.x & 31; + const int warps_per_block = blockDim.x >> 5; + const int tile = blockIdx.x * warps_per_block + warp_in_block; + const int byte_lane = blockIdx.y; + + extern __shared__ uint16_t shared_encode_tables[]; + uint16_t* frequency = shared_encode_tables; + uint16_t* cdf = frequency + 256; + const uint16_t* source_frequency = frequencies + byte_lane * 256; + const uint16_t* source_cdf = cdfs + byte_lane * 256; + for (int index = threadIdx.x; index < 256; index += blockDim.x) { + frequency[index] = source_frequency[index]; + cdf[index] = source_cdf[index]; + } + __syncthreads(); + + if (tile >= num_tiles || byte_lane >= num_lanes) { + return; + } + const int stream = byte_lane * num_tiles + tile; + const int64_t tile_begin = static_cast(tile) * tile_elements; + const int tile_count = static_cast( + (num_elements - tile_begin < tile_elements) + ? (num_elements - tile_begin) + : tile_elements); + + uint16_t* stream_scratch = scratch + static_cast(stream) * tile_elements; + if (lane_modes[byte_lane] != 0) { + const int raw_words = (tile_count + 1) >> 1; + for (int word = lane; word < raw_words; word += kNumStates) { + const int first = word << 1; + const uint32_t low = input[(tile_begin + first) * num_lanes + byte_lane]; + const uint32_t high = (first + 1 < tile_count) + ? input[(tile_begin + first + 1) * num_lanes + byte_lane] + : 0u; + stream_scratch[word] = static_cast(low | (high << 8)); + } + if (lane == 0) { + word_counts[stream] = raw_words; + } + return; + } + + int ans_lane = 0; + for (int prior_lane = 0; prior_lane < byte_lane; ++prior_lane) { + ans_lane += lane_modes[prior_lane] == 0; + } + const int ans_stream = ans_lane * num_tiles + tile; + uint32_t state = kStateMin; + uint32_t word_count = 0; + constexpr uint32_t state_check_mul = 1u << (31 - kProbBits); + + for (int base = 0; base < tile_count; base += kNumStates) { + const bool valid = base + lane < tile_count; + const uint32_t symbol = valid + ? input[(tile_begin + base + lane) * num_lanes + byte_lane] + : 0u; + const uint32_t freq = valid ? frequency[symbol] : 1u; + const bool emit = valid && state >= freq * state_check_mul; + const uint32_t vote = __ballot_sync(0xFFFFFFFFu, emit); + const uint32_t prefix = __popc(vote & lane_mask_lt()); + if (emit) { + stream_scratch[word_count + prefix] = static_cast(state); + state >>= 16; + } + word_count += __popc(vote); + if (valid) { + state = (state / freq) * kTableSize + (state % freq) + cdf[symbol]; + } + } + + final_states[ans_stream * kNumStates + lane] = state; + if (lane == 0) { + word_counts[stream] = word_count; + } +} + +extern "C" __global__ void tile_ans_compact_kernel( + const uint16_t* __restrict__ scratch, + const uint32_t* __restrict__ offsets, + uint16_t* __restrict__ payload, + int tile_elements, + int num_streams) { + const int stream = blockIdx.x; + if (stream >= num_streams) { + return; + } + const uint32_t begin = offsets[stream]; + const uint32_t count = offsets[stream + 1] - begin; + const uint16_t* source = scratch + static_cast(stream) * tile_elements; + for (uint32_t index = threadIdx.x; index < count; index += blockDim.x) { + payload[begin + index] = source[index]; + } +} + +extern "C" __global__ void tile_ans_decode_raw0_ans1_kernel( + const uint16_t* __restrict__ payload, + const uint32_t* __restrict__ offsets, + const uint32_t* __restrict__ states, + const uint32_t* __restrict__ decode_tables, + const uint8_t* __restrict__ lane_modes, + uint8_t* __restrict__ output, + int64_t num_elements, + int tile_elements, + int num_lanes, + int num_tiles) { + extern __shared__ uint32_t shared_table[]; + const uint4* src4 = reinterpret_cast(decode_tables + kTableSize); + uint4* dst4 = reinterpret_cast(shared_table); + for (int i = threadIdx.x; i < kTableSize / 4; i += blockDim.x) dst4[i] = src4[i]; + __syncthreads(); + const uint32_t* __restrict__ table = shared_table; + + const int warp_in_block = threadIdx.x >> 5; + const int lane = threadIdx.x & 31; + const int warps_per_block = blockDim.x >> 5; + const int tile = blockIdx.x * warps_per_block + warp_in_block; + if (tile >= num_tiles || num_lanes != 2) { + return; + } + + const int64_t tile_begin = static_cast(tile) * tile_elements; + const int tile_count = static_cast( + (num_elements - tile_begin < tile_elements) + ? (num_elements - tile_begin) + : tile_elements); + const int raw_stream = tile; + const int ans_stream = num_tiles + tile; + const uint32_t expected_raw_words = (tile_count + 1) >> 1; + if (offsets[raw_stream + 1] - offsets[raw_stream] != expected_raw_words) { + return; + } + const uint16_t* raw_words = payload + offsets[raw_stream]; + const uint16_t* input_begin = payload + offsets[ans_stream]; + const uint16_t* input = payload + offsets[ans_stream + 1]; + uint32_t state = states[tile * kNumStates + lane]; + uint16_t* output16 = reinterpret_cast(output); + const uint32_t mask_ge = lane_mask_ge(); + + const int remainder = tile_count & (kNumStates - 1); + int output_offset = tile_count - remainder; + if (remainder) { + const bool valid = lane < remainder; + const int local_index = output_offset + lane; + if (valid) { + const uint32_t high = rans_decode_symbol(state, table); + const uint32_t packed_low = raw_words[local_index >> 1]; + const uint32_t low = (packed_low >> ((local_index & 1) * 8)) & 0xFFu; + output16[tile_begin + local_index] = static_cast(low | (high << 8)); + } + rans_renormalize_unchecked(valid, state, input, mask_ge); + } + + while (output_offset > 0) { + output_offset -= kNumStates; + const int local_index = output_offset + lane; + const uint32_t high = rans_decode_symbol(state, table); + const uint32_t packed_low = raw_words[local_index >> 1]; + const uint32_t low = (packed_low >> ((local_index & 1) * 8)) & 0xFFu; + output16[tile_begin + local_index] = static_cast(low | (high << 8)); + rans_renormalize_unchecked(true, state, input, mask_ge); + } +} + +__device__ __forceinline__ void decode_group_raw3_ans1( + bool valid, + uint32_t& state, + const uint32_t* __restrict__ table, + const uint16_t*& input, + const uint16_t* __restrict__ input_begin, + const uint16_t* __restrict__ raw0, + const uint16_t* __restrict__ raw1, + const uint16_t* __restrict__ raw2, + uint32_t* __restrict__ output, + int64_t output_index, + int local_index) { + if (valid) { + const uint32_t high = rans_decode_symbol(state, table); + const int word = local_index >> 1; + const int shift = (local_index & 1) * 8; + const uint32_t b0 = (raw0[word] >> shift) & 0xFFu; + const uint32_t b1 = (raw1[word] >> shift) & 0xFFu; + const uint32_t b2 = (raw2[word] >> shift) & 0xFFu; + output[output_index] = b0 | (b1 << 8) | (b2 << 16) | (high << 24); + } + (void)rans_renormalize_checked(valid, state, input, input_begin); +} + +extern "C" __global__ void tile_ans_decode_raw3_ans1_kernel( + const uint16_t* __restrict__ payload, + const uint32_t* __restrict__ offsets, + const uint32_t* __restrict__ states, + const uint32_t* __restrict__ decode_tables, + const uint8_t* __restrict__ lane_modes, + uint8_t* __restrict__ output, + int64_t num_elements, + int tile_elements, + int num_lanes, + int num_tiles) { + const int warp_in_block = threadIdx.x >> 5; + const int lane = threadIdx.x & 31; + const int warps_per_block = blockDim.x >> 5; + const int tile = blockIdx.x * warps_per_block + warp_in_block; + + const uint32_t* table = decode_tables + 3 * kTableSize; + if (tile >= num_tiles || num_lanes != 4) { + return; + } + const int64_t tile_begin = static_cast(tile) * tile_elements; + const int tile_count = static_cast( + (num_elements - tile_begin < tile_elements) + ? (num_elements - tile_begin) + : tile_elements); + + const uint32_t expected_raw_words = (tile_count + 1) >> 1; + if (offsets[tile + 1] - offsets[tile] != expected_raw_words || + offsets[num_tiles + tile + 1] - offsets[num_tiles + tile] != expected_raw_words || + offsets[2 * num_tiles + tile + 1] - offsets[2 * num_tiles + tile] != expected_raw_words) { + return; + } + const uint16_t* raw0 = payload + offsets[tile]; + const uint16_t* raw1 = payload + offsets[num_tiles + tile]; + const uint16_t* raw2 = payload + offsets[2 * num_tiles + tile]; + const int ans_stream = 3 * num_tiles + tile; + const uint16_t* input_begin = payload + offsets[ans_stream]; + const uint16_t* input = payload + offsets[ans_stream + 1]; + uint32_t state = states[tile * kNumStates + lane]; + uint32_t* output32 = reinterpret_cast(output); + + const int remainder = tile_count & (kNumStates - 1); + int output_offset = tile_count - remainder; + if (remainder) { + decode_group_raw3_ans1( + lane < remainder, state, table, input, input_begin, + raw0, raw1, raw2, output32, + tile_begin + output_offset + lane, output_offset + lane); + } + while (output_offset > 0) { + output_offset -= kNumStates; + decode_group_raw3_ans1( + true, state, table, input, input_begin, + raw0, raw1, raw2, output32, + tile_begin + output_offset + lane, output_offset + lane); + } +} + +extern "C" __global__ void tile_ans_decode_all_raw_kernel( + const uint16_t* __restrict__ payload, + const uint32_t* __restrict__ offsets, + const uint32_t* __restrict__ states, + const uint32_t* __restrict__ decode_tables, + const uint8_t* __restrict__ lane_modes, + uint8_t* __restrict__ output, + int64_t num_elements, + int tile_elements, + int num_lanes, + int num_tiles) { + const int tile = blockIdx.x; + if (tile >= num_tiles) { + return; + } + const int64_t tile_begin = static_cast(tile) * tile_elements; + const int tile_count = static_cast( + (num_elements - tile_begin < tile_elements) + ? (num_elements - tile_begin) + : tile_elements); + + for (int element = threadIdx.x; element < tile_count; element += blockDim.x) { + uint64_t value = 0; +#pragma unroll + for (int byte_lane = 0; byte_lane < num_lanes; ++byte_lane) { + const int stream = byte_lane * num_tiles + tile; + const uint32_t expected_words = (tile_count + 1) >> 1; + if (offsets[stream + 1] - offsets[stream] != expected_words) { + return; + } + const uint16_t packed = payload[offsets[stream] + (element >> 1)]; + const uint32_t byte = (packed >> ((element & 1) * 8)) & 0xFFu; + value |= static_cast(byte) << (byte_lane * 8); + } + const int64_t output_index = tile_begin + element; + if (num_lanes == 8) { + reinterpret_cast(output)[output_index] = value; + } else if (num_lanes == 4) { + reinterpret_cast(output)[output_index] = static_cast(value); + } else if (num_lanes == 2) { + reinterpret_cast(output)[output_index] = static_cast(value); + } else { + output[output_index] = static_cast(value); + } + } +} + +extern "C" __global__ void tile_ans_decode_kernel( + const uint16_t* __restrict__ payload, + const uint32_t* __restrict__ offsets, + const uint32_t* __restrict__ states, + const uint32_t* __restrict__ decode_tables, + const uint8_t* __restrict__ lane_modes, + uint8_t* __restrict__ output, + int64_t num_elements, + int tile_elements, + int num_lanes, + int num_tiles) { + const int warp_in_block = threadIdx.x >> 5; + const int lane = threadIdx.x & 31; + const int warps_per_block = blockDim.x >> 5; + const int tile = blockIdx.x * warps_per_block + warp_in_block; + const int byte_lane = blockIdx.y; + + const bool raw_lane = lane_modes[byte_lane] != 0; + extern __shared__ uint32_t shared_decode_tables[]; + uint32_t* table = shared_decode_tables; + const uint32_t* source_table = decode_tables + byte_lane * kTableSize; + if (!raw_lane) { + for (int index = threadIdx.x; index < static_cast(kTableSize); index += blockDim.x) { + table[index] = source_table[index]; + } + } + __syncthreads(); + + if (tile >= num_tiles || byte_lane >= num_lanes) { + return; + } + const int stream = byte_lane * num_tiles + tile; + const int64_t tile_begin = static_cast(tile) * tile_elements; + const int tile_count = static_cast( + (num_elements - tile_begin < tile_elements) + ? (num_elements - tile_begin) + : tile_elements); + + const uint32_t begin_word = offsets[stream]; + const uint32_t end_word = offsets[stream + 1]; + const uint16_t* input_begin = payload + begin_word; + const uint16_t* input = payload + end_word; + + if (raw_lane) { + const int raw_words = (tile_count + 1) >> 1; + if (end_word - begin_word != static_cast(raw_words)) { + return; + } + for (int word = lane; word < raw_words; word += kNumStates) { + const uint32_t packed = input_begin[word]; + const int first = word << 1; + output[(tile_begin + first) * num_lanes + byte_lane] = + static_cast(packed); + if (first + 1 < tile_count) { + output[(tile_begin + first + 1) * num_lanes + byte_lane] = + static_cast(packed >> 8); + } + } + return; + } + + int ans_lane = 0; + for (int prior_lane = 0; prior_lane < byte_lane; ++prior_lane) { + ans_lane += lane_modes[prior_lane] == 0; + } + const int ans_stream = ans_lane * num_tiles + tile; + uint32_t state = states[ans_stream * kNumStates + lane]; + const int remainder = tile_count & (kNumStates - 1); + int output_offset = tile_count - remainder; + if (remainder) { + const bool valid = lane < remainder; + decode_group( + valid, + state, + table, + input, + input_begin, + output, + tile_begin + output_offset + lane, + num_lanes, + byte_lane); + } + + while (output_offset > 0) { + output_offset -= kNumStates; + decode_group( + true, + state, + table, + input, + input_begin, + output, + tile_begin + output_offset + lane, + num_lanes, + byte_lane); + } +} diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..6c31076 --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,36 @@ +[build-system] +requires = ["setuptools>=77.0.3", "wheel"] +build-backend = "setuptools.build_meta" + +[project] +name = "entropack" +version = "0.1.0" +description = "EntroPack: general-purpose tensor compression for PyTorch with lossless and rate-controlled lossy schemes and GPU encoding and decoding." +readme = "README.md" +requires-python = ">=3.10" +license = "Apache-2.0" +license-files = ["LICENSE"] +authors = [{ name = "DiffSynth-Studio" }] +# Runtime deps. The CUDA lane additionally needs cupy for the running CUDA major version: +# `pip install entropack[cuda13]` (CUDA 13) or `[cuda12]` (CUDA 12). Without cupy +# the package degrades to the pure-torch eager lane instead of failing. +dependencies = [ + "torch>=2.10", + "numpy>=1.23", + "dahuffman>=0.4", +] + +[project.optional-dependencies] +cuda13 = ["cupy-cuda13x>=14"] +cuda12 = ["cupy-cuda12x>=14"] + +[tool.setuptools.packages.find] +include = ["entropack*"] + +[tool.setuptools.package-data] +"entropack.schemes.dfloat11" = ["*.cu"] +"entropack.schemes.tile_ans" = ["*.cu", "*.cuh"] +"entropack.schemes.lattice_rans" = ["*.cu"] + +[tool.pytest.ini_options] +testpaths = ["tests"] From 447fc82ff48207874edb4599f04f21f709c82034 Mon Sep 17 00:00:00 2001 From: mi804 <1576993271@qq.com> Date: Sun, 27 Sep 2026 21:57:10 +0800 Subject: [PATCH 2/2] Refactor compressed tensors and Linear integration --- docs/en/API_Reference/index.md | 29 +-- docs/en/Usage/Tensor-compression.md | 1 + docs/zh/API_Reference/index.md | 25 ++- docs/zh/Usage/Tensor-compression.md | 1 + entropack/compression/api.py | 17 +- entropack/compression/compressed_tensor.py | 165 +++++++++----- entropack/linear/linear.py | 249 ++++++++------------- entropack/schemes/base.py | 11 +- entropack/schemes/lattice_rans/cuda.py | 2 +- 9 files changed, 247 insertions(+), 253 deletions(-) diff --git a/docs/en/API_Reference/index.md b/docs/en/API_Reference/index.md index 6faaf4a..03c0360 100644 --- a/docs/en/API_Reference/index.md +++ b/docs/en/API_Reference/index.md @@ -14,7 +14,7 @@ compress(tensor: torch.Tensor, config: CompressionConfig) -> CompressedTensor ``` Compresses `tensor` using the scheme selected by `config`. Returns a `CompressedTensor` -that decompresses to the same shape and dtype as the input. +that decompresses to the same shape and dtype as the input by default. Input requirements depend on the scheme: DFloat11 accepts BF16 tensors, Tile-ANS supports multiple dtypes, and lattice quantization requires @@ -35,9 +35,9 @@ failures raise errors. decompress(compressed: CompressedTensor, config: CompressionConfig) -> torch.Tensor ``` -Returns a tensor with the input's original shape and dtype. Lossless schemes restore its values -bit for bit; lossy schemes return an approximate reconstruction. The result is on the original -input device unless the container has been moved, for example with `compressed.to(device)`. +Returns a tensor with `compressed.shape`, `compressed.dtype`, and `compressed.device`. +Without an output dtype conversion, lossless schemes restore input values bit for bit; +lossy schemes return an approximate reconstruction. | Parameter | Meaning | | --- | --- | @@ -49,15 +49,17 @@ or requantize the tensor. ## CompressedTensor -Holds the compressed representation of one tensor. Usually returned by `compress` or restored -from a checkpoint with `from_state_dict`. +A `torch.Tensor` subclass holding the compressed representation of one tensor. Usually returned +by `compress` or restored from a checkpoint with `from_state_dict`. Use `decompress` before +performing numerical operations. ### Common properties | Property | Type | Meaning | | --- | --- | --- | -| `shape` | `tuple[int, ...]` | Original tensor shape | +| `shape` | `torch.Size` | Original tensor shape | | `dtype` | `torch.dtype` | Reconstructed tensor dtype | +| `encoded_dtype` | `torch.dtype` | Dtype used for encoding | | `compress_method` | `str` | Scheme actually used by the container | | `lossless` | `bool` | Whether the scheme is lossless | | `actual_bpp` | `float` | Stored bits per element, including metadata | @@ -69,7 +71,7 @@ This measures the compressed representation, not checkpoint file size or runtime | Method | Returns | Meaning | | --- | --- | --- | -| `to(device)` | `CompressedTensor` | Returns a new container on the specified device without changing its dtype or values; accepts only a device argument | +| `to(...)` | `CompressedTensor` | Changes device or output dtype without recompression; `copy=True` copies storage | | `storage_nbytes(include_header=True)` | `int` | Total compressed size in bytes; `include_header=False` excludes the container header | | `state_dict(prefix="")` | `dict[str, torch.Tensor]` | Exports the compressed tensor for saving | | `CompressedTensor.from_state_dict(state, prefix="")` | `CompressedTensor` | Restores the container from that dictionary without recompression | @@ -77,6 +79,7 @@ This measures the compressed representation, not checkpoint file size or runtime Use the same `prefix` when saving and restoring. The dictionary can be saved with `torch.save` and loaded with `torch.load(..., weights_only=True)`. Set `map_location` to choose the device on which it will be restored. +Loading restores the encoded dtype. Call `.to(dtype=...)` afterwards if a different output dtype is needed. ## CompressedLinear @@ -113,15 +116,15 @@ or load a checkpoint before running inference. | Interface | Returns | Meaning | | --- | --- | --- | -| `compress_weight(weight)` | `None` | Compresses and replaces the weights; requires shape `(out_features, in_features)` | -| `dequantize(device=None)` | `torch.Tensor` | Returns dense weights with `container_dtype`, on the layer's device unless `device` is specified | +| `compress_weight(weight)` | `None` | Initializes compressed weights with shape `(out_features, in_features)` | +| `dequantize(device=None)` | `torch.Tensor` | Returns dense weights with `weight.dtype`, on the layer's device unless `device` is specified | | `forward(x)` | `torch.Tensor` | Applies the layer to `x` of shape `(..., in_features)` and returns shape `(..., out_features)` | -| `compressed_weight` | `CompressedTensor` | Compressed weight container held by the layer | +| `weight` | `CompressedTensor` | Frozen compressed weight parameter held by the layer | | `container_dtype` | `torch.dtype` | Dtype of the weights or quantized codes in the compressed container | | `stored_nbytes` | `int` | Weight storage bytes, including metadata and low-precision quantization scales, excluding bias | | `compressed_bits` | `float` | `8 * stored_nbytes / (in_features * out_features)` | -Invoke the forward operation as `layer(x)`. The dense `.weight` is `None`; use `dequantize()` +Invoke the forward operation as `layer(x)`. `.weight` is a compressed tensor; use `dequantize()` when numerical weights are needed. Use standard `state_dict()` / `load_state_dict()` calls to save and restore layer state. @@ -151,7 +154,7 @@ per-row quantization scales needed to reconstruct weights. | Method | Returns | Meaning | | --- | --- | --- | | `codes(device=None)` | FP8 or INT8 tensor | Restores quantized codes without applying row scales | -| `dequantize(device=None)` | FP32 tensor | Multiplies restored codes by their row scales to obtain numerical weights | +| `dequantize(device=None)` | `torch.Tensor` | Returns the layer's initialization dtype, which `from_linear` defaults to the source weight dtype | Both methods return tensors on the layer's device unless `device` is specified. Lossy compression may change the codes from their initial quantized values. diff --git a/docs/en/Usage/Tensor-compression.md b/docs/en/Usage/Tensor-compression.md index 21a0a05..4fc6a35 100644 --- a/docs/en/Usage/Tensor-compression.md +++ b/docs/en/Usage/Tensor-compression.md @@ -73,6 +73,7 @@ print(restored.device, restored.dtype) ``` Decompressing the object returned by `compressed.to(device)` restores the tensor on that device. +`compressed.to(dtype=...)` changes the decompressed output dtype without recompression. ## Save and load diff --git a/docs/zh/API_Reference/index.md b/docs/zh/API_Reference/index.md index 99b5b3d..4424166 100644 --- a/docs/zh/API_Reference/index.md +++ b/docs/zh/API_Reference/index.md @@ -13,7 +13,7 @@ compress(tensor: torch.Tensor, config: CompressionConfig) -> CompressedTensor ``` 按 `config` 指定的方案压缩 `tensor`,返回 `CompressedTensor`。 -解压后的张量与输入具有相同的形状和数据类型。 +解压后的张量默认与输入具有相同的形状和数据类型。 输入要求由方案决定:DFloat11 接受 BF16 张量,Tile-ANS 支持多种数据类型,格量化要求非空、有限值组成的二维张量。 @@ -31,8 +31,8 @@ compress(tensor: torch.Tensor, config: CompressionConfig) -> CompressedTensor decompress(compressed: CompressedTensor, config: CompressionConfig) -> torch.Tensor ``` -返回与原始输入形状和数据类型相同的张量。无损方案逐位恢复输入值,有损方案返回近似重建。 -结果默认位于原始输入设备;通过 `compressed.to(device)` 等方式迁移容器后,结果位于迁移后的设备。 +返回形状为 `compressed.shape`、数据类型为 `compressed.dtype`、设备为 `compressed.device` 的张量。 +未转换输出类型时,无损方案逐位恢复输入值,有损方案返回近似重建。 | 参数 | 含义 | | --- | --- | @@ -43,14 +43,16 @@ decompress(compressed: CompressedTensor, config: CompressionConfig) -> torch.Ten ## CompressedTensor -保存一个张量的压缩表示。通常由 `compress` 返回,或由 `from_state_dict` 从检查点恢复。 +保存一个张量压缩表示的 `torch.Tensor` 子类。通常由 `compress` 返回,或由 `from_state_dict` 从检查点恢复。 +进行数值计算前,需先使用 `decompress` 解压。 ### 常用属性 | 属性 | 类型 | 含义 | | --- | --- | --- | -| `shape` | `tuple[int, ...]` | 原始张量的形状 | +| `shape` | `torch.Size` | 原始张量的形状 | | `dtype` | `torch.dtype` | 解压后的数据类型 | +| `encoded_dtype` | `torch.dtype` | 编码时的数据类型 | | `compress_method` | `str` | 容器实际使用的压缩方案 | | `lossless` | `bool` | 该方案是否无损 | | `actual_bpp` | `float` | 每元素实际存储比特数,包含元数据 | @@ -62,13 +64,14 @@ decompress(compressed: CompressedTensor, config: CompressionConfig) -> torch.Ten | 方法 | 返回值 | 含义 | | --- | --- | --- | -| `to(device)` | `CompressedTensor` | 返回位于指定设备的新容器,不改变数据类型或数值;仅接受设备参数 | +| `to(...)` | `CompressedTensor` | 改变设备或解压输出类型,不重新压缩;`copy=True` 可复制存储 | | `storage_nbytes(include_header=True)` | `int` | 压缩结果的总字节数,`include_header=False` 时不计容器头部 | | `state_dict(prefix="")` | `dict[str, torch.Tensor]` | 将压缩张量导出为可保存的字典 | | `CompressedTensor.from_state_dict(state, prefix="")` | `CompressedTensor` | 从上述字典恢复容器,不重新压缩 | 保存与加载的 `prefix` 必须一致。可用 `torch.save` 保存字典,并用 `torch.load(..., weights_only=True)` 加载,通过 `map_location` 指定恢复后的设备。 +加载后恢复编码时的数据类型;需要其他输出类型时,再调用 `.to(dtype=...)`。 ## CompressedLinear @@ -101,15 +104,15 @@ CompressedLinear.from_linear(linear, **kwargs) -> CompressedLinear | 接口 | 返回值 | 含义 | | --- | --- | --- | -| `compress_weight(weight)` | `None` | 压缩并替换层内权重,要求形状为 `(out_features, in_features)` | -| `dequantize(device=None)` | `torch.Tensor` | 返回数据类型为 `container_dtype` 的稠密权重;未指定 `device` 时位于层所在设备 | +| `compress_weight(weight)` | `None` | 初始化层内压缩权重,形状应为 `(out_features, in_features)` | +| `dequantize(device=None)` | `torch.Tensor` | 返回数据类型为 `weight.dtype` 的稠密权重;未指定 `device` 时位于层所在设备 | | `forward(x)` | `torch.Tensor` | 对形状为 `(..., in_features)` 的输入 `x` 执行线性运算,返回形状为 `(..., out_features)` 的张量 | -| `compressed_weight` | `CompressedTensor` | 层持有的压缩权重容器 | +| `weight` | `CompressedTensor` | 层持有的冻结压缩权重参数 | | `container_dtype` | `torch.dtype` | 压缩容器中权重或量化码的数据类型 | | `stored_nbytes` | `int` | 权重存储字节数,含元数据和低精度层的量化尺度,不含偏置 | | `compressed_bits` | `float` | `8 * stored_nbytes / (in_features * out_features)` | -通过 `layer(x)` 调用前向运算。层的稠密 `.weight` 为 `None`,需要数值权重时使用 `dequantize()`。 +通过 `layer(x)` 调用前向运算。`.weight` 为压缩张量,需要数值权重时使用 `dequantize()`。 使用标准 `state_dict()` / `load_state_dict()` 保存与恢复层状态。 加载前须创建相同层类型、形状、容器数据类型和压缩方案的层,Config 对象本身不会保存在检查点中。 @@ -134,7 +137,7 @@ CompressedLinear.from_linear(linear, **kwargs) -> CompressedLinear | 方法 | 返回值 | 含义 | | --- | --- | --- | | `codes(device=None)` | FP8 或 INT8 张量 | 恢复量化码,尚未乘回行尺度 | -| `dequantize(device=None)` | FP32 张量 | 将恢复的量化码乘回行尺度,得到数值权重 | +| `dequantize(device=None)` | `torch.Tensor` | 返回层初始化时的数据类型,`from_linear` 默认沿用原始权重类型 | 两种方法未指定 `device` 时,返回张量均位于层所在设备。有损压缩后的量化码可能与初始量化结果不同。 diff --git a/docs/zh/Usage/Tensor-compression.md b/docs/zh/Usage/Tensor-compression.md index ed24044..4963fd0 100644 --- a/docs/zh/Usage/Tensor-compression.md +++ b/docs/zh/Usage/Tensor-compression.md @@ -72,6 +72,7 @@ print(restored.device, restored.dtype) ``` 对 `compressed.to(device)` 返回的压缩张量解压,得到的张量也位于该设备上。 +`compressed.to(dtype=...)` 可改变解压输出类型,无需重新压缩。 ## 保存与加载 diff --git a/entropack/compression/api.py b/entropack/compression/api.py index 42a459f..7de5af3 100644 --- a/entropack/compression/api.py +++ b/entropack/compression/api.py @@ -12,7 +12,7 @@ class CompressionFallbackWarning(RuntimeWarning): - """The category :func:`compress` warns under when it stored a tensor uncompressed.""" + """Warning emitted when compression falls back to storing the tensor uncompressed.""" def compress(tensor: torch.Tensor, config: CompressionConfig) -> CompressedTensor: @@ -27,7 +27,10 @@ def compress(tensor: torch.Tensor, config: CompressionConfig) -> CompressedTenso Encoding failures can return an uncompressed ``raw`` container with a :class:`CompressionFallbackWarning`. Its header records the requested scheme and - failure reason. Invalid configurations and backend dispatch failures raise errors.""" + failure reason. Invalid configurations and backend dispatch failures raise errors. + """ + if isinstance(tensor, CompressedTensor): + raise TypeError("compress expects an uncompressed tensor; decompress the container before recompressing") scheme = get_scheme(name_for_config(config)) validate_config(config) try: @@ -40,9 +43,7 @@ def compress(tensor: torch.Tensor, config: CompressionConfig) -> CompressedTenso scheme.name, tuple(tensor.shape), tensor.dtype, error, ) return _compress_raw(tensor, scheme.name, error) - return CompressedTensor( - header={"compress_method": scheme.name}, buffers=packed, shape=tuple(tensor.shape), dtype=tensor.dtype, - ) + return CompressedTensor(header={"compress_method": scheme.name}, buffers=packed, shape=tuple(tensor.shape), dtype=tensor.dtype) def _compress_raw(tensor: torch.Tensor, requested: str, error: Exception) -> CompressedTensor: @@ -68,6 +69,8 @@ def decompress(compressed: CompressedTensor, config: CompressionConfig) -> torch the stored data. Returns: - A tensor with the container's shape and dtype, on the device holding its buffers.""" + A tensor with the container's shape and dtype, on the device holding its buffers. + """ validate_config(config) - return compressed.scheme.decode(compressed.buffers, shape=compressed.shape, dtype=compressed.dtype, config=config) + restored = compressed.scheme.decode(compressed.buffers, shape=compressed.shape, dtype=compressed.encoded_dtype, config=config) + return restored.to(compressed.dtype) diff --git a/entropack/compression/compressed_tensor.py b/entropack/compression/compressed_tensor.py index b706262..3806a49 100644 --- a/entropack/compression/compressed_tensor.py +++ b/entropack/compression/compressed_tensor.py @@ -2,10 +2,10 @@ import json import math from collections.abc import Mapping -from dataclasses import dataclass from typing import Any import torch +from torch.utils._python_dispatch import return_and_correct_aliasing from ..registry import get_scheme from ..schemes import Scheme @@ -20,33 +20,103 @@ def _parse_dtype(name: str) -> torch.dtype: return dtype -@dataclass -class CompressedTensor: - """Compressed data and metadata for one tensor. +class CompressedTensor(torch.Tensor): + """Frozen compressed tensor with a logical output dtype. - The container records the scheme, input shape, and dtype. Use - :func:`entropack.decompress` with the matching scheme's configuration to restore values. - Use :meth:`state_dict` and :meth:`from_state_dict` for tensor-only checkpoint entries - compatible with ``torch.load(..., weights_only=True)``. + ``encoded_dtype`` records the codec's input type. ``to(dtype=...)`` changes + the output dtype without casting packed buffers or re-encoding values. + """ - Construction checks the scheme, supported dtype, and buffer names.""" - - #: ``{"compress_method": }`` for a coded container, plus ``requested`` and ``reason`` - #: when a codec refused the tensor and it was stored verbatim. header: dict[str, Any] - #: The scheme's buffers, named exactly as its ``buffer_names`` declares. buffers: dict[str, torch.Tensor] - #: Shape of the tensor that comes back, before any padding the encode applied. - shape: tuple[int, ...] - #: Its dtype. No scheme changes it: a container holds the format it was given. - dtype: torch.dtype - - def __post_init__(self): - self.header = copy.deepcopy(self.header) - self.buffers = dict(self.buffers) - self.shape = tuple(self.shape) + + @staticmethod + def __new__(cls, header, buffers, shape, dtype, *, encoded_dtype=None): + device = next(iter(buffers.values())).device + return torch.Tensor._make_wrapper_subclass(cls, tuple(shape), dtype=dtype, device=device, requires_grad=False) + + def __init__(self, header, buffers, shape, dtype, *, encoded_dtype=None): + self.header = copy.deepcopy(header) + self.buffers = dict(buffers) + self.encoded_dtype = dtype if encoded_dtype is None else encoded_dtype self.validate() + def __repr__(self) -> str: + return f"{type(self).__name__}(shape={tuple(self.shape)}, dtype={self.dtype}, device={self.device}, scheme={self.compress_method!r})" + + def __tensor_flatten__(self): + names = tuple(self.buffers) + for name, value in self.buffers.items(): + setattr(self, "_packed_" + name, value) + metadata = (names, copy.deepcopy(self.header), tuple(self.shape), self.dtype, self.encoded_dtype) + return ["_packed_" + name for name in names], metadata + + @classmethod + def __tensor_unflatten__(cls, inner_tensors, metadata, outer_size, outer_stride): + names, header, shape, dtype, encoded_dtype = metadata + return cls( + header=header, buffers={name: inner_tensors["_packed_" + name] for name in names}, + shape=shape, dtype=dtype, encoded_dtype=encoded_dtype, + ) + + def _map_buffers(self, fn, *, dtype=None) -> "CompressedTensor": + return type(self)( + header=self.header, buffers={name: fn(value) for name, value in self.buffers.items()}, + shape=self.shape, dtype=self.dtype if dtype is None else dtype, encoded_dtype=self.encoded_dtype, + ) + + def to(self, *args, copy: bool = False, **kwargs) -> "CompressedTensor": + """Move buffers or change the logical dtype, with an optional keyword-only copy.""" + device, dtype, non_blocking, memory_format = torch._C._nn._parse_to(*args, **kwargs) + result = super().to(device=device, dtype=dtype, non_blocking=non_blocking, copy=copy, memory_format=memory_format) + return result.clone() if copy and result.device == self.device else result + + def __copy__(self) -> "CompressedTensor": + result = self._map_buffers(lambda value: value) + if getattr(self, "_is_param", False): + result._is_param = True + return result + + def __deepcopy__(self, memo) -> "CompressedTensor": + if id(self) in memo: + return memo[id(self)] + result = self._map_buffers(lambda value: copy.deepcopy(value, memo)) + if getattr(self, "_is_param", False): + result._is_param = True + memo[id(self)] = result + return result + + def requires_grad_(self, requires_grad: bool = False): + """Compressed values are frozen; gradients may flow through their consumers.""" + if requires_grad: + raise RuntimeError("CompressedTensor cannot require gradients; decompress it to train dense values") + return torch.Tensor.requires_grad_(self, False) + + @classmethod + def __torch_dispatch__(cls, func, types, args=(), kwargs=None): + kwargs = kwargs or {} + source = args[0] if args else None + aten = torch.ops.aten + + if func in (aten.detach.default, aten.alias.default): + result = source._map_buffers(torch.detach) + return return_and_correct_aliasing(func, args, kwargs, result) + + if func in (aten.clone.default, aten._to_copy.default): + if kwargs.get("memory_format") not in (None, torch.preserve_format, torch.contiguous_format): + raise ValueError("CompressedTensor does not support that memory format") + if func == aten.clone.default: + return source._map_buffers(torch.clone) + if kwargs.get("layout", torch.strided) not in (None, torch.strided): + raise ValueError("CompressedTensor only supports strided storage") + options = {name: value for name, value in kwargs.items() if name in ("device", "non_blocking")} + return source._map_buffers(lambda value: value.to(**options), dtype=kwargs.get("dtype")) + + result = func.decompose(*args, **kwargs) + if result is not NotImplemented: + return result + raise NotImplementedError(f"CompressedTensor does not implement {func}; decompress it before numerical operations") + @property def compress_method(self) -> str: """The scheme name the header carries.""" @@ -59,7 +129,7 @@ def scheme(self) -> Scheme: @property def lossless(self) -> bool: - """Whether the scheme that wrote this container reconstructs its input exactly.""" + """Whether the encoding scheme is lossless, before any output dtype conversion.""" return self.scheme.lossless @property @@ -68,43 +138,21 @@ def actual_bpp(self) -> float: return self.storage_nbytes() * 8 / math.prod(self.shape) def validate(self) -> None: - """Check the scheme, supported dtype, and required buffer names. - - This check does not inspect buffer contents.""" + """Check the scheme, dtype, and buffer names without inspecting contents.""" + if any(isinstance(value, CompressedTensor) for value in self.buffers.values()): + raise TypeError("CompressedTensor buffers cannot contain another CompressedTensor") scheme = self.scheme - if not scheme.supports(self.dtype): - raise ValueError(f"'{scheme.name}' does not support format {self.dtype}") + if not scheme.supports(self.encoded_dtype): + raise ValueError(f"'{scheme.name}' does not support format {self.encoded_dtype}") if set(self.buffers) != set(scheme.buffer_names): raise ValueError( f"CompressedTensor buffers for {self.compress_method} must be " f"{list(scheme.buffer_names)}, got {sorted(self.buffers)}" ) - def to(self, device: str | torch.device) -> "CompressedTensor": - """A copy of this container with every buffer on ``device``.""" - target = torch.device(device) - return type(self)( - header=self.header, buffers={name: value.to(device=target) for name, value in self.buffers.items()}, - shape=self.shape, dtype=self.dtype, - ) - - def to_dict(self) -> dict[str, Any]: - """A plain-dict view of the four fields, for a caller that serializes them itself.""" - self.validate() - return {"header": copy.deepcopy(self.header), "buffers": dict(self.buffers), "shape": self.shape, "dtype": self.dtype} - - @classmethod - def from_dict(cls, data: Mapping[str, Any]) -> "CompressedTensor": - """Rebuild the container :meth:`to_dict` produced.""" - required = {"header", "buffers", "shape", "dtype"} - missing = required - data.keys() - if missing: - raise ValueError(f"CompressedTensor data is missing fields: {sorted(missing)}") - return cls(header=data["header"], buffers=data["buffers"], shape=data["shape"], dtype=data["dtype"]) - def _serialized_header_tensor(self) -> torch.Tensor: metadata = { - "compress_method": self.compress_method, "dtype": str(self.dtype).removeprefix("torch."), + "compress_method": self.compress_method, "dtype": str(self.encoded_dtype).removeprefix("torch."), "shape": list(self.shape), "buffer_names": list(self.buffers), "header": self.header, } encoded = json.dumps(metadata, sort_keys=True, separators=(",", ":"), allow_nan=False).encode("utf-8") @@ -112,25 +160,21 @@ def _serialized_header_tensor(self) -> torch.Tensor: def storage_nbytes(self, include_header: bool = True) -> int: """Bytes the buffers occupy, plus the serialized header unless ``include_header`` is false.""" - self.validate() total = sum(value.numel() * value.element_size() for value in self.buffers.values()) if include_header: - header = self._serialized_header_tensor() - total += header.numel() * header.element_size() + total += self._serialized_header_tensor().numel() return total def state_dict(self, prefix: str = "") -> dict[str, torch.Tensor]: """The container as flat tensors under ``prefix``: one 1D uint8 header, then one entry per buffer.""" self.validate() state = {f"{prefix}header": self._serialized_header_tensor()} - state.update({f"{prefix}buffers.{name}": value for name, value in self.buffers.items()}) + state.update({f"{prefix}buffers.{name}": value.detach() for name, value in self.buffers.items()}) return state @classmethod def from_state_dict(cls, state: Mapping[str, torch.Tensor], prefix: str = "") -> "CompressedTensor": - """Restore a container from the tensor entries produced by :meth:`state_dict`. - - Metadata is stored as JSON bytes in a uint8 tensor.""" + """Restore a container in its encoded dtype from a tensor-only checkpoint.""" header_key = f"{prefix}header" if header_key not in state: raise ValueError(f"CompressedTensor state is missing '{header_key}'") @@ -158,7 +202,4 @@ def from_state_dict(cls, state: Mapping[str, torch.Tensor], prefix: str = "") -> raise ValueError(f"CompressedTensor state is missing '{key}'") buffers[name] = state[key] - return cls( - header=metadata["header"], buffers=buffers, shape=tuple(metadata["shape"]), - dtype=_parse_dtype(metadata["dtype"]), - ) + return cls(header=metadata["header"], buffers=buffers, shape=tuple(metadata["shape"]), dtype=_parse_dtype(metadata["dtype"])) diff --git a/entropack/linear/linear.py b/entropack/linear/linear.py index 18e645a..d83f1f0 100644 --- a/entropack/linear/linear.py +++ b/entropack/linear/linear.py @@ -1,7 +1,6 @@ -import copy import dataclasses from numbers import Real -from typing import Any, ClassVar +from typing import ClassVar import torch from torch.nn import functional as F @@ -18,9 +17,9 @@ class CompressedLinear(torch.nn.Linear): """A linear layer with compressed weights, reconstructed during each forward call. - The layer registers compressed buffers for ``state_dict``, device transfers, and - ``deepcopy``. Its dense ``weight`` parameter is ``None``. Each forward reconstructs a - temporary weight, casts it to the activation dtype, and applies ``F.linear``. + The frozen ``weight`` is a ``CompressedTensor`` parameter. Each forward + reconstructs the stored weight, casts it to the activation dtype, and applies + ``F.linear``. Checkpoints retain the original packed representation. CUDA and CuPy are required. Args: @@ -29,11 +28,11 @@ class CompressedLinear(torch.nn.Linear): bias: whether to keep a bias. The bias is not compressed. config: compression configuration. ``None`` selects a lossless scheme by dtype. device: device for the bias. Compressed buffers retain the source weight's device. - dtype: container dtype, which must be supported by the selected scheme.""" + dtype: container dtype, which must be supported by the selected scheme. + """ state_buffer_names: ClassVar[tuple[str, ...]] = () - state_prefix: ClassVar[str] = "_entropack." - accepts_raw_container: ClassVar[bool] = False + state_prefix: ClassVar[str] = "weight._entropack." def __init__( self, in_features: int, out_features: int, bias: bool = True, *, config=None, @@ -45,22 +44,14 @@ def __init__( if bias: self.bias = torch.nn.Parameter(torch.zeros(out_features, dtype=dtype, device=device), requires_grad=False) self.config = config - self._container_dtype = require_dtype(dtype) + self._init_dtype = require_dtype(dtype) if config is not None: scheme = get_scheme(name_for_config(config)) if not scheme.supports(self.container_dtype): raise ValueError(f"'{scheme.name}' does not support format {self.container_dtype}") - self._compressed = None for name in self.state_buffer_names: self.register_buffer(name, None, persistent=False) - @property - def scheme_name(self) -> str: - """The scheme the stored container uses, else the one the config names, else ``"auto"``.""" - if self._compressed is not None: - return self._compressed.compress_method - return "auto" if self.config is None else name_for_config(self.config) - @property def _encode_config(self): if self.config is None: @@ -70,39 +61,26 @@ def _encode_config(self): @property def _decode_config(self): if self.config is None: - return self._compressed.scheme.make_config({"execution_backend": _BACKEND}) + return self.weight.scheme.make_config({"execution_backend": _BACKEND}) return dataclasses.replace(self.config, execution_backend=_BACKEND) @property def container_dtype(self) -> torch.dtype: - """The format the weight is stored in, which the scheme has to serve.""" - return self._container_dtype - - @property - def buffer_names(self) -> tuple[str, ...]: - """The stored container's buffer names; empty while the layer holds no weight.""" - return () if self._compressed is None else tuple(self._compressed.scheme.buffer_names) + """The configured compression dtype, unchanged by layer dtype casts.""" + return self._init_dtype @property def qweight(self) -> torch.Tensor: - """One tensor standing in for the weight, for a caller that must move or measure it.""" - for name in self.buffer_names: - buffer = self._buffers.get(name) - if buffer is not None: - return buffer + """A packed weight buffer exposed for quantization-framework compatibility.""" + if self.weight is not None: + for name in self.weight.scheme.buffer_names: + return self.weight.buffers[name] return self.bias - @property - def compressed_weight(self) -> CompressedTensor: - """The stored container.""" - if self._compressed is None: - raise RuntimeError(f"{type(self).__name__} has no compressed weight; load one or call compress_weight") - return self._compressed - @property def stored_nbytes(self) -> int: - """Bytes this layer's weight occupies: the container, its serialized header, and any W8A8 scale.""" - total = self.compressed_weight.storage_nbytes() + """Stored weight bytes, including the serialized header and any W8A8 scales.""" + total = self.weight.storage_nbytes() for name in self.state_buffer_names: buffer = self._buffers.get(name) if buffer is not None: @@ -111,73 +89,28 @@ def stored_nbytes(self) -> int: @property def compressed_bits(self) -> float: - """Bits per element of the source weight's shape, which is what a network rate is built from.""" - rows, cols = self.compressed_weight.shape + """Stored bits per weight element, including metadata.""" + rows, cols = self.weight.shape return self.stored_nbytes * 8 / (rows * cols) - @property - def _held_buffers(self) -> tuple[str, ...]: - return self.buffer_names + self.state_buffer_names - - def _holds(self, method: str) -> bool: - return self.config is None or method == self.scheme_name or ( - self.accepts_raw_container and method == "raw" - ) - def compress_weight(self, weight: torch.Tensor) -> None: - """Compress ``weight`` and store it, replacing whatever the layer held.""" - self._prepare(weight.detach()) + """Initialize the layer with compressed ``weight``.""" + self.weight = torch.nn.Parameter(self._compress_tensor(weight.detach()), requires_grad=False) - def _prepare(self, weight: torch.Tensor) -> None: - self.set_compressed(self._container_for(weight)) - - def _container_for(self, tensor: torch.Tensor) -> CompressedTensor: + def _compress_tensor(self, tensor: torch.Tensor) -> CompressedTensor: compressed = compress(tensor.to(self.container_dtype), self._encode_config) - if compressed.compress_method == "raw" and not self.accepts_raw_container: + if compressed.compress_method == "raw" and "reason" in compressed.header: raise RuntimeError( - f"{self.scheme_name} cannot store a {tuple(tensor.shape)} {self.container_dtype} weight for " - f"{type(self).__name__}: {compressed.header.get('reason', 'asked to store the weight verbatim')}" + f"{compressed.header['requested']} cannot store a {tuple(tensor.shape)} {self.container_dtype} weight for " + f"{type(self).__name__}: {compressed.header['reason']}" ) return compressed - def set_compressed(self, compressed: CompressedTensor) -> None: - """Adopt a container built elsewhere, such as one read from a checkpoint. - - It has to hold this layer's shape and container format and use the scheme this layer's config - names, so a container written for another layer cannot be loaded into this one by accident. - """ - if not isinstance(compressed, CompressedTensor): - raise TypeError(f"expected an entropack CompressedTensor, got {type(compressed).__name__}") - expected = (self.out_features, self.in_features) - if compressed.shape != expected: - raise ValueError(f"compressed weight shape {compressed.shape} does not match this Linear's {expected}") - if not self._holds(compressed.compress_method): - raise ValueError( - f"compressed weight uses '{compressed.compress_method}', this Linear is '{self.scheme_name}'" - ) - if compressed.dtype != self.container_dtype: - raise ValueError( - f"compressed weight holds {compressed.dtype}, this Linear is configured for {self.container_dtype}" - ) - names = tuple(compressed.scheme.buffer_names) - self._compressed = CompressedTensor( - header=compressed.header, buffers={name: compressed.buffers[name] for name in names}, - shape=compressed.shape, dtype=compressed.dtype, - ) - for name in names: - self.register_buffer(name, self._compressed.buffers[name], persistent=False) - - def _container_on(self, device: str | torch.device | None) -> CompressedTensor: - compressed = self.compressed_weight - if device is None or compressed.buffers[self.buffer_names[0]].device == torch.device(device): - return compressed - return compressed.to(device) - def _reconstruct(self, device: str | torch.device | None = None) -> torch.Tensor: - return decompress(self._container_on(device), self._decode_config) + return decompress(self.weight.to(device=device), self._decode_config) def dequantize(self, device: str | torch.device | None = None) -> torch.Tensor: - """The dense weight, reconstructed on ``device``, or on the container's device by default.""" + """Reconstruct the dense weight on ``device``, defaulting to the weight's device.""" return self._reconstruct(device) @classmethod @@ -197,9 +130,7 @@ def from_linear(cls, linear: torch.nn.Linear, **kwargs) -> "CompressedLinear": if weight is None or weight.device.type == "meta": raise ValueError("cannot compress a Linear whose weight is not materialized") kwargs.setdefault("dtype", weight.dtype) - out = cls( - linear.in_features, linear.out_features, bias=linear.bias is not None, device=weight.device, **kwargs, - ) + out = cls(linear.in_features, linear.out_features, bias=linear.bias is not None, device=weight.device, **kwargs) out.compress_weight(weight.data) if linear.bias is not None: out.bias = torch.nn.Parameter(linear.bias.data.clone(), requires_grad=False) @@ -207,12 +138,8 @@ def from_linear(cls, linear: torch.nn.Linear, **kwargs) -> "CompressedLinear": @property def device(self) -> torch.device: - """The device the layer's buffers are on, or ``meta`` while it holds none.""" - for name in self._held_buffers: - buffer = self._buffers.get(name) - if buffer is not None: - return buffer.device - return self.bias.device if self.bias is not None else torch.device("meta") + """The weight device, or the bias device while the weight is not loaded.""" + return self.weight.device if self.weight is not None else self.bias.device if self.bias is not None else torch.device("meta") def _bias_on(self, device: torch.device) -> torch.Tensor | None: return None if self.bias is None else self.bias.to(device) @@ -222,19 +149,17 @@ def forward(self, x: torch.Tensor) -> torch.Tensor: return F.linear(x, self.dequantize(x.device).to(x.dtype), self._bias_on(x.device)) def extra_repr(self) -> str: - parts = [f"scheme={self.scheme_name}", f"container={str(self.container_dtype).removeprefix('torch.')}"] - if self._compressed is not None: + scheme = "auto" if self.config is None else name_for_config(self.config) + parts = [f"scheme={scheme}", f"container={str(self.container_dtype).removeprefix('torch.')}"] + if self.weight is not None: parts.append(f"bits={self.compressed_bits:.3f}") if self.config is not None: parts.append(f"config={type(self.config).__name__}") return ", ".join(parts) - def _sync_compressed(self) -> None: - if self._compressed is not None: - self._compressed.buffers = {name: self._buffers[name] for name in self.buffer_names} - def _apply(self, fn, recurse=True): - held = {name: self._buffers.pop(name) for name in self._held_buffers if name in self._buffers} + # Preserve auxiliary scale dtypes when Module.to casts the layer. + held = {name: self._buffers.pop(name) for name in self.state_buffer_names if name in self._buffers} try: super()._apply(fn, recurse=recurse) finally: @@ -244,23 +169,16 @@ def _apply(self, fn, recurse=True): continue moved = fn(buffer) self._buffers[name] = moved if moved.dtype == buffer.dtype else buffer.to(device=moved.device) - self._sync_compressed() return self - def __deepcopy__(self, memo: dict[int, Any]) -> "CompressedLinear": - clone = type(self).__new__(type(self)) - memo[id(self)] = clone - for key, value in self.__dict__.items(): - clone.__dict__[key] = copy.deepcopy(value, memo) - clone._sync_compressed() - return clone - def _save_to_state_dict(self, destination: dict, prefix: str, keep_vars: bool) -> None: super()._save_to_state_dict(destination, prefix, keep_vars) - if self._compressed is None: + # Serialize the packed representation instead of the wrapper parameter. + destination.pop(prefix + "weight", None) + if self.weight is None: return container_prefix = prefix + self.state_prefix - written = self._compressed.state_dict(container_prefix) + written = self.weight.state_dict(container_prefix) for name in self.state_buffer_names: buffer = self._buffers[name] if buffer is not None: @@ -268,28 +186,58 @@ def _save_to_state_dict(self, destination: dict, prefix: str, keep_vars: bool) - for key, value in written.items(): destination[key] = value if keep_vars else value.detach() - def _load_from_state_dict( - self, state_dict: dict, prefix: str, local_metadata, strict, missing_keys, unexpected_keys, error_msgs, - ) -> None: + def _load_from_state_dict(self, state_dict: dict, prefix: str, local_metadata, strict, missing_keys, unexpected_keys, error_msgs) -> None: container_prefix = prefix + self.state_prefix header_key = container_prefix + "header" if header_key in state_dict: + consumed = [header_key] try: - self.set_compressed(CompressedTensor.from_state_dict(state_dict, prefix=container_prefix)) + compressed = CompressedTensor.from_state_dict(state_dict, prefix=container_prefix) + consumed.extend(container_prefix + "buffers." + name for name in compressed.buffers) + assign = local_metadata.get("assign_to_params_buffers", False) + + # Copy loading keeps the target device and existing Parameter; assign=True adopts the checkpoint tensors. + if self.weight is not None: + device = self.weight.device + elif self.bias is not None and self.bias.device.type != "meta": + device = self.bias.device + else: + device = compressed.device + if not assign: + compressed = compressed.to(device=device, copy=True) + loaded_weight = torch.nn.Parameter(compressed, requires_grad=False) + if not assign and self.weight is not None: + torch.utils.swap_tensors(self.weight, loaded_weight) + else: + self.weight = loaded_weight + + # Apply the same copy/assign behavior to auxiliary state, such as FP8/INT8 weight scales. for name in self.state_buffer_names: key = container_prefix + name if key not in state_dict: raise ValueError(f"compressed state is missing '{key}'") - self._buffers[name] = state_dict[key] + value = state_dict[key] + if not assign: + if self._buffers[name] is None: + value = value.to(device=device, copy=True) + else: + self._buffers[name].copy_(value) + value = self._buffers[name] + self._buffers[name] = value + consumed.append(key) except (TypeError, ValueError) as error: error_msgs.append(f"{prefix[:-1]}: {error}") - for key in [key for key in state_dict if key.startswith(container_prefix)]: + + # Leave unrecognized keys for the parent's strict checks. + for key in consumed: state_dict.pop(key) elif strict: missing_keys.append(header_key) - super()._load_from_state_dict( - state_dict, prefix, local_metadata, strict, missing_keys, unexpected_keys, error_msgs, - ) + + # Let the parent load bias without expecting a dense weight entry. + weight = self._parameters.pop("weight") + super()._load_from_state_dict(state_dict, prefix, local_metadata, strict, missing_keys, unexpected_keys, error_msgs) + self._parameters["weight"] = weight class _W8A8LinearFunction(torch.autograd.Function): @@ -317,7 +265,6 @@ class QuantizedLinear(CompressedLinear): min_tokens: ClassVar[int] min_capability: ClassVar[tuple[int, int]] state_buffer_names = ("weight_scale",) - accepts_raw_container = True def __init__( self, in_features: int, out_features: int, bias: bool = True, *, config=None, @@ -356,19 +303,20 @@ def _quantize_rows(self, tensor: torch.Tensor) -> tuple[torch.Tensor, torch.Tens scaled = scaled.round() return scaled.clamp(-self.code_max, self.code_max).to(self.code_dtype), scale - def _prepare(self, weight: torch.Tensor) -> None: - codes, scale = self._quantize_rows(weight) - self.set_compressed(self._container_for(codes)) + def compress_weight(self, weight: torch.Tensor) -> None: + """Quantize the source weight, compress its codes, and store the row scales.""" + codes, scale = self._quantize_rows(weight.detach()) + self.weight = torch.nn.Parameter(self._compress_tensor(codes), requires_grad=False) self.weight_scale = scale def codes(self, device: str | torch.device | None = None) -> torch.Tensor: - """The stored codes as a dense tensor. The weight is these times their per-row scale.""" - return self._reconstruct(device) + """Decode the weight codes in ``code_dtype`` for low-precision matrix multiplication.""" + return decompress(self.weight.to(device=device, dtype=self.code_dtype), self._decode_config) def dequantize(self, device: str | torch.device | None = None) -> torch.Tensor: - """The dense weight: :meth:`codes` times the per-row scale the quantization fitted.""" + """Reconstruct dense weights in the dtype used to initialize the layer.""" codes = self.codes(device) - return codes.float() * self.weight_scale.to(codes.device).unsqueeze(1) + return (codes.float() * self.weight_scale.to(codes.device).unsqueeze(1)).to(self._init_dtype) def _require_8bit_gemm(self, device: torch.device) -> None: if device.type != "cuda": @@ -391,15 +339,15 @@ def _quantize_activation(self, flat: torch.Tensor) -> tuple[torch.Tensor, torch. return kernels.quantize_rows(flat, self.code_dtype, self.code_max, self.rounds_to_integer) def _gemm_shapes(self, tokens: int) -> tuple[int, int, int]: - return (max(tokens, self.min_tokens), round_up(self.out_features, self.code_alignment), - round_up(self.in_features, self.code_alignment)) + return (max(tokens, self.min_tokens), round_up(self.out_features, self.code_alignment), round_up(self.in_features, self.code_alignment)) def forward(self, x: torch.Tensor) -> torch.Tensor: """Apply the low-precision linear operation. The layer recovers its FP8 or INT8 codes, quantizes activations per row, and runs the corresponding matrix multiplication. Input gradients use the reconstructed numerical - weight. The compressed base weights remain frozen.""" + weight. The compressed base weights remain frozen. + """ self._require_8bit_gemm(x.device) if x.requires_grad: return _W8A8LinearFunction.apply(x, self) @@ -414,18 +362,14 @@ def _forward8(self, x: torch.Tensor) -> torch.Tensor: rows, outs, cols = self._gemm_shapes(tokens) padded_activation = pad(pad(activation, 0, rows), 1, cols) padded_codes = pad(pad(codes, 0, outs), 1, cols) - out = self._gemm(padded_activation, padded_codes, pad(scale, 0, rows), - pad(weight_scale, 0, outs), x.dtype) + out = self._gemm(padded_activation, padded_codes, pad(scale, 0, rows), pad(weight_scale, 0, outs), x.dtype) return out[:tokens, : self.out_features].reshape(*x.shape[:-1], self.out_features) def _bias_padded(self, device: torch.device, columns: int) -> torch.Tensor | None: bias = self._bias_on(device) return None if bias is None else pad(bias, 0, columns) - def _epilogue( - self, product: torch.Tensor, activation_scale: torch.Tensor, weight_scale: torch.Tensor, - out_dtype: torch.dtype, - ) -> torch.Tensor: + def _epilogue(self, product: torch.Tensor, activation_scale: torch.Tensor, weight_scale: torch.Tensor, out_dtype: torch.dtype) -> torch.Tensor: out = product.to(torch.float32) out.mul_(activation_scale.unsqueeze(1)) bias = self._bias_padded(out.device, out.shape[1]) @@ -447,7 +391,8 @@ class CompressedFP8Linear(QuantizedLinear): With no compression configuration, quantized codes are stored directly. :class:`~entropack.LatticeRANSConfig` additionally compresses them at targets from 1 up to, but excluding, 8 bits per element. Inference decodes the FP8 codes before matrix - multiplication. Stored size also includes metadata and per-row weight scales.""" + multiplication. Stored size also includes metadata and per-row weight scales. + """ code_dtype = torch.float8_e4m3fn code_max = float(torch.finfo(torch.float8_e4m3fn).max) @@ -473,7 +418,8 @@ class CompressedINT8Linear(QuantizedLinear): available and ``torch._int_mm`` otherwise. With no compression configuration, quantized codes are stored directly. :class:`~entropack.LatticeRANSConfig` additionally compresses them at targets from 1 up to, but excluding, 8 bits per element. Stored size also - includes metadata and per-row weight scales.""" + includes metadata and per-row weight scales. + """ code_dtype = torch.int8 code_max = 127.0 @@ -494,6 +440,5 @@ def _gemm(self, activation, codes, activation_scale, weight_scale, out_dtype): return kernels.int8_gemm(activation, codes, activation_scale, weight_scale, bias, out_dtype) return self._epilogue(torch._int_mm(activation, codes.t()), activation_scale, weight_scale, out_dtype) -__all__ = [ - "CompressedFP8Linear", "CompressedINT8Linear", "CompressedLinear", "QuantizedLinear", -] + +__all__ = ["CompressedFP8Linear", "CompressedINT8Linear", "CompressedLinear", "QuantizedLinear"] diff --git a/entropack/schemes/base.py b/entropack/schemes/base.py index 1628d55..39c7bce 100644 --- a/entropack/schemes/base.py +++ b/entropack/schemes/base.py @@ -19,10 +19,7 @@ def buffers_fingerprint(buffers: dict, shape, dtype) -> Any: tuple(shape) if shape is not None else None, dtype, tuple( - ( - name, tensor.data_ptr(), tensor._version, tuple(tensor.shape), tuple(tensor.stride()), tensor.dtype, - tensor.device, - ) + (name, tensor.data_ptr(), tuple(tensor.shape), tuple(tensor.stride()), tensor.dtype, tensor.device) for name, tensor in sorted(buffers.items()) ), ) @@ -34,10 +31,10 @@ def packed_buffers(packed: dict, kind: type) -> Any: def cached_parse(layout: torch.Tensor, parse, attribute: str): cached = getattr(layout, attribute, None) - if cached is None or cached[0] != layout._version: - cached = (layout._version, parse(layout)) + if cached is None: + cached = parse(layout) setattr(layout, attribute, cached) - return cached[1] + return cached class Scheme(ABC): diff --git a/entropack/schemes/lattice_rans/cuda.py b/entropack/schemes/lattice_rans/cuda.py index e6acad4..6d4c9ca 100644 --- a/entropack/schemes/lattice_rans/cuda.py +++ b/entropack/schemes/lattice_rans/cuda.py @@ -561,7 +561,7 @@ class _DecodePlan: def _decode_plan(layout, stream_meta, freq_tables, info, caps, threads: int) -> _DecodePlan: device = freq_tables.device - fingerprint = (stream_meta.data_ptr(), stream_meta._version, freq_tables.data_ptr(), freq_tables._version, device, threads) + fingerprint = (stream_meta.data_ptr(), freq_tables.data_ptr(), device, threads) cached = getattr(layout, "_lattice_rans_decode_plan", None) if cached is not None and cached[0] == fingerprint: return cached[1]