| 1 | #!/usr/bin/env python3 |
| 2 | """Writes Noto Sans CJK SC at its regular weight, cut to the characters GB 2312, JIS X 0208 |
| 3 | and KS X 1001's Hangul hold, as the web build's fallback for Chinese, Japanese and Korean. |
| 4 | |
| 5 | uv run --with fonttools python tools/web/subset_cjk.py NotoSansCJK-VF.otf.ttc OUT.otf |
| 6 | """ |
| 7 | import sys |
| 8 | |
| 9 | from fontTools import subset |
| 10 | from fontTools.ttLib import TTCollection |
| 11 | from fontTools.varLib import instancer |
| 12 | |
| 13 | |
| 14 | def main(collection, out): |
| 15 | font = next(face for face in TTCollection(collection).fonts if face['name'].getDebugName(1) == 'Noto Sans CJK SC') |
| 16 | font = instancer.instantiateVariableFont(font, {'wght': 400}) |
| 17 | # Punctuation, kana, Hangul compatibility jamo and full-width forms whole; ideographs and |
| 18 | # syllables as the national standards list them. |
| 19 | codes = set(range(0x3000, 0x3100)) | set(range(0x3130, 0x3190)) | set(range(0xff00, 0xfff0)) |
| 20 | for code, encodings in [*((code, ('gb2312', 'shift_jis')) for code in range(0x4e00, 0xa000)), |
| 21 | *((code, ('euc_kr',)) for code in range(0xac00, 0xd7a4))]: |
| 22 | for encoding in encodings: |
| 23 | try: |
| 24 | chr(code).encode(encoding) |
| 25 | except UnicodeEncodeError: |
| 26 | continue |
| 27 | codes.add(code) |
| 28 | break |
| 29 | options = subset.Options() |
| 30 | options.layout_features = ['*'] |
| 31 | options.name_IDs = ['*'] |
| 32 | options.notdef_outline = True |
| 33 | cut = subset.Subsetter(options) |
| 34 | cut.populate(unicodes=codes) |
| 35 | cut.subset(font) |
| 36 | font.save(out) |
| 37 | |
| 38 | |
| 39 | if __name__ == '__main__': |
| 40 | main(*sys.argv[1:]) |