PaddleSpeech/utils/spm_encode

#!/usr/bin/env python3
# Copyright (c) Facebook, Inc. and its affiliates.
# All rights reserved.
#
# This source code is licensed under the license found in
# https://github.com/pytorch/fairseq/blob/master/LICENSE
from __future__ import absolute_import
from __future__ import division
from __future__ import print_function
from __future__ import unicode_literals

import argparse
import contextlib
import sys

import sentencepiece as spm


def main():
    parser = argparse.ArgumentParser()
    parser.add_argument("--model", required=True,
                        help="sentencepiece model to use for encoding")
    parser.add_argument("--inputs", nargs="+", default=['-'],
                        help="input files to filter/encode")
    parser.add_argument("--outputs", nargs="+", default=['-'],
                        help="path to save encoded outputs")
    parser.add_argument("--output_format", choices=["piece", "id"], default="piece")
    parser.add_argument("--min-len", type=int, metavar="N",
                        help="filter sentence pairs with fewer than N tokens")
    parser.add_argument("--max-len", type=int, metavar="N",
                        help="filter sentence pairs with more than N tokens")
    args = parser.parse_args()

    assert len(args.inputs) == len(args.outputs), \
        "number of input and output paths should match"

    sp = spm.SentencePieceProcessor()
    sp.Load(args.model)

    if args.output_format == "piece":
        def encode(l):
            return sp.EncodeAsPieces(l)
    elif args.output_format == "id":
        def encode(l):
            return list(map(str, sp.EncodeAsIds(l)))
    else:
        raise NotImplementedError

    if args.min_len is not None or args.max_len is not None:
        def valid(line):
            return (
                (args.min_len is None or len(line) >= args.min_len) and
                (args.max_len is None or len(line) <= args.max_len)
            )
    else:
        def valid(lines):
            return True

    with contextlib.ExitStack() as stack:
        inputs = [
            stack.enter_context(open(input, "r", encoding="utf-8"))
            if input != "-" else sys.stdin
            for input in args.inputs
        ]
        outputs = [
            stack.enter_context(open(output, "w", encoding="utf-8"))
            if output != "-" else sys.stdout
            for output in args.outputs
        ]

        stats = {
            "num_empty": 0,
            "num_filtered": 0,
        }

        def encode_line(line):
            line = line.strip()
            if len(line) > 0:
                line = encode(line)
                if valid(line):
                    return line
                else:
                    stats["num_filtered"] += 1
            else:
                stats["num_empty"] += 1
            return None

        for i, lines in enumerate(zip(*inputs), start=1):
            enc_lines = list(map(encode_line, lines))
            if not any(enc_line is None for enc_line in enc_lines):
                for enc_line, output_h in zip(enc_lines, outputs):
                    print(" ".join(enc_line), file=output_h)
            if i % 10000 == 0:
                print("processed {} lines".format(i), file=sys.stderr)

        print("skipped {} empty lines".format(stats["num_empty"]), file=sys.stderr)
        print("filtered {} lines".format(stats["num_filtered"]), file=sys.stderr)


if __name__ == "__main__":
    main()
refactor g2p egs (#630) * refactor g2p egs * add sha-bone; remove avg.sh from egs; 4 years ago			`#!/usr/bin/env python3`
E2E/Streaming Transformer/Conformer ASR (#578) * add cmvn and label smoothing loss layer * add layer for transformer * add glu and conformer conv * add torch compatiable hack, mask funcs * not hack size since it exists * add test; attention * add attention, common utils, hack paddle * add audio utils * conformer batch padding mask bug fix #223 * fix typo, python infer fix rnn mem opt name error and batchnorm1d, will be available at 2.0.2 * fix ci * fix ci * add encoder * refactor egs * add decoder * refactor ctc, add ctc align, refactor ckpt, add warmup lr scheduler, cmvn utils * refactor docs * add fix * fix readme * fix bugs, refactor collator, add pad_sequence, fix ckpt bugs * fix docstring * refactor data feed order * add u2 model * refactor cmvn, test * add utils * add u2 config * fix bugs * fix bugs * fix autograd maybe has problem when using inplace operation * refactor data, build vocab; add format data * fix text featurizer * refactor build vocab * add fbank, refactor feature of speech * refactor audio feat * refactor data preprare * refactor data * model init from config * add u2 bins * flake8 * can train * fix bugs, add coverage, add scripts * test can run * fix data * speed perturb with sox * add spec aug * fix for train * fix train logitc * fix logger * log valid loss, time dataset process * using np for speed perturb, remove some debug log of grad clip * fix logger * fix build vocab * fix logger name * using module logger as default * fix * fix install * reorder imports * fix board logger * fix logger * kaldi fbank and mfcc * fix cmvn and print prarams * fix add_eos_sos and cmvn * fix cmvn compute * fix logger and cmvn * fix subsampling, label smoothing loss, remove useless * add notebook test * fix log * fix tb logger * multi gpu valid * fix log * fix log * fix config * fix compute cmvn, need paddle 2.1 * add cmvn notebook * fix layer tools * fix compute cmvn * add rtf * fix decoding * fix layer tools * fix log, add avg script * more avg and test info * fix dataset pickle problem; using 2.1 paddle; num_workers can > 0; ckpt save in exp dir;fix setup.sh; * add vimrc * refactor tiny script, add transformer and stream conf * spm demo; librisppech scripts and confs * fix log * add librispeech scripts * refactor data pipe; fix conf; fix u2 default params * fix bugs * refactor aishell scripts * fix test * fix cmvn * fix s0 scripts * fix ds2 scripts and bugs * fix dev & test dataset filter * fix dataset filter * filter dev * fix ckpt path * filter test, since librispeech will cause OOM, but all test wer will be worse, since mismatch train with test * add comment * add syllable doc * fix ds2 configs * add doc * add pypinyin tools * fix decoder using blank_id=0 * mmseg with pybind11 * format code 4 years ago			`# Copyright (c) Facebook, Inc. and its affiliates.`
			`# All rights reserved.`
			`#`
			`# This source code is licensed under the license found in`
			`# https://github.com/pytorch/fairseq/blob/master/LICENSE`
			`from __future__ import absolute_import`
			`from __future__ import division`
			`from __future__ import print_function`
			`from __future__ import unicode_literals`

			`import argparse`
			`import contextlib`
			`import sys`

			`import sentencepiece as spm`


			`def main():`
			`parser = argparse.ArgumentParser()`
			`parser.add_argument("--model", required=True,`
			`help="sentencepiece model to use for encoding")`
			`parser.add_argument("--inputs", nargs="+", default=['-'],`
			`help="input files to filter/encode")`
			`parser.add_argument("--outputs", nargs="+", default=['-'],`
			`help="path to save encoded outputs")`
			`parser.add_argument("--output_format", choices=["piece", "id"], default="piece")`
			`parser.add_argument("--min-len", type=int, metavar="N",`
			`help="filter sentence pairs with fewer than N tokens")`
			`parser.add_argument("--max-len", type=int, metavar="N",`
			`help="filter sentence pairs with more than N tokens")`
			`args = parser.parse_args()`

			`assert len(args.inputs) == len(args.outputs), \`
			`"number of input and output paths should match"`

			`sp = spm.SentencePieceProcessor()`
			`sp.Load(args.model)`

			`if args.output_format == "piece":`
			`def encode(l):`
			`return sp.EncodeAsPieces(l)`
			`elif args.output_format == "id":`
			`def encode(l):`
			`return list(map(str, sp.EncodeAsIds(l)))`
			`else:`
			`raise NotImplementedError`

			`if args.min_len is not None or args.max_len is not None:`
			`def valid(line):`
			`return (`
			`(args.min_len is None or len(line) >= args.min_len) and`
			`(args.max_len is None or len(line) <= args.max_len)`
			`)`
			`else:`
			`def valid(lines):`
			`return True`

			`with contextlib.ExitStack() as stack:`
			`inputs = [`
			`stack.enter_context(open(input, "r", encoding="utf-8"))`
			`if input != "-" else sys.stdin`
			`for input in args.inputs`
			`]`
			`outputs = [`
			`stack.enter_context(open(output, "w", encoding="utf-8"))`
			`if output != "-" else sys.stdout`
			`for output in args.outputs`
			`]`

			`stats = {`
			`"num_empty": 0,`
			`"num_filtered": 0,`
			`}`

			`def encode_line(line):`
			`line = line.strip()`
			`if len(line) > 0:`
			`line = encode(line)`
			`if valid(line):`
			`return line`
			`else:`
			`stats["num_filtered"] += 1`
			`else:`
			`stats["num_empty"] += 1`
			`return None`

			`for i, lines in enumerate(zip(*inputs), start=1):`
			`enc_lines = list(map(encode_line, lines))`
			`if not any(enc_line is None for enc_line in enc_lines):`
			`for enc_line, output_h in zip(enc_lines, outputs):`
			`print(" ".join(enc_line), file=output_h)`
			`if i % 10000 == 0:`
			`print("processed {} lines".format(i), file=sys.stderr)`

			`print("skipped {} empty lines".format(stats["num_empty"]), file=sys.stderr)`
			`print("filtered {} lines".format(stats["num_filtered"]), file=sys.stderr)`


			`if __name__ == "__main__":`
			`main()`