PaddleSpeech/examples/wenetspeech/asr1/local/extract_meta.py

# Copyright (c) 2021 PaddlePaddle Authors. All Rights Reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
#     http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# Copyright 2021  Xiaomi Corporation (Author: Yongqing Wang)
#                 Mobvoi Inc(Author: Di Wu, Binbin Zhang)
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
#     http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
import argparse
import json
import os
import sys


def get_args():
    parser = argparse.ArgumentParser(description="""
      This script is used to process raw json dataset of WenetSpeech,
      where the long wav is splitinto segments and
      data of wenet format is generated.
      """)
    parser.add_argument('input_json', help="""Input json file of WenetSpeech""")
    parser.add_argument('output_dir', help="""Output dir for prepared data""")

    args = parser.parse_args()
    return args


def meta_analysis(input_json, output_dir):
    input_dir = os.path.dirname(input_json)

    if not os.path.exists(output_dir):
        os.makedirs(output_dir)

    try:
        with open(input_json, 'r') as injson:
            json_data = json.load(injson)
    except Exception:
        sys.exit(f'Failed to load input json file: {input_json}')
    else:
        if json_data['audios'] is not None:
            with open(f'{output_dir}/text', 'w') as utt2text, \
                 open(f'{output_dir}/segments', 'w') as segments, \
                 open(f'{output_dir}/utt2dur', 'w') as utt2dur, \
                 open(f'{output_dir}/wav.scp', 'w') as wavscp, \
                 open(f'{output_dir}/utt2subsets', 'w') as utt2subsets, \
                 open(f'{output_dir}/reco2dur', 'w') as reco2dur:
                for long_audio in json_data['audios']:
                    try:
                        long_audio_path = os.path.realpath(
                            os.path.join(input_dir, long_audio['path']))
                        aid = long_audio['aid']
                        segments_lists = long_audio['segments']
                        duration = long_audio['duration']
                        assert (os.path.exists(long_audio_path))
                    except AssertionError:
                        print(f'''Warning: {aid} something is wrong,
                                  maybe AssertionError, skipped''')
                        continue
                    except Exception:
                        print(f'''Warning: {aid} something is wrong, maybe the
                                  error path: {long_audio_path}, skipped''')
                        continue
                    else:
                        wavscp.write(f'{aid}\t{long_audio_path}\n')
                        reco2dur.write(f'{aid}\t{duration}\n')
                        for segment_file in segments_lists:
                            try:
                                sid = segment_file['sid']
                                start_time = segment_file['begin_time']
                                end_time = segment_file['end_time']
                                dur = end_time - start_time
                                text = segment_file['text']
                                segment_subsets = segment_file["subsets"]
                            except Exception:
                                print(f'''Warning: {segment_file} something
                                          is wrong, skipped''')
                                continue
                            else:
                                utt2text.write(f'{sid}\t{text}\n')
                                segments.write(
                                    f'{sid}\t{aid}\t{start_time}\t{end_time}\n')
                                utt2dur.write(f'{sid}\t{dur}\n')
                                segment_sub_names = " ".join(segment_subsets)
                                utt2subsets.write(
                                    f'{sid}\t{segment_sub_names}\n')


def main():
    args = get_args()

    meta_analysis(args.input_json, args.output_dir)


if __name__ == '__main__':
    main()
fix the prc-commit 3 years ago			`# Copyright (c) 2021 PaddlePaddle Authors. All Rights Reserved.`
			`#`
			`# Licensed under the Apache License, Version 2.0 (the "License");`
			`# you may not use this file except in compliance with the License.`
			`# You may obtain a copy of the License at`
			`#`
			`# http://www.apache.org/licenses/LICENSE-2.0`
			`#`
			`# Unless required by applicable law or agreed to in writing, software`
			`# distributed under the License is distributed on an "AS IS" BASIS,`
			`# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.`
			`# See the License for the specific language governing permissions and`
			`# limitations under the License.`
add wenetspeech egs 3 years ago			`# Copyright 2021 Xiaomi Corporation (Author: Yongqing Wang)`
			`# Mobvoi Inc(Author: Di Wu, Binbin Zhang)`
			`# Licensed under the Apache License, Version 2.0 (the "License");`
			`# you may not use this file except in compliance with the License.`
			`# You may obtain a copy of the License at`
			`#`
			`# http://www.apache.org/licenses/LICENSE-2.0`
			`#`
			`# Unless required by applicable law or agreed to in writing, software`
			`# distributed under the License is distributed on an "AS IS" BASIS,`
			`# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.`
			`# See the License for the specific language governing permissions and`
			`# limitations under the License.`
			`import argparse`
			`import json`
fix the prc-commit 3 years ago			`import os`
			`import sys`
add wenetspeech egs 3 years ago

			`def get_args():`
			`parser = argparse.ArgumentParser(description="""`
			`This script is used to process raw json dataset of WenetSpeech,`
			`where the long wav is splitinto segments and`
			`data of wenet format is generated.`
			`""")`
			`parser.add_argument('input_json', help="""Input json file of WenetSpeech""")`
			`parser.add_argument('output_dir', help="""Output dir for prepared data""")`

			`args = parser.parse_args()`
			`return args`


			`def meta_analysis(input_json, output_dir):`
			`input_dir = os.path.dirname(input_json)`

			`if not os.path.exists(output_dir):`
			`os.makedirs(output_dir)`

			`try:`
			`with open(input_json, 'r') as injson:`
			`json_data = json.load(injson)`
			`except Exception:`
			`sys.exit(f'Failed to load input json file: {input_json}')`
			`else:`
			`if json_data['audios'] is not None:`
			`with open(f'{output_dir}/text', 'w') as utt2text, \`
			`open(f'{output_dir}/segments', 'w') as segments, \`
			`open(f'{output_dir}/utt2dur', 'w') as utt2dur, \`
			`open(f'{output_dir}/wav.scp', 'w') as wavscp, \`
			`open(f'{output_dir}/utt2subsets', 'w') as utt2subsets, \`
			`open(f'{output_dir}/reco2dur', 'w') as reco2dur:`
			`for long_audio in json_data['audios']:`
			`try:`
			`long_audio_path = os.path.realpath(`
			`os.path.join(input_dir, long_audio['path']))`
			`aid = long_audio['aid']`
			`segments_lists = long_audio['segments']`
			`duration = long_audio['duration']`
			`assert (os.path.exists(long_audio_path))`
			`except AssertionError:`
			`print(f'''Warning: {aid} something is wrong,`
			`maybe AssertionError, skipped''')`
			`continue`
			`except Exception:`
			`print(f'''Warning: {aid} something is wrong, maybe the`
			`error path: {long_audio_path}, skipped''')`
			`continue`
			`else:`
			`wavscp.write(f'{aid}\t{long_audio_path}\n')`
			`reco2dur.write(f'{aid}\t{duration}\n')`
			`for segment_file in segments_lists:`
			`try:`
			`sid = segment_file['sid']`
			`start_time = segment_file['begin_time']`
			`end_time = segment_file['end_time']`
			`dur = end_time - start_time`
			`text = segment_file['text']`
			`segment_subsets = segment_file["subsets"]`
			`except Exception:`
			`print(f'''Warning: {segment_file} something`
			`is wrong, skipped''')`
			`continue`
			`else:`
			`utt2text.write(f'{sid}\t{text}\n')`
			`segments.write(`
fix the prc-commit 3 years ago			`f'{sid}\t{aid}\t{start_time}\t{end_time}\n')`
add wenetspeech egs 3 years ago			`utt2dur.write(f'{sid}\t{dur}\n')`
			`segment_sub_names = " ".join(segment_subsets)`
			`utt2subsets.write(`
			`f'{sid}\t{segment_sub_names}\n')`

fix the prc-commit 3 years ago
add wenetspeech egs 3 years ago			`def main():`
			`args = get_args()`

			`meta_analysis(args.input_json, args.output_dir)`


			`if __name__ == '__main__':`
fix the prc-commit 3 years ago			`main()`