Files
sysconfig/nuance/tools/gather_transcription_from_wordsfile.pl
luk 9742f6dab2 refactor: 脚本及配置文件命名统一用 _ 分隔
将 .sh/.bat 脚本与 .js/.plist/.Dockerfile 等文件中以 - 分隔的文件名
改为 _ 分隔,dcloud-renew 目录改为 dcloud、script_nuance 目录改为
nuance,并同步更新脚本内部引用与 README 路径。

Co-Authored-By: AtomCode (deepseek-v4-flash) <noreply@atomgit.com>
2026-08-25 15:11:50 +08:00

36 lines
902 B
Perl

print "Usage: perl $0 transcription_dir path_prefix output_file\n";
print "Example: perl $0 /direct/datadigest/read_English_SG/gitm/transcription /direct/datadigest/read_English_SG/gitm/calls/ ~/gitm.list\n";
use File::Find;
use File::Copy;
open CORPUS, ">$ARGV[2]" or die "Cannot open corpus file $ARGV[1] for write.\n";
if ($ARGV[1] =~ m|/$|) # the parameter "path_prefix" is ended with /
{
$prefix = $ARGV[1];
}else
{
$prefix = "$ARGV[1]/";
}
@dirs = ($ARGV[0]);
find ( {wanted => \&wanted},
@dirs );
sub wanted
{
if (m|^([a-zA-Z0-9_]+)_(utt\d+)\.words$|)
{
$folder = $1;
$utt = $2;
$folder =~ m|^[A-Za-z]+(\d\d\d)|;
$group = $1; # usually it's 000, but not always. So $group need be extracted.
open WORDS, "$_" or die "Cannot open words file $_\n";
$words = <WORDS>;
chomp ($words);
print CORPUS "$prefix$group/$folder/${folder}_${utt}.ulaw\t$words\n";
}
}