Repository navigation
Expand file tree
/
Copy pathParametersAlgo.py
More file actions
executable file
·142 lines (100 loc) · 4.27 KB
/
Copy pathParametersAlgo.py
File metadata and controls
executable file
·142 lines (100 loc) · 4.27 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
# Copyright (C) 2014-2017 Music Technology Group - Universitat Pompeu Fabra
#
# This file is part of AlignmentDuration: tool for Lyrics-to-audio alignment with syllable duration modeling
#
# AlignmentDuration is free software: you can redistribute it and/or modify it under
# the terms of the GNU Affero General Public License as published by the Free
# Software Foundation (FSF), either version 3 of the License, or (at your
# option) any later version.
#
# This program is distributed in the hope that it will be useful, but WITHOUT
# ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or FITNESS
# FOR A PARTICULAR PURPOSE. See the GNU General Public License for more
# details.
#
# You should have received a copy of the Affero GNU General Public License
# version 3 along with this program. If not, see http://www.gnu.org/licenses/
'''
Created on May 28, 2015
@author: joro
'''
### include src folder
import os
import sys
parentDir = os.path.abspath(os.path.join(os.path.dirname(os.path.realpath(__file__) ), os.path.pardir, os.pardir))
if parentDir not in sys.path:
sys.path.append(parentDir)
import logging
from numpy.ma.core import floor
import os
from src.parse.TextGrid_Parsing import tierAliases
######### PARAMS:
class ParametersAlgo(object):
ALPHA = 0.97
FOR_JINGJU = 0
FOR_MAKAM = 0
OBS_MODEL = 'GMM'
OBS_MODEL = 'MLP'
OBS_MODEL = 'MLP_fuzzy'
EVAL_LEVEL = tierAliases.words
# eval level phonemes does not work
# EVAL_LEVEL = tierAliases.pinyin # in Jingju only level is syllable
# use duraiton-based decoding (HMMDuraiton package) or just plain viterbi (HMM package)
# if false, use transition probabilities from htkModels
WITH_DURATIONS= 1
USE_PERSISTENT_PPGs = 0
# level into which to segments decoded result stateNetwork
# DETECTION_TOKEN_LEVEL= 'syllables'
DETECTION_TOKEN_LEVEL= 'words'
# DETECTION_TOKEN_LEVEL= 'phonemes'
Q_WEIGHT_TRANSITION = 3.5
DECODE_WITH_HTK = 0
GLOBAL_WAIT_PROB = 0.9
THRESHOLD_PEAKS = -70
DEVIATION_IN_SEC = 0.1
# unit: num frames
NUMFRAMESPERSECOND = 100
# same as WINDOWSIZE in wavconfig singing. unit: seconds. TOOD: read from there automatically
WINDOW_SIZE = 0.025
# in frames
ONLY_MIDDLE_STATE = 1
WITH_SHORT_PAUSES = 0
# padded a short pause state at beginning and end of sequence
WITH_PADDED_SILENCE = 0
# no feature vectors at all. all observ, probs. set to 1
# WITH_ORACLE_PHONEMES = -1
WITH_ORACLE_PHONEMES = 0
PATH_TO_HCOPY= '/usr/local/bin/HCopy'
PATH_TO_HVITE = '/usr/local/bin/HVite'
# On kora.s.upf.edu
# PATH_TO_HCOPY = '/homedtic/georgid/htkBuilt/bin/HCopy'
projDir = os.path.abspath(os.path.join(os.path.dirname(os.path.realpath(__file__)) , os.path.pardir ))
PATH_TO_CONFIG_FILES= projDir + '/models_makam/input_files/'
parentDir = os.path.abspath(os.path.join(os.path.dirname(os.path.realpath(__file__) ), os.path.pardir))
MODELS_DIR = os.path.join(parentDir, 'models_jingju/' + '3' + 'folds/')
POLYPHONIC = 1
WITH_ORACLE_ONSETS = -1
### no onsets at all.
# WITH_ORACLE_ONSETS = -1
# Sigma of onset smoothing function g: normal distribution
ONSET_SIGMA = 0.075
# ONSET_SIGMA = 0.15
ONSET_SIGMA_IN_FRAMES = int(floor(ONSET_SIGMA * NUMFRAMESPERSECOND))
if ONSET_SIGMA_IN_FRAMES % 2 == 0:
ONSET_SIGMA_IN_FRAMES += 1
# ONSET_TOLERANCE_WINDOW = 0.02 # seconds. to work implement decoding with one onset only
ONSET_TOLERANCE_WINDOW = 0 # seconds
# in _ContinousHMM.b_map cut probabilities
CUTOFF_BIN_OBS_PROBS = 30
# for for_jingju
CONSONANT_DURATION_IN_SEC = 0.3
# for for_makam
# CONSONANT_DURATION_IN_SEC = 0.1
CONSONANT_DURATION = NUMFRAMESPERSECOND * CONSONANT_DURATION_IN_SEC;
CONSONANT_DURATION_DEVIATION = 0.7
#####
LOGGING_LEVEL = logging.INFO
VISUALIZE = 0
ANNOTATION_RULES_ONSETS_EXT = 'annotationOnsets.txt'
ANNOTATION_SCORE_ONSETS_EXT = 'alignedNotes.txt' # use this ont to get better impression on recall, compared to annotationOnsets.txt, which are only on note onsets with rules of interest
WRITE_TO_FILE = True