44namespace LinguisticProcessing {
45namespace AnalysisDumpers {
49 const MediaId& language,
50 std::shared_ptr<StopList> stopList) : m_language(language), m_stopList(stopList) {
69 boost::shared_ptr<LTR_Text> textRep,
74 uint64_t tokenCounter = 0;
75 this->addTokensToLTRTextFrom(
85 LDEBUG <<
"LTR: add sentence bound at token" << tokenCounter;
86 textRep->addSentenceBound(tokenCounter);
90 std::vector<Segment>::iterator sbIt = (sb->
getSegments()).begin();
91 uint64_t tokenCounter = 0;
96 this->addTokensToLTRTextFrom(
104 textRep->addSentenceBound(tokenCounter);
110void LTRTextBuilder::addTokensToLTRTextFrom(
115 boost::shared_ptr<LTR_Text> textRep,
117 uint64_t* tokenCounter) {
120 m_verticesToExplore.clear();
121 m_exploredVertices.clear();
122 this->exploreVerticesFrom(firstVertex, graphLastVertex, graph);
123 if (m_verticesToExplore.size() != 0) {
127 m_currentOffset = token->
position() + offset;
128 boost::shared_ptr<LTR_Token> currentLtrTok(
new LTR_Token() );
129 textRep->addToken(currentLtrTok);
132 bool endVertexFlag =
false;
133 while (! endVertexFlag && ! m_verticesToExplore.empty()) {
136 uint64_t smallestPosDiff = std::numeric_limits<uint64_t>::max();
137 for (VERTICES_TO_EXPLORE_T::const_iterator itVert = m_verticesToExplore.begin();
138 itVert != m_verticesToExplore.end(); itVert ++) {
140 uint64_t posDiff = token->
position() + offset - m_currentOffset;
141 if (posDiff < smallestPosDiff) {
142 smallestPosDiff = posDiff;
143 closestVertex = *itVert;
147 if (smallestPosDiff > 0) {
148 currentLtrTok = boost::shared_ptr<LTR_Token>(
new LTR_Token());
149 m_currentOffset = token->
position() + offset;
150 textRep->addToken(currentLtrTok);
154 this->updateLTR_TokenFromVertex(closestVertex, graph,
155 currentLtrTok, offset);
158 m_verticesToExplore.remove(closestVertex);
159 m_exploredVertices.insert(closestVertex);
162 this->exploreVerticesFrom(closestVertex, graphLastVertex, graph);
163 endVertexFlag = (closestVertex == lastVertex);
165 if (! m_verticesToExplore.empty()) {
166 LWARN <<
"all vertices not explored between " << firstVertex <<
" and "
172void LTRTextBuilder::exploreVerticesFrom(
178 boost::tie(it, itEnd) = out_edges(vertex, graph);
179 while (it != itEnd) {
181 if ((outVertex != graphLastVertex) &&
182 (m_exploredVertices.find(outVertex) == m_exploredVertices.end())) {
183 m_verticesToExplore.push_back(outVertex);
189void LTRTextBuilder::updateLTR_TokenFromVertex(
192 boost::shared_ptr<LTR_Token> tokenRep,
193 uint64_t offset)
const
202 if (data->size()==0) {
204 LERROR <<
"Empty MorphoSyntacticData for vertex" << vertex <<
", token=" << fullToken->
stringForm();
209 StringsPoolIndex norm(0),lastNorm(0);
211 for (MorphoSyntacticData::const_iterator elemItr=data->begin();
212 elemItr!=data->end(); elemItr++) {
213 norm = elemItr->normalizedForm;
214 macro = m_macroAccessor->
readValue(elemItr->properties);
215 if (norm == lastNorm && macro == lastMacro) {
223 bool selectionFlag =
true;
224 LTR_Token::const_iterator itTok = tokenRep->begin();
225 while (selectionFlag && (itTok != tokenRep->end())) {
226 selectionFlag = (itTok->first->getLemma() != normStr) ||
227 (itTok->first->getCategory() != macro);
234 boost::shared_ptr< BoWToken> bowToken (
new BoWToken(normStr,macro,
237 bowToken->setInflectedForm(fullToken->
stringForm());
239 tokenRep->push_back(make_pair(bowToken, plainWordFlag));
271 if (m_stopList && (m_stopList->find(lemma) != m_stopList->end())) {
=========================================================================
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
LinguisticGraph::out_edge_iterator LinguisticGraphOutEdgeIt
boost::adjacency_list< boost::vecS, boost::vecS, boost::bidirectionalS, LinguisticVertexProperties > LinguisticGraph
Property to identify the chains in the graph.
This class contains the representation of an element of the bag of words.
a token of a linear text representation Each LTR_Token represents the set of words that start at a po...
LinguisticCode readValue(const LinguisticCode &code) const
read a property in a coded int.
LTRTextBuilder(const MediaId &language, std::shared_ptr< StopList > stopList)
void buildLTRTextFrom(const LinguisticGraph &graph, Lima::LinguisticProcessing::SegmentationData *sb, const LinguisticGraphVertex &graphFirstVertex, const LinguisticGraphVertex &graphLastVertex, boost::shared_ptr< Lima::Common::BagOfWords::LTR_Text > textRep, uint64_t offset)
build a LTRText representation of the analyzed text
bool isWordToSelect(const Lima::LimaString &lemma, LinguisticCode macroCategory) const
Holds morphosyntactic informations.
holds surface data of a token
uint64_t position() const
const LimaString & stringForm() const
const std::vector< Segment > & getSegments() const
static const MediaticData & single()
const singleton accessor
=========================================================================
PUGI__FN void sort(I begin, I end, const Pred &pred)