LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
SentenceBoundariesFinder.cpp
Go to the documentation of this file.
1// Copyright 2002-2020 CEA LIST
2// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
3//
4// SPDX-License-Identifier: MIT
5
6/***************************************************************************
7 * Copyright (C) 2004-2020 by CEA LIST *
8 * *
9 ***************************************************************************/
11#include "SegmentationData.h"
20
21using namespace std;
23
24namespace Lima {
25namespace LinguisticProcessing {
26namespace LinguisticAnalysisStructure {
27
29
32 m_microAccessor(0),
33 m_graph(),
34 m_boundaryValues(),
35 m_forbidBoundaryValues(),
36 m_boundaryMicros()
37{}
38
39
42
45 Manager* manager)
46
47{
52
53 MediaId language=manager->getInitializationParameters().media;
54 m_microAccessor=&(static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(language)).getPropertyCodeManager().getPropertyAccessor("MICRO"));
55 try
56 {
57 m_graph=unitConfiguration.getParamsValueAtKey("graph");
58 }
60 {
61 m_graph=string("PosGraph");
62 }
63
64 try
65 {
66 m_data=unitConfiguration.getParamsValueAtKey("data");
67 }
69 {
70 m_data=string("SentenceBoundaries");
71 }
72
73 try
74 {
75 deque<string> boundariesRestrictions=unitConfiguration.getListsValueAtKey("values");
76 for (deque<string>::const_iterator it=boundariesRestrictions.begin(),it_end=boundariesRestrictions.end();
77 it!=it_end; it++)
78 {
79#ifdef DEBUG_LP
80 LDEBUG << "init(): add filter for value " << *it;
81#endif
82 m_boundaryValues.insert(Common::Misc::utf8stdstring2limastring(*it));
83 }
84 }
85 catch (Common::XMLConfigurationFiles::NoSuchList& ) {} // optional
86
87
88 try
89 {
90 deque<string> exclBoundariesRestrictions=unitConfiguration.getListsValueAtKey("forbid-values");
91 for (deque<string>::const_iterator it=exclBoundariesRestrictions.begin(),it_end=exclBoundariesRestrictions.end();
92 it!=it_end; it++)
93 {
94#ifdef DEBUG_LP
95 LDEBUG << "init(): add filter for forbidden value " << *it;
96#endif
97 m_forbidBoundaryValues.insert(Common::Misc::utf8stdstring2limastring(*it));
98 }
99 }
100 catch (Common::XMLConfigurationFiles::NoSuchList& ) {} // optional
101
102
103 try
104 {
105 const Common::PropertyCode::PropertyManager& microManager=
106 static_cast<const Common::MediaticData::LanguageData&>(Common::MediaticData::MediaticData::single().mediaData(language)).getPropertyCodeManager().getPropertyManager("MICRO");
107 deque<string> boundariesRestrictions=unitConfiguration.getListsValueAtKey("micros");
108 for (deque<string>::const_iterator it=boundariesRestrictions.begin(),it_end=boundariesRestrictions.end();
109 it!=it_end; it++)
110 {
111 LinguisticCode micro=microManager.getPropertyValue(*it);
112 if (micro == L_NONE)
113 {
114 LERROR << "init(): cannot find linguistic code for micro " << *it;
115 }
116 else
117 {
118#ifdef DEBUG_LP
119 LDEBUG << "init(): add filter for micro " << micro;
120#endif
121 m_boundaryMicros.push_back(micro);
122 }
123 }
124 }
126 {
127 LERROR << "Warning: No boundaries categories defined for language " << language;
128 //throw InvalidConfiguration();
129 }
130}
131
132
134 AnalysisContent& analysis) const
135{
136 Lima::TimeUtilsController timer("SentenceBoundariesFinder");
137
139 LINFO << "start finding sentence bounds";
140 auto anagraph = std::dynamic_pointer_cast<AnalysisGraph>(analysis.getData(m_graph));
141 if (anagraph==0)
142 {
143 LERROR << "no graph '" << m_graph << "' available !";
144 return MISSING_DATA;
145 }
146
147 LinguisticGraphVertex lastVx=anagraph->lastVertex();
148 LinguisticGraphVertex beginSentence=anagraph->firstVertex();
149#ifdef DEBUG_LP
150 LDEBUG << "found beginSentence at " << beginSentence;
151#endif
152
153 SegmentationData* sb=new SegmentationData(m_graph);
154 analysis.setData(m_data,sb);
155
156 if (m_boundaryValues.empty() && m_forbidBoundaryValues.empty())
157 {
158 while (beginSentence!=lastVx)
159 {
160 LinguisticGraphVertex endSentence=anagraph->nextMainPathVertex(beginSentence,*m_microAccessor,m_boundaryMicros,lastVx);
161 if (endSentence == lastVx)
162 {
163 set<LinguisticGraphVertex> prevVx = getPrecedingNodes<set<LinguisticGraphVertex>>(*anagraph, lastVx);
164 /*if (prevVx.size() != 1)
165 {
166 throw LimaException("Many paths lead to the last vertex of the text");
167 }
168 endSentence = *prevVx.begin();
169 */
170 if (beginSentence != endSentence && prevVx.end() == prevVx.find(beginSentence))
171 {
172 sb->add(Segment("sentence",beginSentence,endSentence,anagraph.get()));
173 }
174 break;
175 }
176#ifdef DEBUG_LP
177 LDEBUG << "found endSentence at " << endSentence;
178#endif
179 sb->add(Segment("sentence",beginSentence,endSentence,anagraph.get()));
180 beginSentence=endSentence;
181 }
182 }
183 else
184 {
185 // Apply restriction on values for sentence boundaries
186 // cannot set endSentence from beginSentence inside the loop, because we have to continue
187 // moving forward even if there is no match (with restricted values)
188 LinguisticGraphVertex endSentence=anagraph->nextMainPathVertex(beginSentence,*m_microAccessor,m_boundaryMicros,lastVx);
189 while (endSentence!=lastVx)
190 {
191 Token* t=get(vertex_token,*(anagraph->getGraph()),endSentence);
192#ifdef DEBUG_LP
193 if (t!=0)
194 {
195 LDEBUG << "found endSentence at " << endSentence << "("
197 }
198 else
199 {
200 LDEBUG << "found endSentence at " << endSentence;
201 }
202#endif
203
204 if (t==0 || (!m_forbidBoundaryValues.empty() && m_forbidBoundaryValues.find(t->stringForm())!=m_forbidBoundaryValues.end()) )
205 {
206#ifdef DEBUG_LP
207 LDEBUG << " -> not kept (bcz in forbidden values)";
208#endif
209 }
210 else if (t==0 || (!m_boundaryValues.empty() && m_boundaryValues.find(t->stringForm())==m_boundaryValues.end()) )
211 {
212#ifdef DEBUG_LP
213 LDEBUG << " -> not kept (bcz not in restricted values)";
214#endif
215 }
216 else
217 {
218#ifdef DEBUG_LP
219 LDEBUG << "add sentence " << beginSentence << "-" << endSentence;
220#endif
221 sb->add(Segment("sentence",beginSentence,endSentence,anagraph.get()));
222 beginSentence=endSentence;
223 }
224 endSentence=anagraph->nextMainPathVertex(endSentence,*m_microAccessor,m_boundaryMicros,lastVx);
225 }
226 // add last sentence (the one ending with lastVx)
227 if(beginSentence<endSentence) {
228#ifdef DEBUG_LP
229 LDEBUG << "add last sentence " << beginSentence << "-" << endSentence;
230#endif
231 sb->add(Segment("sentence",beginSentence,endSentence,anagraph.get()));
232 }
233 }
234 return SUCCESS_ID;
235}
236
237
238}
239
240}
241
242}
#define LDEBUG
Definition LimaCommon.h:157
#define LINFO
Definition LimaCommon.h:158
#define LERROR
Definition LimaCommon.h:161
@ vertex_token
LinguisticGraph::vertex_descriptor LinguisticGraphVertex
#define SENTBOUNDLOGINIT
#define SENTENCEBOUNDARIESFINDER_CLASSID
Defines a Factory to create Object of type Base.
#define L_NONE
Definition StdBitset.h:338
Holds all data that pass through the ProcessUnits Analysis data are shared pointers,...
std::shared_ptr< AnalysisData > getData(const QString &id)
return AnalysisData by id
void setData(const QString &id, std::shared_ptr< AnalysisData > data)
set an analysisData with the given id.
Holds linguistic data for one language.
const MediaData & mediaData(MediaId media) const
Provide tools to manage a specific property.
LinguisticCode getPropertyValue(const std::string &symbolicValue) const
Get the coded property value from the symbolic value.
std::deque< std::string > & getListsValueAtKey(const std::string &key)
return a message when a 'param' was not found
LimaStatusCode process(AnalysisContent &analysis) const override
Process on data in analysisContent.
void init(Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
initialize with parameters from configuration file.
static const MediaticData & single()
const singleton accessor
Definition Singleton.h:51
This file contains a class to control log of informations about time, such as logging cumulated time ...
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.
LimaString utf8stdstring2limastring(const std::string &src)
SimpleFactory< MediaProcessUnit, SentenceBoundariesFinder > sentenceBoundariesFinderFactory(SENTENCEBOUNDARIESFINDER_CLASSID)
NAUTITIA.
LimaStatusCode
Definition LimaCommon.h:236
@ SUCCESS_ID
Definition LimaCommon.h:237
@ MISSING_DATA
Definition LimaCommon.h:243
STL namespace.
launch exception related to the configuration file parsing