LIMA
Libre Multilingual Analyzer — C++ API
Loading...
Searching...
No Matches
structuredDocumentXMLParser.cpp
Go to the documentation of this file.
1// SPDX-FileCopyrightText: 2022 CEA LIST <gael.de-chalendar@cea.fr>
2//
3// SPDX-License-Identifier: MIT
4
5/************************************************************************
6 * @file StructuredDocumentXMLParser.cpp
7 * @author Olivier Mesnard
8 * @date ??
9 * modified by Romaric Besancon on Tue Oct 28 18:29:19 2003
10 *
11 *
12 ***********************************************************************/
15#include "contentDocument.h"
16
19
22
23#include <QXmlStreamAttributes>
24
25#include <iostream>
26#include <string>
27#include <deque>
28#include <cassert>
29
30using namespace std;
31
32using namespace Lima::Common;
33using namespace Lima::Common::Misc;
35
36 // to handle dates (parsing)
37
38namespace Lima {
39namespace DocumentsReader {
40
41
42Lima::SimpleFactory<AbstractReaderResource,
45
47 m_processor(nullptr),
48 m_currentDocument(),
49 m_elementPointerHasBeenReturned(false),
50 m_fields(),
51 m_addAbsoluteOffsetToTokens(true)
52{
53#ifdef DEBUG_LP
55 LDEBUG << "StructuredDocumentXMLParser::StructuredDocumentXMLParser()";
56#endif
57}
58
62
63void StructuredDocumentXMLParser::setShiftFrom(std::shared_ptr<const ShiftFrom> shiftFrom)
64{
65#ifdef DEBUG_LP
67 LDEBUG << "StructuredDocumentXMLParser::setShiftFrom";
68#endif
69 m_shiftFrom = shiftFrom;
70}
71
74 Manager* )
75{
76#ifdef DEBUG_LP
78 LDEBUG << "StructuredDocumentXMLParser::init(): readPropertiesMetadata ...";
79#endif
80 readPropertiesMetadata ( unitConfiguration ) ;
81
82 // Read list of tag for identified fields in document
83 vector<pair<QString,FieldType> > fields;
84 for ( unsigned int f=0; f < MAX_NODE_TYPE; f++ )
85 {
86#ifdef DEBUG_LP
87 LDEBUG << "StructuredDocumentXMLParser::init(): readFieldTags " << f;
88#endif
89 readFieldTags ( unitConfiguration,static_cast<FieldType> ( f ),fields );
90 }
91
92 for ( vector<pair<QString,FieldType> >::const_iterator tag=fields.begin();
93 tag!=fields.end(); tag++ )
94 {
95#ifdef DEBUG_LP
96 LDEBUG << "StructuredDocumentXMLParser::init(): addField "
97 << ( *tag ).first << " to list of type " << ( *tag ).second;
98#endif
99 addField ( ( *tag ).first, ( *tag ).second );
100 }
101
102 // read parameters for treatment of XML entities
103// map<Lima::LimaChar,unsigned int> specialCharSize;
104#ifdef DEBUG_LP
105 LDEBUG << "StructuredDocumentXMLParser::init(): getMapAtKey(specialCharacterSize)";
106#endif
107 try
108 {
109 const map<string,string>& specialCharSizeMap=unitConfiguration.
110 getMapAtKey ( "specialCharacterSize" );
111 map<string,string>::const_iterator
112 it=specialCharSizeMap.begin(),
113 it_end=specialCharSizeMap.end();
114 for ( ; it!=it_end; it++ )
115 {
116 m_specialCharacterSize[ ( *it ).first[0]]=atoi ( ( *it ).second.c_str() );
117 }
118 }
119 catch ( NoSuchMap& e )
120 {
121 DRLOGINIT;
122 LWARN << "StructuredDocumentXMLParser::init: no such map specialCharacterSize";
123 // ignored: keep empty map
124 }
125
126 try
127 {
128 if (unitConfiguration.getParamsValueAtKey ( "addAbsoluteOffsetToTokens" ) == "no")
129 m_addAbsoluteOffsetToTokens = false;
130 }
131 catch ( NoSuchParam& e )
132 {
133 DRLOGINIT;
134 LINFO << "StructuredDocumentXMLParser: No parameter addAbsoluteOffsetToTokens. Using default 'yes'";
135 }
136
137#ifdef DEBUG_LP
138 LDEBUG << "StructuredDocumentXMLParser::init() done";
139#endif
140}
141
142//**********************************************************************
143// functions to deal with special characters (predefined XML entities)
144bool StructuredDocumentXMLParser::
145isSpecialCharacter ( const Lima::LimaChar c )
146{
147 return ( m_specialCharacterSize.find ( c )
148 != m_specialCharacterSize.end() );
149}
150
151unsigned int StructuredDocumentXMLParser::
152getSpecialCharSize ( const Lima::LimaChar c )
153{
154 return m_specialCharacterSize[c];
155}
156
157
158void StructuredDocumentXMLParser::readPropertiesMetadata (
160{
161#ifdef DEBUG_LP
162 DRLOGINIT;
163#endif
164
165 // Read list of standard properties
166 try
167 {
168 deque<string> standardPropertyList = groupConf.getListsValueAtKey ( "standard-properties-list" );
169 // get list of identifier of attributes from config file
170 for ( deque<string>::const_iterator attributeIdentifier = standardPropertyList.begin();
171 attributeIdentifier != standardPropertyList.end(); attributeIdentifier++ )
172 {
173 // build attribute from identifier
174 DocumentPropertyType attr = *DocumentsReaderResources::single().getProperty ( *attributeIdentifier );
175 // add new property to map elementTag->property*
176 const std::deque<QString> elementTagSet = attr.getElementTagNames();
177 for ( std::deque<QString>::const_iterator it = elementTagSet.begin() ;
178 it != elementTagSet.end() ; it++ )
179 {
180 const QString elementTagName = *it;
181 auto ite = m_elementTag2DocumentPropertyType.find( elementTagName );
182 if( ite != m_elementTag2DocumentPropertyType.end() ){
183 if( ite->second.getId() != attr.getId() ) {
184 DRLOGINIT;
185 LWARN << "ElementTag" << elementTagName << "cannot be used for"<< attr.getId() <<", it is already used for Document Property" << ite->second.getId() << ". Skip";
186 }
187 continue;
188 }
189#ifdef DEBUG_LP
190 LDEBUG << "ElementTag" << elementTagName << "is used for property" << attr.getId();
191#endif
192 m_elementTag2DocumentPropertyType.insert ( std::make_pair( elementTagName, attr ) );
193 }
194 // add property to map elementTag->property*
195 const std::deque< std::pair<QString,QString> >& elementTagOfPairSet = attr.getAttributeTagNames();
196 for ( std::deque< std::pair<QString,QString> >::const_iterator it = elementTagOfPairSet.begin() ;
197 it != elementTagOfPairSet.end() ; it++ )
198 {
199 const QString elementTagName = (*it).first;
200 const QString attributeTagName = (*it).second;
201 const auto ite = m_elementTagOfPair2DocumentPropertyType.find( elementTagName );
202 if( ite != m_elementTagOfPair2DocumentPropertyType.end() ){
203 if( ite->second.getId() != attr.getId() ) {
204 DRLOGINIT;
205 LWARN << "ElementTag" << elementTagName << "AttributeTagName" << attributeTagName << "cannot be used for"<< attr.getId() <<", it is already used for Document Property" << ite->second.getId() << ". Skip";
206 }
207 continue;
208 }
209#ifdef DEBUG_LP
210 LDEBUG << "ElementTag" << elementTagName << "AttributeTagName" << attributeTagName << "is used for property" << attr.getId();
211#endif
212 m_elementTagOfPair2DocumentPropertyType.insert ( std::make_pair( elementTagName, attr ) );
213 }
214 }
215 }
216 catch ( NoSuchList & e )
217 {
218 DRLOGINIT;
219 LWARN << "StructuredDocumentXMLParser: NoSuchList Exception: standard-properties-list";
220 }
221
222 // Read list of specific attribute
223 try
224 {
225 deque<string> extendedPropertyList= groupConf.getListsValueAtKey ( "extended-properties-list" );
226 // get list of identifier of attributes from config file
227 for ( deque<string>::const_iterator attributeIdentifier = extendedPropertyList.begin();
228 attributeIdentifier != extendedPropertyList.end(); attributeIdentifier++ )
229 {
230 // build attribute from identifier
231 DocumentPropertyType attr = *DocumentsReaderResources::single().getProperty ( *attributeIdentifier );
232 // add new property to map elementTag->property*
233 const std::deque<QString> elementTagSet = attr.getElementTagNames();
234 for ( std::deque<QString>::const_iterator it = elementTagSet.begin() ;
235 it != elementTagSet.end() ; it++ )
236 {
237 const QString elementTagName = *it;
238 const auto ite = m_elementTag2DocumentPropertyType.find( elementTagName );
239 if( ite != m_elementTag2DocumentPropertyType.end() ){
240 if( ite->second.getId() != attr.getId() ) {
241 DRLOGINIT;
242 LWARN << "ElementTag" << elementTagName << "cannot be used for"<< attr.getId() <<", it is already used for Document Property" << ite->second.getId() << ". Skip";
243 }
244 continue;
245 }
246#ifdef DEBUG_LP
247 LDEBUG << "ElementTag" << elementTagName << "is used for property" << attr.getId();
248#endif
249
250 m_elementTag2DocumentPropertyType.insert ( std::make_pair( elementTagName, attr ) );
251 }
252 // add property to map elementTag.attribute -> property*
253 const std::deque< std::pair<QString,QString> > elementTagOfPairSet = attr.getAttributeTagNames();
254 for ( std::deque< std::pair<QString,QString> >::const_iterator it = elementTagOfPairSet.begin() ;
255 it != elementTagOfPairSet.end() ; it++ )
256 {
257 const QString elementTagName = (*it).first;
258 const QString attributeTagName = (*it).second;
259 const auto ite = m_elementTagOfPair2DocumentPropertyType.find( elementTagName );
260 if( ite != m_elementTagOfPair2DocumentPropertyType.end() ){
261 if( ite->second.getId() != attr.getId() ) {
262 DRLOGINIT;
263 LWARN << "ElementTag" << elementTagName << "AttributeTagName" << attributeTagName << "cannot be used for"<< attr.getId() <<", it is already used for Document Property" << ite->second.getId() << ". Skip";
264 }
265 continue;
266 }
267#ifdef DEBUG_LP
268 LDEBUG << "ElementTag" << elementTagName << "AttributeTagName" << attributeTagName << "is used for property" << attr.getId();
269#endif
270 m_elementTagOfPair2DocumentPropertyType.insert ( std::make_pair( elementTagName, attr ) );
271 }
272 }
273 }
274 catch ( NoSuchList & e )
275 {
276 DRLOGINIT;
277 LWARN << "StructuredDocumentXMLParser: NoSuchList Exception: extended-properties-list";
278 }
279
280}
281
282void StructuredDocumentXMLParser::readFieldTags (
283 GroupConfigurationStructure& groupConf, const Lima::DocumentsReader::FieldType type, vector< pair< QString, Lima::DocumentsReader::FieldType > >& fields )
284{
285#ifdef DEBUG_LP
286 DRLOGINIT;
287 LDEBUG << "StructuredDocumentXMLParser::readFieldTags" << type;
288#endif
289 QString errMess ( "" );
290
291 const QString& fieldTagListName=tagSemanticItem[type];
292#ifdef DEBUG_LP
293 LDEBUG << "StructuredDocumentXMLParser::readFieldTags fieldTagListName:" << fieldTagListName;
294#endif
295
296 // Read list of tag for XML root
297 try
298 {
299 deque<string> tagList= groupConf.getListsValueAtKey ( fieldTagListName.toUtf8().constData() );
300 for ( deque<string>::const_iterator tag=tagList.begin();
301 tag!=tagList.end(); tag++ )
302 {
303 fields.push_back ( make_pair ( QString::fromUtf8(tag->c_str()),type ) );
304 }
305 }
306 catch ( NoSuchGroup & e )
307 {
308 errMess = QString ( "NoSuchGroup Exception: tagSemantic" );
309 //errMess = e.what();
310 }
311 catch ( NoSuchList & e )
312 {
313 errMess = QString ( "NoSuchList Exception: " ) +fieldTagListName;
314 //errMess = e.what();
315 }
316 if ( errMess.compare ( "" ) )
317 {
318#ifdef DEBUG_LP
319 DRLOGINIT;
320 LWARN << "DocumentReader: " << errMess;
321#endif
322 }
323}
324
325void StructuredDocumentXMLParser::addField ( const QString& fieldName, const Lima::DocumentsReader::FieldType type )
326{
327#ifdef DEBUG_LP
328 DRLOGINIT;
329 LDEBUG << "StructuredDocumentXMLParser::addField" << fieldName << type;
330#endif
331 //uint64_t spaceOffset=fieldName.find(' '); portage 32 64
332 int spaceOffset=fieldName.indexOf ( ' ' );
333 if ( spaceOffset == -1 )
334 {
335 m_fields.insert ( std::pair<QString,FieldTypeElement> ( fieldName,FieldTypeElement ( type ) ) );
336 }
337 else
338 {
339 int equalOffset=fieldName.indexOf ( '=',spaceOffset+1 );
340 QString attributeName ( fieldName.mid(spaceOffset+1,equalOffset-spaceOffset-1) );
341 QString newFieldName ( fieldName.left(spaceOffset) );
342
343 m_fields.insert ( std::pair<QString,FieldTypeElement> ( newFieldName,FieldTypeElement ( type,attributeName ) ) );
344 }
345}
346
347// -----------------------------------------------------------------------
348// Implementations of the SAX DocumentHandler interface
349// -----------------------------------------------------------------------
350bool StructuredDocumentXMLParser::startDocument(unsigned int parserOffset)
351{
352#ifdef DEBUG_LP
353 DRLOGINIT;
354 LDEBUG << "StructuredDocumentXMLParser: startDocument()";
355#endif
356// parserOffset = m_shiftFrom->correct_offset(0, parserOffset);
357 m_currentDocument = boost::shared_ptr< ContentStructuredDocument >(new ContentStructuredDocument);
358 // creation d'un element artificiel = ROOT
359 DocumentPropertyType noProperty;
360 QString rootName ( "ROOT" );
361 m_currentDocument->pushHierarchyChild ( rootName, m_shiftFrom->correct_offset(0, parserOffset), noProperty );
362 setCurrentByteOffset ( parserOffset );
363 return true;
364}
365
367{
368#ifdef DEBUG_LP
369 DRLOGINIT;
370 LDEBUG << "StructuredDocumentXMLParser: endDocument()";
371#endif
372
373// assert( m_currentDocument->size() == 1);
374// AbstractStructuredDocumentElement* currentElement = m_currentDocument->back();
375// delete currentElement;
376// currentElement->pop_back();
377// delete m_currentDocument; m_currentDocument = 0;
378 return true;
379}
380
381// -----------------------------------------------------------------------
382// Implementations of the SAX DocumentHandler interface
383// -----------------------------------------------------------------------
384
385bool StructuredDocumentXMLParser::startElement ( const QString& namespaceURI, const QString& name, const QString& qName, const QXmlStreamAttributes& attributes, unsigned int parserOffset )
386{
387#ifdef DEBUG_LP
388 DRLOGINIT;
389#endif
390 LIMA_UNUSED(namespaceURI)
391 LIMA_UNUSED(qName)
392
393#ifdef DEBUG_LP
394 LDEBUG << "StructuredDocumentXMLParser::startElement(" << name << parserOffset << "), document.size="
395 << m_currentDocument->size();
396#endif
397
398 AbstractStructuredDocumentElement* newElement = 0;
399
400 // Trois situations contextuelles:
401 // - la pile est vide,
402 // - le dernier element est "discardable" ou
403 // - il y a un element "non discardable"
404
405#ifdef DEBUG_LP
406 assert(!m_currentDocument->empty());
407#endif
408
409
410 DocumentPropertyType propType;
411 if ( isMetaData ( name ) )
412 {
413#ifdef DEBUG_LP
414 LDEBUG << "StructuredDocumentXMLParser::startElement: " << name << " is metadata" ;
415#endif
416 // recupere la propriete liee a ce tag
417 propType = getMetaDataFromName ( name );
418 }
419
420
421 AbstractStructuredDocumentElement* currentElement = m_currentDocument->back();
422 switch (currentElement->nodeType())
423 {
424 case NODE_DISCARDABLE:
425 // current element is discardable
426#ifdef DEBUG_LP
427 LDEBUG << "StructuredDocumentXMLParser::startElement: current" << currentElement->getElementName() << "is discardable, discards element " << name;
428#endif
429 newElement = m_currentDocument->pushDiscardableChild ( name, parserOffset );
430 break;
431 // current element is indexing
432 case NODE_INDEXING:
433#ifdef DEBUG_LP
434 LDEBUG << "StructuredDocumentXMLParser::startElement: current" << currentElement->getElementName() << "is indexing";
435#endif
436 // field is metadata (exemple: <TITLE>document 1</TITLE> or <filename>d1.xml<filename>
437 // exemple: <USELESS>.....</USELESS>
438 if ( isDiscardable ( name ) )
439 {
440#ifdef DEBUG_LP
441 LDEBUG << "StructuredDocumentXMLParser::startElement:" << name << "discardable element" ;
442#endif
443 // Element discardable. Don't move
444 newElement = m_currentDocument->pushDiscardableChild ( name, parserOffset );
445 }
446 else // current element is indexing. All its not discardable children are presentation
447 {
448#ifdef DEBUG_LP
449 LDEBUG << "StructuredDocumentXMLParser::startElement:" << name << "presentation element" ;
450#endif
451 newElement = m_currentDocument->pushPresentationChild ( name, parserOffset );
452 }
453 break;
454 // current element is hierarchy
455 case NODE_HIERARCHY:
456 // hierarchy elements can have indexing, hierarchy and ignored children
457#ifdef DEBUG_LP
458 LDEBUG << "StructuredDocumentXMLParser::startElement: current" << currentElement->getElementName() << "is hierarchy";
459#endif
460 if ( isHierarchy ( name ) )
461 {
462#ifdef DEBUG_LP
463 LDEBUG << "StructuredDocumentXMLParser::startElement: " << name << "hierarchy element" ;
464#endif
465// parserOffset = m_shiftFrom->correct_offset(0, parserOffset);
466 newElement = m_currentDocument->pushHierarchyChild ( name, m_shiftFrom->correct_offset(0, parserOffset), propType );
467 m_processor->startHierarchy ( *m_currentDocument );
468 }
469 else if ( isIndexing ( name ) )
470 {
471#ifdef DEBUG_LP
472 LDEBUG << "StructuredDocumentXMLParser::startElement: " << name << "indexing element" ;
473#endif
474 newElement = m_currentDocument->pushIndexingChild ( name, parserOffset, propType );
475 m_processor->startIndexing ( *m_currentDocument );
476 }
477 else
478 {
479#ifdef DEBUG_LP
480 LDEBUG << "StructuredDocumentXMLParser::startElement: " << name << "ignored element" ;
481#endif
482 newElement = m_currentDocument->pushIgnoredChild ( name, parserOffset, propType );
483 }
484 break;
486 // presentation elements can only have discardable and presentation children
487#ifdef DEBUG_LP
488 LDEBUG << "StructuredDocumentXMLParser::startElement: current" << currentElement->getElementName() << "is presentation";
489#endif
490 if ( isDiscardable ( name ) )
491 {
492#ifdef DEBUG_LP
493 LDEBUG << "StructuredDocumentXMLParser::startElement: " << name << " is discardable " ;
494#endif
495 // Element discardable. Don't move
496 newElement = m_currentDocument->pushDiscardableChild ( name, parserOffset );
497 }
498 else // child is presentation
499 {
500 currentElement->addSpaces(parserOffset-currentElement->getOffset());
501 newElement = m_currentDocument->pushPresentationChild ( name, parserOffset );
502 }
503 break;
504 case NODE_IGNORED:
505 // all ignored elements children are also ignored
506#ifdef DEBUG_LP
507 LDEBUG << "StructuredDocumentXMLParser::startElement: current" << currentElement->getElementName() << "is ignored. Ignoring also" << name;
508#endif
509 newElement = m_currentDocument->pushIgnoredChild ( name, parserOffset, propType );
510 break;
511 default:
512#ifdef DEBUG_LP
513 LDEBUG << "StructuredDocumentXMLParser::startElement: all inheritance cases are handled for current" << currentElement->getElementName() << ". We should not get here while handling" << name;
514 // All inheritance cases are handled. We should not get here
515 assert(false);
516#endif
517 ;
518 }
519#ifdef DEBUG_LP
520 assert ( newElement );
521#endif
522 if ( hasMetaData ( name ) && newElement->nodeType() != NODE_DISCARDABLE )
523 {
524 // TODO: changer l'implementation, trop complique!
525 // field has metadata (example: <document id="number_one">qsdfhqmsfjh</document>
526 #ifdef DEBUG_LP
527 LDEBUG << "StructuredDocumentXMLParser::startElement: " << name << " has metadata";
528#endif
529 auto range = m_elementTagOfPair2DocumentPropertyType.equal_range ( name );
530 // iteration sur toutes les proprietes suggerees par le nom de l'element
531 for ( auto propIt = range.first ; propIt != range.second ; ( propIt ) ++ )
532 {
533 DocumentPropertyType& propType = ( *propIt ).second;
534#ifdef DEBUG_LP
535 LDEBUG << "StructuredDocumentXMLParser::startElement: try " << propType.getId();
536#endif
537 auto attrAndElementNames = propType.getAttributeTagNames();
538 for (auto namesIt = attrAndElementNames.begin(); namesIt != attrAndElementNames.end() ; namesIt++ )
539 {
540 if ( ! ( *namesIt ).first.compare ( name ) )
541 {
542 // field contain metadata in attribute
543 QString attributeName = (*namesIt ).second;
544 Lima::LimaString lic2mValue=attributes.value ( attributeName ).toString();
545 std::string utf8Value = Misc::limastring2utf8stdstring ( lic2mValue );
546#ifdef DEBUG_LP
547 LDEBUG << "StructuredDocumentXMLParser::startElement: found " << ( *namesIt ).first
548 << " as element with attribute " << ( *namesIt ).second
549 << " and value " << utf8Value;
550#endif
551 m_currentDocument->setDataToLastElement ( propType, utf8Value, m_processor );
552 }
553 }
554 }
555 }
556
557#ifdef DEBUG_LP
558 LDEBUG << "StructuredDocumentXMLParser::startElement() end";
559#endif
560 return true;
561}
562
563bool StructuredDocumentXMLParser::endElement(const QString& namespaceURI,
564 const QString& qsname,
565 const QString& qName ,
566 unsigned int parserOffset)
567{
568#ifdef DEBUG_LP
569 DRLOGINIT;
570#endif
571 LIMA_UNUSED(namespaceURI)
572 LIMA_UNUSED(qName)
573
574 AbstractStructuredDocumentElement* currentElement = m_currentDocument->back();
575#ifdef DEBUG_LP
576 LDEBUG << "StructuredDocumentXMLParser::endElement" << qsname << parserOffset
577 << ". currentElement=" << currentElement->getElementName();
578
579 assert ( currentElement->getElementName() == qsname );
580#endif
581 switch (currentElement->nodeType())
582 {
583 case NODE_DISCARDABLE:
584#ifdef DEBUG_LP
585 LDEBUG << "StructuredDocumentXMLParser::endElement: pop discardable element " << qsname;
586#endif
587 m_currentDocument->popDiscardableElement(parserOffset);
588#ifdef DEBUG_LP
589 LDEBUG << "StructuredDocumentXMLParser::endElement(" << qsname
590 << "), document.size = " << m_currentDocument->size() << " before return";
591#endif
592 return true;
593 break;
594 case NODE_INDEXING:
595#ifdef DEBUG_LP
596 LDEBUG << "StructuredDocumentXMLParser::endElement: pop indexing element " << qsname;
597 assert(currentElement->size() > 0);
598#endif
599
600 try {
601 m_processor->handle(*m_currentDocument, currentElement->front()->getText(),
602 m_addAbsoluteOffsetToTokens ? currentElement->front()->getOffset() : 0,
603 qsname.toUtf8().constData());
604 }
606 {
607 DRLOGINIT;
608 LERROR << "StructuredDocumentXMLParser::endElement: error while handling indexing element"
609 << qsname << "absolute offset:" << currentElement->front()->getOffset() << ". The content will be ignored.";
610 }
611
612#ifdef DEBUG_LP
613 LDEBUG << "StructuredDocumentXMLParser::endElement: pop indexing element handled" << qsname;
614#endif
615 m_processor->endIndexing ( *m_currentDocument ); //m_processor = CoreXmlReaderClient
616#ifdef DEBUG_LP
617 LDEBUG << "StructuredDocumentXMLParser::endElement: pop indexing element ended" << qsname;
618#endif
619 m_currentDocument->popIndexingElement(parserOffset);
620#ifdef DEBUG_LP
621 LDEBUG << "StructuredDocumentXMLParser::endElement: pop indexing element poped" << qsname;
622#endif
623 return true;
624 break;
625 case NODE_HIERARCHY:
626#ifdef DEBUG_LP
627 LDEBUG << "StructuredDocumentXMLParser::endElement: pop hierarchy element " << qsname;
628#endif
629 m_processor->endHierarchy ( *m_currentDocument );
630 m_currentDocument->popHierarchyElement(parserOffset);
631 return true;
632 break;
634#ifdef DEBUG_LP
635 LDEBUG << "StructuredDocumentXMLParser::endElement: pop presentation element " << qsname;
636#endif
637 m_currentDocument->popPresentationElement ( parserOffset );
638#ifdef DEBUG_LP
639 LDEBUG << "StructuredDocumentXMLParser::endElement(" << qsname << "), document.size = "
640 << m_currentDocument->size() << " before return";
641#endif
642 return true;
643 break;
644 case NODE_PROPERTY:
645#ifdef DEBUG_LP
646 LDEBUG << "StructuredDocumentXMLParser::endElement: pop property element " << qsname;
647#endif
648 if (currentElement->getPropType().getValueCardinality() != CARDINALITY_NONE && !currentElement->empty() )
649 m_currentDocument->setDataToElement(currentElement, currentElement->getPropType(), currentElement->back()->getText().toUtf8().constData(), m_processor);
650 m_currentDocument->popPropertyElement( parserOffset );
651#ifdef DEBUG_LP
652 LDEBUG << "StructuredDocumentXMLParser::endElement(" << qsname << "), document.size = "
653 << m_currentDocument->size() << " before return";
654#endif
655 return true;
656 break;
657 case NODE_IGNORED:
658#ifdef DEBUG_LP
659 LDEBUG << "StructuredDocumentXMLParser::endElement: pop ignored element " << qsname;
660#endif
661 if (currentElement->getPropType().getValueCardinality() != CARDINALITY_NONE && !currentElement->empty() )
662 m_currentDocument->setDataToElement(currentElement, currentElement->getPropType(), currentElement->back()->getText().toUtf8().constData(), m_processor);
663 m_currentDocument->popIgnoredElement ( parserOffset );
664#ifdef DEBUG_LP
665 LDEBUG << "StructuredDocumentXMLParser::endElement(" << qsname << "), document.size = "
666 << m_currentDocument->size() << " before return";
667#endif
668 return true;
669 break;
670 default:
671 // All inheritance cases are handled. We should not get here
672 DRLOGINIT;
673 LERROR << "All inheritance cases are handled. We should not get here";
674 assert(false);
675 return false;
676 }
677 return true;
678}
679
680
681// -----------------------------------------------------------------------
682// Handling of offset
683// -----------------------------------------------------------------------
684
685void StructuredDocumentXMLParser::setCurrentByteOffset ( const unsigned int offset )
686{
687#ifdef DEBUG_LP
688 DRLOGINIT;
689 LDEBUG << "StructuredDocumentXMLParser::setCurrentByteOffset(" << offset << ")";
690#else
691 LIMA_UNUSED(offset);
692#endif
693 if ( m_currentDocument->size() != 0 )
694 {
695 // AbstractStructuredDocumentElement* currentElement = m_currentDocument->back();
696 // currentElement->moveByteReaderPosTo ( offset );
697 }
698}
699
700//**********************************************************************
702 unsigned int parserOffset)
703{
704 QString value = ch;
705#ifdef DEBUG_LP
706 DRLOGINIT;
707 LDEBUG << "StructuredDocumentXMLParser::characters" << value.left(50)
708 << "(...), length=" << value.size() << ", parserOffset=" << parserOffset;
709#endif
710 auto currentElement = m_currentDocument->back();
711 currentElement->setOffset(parserOffset);
712 if ( !value.isEmpty() )
713 {
714 Lima::LimaChar firstChar=value[0];
715 if ( m_shiftFrom->contains(parserOffset) && isSpecialCharacter ( firstChar ) )
716 {
717#ifdef DEBUG_LP
718 LDEBUG << "StructuredDocumentXMLParser::characters m_shiftFrom:";
719 LDEBUG << "StructuredDocumentXMLParser::characters: first char "
720 << firstChar << " is special character: add "
721 << getSpecialCharSize ( firstChar )-1 << " spaces";
722#endif
723 Lima::LimaString spaces ( getSpecialCharSize ( firstChar )-1,' ' );
724 value.insert ( 1,spaces );
725 }
726 else
727 {
728#ifdef DEBUG_LP
729 LDEBUG << "StructuredDocumentXMLParser::characters: first char "
730 << firstChar << "(" << firstChar
731 << ") is not a special character";
732#endif
733 }
734 currentElement->addToCurrentOffset ( value );
735 }
736
737 return true;
738}
739
740bool StructuredDocumentXMLParser::isFieldType ( const QString& elementName, const Lima::DocumentsReader::FieldType& type ) const
741{
742 auto range=m_fields.equal_range ( elementName );
743 for ( auto it=range.first; it!=range.second; it++ )
744 {
745 if ( ( *it ).second.getType() == type )
746 {
747 return true;
748 }
749 }
750 return false;
751}
752
753bool StructuredDocumentXMLParser::isHierarchy ( const QString& elementName ) const
754{
755 return isFieldType ( elementName,NODE_HIERARCHY );
756}
757
758bool StructuredDocumentXMLParser::isIndexing ( const QString& elementName ) const
759{
760 return isFieldType ( elementName,NODE_INDEXING );
761}
762
763bool StructuredDocumentXMLParser::isRoot ( const QString& elementName ) const
764{
765 return isFieldType ( elementName,NODE_ROOT );
766}
767
768bool StructuredDocumentXMLParser::isDiscardable ( const QString& elementName ) const
769{
770 return isFieldType ( elementName,NODE_DISCARDABLE );
771}
772
773// test si le nom de l'element correspond a
774// 1) un element ayant des attributs dont la valeur deviendra une propriete du document
775// (ex <file source="doc0.txt"> ou <document id="d1"> ...</document>
776bool StructuredDocumentXMLParser::hasMetaData ( const QString& elementName ) const
777{
778 auto range=m_elementTagOfPair2DocumentPropertyType.equal_range ( elementName );
779
780 if ( range.first != range.second )
781 {
782 return true;
783 }
784 else
785 {
786 return false;
787 }
788}
789
790// test si le nom de l'element correspond a
791// 2) un element contenant du texte dont la valeur est la propriete du document
792// (ex <filename>doc0.txt<filename>)
793bool StructuredDocumentXMLParser::isMetaData ( const QString& elementName ) const
794{
795 auto pos = m_elementTag2DocumentPropertyType.find ( elementName );
796 if ( pos != m_elementTag2DocumentPropertyType.end() )
797 {
798 return true;
799 }
800 else
801 {
802 return false;
803 }
804}
805
806// retourne la propriete associee au nom de l'element
807DocumentPropertyType StructuredDocumentXMLParser::getMetaDataFromName ( const QString& elementName ) const
808{
809
810 auto pos = m_elementTag2DocumentPropertyType.find ( elementName );
811 if ( pos != m_elementTag2DocumentPropertyType.end() )
812 {
813#ifdef DEBUG_LP
814 DRLOGINIT;
815 LDEBUG << "StructuredDocumentXMLParser::getMetaDataFromName(" << elementName << ")="
816 << ( ( *pos ).second ).getId();
817#endif
818 return ( *pos ).second;
819 }
820 else
821 {
822 DRLOGINIT;
823 LWARN << "StructuredDocumentXMLParser::getMetaDataFromName: no property associated with element name "
824 << elementName;
825 return DocumentPropertyType();
826 }
827}
828
829
830} // namespace DocumentsReader
831} // namespace Lima
832
#define LWARN
Definition LimaCommon.h:160
#define LIMA_UNUSED(x)
Definition LimaCommon.h:224
#define LDEBUG
Definition LimaCommon.h:157
#define LINFO
Definition LimaCommon.h:158
#define LERROR
Definition LimaCommon.h:161
Defines a Factory to create Object of type Base.
#define DRLOGINIT
std::deque< std::string > & getListsValueAtKey(const std::string &key)
return a message when a 'group' was not found
return a message when a 'param' was not found
virtual const DocumentPropertyType & getPropType()
Partie d'un document structure decodee avant analyse.
const std::deque< std::pair< QString, QString > > & getAttributeTagNames() const
const CardinalityType & getValueCardinality() const
void init(Lima::Common::XMLConfigurationFiles::GroupConfigurationStructure &unitConfiguration, Manager *manager) override
bool characters(const QString &ch, unsigned int parserOffset)
bool endElement(const QString &namespaceURI, const QString &qsname, const QString &qName, unsigned int parserOffset)
void setShiftFrom(std::shared_ptr< const ShiftFrom > shiftFrom)
bool startElement(const QString &namespaceURI, const QString &name, const QString &qName, const QXmlStreamAttributes &attributes, unsigned int parserOffset)
Manage initialization of InitializableObjects using configuration module and parameters.
const std::string & getId() const
get the object id
static const DocumentsReaderResources & single()
const singleton accessor
Definition Singleton.h:51
virtual void startIndexing(const DocumentsReader::ContentStructuredDocument &contentDocument)=0
virtual void handle(const DocumentsReader::ContentStructuredDocument &contentDocument, const Lima::LimaString &text, unsigned long int offset, std::string langOfAnalysis)=0
virtual void endHierarchy(const DocumentsReader::ContentStructuredDocument &contentDocument)=0
virtual void endIndexing(const DocumentsReader::ContentStructuredDocument &contentDocument)=0
virtual void startHierarchy(const DocumentsReader::ContentStructuredDocument &contentDocument)=0
std::string limastring2utf8stdstring(const Lima::LimaString &phrase, uint32_t size0)
Convert a wide string to a string , in dest up to size bytes.
static const QString tagSemanticItem[MAX_NODE_TYPE]
FieldType
a type containing the type of an element
@ NODE_DISCARDABLE
root of XML file
@ NODE_PRESENTATION
node that can contain indexing nodes in its hierarchy.
@ NODE_ROOT
no type: field is ignored
@ NODE_INDEXING
node inside an indexing node whose content (including its children) is replaced by spaces
@ NODE_HIERARCHY
node whose content is analyzed and indexed
@ NODE_PROPERTY
nodes which are not children of hierarchy nodes are ignored
@ NODE_IGNORED
node inside an indexing node with content replaced by spaces, but which content is kept
Lima::SimpleFactory< AbstractReaderResource, Lima::DocumentsReader::StructuredDocumentXMLParser > structuredDocumentXMLParserFactory(STRUCTUREDDOCUMENTXMLPARSER_CLASSID)
NAUTITIA.
QChar LimaChar
Definition LimaString.h:30
QString LimaString
Definition LimaString.h:33
STL namespace.
#define STRUCTUREDDOCUMENTXMLPARSER_CLASSID
#define MAX_NODE_TYPE