// This file is part of The New Aspell // Copyright (C) 2004 by Tom Snyder // Copyright (C) 2001-2004 by Kevin Atkinson under the GNU LGPL license // version 2.0 or 2.1. You should have received a copy of the LGPL // license along with this library if you did not you can find // it at http://www.gnu.org/. // // The orignal filter was written by Kevin Atkinson. // Tom Snyder rewrote the filter to support skipping SGML tags // // This filter enables the spell checking of sgml, html, and xhtml files // by skipping the and such. // The overall strategy is based on http://www.w3.org/Library/SGML.c. // We don't use that code (nor the sourceforge 'expat' project code) for // simplicity's sake. We don't need to fully parse all aspects of HTML - // we just need to skip and handle a few aspects. The w3.org code had too // many linkages into their overall SGML/HTML processing engine. // // See the comment at the end of this file for examples of what we handle. // See the config setting docs regarding our config lists: check and skip. #include // needed for sprintf #include "settings.h" #include "asc_ctype.hpp" #include "config.hpp" #include "indiv_filter.hpp" #include "string_map.hpp" #include "mutable_container.hpp" #include "clone_ptr-t.hpp" #include "filter_char_vector.hpp" //right now unused option // static const KeyInfo sgml_options[] = { // {"sgml-extension", KeyInfoList, "html,htm,php,sgml", // N_("sgml file extensions")} // }; namespace { using namespace acommon; class ToLowerMap : public StringMap { public: PosibErr add(ParmStr to_add) { String new_key; for (const char * i = to_add; *i; ++i) new_key += asc_tolower(*i); return StringMap::add(new_key); } PosibErr remove(ParmStr to_rem) { String new_key; for (const char * i = to_rem; *i; ++i) new_key += asc_tolower(*i); return StringMap::remove(new_key); } }; class SgmlFilter : public IndividualFilter { // State enum. These states track where we are in the HTML/tag/element constructs. // This diagram shows the main states. The marked number is the state we enter // *after* we've read that char. Note that some of the state transitions handle // illegal HTML such as . // // real text   { // | | | | || | | | | | | // 1 2 3 4 56 7 8 10 11 9 12 enum ScanState { S_text, // 1. raw user text outside of any markup. S_tag, // 2. reading the 'tag' in S_tag_gap,// 3. gap between attributes within an element: S_attr, // 4. Looking at an attrib name S_attr_gap,// 5. optional gap after attrib name S_equals, // 6. Attrib equals sign, also space after the equals. S_value, // 7. In attrib value. S_quoted, // 8. In quoted attrib value. S_end, // 9. Same as S_tag, but this is a type end tag. S_ignore_junk, // special case invalid area to ignore. S_ero, // 10. in the &code; special encoding within HTML. S_entity, // 11. in the alpha named &nom; special encoding.. S_cro, // 12. after the # of a &#nnn; numerical char reference encoding. // SGML.. etc can have these special "declarations" within them. We skip them // in a more raw manners since they don't abide by the attrib= rules. // Most importantly, some of the attrib quoting rules don't apply. // // | | || | // 20 21 23 24 25 S_md, // 20. In a declaration (or beginning a comment). S_mdq, // 21. Declaration in quotes - double or single quotes. S_com_1, // 23. perhaps a comment beginning. S_com, // 24. Fully in a comment S_com_e, // 25. Perhaps ending a comment. //S_literal, // within a tag pair that means all content should be interpreted literally:
               // NOT CURRENTLY SUPPORTED FULLY.
               
    //S_esc,S_dollar,S_paren,S_nonasciitext // WOULD BE USED FOR ISO_2022_JP support.
                                          // NOT CURRENTLY SUPPORTED.               
    };
    
    ScanState in_what;
	     // which quote char is quoting this attrib value.
	
    FilterChar::Chr  quote_val;   
	    // one char prior to this one. For escape handling and such.
    FilterChar::Chr  lookbehind;   

    String tag_name;    // we accumulate the current tag name here.
    String attrib_name; // we accumulate the current attribute name here.
    
    bool include_attrib;  // are we in one of the attribs that *should* be spell checked (alt=..)
    int  skipall;         // are we in one of the special skip-all content tags? This is treated
			  // as a bool and as a nesting level count.
    String tag_endskip;  // tag name that will end that.
    
    StringMap check_attribs; // list of attribs that we *should* spell check.
    StringMap skip_tags;   // list of tags that start a no-check-at-all zone.

    String which;

    bool process_char(FilterChar::Chr c);
 
  public:

    SgmlFilter(const char * n) : which(n) {}

    PosibErr setup(Config *);
    void reset();
    void process(FilterChar * &, FilterChar * &);
  };

  PosibErr SgmlFilter::setup(Config * opts) 
  {
    name_ = which + "-filter";
    order_num_ = 0.35;
    check_attribs.clear();
    skip_tags.clear();
    opts->retrieve_list("f-" + which + "-skip",  &skip_tags);
    opts->retrieve_list("f-" + which + "-check", &check_attribs);
    reset();
    return true;
  }
  
  void SgmlFilter::reset() 
  {
    in_what = S_text;
    quote_val = lookbehind = '\0';
    skipall = 0;
    include_attrib = false;
  }

  // yes this should be inlines, it is only called once
  
  // RETURNS: TRUE if the caller should skip the passed char and
  //  not do any spell check on it. FALSE if char is a part of the text
  //  of the document.
  bool SgmlFilter::process_char(FilterChar::Chr c) {
  
    bool retval = true;  // DEFAULT RETURN VALUE. All returns are done
    			 // via retval and falling out the bottom. Except for
    			 // one case that must manage the lookbehind char.
    
    // PS: this switch will be fast since S_ABCs are an enum and
    //  any good compiler will build a jump table for it.
    // RE the gotos: Sometimes considered bad practice but that is
    //  how the W3C code (1995) handles it. Could be done also with recursion
    //  but I doubt that will clarify it. The gotos are done in cases where several
    //  state changes occur on a single input char.

    switch( in_what ) {
    
      case S_text:   // 1. raw user text outside of any markup.
	   s_text:
        switch( c ) {
          case '&': in_what = S_ero; 
		    break;
          case '<': in_what = S_tag; tag_name.clear(); 
		    break;
          default:
                retval = skipall;  // ********** RETVAL ASSIGNED
        }			    // **************************
        break;
        
      case S_tag:    // 2. reading the 'tag' in 
      		     //  heads up: ': goto all_end_tags;
          case '/': in_what = S_end; 
		    tag_name.clear(); 
		    break;
          case '!': in_what = S_md;  
		    break;
          default: // either more alphanum of the tag, or end of tagname:
            if( asc_isalpha(c) || asc_isdigit(c) ) {
                tag_name += asc_tolower(c);            
            }
            else {  // End of the tag:
                in_what = S_tag_gap;
		goto s_tag_gap;  // handle content in that zone.
	    }
	}
	break;

	// '>'  '>'  '>'  '>' 
      all_end_tags:   // this gets called by several states to handle the
		       // possibility of a '>' ending a whole  guy.
	if( c != '>' ) break;
	in_what = S_text;

	if( lookbehind == '/' ) {
	    // Wowza: this is how we handle the  reportthisyes
 reportthisyes
 reportthisyes
hello nonoreportreportthisyes


reportthisphaseTHREE




reportthisphaseoneFOUR



still in dontcheck\\\" still in dontcheck"> reportthisyes.
 reportthisyes.
still in dontcheck\\\' still in dontcheck'> reportthisyes.
 reportthisyes.



reportthisphaseFIVE
 reportthisyes reportthisyes

cool stuff
real

*/