2 * \file output_xhtml.cpp
3 * This file is part of LyX, the document processor.
4 * Licence details can be found in the file COPYING.
8 * This code is based upon output_docbook.cpp
10 * Full author contact details are available in file CREDITS.
15 #include "output_xhtml.h"
18 #include "buffer_funcs.h"
19 #include "BufferParams.h"
22 #include "OutputParams.h"
23 #include "Paragraph.h"
24 #include "ParagraphList.h"
25 #include "ParagraphParameters.h"
28 #include "TextClass.h"
30 #include "support/lassert.h"
31 #include "support/debug.h"
32 #include "support/lstrings.h"
37 using namespace lyx::support;
43 docstring escapeChar(char_type c)
67 // escape what needs escaping
68 docstring htmlize(docstring const & str) {
70 docstring::const_iterator it = str.begin();
71 docstring::const_iterator en = str.end();
72 for (; it != en; ++it)
78 string escapeChar(char c)
102 // escape what needs escaping
103 string htmlize(string const & str) {
105 string::const_iterator it = str.begin();
106 string::const_iterator en = str.end();
107 for (; it != en; ++it)
108 d << escapeChar(*it);
113 string cleanAttr(string const & str)
116 string::const_iterator it = str.begin();
117 string::const_iterator en = str.end();
118 for (; it != en; ++it)
119 newname += isalnum(*it) ? *it : '_';
124 docstring cleanAttr(docstring const & str)
127 docstring::const_iterator it = str.begin();
128 docstring::const_iterator en = str.end();
129 for (; it != en; ++it)
138 bool isFontTag(string const & s)
140 return s == "em" || s == "strong"; // others?
145 docstring StartTag::asTag() const
147 string output = "<" + tag_;
149 output += " " + html::htmlize(attr_);
151 return from_utf8(output);
155 docstring StartTag::asEndTag() const
157 string output = "</" + tag_ + ">";
158 return from_utf8(output);
162 docstring EndTag::asEndTag() const
164 string output = "</" + tag_ + ">";
165 return from_utf8(output);
169 docstring CompTag::asTag() const
171 string output = "<" + tag_;
173 output += " " + html::htmlize(attr_);
175 return from_utf8(output);
179 ////////////////////////////////////////////////////////////////
183 ////////////////////////////////////////////////////////////////
185 XHTMLStream::XHTMLStream(odocstream & os)
186 : os_(os), nextraw_(false)
190 void XHTMLStream::cr()
193 os_ << from_ascii("\n");
197 void XHTMLStream::writeError(std::string const & s)
200 os_ << from_utf8("<!-- Output Error: " + s + " -->");
204 bool XHTMLStream::closeFontTags()
206 if (tag_stack_.empty())
208 // first, we close any open font tags we can close
209 StartTag curtag = tag_stack_.back();
210 while (html::isFontTag(curtag.tag_)) {
211 os_ << curtag.asEndTag();
212 tag_stack_.pop_back();
213 if (tag_stack_.empty())
214 // this probably shouldn't happen, since then the
215 // font tags weren't in any other tag. but that
216 // problem will likely be caught elsewhere.
218 curtag = tag_stack_.back();
220 // so we've hit a non-font tag. let's see if any of the
221 // remaining tags are font tags.
222 TagStack::const_iterator it = tag_stack_.begin();
223 TagStack::const_iterator en = tag_stack_.end();
224 bool noFontTags = true;
225 for (; it != en; ++it) {
226 if (html::isFontTag(it->tag_)) {
227 writeError("Font tag `" + it->tag_ + "' still open in closeFontTags().");
235 void XHTMLStream::clearTagDeque()
237 while (!pending_tags_.empty()) {
238 StartTag const & tag = pending_tags_.front();
241 tag_stack_.push_back(tag);
242 pending_tags_.pop_front();
247 XHTMLStream & XHTMLStream::operator<<(docstring const & d)
254 os_ << html::htmlize(d);
259 XHTMLStream & XHTMLStream::operator<<(const char * s)
262 docstring const d = from_ascii(s);
267 os_ << html::htmlize(d);
272 XHTMLStream & XHTMLStream::operator<<(char_type c)
279 os_ << html::escapeChar(c);
284 XHTMLStream & XHTMLStream::operator<<(NextRaw const &)
291 XHTMLStream & XHTMLStream::operator<<(StartTag const & tag)
293 if (tag.tag_.empty())
295 pending_tags_.push_back(tag);
302 XHTMLStream & XHTMLStream::operator<<(CompTag const & tag)
304 if (tag.tag_.empty())
314 bool XHTMLStream::isTagOpen(string const & stag)
316 TagStack::const_iterator sit = tag_stack_.begin();
317 TagStack::const_iterator const sen = tag_stack_.end();
318 for (; sit != sen; ++sit)
319 // we could check for the
320 if (sit->tag_ == stag)
326 // this is complicated, because we want to make sure that
327 // everything is properly nested. the code ought to make
328 // sure of that, but we won't assert (yet) if we run into
329 // a problem. we'll just output error messages and try our
330 // best to make things work.
331 XHTMLStream & XHTMLStream::operator<<(EndTag const & etag)
333 if (etag.tag_.empty())
335 // first make sure we're not closing an empty tag
336 if (!pending_tags_.empty()) {
337 StartTag const & stag = pending_tags_.back();
338 if (etag.tag_ == stag.tag_) {
339 // we have <tag></tag>, so we discard it and remove it
340 // from the pending_tags_.
341 pending_tags_.pop_back();
344 // there is a pending tag that isn't the one we are trying
346 // is this tag itself pending?
347 // non-const iterators because we may call erase().
348 TagDeque::iterator dit = pending_tags_.begin();
349 TagDeque::iterator const den = pending_tags_.end();
350 for (; dit != den; ++dit) {
351 if (dit->tag_ == etag.tag_) {
352 // it was pending, so we just erase it
353 writeError("Tried to close pending tag `" + etag.tag_
354 + "' when other tags were pending. Last pending tag is `"
355 + pending_tags_.back().tag_ + "'. Tag discarded.");
356 pending_tags_.erase(dit);
360 // so etag isn't itself pending. is it even open?
361 if (!isTagOpen(etag.tag_)) {
362 writeError("Tried to close `" + etag.tag_
363 + "' when tag was not open. Tag discarded.");
366 // ok, so etag is open.
367 // our strategy will be as below: we will do what we need to
368 // do to close this tag.
369 string estr = "Closing tag `" + etag.tag_
370 + "' when other tags are pending. Discarded pending tags:\n";
371 for (dit = pending_tags_.begin(); dit != den; ++dit)
372 estr += dit->tag_ + "\n";
374 // clear the pending tags...
375 pending_tags_.clear();
376 // ...and then just fall through.
379 // is the tag we are closing the last one we opened?
380 if (etag.tag_ == tag_stack_.back().tag_) {
382 os_ << etag.asEndTag();
383 // ...and forget about it
384 tag_stack_.pop_back();
388 // we are trying to close a tag other than the one last opened.
389 // let's first see if this particular tag is still open somehow.
390 if (!isTagOpen(etag.tag_)) {
391 writeError("Tried to close `" + etag.tag_
392 + "' when tag was not open. Tag discarded.");
396 // so the tag was opened, but other tags have been opened since
397 // and not yet closed.
398 // if it's a font tag, though...
399 if (html::isFontTag(etag.tag_)) {
400 // it won't be a problem if the other tags open since this one
401 // are also font tags.
402 TagStack::const_reverse_iterator rit = tag_stack_.rbegin();
403 TagStack::const_reverse_iterator ren = tag_stack_.rend();
404 for (; rit != ren; ++rit) {
405 if (rit->tag_ == etag.tag_)
407 if (!html::isFontTag(rit->tag_)) {
408 // we'll just leave it and, presumably, have to close it later.
409 writeError("Unable to close font tag `" + etag.tag_
410 + "' due to open non-font tag `" + rit->tag_ + "'.");
416 // <em>this is <strong>bold
417 // and are being asked to closed em. we want:
418 // <em>this is <strong>bold</strong></em><strong>
419 // first, we close the intervening tags...
420 StartTag curtag = tag_stack_.back();
421 // ...remembering them in a stack.
423 while (curtag.tag_ != etag.tag_) {
424 os_ << curtag.asEndTag();
425 fontstack.push_back(curtag);
426 tag_stack_.pop_back();
427 curtag = tag_stack_.back();
429 // now close our tag...
430 os_ << etag.asEndTag();
431 tag_stack_.pop_back();
433 // ...and restore the other tags.
434 rit = fontstack.rbegin();
435 ren = fontstack.rend();
436 for (; rit != ren; ++rit)
437 pending_tags_.push_back(*rit);
441 // it wasn't a font tag.
442 // so other tags were opened before this one and not properly closed.
443 // so we'll close them, too. that may cause other issues later, but it
444 // at least guarantees proper nesting.
445 writeError("Closing tag `" + etag.tag_
446 + "' when other tags are open, namely:");
447 StartTag curtag = tag_stack_.back();
448 while (curtag.tag_ != etag.tag_) {
449 writeError(curtag.tag_);
450 os_ << curtag.asEndTag();
451 tag_stack_.pop_back();
452 curtag = tag_stack_.back();
454 // curtag is now the one we actually want.
455 os_ << curtag.asEndTag();
456 tag_stack_.pop_back();
461 // End code for XHTMLStream
465 // convenience functions
467 inline void openTag(XHTMLStream & xs, Layout const & lay)
469 xs << StartTag(lay.htmltag(), lay.htmlattr());
473 inline void closeTag(XHTMLStream & xs, Layout const & lay)
475 xs << EndTag(lay.htmltag());
479 inline void openLabelTag(XHTMLStream & xs, Layout const & lay)
481 xs << StartTag(lay.htmllabeltag(), lay.htmllabelattr());
485 inline void closeLabelTag(XHTMLStream & xs, Layout const & lay)
487 xs << EndTag(lay.htmllabeltag());
491 inline void openItemTag(XHTMLStream & xs, Layout const & lay)
493 xs << StartTag(lay.htmlitemtag(), lay.htmlitemattr(), true);
497 inline void closeItemTag(XHTMLStream & xs, Layout const & lay)
499 xs << EndTag(lay.htmlitemtag());
502 // end of convenience functions
504 ParagraphList::const_iterator searchParagraphHtml(
505 ParagraphList::const_iterator p,
506 ParagraphList::const_iterator const & pend)
508 for (++p; p != pend && p->layout().latextype == LATEX_PARAGRAPH; ++p)
515 ParagraphList::const_iterator searchEnvironmentHtml(
516 ParagraphList::const_iterator const pstart,
517 ParagraphList::const_iterator const & pend)
519 ParagraphList::const_iterator p = pstart;
520 Layout const & bstyle = p->layout();
521 size_t const depth = p->params().depth();
522 for (++p; p != pend; ++p) {
523 Layout const & style = p->layout();
524 // It shouldn't happen that e.g. a section command occurs inside
525 // a quotation environment, at a higher depth, but as of 6/2009,
526 // it can happen. We pretend that it's just at lowest depth.
527 if (style.latextype == LATEX_COMMAND)
529 // If depth is down, we're done
530 if (p->params().depth() < depth)
532 // If depth is up, we're not done
533 if (p->params().depth() > depth)
535 // Now we know we are at the same depth
536 if (style.latextype == LATEX_PARAGRAPH
537 || style.latexname() != bstyle.latexname())
544 ParagraphList::const_iterator makeParagraphs(Buffer const & buf,
546 OutputParams const & runparams,
548 ParagraphList::const_iterator const & pbegin,
549 ParagraphList::const_iterator const & pend)
551 ParagraphList::const_iterator const begin = text.paragraphs().begin();
552 ParagraphList::const_iterator par = pbegin;
553 for (; par != pend; ++par) {
554 Layout const & lay = par->layout();
555 if (!lay.counter.empty())
556 buf.params().documentClass().counters().step(lay.counter);
557 // FIXME We should see if there's a label to be output and
558 // do something with it.
562 // If we are already in a paragraph, and this is the first one, then we
563 // do not want to open the paragraph tag.
564 // we also do not want to open it if the current layout does not permit
565 // multiple paragraphs.
566 bool const opened = runparams.html_make_pars &&
567 (par != pbegin || !runparams.html_in_par);
570 docstring const deferred =
571 par->simpleLyXHTMLOnePar(buf, xs, runparams, text.outerFont(distance(begin, par)));
573 // We want to issue the closing tag if either:
574 // (i) We opened it, and either html_in_par is false,
575 // or we're not in the last paragraph, anyway.
576 // (ii) We didn't open it and html_in_par is true,
577 // but we are in the first par, and there is a next par.
578 ParagraphList::const_iterator nextpar = par;
580 bool const needclose =
581 (opened && (!runparams.html_in_par || nextpar != pend))
582 || (!opened && runparams.html_in_par && par == pbegin && nextpar != pend);
587 if (!deferred.empty()) {
588 xs << XHTMLStream::NextRaw() << deferred;
596 ParagraphList::const_iterator makeBibliography(Buffer const & buf,
598 OutputParams const & runparams,
600 ParagraphList::const_iterator const & pbegin,
601 ParagraphList::const_iterator const & pend)
603 xs << StartTag("h2", "class='bibliography'");
604 xs << pbegin->layout().labelstring(false);
607 xs << StartTag("div", "class='bibliography'");
609 makeParagraphs(buf, xs, runparams, text, pbegin, pend);
615 bool isNormalEnv(Layout const & lay)
617 return lay.latextype == LATEX_ENVIRONMENT
618 || lay.latextype == LATEX_BIB_ENVIRONMENT;
622 ParagraphList::const_iterator makeEnvironmentHtml(Buffer const & buf,
624 OutputParams const & runparams,
626 ParagraphList::const_iterator const & pbegin,
627 ParagraphList::const_iterator const & pend)
629 ParagraphList::const_iterator const begin = text.paragraphs().begin();
630 ParagraphList::const_iterator par = pbegin;
631 Layout const & bstyle = par->layout();
632 depth_type const origdepth = pbegin->params().depth();
634 // open tag for this environment
638 // we will on occasion need to remember a layout from before.
639 Layout const * lastlay = 0;
641 while (par != pend) {
642 Layout const & style = par->layout();
643 // the counter only gets stepped if we're in some kind of list,
644 // or if it's the first time through.
645 // note that enum, etc, are handled automatically.
646 // FIXME There may be a bug here about user defined enumeration
647 // types. If so, then we'll need to take the counter and add "i",
648 // "ii", etc, as with enum.
649 Counters & cnts = buf.params().documentClass().counters();
650 docstring const & cntr = style.counter;
651 if (!style.counter.empty()
652 && (par == pbegin || !isNormalEnv(style))
653 && cnts.hasCounter(cntr)
656 ParagraphList::const_iterator send;
657 // this will be positive, if we want to skip the initial word
658 // (if it's been taken for the label).
661 switch (style.latextype) {
662 case LATEX_ENVIRONMENT:
663 case LATEX_LIST_ENVIRONMENT:
664 case LATEX_ITEM_ENVIRONMENT: {
665 // There are two possiblities in this case.
666 // One is that we are still in the environment in which we
667 // started---which we will be if the depth is the same.
668 if (par->params().depth() == origdepth) {
669 LASSERT(bstyle == style, /* */);
671 closeItemTag(xs, *lastlay);
674 if (isNormalEnv(style)) {
675 // in this case, we print the label only for the first
676 // paragraph (as in a theorem).
677 openItemTag(xs, style);
678 if (par == pbegin && style.htmllabeltag() != "NONE") {
679 docstring const lbl =
680 pbegin->expandLabel(style, buf.params(), false);
682 openLabelTag(xs, style);
684 closeLabelTag(xs, style);
688 } else { // some kind of list
689 bool const labelfirst = style.htmllabelfirst();
691 openItemTag(xs, style);
692 if (style.labeltype == LABEL_MANUAL
693 && style.htmllabeltag() != "NONE") {
694 openLabelTag(xs, style);
695 sep = par->firstWordLyXHTML(xs, runparams);
696 closeLabelTag(xs, style);
699 else if (style.labeltype != LABEL_NO_LABEL
700 && style.htmllabeltag() != "NONE") {
701 openLabelTag(xs, style);
702 xs << par->expandLabel(style, buf.params(), false);
703 closeLabelTag(xs, style);
707 openItemTag(xs, style);
709 par->simpleLyXHTMLOnePar(buf, xs, runparams,
710 text.outerFont(distance(begin, par)), false, sep);
712 // We may not want to close the tag yet, in particular,
713 // if we're not at the end...
715 // and are doing items...
716 && !isNormalEnv(style)
717 // and if the depth has changed...
718 && par->params().depth() != origdepth) {
719 // then we'll save this layout for later, and close it when
720 // we get another item.
723 closeItemTag(xs, style);
726 // The other possibility is that the depth has increased, in which
727 // case we need to recurse.
729 send = searchEnvironmentHtml(par, pend);
730 par = makeEnvironmentHtml(buf, xs, runparams, text, par, send);
734 case LATEX_PARAGRAPH:
735 send = searchParagraphHtml(par, pend);
736 par = makeParagraphs(buf, xs, runparams, text, par, send);
739 case LATEX_BIB_ENVIRONMENT:
742 par = makeParagraphs(buf, xs, runparams, text, par, send);
752 closeItemTag(xs, *lastlay);
753 closeTag(xs, bstyle);
759 void makeCommand(Buffer const & buf,
761 OutputParams const & runparams,
763 ParagraphList::const_iterator const & pbegin)
765 Layout const & style = pbegin->layout();
766 if (!style.counter.empty())
767 buf.params().documentClass().counters().step(style.counter);
771 // Label around sectioning number:
772 // FIXME Probably need to account for LABEL_MANUAL
773 if (style.labeltype != LABEL_NO_LABEL) {
774 openLabelTag(xs, style);
775 xs << pbegin->expandLabel(style, buf.params(), false);
776 closeLabelTag(xs, style);
777 // Otherwise the label might run together with the text
778 xs << from_ascii(" ");
781 ParagraphList::const_iterator const begin = text.paragraphs().begin();
782 pbegin->simpleLyXHTMLOnePar(buf, xs, runparams,
783 text.outerFont(distance(begin, pbegin)));
788 } // end anonymous namespace
791 void xhtmlParagraphs(Text const & text,
794 OutputParams const & runparams)
796 ParagraphList const & paragraphs = text.paragraphs();
797 ParagraphList::const_iterator par = paragraphs.begin();
798 ParagraphList::const_iterator pend = paragraphs.end();
800 OutputParams ourparams = runparams;
801 while (par != pend) {
802 Layout const & style = par->layout();
803 ParagraphList::const_iterator lastpar = par;
804 ParagraphList::const_iterator send;
806 switch (style.latextype) {
807 case LATEX_COMMAND: {
808 // The files with which we are working never have more than
809 // one paragraph in a command structure.
811 // if (ourparams.html_in_par)
812 // fix it so we don't get sections inside standard, e.g.
813 // note that we may then need to make runparams not const, so we
814 // can communicate that back.
815 // FIXME Maybe this fix should be in the routines themselves, in case
816 // they are called from elsewhere.
817 makeCommand(buf, xs, ourparams, text, par);
821 case LATEX_ENVIRONMENT:
822 case LATEX_LIST_ENVIRONMENT:
823 case LATEX_ITEM_ENVIRONMENT: {
824 // FIXME Same fix here.
825 send = searchEnvironmentHtml(par, pend);
826 par = makeEnvironmentHtml(buf, xs, ourparams, text, par, send);
829 case LATEX_BIB_ENVIRONMENT: {
830 // FIXME Same fix here.
831 send = searchEnvironmentHtml(par, pend);
832 par = makeBibliography(buf, xs, ourparams, text, par, send);
835 case LATEX_PARAGRAPH:
836 send = searchParagraphHtml(par, pend);
837 par = makeParagraphs(buf, xs, ourparams, text, par, send);
841 // makeEnvironment may process more than one paragraphs and bypass pend
842 if (distance(lastpar, par) >= distance(lastpar, pend))