diff --git a/src/Foliage-Markdown-Tests/FOHTMLRenderVisitorTest.class.st b/src/Foliage-Markdown-Tests/FOHTMLRenderVisitorTest.class.st index c313f41..c6163ad 100644 --- a/src/Foliage-Markdown-Tests/FOHTMLRenderVisitorTest.class.st +++ b/src/Foliage-Markdown-Tests/FOHTMLRenderVisitorTest.class.st @@ -171,3 +171,29 @@ FOHTMLRenderVisitorTest >> testRenderParagraphWithMixedInlineContent [ html := FOHTMLRenderVisitor render: doc. self assert: html equals: '
Text can also contain some bold.
' ] + +{ #category : #tests } +FOHTMLRenderVisitorTest >> testRenderRawHtmlBlockIsVerbatim [ + | html source | + source := ''. + html := FOHTMLRenderVisitor render: (FORawHTMLBlock new html: source; yourself). + self assert: html equals: source +] + +{ #category : #tests } +FOHTMLRenderVisitorTest >> testRenderDocumentWithRawHtmlBlock [ + | doc html iframe | + iframe := ''. + doc := FOMarkdownParser parse: 'Some text.', self nl, self nl, iframe. + html := FOHTMLRenderVisitor render: doc. + self assert: html equals: 'Some text.
', self nl, iframe +] + +{ #category : #tests } +FOHTMLRenderVisitorTest >> testRenderParagraphStillEscapesLessThan [ + "Only a line that starts like a tag becomes raw HTML; a '<' inside prose is still escaped." + | doc html | + doc := FOMarkdownParser parse: '5 < 3 and not a block'. + html := FOHTMLRenderVisitor render: doc. + self assert: html equals: '5 < 3 and <b>not a block</b>
' +] diff --git a/src/Foliage-Markdown-Tests/FOLineClassifierTest.class.st b/src/Foliage-Markdown-Tests/FOLineClassifierTest.class.st index 35a996d..8100aca 100644 --- a/src/Foliage-Markdown-Tests/FOLineClassifierTest.class.st +++ b/src/Foliage-Markdown-Tests/FOLineClassifierTest.class.st @@ -127,3 +127,54 @@ FOLineClassifierTest >> testParagraphText [ c := FOLineClassifier classify: 'Lorem ipsum dolor sit amet.'. self assert: c kind equals: #paragraphText ] + +{ #category : #tests } +FOLineClassifierTest >> testHtmlBlockOpeningTag [ + | c | + c := FOLineClassifier classify: ''. + self assert: c kind equals: #htmlBlock +] + +{ #category : #tests } +FOLineClassifierTest >> testHtmlBlockClosingTag [ + | c | + c := FOLineClassifier classify: ''. + self assert: c kind equals: #htmlBlock +] + +{ #category : #tests } +FOLineClassifierTest >> testHtmlBlockComment [ + | c | + c := FOLineClassifier classify: ''. + self assert: c kind equals: #htmlBlock +] + +{ #category : #tests } +FOLineClassifierTest >> testLessThanFollowedBySpaceIsParagraphText [ + | c | + c := FOLineClassifier classify: '5 < 3 and more text'. + self assert: c kind equals: #paragraphText. + c := FOLineClassifier classify: '< not a tag'. + self assert: c kind equals: #paragraphText +] + +{ #category : #tests } +FOLineClassifierTest >> testLessThanFollowedByDigitIsParagraphText [ + | c | + c := FOLineClassifier classify: '<3 is a heart'. + self assert: c kind equals: #paragraphText +] + +{ #category : #tests } +FOLineClassifierTest >> testLoneLessThanIsParagraphText [ + | c | + c := FOLineClassifier classify: '<'. + self assert: c kind equals: #paragraphText +] + +{ #category : #tests } +FOLineClassifierTest >> testIndentedHtmlStaysIndentedCode [ + | c | + c := FOLineClassifier classify: '', self nl, 'inner', self nl, '
'. + self assert: doc children size equals: 2. + self assert: doc children last html equals: '', self nl, 'inner', self nl, '
' +] + +{ #category : #tests } +FOMarkdownBlockParserTest >> testLessThanInParagraphIsNotAnHtmlBlock [ + | doc | + doc := FOMarkdownParser parse: '5 < 3 and more text'. + self assert: doc children size equals: 1. + self assert: (doc children first isKindOf: FOParagraph). + self assert: (self plainTextOf: doc children first) equals: '5 < 3 and more text' +] diff --git a/src/Foliage-Markdown-Tests/FOMarkdownIntegrationTest.class.st b/src/Foliage-Markdown-Tests/FOMarkdownIntegrationTest.class.st index 0bf3cc4..c261a34 100644 --- a/src/Foliage-Markdown-Tests/FOMarkdownIntegrationTest.class.st +++ b/src/Foliage-Markdown-Tests/FOMarkdownIntegrationTest.class.st @@ -168,3 +168,21 @@ FOMarkdownIntegrationTest >> testFullDocumentRendersWithoutError [ self assert: (html includesSubstring: '', (self escapeHtml: aCodeBlock code), ''
]
+{ #category : #visiting }
+FOHTMLRenderVisitor >> visitRawHTMLBlock: aRawHTMLBlock [
+ "Deliberately not escaped and not wrapped: the author wrote HTML on purpose."
+ ^ aRawHTMLBlock html
+]
+
{ #category : #visiting }
FOHTMLRenderVisitor >> visitBlockquote: aBlockquote [
| rendered |
diff --git a/src/Foliage-Markdown/FOLineClassifier.class.st b/src/Foliage-Markdown/FOLineClassifier.class.st
index 1c669c1..c656e92 100644
--- a/src/Foliage-Markdown/FOLineClassifier.class.st
+++ b/src/Foliage-Markdown/FOLineClassifier.class.st
@@ -25,6 +25,8 @@ FOLineClassifier class >> classify: aLine [
^ self classifyImage: trimmed indent: indent ].
indent >= 4 ifTrue: [
^ FOLineClassification new kind: #indentedCode; indent: indent; yourself ].
+ (self isHTMLBlockStart: trimmed) ifTrue: [
+ ^ FOLineClassification new kind: #htmlBlock; indent: indent; yourself ].
^ FOLineClassification new kind: #paragraphText; indent: indent; yourself
]
@@ -119,6 +121,18 @@ FOLineClassifier class >> isImageLine: aLine [
^ (aLine beginsWith: '![') and: [ aLine notEmpty and: [ (aLine indexOf: $)) = aLine size ] ]
]
+{ #category : #testing }
+FOLineClassifier class >> isHTMLBlockStart: aLine [
+ "Simplified CommonMark detection: an opening tag, a closing tag or a
+ comment/doctype. A lone '<' or one followed by a space or digit
+ (as in '5 < 3') stays ordinary paragraph text."
+ | second |
+ aLine size < 2 ifTrue: [ ^ false ].
+ aLine first = $< ifFalse: [ ^ false ].
+ second := aLine at: 2.
+ ^ second isLetter or: [ second = $/ or: [ second = $! ] ]
+]
+
{ #category : #classifying }
FOLineClassifier class >> classifyImage: aLine indent: anIndent [
| closeBracket inner spaceIndex src attributes attrString |
diff --git a/src/Foliage-Markdown/FOMarkdownParser.class.st b/src/Foliage-Markdown/FOMarkdownParser.class.st
index 7c3dd0f..ee86d37 100644
--- a/src/Foliage-Markdown/FOMarkdownParser.class.st
+++ b/src/Foliage-Markdown/FOMarkdownParser.class.st
@@ -30,6 +30,7 @@ FOMarkdownParser >> consumeBlock: aReader classification: aClassification [
aClassification kind = #blockquote ifTrue: [ ^ self consumeBlockquote: aReader ].
(aClassification kind = #unorderedListItem or: [ aClassification kind = #orderedListItem ])
ifTrue: [ ^ FOListParser parseFrom: aReader ].
+ aClassification kind = #htmlBlock ifTrue: [ ^ self consumeHTMLBlock: aReader ].
^ self consumeParagraph: aReader
]
@@ -87,6 +88,17 @@ FOMarkdownParser >> consumeFencedCode: aReader classification: aClassification [
yourself
]
+{ #category : #private }
+FOMarkdownParser >> consumeHTMLBlock: aReader [
+ "Raw lines are kept verbatim (no trimming, no inline parsing) up to the next
+ blank line or the end of input. No attempt is made to balance tags."
+ | lines |
+ lines := OrderedCollection new.
+ [ aReader atEnd not and: [ (FOLineClassifier classify: aReader peek) kind ~= #blank ] ]
+ whileTrue: [ lines add: aReader next ].
+ ^ FORawHTMLBlock new html: (self joinWithNewline: lines); yourself
+]
+
{ #category : #private }
FOMarkdownParser >> consumeIndentedCode: aReader [
| lines line |
diff --git a/src/Foliage-Markdown/FOPlainTextVisitor.class.st b/src/Foliage-Markdown/FOPlainTextVisitor.class.st
index 699af0e..8df1831 100644
--- a/src/Foliage-Markdown/FOPlainTextVisitor.class.st
+++ b/src/Foliage-Markdown/FOPlainTextVisitor.class.st
@@ -32,6 +32,12 @@ FOPlainTextVisitor >> visitCodeBlock: aCodeBlock [
^ aCodeBlock code
]
+{ #category : #visiting }
+FOPlainTextVisitor >> visitRawHTMLBlock: aRawHTMLBlock [
+ "Markup has no textual content worth including in a plain-text abstract."
+ ^ ''
+]
+
{ #category : #visiting }
FOPlainTextVisitor >> visitBlockquote: aBlockquote [
^ self renderChildren: aBlockquote
diff --git a/src/Foliage-Markdown/FORawHTMLBlock.class.st b/src/Foliage-Markdown/FORawHTMLBlock.class.st
new file mode 100644
index 0000000..87a5ce1
--- /dev/null
+++ b/src/Foliage-Markdown/FORawHTMLBlock.class.st
@@ -0,0 +1,29 @@
+"
+A block of raw HTML copied verbatim from the markdown source. It starts at a line
+that looks like an HTML tag or comment and runs until the next blank line. Its
+content is neither inline-parsed nor escaped when rendered, so authors can embed
+elements the markdown syntax has no notation for (iframes, custom markup, ...).
+"
+Class {
+ #name : #FORawHTMLBlock,
+ #superclass : #FOBlockNode,
+ #instVars : [
+ 'html'
+ ],
+ #category : #'Foliage-Markdown'
+}
+
+{ #category : #accessing }
+FORawHTMLBlock >> html [
+ ^ html
+]
+
+{ #category : #accessing }
+FORawHTMLBlock >> html: aString [
+ html := aString
+]
+
+{ #category : #visiting }
+FORawHTMLBlock >> acceptVisitor: aVisitor [
+ ^ aVisitor visitRawHTMLBlock: self
+]
diff --git a/src/Foliage-Markdown/FOVisitor.class.st b/src/Foliage-Markdown/FOVisitor.class.st
index e652748..e5ee163 100644
--- a/src/Foliage-Markdown/FOVisitor.class.st
+++ b/src/Foliage-Markdown/FOVisitor.class.st
@@ -24,6 +24,11 @@ FOVisitor >> visitCodeBlock: aCodeBlock [
^ self subclassResponsibility
]
+{ #category : #visiting }
+FOVisitor >> visitRawHTMLBlock: aRawHTMLBlock [
+ ^ self subclassResponsibility
+]
+
{ #category : #visiting }
FOVisitor >> visitBlockquote: aBlockquote [
^ self subclassResponsibility