factor/basis/xml/tokenize/tokenize.factor

! Copyright (C) 2005, 2006 Daniel Ehrenberg
! See http://factorcode.org/license.txt for BSD license.
USING: accessors arrays ascii assocs combinators locals
combinators.short-circuit fry io.encodings io.encodings.iana
io.encodings.string io.encodings.utf16 io.encodings.utf8 kernel make
math math.parser namespaces sequences sets splitting xml.state-parser
strings xml.char-classes xml.data xml.entities xml.errors hashtables
circular ;
IN: xml.tokenize

! XML namespace processing: ns = namespace

! A stack of hashtables
SYMBOL: ns-stack

SYMBOL: depth

: attrs>ns ( attrs-alist -- hash )
    ! this should check to make sure URIs are valid
    [
        [
            swap dup space>> "xmlns" =
            [ main>> set ]
            [
                T{ name f "" "xmlns" f } names-match?
                [ "" set ] [ drop ] if
            ] if
        ] assoc-each
    ] { } make-assoc f like ;

: add-ns ( name -- )
    dup space>> dup ns-stack get assoc-stack
    [ nip ] [ nonexist-ns ] if* >>url drop ;

: push-ns ( hash -- )
    ns-stack get push ;

: pop-ns ( -- )
    ns-stack get pop* ;

: init-ns-stack ( -- )
    V{ H{
        { "xml" "http://www.w3.org/XML/1998/namespace" }
        { "xmlns" "http://www.w3.org/2000/xmlns" }
        { "" "" }
    } } clone
    ns-stack set ;

: tag-ns ( name attrs-alist -- name attrs )
    dup attrs>ns push-ns
    [ dup add-ns ] dip dup [ drop add-ns ] assoc-each <attrs> ;

! Parsing names

: valid-name? ( str -- ? )
    [ f ] [
        version=1.0? swap {
            [ first name-start? ]
            [ rest-slice [ name-char? ] with all? ]
        } 2&&
    ] if-empty ;

: prefixed-name ( str -- name/f )
    ":" split dup length 2 = [
        [ [ valid-name? ] all? ]
        [ first2 f <name> ] bi and
    ] [ drop f ] if ;

: interpret-name ( str -- name )
    dup prefixed-name [ ] [
        dup valid-name?
        [ <simple-name> ] [ bad-name ] if
    ] ?if ;

: take-name ( -- string )
    version=1.0? '[ _ get-char name-char? not ] take-until ;

: parse-name ( -- name )
    take-name interpret-name ;

: parse-name-starting ( string -- name )
    take-name append interpret-name ;

: parse-simple-name ( -- name )
    take-name <simple-name> ;

!   -- Parsing strings

: parse-named-entity ( string -- )
    dup entities at [ , ] [
        dup extra-entities get at
        [ % ] [ no-entity ] ?if
    ] ?if ;

: parse-entity ( -- )
    next CHAR: ; take-char next
    "#" ?head [
        "x" ?head 16 10 ? base> ,
    ] [ parse-named-entity ] if ;

:: (parse-char) ( quot: ( ch -- ? ) -- )
    get-char :> char
    {
        { [ char not ] [ ] }
        { [ char quot call ] [ next ] }
        { [ char CHAR: & = ] [ parse-entity quot (parse-char) ] }
        [ char , next quot (parse-char) ]
    } cond ; inline recursive

: parse-char ( quot: ( ch -- ? ) -- seq )
    [ (parse-char) ] "" make ; inline

: assure-no-]]> ( circular -- )
    "]]>" sequence= [ text-w/]]> ] when ;

:: parse-text ( -- string )
    3 f <array> <circular> :> circ
    depth get zero? :> no-text [| char |
        char circ push-circular
        circ assure-no-]]>
        no-text [ char blank? char CHAR: < = or [
            char 1string t pre/post-content
        ] unless ] when
        char CHAR: < =
    ] parse-char ;

! Parsing tags

: start-tag ( -- name ? )
    #! Outputs the name and whether this is a closing tag
    get-char CHAR: / = dup [ next ] when
    parse-name swap ;

: (parse-quote) ( <-disallowed? ch -- string )
    swap '[
        dup _ = [ drop t ]
        [ CHAR: < = _ and [ attr-w/< ] [ f ] if ] if
    ] parse-char get-char
    [ unclosed-quote ] unless ; inline

: parse-quote* ( <-disallowed? -- seq )
    pass-blank get-char dup "'\"" member?
    [ next (parse-quote) ] [ quoteless-attr ] if ; inline

: parse-quote ( -- seq )
   f parse-quote* ;

: normalize-quot ( str -- str )
    [ dup "\t\r\n" member? [ drop CHAR: \s ] when ] map ;

: parse-attr ( -- )
    parse-name pass-blank CHAR: = expect pass-blank
    t parse-quote* normalize-quot 2array , ;

: (middle-tag) ( -- )
    pass-blank version=1.0? get-char name-start?
    [ parse-attr (middle-tag) ] when ;

: assure-no-duplicates ( attrs-alist -- attrs-alist )
    H{ } clone 2dup '[ swap _ push-at ] assoc-each
    [ nip length 2 >= ] assoc-filter >alist
    [ first first2 duplicate-attr ] unless-empty ;

: middle-tag ( -- attrs-alist )
    ! f make will make a vector if it has any elements
    [ (middle-tag) ] f make pass-blank
    assure-no-duplicates ;

: close ( -- )
    pass-blank CHAR: > expect ;

: end-tag ( name attrs-alist -- tag )
    tag-ns pass-blank get-char CHAR: / =
    [ pop-ns <contained> next CHAR: > expect ]
    [ depth inc <opener> close ] if ;

: take-comment ( -- comment )
    "--" expect-string
    "--" take-string
    <comment>
    CHAR: > expect ;

: take-cdata ( -- string )
    depth get zero? [ bad-cdata ] when
    "[CDATA[" expect-string "]]>" take-string ;

: take-word ( -- string )
    [ get-char blank? ] take-until ;

: take-decl-contents ( -- first second )
    pass-blank take-word pass-blank ">" take-string ;

: take-element-decl ( -- element-decl )
    take-decl-contents <element-decl> ;

: take-attlist-decl ( -- doctype-decl )
    take-decl-contents <attlist-decl> ;

: take-notation-decl ( -- notation-decl )
    take-decl-contents <notation-decl> ; 

: take-until-one-of ( seps -- str sep )
    '[ get-char _ member? ] take-until get-char ;

: take-system-id ( -- system-id )
    parse-quote <system-id> close ;

: take-public-id ( -- public-id )
    parse-quote parse-quote <public-id> close ;

DEFER: direct

: (take-internal-subset) ( -- )
    pass-blank get-char {
        { CHAR: ] [ next ] }
        [ drop "<!" expect-string direct , (take-internal-subset) ]
    } case ;

: take-internal-subset ( -- seq )
    [ (take-internal-subset) ] { } make ;

: (take-external-id) ( token -- external-id )
    pass-blank {
        { "SYSTEM" [ take-system-id ] }
        { "PUBLIC" [ take-public-id ] }
        [ bad-external-id ]
    } case ;

: take-external-id ( -- external-id )
    take-word (take-external-id) ;

: only-blanks ( str -- )
    [ blank? ] all? [ bad-decl ] unless ;

: take-doctype-decl ( -- doctype-decl )
    pass-blank " >" take-until-one-of {
        { CHAR: \s [
            pass-blank get-char CHAR: [ = [
                next take-internal-subset f swap
                close
            ] [
                " >" take-until-one-of {
                    { CHAR: \s [ (take-external-id) ] }
                    { CHAR: > [ only-blanks f ] }
                } case f
            ] if
        ] }
        { CHAR: > [ f f ] }
    } case <doctype-decl> ;

: take-entity-def ( -- entity-name entity-def )
    take-word pass-blank get-char {
        { CHAR: ' [ parse-quote ] }
        { CHAR: " [ parse-quote ] }
        [ drop take-external-id ]
    } case ;

: associate-entity ( entity-name entity-def -- )
    swap extra-entities get set-at ;

: take-entity-decl ( -- entity-decl )
    pass-blank get-char {
        { CHAR: % [ next pass-blank take-entity-def ] }
        [ drop take-entity-def 2dup associate-entity ]
    } case
    close <entity-decl> ;

: take-directive ( -- directive )
    take-name {
        { "ELEMENT" [ take-element-decl ] }
        { "ATTLIST" [ take-attlist-decl ] }
        { "DOCTYPE" [ take-doctype-decl ] }
        { "ENTITY" [ take-entity-decl ] }
        { "NOTATION" [ take-notation-decl ] }
        [ bad-directive ]
    } case ;

: direct ( -- object )
    get-char {
        { CHAR: - [ take-comment ] }
        { CHAR: [ [ take-cdata ] }
        [ drop take-directive ]
    } case ;

: yes/no>bool ( string -- t/f )
    {
        { "yes" [ t ] }
        { "no" [ f ] }
        [ not-yes/no ]
    } case ;

: assure-no-extra ( seq -- )
    [ first ] map {
        T{ name f "" "version" f }
        T{ name f "" "encoding" f }
        T{ name f "" "standalone" f }
    } diff
    [ extra-attrs ] unless-empty ; 

: good-version ( version -- version )
    dup { "1.0" "1.1" } member? [ bad-version ] unless ;

: prolog-version ( alist -- version )
    T{ name f "" "version" f } swap at
    [ good-version ] [ versionless-prolog ] if* ;

: prolog-encoding ( alist -- encoding )
    T{ name f "" "encoding" f } swap at "UTF-8" or ;

: prolog-standalone ( alist -- version )
    T{ name f "" "standalone" f } swap at
    [ yes/no>bool ] [ f ] if* ;

: prolog-attrs ( alist -- prolog )
    [ prolog-version ]
    [ prolog-encoding ]
    [ prolog-standalone ]
    tri <prolog> ;

SYMBOL: string-input?
: decode-input-if ( encoding -- )
    string-input? get [ drop ] [ decode-input ] if ;

: parse-prolog ( -- prolog )
    pass-blank middle-tag "?>" expect-string
    dup assure-no-extra prolog-attrs
    dup encoding>> dup "UTF-16" =
    [ drop ] [ name>encoding [ decode-input-if ] when* ] if
    dup prolog-data set ;

: instruct ( -- instruction )
    take-name {
        { [ dup "xml" = ] [ drop parse-prolog ] }
        { [ dup >lower "xml" = ] [ capitalized-prolog ] }
        { [ dup valid-name? not ] [ bad-name ] }
        [ "?>" take-string append <instruction> ]
    } cond ;

: make-tag ( -- tag )
    {
        { [ get-char dup CHAR: ! = ] [ drop next direct ] }
        { [ CHAR: ? = ] [ next instruct ] }
        [
            start-tag [ dup add-ns pop-ns <closer> depth dec close ]
            [ middle-tag end-tag ] if
        ]
    } cond ;

! Autodetecting encodings

: continue-make-tag ( str -- tag )
    parse-name-starting middle-tag end-tag ;

: start-utf16le ( -- tag )
    utf16le decode-input-if
    CHAR: ? expect
    0 expect check instruct ;

: 10xxxxxx? ( ch -- ? )
    -6 shift 3 bitand 2 = ;
          
: start<name ( ch -- tag )
    ascii?
    [ utf8 decode-input-if next make-tag ] [
        next
        [ get-next 10xxxxxx? not ] take-until
        get-char suffix utf8 decode
        utf8 decode-input-if next
        continue-make-tag
    ] if ;
          
: start< ( -- tag )
    get-next {
        { 0 [ next next start-utf16le ] }
        { CHAR: ? [ check next next instruct ] } ! XML prolog parsing sets the encoding
        { CHAR: ! [ check utf8 decode-input next next direct ] }
        [ check start<name ]
    } case ;

: skip-utf8-bom ( -- tag )
    "\u0000bb\u0000bf" expect utf8 decode-input
    CHAR: < expect check make-tag ;

: decode-expecting ( encoding string -- tag )
    [ decode-input-if next ] [ expect-string ] bi* check make-tag ;

: start-utf16be ( -- tag )
    utf16be "<" decode-expecting ;

: skip-utf16le-bom ( -- tag )
    utf16le "\u0000fe<" decode-expecting ;

: skip-utf16be-bom ( -- tag )
    utf16be "\u0000ff<" decode-expecting ;

: start-document ( -- tag )
    get-char {
        { CHAR: < [ start< ] }
        { 0 [ start-utf16be ] }
        { HEX: EF [ skip-utf8-bom ] }
        { HEX: FF [ skip-utf16le-bom ] }
        { HEX: FE [ skip-utf16be-bom ] }
        { f [ "" ] }
        [ drop utf8 decode-input-if f ]
        ! Same problem as with <e`>, in the case of XML chunks?
    } case check ;
Initial import 2007-09-20 18:09:08 -04:00			`! Copyright (C) 2005, 2006 Daniel Ehrenberg`
			`! See http://factorcode.org/license.txt for BSD license.`
Various XML fixes, XML test suite 2009-01-19 23:25:15 -05:00			`USING: accessors arrays ascii assocs combinators locals`
Fix subtle Unicode encodings autodetection bug 2009-01-15 16:25:00 -05:00			`combinators.short-circuit fry io.encodings io.encodings.iana`
			`io.encodings.string io.encodings.utf16 io.encodings.utf8 kernel make`
Various XML fixes, XML test suite 2009-01-19 23:25:15 -05:00			`math math.parser namespaces sequences sets splitting xml.state-parser`
			`strings xml.char-classes xml.data xml.entities xml.errors hashtables`
			`circular ;`
Initial import 2007-09-20 18:09:08 -04:00			`IN: xml.tokenize`

			`! XML namespace processing: ns = namespace`

			`! A stack of hashtables`
			`SYMBOL: ns-stack`

Going further towards conformance 2009-01-20 16:37:21 -05:00			`SYMBOL: depth`

Initial import 2007-09-20 18:09:08 -04:00			`: attrs>ns ( attrs-alist -- hash )`
			`! this should check to make sure URIs are valid`
			`[`
			`[`
XML updates 2008-08-27 18:02:54 -04:00			`swap dup space>> "xmlns" =`
			`[ main>> set ]`
Initial import 2007-09-20 18:09:08 -04:00			`[`
			`T{ name f "" "xmlns" f } names-match?`
			`[ "" set ] [ drop ] if`
			`] if`
			`] assoc-each`
			`] { } make-assoc f like ;`

			`: add-ns ( name -- )`
XML updates 2008-08-27 18:02:54 -04:00			`dup space>> dup ns-stack get assoc-stack`
Update XML library to parse <! directives better; modernize the code a bit 2008-12-02 20:59:16 -05:00			`[ nip ] [ nonexist-ns ] if* >>url drop ;`
Initial import 2007-09-20 18:09:08 -04:00
			`: push-ns ( hash -- )`
			`ns-stack get push ;`

			`: pop-ns ( -- )`
			`ns-stack get pop* ;`

			`: init-ns-stack ( -- )`
			`V{ H{`
			`{ "xml" "http://www.w3.org/XML/1998/namespace" }`
			`{ "xmlns" "http://www.w3.org/2000/xmlns" }`
			`{ "" "" }`
			`} } clone`
			`ns-stack set ;`

			`: tag-ns ( name attrs-alist -- name attrs )`
			`dup attrs>ns push-ns`
Update XML library to parse <! directives better; modernize the code a bit 2008-12-02 20:59:16 -05:00			`[ dup add-ns ] dip dup [ drop add-ns ] assoc-each <attrs> ;`
Initial import 2007-09-20 18:09:08 -04:00
			`! Parsing names`

Going further towards conformance 2009-01-20 16:37:21 -05:00			`: valid-name? ( str -- ? )`
			`[ f ] [`
			`version=1.0? swap {`
			`[ first name-start? ]`
			`[ rest-slice [ name-char? ] with all? ]`
			`} 2&&`
			`] if-empty ;`

			`: prefixed-name ( str -- name/f )`
			`":" split dup length 2 = [`
			`[ [ valid-name? ] all? ]`
			`[ first2 f <name> ] bi and`
			`] [ drop f ] if ;`

			`: interpret-name ( str -- name )`
			`dup prefixed-name [ ] [`
			`dup valid-name?`
			`[ <simple-name> ] [ bad-name ] if`
			`] ?if ;`
Initial import 2007-09-20 18:09:08 -04:00
Going further towards conformance 2009-01-20 16:37:21 -05:00			`: take-name ( -- string )`
			`version=1.0? '[ _ get-char name-char? not ] take-until ;`
Initial import 2007-09-20 18:09:08 -04:00
Going further towards conformance 2009-01-20 16:37:21 -05:00			`: parse-name ( -- name )`
			`take-name interpret-name ;`
Fix subtle Unicode encodings autodetection bug 2009-01-15 16:25:00 -05:00
Going further towards conformance 2009-01-20 16:37:21 -05:00			`: parse-name-starting ( string -- name )`
			`take-name append interpret-name ;`
Fix subtle Unicode encodings autodetection bug 2009-01-15 16:25:00 -05:00
Going further towards conformance 2009-01-20 16:37:21 -05:00			`: parse-simple-name ( -- name )`
			`take-name <simple-name> ;`
Initial import 2007-09-20 18:09:08 -04:00
			`! -- Parsing strings`

XML parses entities now 2009-01-15 23:20:24 -05:00			`: parse-named-entity ( string -- )`
Various XML fixes, XML test suite 2009-01-19 23:25:15 -05:00			`dup entities at [ , ] [`
XML parses entities now 2009-01-15 23:20:24 -05:00			`dup extra-entities get at`
Various XML fixes, XML test suite 2009-01-19 23:25:15 -05:00			`[ % ] [ no-entity ] ?if`
Initial import 2007-09-20 18:09:08 -04:00			`] ?if ;`

			`: parse-entity ( -- )`
			`next CHAR: ; take-char next`
			`"#" ?head [`
			`"x" ?head 16 10 ? base> ,`
XML parses entities now 2009-01-15 23:20:24 -05:00			`] [ parse-named-entity ] if ;`
Initial import 2007-09-20 18:09:08 -04:00
Various XML fixes, XML test suite 2009-01-19 23:25:15 -05:00			`:: (parse-char) ( quot: ( ch -- ? ) -- )`
			`get-char :> char`
			`{`
			`{ [ char not ] [ ] }`
			`{ [ char quot call ] [ next ] }`
			`{ [ char CHAR: & = ] [ parse-entity quot (parse-char) ] }`
			`[ char , next quot (parse-char) ]`
			`} cond ; inline recursive`

			`: parse-char ( quot: ( ch -- ? ) -- seq )`
			`[ (parse-char) ] "" make ; inline`
Initial import 2007-09-20 18:09:08 -04:00
Various XML fixes, XML test suite 2009-01-19 23:25:15 -05:00			`: assure-no-]]> ( circular -- )`
			`"]]>" sequence= [ text-w/]]> ] when ;`
Initial import 2007-09-20 18:09:08 -04:00
Going further towards conformance 2009-01-20 16:37:21 -05:00			`:: parse-text ( -- string )`
			`3 f <array> <circular> :> circ`
			`depth get zero? :> no-text [\| char \|`
			`char circ push-circular`
			`circ assure-no-]]>`
			`no-text [ char blank? char CHAR: < = or [`
			`char 1string t pre/post-content`
			`] unless ] when`
			`char CHAR: < =`
Various XML fixes, XML test suite 2009-01-19 23:25:15 -05:00			`] parse-char ;`

Initial import 2007-09-20 18:09:08 -04:00			`! Parsing tags`

			`: start-tag ( -- name ? )`
			`#! Outputs the name and whether this is a closing tag`
			`get-char CHAR: / = dup [ next ] when`
			`parse-name swap ;`

Various XML fixes, XML test suite 2009-01-19 23:25:15 -05:00			`: (parse-quote) ( <-disallowed? ch -- string )`
			`swap '[`
			`dup _ = [ drop t ]`
			`[ CHAR: < = _ and [ attr-w/< ] [ f ] if ] if`
			`] parse-char get-char`
			`[ unclosed-quote ] unless ; inline`
XML parses entities now 2009-01-15 23:20:24 -05:00
Various XML fixes, XML test suite 2009-01-19 23:25:15 -05:00			`: parse-quote* ( <-disallowed? -- seq )`
XML parses entities now 2009-01-15 23:20:24 -05:00			`pass-blank get-char dup "'\"" member?`
Various XML fixes, XML test suite 2009-01-19 23:25:15 -05:00			`[ next (parse-quote) ] [ quoteless-attr ] if ; inline`

			`: parse-quote ( -- seq )`
			`f parse-quote* ;`

			`: normalize-quot ( str -- str )`
			`[ dup "\t\r\n" member? [ drop CHAR: \s ] when ] map ;`
Initial import 2007-09-20 18:09:08 -04:00
			`: parse-attr ( -- )`
Going further towards conformance 2009-01-20 16:37:21 -05:00			`parse-name pass-blank CHAR: = expect pass-blank`
Various XML fixes, XML test suite 2009-01-19 23:25:15 -05:00			`t parse-quote* normalize-quot 2array , ;`
Initial import 2007-09-20 18:09:08 -04:00
			`: (middle-tag) ( -- )`
			`pass-blank version=1.0? get-char name-start?`
			`[ parse-attr (middle-tag) ] when ;`

Various XML fixes, XML test suite 2009-01-19 23:25:15 -05:00			`: assure-no-duplicates ( attrs-alist -- attrs-alist )`
			`H{ } clone 2dup '[ swap _ push-at ] assoc-each`
			`[ nip length 2 >= ] assoc-filter >alist`
			`[ first first2 duplicate-attr ] unless-empty ;`

Initial import 2007-09-20 18:09:08 -04:00			`: middle-tag ( -- attrs-alist )`
XML combinator refactoring 2007-12-23 14:57:39 -05:00			`! f make will make a vector if it has any elements`
Various XML fixes, XML test suite 2009-01-19 23:25:15 -05:00			`[ (middle-tag) ] f make pass-blank`
			`assure-no-duplicates ;`
Initial import 2007-09-20 18:09:08 -04:00
Going further towards conformance 2009-01-20 16:37:21 -05:00			`: close ( -- )`
			`pass-blank CHAR: > expect ;`

Initial import 2007-09-20 18:09:08 -04:00			`: end-tag ( name attrs-alist -- tag )`
			`tag-ns pass-blank get-char CHAR: / =`
Going further towards conformance 2009-01-20 16:37:21 -05:00			`[ pop-ns <contained> next CHAR: > expect ]`
			`[ depth inc <opener> close ] if ;`
Initial import 2007-09-20 18:09:08 -04:00
			`: take-comment ( -- comment )`
			`"--" expect-string`
			`"--" take-string`
			`<comment>`
			`CHAR: > expect ;`

			`: take-cdata ( -- string )`
Going further towards conformance 2009-01-20 16:37:21 -05:00			`depth get zero? [ bad-cdata ] when`
Fixed CDATA parsing bug 2007-10-12 16:28:23 -04:00			`"[CDATA[" expect-string "]]>" take-string ;`
Initial import 2007-09-20 18:09:08 -04:00
Various XML fixes, XML test suite 2009-01-19 23:25:15 -05:00			`: take-word ( -- string )`
			`[ get-char blank? ] take-until ;`

			`: take-decl-contents ( -- first second )`
			`pass-blank take-word pass-blank ">" take-string ;`

Update XML library to parse <! directives better; modernize the code a bit 2008-12-02 20:59:16 -05:00			`: take-element-decl ( -- element-decl )`
Various XML fixes, XML test suite 2009-01-19 23:25:15 -05:00			`take-decl-contents <element-decl> ;`
Update XML library to parse <! directives better; modernize the code a bit 2008-12-02 20:59:16 -05:00
			`: take-attlist-decl ( -- doctype-decl )`
Various XML fixes, XML test suite 2009-01-19 23:25:15 -05:00			`take-decl-contents <attlist-decl> ;`
Update XML library to parse <! directives better; modernize the code a bit 2008-12-02 20:59:16 -05:00
Going further towards conformance 2009-01-20 16:37:21 -05:00			`: take-notation-decl ( -- notation-decl )`
			`take-decl-contents <notation-decl> ;`

Update XML library to parse <! directives better; modernize the code a bit 2008-12-02 20:59:16 -05:00			`: take-until-one-of ( seps -- str sep )`
			`'[ get-char _ member? ] take-until get-char ;`

			`: take-system-id ( -- system-id )`
Going further towards conformance 2009-01-20 16:37:21 -05:00			`parse-quote <system-id> close ;`
Update XML library to parse <! directives better; modernize the code a bit 2008-12-02 20:59:16 -05:00
			`: take-public-id ( -- public-id )`
Going further towards conformance 2009-01-20 16:37:21 -05:00			`parse-quote parse-quote <public-id> close ;`
Update XML library to parse <! directives better; modernize the code a bit 2008-12-02 20:59:16 -05:00
			`DEFER: direct`

			`: (take-internal-subset) ( -- )`
			`pass-blank get-char {`
			`{ CHAR: ] [ next ] }`
			`[ drop "<!" expect-string direct , (take-internal-subset) ]`
			`} case ;`

			`: take-internal-subset ( -- seq )`
			`[ (take-internal-subset) ] { } make ;`

			`: (take-external-id) ( token -- external-id )`
			`pass-blank {`
			`{ "SYSTEM" [ take-system-id ] }`
			`{ "PUBLIC" [ take-public-id ] }`
			`[ bad-external-id ]`
			`} case ;`

			`: take-external-id ( -- external-id )`
Various XML fixes, XML test suite 2009-01-19 23:25:15 -05:00			`take-word (take-external-id) ;`

			`: only-blanks ( str -- )`
			`[ blank? ] all? [ bad-decl ] unless ;`
Update XML library to parse <! directives better; modernize the code a bit 2008-12-02 20:59:16 -05:00
			`: take-doctype-decl ( -- doctype-decl )`
			`pass-blank " >" take-until-one-of {`
			`{ CHAR: \s [`
			`pass-blank get-char CHAR: [ = [`
			`next take-internal-subset f swap`
Going further towards conformance 2009-01-20 16:37:21 -05:00			`close`
Update XML library to parse <! directives better; modernize the code a bit 2008-12-02 20:59:16 -05:00			`] [`
			`" >" take-until-one-of {`
			`{ CHAR: \s [ (take-external-id) ] }`
			`{ CHAR: > [ only-blanks f ] }`
			`} case f`
			`] if`
			`] }`
			`{ CHAR: > [ f f ] }`
			`} case <doctype-decl> ;`

			`: take-entity-def ( -- entity-name entity-def )`
Various XML fixes, XML test suite 2009-01-19 23:25:15 -05:00			`take-word pass-blank get-char {`
XML parses entities now 2009-01-15 23:20:24 -05:00			`{ CHAR: ' [ parse-quote ] }`
			`{ CHAR: " [ parse-quote ] }`
Update XML library to parse <! directives better; modernize the code a bit 2008-12-02 20:59:16 -05:00			`[ drop take-external-id ]`
			`} case ;`

XML parses entities now 2009-01-15 23:20:24 -05:00			`: associate-entity ( entity-name entity-def -- )`
Going further towards conformance 2009-01-20 16:37:21 -05:00			`swap extra-entities get set-at ;`
XML parses entities now 2009-01-15 23:20:24 -05:00
Update XML library to parse <! directives better; modernize the code a bit 2008-12-02 20:59:16 -05:00			`: take-entity-decl ( -- entity-decl )`
			`pass-blank get-char {`
			`{ CHAR: % [ next pass-blank take-entity-def ] }`
XML parses entities now 2009-01-15 23:20:24 -05:00			`[ drop take-entity-def 2dup associate-entity ]`
Update XML library to parse <! directives better; modernize the code a bit 2008-12-02 20:59:16 -05:00			`} case`
Going further towards conformance 2009-01-20 16:37:21 -05:00			`close <entity-decl> ;`
Update XML library to parse <! directives better; modernize the code a bit 2008-12-02 20:59:16 -05:00
Initial import 2007-09-20 18:09:08 -04:00			`: take-directive ( -- directive )`
Going further towards conformance 2009-01-20 16:37:21 -05:00			`take-name {`
Update XML library to parse <! directives better; modernize the code a bit 2008-12-02 20:59:16 -05:00			`{ "ELEMENT" [ take-element-decl ] }`
			`{ "ATTLIST" [ take-attlist-decl ] }`
			`{ "DOCTYPE" [ take-doctype-decl ] }`
			`{ "ENTITY" [ take-entity-decl ] }`
Going further towards conformance 2009-01-20 16:37:21 -05:00			`{ "NOTATION" [ take-notation-decl ] }`
Update XML library to parse <! directives better; modernize the code a bit 2008-12-02 20:59:16 -05:00			`[ bad-directive ]`
			`} case ;`
Initial import 2007-09-20 18:09:08 -04:00
			`: direct ( -- object )`
			`get-char {`
			`{ CHAR: - [ take-comment ] }`
			`{ CHAR: [ [ take-cdata ] }`
			`[ drop take-directive ]`
			`} case ;`

			`: yes/no>bool ( string -- t/f )`
			`{`
			`{ "yes" [ t ] }`
			`{ "no" [ f ] }`
Update XML library to parse <! directives better; modernize the code a bit 2008-12-02 20:59:16 -05:00			`[ not-yes/no ]`
Initial import 2007-09-20 18:09:08 -04:00			`} case ;`

			`: assure-no-extra ( seq -- )`
			`[ first ] map {`
			`T{ name f "" "version" f }`
			`T{ name f "" "encoding" f }`
			`T{ name f "" "standalone" f }`
reverse args for assoc-diff, diff fix lots of usings fix help lint 2008-04-26 03:01:43 -04:00			`} diff`
Update XML library to parse <! directives better; modernize the code a bit 2008-12-02 20:59:16 -05:00			`[ extra-attrs ] unless-empty ;`
Initial import 2007-09-20 18:09:08 -04:00
			`: good-version ( version -- version )`
Update XML library to parse <! directives better; modernize the code a bit 2008-12-02 20:59:16 -05:00			`dup { "1.0" "1.1" } member? [ bad-version ] unless ;`
Initial import 2007-09-20 18:09:08 -04:00
XML parses entities now 2009-01-15 23:20:24 -05:00			`: prolog-version ( alist -- version )`
			`T{ name f "" "version" f } swap at`
			`[ good-version ] [ versionless-prolog ] if* ;`

			`: prolog-encoding ( alist -- encoding )`
			`T{ name f "" "encoding" f } swap at "UTF-8" or ;`

			`: prolog-standalone ( alist -- version )`
Initial import 2007-09-20 18:09:08 -04:00			`T{ name f "" "standalone" f } swap at`
XML parses entities now 2009-01-15 23:20:24 -05:00			`[ yes/no>bool ] [ f ] if* ;`

			`: prolog-attrs ( alist -- prolog )`
			`[ prolog-version ]`
			`[ prolog-encoding ]`
			`[ prolog-standalone ]`
			`tri <prolog> ;`
Initial import 2007-09-20 18:09:08 -04:00
XML encoding auto-detection 2009-01-15 01:11:23 -05:00			`SYMBOL: string-input?`
			`: decode-input-if ( encoding -- )`
			`string-input? get [ drop ] [ decode-input ] if ;`

Initial import 2007-09-20 18:09:08 -04:00			`: parse-prolog ( -- prolog )`
			`pass-blank middle-tag "?>" expect-string`
			`dup assure-no-extra prolog-attrs`
XML encoding auto-detection 2009-01-15 01:11:23 -05:00			`dup encoding>> dup "UTF-16" =`
			`[ drop ] [ name>encoding [ decode-input-if ] when* ] if`
Initial import 2007-09-20 18:09:08 -04:00			`dup prolog-data set ;`

			`: instruct ( -- instruction )`
Going further towards conformance 2009-01-20 16:37:21 -05:00			`take-name {`
			`{ [ dup "xml" = ] [ drop parse-prolog ] }`
			`{ [ dup >lower "xml" = ] [ capitalized-prolog ] }`
			`{ [ dup valid-name? not ] [ bad-name ] }`
			`[ "?>" take-string append <instruction> ]`
			`} cond ;`
Initial import 2007-09-20 18:09:08 -04:00
			`: make-tag ( -- tag )`
			`{`
			`{ [ get-char dup CHAR: ! = ] [ drop next direct ] }`
XML parses entities now 2009-01-15 23:20:24 -05:00			`{ [ CHAR: ? = ] [ next instruct ] }`
case/cond 2008-04-11 13:57:43 -04:00			`[`
Going further towards conformance 2009-01-20 16:37:21 -05:00			`start-tag [ dup add-ns pop-ns <closer> depth dec close ]`
Initial import 2007-09-20 18:09:08 -04:00			`[ middle-tag end-tag ] if`
case/cond 2008-04-11 13:57:43 -04:00			`]`
Initial import 2007-09-20 18:09:08 -04:00			`} cond ;`
XML encoding auto-detection 2009-01-15 01:11:23 -05:00
			`! Autodetecting encodings`

Fix subtle Unicode encodings autodetection bug 2009-01-15 16:25:00 -05:00			`: continue-make-tag ( str -- tag )`
Going further towards conformance 2009-01-20 16:37:21 -05:00			`parse-name-starting middle-tag end-tag ;`
Fix subtle Unicode encodings autodetection bug 2009-01-15 16:25:00 -05:00
XML encoding auto-detection 2009-01-15 01:11:23 -05:00			`: start-utf16le ( -- tag )`
			`utf16le decode-input-if`
			`CHAR: ? expect`
Various XML fixes, XML test suite 2009-01-19 23:25:15 -05:00			`0 expect check instruct ;`
XML encoding auto-detection 2009-01-15 01:11:23 -05:00
Fix subtle Unicode encodings autodetection bug 2009-01-15 16:25:00 -05:00			`: 10xxxxxx? ( ch -- ? )`
			`-6 shift 3 bitand 2 = ;`

			`: start<name ( ch -- tag )`
			`ascii?`
			`[ utf8 decode-input-if next make-tag ] [`
			`next`
			`[ get-next 10xxxxxx? not ] take-until`
			`get-char suffix utf8 decode`
			`utf8 decode-input-if next`
			`continue-make-tag`
			`] if ;`

XML encoding auto-detection 2009-01-15 01:11:23 -05:00			`: start< ( -- tag )`
			`get-next {`
			`{ 0 [ next next start-utf16le ] }`
Various XML fixes, XML test suite 2009-01-19 23:25:15 -05:00			`{ CHAR: ? [ check next next instruct ] } ! XML prolog parsing sets the encoding`
			`{ CHAR: ! [ check utf8 decode-input next next direct ] }`
			`[ check start<name ]`
XML encoding auto-detection 2009-01-15 01:11:23 -05:00			`} case ;`

			`: skip-utf8-bom ( -- tag )`
			`"\u0000bb\u0000bf" expect utf8 decode-input`
Various XML fixes, XML test suite 2009-01-19 23:25:15 -05:00			`CHAR: < expect check make-tag ;`
XML encoding auto-detection 2009-01-15 01:11:23 -05:00
XML parses entities now 2009-01-15 23:20:24 -05:00			`: decode-expecting ( encoding string -- tag )`
Various XML fixes, XML test suite 2009-01-19 23:25:15 -05:00			`[ decode-input-if next ] [ expect-string ] bi* check make-tag ;`
XML parses entities now 2009-01-15 23:20:24 -05:00
XML encoding auto-detection 2009-01-15 01:11:23 -05:00			`: start-utf16be ( -- tag )`
XML parses entities now 2009-01-15 23:20:24 -05:00			`utf16be "<" decode-expecting ;`
XML encoding auto-detection 2009-01-15 01:11:23 -05:00
			`: skip-utf16le-bom ( -- tag )`
XML parses entities now 2009-01-15 23:20:24 -05:00			`utf16le "\u0000fe<" decode-expecting ;`
XML encoding auto-detection 2009-01-15 01:11:23 -05:00
			`: skip-utf16be-bom ( -- tag )`
XML parses entities now 2009-01-15 23:20:24 -05:00			`utf16be "\u0000ff<" decode-expecting ;`
XML encoding auto-detection 2009-01-15 01:11:23 -05:00
			`: start-document ( -- tag )`
			`get-char {`
			`{ CHAR: < [ start< ] }`
			`{ 0 [ start-utf16be ] }`
			`{ HEX: EF [ skip-utf8-bom ] }`
			`{ HEX: FF [ skip-utf16le-bom ] }`
			`{ HEX: FE [ skip-utf16be-bom ] }`
			`{ f [ "" ] }`
XML parses entities now 2009-01-15 23:20:24 -05:00			`[ drop utf8 decode-input-if f ]`
			! Same problem as with <e`>, in the case of XML chunks?
Various XML fixes, XML test suite 2009-01-19 23:25:15 -05:00			`} case check ;`