diff --git a/.gitattributes b/.gitattributes index 94af433ee123..0362972a38dc 100644 --- a/.gitattributes +++ b/.gitattributes @@ -24,6 +24,13 @@ *.lockb binary diff=lockb +# XML loader fixtures in specific byte encodings (BOMs, UTF-16, ISO-8859-1); +# any newline or encoding normalization would change what they test. +test/js/bun/resolve/xml/xml-utf16le-bom.xml -text +test/js/bun/resolve/xml/xml-utf16be-bom.xml -text +test/js/bun/resolve/xml/xml-utf8-bom.xml -text +test/js/bun/resolve/xml/xml-latin1.xml -text + .vscode/launch.json linguist-generated src/api/schema.d.ts linguist-generated fixture.*.c linguist-generated diff --git a/bench/xml/bun.lock b/bench/xml/bun.lock new file mode 100644 index 000000000000..73d263753256 --- /dev/null +++ b/bench/xml/bun.lock @@ -0,0 +1,36 @@ +{ + "lockfileVersion": 2, + "configVersion": 1, + "workspaces": { + "": { + "name": "xml-benchmark", + "dependencies": { + "fast-xml-parser": "^5.2.5", + "xml2js": "^0.6.2", + }, + }, + }, + "packages": { + "@nodable/entities": ["@nodable/entities@3.0.0", "", {}, "sha512-8L9xFeTYKhm49xfIypoe2W5wV1m/3Z58kT+7kR9A8OyFxcPduI4VmxaUMQyKYrRjUoLLSXv6EKKID5Tvj9cUVw=="], + + "anynum": ["anynum@1.0.1", "", {}, "sha512-N6//FLET/tXYNM/F6ABca1oH6fWB+KlTt909Le28WMDBk8oaT4vY17DCrwg2MvmuqUKt3Ni4N5dGJ/EoBgcO6A=="], + + "fast-xml-builder": ["fast-xml-builder@1.3.0", "", { "dependencies": { "path-expression-matcher": "^1.6.2", "xml-naming": "^0.3.0" } }, "sha512-F74cZEdCvuw9P41GAC3rod4X04jjWGM1JPEv/GWSqFTWLsdyMSBMBMlm9Hk3GLBgLBbdBNY8yee0pQh2RBVESQ=="], + + "fast-xml-parser": ["fast-xml-parser@5.10.1", "", { "dependencies": { "@nodable/entities": "^3.0.0", "fast-xml-builder": "^1.2.0", "is-unsafe": "^2.0.0", "path-expression-matcher": "^1.6.2", "strnum": "^2.4.1", "xml-naming": "^0.3.0" }, "bin": { "fxparser": "src/cli/cli.js" } }, "sha512-IEMIf7298kXuZSRFoGfMYrl7is8LpavODgbNz1cwIudv7KwVFnuU+UsMporfq6PD6aXSlawZlARiA3UywCTfMw=="], + + "is-unsafe": ["is-unsafe@2.0.0", "", {}, "sha512-2LdV822R+wmI86unXA93WCFpL6g+av8ynWk0nrHyJqGop5VoocYsSLFgN8jrfalT6iGeLNM4KXuVSsULP53kEA=="], + + "path-expression-matcher": ["path-expression-matcher@1.6.2", "", {}, "sha512-enSlaiat05iasnzmgNxRj8reFdj3puY2QpNgP1aPIaVfT6nn9ICuPoFlKHk8EN22HcwewshO+mN2DGbkCEOtqQ=="], + + "sax": ["sax@1.6.1", "", {}, "sha512-42tBVwLWnaQvW5zc4HbZrTuWccECCZfBi92FDuwtqxasH+JbPB3/FOKb1m222K42R4WxuxzzMsTswfzgtSu64Q=="], + + "strnum": ["strnum@2.4.1", "", { "dependencies": { "anynum": "^1.0.1" } }, "sha512-M9eUSMT2dCB2cTNPG7UYj6KuK7RJR2SN2+yCV/fTW3xzTCS6EaGZ5pSMgDIjB7r8zSfTGk+dvvn9rTjpVS9Mwg=="], + + "xml-naming": ["xml-naming@0.3.0", "", {}, "sha512-ghig2TBE/H11aOVgmahA3MhimvkBr6JIYknH/Dhdk10nXwdbIqBJsbfMxpvFPG8bAw77gN29aQWvKpmVoPlvPQ=="], + + "xml2js": ["xml2js@0.6.2", "", { "dependencies": { "sax": ">=0.6.0", "xmlbuilder": "~11.0.0" } }, "sha512-T4rieHaC1EXcES0Kxxj4JWgaUQHDk+qwHcYOCFHfiwKz7tOVPLq7Hjq9dM1WCMhylqMEfP7hMcOIChvotiZegA=="], + + "xmlbuilder": ["xmlbuilder@11.0.1", "", {}, "sha512-fDlsI/kFEx7gLvbecc0/ohLG50fugQp8ryHzMTuW9vSa1GJ0XYWKnhsUx7oie3G98+r56aTQIUB4kht42R3JvA=="], + } +} diff --git a/bench/xml/package.json b/bench/xml/package.json new file mode 100644 index 000000000000..a7534edd3106 --- /dev/null +++ b/bench/xml/package.json @@ -0,0 +1,8 @@ +{ + "name": "xml-benchmark", + "version": "1.0.0", + "dependencies": { + "fast-xml-parser": "^5.2.5", + "xml2js": "^0.6.2" + } +} diff --git a/bench/xml/xml.mjs b/bench/xml/xml.mjs new file mode 100644 index 000000000000..ec08badd780b --- /dev/null +++ b/bench/xml/xml.mjs @@ -0,0 +1,100 @@ +import { XMLBuilder, XMLParser } from "fast-xml-parser"; +import xml2js from "xml2js"; +import { bench, group, run } from "../runner.mjs"; + +const isBun = typeof Bun !== "undefined" && Bun.XML; + +function sizeLabel(n) { + if (n >= 1024 * 1024) return `${(n / 1024 / 1024).toFixed(1)}MB`; + if (n >= 1024) return `${(n / 1024).toFixed(0)}KB`; + return `${n}B`; +} + +// -- parse inputs -- + +const small = ` + + John Doe + john@example.com + admineditor +`; + +// An S3 ListObjectsV2-style response. +function listing(count) { + const parts = [ + `\n\n bucket\n \n ${count}\n 1000\n false\n`, + ]; + for (let i = 0; i < count; i++) { + parts.push(` + photos/2024/${i.toString(16)}/image_${i}.jpg + 2024-01-${String((i % 28) + 1).padStart(2, "0")}T12:00:00.000Z + "${(i * 2654435761).toString(16)}" + ${(i * 7919) % 1000000} + STANDARD + \n`); + } + parts.push(`\n`); + return parts.join(""); +} + +// An Atom-feed-style document with mixed content and CDATA. +function feed(count) { + const parts = [ + `\n\n Example Feed\n 2024-01-13T18:30:02Z\n`, + ]; + for (let i = 0; i < count; i++) { + parts.push(` + Post number ${i} & other <things> + + urn:uuid:1225c695-cfb8-4ebb-aaaa-${String(i).padStart(12, "0")} + 2024-01-13T18:30:02Z + Some HTML in entry ${i}, kept verbatim.

]]>
+ Author ${i % 7} +
\n`); + } + parts.push(`
\n`); + return parts.join(""); +} + +const large = listing(1000); +const mixed = feed(500); + +const fxp = new XMLParser({ ignoreAttributes: false, parseTagValue: false }); +const parseXml2js = doc => { + let result; + xml2js.parseString(doc, (err, r) => { + if (err) throw err; + result = r; + }); + return result; +}; + +for (const [label, doc] of [ + ["small", small], + ["S3 listing", large], + ["Atom feed", mixed], +]) { + group(`parse ${label} (${sizeLabel(doc.length)})`, () => { + if (isBun) { + bench("Bun.XML.parse", () => Bun.XML.parse(doc)); + bench("Bun.XML.parse { compact: false }", () => Bun.XML.parse(doc, { compact: false })); + } + bench("fast-xml-parser", () => fxp.parse(doc)); + bench("xml2js", () => parseXml2js(doc)); + }); +} + +// -- stringify -- + +// Each serializer gets the object its own parser produced ("@" vs "@_" +// attribute keys), so both do the same work. +const bunObject = isBun ? Bun.XML.parse(large) : undefined; +const fxpObject = fxp.parse(large); +const builder = new XMLBuilder({ ignoreAttributes: false }); + +group(`stringify S3 listing`, () => { + if (isBun) bench("Bun.XML.stringify", () => Bun.XML.stringify(bunObject)); + bench("fast-xml-parser XMLBuilder", () => builder.build(fxpObject)); +}); + +await run(); diff --git a/docs/bundler/loaders.mdx b/docs/bundler/loaders.mdx index a341ee63b3ff..addd28b4f19a 100644 --- a/docs/bundler/loaders.mdx +++ b/docs/bundler/loaders.mdx @@ -216,6 +216,59 @@ export default { --- +### `xml` + +**XML loader.** Default for `.xml`. + +XML files can be directly imported. Bun parses them with its native XML 1.0 parser into the compact object shape of [`Bun.XML.parse`](/runtime/xml): one key for the root element, `"@name"` keys for attributes, arrays for repeated child elements, `"#text"` for text next to attributes or children, and every value a string. + +```ts +import doc from "./config.xml"; +console.log(doc.config["@version"]); + +// via import attribute: +import feed from "./export.rss" with { type: "xml" }; +``` + +During bundling, the parsed XML is inlined into the bundle as a JavaScript object. + +```ts +var doc = { + config: { + "@version": "2", + // ...other fields + }, +}; +``` + +If a `.xml` file is passed as an entrypoint, it is converted to a `.js` module that `export default`s the parsed object. + + + +```xml Input + + John Doe + johndoe@example.com + admin + editor + +``` + +```ts Output +export default { + user: { + "@id": "1", + name: "John Doe", + email: "johndoe@example.com", + role: ["admin", "editor"], + }, +}; +``` + + + +--- + ### `text` **Text loader.** Default for `.txt`. diff --git a/docs/docs.json b/docs/docs.json index 9531ea524f24..54f3545a1c1c 100644 --- a/docs/docs.json +++ b/docs/docs.json @@ -161,6 +161,7 @@ "/runtime/yaml", "/runtime/markdown", "/runtime/json5", + "/runtime/xml", "/runtime/jsonl", "/runtime/html-rewriter", "/runtime/image", @@ -513,6 +514,7 @@ "/guides/runtime/import-toml", "/guides/runtime/import-yaml", "/guides/runtime/import-json5", + "/guides/runtime/import-xml", "/guides/runtime/import-html", "/guides/util/import-meta-dir", "/guides/util/import-meta-file", diff --git a/docs/guides/runtime/import-xml.mdx b/docs/guides/runtime/import-xml.mdx new file mode 100644 index 000000000000..e7ae274c37b0 --- /dev/null +++ b/docs/guides/runtime/import-xml.mdx @@ -0,0 +1,63 @@ +--- +title: Import an XML file +sidebarTitle: Import XML +mode: center +--- + +Bun natively supports `.xml` imports. + +```xml config.xml icon="file-code" + + + + + + + +``` + +--- + +Import the file like any other source file. The module is the parsed document: one key for the root element, `"@name"` keys for attributes, arrays for repeated elements, and every value a string. + +```ts config.ts icon="/icons/typescript.svg" +import doc from "./config.xml"; + +doc.config["@env"]; // => "production" +doc.config.database["@host"]; // => "localhost" +doc.config.server["@port"]; // => "3000" +doc.config.feature.map(f => f["@name"]); // => ["auth", "rateLimit"] +``` + +--- + +The root element is also available as a named import: + +```ts config.ts icon="/icons/typescript.svg" +import { config } from "./config.xml"; + +console.log(config.database["@name"]); // => "myapp" +console.log(Number(config.server["@timeout"])); // => 30 +``` + +--- + +For parsing XML strings at runtime, use `Bun.XML.parse()`: + +```ts config.ts icon="/icons/typescript.svg" +const data = Bun.XML.parse(` + + John Doe + reading + coding + +`); + +console.log(data.user.name); // => "John Doe" +console.log(data.user.hobby); // => ["reading", "coding"] +console.log(data.user["@id"]); // => "7" +``` + +--- + +See [XML](/runtime/xml) for the rest of Bun's XML support, including the ordered `{ compact: false }` node tree and `Bun.XML.stringify()`. diff --git a/docs/runtime/bun-apis.mdx b/docs/runtime/bun-apis.mdx index 8b3b9641d9dc..a993ed36e4d0 100644 --- a/docs/runtime/bun-apis.mdx +++ b/docs/runtime/bun-apis.mdx @@ -57,5 +57,5 @@ Use the links in the table to jump to the associated documentation. | Stream Processing | [`Bun.readableStreamTo*()`](/runtime/utils#bun-readablestreamto), `Bun.readableStreamToBytes()`, `Bun.readableStreamToBlob()`, `Bun.readableStreamToFormData()`, `Bun.readableStreamToJSON()`, `Bun.readableStreamToArray()` | | Memory & Buffer Management | `Bun.ArrayBufferSink`, `Bun.allocUnsafe`, `Bun.concatArrayBuffers` | | Module Resolution | [`Bun.resolveSync()`](/runtime/utils#bun-resolvesync) | -| Parsing & Formatting | [`Bun.semver`](/runtime/semver), [`Bun.TOML.parse`](/runtime/toml), [`Bun.markdown`](/runtime/markdown), [`Bun.color`](/runtime/color), [`Bun.Image`](/runtime/image) | +| Parsing & Formatting | [`Bun.semver`](/runtime/semver), [`Bun.TOML.parse`](/runtime/toml), [`Bun.XML`](/runtime/xml), [`Bun.markdown`](/runtime/markdown), [`Bun.color`](/runtime/color), [`Bun.Image`](/runtime/image) | | Low-level / Internals | `Bun.mmap`, `Bun.gc`, `Bun.generateHeapSnapshot`, [`bun:jsc`](https://bun.com/reference/bun/jsc) | diff --git a/docs/runtime/file-types.mdx b/docs/runtime/file-types.mdx index b37f42cb4670..19e04e182760 100644 --- a/docs/runtime/file-types.mdx +++ b/docs/runtime/file-types.mdx @@ -5,7 +5,7 @@ description: "File types and loaders supported by Bun's bundler and runtime" The Bun bundler implements a set of default loaders. As a rule of thumb, the bundler and the runtime support the same set of file types. -`.js` `.cjs` `.mjs` `.mts` `.cts` `.ts` `.tsx` `.jsx` `.css` `.json` `.jsonc` `.json5` `.toml` `.yaml` `.yml` `.txt` `.wasm` `.node` `.html` `.sh` +`.js` `.cjs` `.mjs` `.mts` `.cts` `.ts` `.tsx` `.jsx` `.css` `.json` `.jsonc` `.json5` `.toml` `.yaml` `.yml` `.xml` `.txt` `.wasm` `.node` `.html` `.sh` Bun uses the file extension to pick the built-in _loader_ that parses the file. Every loader has a name, such as `js`, `tsx`, or `json`. These names are used when building [plugins](/bundler/plugins) that extend Bun with custom loaders. @@ -244,6 +244,57 @@ export default { +### `xml` + +**XML loader**. Default for `.xml`. + +XML files can be directly imported. Bun parses them with its native XML 1.0 parser into the compact object shape of [`Bun.XML.parse`](/runtime/xml): one key for the root element, `"@name"` keys for attributes, arrays for repeated child elements, `"#text"` for text next to attributes or children, and every value a string. + +```ts +import doc from "./config.xml"; +console.log(doc.config["@version"]); + +// via import attribute: +import feed from "./export.rss" with { type: "xml" }; +``` + +During bundling, the parsed XML is inlined into the bundle as a JavaScript object. + +```ts +var doc = { + config: { + "@version": "2", + // ...other fields + }, +}; +``` + +If a `.xml` file is passed as an entrypoint, it is converted to a `.js` module that `export default`s the parsed object. + + + +```xml Input + + John Doe + johndoe@example.com + admin + editor + +``` + +```ts Output +export default { + user: { + "@id": "1", + name: "John Doe", + email: "johndoe@example.com", + role: ["admin", "editor"], + }, +}; +``` + + + ### `text` **Text loader**. Default for `.txt`. diff --git a/docs/runtime/xml.mdx b/docs/runtime/xml.mdx new file mode 100644 index 000000000000..3f457582d920 --- /dev/null +++ b/docs/runtime/xml.mdx @@ -0,0 +1,247 @@ +--- +title: XML +description: Use Bun's built-in support for XML through both runtime APIs and bundler integration +--- + +In Bun, XML is a first-class citizen alongside JSON, TOML, YAML, and JSON5. You can: + +- Parse and stringify XML with `Bun.XML.parse` and `Bun.XML.stringify` +- `import` & `require` XML files as modules at runtime (including hot reloading & watch mode support) +- `import` & `require` XML files in frontend apps with Bun's bundler + +--- + +## Conformance + +Bun's XML parser is written in Rust and implements [XML 1.0 (Fifth Edition)](https://www.w3.org/TR/2008/REC-xml-20081126/) as a **non-validating processor that does not read external entities**: + +- The whole document, including the internal DTD subset, must be well-formed — anything else throws a `SyntaxError`. +- Internal entities declared in the document are expanded (with expansion limits, so "billion laughs" payloads fail instead of exhausting memory), attribute values are normalized, and attribute defaults declared in the internal subset are applied. +- External DTDs and external entities are never fetched or read, so there is no XXE surface. In a document with no DTD, a reference to an undeclared entity is an error; when the DOCTYPE points at an external subset (or uses parameter entities) that could have declared it, the reference is kept as written (` ` stays ` `), unless the document says `standalone="yes"`. +- Nothing is validated against the DTD, namespaces are not resolved (prefixed names are kept verbatim), and comments and processing instructions are skipped. + +It is run against the [W3C XML Conformance Test Suite](https://www.w3.org/XML/Test/): all 1,679 cases that have a required outcome for this class of processor pass — not-well-formed documents are rejected, well-formed ones are accepted and, where the suite gives one, their element tree matches its canonical output byte for byte. The [translated test suite](https://github.com/oven-sh/bun/blob/main/test/js/bun/xml/xml-test-suite.test.ts) lists every case, including the ones whose outcome legitimately depends on not reading external entities. + +--- + +## Runtime API + +### `Bun.XML.parse()` + +Parse an XML document into a plain JavaScript object. + +```ts +import { XML } from "bun"; + +const data = XML.parse(` + + Ada + Green tea + Mug + + +`); + +console.log(data); +// { +// order: { +// "@id": "A1", +// "@currency": "USD", +// customer: "Ada", +// item: [ +// { "@sku": "tea", "@qty": "2", "#text": "Green tea" }, +// { "@sku": "mug", "@qty": "1", "#text": "Mug" }, +// ], +// paid: "", +// }, +// } +``` + +By default the result is a **compact object** keyed by element name — the shape most XML-to-object libraries use: + +- The result has one key, the root element's name. +- An element with no attributes and no child elements becomes its text content, trimmed of surrounding whitespace (`""` when empty). +- Any other element becomes an object with a `"@name"` key per attribute, one key per distinct child element name — an **array** when that name repeats, in document order — and `"#text"` for its trimmed character data, if any. +- CDATA sections and entity references are already expanded into the text. Comments and processing instructions are dropped. +- All values are strings. Nothing is coerced to numbers, booleans, or `null`. + +The compact shape does not keep the relative order of differently named siblings or of text between child elements. When that matters — documents rather than data — pass `{ compact: false }` to get the root element as a **node tree** that keeps everything in document order: + +```ts +const p = XML.parse(`

Hello world!

`, { compact: false }); + +console.log(p); +// { +// name: "p", +// attributes: { class: "lead" }, +// children: [ +// "Hello ", +// { name: "b", attributes: {}, children: ["world"] }, +// "!", +// ], +// } +``` + +Every element is `{ name, attributes, children }`; `children` holds child elements and strings, and text is passed through exactly (including whitespace-only runs between elements). + +#### Input types and encodings + +`XML.parse` accepts a string, or bytes as a `Buffer`, `TypedArray`, `ArrayBuffer`, or `Blob`. + +A string is already-decoded text, so its `encoding` declaration is checked for syntax but otherwise ignored. Bytes are decoded per the XML rules: a byte-order mark or the `encoding` in `` selects **UTF-8** (the default), **UTF-16** (either byte order), or **ISO-8859-1**. Other encodings throw. + +```ts +XML.parse(await Bun.file("feed.xml").bytes()); +``` + +#### Error handling + +`Bun.XML.parse()` throws a `SyntaxError` when the document is not well-formed: + +```ts +try { + XML.parse(""); +} catch (error) { + console.error(error.message); // "XML Parse error: Expected closing tag but found " +} +``` + +### `Bun.XML.stringify()` + +Serialize either shape back to XML. The output has no XML declaration and is always well-formed: `&`, `<`, `>` (and, in attributes, quotes, tabs and newlines) are escaped, and element or attribute names that are not XML names throw. + +```ts +import { XML } from "bun"; + +XML.stringify({ + order: { + "@id": "A1", + customer: "Ada", + item: [{ "@sku": "tea", "#text": "Green tea" }, { "@sku": "mug" }], + paid: null, + }, +}); +// 'AdaGreen tea' + +XML.stringify({ + name: "p", + attributes: { class: "lead" }, + children: ["Hello ", { name: "b", children: ["world"] }, "!"], +}); +// '

Hello world!

' +``` + +A value with a string `name` and a `children` or `attributes` property is written as a node; anything else is a compact object and must have exactly one key naming the root element. Strings, numbers, booleans, bigints and `Date`s (as ISO strings) become text, `null` becomes an empty element, and `undefined`, functions and symbols are skipped like `JSON.stringify` skips them (unlike `JSON.stringify`, a bigint is written as its decimal digits rather than rejected). + +#### Pretty printing + +Pass a `space` argument (a number of spaces or an indent string, as with `JSON.stringify`) to indent element-only content. Elements that contain text are written inline so character data is unchanged: + +```ts +console.log(XML.stringify(data, null, 2)); +// +// Ada +// Green tea +// Mug +// +// +``` + +`XML.parse(XML.stringify(value))` gives back `value` for anything `XML.parse` produced, in either shape. + +--- + +## Module Import + +### ES Modules + +You can import XML files directly. Files are decoded like bytes passed to `XML.parse` (UTF-8, UTF-16, or ISO-8859-1 per the byte-order mark or declaration), and the module's value is the compact object described above: + +```xml config.xml + + + + + + +``` + +#### Default Import + +```ts app.ts icon="/icons/typescript.svg" +import doc from "./config.xml"; + +console.log(doc.config["@env"]); // "production" +console.log(doc.config.database["@host"]); // "localhost" +console.log(doc.config.feature.map(f => f["@name"])); // ["auth", "rateLimit"] +``` + +#### Named Import + +The root element is also available as a named import: + +```ts app.ts icon="/icons/typescript.svg" +import { config } from "./config.xml"; + +console.log(config.database["@port"]); // "5432" +``` + +### CommonJS + +```ts app.ts icon="/icons/typescript.svg" +const { config } = require("./config.xml"); +console.log(config.database["@name"]); // "myapp" +``` + +### Import Attributes + +Use `with { type: "xml" }` to parse a file with another extension as XML: + +```ts +import feed from "./export.rss" with { type: "xml" }; +``` + +--- + +## Hot Reloading with XML + +When you run your application with `bun --hot`, Bun reloads XML files when they change: + +```ts server.ts icon="/icons/typescript.svg" +import { config } from "./config.xml"; + +Bun.serve({ + port: 3000, + fetch(req) { + return new Response(`Running in ${config["@env"]} against ${config.database["@host"]}`); + }, +}); +``` + +```bash terminal icon="terminal" +bun --hot server.ts +``` + +--- + +## Bundler Integration + +When you bundle with Bun, imported XML files are parsed at build time and inlined as JavaScript objects: + +```bash terminal icon="terminal" +bun build app.ts --outdir=dist +``` + +Parsing at build time means: + +- Zero runtime XML parsing overhead in production +- Smaller bundle sizes +- Tree shaking of unused properties + +### Dynamic Imports + +XML files can be dynamically imported: + +```ts +const { default: doc } = await import("./config.xml"); +``` diff --git a/packages/bun-native-bundler-plugin-api/bundler_plugin.h b/packages/bun-native-bundler-plugin-api/bundler_plugin.h index 5578e50f105e..a4a15fa08aec 100644 --- a/packages/bun-native-bundler-plugin-api/bundler_plugin.h +++ b/packages/bun-native-bundler-plugin-api/bundler_plugin.h @@ -20,9 +20,10 @@ typedef enum { BUN_LOADER_TEXT = 12, BUN_LOADER_HTML = 17, BUN_LOADER_YAML = 18, + BUN_LOADER_XML = 21, } BunLoader; -const BunLoader BUN_LOADER_MAX = BUN_LOADER_YAML; +const BunLoader BUN_LOADER_MAX = BUN_LOADER_XML; typedef struct BunLogOptions { size_t __struct_size; diff --git a/packages/bun-types/bun.d.ts b/packages/bun-types/bun.d.ts index 98e7561841b8..a334f32d3e4f 100644 --- a/packages/bun-types/bun.d.ts +++ b/packages/bun-types/bun.d.ts @@ -821,6 +821,144 @@ declare module "bun" { export function stringify(input: unknown, replacer?: undefined | null, space?: string | number): string | undefined; } + /** + * XML related APIs + */ + namespace XML { + /** + * An element in the node tree returned by {@link parse} with `{ compact: false }` + * and accepted by {@link stringify}. + */ + interface Node { + /** The element name as written, including any namespace prefix (`"soap:Envelope"`). */ + name: string; + /** + * Attribute values by name, in document order, after attribute-value + * normalization and with defaults declared in the internal DTD subset applied. + * Namespace declarations (`xmlns`, `xmlns:*`) appear as ordinary attributes. + */ + attributes: Record; + /** + * Child elements and character data in document order. Text is passed through + * exactly (whitespace-only runs between elements included); CDATA sections, + * character references and internal entities are already expanded into the + * surrounding text, while a reference to an entity that only an (unread) external + * DTD could declare is kept as written (`"&name;"`). Comments and processing + * instructions are not represented. + */ + children: Array; + } + + interface ParseOptions { + /** + * Selects the shape of the result. + * + * - `true` (default): a compact object — `{ [rootName]: value }`, where an + * element with no attributes and no child elements becomes its text (trimmed + * of surrounding whitespace, `""` when empty), and any other element becomes + * an object with a `"@name"` key per attribute, one key per distinct child + * element name (an array when that name repeats, in document order), and + * `"#text"` for its trimmed character data if any. The relative order of + * differently named siblings and of text between them is not kept. + * - `false`: the root element as a {@link Node} tree, which keeps everything + * in document order. + * + * @default true + */ + compact?: boolean | undefined; + } + + /** + * Parse an XML 1.0 document. + * + * `Bun.XML` is a non-validating processor: the document (including its internal + * DTD subset) must be well-formed, internal entities are expanded, and attribute + * defaults declared in the internal subset are applied, but external DTDs and + * external entities are never loaded. Comments and processing instructions are + * skipped. All values are strings; nothing is coerced to numbers or booleans. + * + * A string is parsed as already-decoded text. Bytes (`Buffer`, `TypedArray`, + * `DataView`, `ArrayBuffer`, `Blob`) are decoded per the XML rules: a byte-order + * mark or the `encoding` declared in `` selects UTF-8, UTF-16, or + * ISO-8859-1. + * + * @category Utilities + * + * @param input The XML document + * @throws {SyntaxError} If the document is not well-formed (which, in a document + * without an external DTD, includes referencing an undeclared entity), uses an + * unsupported encoding, or exceeds the entity-expansion limits + * + * @example + * ```ts + * import { XML } from "bun"; + * + * XML.parse(`TeaMug`); + * // { + * // order: { + * // "@id": "A1", + * // item: [ { "@sku": "x", "#text": "Tea" }, { "@sku": "y", "#text": "Mug" } ], + * // paid: "", + * // }, + * // } + * + * XML.parse(`

Hello world!

`, { compact: false }); + * // { + * // name: "p", + * // attributes: {}, + * // children: [ "Hello ", { name: "b", attributes: {}, children: ["world"] }, "!" ], + * // } + * ``` + */ + function parse( + input: string | NodeJS.TypedArray | DataView | ArrayBufferLike | Blob, + options?: ParseOptions & { compact?: true }, + ): Record; + function parse( + input: string | NodeJS.TypedArray | DataView | ArrayBufferLike | Blob, + options: ParseOptions & { compact: false }, + ): Node; + function parse( + input: string | NodeJS.TypedArray | DataView | ArrayBufferLike | Blob, + options?: ParseOptions, + ): Record | Node; + + /** + * Serialize a value to an XML document (without an XML declaration). + * + * Accepts either shape {@link parse} produces: a {@link Node} (anything with a + * string `name` and a `children` or `attributes` property), or a compact + * object with exactly one key naming the root element, using the same `"@name"` / + * `"#text"` / array conventions. Strings, numbers, booleans, bigints and `Date`s + * (as ISO strings) become text; `null` becomes an empty element; `undefined`, + * functions and symbols are skipped. The output is always well-formed: `& < >` + * (and, in attributes, quotes and whitespace other than space) are escaped, and + * names that are not XML names, characters XML cannot contain, and circular + * structures throw. + * + * @category Utilities + * + * @param value The {@link Node} or compact object to serialize + * @param replacer Not supported; pass `undefined` or `null` + * @param space Indentation for element-only content, as in `JSON.stringify`: a + * number of spaces (at most 10) or a string (its first 10 characters). Elements + * that contain text are always written inline so character data is unchanged. + * @returns The XML, or `undefined` if `value` is `undefined`, a function, or a symbol + * + * @example + * ```ts + * import { XML } from "bun"; + * + * XML.stringify({ order: { "@id": "A1", item: ["Tea", "Mug"], paid: null } }); + * // 'TeaMug' + * + * XML.stringify({ name: "p", attributes: { class: "x" }, children: ["Hi ", { name: "b", children: ["!"] }] }, null, 2); + * // '

Hi !

' + * ``` + */ + function stringify(value: unknown, replacer?: undefined | null, space?: string | number): string | undefined; + } + /** * JSONC related APIs */ @@ -5410,6 +5548,7 @@ declare module "bun" { | "jsonc" | "toml" | "yaml" + | "xml" | "file" | "napi" | "wasm" diff --git a/packages/bun-types/extensions.d.ts b/packages/bun-types/extensions.d.ts index 07e7f3a63057..52628306b7ff 100644 --- a/packages/bun-types/extensions.d.ts +++ b/packages/bun-types/extensions.d.ts @@ -28,6 +28,11 @@ declare module "*.json5" { export = contents; } +declare module "*.xml" { + var contents: any; + export = contents; +} + declare module "*/bun.lock" { var contents: import("bun").BunLockFile; export = contents; diff --git a/scripts/build/codegen.ts b/scripts/build/codegen.ts index 221fc68cb8ff..65aa0dd6a998 100644 --- a/scripts/build/codegen.ts +++ b/scripts/build/codegen.ts @@ -737,6 +737,9 @@ function emitJsModules({ n, cfg, sources, o, dirStamp }: Ctx): void { // ($makeErrorWithCode(N, ...)); without this dep an ErrorCode.ts edit leaves // stale error numbers in the JS bundles while the C++ enum regenerates. const errorCodeInput = resolve(cfg.cwd, "src", "jsc", "bindings", "ErrorCode.ts"); + // replacements.ts also bakes the loader tables ($LoaderLabelToId / + // $LoaderIdToLabel) from src/api/schema.js into BundlerPlugin.ts. + const schemaInput = resolve(cfg.cwd, "src", "api", "schema.js"); const outputs = [ resolve(cfg.codegenDir, "WebCoreJSBuiltins.cpp"), @@ -763,7 +766,7 @@ function emitJsModules({ n, cfg, sources, o, dirStamp }: Ctx): void { n.build({ outputs, rule: "codegen", - inputs: [script, ...sources.js, ...sources.jsCodegen, extraInput, errorCodeInput], + inputs: [script, ...sources.js, ...sources.jsCodegen, extraInput, errorCodeInput, schemaInput], orderOnlyInputs: [dirStamp], vars: { cwd: cfg.cwd, diff --git a/src/analytics/lib.rs b/src/analytics/lib.rs index 109e1f86ad48..cc9ad51f3c6d 100644 --- a/src/analytics/lib.rs +++ b/src/analytics/lib.rs @@ -298,6 +298,7 @@ pub mod features { 56 => (webview_chrome, "webview_chrome"), #[unsafe(export_name = "Bun__Feature__webview_webkit")] 57 => (webview_webkit, "webview_webkit"), + 58 => (xml_parse, "xml_parse", core = XML_PARSE), } // C++ declares these as `extern "C" size_t Bun__...;` and diff --git a/src/api/schema.d.ts b/src/api/schema.d.ts index eab2dd8ebb8a..b14355f04d34 100644 --- a/src/api/schema.d.ts +++ b/src/api/schema.d.ts @@ -33,6 +33,9 @@ export const enum Loader { sqlite_embedded = 17, html = 18, yaml = 19, + json5 = 20, + md = 21, + xml = 22, } export const LoaderKeys: { 1: "jsx"; @@ -54,6 +57,9 @@ export const LoaderKeys: { 17: "sqlite_embedded"; 18: "html"; 19: "yaml"; + 20: "json5"; + 21: "md"; + 22: "xml"; jsx: 1; js: 2; ts: 3; @@ -73,6 +79,9 @@ export const LoaderKeys: { sqlite_embedded: 17; html: 18; yaml: 19; + json5: 20; + md: 21; + xml: 22; }; export const enum FrameworkEntryPointType { client = 1, diff --git a/src/api/schema.js b/src/api/schema.js index 99dc2331a9a0..5c794ff8dd4c 100644 --- a/src/api/schema.js +++ b/src/api/schema.js @@ -18,6 +18,9 @@ const Loader = { "17": "sqlite_embedded", "18": "html", "19": "yaml", + "20": "json5", + "21": "md", + "22": "xml", jsx: 1, js: 2, ts: 3, @@ -37,6 +40,9 @@ const Loader = { sqlite_embedded: 17, html: 18, yaml: 19, + json5: 20, + md: 21, + xml: 22, }; const LoaderKeys = { "1": "jsx", @@ -58,6 +64,9 @@ const LoaderKeys = { "17": "sqlite_embedded", "18": "html", "19": "yaml", + "20": "json5", + "21": "md", + "22": "xml", jsx: "jsx", js: "js", ts: "ts", @@ -77,6 +86,9 @@ const LoaderKeys = { sqlite_embedded: "sqlite_embedded", html: "html", yaml: "yaml", + json5: "json5", + md: "md", + xml: "xml", }; const FrameworkEntryPointType = { "1": 1, diff --git a/src/ast/loader.rs b/src/ast/loader.rs index 2bd07b85a1d7..fbb7dcbb46b8 100644 --- a/src/ast/loader.rs +++ b/src/ast/loader.rs @@ -50,6 +50,7 @@ pub enum Loader { Yaml = 18, Json5 = 19, Md = 20, + Xml = 21, } // Crosses FFI as `uint8_t default_loader` / `uint8_t loader` in @@ -62,6 +63,7 @@ bun_core::assert_ffi_discr!( Jsx = 0, Js = 1, Ts = 2, Tsx = 3, Css = 4, File = 5, Json = 6, Jsonc = 7, Toml = 8, Wasm = 9, Napi = 10, Base64 = 11, Dataurl = 12, Text = 13, Bunsh = 14, Sqlite = 15, SqliteEmbedded = 16, Html = 17, + Yaml = 18, Json5 = 19, Md = 20, Xml = 21, ); // E0658: inherent assoc types are nightly-only; lifted to module scope. @@ -84,6 +86,7 @@ bun_core::comptime_string_map! { b"toml" => Loader::Toml, b"yaml" => Loader::Yaml, b"json5" => Loader::Json5, + b"xml" => Loader::Xml, b"wasm" => Loader::Wasm, b"napi" => Loader::Napi, b"node" => Loader::Napi, @@ -154,6 +157,7 @@ impl Loader { Loader::Toml => "input.toml", Loader::Yaml => "input.yaml", Loader::Json5 => "input.json5", + Loader::Xml => "input.xml", Loader::Wasm => "input.wasm", Loader::Napi => "input.node", Loader::Text => "input.txt", @@ -217,10 +221,11 @@ impl Loader { | Loader::Tsx | Loader::Json | Loader::Jsonc - // toml, yaml, and json5 are included because we can serialize to the same AST as JSON + // toml, yaml, json5, and xml are included because we can serialize to the same AST as JSON | Loader::Toml | Loader::Yaml | Loader::Json5 + | Loader::Xml ) } @@ -235,6 +240,7 @@ impl Loader { | Loader::Toml | Loader::Yaml | Loader::Json5 + | Loader::Xml | Loader::File | Loader::Md => SideEffects::NoSideEffectsPureData, _ => SideEffects::HasSideEffects, diff --git a/src/bun_core/Global.rs b/src/bun_core/Global.rs index c874ea12a9c4..b9e3932a4f1e 100644 --- a/src/bun_core/Global.rs +++ b/src/bun_core/Global.rs @@ -297,7 +297,7 @@ pub mod features { SHELL, SPAWN, STANDALONE_EXECUTABLE, STANDALONE_SHELL, TODO_PANIC, TRANSPILER_CACHE, TSCONFIG, TSCONFIG_PATHS, VIRTUAL_MODULES, WORKERS_SPAWNED, WORKERS_TERMINATED, NAPI_MODULE_REGISTER, EXITED, YAML_PARSE, YARN_MIGRATION, PNPM_MIGRATION, - VALKEY, + VALKEY, XML_PARSE, } /// dotenv crate calls `bun_core::analytics::Features::dotenv_inc()`. #[inline] @@ -319,6 +319,11 @@ pub mod features { pub fn yaml_parse_inc() { YAML_PARSE.fetch_add(1, core::sync::atomic::Ordering::Relaxed); } + /// parsers crate calls `bun_core::analytics::Features::xml_parse_inc()`. + #[inline] + pub fn xml_parse_inc() { + XML_PARSE.fetch_add(1, core::sync::atomic::Ordering::Relaxed); + } /// install/yarn crate calls `bun_core::analytics::Features::yarn_migration_inc(1)`. #[inline] pub fn yarn_migration_inc(n: usize) { diff --git a/src/bundler/LinkerContext.rs b/src/bundler/LinkerContext.rs index 9c3e45bc8132..1de411454089 100644 --- a/src/bundler/LinkerContext.rs +++ b/src/bundler/LinkerContext.rs @@ -2970,6 +2970,7 @@ impl<'a> LinkerContext<'a> { | Loader::Json | Loader::Jsonc | Loader::Json5 + | Loader::Xml | Loader::Yaml | Loader::Html | Loader::SqliteEmbedded diff --git a/src/bundler/ParseTask.rs b/src/bundler/ParseTask.rs index fb71210f27b1..3033a5360809 100644 --- a/src/bundler/ParseTask.rs +++ b/src/bundler/ParseTask.rs @@ -838,6 +838,35 @@ pub mod parse_worker { let _ = temp_log.clone_to_with_recycled(log, true); return result; } + Loader::Xml => { + let _trace = perf::trace("Bundler.ParseXML"); + let mut temp_log = Log::init(); + let result = (|| -> core::result::Result, AnyError> { + let root: Expr = bun_parsers::xml::XML::parse( + source, + &mut temp_log, + bump, + bun_parsers::xml::Options { + compact: true, + encoding: bun_parsers::xml::InputEncoding::File, + }, + )?; + Ok(JSAst::init( + js_parser::new_lazy_export_ast( + bump, + &mut topts.define, + opts, + &mut temp_log, + root, + source, + b"", + )? + .unwrap(), + )) + })(); + let _ = temp_log.clone_to_with_recycled(log, true); + return result; + } Loader::Text => { let root = Expr::init( E::String { diff --git a/src/bundler/options.rs b/src/bundler/options.rs index 66155da85826..15a8283ddec5 100644 --- a/src/bundler/options.rs +++ b/src/bundler/options.rs @@ -359,9 +359,12 @@ impl LoaderExt for Loader { match self { Loader::Jsx | Loader::Js | Loader::Ts | Loader::Tsx => MimeType::JAVASCRIPT, Loader::Css => MimeType::CSS, - Loader::Toml | Loader::Yaml | Loader::Json | Loader::Jsonc | Loader::Json5 => { - MimeType::JSON - } + Loader::Toml + | Loader::Yaml + | Loader::Json + | Loader::Jsonc + | Loader::Json5 + | Loader::Xml => MimeType::JSON, Loader::Wasm => MimeType::WASM, Loader::Html | Loader::Md => MimeType::HTML, _ => { @@ -600,6 +603,7 @@ const DEFAULT_LOADERS_POSIX: &[(&[u8], Loader)] = &[ (b".html", Loader::Html), (b".jsonc", Loader::Jsonc), (b".json5", Loader::Json5), + (b".xml", Loader::Xml), (b".md", Loader::Md), (b".markdown", Loader::Md), ]; @@ -611,7 +615,7 @@ const DEFAULT_LOADERS_WIN32_EXTRA: &[(&[u8], Loader)] = &[(b".sh", Loader::Bunsh /// /// PERF: deliberately not a hashed map (the old `phf::Map` SipHash-ed the full /// key, probed a displacement table, and finished with a memcmp on every -/// lookup). With only 22 keys bucketing into 5 distinct lengths +/// lookup). With only 23 keys bucketing into 5 distinct lengths /// (3/4/5/6/9, all `.`-prefixed), a length-gated `match` is cheaper: one /// `usize` compare rejects every wrong-length probe, and within each bucket /// rustc lowers the fixed-width byte-slice arms to single u32/u64 compares (no @@ -651,6 +655,7 @@ impl DefaultLoaders { b".cts" => Some(&Loader::Ts), b".css" => Some(&Loader::Css), b".yml" => Some(&Loader::Yaml), + b".xml" => Some(&Loader::Xml), b".txt" => Some(&Loader::Text), _ => None, }, diff --git a/src/bundler/transpiler.rs b/src/bundler/transpiler.rs index d80c75a73613..0036a9396535 100644 --- a/src/bundler/transpiler.rs +++ b/src/bundler/transpiler.rs @@ -1776,7 +1776,8 @@ impl<'a> Transpiler<'a> { | options::Loader::Yaml | options::Loader::Json | options::Loader::Jsonc - | options::Loader::Json5 => { + | options::Loader::Json5 + | options::Loader::Xml => { return parse_data_loader( source, loader, @@ -1871,7 +1872,17 @@ fn parse_data_loader<'a>( Err(_) => return None, } } - // SAFETY: outer match arm guarantees one of the five. + options::Loader::Xml => { + let options = bun_parsers::xml::Options { + compact: true, + encoding: bun_parsers::xml::InputEncoding::File, + }; + match bun_parsers::xml::XML::parse(source, log, arena, options) { + Ok(e) => e, + Err(_) => return None, + } + } + // SAFETY: outer match arm guarantees one of the six. _ => unsafe { core::hint::unreachable_unchecked() }, }; let mut expr = value_expr; @@ -2869,6 +2880,7 @@ impl<'a> Transpiler<'a> { | options::Loader::Toml | options::Loader::Yaml | options::Loader::Json5 + | options::Loader::Xml | options::Loader::Text | options::Loader::Md => { // borrowck — `parse` consumes `&mut self`, so capture diff --git a/src/bundler_jsc/options_jsc.rs b/src/bundler_jsc/options_jsc.rs index 7691afb36eea..ea381ce971b6 100644 --- a/src/bundler_jsc/options_jsc.rs +++ b/src/bundler_jsc/options_jsc.rs @@ -41,7 +41,7 @@ pub fn loader_from_js( let Some(v) = bun_ast::Loader::from_string(slice.slice()) else { return Err(global.throw_invalid_arguments(format_args!( - "invalid loader - must be js, jsx, tsx, ts, css, json, jsonc, json5, toml, yaml, text, wasm, or md" + "invalid loader - must be js, jsx, tsx, ts, css, json, jsonc, json5, toml, yaml, xml, text, wasm, or md" ))); }; // These are valid `Loader` variants for the bundler but have no source-text diff --git a/src/js_printer/lib.rs b/src/js_printer/lib.rs index 86627cc0dd8a..0b95258cc04f 100644 --- a/src/js_printer/lib.rs +++ b/src/js_printer/lib.rs @@ -5845,6 +5845,9 @@ pub(crate) mod __gated_printer { Loader::Json5 => { self.print_whitespacer(ws!(b" with { type: \"json5\" }")) } + Loader::Xml => { + self.print_whitespacer(ws!(b" with { type: \"xml\" }")) + } Loader::Wasm => { self.print_whitespacer(ws!(b" with { type: \"wasm\" }")) } @@ -5911,6 +5914,7 @@ pub(crate) mod __gated_printer { } Loader::Html => FP::host_defined(mi.str(b"html")), Loader::Json5 => FP::host_defined(mi.str(b"json5")), + Loader::Xml => FP::host_defined(mi.str(b"xml")), Loader::Md => FP::host_defined(mi.str(b"md")), } } else { diff --git a/src/jsc/bindings/BunObject+exports.h b/src/jsc/bindings/BunObject+exports.h index 47a1f8d280db..31319ae85641 100644 --- a/src/jsc/bindings/BunObject+exports.h +++ b/src/jsc/bindings/BunObject+exports.h @@ -23,6 +23,7 @@ macro(SHA512) \ macro(SHA512_256) \ macro(TOML) \ + macro(XML) \ macro(YAML) \ macro(Terminal) \ macro(Transpiler) \ diff --git a/src/jsc/bindings/BunObject.cpp b/src/jsc/bindings/BunObject.cpp index 696460b194a6..2067f2f9e461 100644 --- a/src/jsc/bindings/BunObject.cpp +++ b/src/jsc/bindings/BunObject.cpp @@ -944,6 +944,7 @@ JSC_DEFINE_HOST_FUNCTION(functionFileURLToPath, (JSC::JSGlobalObject * globalObj JSONL constructJSONLObject ReadOnly|DontDelete|PropertyCallback markdown BunObject_lazyPropCb_wrap_markdown DontDelete|PropertyCallback TOML BunObject_lazyPropCb_wrap_TOML DontDelete|PropertyCallback + XML BunObject_lazyPropCb_wrap_XML DontDelete|PropertyCallback YAML BunObject_lazyPropCb_wrap_YAML DontDelete|PropertyCallback Transpiler BunObject_lazyPropCb_wrap_Transpiler DontDelete|PropertyCallback embeddedFiles BunObject_lazyPropCb_wrap_embeddedFiles DontDelete|PropertyCallback diff --git a/src/jsc/bindings/ModuleLoader.cpp b/src/jsc/bindings/ModuleLoader.cpp index 7c5ba80bb6a3..21c75a0f3483 100644 --- a/src/jsc/bindings/ModuleLoader.cpp +++ b/src/jsc/bindings/ModuleLoader.cpp @@ -272,13 +272,15 @@ OnLoadResult handleOnLoadResultNotPromise(Zig::GlobalObject* globalObject, JSC:: loader = BunLoaderTypeYAML; } else if (loaderString == "md"_s) { loader = BunLoaderTypeMD; + } else if (loaderString == "xml"_s) { + loader = BunLoaderTypeXML; } } } } if (loader == BunLoaderTypeNone) [[unlikely]] { - throwException(globalObject, scope, createError(globalObject, "Expected loader to be one of \"js\", \"jsx\", \"object\", \"ts\", \"tsx\", \"toml\", \"yaml\", \"json\", or \"md\""_s)); + throwException(globalObject, scope, createError(globalObject, "Expected loader to be one of \"js\", \"jsx\", \"object\", \"ts\", \"tsx\", \"toml\", \"yaml\", \"json\", \"xml\", or \"md\""_s)); result.value.error = scope.exception(); (void)scope.tryClearException(); return result; diff --git a/src/jsc/bindings/headers-handwritten.h b/src/jsc/bindings/headers-handwritten.h index c07b613f42b4..6cb74be5243e 100644 --- a/src/jsc/bindings/headers-handwritten.h +++ b/src/jsc/bindings/headers-handwritten.h @@ -267,7 +267,8 @@ inline constexpr BunLoaderType BunLoaderTypeTOML = 9; inline constexpr BunLoaderType BunLoaderTypeWASM = 10; inline constexpr BunLoaderType BunLoaderTypeNAPI = 11; inline constexpr BunLoaderType BunLoaderTypeYAML = 19; -inline constexpr BunLoaderType BunLoaderTypeMD = 20; +inline constexpr BunLoaderType BunLoaderTypeMD = 21; +inline constexpr BunLoaderType BunLoaderTypeXML = 22; #pragma mark - Stream diff --git a/src/options_types/bundle_enums.rs b/src/options_types/bundle_enums.rs index bc006a092c8d..ba7eb1429491 100644 --- a/src/options_types/bundle_enums.rs +++ b/src/options_types/bundle_enums.rs @@ -177,6 +177,7 @@ pub static LOADER_API_NAMES: api::Loader = { b"toml" => api::Loader::toml, b"yaml" => api::Loader::yaml, b"json5" => api::Loader::json5, + b"xml" => api::Loader::xml, b"wasm" => api::Loader::wasm, b"node" => api::Loader::napi, b"dataurl" => api::Loader::dataurl, @@ -212,6 +213,7 @@ impl LoaderExt for Loader { Loader::Toml => api::Loader::toml, Loader::Yaml => api::Loader::yaml, Loader::Json5 => api::Loader::json5, + Loader::Xml => api::Loader::xml, Loader::Wasm => api::Loader::wasm, Loader::Napi => api::Loader::napi, Loader::Base64 => api::Loader::base64, @@ -236,6 +238,7 @@ impl LoaderExt for Loader { api::Loader::toml => Loader::Toml, api::Loader::yaml => Loader::Yaml, api::Loader::json5 => Loader::Json5, + api::Loader::xml => Loader::Xml, api::Loader::wasm => Loader::Wasm, api::Loader::napi => Loader::Napi, api::Loader::base64 => Loader::Base64, diff --git a/src/options_types/schema.rs b/src/options_types/schema.rs index 9d1dad64afa4..046f30bc7faf 100644 --- a/src/options_types/schema.rs +++ b/src/options_types/schema.rs @@ -393,6 +393,7 @@ pub mod api { yaml = 19, json5 = 20, md = 21, + xml = 22, } impl Loader { @@ -423,6 +424,7 @@ pub mod api { 19 => Loader::yaml, 20 => Loader::json5, 21 => Loader::md, + 22 => Loader::xml, _ => Loader::_none, } } diff --git a/src/parsers/lib.rs b/src/parsers/lib.rs index 73890a72e2f9..83bed221c01f 100644 --- a/src/parsers/lib.rs +++ b/src/parsers/lib.rs @@ -24,3 +24,6 @@ pub mod toml; #[path = "yaml.rs"] pub mod yaml; + +#[path = "xml.rs"] +pub mod xml; diff --git a/src/parsers/xml.rs b/src/parsers/xml.rs new file mode 100644 index 000000000000..cee6b1fcfa8e --- /dev/null +++ b/src/parsers/xml.rs @@ -0,0 +1,3096 @@ +//! XML 1.0 (Fifth Edition) scanner/parser — a non-validating processor that +//! does not read external entities (§5.1). +//! +//! Architecture (mirrors `yaml.rs`): the scanner turns bytes into tokens and +//! the parser is recursive descent over tokens, never touching source bytes. +//! Outside element content XML's lexical grammar is uniform — names, quoted +//! literals, a handful of punctuation marks and the ``), the parser +//! checks the token's `spaced` flag. The one context-sensitive lexeme is the +//! quoted literal (`AttValue`, `EntityValue`, `SystemLiteral` and +//! `PubidLiteral` decode differently), so `next` takes the `Literal` kind the +//! parser's grammar position calls for. Element content, where whitespace is +//! character data, has its own loop (`Scanner::next_content`). +//! +//! Entity replacement (§4.4) is character-level substitution, so it lives in +//! the scanner: an entity reference in a context where the spec says +//! "included" pushes the replacement text as a new input frame and scanning +//! continues there. Tokens carry the id of the frame they came from so the +//! parser can enforce the structural rules (an element or markup declaration +//! must start and end in the same entity). The parser feeds declarations from +//! the internal DTD subset back to the scanner's entity tables; per §5.1 those +//! declarations are used to expand internal entities, supply attribute +//! defaults, and normalize attribute values, and declarations after a +//! reference to a parameter entity that is not read are ignored (unless +//! `standalone="yes"`). +//! +//! Two JS value shapes are built from the same token stream (see `Sink`): the +//! compact object (`{"@attr": .., child: .., "#text": ..}`) used by +//! `Bun.XML.parse` by default and by the module loader, and the ordered node +//! tree (`{name, attributes, children}`) for `{ compact: false }`. + +use bun_alloc::Arena as Bump; +use bun_alloc::ArenaVec; +use bun_alloc::ArenaVecExt as _; +use bun_ast::{self as ast, E, Expr, G, Loc, Log, Source}; +use bun_collections::{HashMap, VecExt}; +use bun_core::{StackCheck, strings}; +use bun_simdutf_sys::simdutf; + +// ── public entry point ────────────────────────────────────────────────────── + +pub struct XML; + +#[derive(Copy, Clone)] +pub struct Options { + /// Build the compact object shape (`true`) or the node tree (`false`). + pub compact: bool, + pub encoding: InputEncoding, +} + +/// What the bytes handed to the parser are, which decides how much of +/// §4.3.3 (BOM / UTF-16 detection, the `encoding` declaration) applies. +#[derive(Copy, Clone, PartialEq, Eq)] +pub enum InputEncoding { + /// Raw bytes (`XML.parse` of a Buffer/Blob): full detection, and a + /// declaration that contradicts the bytes is an error. + Bytes, + /// A file read by the module loader: detection applies, but the file + /// reader has already converted UTF-16LE-with-BOM to UTF-8 and stripped + /// UTF-8 byte-order marks, so a UTF-16 declaration on UTF-8 input is + /// taken as already satisfied rather than as a contradiction. + File, + /// Already-decoded text (a JS string, re-encoded as UTF-8): nothing to + /// detect; the declaration is checked for syntax but not acted upon. + Text, +} + +impl XML { + pub fn parse<'a>( + source: &'a Source, + log: &mut Log, + bump: &'a Bump, + options: Options, + ) -> crate::Result { + bun_core::analytics::Features::xml_parse_inc(); + let result = if options.compact { + Parser::new(source, log, bump, options, CompactSink::new(bump)).parse_document() + } else { + Parser::new(source, log, bump, options, NodeSink::new(bump)).parse_document() + }; + match result { + Ok(root) => Ok(root), + Err(PErr::Syntax) => Err(crate::Error::SyntaxError), + Err(PErr::Oom) => Err(crate::Error::Alloc(bun_alloc::AllocError)), + Err(PErr::StackOverflow) => Err(crate::Error::StackOverflow), + } + } +} + +#[derive(Copy, Clone, PartialEq, Eq, Debug)] +enum PErr { + /// Already logged. + Syntax, + Oom, + StackOverflow, +} + +impl From for PErr { + fn from(_: bun_alloc::AllocError) -> Self { + PErr::Oom + } +} + +type PResult = Result; + +/// Entity expansion is bounded the way expat bounds it (billion laughs): once +/// the text produced by parsing passes the threshold, it may not exceed +/// `MAX_AMPLIFICATION` times the size of the document. (expat activates at +/// 8 MiB; internal entities are the only kind expanded here, and a document +/// that small legitimately producing more than 1 MiB from them is unheard of, +/// so the work an attacker can force is capped lower.) +const AMPLIFICATION_THRESHOLD: u64 = 1024 * 1024; +const MAX_AMPLIFICATION: u64 = 100; +/// Entity references open at any one time (the depth of the reference chain). +const MAX_ENTITY_DEPTH: usize = 256; + +// ── character classes ─────────────────────────────────────────────────────── + +/// `S` (§2.3 [3]). +#[inline] +fn is_ws(c: u8) -> bool { + matches!(c, b' ' | b'\t' | b'\n' | b'\r') +} + +#[inline] +fn is_name_start_ascii(c: u8) -> bool { + c.is_ascii_alphabetic() || c == b'_' || c == b':' +} + +#[inline] +fn is_name_char_ascii(c: u8) -> bool { + is_name_start_ascii(c) || c.is_ascii_digit() || c == b'-' || c == b'.' +} + +/// `NameStartChar` (§2.3 [4]) above ASCII. +fn is_name_start_code_point(cp: u32) -> bool { + matches!( + cp, + 0xC0..=0xD6 + | 0xD8..=0xF6 + | 0xF8..=0x2FF + | 0x370..=0x37D + | 0x37F..=0x1FFF + | 0x200C..=0x200D + | 0x2070..=0x218F + | 0x2C00..=0x2FEF + | 0x3001..=0xD7FF + | 0xF900..=0xFDCF + | 0xFDF0..=0xFFFD + | 0x10000..=0xEFFFF + ) +} + +/// `NameChar` (§2.3 [4a]) above ASCII. +fn is_name_code_point(cp: u32) -> bool { + is_name_start_code_point(cp) || cp == 0xB7 || matches!(cp, 0x300..=0x36F | 0x203F..=0x2040) +} + +/// `Char` (§2.2 [2]). +pub fn is_xml_char(cp: u32) -> bool { + matches!(cp, 0x9 | 0xA | 0xD | 0x20..=0xD7FF | 0xE000..=0xFFFD | 0x10000..=0x10FFFF) +} + +/// `NameStartChar` (§2.3 [4]) for any code point; used by `XML.stringify` +/// to refuse names the parser would reject. +pub fn is_name_start_char(cp: u32) -> bool { + if cp < 0x80 { + is_name_start_ascii(cp as u8) + } else { + is_name_start_code_point(cp) + } +} + +/// `NameChar` (§2.3 [4a]) for any code point. +pub fn is_name_char(cp: u32) -> bool { + if cp < 0x80 { + is_name_char_ascii(cp as u8) + } else { + is_name_code_point(cp) + } +} + +/// `PubidChar` (§2.3 [13]). +fn is_pubid_char(c: u8) -> bool { + c.is_ascii_alphanumeric() + || matches!( + c, + b' ' | b'\r' + | b'\n' + | b'-' + | b'\'' + | b'(' + | b')' + | b'+' + | b',' + | b'.' + | b'/' + | b':' + | b'=' + | b'?' + | b';' + | b'!' + | b'*' + | b'#' + | b'@' + | b'$' + | b'_' + | b'%' + ) +} + +/// The `S` characters, for `strings::trim`. +const XML_WS: &[u8] = b" \t\n\r"; + +// ── tokens ────────────────────────────────────────────────────────────────── + +/// What the scanner hands the parser: `Scanner::next` produces everything +/// but `Text`; `Scanner::next_content` produces `Text`, the tags and `Eof`. +#[derive(Clone, Copy)] +struct Token<'a> { + kind: Kind<'a>, + /// Byte offset in the document, for diagnostics. + pos: usize, + /// The input frame (document or entity replacement text) the token was + /// read from; elements and declarations must begin and end in the same + /// one. + frame: u32, + /// Whether whitespace came directly before the token. `S` is required, + /// optional or forbidden depending on the position, and the parser + /// checks that with this flag. + spaced: bool, +} + +#[derive(Clone, Copy, PartialEq, Eq)] +enum Kind<'a> { + /// The end of the document or, with its name, of an entity's replacement + /// text where that is not simply the end of an inclusion. + Eof(Option<&'a [u8]>), + /// `Name` (§2.3 [5]); keywords such as `SYSTEM` or `CDATA` are names too. + Name(&'a [u8]), + /// A run of name characters that does not start with a `NameStartChar`, + /// so only an `Nmtoken` (§2.3 [7]). + Nmtoken(&'a [u8]), + /// `#` and a name: `#PCDATA`, `#REQUIRED`, `#IMPLIED`, `#FIXED`. + Hash(&'a [u8]), + /// `%Name;` outside parameter-entity replacement text (inside it, a + /// reference is included in place, §4.4.8, and never surfaces). + PeReference(&'a [u8]), + /// `%` not followed by a name: the parameter-entity declaration marker. + Percent, + /// `%Name` with no `;`: a malformed reference — or, right after + /// ` { + /// How a token is named in "but found …" diagnostics. + fn fmt(&self, f: &mut core::fmt::Formatter<'_>) -> core::fmt::Result { + let name = |f: &mut core::fmt::Formatter<'_>, prefix: &str, name: &[u8], suffix: &str| { + write!(f, "'{}{}{}'", prefix, bstr::BStr::new(name), suffix) + }; + match *self { + Kind::Eof(Some(entity)) => write!(f, "the end of entity '{}'", bstr::BStr::new(entity)), + Kind::Eof(None) => f.write_str("end of input"), + Kind::Name(n) | Kind::Nmtoken(n) => name(f, "", n, ""), + Kind::Hash(n) => name(f, "#", n, ""), + Kind::PeReference(n) => name(f, "%", n, ";"), + Kind::Percent => f.write_str("'%'"), + Kind::PercentName(n) => name(f, "%", n, ""), + Kind::Literal { .. } => f.write_str("a quoted string"), + Kind::Eq => f.write_str("'='"), + Kind::Gt => f.write_str("'>'"), + Kind::SlashGt => f.write_str("'/>'"), + Kind::ParenOpen => f.write_str("'('"), + Kind::ParenClose => f.write_str("')'"), + Kind::Bar => f.write_str("'|'"), + Kind::Comma => f.write_str("','"), + Kind::Question => f.write_str("'?'"), + Kind::Star => f.write_str("'*'"), + Kind::Plus => f.write_str("'+'"), + Kind::BracketOpen => f.write_str("'['"), + Kind::BracketClose => f.write_str("']'"), + Kind::Decl(kind) => f.write_str(kind.opener()), + Kind::XmlDecl => f.write_str("' f.write_str("a comment"), + Kind::Pi => f.write_str("a processing instruction"), + Kind::StartTag(n) => name(f, "<", n, ""), + Kind::EndTag(n) => name(f, " f.write_str("text"), + Kind::Unexpected(cp) => match char::from_u32(cp) { + Some(c) if c.is_ascii_graphic() => write!(f, "'{}'", c), + Some(c) if !c.is_control() => write!(f, "'{}' (U+{:04X})", c, cp), + _ => write!(f, "U+{:04X}", cp), + }, + } + } +} + +#[derive(Copy, Clone, PartialEq, Eq)] +enum DeclKind { + Doctype, + Element, + Attlist, + Entity, + Notation, +} + +impl DeclKind { + fn opener(self) -> &'static str { + match self { + DeclKind::Doctype => "' "' "' "' "'` are rejected + /// so a missing quote cannot swallow markup. + Plain, + /// `AttValue`: references included and whitespace normalized (§4.4.5, + /// §3.3.3); `collapse` when the attribute is declared with a tokenized + /// or enumerated type, whose values additionally have leading and + /// trailing spaces dropped and inner runs collapsed. + AttValue { collapse: bool }, + /// `EntityValue`: character and parameter-entity references included, + /// general entity references bypassed (§4.4.5, §4.4.7). + EntityValue, + /// `SystemLiteral`: anything up to the closing quote. + System, + /// `PubidLiteral`: `PubidChar`s only. + Pubid, +} + +// ── entities and input frames ─────────────────────────────────────────────── + +#[derive(Copy, Clone)] +enum EntityValue<'a> { + /// Replacement text: character references (and, in text that came from + /// a parameter entity, parameter-entity references) already resolved, + /// general entity references bypassed (§4.4.7), line ends normalized. + Internal(&'a [u8]), + /// Declared SYSTEM/PUBLIC; never read. + External, + /// External with NDATA — not a parsed entity at all. + Unparsed, +} + +struct Entities<'a> { + general: HashMap<&'a [u8], EntityValue<'a>>, + parameter: HashMap<&'a [u8], EntityValue<'a>>, +} + +fn predefined_entity(name: &[u8]) -> Option { + match name { + b"lt" => Some(b'<'), + b"gt" => Some(b'>'), + b"amp" => Some(b'&'), + b"apos" => Some(b'\''), + b"quot" => Some(b'"'), + _ => None, + } +} + +/// What a general entity reference contributes where it is included. +enum Resolved<'a> { + /// A predefined entity's character. + Byte(u8), + /// Replacement text to scan in a new input frame. + Text(&'a [u8]), + /// Nothing known: the reference itself is kept as character data. + Unexpanded, +} + +#[derive(Copy, Clone, PartialEq, Eq)] +enum FrameKind { + /// The document entity. Line ends are normalized while scanning it. + Document, + /// A general entity included in content (§4.4.2). + Content, + /// An entity included in a literal (§4.4.5): quotes inside it are data. + Literal, + /// A parameter entity included in the DTD (§4.4.8). + Declarations, +} + +struct Frame<'a> { + src: &'a [u8], + pos: usize, + id: u32, + kind: FrameKind, + /// The entity this frame is the replacement text of: (name, is-parameter). + entity: Option<(&'a [u8], bool)>, + /// Where diagnostics for tokens read from this frame point: the position + /// of the outermost reference in the document. + report_pos: usize, +} + +// ── scanner ───────────────────────────────────────────────────────────────── + +/// Owns the byte cursor, the input-frame stack and the entity tables; the +/// only component that reads bytes. +struct Scanner<'a, 'log> { + /// The frame being read (the fields of `Frame`, unpacked for the hot + /// path); enclosing frames wait in `suspended`. + src: &'a [u8], + pos: usize, + frame_id: u32, + frame_kind: FrameKind, + frame_entity: Option<(&'a [u8], bool)>, + frame_report_pos: usize, + suspended: Vec>, + next_frame_id: u32, + + entities: Entities<'a>, + /// Bytes of replacement text pushed so far, for the amplification limit. + expanded_bytes: u64, + document_len: u64, + + /// Facts about the document that well-formedness rules depend on. + standalone: bool, + has_external_subset: bool, + /// A parameter-entity reference was seen in the DTD. + saw_pe_reference: bool, + /// A reference to a parameter entity that was not read (external or + /// undeclared) was seen: later declarations may depend on it. + saw_unread_pe: bool, + + /// Positions cannot be mapped back to `source` once the input has been + /// transcoded, so diagnostics carry no location in that case. + transcoded: bool, + saw_utf8_bom: bool, + needs_utf16_declaration: bool, + /// Where the document proper starts (after a byte-order mark): the only + /// place an XML declaration may stand. + content_start: usize, + encoding: InputEncoding, + bump: &'a Bump, + source: &'a Source, + log: &'log mut Log, +} + +impl<'a, 'log> Scanner<'a, 'log> { + // ── error helpers ────────────────────────────────────────────────────── + + fn loc(&self, pos: usize) -> Loc { + if self.transcoded { + Loc::EMPTY + } else { + Loc { + start: i32::try_from(pos).unwrap_or(i32::MAX), + } + } + } + + fn err(&mut self, pos: usize, msg: &'static str) -> PErr { + self.err_fmt(pos, format_args!("{}", msg)) + } + + fn err_fmt(&mut self, pos: usize, args: core::fmt::Arguments<'_>) -> PErr { + let loc = self.loc(pos); + self.log.add_error_fmt_opts( + args, + ast::AddErrorOptions { + source: Some(self.source), + loc, + len: 0, + redact_sensitive_information: false, + }, + ); + PErr::Syntax + } + + /// `{before} '{name}'{after}`. + fn err_named( + &mut self, + pos: usize, + before: &'static str, + name: &[u8], + after: &'static str, + ) -> PErr { + self.err_fmt( + pos, + format_args!("{} '{}'{}", before, bstr::BStr::new(name), after), + ) + } + + /// `{what} {the character at the cursor}` — for "expected X but found" + /// diagnostics. + fn err_here(&mut self, what: &'static str) -> PErr { + let pos = self.here(); + if self.at_end() { + return match self.frame_entity { + Some((name, _)) => self.err_fmt( + pos, + format_args!("{} the end of entity '{}'", what, bstr::BStr::new(name)), + ), + None => self.err_fmt(pos, format_args!("{} end of input", what)), + }; + } + let c = self.peek(); + if c.is_ascii_graphic() { + self.err_fmt(pos, format_args!("{} '{}'", what, c as char)) + } else if c >= 0x80 { + let (cp, _) = self.decode_utf8(); + match char::from_u32(cp) { + Some(ch) if !ch.is_control() => { + self.err_fmt(pos, format_args!("{} '{}' (U+{:04X})", what, ch, cp)) + } + _ => self.err_fmt(pos, format_args!("{} U+{:04X}", what, cp)), + } + } else { + let name = match c { + b' ' => "space", + b'\t' => "tab", + b'\n' | b'\r' => "newline", + _ => "", + }; + if name.is_empty() { + self.err_fmt(pos, format_args!("{} control character 0x{:02X}", what, c)) + } else { + self.err_fmt(pos, format_args!("{} {}", what, name)) + } + } + } + + fn err_invalid_char(&mut self) -> PErr { + self.err_here("Invalid character in XML:") + } + + // ── byte cursor ──────────────────────────────────────────────────────── + + /// Document position for diagnostics about the byte at the cursor. + #[inline] + fn here(&self) -> usize { + if self.in_document() { + self.pos + } else { + self.frame_report_pos + } + } + + #[inline] + fn peek(&self) -> u8 { + self.peek_at(self.pos) + } + + #[inline] + fn peek_at(&self, pos: usize) -> u8 { + if pos < self.src.len() { + self.src[pos] + } else { + 0 + } + } + + /// End of the current frame (not necessarily of the document). + #[inline] + fn at_end(&self) -> bool { + self.pos >= self.src.len() + } + + #[inline] + fn starts_with(&self, s: &[u8]) -> bool { + self.src[self.pos.min(self.src.len())..].starts_with(s) + } + + #[inline] + fn in_document(&self) -> bool { + self.frame_kind == FrameKind::Document + } + + /// Decodes the UTF-8 sequence at the cursor: (code point, byte length). + /// Only the XML declaration is tokenized before the input is validated; + /// there a malformed sequence decodes as (0, len) or, when cut off by + /// the end of the frame, as (lead byte, 1) — never past the end, and + /// never as a character a name or the parser accepts. + fn decode_utf8(&self) -> (u32, usize) { + let first = self.peek(); + let len = strings::wtf8_byte_sequence_length(first); + if len == 1 || self.pos + usize::from(len) > self.src.len() { + return (u32::from(first), 1); + } + let mut bytes = [0u8; 4]; + bytes[..usize::from(len)].copy_from_slice(&self.src[self.pos..self.pos + usize::from(len)]); + ( + strings::decode_wtf8_rune_t(bytes, len, 0u32), + usize::from(len), + ) + } + + /// Validates the non-ASCII sequence at the cursor as a `Char` (in valid + /// UTF-8 only U+FFFE and U+FFFF are excluded) and returns its byte + /// length. A malformed sequence can only be met inside the XML + /// declaration, before the input has been validated. + fn check_non_ascii_char(&mut self) -> PResult { + let (cp, len) = self.decode_utf8(); + if len == 1 || cp == 0 { + return Err(self.err(self.here(), "Invalid UTF-8")); + } + if cp == 0xFFFE || cp == 0xFFFF { + return Err(self.err_invalid_char()); + } + Ok(len) + } + + // ── input frames and entities ────────────────────────────────────────── + + fn push_frame( + &mut self, + text: &'a [u8], + kind: FrameKind, + entity: (&'a [u8], bool), + ref_pos: usize, + ) -> PResult<()> { + if self.suspended.len() >= MAX_ENTITY_DEPTH { + return Err(self.err(ref_pos, "Entity references are nested too deeply")); + } + // WFC: No Recursion. + if self.frame_entity == Some(entity) + || self.suspended.iter().any(|f| f.entity == Some(entity)) + { + return Err(self.err_named(ref_pos, "Entity", entity.0, " refers to itself")); + } + self.expanded_bytes += text.len() as u64; + let produced = self.document_len + self.expanded_bytes; + if produced > AMPLIFICATION_THRESHOLD + && produced > MAX_AMPLIFICATION * self.document_len.max(1) + { + return Err(self.err(ref_pos, "Entity expansion exceeds the amplification limit")); + } + let report_pos = if self.in_document() { + ref_pos + } else { + self.frame_report_pos + }; + self.suspended.push(Frame { + src: self.src, + pos: self.pos, + id: self.frame_id, + kind: self.frame_kind, + entity: self.frame_entity, + report_pos: self.frame_report_pos, + }); + self.src = text; + self.pos = 0; + self.frame_id = self.next_frame_id; + self.next_frame_id += 1; + self.frame_kind = kind; + self.frame_entity = Some(entity); + self.frame_report_pos = report_pos; + Ok(()) + } + + fn pop_frame(&mut self) { + let frame = self + .suspended + .pop() + .expect("pop_frame without a suspended frame"); + self.src = frame.src; + self.pos = frame.pos; + self.frame_id = frame.id; + self.frame_kind = frame.kind; + self.frame_entity = frame.entity; + self.frame_report_pos = frame.report_pos; + } + + /// Resolves `&name;` for inclusion in content or an attribute value. + fn resolve_general_entity( + &mut self, + name: &'a [u8], + ref_pos: usize, + in_attribute: bool, + ) -> PResult> { + if let Some(c) = predefined_entity(name) { + return Ok(Resolved::Byte(c)); + } + match self.entities.general.get(name).copied() { + Some(EntityValue::Internal(text)) => Ok(Resolved::Text(text)), + // WFC: No External Entity References. + Some(EntityValue::External) if in_attribute => Err(self.err_named( + ref_pos, + "Attribute values cannot reference external entity", + name, + "", + )), + // A non-validating processor may decline to include an external + // entity but must let the application know it was there + // (§4.4.3): the reference is kept as written. + Some(EntityValue::External) => Ok(Resolved::Unexpanded), + // WFC: Parsed Entity. + Some(EntityValue::Unparsed) => { + Err(self.err_named(ref_pos, "Unparsed entity", name, " cannot be referenced")) + } + // WFC: Entity Declared applies to documents without a DTD, with + // standalone="yes", or whose DTD has no parameter-entity + // references and no external subset. Otherwise the declaration + // may live in the part of the DTD that is not loaded, an + // undeclared entity is only a validity error, and the reference + // is kept as written. + None if self.standalone || !(self.has_external_subset || self.saw_pe_reference) => { + Err(self.err_named(ref_pos, "Entity", name, " is not declared")) + } + None => Ok(Resolved::Unexpanded), + } + } + + /// Appends `&name;` for a reference that is kept rather than expanded. + fn push_reference(buf: &mut ArenaVec<'a, u8>, name: &[u8], is_ascii: &mut bool) { + *is_ascii &= name.is_ascii(); + buf.push(b'&'); + buf.extend_from_slice(name); + buf.push(b';'); + } + + /// Includes the parameter entity `name` as declarations (§4.4.8): pushes + /// its replacement text (the caller accounts for the space it counts as + /// on either side), or records that an entity that is not read was + /// referenced. + fn include_parameter_entity(&mut self, name: &'a [u8], ref_pos: usize) -> PResult<()> { + self.saw_pe_reference = true; + match self.entities.parameter.get(name).copied() { + Some(EntityValue::Internal(text)) => { + self.push_frame(text, FrameKind::Declarations, (name, true), ref_pos) + } + Some(_) => { + self.saw_unread_pe = true; + Ok(()) + } + // Undeclared: only a well-formedness error when standalone (WFC: + // Entity Declared); otherwise the DTD is merely incomplete. + None if self.standalone => { + Err(self.err_named(ref_pos, "Parameter entity", name, " is not declared")) + } + None => { + self.saw_unread_pe = true; + Ok(()) + } + } + } + + // ── document setup ───────────────────────────────────────────────────── + + /// Byte-order mark handling and UTF-16 detection (§4.3.3, Appendix F); + /// UTF-16 input is transcoded to UTF-8 up front. Returns whether an XML + /// declaration may follow, in which case validating the input has to + /// wait for the encoding it declares. + fn init_document(&mut self) -> PResult { + let bytes = self.src; + if bytes.starts_with(b"\xEF\xBB\xBF") { + self.pos = 3; + self.saw_utf8_bom = true; + } else if self.encoding == InputEncoding::Text { + // A JS string is characters, not bytes: nothing to detect. + } else if bytes.starts_with(b"\xFE\xFF") { + self.transcode_utf16(&bytes[2..], true)?; + } else if bytes.starts_with(b"\xFF\xFE") { + self.transcode_utf16(&bytes[2..], false)?; + } else if bytes.starts_with(b"\x00<") { + self.transcode_utf16(bytes, true)?; + self.needs_utf16_declaration = true; + } else if bytes.starts_with(b"<\x00") { + self.transcode_utf16(bytes, false)?; + self.needs_utf16_declaration = true; + } + self.content_start = self.pos; + Ok(self.starts_with(b" PResult<()> { + let result = simdutf::validate::with_errors::utf8(self.src); + if result.is_successful() { + Ok(()) + } else { + Err(self.err(result.count, "Invalid UTF-8")) + } + } + + fn transcode_utf16(&mut self, payload: &[u8], big_endian: bool) -> PResult<()> { + let (pairs, rest) = payload.as_chunks::<2>(); + if !rest.is_empty() { + return Err(self.err(payload.len(), "UTF-16 input has an odd number of bytes")); + } + let units: Vec = pairs + .iter() + .map(|&p| { + if big_endian { + u16::from_be_bytes(p) + } else { + u16::from_le_bytes(p) + } + }) + .collect(); + let mut utf8 = vec![0u8; simdutf::length::utf8::from::utf16::le(&units)]; + let result = simdutf::convert::utf16::to::utf8::with_errors::le(&units, &mut utf8); + if !result.is_successful() { + return Err(self.err(result.count * 2, "Invalid UTF-16")); + } + utf8.truncate(result.count); + self.src = self.bump.alloc_slice_copy(&utf8); + self.pos = 0; + self.transcoded = true; + Ok(()) + } + + /// UTF-16 without a byte-order mark is only legal with an encoding + /// declaration naming UTF-16 (§4.3.3); `apply_declared_encoding` clears + /// the requirement when it sees one. + fn check_utf16_declaration(&mut self) -> PResult<()> { + if self.needs_utf16_declaration { + return Err(self.err( + 0, + "UTF-16 input must start with a byte-order mark or declare encoding=\"UTF-16\"", + )); + } + Ok(()) + } + + /// Acts on the encoding named by the XML declaration (the cursor is just + /// past the declaration, which is ASCII in every supported encoding). + fn apply_declared_encoding(&mut self, name: &[u8], pos: usize) -> PResult<()> { + if self.encoding == InputEncoding::Text { + return Ok(()); + } + let is = |canonical: &str| name.eq_ignore_ascii_case(canonical.as_bytes()); + if is("UTF-8") || is("UTF8") || is("US-ASCII") || is("ASCII") { + if self.transcoded { + return Err(self.err_named( + pos, + "Document is UTF-16 but declares encoding", + name, + "", + )); + } + Ok(()) + } else if is("UTF-16") || is("UTF-16LE") || is("UTF-16BE") { + if !self.transcoded && self.encoding == InputEncoding::Bytes { + return Err(self.err_named( + pos, + "Document is not UTF-16 but declares encoding", + name, + "", + )); + } + self.needs_utf16_declaration = false; + Ok(()) + } else if is("ISO-8859-1") || is("ISO_8859-1") || is("LATIN1") || is("L1") || is("CP819") { + if self.transcoded { + return Err(self.err_named( + pos, + "Document is UTF-16 but declares encoding", + name, + "", + )); + } + if self.saw_utf8_bom { + return Err(self.err_named( + pos, + "Document has a UTF-8 byte-order mark but declares encoding", + name, + "", + )); + } + if let Some(utf8) = strings::to_utf8_from_latin1(&self.src[self.pos..]) { + self.src = self.bump.alloc_slice_copy(&utf8); + self.pos = 0; + self.content_start = usize::MAX; + self.transcoded = true; + } + Ok(()) + } else { + Err(self.err_named( + pos, + "Unsupported encoding", + name, + " (supported: UTF-8, UTF-16, ISO-8859-1)", + )) + } + } + + // ── names, references, literals ──────────────────────────────────────── + + /// `Name` (§2.3 [5]) at the cursor, where the grammar allows nothing + /// else (after `<`, ` PResult<&'a [u8]> { + let start = self.pos; + if !self.at_name_start() { + return Err(self.err_here(what)); + } + let (_, len) = self.decode_utf8(); + self.pos += len; + self.scan_name_chars(); + Ok(&self.src[start..self.pos]) + } + + /// Whether a `NameStartChar` is at the cursor. + fn at_name_start(&self) -> bool { + let c = self.peek(); + if c < 0x80 { + is_name_start_ascii(c) + } else { + is_name_start_code_point(self.decode_utf8().0) + } + } + + fn scan_name_chars(&mut self) { + loop { + let c = self.peek(); + if is_name_char_ascii(c) { + self.pos += 1; + } else if c >= 0x80 { + let (cp, len) = self.decode_utf8(); + if !is_name_code_point(cp) { + return; + } + self.pos += len; + } else { + return; + } + } + } + + /// A maximal run of `NameChar`s at the cursor and whether it starts with + /// a `NameStartChar` (a `Name`) or not (only an `Nmtoken`); `None` if the + /// character at the cursor cannot start either. + fn scan_name_run(&mut self) -> Option<(&'a [u8], bool)> { + let start = self.pos; + let c = self.peek(); + let (is_name, len) = if c < 0x80 { + if is_name_start_ascii(c) { + (true, 1) + } else if is_name_char_ascii(c) { + (false, 1) + } else { + return None; + } + } else { + let (cp, len) = self.decode_utf8(); + if is_name_start_code_point(cp) { + (true, len) + } else if is_name_code_point(cp) { + (false, len) + } else { + return None; + } + }; + self.pos += len; + self.scan_name_chars(); + Some((&self.src[start..self.pos], is_name)) + } + + /// `Name ';'` after `&` or `%`. + fn scan_reference_name(&mut self, what: &'static str) -> PResult<&'a [u8]> { + let name = self.scan_name(what)?; + if self.peek() != b';' { + return Err(self.err_here("Expected ';' after the entity name but found")); + } + self.pos += 1; + Ok(name) + } + + /// The rest of `&#...;` / `&#x...;` after `&#`. Returns the code point, + /// which must be a `Char` (WFC: Legal Character). + fn scan_char_ref(&mut self, ref_pos: usize) -> PResult { + let text_start = self.pos - 2; + let hex = self.peek() == b'x'; + if hex { + self.pos += 1; + } + let mut value: u32 = 0; + let mut digits = 0; + loop { + let c = self.peek(); + let digit = match c { + b'0'..=b'9' => u32::from(c - b'0'), + b'a'..=b'f' if hex => u32::from(c - b'a' + 10), + b'A'..=b'F' if hex => u32::from(c - b'A' + 10), + _ => break, + }; + value = value + .saturating_mul(if hex { 16 } else { 10 }) + .saturating_add(digit); + digits += 1; + self.pos += 1; + } + if digits == 0 || self.peek() != b';' { + return Err(self.err_here( + "Invalid character reference: expected a number followed by ';' but found", + )); + } + self.pos += 1; + if !is_xml_char(value) { + let text = &self.src[text_start..self.pos]; + return Err(self.err_named( + ref_pos, + "Character reference", + text, + " is not a valid XML character", + )); + } + Ok(value) + } + + fn push_code_point(buf: &mut ArenaVec<'a, u8>, cp: u32, is_ascii: &mut bool) { + let mut tmp = [0u8; 4]; + let n = strings::encode_wtf8_rune(&mut tmp, cp); + if cp >= 0x80 { + *is_ascii = false; + } + buf.extend_from_slice(&tmp[..n]); + } + + /// Copies the borrowed run `src[start..end]` into a buffer the first + /// time decoding has to diverge from the source bytes. + fn materialize<'b>( + bump: &'a Bump, + src: &[u8], + start: usize, + end: usize, + buf: &'b mut Option>, + ) -> &'b mut ArenaVec<'a, u8> { + if buf.is_none() { + let mut b: ArenaVec<'a, u8> = ArenaVec::with_capacity_in(end - start + 32, bump); + b.extend_from_slice(&src[start..end]); + *buf = Some(b); + } + buf.as_mut().expect("just set") + } + + /// A `Plain`, `System` or `Pubid` literal after the opening quote: no + /// references are recognized; the kinds differ only in the characters + /// they admit. Returns (value, is_ascii). + fn scan_simple_literal( + &mut self, + quote: u8, + open: usize, + literal: Literal, + ) -> PResult<(&'a [u8], bool)> { + let start = self.pos; + let mut is_ascii = true; + loop { + match self.peek() { + _ if self.at_end() => return Err(self.err(open, "Unterminated quoted string")), + c if c == quote => { + let value = &self.src[start..self.pos]; + self.pos += 1; + return Ok((value, is_ascii)); + } + c if literal == Literal::Pubid && !is_pubid_char(c) => { + return Err(self.err_here("Invalid character in a public identifier:")); + } + b'<' | b'>' if literal == Literal::Plain => { + return Err(self.err_here("Invalid character in a quoted string:")); + } + c if c >= 0x80 => { + is_ascii = false; + self.pos += self.check_non_ascii_char()?; + } + c if c < 0x20 && !is_ws(c) => return Err(self.err_invalid_char()), + _ => self.pos += 1, + } + } + } + + /// `AttValue` (§2.3 [10]) after the opening quote, normalized per §3.3.3: + /// a character reference appends the character, an entity reference + /// appends its (recursively normalized) replacement text, a whitespace + /// character appends a space; then, for a tokenized type (`collapse`), + /// spaces are trimmed and collapsed. Returns (value, is_ascii). + fn scan_att_value(&mut self, quote: u8, collapse: bool) -> PResult<(&'a [u8], bool)> { + let literal_frame = self.frame_id; + let open_pos = self.here(); + // The value borrows `src[start..]` until normalization or a + // reference forces a copy; once `buf` exists everything is appended. + let start = self.pos; + let mut buf: Option> = None; + let mut is_ascii = true; + loop { + let c = self.peek(); + match c { + _ if self.at_end() => { + if self.frame_id != literal_frame { + self.pop_frame(); + continue; + } + return Err(self.err(open_pos, "Unterminated attribute value")); + } + _ if c == quote && self.frame_id == literal_frame => { + let end = self.pos; + self.pos += 1; + let value = match buf { + Some(b) => b.into_bump_slice(), + None => &self.src[start..end], + }; + let value = if collapse { + collapse_spaces(self.bump, value) + } else { + value + }; + return Ok((value, is_ascii)); + } + // WFC: No < in Attribute Values (also via replacement text). + b'<' => return Err(self.err(self.here(), "'<' is not allowed in attribute values")), + b'&' => { + let ref_pos = self.here(); + let b = Self::materialize(self.bump, self.src, start, self.pos, &mut buf); + self.pos += 1; + if self.peek() == b'#' { + self.pos += 1; + let cp = self.scan_char_ref(ref_pos)?; + Self::push_code_point(b, cp, &mut is_ascii); + } else { + let name = self + .scan_reference_name("Expected an entity name after '&' but found")?; + match self.resolve_general_entity(name, ref_pos, true)? { + Resolved::Byte(byte) => b.push(byte), + Resolved::Text(text) => { + self.push_frame(text, FrameKind::Literal, (name, false), ref_pos)? + } + Resolved::Unexpanded => Self::push_reference(b, name, &mut is_ascii), + } + } + } + b'\r' if self.in_document() => { + // A line end in the document (CR or CRLF) is one #xA, + // hence one space. + Self::materialize(self.bump, self.src, start, self.pos, &mut buf).push(b' '); + self.pos += 1; + if self.peek() == b'\n' { + self.pos += 1; + } + } + b'\t' | b'\n' | b'\r' => { + Self::materialize(self.bump, self.src, start, self.pos, &mut buf).push(b' '); + self.pos += 1; + } + _ if c < 0x20 => return Err(self.err_invalid_char()), + _ => { + let len = if c >= 0x80 { + is_ascii = false; + self.check_non_ascii_char()? + } else { + 1 + }; + if let Some(b) = buf.as_mut() { + b.extend_from_slice(&self.src[self.pos..self.pos + len]); + } + self.pos += len; + } + } + } + } + + /// `EntityValue` (§2.3 [9]) after the opening quote: character + /// references are included; parameter-entity references are included in + /// literal, which is only legal outside the internal subset proper (WFC: + /// PEs in Internal Subset); general entity references are bypassed — + /// checked for form and kept verbatim (§4.4.7). + fn scan_entity_value(&mut self, quote: u8) -> PResult<(&'a [u8], bool)> { + let literal_frame = self.frame_id; + let in_internal_subset = self.in_document(); + let open_pos = self.here(); + let mut buf: ArenaVec<'a, u8> = ArenaVec::with_capacity_in(32, self.bump); + let mut is_ascii = true; + loop { + let c = self.peek(); + match c { + _ if self.at_end() => { + if self.frame_id != literal_frame { + self.pop_frame(); + continue; + } + return Err(self.err(open_pos, "Unterminated entity value")); + } + _ if c == quote && self.frame_id == literal_frame => { + self.pos += 1; + return Ok((buf.into_bump_slice(), is_ascii)); + } + b'%' => { + let ref_pos = self.here(); + self.pos += 1; + let name = self.scan_reference_name( + "Expected a parameter entity name after '%' but found", + )?; + if in_internal_subset { + return Err(self.err(ref_pos, "Parameter entity references are not allowed inside markup declarations in the internal subset")); + } + self.saw_pe_reference = true; + match self.entities.parameter.get(name).copied() { + Some(EntityValue::Internal(text)) => { + self.push_frame(text, FrameKind::Literal, (name, true), ref_pos)? + } + Some(_) => { + return Err(self.err_named( + ref_pos, + "External parameter entity", + name, + " cannot be included because external entities are not loaded", + )); + } + None => { + return Err(self.err_named( + ref_pos, + "Parameter entity", + name, + " is not declared", + )); + } + } + } + b'&' => { + let ref_pos = self.here(); + self.pos += 1; + if self.peek() == b'#' { + self.pos += 1; + let cp = self.scan_char_ref(ref_pos)?; + Self::push_code_point(&mut buf, cp, &mut is_ascii); + } else { + let name = self + .scan_reference_name("Expected an entity name after '&' but found")?; + Self::push_reference(&mut buf, name, &mut is_ascii); + } + } + b'\r' if self.in_document() => { + buf.push(b'\n'); + self.pos += 1; + if self.peek() == b'\n' { + self.pos += 1; + } + } + _ if c < 0x20 && !is_ws(c) => return Err(self.err_invalid_char()), + _ => { + let len = if c >= 0x80 { + is_ascii = false; + self.check_non_ascii_char()? + } else { + 1 + }; + buf.extend_from_slice(&self.src[self.pos..self.pos + len]); + self.pos += len; + } + } + } + } + + // ── comments and processing instructions ─────────────────────────────── + + /// The rest of a comment after ` + + + + + + watch on + + diff --git a/test/js/bun/resolve/xml/xml-fixture.xml.txt b/test/js/bun/resolve/xml/xml-fixture.xml.txt new file mode 100644 index 000000000000..c84e108f7172 --- /dev/null +++ b/test/js/bun/resolve/xml/xml-fixture.xml.txt @@ -0,0 +1 @@ +remember diff --git a/test/js/bun/resolve/xml/xml-latin1.xml b/test/js/bun/resolve/xml/xml-latin1.xml new file mode 100644 index 000000000000..36a13d76c8f2 --- /dev/null +++ b/test/js/bun/resolve/xml/xml-latin1.xml @@ -0,0 +1 @@ +café diff --git a/test/js/bun/resolve/xml/xml-malformed.xml b/test/js/bun/resolve/xml/xml-malformed.xml new file mode 100644 index 000000000000..303c11bcb9dc --- /dev/null +++ b/test/js/bun/resolve/xml/xml-malformed.xml @@ -0,0 +1,4 @@ + + app + 8080 + diff --git a/test/js/bun/resolve/xml/xml-utf16be-bom.xml b/test/js/bun/resolve/xml/xml-utf16be-bom.xml new file mode 100644 index 000000000000..bf84498adb96 Binary files /dev/null and b/test/js/bun/resolve/xml/xml-utf16be-bom.xml differ diff --git a/test/js/bun/resolve/xml/xml-utf16le-bom.xml b/test/js/bun/resolve/xml/xml-utf16le-bom.xml new file mode 100644 index 000000000000..86bab01f0124 Binary files /dev/null and b/test/js/bun/resolve/xml/xml-utf16le-bom.xml differ diff --git a/test/js/bun/resolve/xml/xml-utf8-bom.xml b/test/js/bun/resolve/xml/xml-utf8-bom.xml new file mode 100644 index 000000000000..8f0717c7e365 --- /dev/null +++ b/test/js/bun/resolve/xml/xml-utf8-bom.xml @@ -0,0 +1 @@ +café 日本 diff --git a/test/js/bun/resolve/xml/xml.test.js b/test/js/bun/resolve/xml/xml.test.js new file mode 100644 index 000000000000..e46a07c7e998 --- /dev/null +++ b/test/js/bun/resolve/xml/xml.test.js @@ -0,0 +1,90 @@ +import { expect, it } from "bun:test"; +import { tempDir } from "harness"; +import { join } from "node:path"; +import empty from "./xml-empty.xml"; +import fixture, * as namespace from "./xml-fixture.xml"; +import xmlFromCustomTypeAttribute from "./xml-fixture.xml.txt" with { type: "xml" }; + +const expectedFixture = { + config: { + "@framework": "next", + bundle: { + package: [ + { "@name": "@emotion/react", "@enabled": "true" }, + { "@name": "lodash", "@enabled": "false" }, + ], + }, + dev: { "@port": "3000", b: "on", "#text": "watch" }, + empty: "", + }, +}; + +it("via import statement", () => { + expect(fixture).toEqual(expectedFixture); +}); + +it("the root element is also a named export", () => { + expect(namespace.config).toEqual(expectedFixture.config); + expect(namespace.default).toEqual(expectedFixture); +}); + +it("via dynamic import", async () => { + const xml = (await import("./xml-fixture.xml")).default; + expect(xml).toEqual(expectedFixture); +}); + +it("via require", () => { + // Already in the module registry from the import above, so this is the + // namespace object: the root element is reachable either way. + expect(require("./xml-fixture.xml").config).toEqual(expectedFixture.config); +}); + +it("via import type xml", () => { + expect(xmlFromCustomTypeAttribute).toEqual({ note: { "@importance": "high", "#text": "remember" } }); +}); + +it("via dynamic import with type attribute", async () => { + delete require.cache[require.resolve("./xml-fixture.xml.txt")]; + const xml = (await import("./xml-fixture.xml.txt", { with: { type: "xml" } })).default; + expect(xml).toEqual({ note: { "@importance": "high", "#text": "remember" } }); +}); + +it("an empty file is an empty object, like the other data loaders", () => { + expect(empty).toEqual({}); +}); + +it("files in the encodings XML requires are decoded before parsing", async () => { + const expected = { doc: { "@lang": "français", "#text": "café 日本" } }; + expect((await import("./xml-utf16le-bom.xml")).default).toEqual(expected); + expect((await import("./xml-utf16be-bom.xml")).default).toEqual(expected); + expect((await import("./xml-utf8-bom.xml")).default).toEqual(expected); + expect((await import("./xml-latin1.xml")).default).toEqual({ doc: { "@lang": "français", "#text": "café" } }); +}); + +it("a malformed file is a build error with the parser's message and location", async () => { + let error; + try { + await import("./xml-malformed.xml"); + } catch (e) { + error = e; + } + expect(error?.name).toBe("BuildMessage"); + expect(error.message).toBe("Expected closing tag but found "); + expect(error.position).toMatchObject({ line: 3, column: 13, lineText: " 8080" }); +}); + +it("a document nested too deeply to parse is a build error, not a crash or a missing module", async () => { + // Deep enough to overflow any native stack, whatever the frame size. + const depth = 2_000_000; + using dir = tempDir("xml-deep", { + "deep.xml": Buffer.alloc(depth * 3, "").toString() + Buffer.alloc(depth * 4, "").toString(), + }); + let error; + try { + await import(join(String(dir), "deep.xml")); + } catch (e) { + error = e; + } + expect(error?.name).toBe("BuildMessage"); + expect(error.message).toBe("Nesting is too deep"); +}); diff --git a/test/js/bun/transpiler/transpiler-unsupported-loader.test.ts b/test/js/bun/transpiler/transpiler-unsupported-loader.test.ts index 496c2a44fa22..2cc5d87d5a41 100644 --- a/test/js/bun/transpiler/transpiler-unsupported-loader.test.ts +++ b/test/js/bun/transpiler/transpiler-unsupported-loader.test.ts @@ -30,7 +30,7 @@ describe("Bun.Transpiler rejects non-transpilable loaders", () => { test("unknown-loader message lists only transpilable loaders", () => { expect(() => t.transformSync("let x = 1", "bogus" as any)).toThrow(TypeError); expect(() => t.transformSync("let x = 1", "bogus" as any)).toThrow( - "invalid loader - must be js, jsx, tsx, ts, css, json, jsonc, json5, toml, yaml, text, wasm, or md", + "invalid loader - must be js, jsx, tsx, ts, css, json, jsonc, json5, toml, yaml, xml, text, wasm, or md", ); }); }); @@ -46,6 +46,13 @@ describe("Bun.Transpiler still accepts data-format loaders", () => { expect(t.transformSync("a = 1", "toml")).toContain("export default"); }); + test("xml", () => { + const out = t.transformSync(`x`, "xml"); + expect(out).toContain("export default"); + expect(out).toContain('"@b": "1"'); + expect(out).toContain('c: "x"'); + }); + test("text", () => { expect(t.transformSync("hello", "text")).toContain("export default"); }); diff --git a/test/js/bun/xml/generate_xml_test_suite.ts b/test/js/bun/xml/generate_xml_test_suite.ts new file mode 100644 index 000000000000..653f8ae70de8 --- /dev/null +++ b/test/js/bun/xml/generate_xml_test_suite.ts @@ -0,0 +1,648 @@ +#!/usr/bin/env bun +/** + * Generates xml-test-suite.test.ts from the W3C XML Conformance Test Suite + * (xmlts20130923, https://www.w3.org/XML/Test/). + * + * Usage: + * bun bd test/js/bun/xml/generate_xml_test_suite.ts [path-to-xmlconf-dir] [--check] + * + * Must run under the debug build (`bun bd`): rejection-test error messages, and + * the behaviour pinned for cases whose verdict depends on reading external + * entities, are captured from the in-tree Bun.XML at generation time. Without + * an in-tree Bun.XML those assertions are omitted / emitted as test.todo. + * + * If no path is given, downloads the pinned archive from w3.org into a temp + * directory (checked against ARCHIVE_SHA256) and extracts it with `tar`. + * --check regenerates beside the committed suite and exits 1 if it differs. + * + * Scope: every catalogued case that applies to an XML 1.0 (Fifth Edition) + * processor — cases for XML 1.1 / Namespaces 1.1, and cases restricted to + * editions 1-4 (EDITION="1 2 3 4", superseded by Fifth Edition erratum E09), + * are skipped. The six 180-313 KB `japanese/pr-xml-*` samples are skipped too: + * they are ENTITIES="parameter" (advisory for a processor that does not read + * external entities) and the same six encodings are covered by the 2-3 KB + * `japanese/weekly-*` documents. + * + * How a case is asserted (Bun.XML is a non-validating processor that never + * reads external entities — XML 1.0 §5.1): + * - TYPE="not-wf", ENTITIES="none": XML.parse throws SyntaxError (exact + * message asserted when captured at generation time). + * - TYPE="valid" | "invalid", ENTITIES="none": XML.parse accepts in both + * shapes (non-validating processors accept invalid-but-well-formed + * documents). When the catalogue gives a canonical OUTPUT file, the node + * tree ({ compact: false }) must serialize to it exactly (Second Canonical + * Form minus the DOCTYPE/notation block and processing instructions, + * which Bun.XML does not represent), the compact object must equal the + * projection of that output, and stringify() of either shape must parse + * back to the same result. + * - ENTITIES != "none", TYPE="error", and the Namespaces-1.0 collection + * (Bun.XML treats names as opaque strings and does not enforce namespace + * constraints): the verdict legitimately depends on processor class, so the + * in-tree behaviour is pinned at generation time with the upstream verdict + * noted in a comment. + */ + +import { execFileSync } from "node:child_process"; +import { createHash } from "node:crypto"; +import { existsSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; + +// --------------------------------------------------------------------------- +// 1. Locate the suite +// --------------------------------------------------------------------------- +const ARCHIVE_URL = "https://www.w3.org/XML/Test/xmlts20130923.tar.gz"; +const ARCHIVE_SHA256 = "9b61db9f5dbffa545f4b8d78422167083a8568c59bd1129f94138f936cf6fc1f"; + +const checkMode = process.argv.includes("--check"); +let suiteDir = process.argv.slice(2).find(a => a !== "--check"); +if (!suiteDir) { + const tmp = mkdtempSync(join(tmpdir(), "xmlconf-")); + const archive = join(tmp, "xmlts20130923.tar.gz"); + console.log(`Downloading ${ARCHIVE_URL} into ${tmp} ...`); + execFileSync("curl", ["-fsSL", "-o", archive, ARCHIVE_URL], { stdio: "inherit" }); + const digest = createHash("sha256").update(readFileSync(archive)).digest("hex"); + if (digest !== ARCHIVE_SHA256) { + console.error(`SHA-256 mismatch for ${archive}: got ${digest}, expected ${ARCHIVE_SHA256}`); + process.exit(1); + } + execFileSync("tar", ["-xzf", archive, "-C", tmp], { stdio: "inherit" }); + suiteDir = join(tmp, "xmlconf"); +} +if (!existsSync(join(suiteDir, "xmlconf.xml"))) { + console.error(`${suiteDir} does not look like an extracted xmlconf directory (no xmlconf.xml)`); + process.exit(1); +} + +const XML = (Bun as unknown as { XML?: { parse: Function; stringify: Function } }).XML; +if (!XML) { + console.warn( + "Bun.XML is not available in this build: error messages will not be asserted and advisory cases become test.todo.", + ); +} + +// --------------------------------------------------------------------------- +// 2. Read the catalogue +// --------------------------------------------------------------------------- +// xmlconf.xml stitches the per-collection catalogues together with external +// entities; each collection's TESTCASES element carries the xml:base its URIs +// resolve against. Listed in xmlconf.xml order. +const COLLECTIONS: { describe: string; catalog: string; base: string }[] = [ + { describe: "xmltest", catalog: "xmltest/xmltest.xml", base: "xmltest/" }, + { describe: "japanese", catalog: "japanese/japanese.xml", base: "japanese/" }, + { describe: "sun", catalog: "sun/sun-valid.xml", base: "sun/" }, + { describe: "sun", catalog: "sun/sun-invalid.xml", base: "sun/" }, + { describe: "sun", catalog: "sun/sun-not-wf.xml", base: "sun/" }, + { describe: "sun", catalog: "sun/sun-error.xml", base: "sun/" }, + { describe: "oasis", catalog: "oasis/oasis.xml", base: "oasis/" }, + { describe: "ibm", catalog: "ibm/ibm_oasis_invalid.xml", base: "ibm/" }, + { describe: "ibm", catalog: "ibm/ibm_oasis_not-wf.xml", base: "ibm/" }, + { describe: "ibm", catalog: "ibm/ibm_oasis_valid.xml", base: "ibm/" }, + { describe: "eduni/errata-2e", catalog: "eduni/errata-2e/errata2e.xml", base: "eduni/errata-2e/" }, + { describe: "eduni/errata-3e", catalog: "eduni/errata-3e/errata3e.xml", base: "eduni/errata-3e/" }, + { describe: "eduni/errata-4e", catalog: "eduni/errata-4e/errata4e.xml", base: "eduni/errata-4e/" }, + { describe: "eduni/namespaces-1.0", catalog: "eduni/namespaces/1.0/rmt-ns10.xml", base: "eduni/namespaces/1.0/" }, + { + describe: "eduni/namespaces-1.0", + catalog: "eduni/namespaces/errata-1e/errata1e.xml", + base: "eduni/namespaces/errata-1e/", + }, + { describe: "eduni/misc", catalog: "eduni/misc/ht-bh.xml", base: "eduni/misc/" }, + // Not listed: ibm/xml-1.1/*, eduni/xml-1.1/xml11.xml, eduni/namespaces/1.1 — + // XML 1.1 / Namespaces 1.1 only. +]; + +interface Case { + collection: string; + id: string; + type: "valid" | "invalid" | "not-wf" | "error"; + entities: "none" | "general" | "parameter" | "both"; + sections: string; + description: string; + recommendation: string; + namespace: boolean; + input: Buffer; + /** Expected Second Canonical Form, when the catalogue provides one. */ + output: string | undefined; +} + +const attrRe = /([A-Za-z_:][-A-Za-z0-9_:.]*)\s*=\s*(?:"([^"]*)"|'([^']*)')/g; +const cases: Case[] = []; +let skipped11 = 0, + skippedEdition = 0, + skippedBig = 0; +for (const { describe, catalog, base } of COLLECTIONS) { + // The catalogues are ASCII/Latin-1; latin1 keeps byte offsets stable. + const src = readFileSync(join(suiteDir, catalog), "latin1"); + for (const m of src.matchAll(/]*)>([\s\S]*?)<\/TEST>/g)) { + const attrs: Record = {}; + for (const a of m[1].matchAll(attrRe)) attrs[a[1]] = a[2] ?? a[3]; + const recommendation = attrs.RECOMMENDATION ?? "XML1.0"; + if (recommendation === "XML1.1" || recommendation === "NS1.1" || attrs.VERSION === "1.1") { + skipped11++; + continue; + } + if (attrs.EDITION !== undefined && !attrs.EDITION.split(/\s+/).includes("5")) { + skippedEdition++; + continue; + } + if (base === "japanese/" && /^pr-xml-/.test(attrs.ID)) { + skippedBig++; + continue; + } + const input = readFileSync(join(suiteDir, base, attrs.URI)); + let output: string | undefined; + if (attrs.OUTPUT) + output = new TextDecoder("utf-8", { fatal: true }).decode(readFileSync(join(suiteDir, base, attrs.OUTPUT))); + cases.push({ + collection: describe, + id: attrs.ID, + type: attrs.TYPE as Case["type"], + entities: (attrs.ENTITIES ?? "none") as Case["entities"], + sections: attrs.SECTIONS, + description: m[2] + .replace(/<[^>]+>/g, "") + .replace(/</g, "<") + .replace(/>/g, ">") + .replace(/"/g, '"') + .replace(/'/g, "'") + .replace(/&/g, "&") + .replace(/\s+/g, " ") + .trim(), + recommendation, + namespace: attrs.NAMESPACE !== "no", + input, + output, + }); + } +} + +// --------------------------------------------------------------------------- +// 3. Expected values: read Second Canonical Form, project to both shapes +// --------------------------------------------------------------------------- +// The OUTPUT files use "Second Canonical Form" (sun/cxml.html in the suite): +// James Clark's Canonical XML — UTF-8, no XML declaration, attributes sorted, +// & < > " as the only references, empty +// elements as start/end pairs, no comments, PIs kept — optionally preceded by +// a DOCTYPE listing notations. That grammar is small enough to read exactly. +type XNode = { name: string; attributes: Record; children: (XNode | string)[] }; + +function readCanonical(src: string, id: string): XNode { + let i = 0; + const fail = (msg: string) => new Error(`${id}: cannot read canonical output: ${msg} at ${i}`); + const unescape = (s: string) => + s.replace(/&(amp|lt|gt|quot|apos|#([0-9]+));/g, (_, name, dec) => + dec ? String.fromCodePoint(Number(dec)) : { amp: "&", lt: "<", gt: ">", quot: '"', apos: "'" }[name as "amp"], + ); + const skipPI = () => { + const end = src.indexOf("?>", i); + if (end < 0) throw fail("unterminated PI"); + i = end + 2; + }; + const readElement = (): XNode => { + const m = /^<([^\s>/]+)((?: [^\s=]+="[^"]*")*)>/.exec(src.slice(i)); + if (!m) throw fail("bad start tag"); + i += m[0].length; + const node: XNode = { name: m[1], attributes: {}, children: [] }; + for (const a of m[2].matchAll(/ ([^\s=]+)="([^"]*)"/g)) node.attributes[a[1]] = unescape(a[2]); + let text = ""; + const flush = () => { + // Text split only by a (dropped) PI is one text node in Bun's tree. + if (!text) return; + const last = node.children.length - 1; + if (last >= 0 && typeof node.children[last] === "string") node.children[last] += unescape(text); + else node.children.push(unescape(text)); + text = ""; + }; + while (i < src.length) { + if (src.startsWith("", i); + if (src.slice(i + 2, end) !== node.name) throw fail("mismatched end tag"); + i = end + 1; + return node; + } else if (src.startsWith("\n", i); + if (end < 0) throw fail("unterminated DOCTYPE"); + i = end + 3; + } else if (root === undefined && src[i] === "<") root = readElement(); + else throw fail("unexpected content at top level"); + } + if (!root) throw fail("no root element"); + return root; +} + +const CANON_ESCAPES: Record = { + "&": "&", + "<": "<", + ">": ">", + '"': """, + "\t": " ", + "\n": " ", + "\r": " ", +}; +function codePointCompare(a: string, b: string): number { + const x = [...a], + y = [...b]; + for (let k = 0; k < x.length && k < y.length; k++) { + const d = x[k].codePointAt(0)! - y[k].codePointAt(0)!; + if (d !== 0) return d; + } + return x.length - y.length; +} +function writeCanonical(node: XNode | string): string { + if (typeof node === "string") return node.replace(/[&<>"\t\n\r]/g, c => CANON_ESCAPES[c]); + const attrs = Object.keys(node.attributes) + .sort(codePointCompare) + .map(k => ` ${k}="${node.attributes[k].replace(/[&<>"\t\n\r]/g, c => CANON_ESCAPES[c])}"`) + .join(""); + return `<${node.name}${attrs}>${node.children.map(writeCanonical).join("")}`; +} + +// Reference implementation of Bun.XML's compact projection (must match +// CompactSink in src/parsers/xml.rs): attributes as "@name", repeated +// same-name children as arrays in document order, all text concatenated and +// trimmed of XML whitespace, a leaf with no attributes collapsing to its text. +const XML_WS = /^[ \t\r\n]+|[ \t\r\n]+$/g; +function toCompact(node: XNode): unknown { + const obj: Record = {}; + let hasAttributes = false; + for (const [k, v] of Object.entries(node.attributes)) { + defineOwn(obj, "@" + k, v); + hasAttributes = true; + } + let text = ""; + let hasElements = false; + const groups = new Map(); + for (const child of node.children) { + if (typeof child === "string") { + text += child; + continue; + } + hasElements = true; + const value = toCompact(child); + const group = groups.get(child.name); + if (group) group.push(value); + else groups.set(child.name, [value]); + } + const trimmed = text.replace(XML_WS, ""); + if (!hasAttributes && !hasElements) return trimmed; + for (const [name, values] of groups) defineOwn(obj, name, values.length === 1 ? values[0] : values); + if (trimmed !== "") defineOwn(obj, "#text", trimmed); + return obj; +} +function defineOwn(obj: Record, key: string, value: unknown) { + Object.defineProperty(obj, key, { value, enumerable: true, configurable: true, writable: true }); +} + +// --------------------------------------------------------------------------- +// 4. Classify inputs and capture in-tree behaviour +// --------------------------------------------------------------------------- +// A JS string handed to XML.parse is already-decoded text, so its encoding +// declaration is (correctly) not acted upon. Cases whose bytes carry encoding +// information the processor must act on — UTF-16, non-UTF-8 8-bit encodings, +// deliberately broken byte sequences, or any encoding declaration at all — are +// therefore passed as bytes; everything else is inlined as a string. +const utf8Strict = new TextDecoder("utf-8", { fatal: true, ignoreBOM: true }); +type Input = + | { kind: "string"; text: string } + | { kind: "utf8-bytes"; text: string } + | { kind: "bytes"; base64: string }; +function classifyInput(bytes: Buffer): Input { + const looks16 = + (bytes[0] === 0xfe && bytes[1] === 0xff) || + (bytes[0] === 0xff && bytes[1] === 0xfe) || + (bytes[0] === 0x00 && bytes[1] === 0x3c) || + (bytes[0] === 0x3c && bytes[1] === 0x00); + if (!looks16) { + try { + const text = utf8Strict.decode(bytes); + if (/^\ufeff?<\?xml[^>]*encoding/.test(text)) return { kind: "utf8-bytes", text }; + return { kind: "string", text }; + } catch {} + } + return { kind: "bytes", base64: bytes.toString("base64") }; +} +function materialize(input: Input): string | Buffer { + if (input.kind === "string") return input.text; + if (input.kind === "utf8-bytes") return Buffer.from(input.text); + return Buffer.from(input.base64, "base64"); +} +function inputDecl(input: Input): string { + if (input.kind === "string") return `const input: string = ${jsString(input.text)};`; + if (input.kind === "utf8-bytes") return `const input = Buffer.from(${jsString(input.text)});`; + return `const input = Buffer.from(${jsString(input.base64)}, "base64");`; +} + +type Observed = { threw: true; message: string | undefined } | { threw: false; canonical: string; compact: unknown }; +function observe(input: string | Buffer): Observed | undefined { + if (!XML) return undefined; + try { + const node = XML.parse(input, { compact: false }) as XNode; + return { threw: false, canonical: writeCanonical(node), compact: XML.parse(input) }; + } catch (e) { + return { threw: true, message: e instanceof SyntaxError ? e.message : undefined }; + } +} + +// --------------------------------------------------------------------------- +// 5. Code generation helpers +// --------------------------------------------------------------------------- +// JSON.stringify covers C0 controls, quotes, and backslashes; additionally +// escape DEL/C1 controls, U+2028/U+2029, U+FEFF and noncharacters so the +// generated source stays visibly ASCII-clean where it matters. +function jsString(s: string): string { + return JSON.stringify(s).replace( + /[\u007f-\u009f\u2028\u2029\ufeff\ufffe\uffff]/g, + c => `\\u${c.charCodeAt(0).toString(16).padStart(4, "0")}`, + ); +} + +function valueToJS(val: unknown, indent: number = 0): string { + if (typeof val === "string") return jsString(val); + if (Array.isArray(val)) { + if (val.length === 0) return "[]"; + const items = val.map(v => valueToJS(v, indent + 1)); + const oneLine = `[${items.join(", ")}]`; + if (oneLine.length < 80 && !oneLine.includes("\n")) return oneLine; + const pad = " ".repeat(indent + 1); + const endPad = " ".repeat(indent); + return `[\n${items.map(i => `${pad}${i},`).join("\n")}\n${endPad}]`; + } + if (val !== null && typeof val === "object") { + const entries = Object.entries(val as Record); + if (entries.length === 0) return "{}"; + const parts = entries.map(([k, v]) => { + // A literal "__proto__" key would set the prototype; the computed form + // creates an own property. + const key = k === "__proto__" ? '["__proto__"]' : /^[a-zA-Z_$][a-zA-Z0-9_$]*$/.test(k) ? k : jsString(k); + return `${key}: ${valueToJS(v, indent + 1)}`; + }); + const oneLine = `{ ${parts.join(", ")} }`; + if (oneLine.length < 80 && !oneLine.includes("\n")) return oneLine; + const pad = " ".repeat(indent + 1); + const endPad = " ".repeat(indent); + return `{\n${parts.map(p => `${pad}${p},`).join("\n")}\n${endPad}}`; + } + throw new Error(`Cannot serialize ${String(val)}`); +} + +function comment(c: Case, extra?: string): string { + const text = `${c.sections} — ${c.description}${extra ? ` (${extra})` : ""}`; + // Wrap long descriptions so prettier leaves the lines alone. + const words = text.split(" "); + const lines: string[] = []; + let line = ""; + for (const w of words) { + if (line && line.length + 1 + w.length > 100) { + lines.push(line); + line = w; + } else { + line = line ? `${line} ${w}` : w; + } + } + if (line) lines.push(line); + return lines.map(l => ` // ${l}\n`).join(""); +} + +function emitAccept(c: Case, decl: string, canonical: string | undefined, compact: unknown, extra?: string): string { + let body = ` test(${jsString(c.id)}, () => {\n`; + body += comment(c, extra); + body += ` ${decl}\n`; + if (canonical !== undefined) { + body += ` const canonical = ${jsString(canonical)};\n`; + body += ` const compact: unknown = ${valueToJS(compact, 2)};\n`; + body += ` expectParses(input, canonical, compact);\n`; + } else { + body += ` expectParses(input);\n`; + } + body += ` });\n\n`; + return body; +} + +function emitReject(c: Case, decl: string, message: string | undefined, extra?: string): string { + let body = ` test(${jsString(c.id)}, () => {\n`; + body += comment(c, extra); + body += ` ${decl}\n`; + body += message === undefined ? ` expectRejects(input);\n` : ` expectRejects(input, ${jsString(message)});\n`; + body += ` });\n\n`; + return body; +} + +function emitTodo(c: Case, decl: string, why: string): string { + let body = ` test.todo(${jsString(c.id)}, () => {\n`; + body += comment(c, why); + body += ` ${decl}\n`; + body += c.type === "not-wf" ? ` expectRejects(input);\n` : ` expectParses(input);\n`; + body += ` });\n\n`; + return body; +} + +// --------------------------------------------------------------------------- +// 6. Emit +// --------------------------------------------------------------------------- +const counts = { reject: 0, accept: 0, acceptWithOutput: 0, pinned: 0, todo: 0 }; +const bodies = new Map(); +for (const c of cases) { + const input = classifyInput(c.input); + const decl = inputDecl(input); + const isNamespaceCase = c.recommendation.startsWith("NS1.0"); + const hard = c.entities === "none" && c.type !== "error" && !isNamespaceCase; + let body: string; + if (hard && c.type === "not-wf") { + const observed = observe(materialize(input)); + body = emitReject(c, decl, observed?.threw ? observed.message : undefined); + counts.reject++; + } else if (hard) { + let canonical: string | undefined; + let compact: unknown; + if (c.output !== undefined) { + const tree = readCanonical(c.output, c.id); + canonical = writeCanonical(tree); + compact = { [tree.name]: toCompact(tree) }; + counts.acceptWithOutput++; + } + body = emitAccept(c, decl, canonical, compact); + counts.accept++; + } else { + // Verdict depends on processor class; pin what Bun.XML does. + const why = isNamespaceCase + ? `upstream: ${c.type}; namespace constraints are not enforced` + : c.type === "error" + ? `upstream: optional error` + : `upstream: ${c.type}; external ${c.entities === "both" ? "general and parameter" : c.entities} entities are not read`; + const observed = observe(materialize(input)); + if (!observed) { + body = emitTodo(c, decl, why); + counts.todo++; + } else if (observed.threw) { + body = emitReject(c, decl, observed.message, why); + counts.pinned++; + } else { + let canonical: string | undefined; + let compact: unknown; + if (c.output !== undefined) { + const tree = readCanonical(c.output, c.id); + if (writeCanonical(tree) === observed.canonical) { + canonical = observed.canonical; + compact = { [tree.name]: toCompact(tree) }; + } + } + body = emitAccept( + c, + decl, + canonical, + compact, + canonical === undefined && c.output ? `${why}; output depends on them` : why, + ); + counts.pinned++; + } + } + bodies.set(c.collection, (bodies.get(c.collection) ?? "") + body); +} + +let output = `// Tests generated from the W3C XML Conformance Test Suite, 20130923 release +// (https://www.w3.org/XML/Test/ — xmlts20130923.tar.gz, SHA-256 ${ARCHIVE_SHA256}). +// The suite was contributed by James Clark, Sun Microsystems, OASIS/NIST, IBM, +// Fuji Xerox and Richard Tobin / the University of Edinburgh; test documents are +// (c) their contributors ("Copyright 1998-1999 by Sun Microsystems, Inc.", +// "Modifications copyright 1999-2001 by OASIS", IBM 2000-2003, ...) and are +// inlined here verbatim as conformance inputs. +// +// Scope: ${cases.length} cases applicable to an XML 1.0 (Fifth Edition) processor — +// ${counts.reject} must-reject + ${counts.accept} must-accept (${counts.acceptWithOutput} with canonical output) + ${counts.pinned + counts.todo} whose +// verdict depends on processor class (pinned to Bun's behaviour, see below). +// Skipped: ${skipped11} XML 1.1 / Namespaces 1.1 cases, ${skippedEdition} cases restricted to editions 1-4 +// (superseded by Fifth Edition erratum E09), ${skippedBig} large japanese/pr-xml-* samples +// (advisory only; the japanese/weekly-* cases cover the same encodings). +// Regenerate with: bun bd test/js/bun/xml/generate_xml_test_suite.ts [path-to-xmlconf] +// +// Bun.XML is a non-validating processor that does not read external entities +// (XML 1.0 §5.1), so: +// - not-wf cases with ENTITIES="none" must throw SyntaxError (exact message +// asserted); +// - valid and invalid cases with ENTITIES="none" must parse (non-validating +// processors accept invalid-but-well-formed documents); when the suite gives +// a canonical output, the { compact: false } tree must serialize to it +// exactly (Second Canonical Form, minus the DOCTYPE/notation block and +// processing instructions, which Bun.XML does not represent), the compact +// object must equal the projection of that output, and XML.stringify of +// either shape must parse back to the same value; +// - cases with external entities, TYPE="error" cases, and the Namespaces 1.0 +// collection (names are opaque strings to Bun.XML; namespace constraints are +// not enforced) are pinned to the in-tree behaviour, with the upstream +// verdict in the comment. +// Inputs whose bytes carry encoding information the processor must act on (an +// encoding declaration, a UTF-16 BOM, non-UTF-8 bytes) are passed as bytes; a JS +// string is already-decoded text and its encoding declaration is not acted upon. +import { XML } from "bun"; +import { describe, expect, test } from "bun:test"; + +type XMLNode = { name: string; attributes: Record; children: (XMLNode | string)[] }; + +const CANON_ESCAPES: Record = { + "&": "&", + "<": "<", + ">": ">", + '"': """, + "\\t": " ", + "\\n": " ", + "\\r": " ", +}; +function codePointCompare(a: string, b: string): number { + const x = [...a]; + const y = [...b]; + for (let i = 0; i < x.length && i < y.length; i++) { + const d = x[i].codePointAt(0)! - y[i].codePointAt(0)!; + if (d !== 0) return d; + } + return x.length - y.length; +} +/** James Clark's Canonical XML for an element tree. */ +function canonicalize(node: XMLNode | string): string { + if (typeof node === "string") return node.replace(/[&<>"\\t\\n\\r]/g, c => CANON_ESCAPES[c]); + const attrs = Object.keys(node.attributes) + .sort(codePointCompare) + .map(k => \` \${k}="\${node.attributes[k].replace(/[&<>"\\t\\n\\r]/g, c => CANON_ESCAPES[c])}"\`) + .join(""); + return \`<\${node.name}\${attrs}>\${node.children.map(canonicalize).join("")}\`; +} + +function tree(input: string | Buffer): XMLNode { + return XML.parse(input, { compact: false }); +} + +function expectParses(input: string | Buffer, canonical?: string, compact?: unknown): void { + const node = tree(input); + const object = XML.parse(input); + if (canonical !== undefined) { + expect(canonicalize(node)).toBe(canonical); + expect(object).toEqual(compact); + } + // stringify of either shape reads back as the same value. + expect(canonicalize(tree(XML.stringify(node)))).toBe(canonicalize(node)); + expect(XML.parse(XML.stringify(object))).toEqual(object); +} + +function expectRejects(input: string | Buffer, message?: string): void { + let err: unknown; + try { + XML.parse(input); + } catch (e) { + err = e; + } + expect(err).toBeInstanceOf(SyntaxError); + if (message !== undefined) expect((err as SyntaxError).message).toBe(message); + expect(() => tree(input)).toThrow(SyntaxError); +} +`; + +for (const name of new Set(COLLECTIONS.map(c => c.describe))) { + const body = bodies.get(name); + if (!body) continue; + output += `\ndescribe(${jsString(name)}, () => {\n${body}});\n`; +} + +const committedPath = join(import.meta.dir, "xml-test-suite.test.ts"); +// The --check comparand must sit beside the committed file: prettier 3 also +// honors .gitignore, whose bare `tmp` rule matches os.tmpdir() on Linux and +// makes prettier silently skip the file there instead of formatting it. +const outPath = checkMode ? join(import.meta.dir, "xml-test-suite.check.ts") : committedPath; +writeFileSync(outPath, output); +const repoRoot = join(import.meta.dir, "../../../.."); +execFileSync( + join(repoRoot, "node_modules/.bin/prettier"), + ["--plugin=prettier-plugin-organize-imports", "--config", join(repoRoot, ".prettierrc"), "--write", outPath], + { stdio: "inherit", cwd: repoRoot }, +); +if (checkMode) { + const fresh = readFileSync(outPath, "utf8"); + const committed = readFileSync(committedPath, "utf8"); + if (fresh !== committed) { + console.error( + `MISMATCH: ${committedPath} is stale; regenerate it (fresh output kept at ${outPath} for comparison).`, + ); + process.exit(1); + } + rmSync(outPath); + console.log(`OK: ${committedPath} is up to date.`); +} else { + console.log( + `Wrote ${outPath}: ${counts.reject} must-reject + ${counts.accept} must-accept (${counts.acceptWithOutput} with output) + ${counts.pinned} pinned + ${counts.todo} todo`, + ); +} diff --git a/test/js/bun/xml/xml-test-suite.test.ts b/test/js/bun/xml/xml-test-suite.test.ts new file mode 100644 index 000000000000..4499bec660ff --- /dev/null +++ b/test/js/bun/xml/xml-test-suite.test.ts @@ -0,0 +1,16815 @@ +// Tests generated from the W3C XML Conformance Test Suite, 20130923 release +// (https://www.w3.org/XML/Test/ — xmlts20130923.tar.gz, SHA-256 9b61db9f5dbffa545f4b8d78422167083a8568c59bd1129f94138f936cf6fc1f). +// The suite was contributed by James Clark, Sun Microsystems, OASIS/NIST, IBM, +// Fuji Xerox and Richard Tobin / the University of Edinburgh; test documents are +// (c) their contributors ("Copyright 1998-1999 by Sun Microsystems, Inc.", +// "Modifications copyright 1999-2001 by OASIS", IBM 2000-2003, ...) and are +// inlined here verbatim as conformance inputs. +// +// Scope: 1995 cases applicable to an XML 1.0 (Fifth Edition) processor — +// 927 must-reject + 752 must-accept (262 with canonical output) + 316 whose +// verdict depends on processor class (pinned to Bun's behaviour, see below). +// Skipped: 1 XML 1.1 / Namespaces 1.1 cases, 310 cases restricted to editions 1-4 +// (superseded by Fifth Edition erratum E09), 6 large japanese/pr-xml-* samples +// (advisory only; the japanese/weekly-* cases cover the same encodings). +// Regenerate with: bun bd test/js/bun/xml/generate_xml_test_suite.ts [path-to-xmlconf] +// +// Bun.XML is a non-validating processor that does not read external entities +// (XML 1.0 §5.1), so: +// - not-wf cases with ENTITIES="none" must throw SyntaxError (exact message +// asserted); +// - valid and invalid cases with ENTITIES="none" must parse (non-validating +// processors accept invalid-but-well-formed documents); when the suite gives +// a canonical output, the { compact: false } tree must serialize to it +// exactly (Second Canonical Form, minus the DOCTYPE/notation block and +// processing instructions, which Bun.XML does not represent), the compact +// object must equal the projection of that output, and XML.stringify of +// either shape must parse back to the same value; +// - cases with external entities, TYPE="error" cases, and the Namespaces 1.0 +// collection (names are opaque strings to Bun.XML; namespace constraints are +// not enforced) are pinned to the in-tree behaviour, with the upstream +// verdict in the comment. +// Inputs whose bytes carry encoding information the processor must act on (an +// encoding declaration, a UTF-16 BOM, non-UTF-8 bytes) are passed as bytes; a JS +// string is already-decoded text and its encoding declaration is not acted upon. +import { XML } from "bun"; +import { describe, expect, test } from "bun:test"; + +type XMLNode = { name: string; attributes: Record; children: (XMLNode | string)[] }; + +const CANON_ESCAPES: Record = { + "&": "&", + "<": "<", + ">": ">", + '"': """, + "\t": " ", + "\n": " ", + "\r": " ", +}; +function codePointCompare(a: string, b: string): number { + const x = [...a]; + const y = [...b]; + for (let i = 0; i < x.length && i < y.length; i++) { + const d = x[i].codePointAt(0)! - y[i].codePointAt(0)!; + if (d !== 0) return d; + } + return x.length - y.length; +} +/** James Clark's Canonical XML for an element tree. */ +function canonicalize(node: XMLNode | string): string { + if (typeof node === "string") return node.replace(/[&<>"\t\n\r]/g, c => CANON_ESCAPES[c]); + const attrs = Object.keys(node.attributes) + .sort(codePointCompare) + .map(k => ` ${k}="${node.attributes[k].replace(/[&<>"\t\n\r]/g, c => CANON_ESCAPES[c])}"`) + .join(""); + return `<${node.name}${attrs}>${node.children.map(canonicalize).join("")}`; +} + +function tree(input: string | Buffer): XMLNode { + return XML.parse(input, { compact: false }); +} + +function expectParses(input: string | Buffer, canonical?: string, compact?: unknown): void { + const node = tree(input); + const object = XML.parse(input); + if (canonical !== undefined) { + expect(canonicalize(node)).toBe(canonical); + expect(object).toEqual(compact); + } + // stringify of either shape reads back as the same value. + expect(canonicalize(tree(XML.stringify(node)))).toBe(canonicalize(node)); + expect(XML.parse(XML.stringify(object))).toEqual(object); +} + +function expectRejects(input: string | Buffer, message?: string): void { + let err: unknown; + try { + XML.parse(input); + } catch (e) { + err = e; + } + expect(err).toBeInstanceOf(SyntaxError); + if (message !== undefined) expect((err as SyntaxError).message).toBe(message); + expect(() => tree(input)).toThrow(SyntaxError); +} + +describe("xmltest", () => { + test("not-wf-sa-001", () => { + // 3.1 [41] — Attribute values must start with attribute names, not "?". + const input: string = "\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected an attribute name, '>' or '/>' in the start tag but found '?'"); + }); + + test("not-wf-sa-002", () => { + // 2.3 [4] — Names may not start with "."; it's not a Letter. + const input: string = "\r\n<.doc>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected an element name after '<' but found '.'"); + }); + + test("not-wf-sa-003", () => { + // 2.6 [16] — Processing Instruction target name is required. + const input: string = "\r\n"; + expectRejects(input, "XML Parse error: Expected a processing instruction target after ' { + // 2.6 [16] — SGML-ism: processing instructions end in '?>' not '>'. + const input: string = "\r\n"; + expectRejects(input, "XML Parse error: Unterminated processing instruction"); + }); + + test("not-wf-sa-005", () => { + // 2.6 [16] — Processing instructions end in '?>' not '?'. + const input: string = "\r\n"; + expectRejects(input, "XML Parse error: Unterminated processing instruction"); + }); + + test("not-wf-sa-006", () => { + // 2.5 [16] — XML comments may not contain "--" + const input: string = "\r\n"; + expectRejects(input, "XML Parse error: '--' is not allowed inside a comment"); + }); + + test("not-wf-sa-007", () => { + // 4.1 [68] — General entity references have no whitespace after the entity name and before the + // semicolon. + const input: string = "& no refc\r\n"; + expectRejects(input, "XML Parse error: Expected ';' after the entity name but found space"); + }); + + test("not-wf-sa-008", () => { + // 2.3 [5] — Entity references must include names, which don't begin with '.' (it's not a Letter or + // other name start character). + const input: string = "&.entity;\r\n"; + expectRejects(input, "XML Parse error: Expected an entity name after '&' but found '.'"); + }); + + test("not-wf-sa-009", () => { + // 4.1 [66] — Character references may have only decimal or numeric strings. + const input: string = "&#RE;\r\n"; + expectRejects( + input, + "XML Parse error: Invalid character reference: expected a number followed by ';' but found 'R'", + ); + }); + + test("not-wf-sa-010", () => { + // 4.1 [68] — Ampersand may only appear as part of a general entity reference. + const input: string = "A & B\r\n"; + expectRejects(input, "XML Parse error: Expected an entity name after '&' but found space"); + }); + + test("not-wf-sa-011", () => { + // 3.1 [41] — SGML-ism: attribute values must be explicitly assigned a value, it can't act as a boolean + // toggle. + const input: string = "\r\n"; + expectRejects(input, "XML Parse error: Expected '=' after the attribute name but found '>'"); + }); + + test("not-wf-sa-012", () => { + // 2.3 [10] — SGML-ism: attribute values must be quoted in all cases. + const input: string = "\r\n"; + expectRejects(input, "XML Parse error: Expected a quoted attribute value but found 'v1'"); + }); + + test("not-wf-sa-013", () => { + // 2.3 [10] — The quotes on both ends of an attribute value must match. + const input: string = "\r\n"; + expectRejects(input, "XML Parse error: '<' is not allowed in attribute values"); + }); + + test("not-wf-sa-014", () => { + // 2.3 [10] — Attribute values may not contain literal '<' characters. + const input: string = '\r\n'; + expectRejects(input, "XML Parse error: '<' is not allowed in attribute values"); + }); + + test("not-wf-sa-015", () => { + // 3.1 [41] — Attribute values need a value, not just an equals sign. + const input: string = "\r\n"; + expectRejects(input, "XML Parse error: Expected a quoted attribute value but found '>'"); + }); + + test("not-wf-sa-016", () => { + // 3.1 [41] — Attribute values need an associated name. + const input: string = '\r\n'; + expectRejects(input, "XML Parse error: Expected an attribute name, '>' or '/>' in the start tag but found '\"'"); + }); + + test("not-wf-sa-017", () => { + // 2.7 [18] — CDATA sections need a terminating ']]>'. + const input: string = "\r\n"; + expectRejects(input, "XML Parse error: Unterminated CDATA section"); + }); + + test("not-wf-sa-018", () => { + // 2.7 [19] — CDATA sections begin with a literal '\r\n"; + expectRejects(input, "XML Parse error: Expected a comment or CDATA section after ' { + // 3.1 [42] — End tags may not be abbreviated as ''. + const input: string = "\r\n"; + expectRejects(input, "XML Parse error: Expected an element name after ''"); + }); + + test("not-wf-sa-020", () => { + // 2.3 [10] — Attribute values may not contain literal '&' characters except as part of an entity + // reference. + const input: string = '\r\n'; + expectRejects(input, "XML Parse error: Expected an entity name after '&' but found space"); + }); + + test("not-wf-sa-021", () => { + // 2.3 [10] — Attribute values may not contain literal '&' characters except as part of an entity + // reference. + const input: string = '\r\n'; + expectRejects(input, "XML Parse error: Expected ';' after the entity name but found '\"'"); + }); + + test("not-wf-sa-022", () => { + // 4.1 [66] — Character references end with semicolons, always! + const input: string = '\r\n'; + expectRejects( + input, + "XML Parse error: Invalid character reference: expected a number followed by ';' but found ':'", + ); + }); + + test("not-wf-sa-023", () => { + // 2.3 [5] — Digits are not valid name start characters. + const input: string = '\r\n'; + expectRejects(input, "XML Parse error: Expected an attribute name, '>' or '/>' in the start tag but found '12'"); + }); + + test("not-wf-sa-024", () => { + // 2.3 [5] — Digits are not valid name start characters. + const input: string = "\r\n<123>\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected an element name after '<' but found '1'"); + }); + + test("not-wf-sa-025", () => { + // 2.4 [14] — Text may not contain a literal ']]>' sequence. + const input: string = "]]>\r\n"; + expectRejects(input, "XML Parse error: ']]>' is only allowed as the end of a CDATA section"); + }); + + test("not-wf-sa-026", () => { + // 2.4 [14] — Text may not contain a literal ']]>' sequence. + const input: string = "]]]>\r\n"; + expectRejects(input, "XML Parse error: ']]>' is only allowed as the end of a CDATA section"); + }); + + test("not-wf-sa-027", () => { + // 2.5 [15] — Comments must be terminated with "-->". + const input: string = "\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x0C"); + }); + + test("not-wf-sa-033", () => { + // 2.2 [2] — An ESC (octal 033) is not a legal XML character. + const input: string = "abc\u001bdef\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x1B"); + }); + + test("not-wf-sa-034", () => { + // 2.2 [2] — A form feed is not a legal XML character. + const input: string = "A form-feed is not white space or a name character\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x0C"); + }); + + test("not-wf-sa-035", () => { + // 3.1 [43] — The '<' character is a markup delimiter and must start an element, CDATA section, PI, or + // comment. + const input: string = "1 < 2 but not in XML\r\n"; + expectRejects(input, "XML Parse error: Expected an element name after '<' but found space"); + }); + + test("not-wf-sa-036", () => { + // 2.8 [27] — Text may not appear after the root element. + const input: string = "\r\nIllegal data\r\n"; + expectRejects(input, "XML Parse error: Unexpected 'Illegal' after the root element"); + }); + + test("not-wf-sa-037", () => { + // 2.8 [27] — Character references may not appear after the root element. + const input: string = "\r\n \r\n"; + expectRejects(input, "XML Parse error: Unexpected '&' after the root element"); + }); + + test("not-wf-sa-038", () => { + // 3.1 — Tests the "Unique Att Spec" WF constraint by providing multiple values for an attribute. + const input: string = '\r\n'; + expectRejects(input, "XML Parse error: Duplicate attribute 'x'"); + }); + + test("not-wf-sa-039", () => { + // 3 — Tests the Element Type Match WFC - end tag name must match start tag name. + const input: string = "\r\n"; + expectRejects(input, "XML Parse error: Expected closing tag but found "); + }); + + test("not-wf-sa-040", () => { + // 2.8 [27] — Provides two document elements. + const input: string = "\r\n\r\n"; + expectRejects(input, "XML Parse error: Only one root element is allowed"); + }); + + test("not-wf-sa-041", () => { + // 2.8 [27] — Provides two document elements. + const input: string = "\r\n\r\n"; + expectRejects(input, "XML Parse error: Only one root element is allowed"); + }); + + test("not-wf-sa-042", () => { + // 3.1 [42] — Invalid End Tag + const input: string = "\r\n"; + expectRejects(input, "XML Parse error: Unexpected ' { + // 2.8 [27] — Provides #PCDATA text after the document element. + const input: string = "\r\nIllegal data\r\n"; + expectRejects(input, "XML Parse error: Unexpected 'Illegal' after the root element"); + }); + + test("not-wf-sa-044", () => { + // 2.8 [27] — Provides two document elements. + const input: string = "\r\n"; + expectRejects(input, "XML Parse error: Only one root element is allowed"); + }); + + test("not-wf-sa-045", () => { + // 3.1 [44] — Invalid Empty Element Tag + const input: string = "\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected '>' after '/' but found newline"); + }); + + test("not-wf-sa-046", () => { + // 3.1 [40] — This start (or empty element) tag was not terminated correctly. + const input: string = "\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected '>' after '/' but found '<'"); + }); + + test("not-wf-sa-047", () => { + // 3.1 [44] — Invalid empty element tag invalid whitespace + const input: string = "\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected '>' after '/' but found space"); + }); + + test("not-wf-sa-048", () => { + // 2.8 [27] — Provides a CDATA section after the root element. + const input: string = "\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: CDATA sections are only allowed inside elements"); + }); + + test("not-wf-sa-049", () => { + // 3.1 [40] — Missing start tag + const input: string = "\r\n\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected closing tag but found "); + }); + + test("not-wf-sa-050", () => { + // 2.1 [1] — Empty document, with no root element. + const input: string = ""; + expectRejects(input, "XML Parse error: XML document must have a root element"); + }); + + test("not-wf-sa-051", () => { + // 2.7 [18] — CDATA is invalid at top level of document. + const input: string = "\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: CDATA sections are only allowed inside elements"); + }); + + test("not-wf-sa-052", () => { + // 4.1 [66] — Invalid character reference. + const input: string = "\r\n \r\n\r\n"; + expectRejects(input, "XML Parse error: Expected the root element but found '&'"); + }); + + test("not-wf-sa-053", () => { + // 3.1 [42] — End tag does not match start tag. + const input: string = "\r\n"; + expectRejects(input, "XML Parse error: Expected closing tag but found "); + }); + + test("not-wf-sa-054", () => { + // 4.2.2 [75] — PUBLIC requires two literals. + const input: string = '\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Expected a quoted system identifier after the public identifier but found '>'", + ); + }); + + test("not-wf-sa-055", () => { + // 2.8 [28] — Invalid Document Type Definition format. + const input: string = "\r\n"; + expectRejects( + input, + "XML Parse error: Expected a markup declaration or ']' in the internal subset but found ' { + // 2.8 [28] — Invalid Document Type Definition format - misplaced comment. + const input: string = "\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '--'", + ); + }); + + test("not-wf-sa-057", () => { + // 3.2 [45] — This isn't SGML; comments can't exist in declarations. + const input: string = '\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected '>' to end the entity declaration but found '--'"); + }); + + test("not-wf-sa-058", () => { + // 3.3.1 [54] — Invalid character , in ATTLIST enumeration + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected '|' or ')' in the enumeration but found ','"); + }); + + test("not-wf-sa-059", () => { + // 3.3.1 [59] — String literal must be in quotes. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected #REQUIRED, #IMPLIED, #FIXED or a quoted default value but found 'v1'", + ); + }); + + test("not-wf-sa-060", () => { + // 3.3.1 [56] — Invalid type NAME defined in ATTLIST. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'NAME'", + ); + }); + + test("not-wf-sa-061", () => { + // 4.2.2 [75] — External entity declarations require whitespace between public and system IDs. + const input: string = '\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before a quoted string"); + }); + + test("not-wf-sa-062", () => { + // 4.2 [71] — Entity declarations need space after the entity name. + const input: string = '\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before a quoted string"); + }); + + test("not-wf-sa-063", () => { + // 2.8 [29] — Conditional sections may only appear in the external DTD subset. + const input: string = "\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Conditional sections are only allowed in the external DTD subset"); + }); + + test("not-wf-sa-064", () => { + // 3.3 [53] — Space is required between attribute type and default values in + // declarations. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before a quoted string"); + }); + + test("not-wf-sa-065", () => { + // 3.3 [53] — Space is required between attribute name and type in declarations. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Whitespace is required before '('"); + }); + + test("not-wf-sa-066", () => { + // 3.3 [52] — Required whitespace is missing. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Whitespace is required before '#IMPLIED'"); + }); + + test("not-wf-sa-067", () => { + // 3.3 [53] — Space is required between attribute type and default values in + // declarations. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before a quoted string"); + }); + + test("not-wf-sa-068", () => { + // 3.3.1 [58] — Space is required between NOTATION keyword and list of enumerated choices in + // declarations. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Whitespace is required before '('"); + }); + + test("not-wf-sa-069", () => { + // 4.2.2 [76] — Space is required before an NDATA entity annotation. + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before 'NDATA'"); + }); + + test("not-wf-sa-070", () => { + // 2.5 [16] — XML comments may not contain "--" + const input: string = "\r\n\r\n"; + expectRejects(input, "XML Parse error: '--' is not allowed inside a comment"); + }); + + test("not-wf-sa-071", () => { + // 4.1 [68] — ENTITY can't reference itself directly or indirectly. + const input: string = + '\r\n\r\n\r\n]>\r\n&e1;\r\n'; + expectRejects(input, "XML Parse error: Entity 'e1' refers to itself"); + }); + + test("not-wf-sa-072", () => { + // 4.1 [68] — Undefined ENTITY foo. + const input: string = "&foo;\r\n"; + expectRejects(input, "XML Parse error: Entity 'foo' is not declared"); + }); + + test("not-wf-sa-073", () => { + // 4.1 [68] — Undefined ENTITY f. + const input: string = '\r\n]>\r\n&f;\r\n'; + expectRejects(input, "XML Parse error: Entity 'f' is not declared"); + }); + + test("not-wf-sa-074", () => { + // 4.3.2 — Internal general parsed entities are only well formed if they match the "content" + // production. + const input: string = '">\r\n]>\r\n\r\n&e;\r\n\r\n'; + expectRejects(input, "XML Parse error: Element 'foo' must start and end within the same entity"); + }); + + test("not-wf-sa-075", () => { + // 4.1 [68] — ENTITY can't reference itself directly or indirectly. + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Entity 'e1' refers to itself"); + }); + + test("not-wf-sa-076", () => { + // 4.1 [68] — Undefined ENTITY foo. + const input: string = '\r\n'; + expectRejects(input, "XML Parse error: Entity 'foo' is not declared"); + }); + + test("not-wf-sa-077", () => { + // 41. [68] — Undefined ENTITY bar. + const input: string = '\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Entity 'bar' is not declared"); + }); + + test("not-wf-sa-078", () => { + // 4.1 [68] — Undefined ENTITY foo. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Entity 'foo' is not declared"); + }); + + test("not-wf-sa-079", () => { + // 4.1 [68] — ENTITY can't reference itself directly or indirectly. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Entity 'e1' refers to itself"); + }); + + test("not-wf-sa-080", () => { + // 4.1 [68] — ENTITY can't reference itself directly or indirectly. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Entity 'e1' refers to itself"); + }); + + test("not-wf-sa-081", () => { + // 3.1 — This tests the No External Entity References WFC, since the entity is referred to within an + // attribute. (upstream: not-wf; external general entities are not read) + const input: string = '\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Attribute values cannot reference external entity 'e'"); + }); + + test("not-wf-sa-082", () => { + // 3.1 — This tests the No External Entity References WFC, since the entity is referred to within an + // attribute. (upstream: not-wf; external general entities are not read) + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Attribute values cannot reference external entity 'e'"); + }); + + test("not-wf-sa-083", () => { + // 4.2.2 [76] — Undefined NOTATION n. + const input: string = '\r\n]>\r\n&e;\r\n'; + expectRejects(input, "XML Parse error: Unparsed entity 'e' cannot be referenced"); + }); + + test("not-wf-sa-084", () => { + // 4.1 — Tests the Parsed Entity WFC by referring to an unparsed entity. (This precedes the error of + // not declaring that entity's notation, which may be detected any time before the DTD parsing is + // completed.) + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Unparsed entity 'e' cannot be referenced"); + }); + + test("not-wf-sa-085", () => { + // 2.3 [13] — Public IDs may not contain "[". + const input: string = '\r\n\r\n'; + expectRejects(input, "XML Parse error: Invalid character in a public identifier: '['"); + }); + + test("not-wf-sa-086", () => { + // 2.3 [13] — Public IDs may not contain "[". + const input: string = '\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Invalid character in a public identifier: '['"); + }); + + test("not-wf-sa-087", () => { + // 2.3 [13] — Public IDs may not contain "[". + const input: string = '\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Invalid character in a public identifier: '['"); + }); + + test("not-wf-sa-088", () => { + // 2.3 [10] — Attribute values are terminated by literal quote characters, and any entity expansion is + // done afterwards. + const input: string = + "\r\n\r\n\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: '<' is not allowed in attribute values"); + }); + + test("not-wf-sa-089", () => { + // 4.2 [74] — Parameter entities "are" always parsed; NDATA annotations are not permitted. + const input: string = '\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Parameter entities cannot have NDATA"); + }); + + test("not-wf-sa-090", () => { + // 2.3 [10] — Attributes may not contain a literal "<" character; this one has one because of reference + // expansion. + const input: string = "\">\r\n]>\r\n&e;\r\n"; + expectRejects(input, "XML Parse error: '<' is not allowed in attribute values"); + }); + + test("not-wf-sa-091", () => { + // 4.2 [74] — Parameter entities "are" always parsed; NDATA annotations are not permitted. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Parameter entities cannot have NDATA"); + }); + + test("not-wf-sa-092", () => { + // 4.5 — The replacement text of this entity has an illegal reference, because the character reference + // is expanded immediately. + const input: string = "\">\r\n]>\r\n&e;\r\n"; + expectRejects(input, "XML Parse error: Expected an entity name after '&' but found '''"); + }); + + test("not-wf-sa-093", () => { + // 4.1 [66] — Hexadecimal character references may not use the uppercase 'X'. + const input: string = "X\r\n"; + expectRejects( + input, + "XML Parse error: Invalid character reference: expected a number followed by ';' but found 'X'", + ); + }); + + test("not-wf-sa-094", () => { + // 2.8 [24] — Prolog VERSION must be lowercase. + const input: string = '\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected version=\"1.0\" in the XML declaration but found 'VERSION'"); + }); + + test("not-wf-sa-095", () => { + // 2.8 [23] — VersionInfo must come before EncodingDecl. + const input = Buffer.from('\r\n\r\n'); + expectRejects(input, 'XML Parse error: The XML declaration must start with version="1.0"'); + }); + + test("not-wf-sa-096", () => { + // 2.9 [32] — Space is required before the standalone declaration. + const input = Buffer.from('\r\n'); + expectRejects(input, "XML Parse error: Whitespace is required before 'encoding'"); + }); + + test("not-wf-sa-097", () => { + // 2.8 [24] — Both quotes surrounding VersionNum must be the same. + const input = Buffer.from('\r\n'); + expectRejects(input, "XML Parse error: Unsupported XML version '1.0' encoding=' (this is an XML 1.0 parser)"); + }); + + test("not-wf-sa-098", () => { + // 2.8 [23] — Only one "version=..." string may appear in an XML declaration. + const input: string = '\r\n'; + expectRejects( + input, + "XML Parse error: Misplaced 'version' in the XML declaration (the order is version, encoding, standalone)", + ); + }); + + test("not-wf-sa-099", () => { + // 2.8 [23] — Only three pseudo-attributes are in the XML declaration, and "valid=..." is not one of + // them. + const input: string = '\r\n'; + expectRejects( + input, + "XML Parse error: Unexpected 'valid' in the XML declaration (expected version, encoding or standalone)", + ); + }); + + test("not-wf-sa-100", () => { + // 2.9 [32] — Only "yes" and "no" are permitted as values of "standalone". + const input: string = '\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Invalid value 'YES' for standalone in the XML declaration (expected yes or no)", + ); + }); + + test("not-wf-sa-101", () => { + // 4.3.3 [81] — Space is not permitted in an encoding name. + const input = Buffer.from('\r\n\r\n'); + expectRejects(input, "XML Parse error: Invalid encoding name ' UTF-8' in the XML declaration"); + }); + + test("not-wf-sa-102", () => { + // 2.8 [26] — Provides an illegal XML version number; spaces are illegal. + const input: string = '\r\n\r\n'; + expectRejects(input, "XML Parse error: Unsupported XML version '1.0 ' (this is an XML 1.0 parser)"); + }); + + test("not-wf-sa-103", () => { + // 4.3.2 — End-tag required for element foo. + const input: string = '">\r\n]>\r\n&e;\r\n'; + expectRejects(input, "XML Parse error: Expected closing tag but found "); + }); + + test("not-wf-sa-104", () => { + // 4.3.2 — Internal general parsed entities are only well formed if they match the "content" + // production. + const input: string = '">\r\n]>\r\n&e;\r\n'; + expectRejects(input, "XML Parse error: Element 'foo' must start and end within the same entity"); + }); + + test("not-wf-sa-105", () => { + // 2.7 — Invalid placement of CDATA section. + const input: string = "\r\n\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: CDATA sections are only allowed inside elements"); + }); + + test("not-wf-sa-106", () => { + // 4.2 — Invalid placement of entity declaration. + const input: string = "\r\n \r\n"; + expectRejects(input, "XML Parse error: Expected the root element but found '&'"); + }); + + test("not-wf-sa-107", () => { + // 2.8 [28] — Invalid document type declaration. CDATA alone is invalid. + const input: string = "\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: CDATA sections are only allowed inside elements"); + }); + + test("not-wf-sa-108", () => { + // 2.7 [19] — No space in '\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected a comment or CDATA section after ' { + // 4.2 [70] — Tags invalid within EntityDecl. + const input: string = '">\r\n]>\r\n&e;\r\n'; + expectRejects(input, "XML Parse error: Expected the root element but found '&'"); + }); + + test("not-wf-sa-110", () => { + // 4.1 [68] — Entity reference must be in content of element. + const input: string = '\r\n]>\r\n\r\n&e;\r\n'; + expectRejects(input, "XML Parse error: Unexpected '&' after the root element"); + }); + + test("not-wf-sa-111", () => { + // 3.1 [43] — Entiry reference must be in content of element not Start-tag. + const input: string = "\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected an attribute name, '>' or '/>' in the start tag but found '&'"); + }); + + test("not-wf-sa-112", () => { + // 2.7 [19] — CDATA sections start '\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected a comment or CDATA section after ' { + // 2.3 [9] — Parameter entity values must use valid reference syntax; this reference is malformed. + const input: string = '\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected an entity name after '&' but found '\"'"); + }); + + test("not-wf-sa-114", () => { + // 2.3 [9] — General entity values must use valid reference syntax; this reference is malformed. + const input: string = '\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected an entity name after '&' but found '\"'"); + }); + + test("not-wf-sa-115", () => { + // 4.5 — The replacement text of this entity is an illegal character reference, which must be rejected + // when it is parsed in the context of an attribute value. + const input: string = '\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected an entity name after '&' but found the end of entity 'e'"); + }); + + test("not-wf-sa-116", () => { + // 4.3.2 — Internal general parsed entities are only well formed if they match the "content" + // production. This is a partial character reference, not a full one. + const input: string = '\r\n]>\r\n&e;7;\r\n'; + expectRejects( + input, + "XML Parse error: Invalid character reference: expected a number followed by ';' but found the end of entity 'e'", + ); + }); + + test("not-wf-sa-117", () => { + // 4.3.2 — Internal general parsed entities are only well formed if they match the "content" + // production. This is a partial character reference, not a full one. + const input: string = '\r\n]>\r\n&e;#97;\r\n'; + expectRejects(input, "XML Parse error: Expected an entity name after '&' but found the end of entity 'e'"); + }); + + test("not-wf-sa-118", () => { + // 4.1 [68] — Entity reference expansion is not recursive. + const input: string = '\r\n]>\r\n&&e;97;\r\n'; + expectRejects(input, "XML Parse error: Expected an entity name after '&' but found '&'"); + }); + + test("not-wf-sa-119", () => { + // 4.3.2 — Internal general parsed entities are only well formed if they match the "content" + // production. This is a partial character reference, not a full one. + const input: string = '\r\n]>\r\n\r\n&e;#38;\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected an entity name after '&' but found the end of entity 'e'"); + }); + + test("not-wf-sa-120", () => { + // 4.5 — Character references are expanded in the replacement text of an internal entity, which is then + // parsed as usual. Accordingly, & must be doubly quoted - encoded either as & or as &#38;. + const input: string = '\r\n]>\r\n\r\n&e;\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected an entity name after '&' but found the end of entity 'e'"); + }); + + test("not-wf-sa-121", () => { + // 4.1 [68] — A name of an ENTITY was started with an invalid character. + const input: string = '\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected an entity name or '%' after ' { + // 3.2.1 [47] — Invalid syntax mixed connectors are used. + const input: string = "\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: A content model group cannot mix ',' and '|'"); + }); + + test("not-wf-sa-123", () => { + // 3.2.1 [48] — Invalid syntax mismatched parenthesis. + const input: string = "\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected '>' to end the element declaration but found ')'"); + }); + + test("not-wf-sa-124", () => { + // 3.2.2 [51] — Invalid format of Mixed-content declaration. + const input: string = "\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: #PCDATA must come first in a content model, as (#PCDATA|a|b)*"); + }); + + test("not-wf-sa-125", () => { + // 3.2.2 [51] — Invalid syntax extra set of parenthesis not necessary. + const input: string = "\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: #PCDATA must come first in a content model, as (#PCDATA|a|b)*"); + }); + + test("not-wf-sa-126", () => { + // 3.2.2 [51] — Invalid syntax Mixed-content must be defined as zero or more. + const input: string = "\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: A mixed content model may only be followed by '*'"); + }); + + test("not-wf-sa-127", () => { + // 3.2.2 [51] — Invalid syntax Mixed-content must be defined as zero or more. + const input: string = "\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: A mixed content model may only be followed by '*'"); + }); + + test("not-wf-sa-128", () => { + // 2.7 [18] — Invalid CDATA syntax. + const input: string = "\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected EMPTY, ANY or '(' in the element declaration but found 'CDATA'"); + }); + + test("not-wf-sa-129", () => { + // 3.2 [45] — Invalid syntax for Element Type Declaration. + const input: string = "\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected EMPTY, ANY or '(' in the element declaration but found '-'"); + }); + + test("not-wf-sa-130", () => { + // 3.2 [45] — Invalid syntax for Element Type Declaration. + const input: string = "\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: An occurrence indicator must directly follow the name or ')' it applies to"); + }); + + test("not-wf-sa-131", () => { + // 3.2 [45] — Invalid syntax for Element Type Declaration. + const input: string = "\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected '>' to end the element declaration but found '-'"); + }); + + test("not-wf-sa-132", () => { + // 3.2.1 [50] — Invalid syntax mixed connectors used. + const input: string = "\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: A content model group cannot mix ',' and '|'"); + }); + + test("not-wf-sa-133", () => { + // 3.2.1 — Illegal whitespace before optional character causes syntax error. + const input: string = "\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: An occurrence indicator must directly follow the name or ')' it applies to"); + }); + + test("not-wf-sa-134", () => { + // 3.2.1 — Illegal whitespace before optional character causes syntax error. + const input: string = "\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: An occurrence indicator must directly follow the name or ')' it applies to"); + }); + + test("not-wf-sa-135", () => { + // 3.2.1 [47] — Invalid character used as connector. + const input: string = "\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected ')', '|' or ',' in the content model but found '&'"); + }); + + test("not-wf-sa-136", () => { + // 3.2 [45] — Tag omission is invalid in XML. + const input: string = "\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected EMPTY, ANY or '(' in the element declaration but found 'O'"); + }); + + test("not-wf-sa-137", () => { + // 3.2 [45] — Space is required before a content model. + const input: string = "\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Whitespace is required before '('"); + }); + + test("not-wf-sa-138", () => { + // 3.2.1 [48] — Invalid syntax for content particle. + const input: string = "\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected ')', '|' or ',' in the content model but found '?'"); + }); + + test("not-wf-sa-139", () => { + // 3.2.1 [46] — The element-content model should not be empty. + const input: string = "\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected an element name or '(' in the content model but found ')'"); + }); + + test("not-wf-sa-142", () => { + // 2.2 [2] — Character #x0000 is not legal anywhere in an XML document. + const input: string = "\r\n]>\r\n�\r\n"; + expectRejects(input, "XML Parse error: Character reference '�' is not a valid XML character"); + }); + + test("not-wf-sa-143", () => { + // 2.2 [2] — Character #x001F is not legal anywhere in an XML document. + const input: string = "\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Character reference '' is not a valid XML character"); + }); + + test("not-wf-sa-144", () => { + // 2.2 [2] — Character #xFFFF is not legal anywhere in an XML document. + const input: string = "\r\n]>\r\n￿\r\n"; + expectRejects(input, "XML Parse error: Character reference '￿' is not a valid XML character"); + }); + + test("not-wf-sa-145", () => { + // 2.2 [2] — Character #xD800 is not legal anywhere in an XML document. (If it appeared in a UTF-16 + // surrogate pair, it'd represent half of a UCS-4 character and so wouldn't really be in the document.) + const input: string = "\r\n]>\r\n�\r\n"; + expectRejects(input, "XML Parse error: Character reference '�' is not a valid XML character"); + }); + + test("not-wf-sa-146", () => { + // 2.2 [2] — Character references must also refer to legal XML characters; #x00110000 is one more than + // the largest legal character. + const input: string = "\r\n]>\r\n�\r\n"; + expectRejects(input, "XML Parse error: Character reference '�' is not a valid XML character"); + }); + + test("not-wf-sa-147", () => { + // 2.8 [22] — XML Declaration may not be preceded by whitespace. + const input: string = '\r\n\r\n\r\n'; + expectRejects( + input, + "XML Parse error: ' { + // 2.8 [22] — XML Declaration may not be preceded by comments or whitespace. + const input: string = '\r\n\r\n\r\n'; + expectRejects( + input, + "XML Parse error: ' { + // 2.8 [28] — XML Declaration may not be within a DTD. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: ' { + // 3.1 [43] — XML declarations may not be within element content. + const input: string = '\r\n\r\n\r\n'; + expectRejects( + input, + "XML Parse error: ' { + // 2.8 [27] — XML declarations may not follow document content. + const input: string = '\r\n\r\n\r\n'; + expectRejects( + input, + "XML Parse error: ' { + // 2.8 [22] — XML declarations must include the "version=..." string. + const input = Buffer.from('\r\n\r\n'); + expectRejects(input, 'XML Parse error: The XML declaration must start with version="1.0"'); + }); + + test("not-wf-sa-153", () => { + // 4.3.2 — Text declarations may not begin internal parsed entities; they may only appear at the + // beginning of external parsed (parameter or general) entities. + const input: string = + "\r\n\">\r\n]>\r\n&e;\r\n"; + expectRejects( + input, + "XML Parse error: ' { + // 2.8 2.6 [23, 17] — '' is neither an XML declaration nor a legal processing instruction + // target name. + const input: string = '\r\n\r\n'; + expectRejects( + input, + "XML Parse error: ' { + // 2.8 2.6 [23, 17] — '' is neither an XML declaration nor a legal processing instruction + // target name. + const input: string = '\r\n\r\n'; + expectRejects( + input, + "XML Parse error: ' { + // 2.8 2.6 [23, 17] — '' is neither an XML declaration nor a legal processing instruction + // target name. + const input: string = '\r\n\r\n\r\n'; + expectRejects( + input, + "XML Parse error: ' { + // 2.6 [17] — '' is not a legal processing instruction target name. + const input: string = "\r\n\r\n\r\n"; + expectRejects( + input, + "XML Parse error: ' { + // 3.3 [52] — SGML-ism: "#NOTATION gif" can't have attributes. + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected an element name after ' { + // 2.3 [9] — Uses '&' unquoted in an entity declaration, which is illegal syntax for an entity + // reference. + const input: string = + '\r\n">\r\n]>\r\n&e;\r\n'; + expectRejects(input, "XML Parse error: Expected an entity name after '&' but found space"); + }); + + test("not-wf-sa-160", () => { + // 2.8 — Violates the PEs in Internal Subset WFC by using a PE reference within a declaration. + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Parameter entity references are not allowed inside markup declarations in the internal subset", + ); + }); + + test("not-wf-sa-161", () => { + // 2.8 — Violates the PEs in Internal Subset WFC by using a PE reference within a declaration. + const input: string = '\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Parameter entity references are not allowed inside markup declarations in the internal subset", + ); + }); + + test("not-wf-sa-162", () => { + // 2.8 — Violates the PEs in Internal Subset WFC by using a PE reference within a declaration. + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Parameter entity references are not allowed inside markup declarations in the internal subset", + ); + }); + + test("not-wf-sa-163", () => { + // 4.1 [69] — Invalid placement of Parameter entity reference. + const input: string = + '\r\n\r\n]>\r\n%e;\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Markup declarations and parameter-entity references are only allowed in the document type declaration", + ); + }); + + test("not-wf-sa-164", () => { + // 4.1 [69] — Invalid placement of Parameter entity reference. + const input: string = + '\r\n\r\n] %e; >\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Parameter entity references are not allowed inside markup declarations in the internal subset", + ); + }); + + test("not-wf-sa-165", () => { + // 4.2 [72] — Parameter entity declarations must have a space before the '%'. + const input: string = '\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before '%'"); + }); + + test("not-wf-sa-166", () => { + // 2.2 [2] — Character FFFF is not legal anywhere in an XML document. + const input: string = "\uffff\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: '\uffff' (U+FFFF)"); + }); + + test("not-wf-sa-167", () => { + // 2.2 [2] — Character FFFE is not legal anywhere in an XML document. + const input: string = "\ufffe\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: '\ufffe' (U+FFFE)"); + }); + + test("not-wf-sa-168", () => { + // 2.2 [2] — An unpaired surrogate (D800) is not legal anywhere in an XML document. + const input = Buffer.from("PGRvYz7toIA8L2RvYz4NCg==", "base64"); + expectRejects(input, "XML Parse error: Invalid UTF-8"); + }); + + test("not-wf-sa-169", () => { + // 2.2 [2] — An unpaired surrogate (DC00) is not legal anywhere in an XML document. + const input = Buffer.from("PGRvYz7tsIA8L2RvYz4NCg==", "base64"); + expectRejects(input, "XML Parse error: Invalid UTF-8"); + }); + + test("not-wf-sa-170", () => { + // 2.2 [2] — Four byte UTF-8 encodings can encode UCS-4 characters which are beyond the range of legal + // XML characters (and can't be expressed in Unicode surrogate pairs). This document holds such a + // character. + const input = Buffer.from("PGRvYz73gICAPC9kb2M+DQo=", "base64"); + expectRejects(input, "XML Parse error: Invalid UTF-8"); + }); + + test("not-wf-sa-171", () => { + // 2.2 [2] — Character FFFF is not legal anywhere in an XML document. + const input: string = "\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: '\uffff' (U+FFFF)"); + }); + + test("not-wf-sa-172", () => { + // 2.2 [2] — Character FFFF is not legal anywhere in an XML document. + const input: string = "\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: '\uffff' (U+FFFF)"); + }); + + test("not-wf-sa-173", () => { + // 2.2 [2] — Character FFFF is not legal anywhere in an XML document. + const input: string = '\r\n'; + expectRejects(input, "XML Parse error: Invalid character in XML: '\uffff' (U+FFFF)"); + }); + + test("not-wf-sa-174", () => { + // 2.2 [2] — Character FFFF is not legal anywhere in an XML document. + const input: string = "\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: '\uffff' (U+FFFF)"); + }); + + test("not-wf-sa-175", () => { + // 2.2 [2] — Character FFFF is not legal anywhere in an XML document. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Invalid character in XML: '\uffff' (U+FFFF)"); + }); + + test("not-wf-sa-176", () => { + // 3 [39] — Start tags must have matching end tags. + const input: string = "\n]>\n\n"; + expectRejects(input, "XML Parse error: Missing closing tag for element 'doc'"); + }); + + test("not-wf-sa-177", () => { + // 2.2 [2] — Character FFFF is not legal anywhere in an XML document. + const input: string = "\r\n]>\r\nA\uffff\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: '\uffff' (U+FFFF)"); + }); + + test("not-wf-sa-178", () => { + // 3.1 [41] — Invalid syntax matching double quote is missing. + const input: string = + '\r\n\r\n]>\r\n { + // 4.1 [66] — Invalid syntax matching double quote is missing. + const input: string = '\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Unterminated entity value"); + }); + + test("not-wf-sa-180", () => { + // 4.1 — The Entity Declared WFC requires entities to be declared before they are used in an attribute + // list declaration. + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Entity 'e' is not declared"); + }); + + test("not-wf-sa-181", () => { + // 4.3.2 — Internal parsed entities must match the content production to be well formed. + const input: string = + '\r\n\r\n]>\r\n&e;]]>\r\n'; + expectRejects(input, "XML Parse error: Unterminated CDATA section"); + }); + + test("not-wf-sa-182", () => { + // 4.3.2 — Internal parsed entities must match the content production to be well formed. + const input: string = + '\r\n\r\n]>\r\n&e;-->\r\n'; + expectRejects(input, "XML Parse error: Unterminated comment"); + }); + + test("not-wf-sa-183", () => { + // 3.2.2 [51] — Mixed content declarations may not include content particles. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Names in a mixed content model cannot have occurrence indicators"); + }); + + test("not-wf-sa-184", () => { + // 3.2.2 [51] — In mixed content models, element names must not be parenthesized. + const input: string = + "\r\n\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Only element names may follow #PCDATA in a mixed content model"); + }); + + test("not-wf-sa-185", () => { + // 4.1 — Tests the Entity Declared WFC. Note: a nonvalidating parser is permitted not to report this + // WFC violation, since it would need to read an external parameter entity to distinguish it from a + // violation of the Standalone Declaration VC. (upstream: not-wf; external parameter entities are not + // read) + const input: string = + '\r\n\r\n&e;\r\n'; + expectRejects(input, "XML Parse error: Entity 'e' is not declared"); + }); + + test("not-wf-sa-186", () => { + // 3.1 [44] — Whitespace is required between attribute/value pairs. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before 'd'"); + }); + + test("not-wf-not-sa-001", () => { + // 3.4 [62] — Conditional sections must be properly terminated ("]>" used instead of "]]>"). (upstream: + // not-wf; external general and parameter entities are not read) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("not-wf-not-sa-002", () => { + // 2.6 [17] — Processing instruction target names may not be "XML" in any combination of cases. + // (upstream: not-wf; external general and parameter entities are not read) + const input: string = + "\r\n\">\r\n%e;\r\n]>\r\n"; + expectRejects( + input, + "XML Parse error: ' { + // 3.4 [62] — Conditional sections must be properly terminated ("]]>" omitted). (upstream: not-wf; + // external general and parameter entities are not read) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("not-wf-not-sa-004", () => { + // 3.4 [62] — Conditional sections must be properly terminated ("]]>" omitted). (upstream: not-wf; + // external general and parameter entities are not read) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("not-wf-not-sa-005", () => { + // 4.1 — Tests the Entity Declared VC by referring to an undefined parameter entity within an external + // entity. (upstream: optional error) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("not-wf-not-sa-006", () => { + // 3.4 [62] — Conditional sections need a '[' after the INCLUDE or IGNORE. (upstream: not-wf; external + // general and parameter entities are not read) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("not-wf-not-sa-007", () => { + // 4.3.2 [79] — A declaration may not begin any external entity; it's only found once, + // in the document entity. (upstream: not-wf; external general and parameter entities are not read) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("not-wf-not-sa-008", () => { + // 4.1 [69] — In DTDs, the '%' character must be part of a parameter entity reference. (upstream: + // not-wf; external general and parameter entities are not read) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("not-wf-not-sa-009", () => { + // 2.8 — This test violates WFC:PE Between Declarations in Production 28a. The last character of a + // markup declaration is not contained in the same parameter-entity text replacement. (upstream: + // not-wf; external general and parameter entities are not read) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("not-wf-ext-sa-001", () => { + // 4.1 — Tests the No Recursion WFC by having an external general entity be self-recursive. (upstream: + // not-wf; external general and parameter entities are not read) + const input: string = '\r\n]>\r\n&e;\r\n'; + expectParses(input); + }); + + test("not-wf-ext-sa-002", () => { + // 4.3.1 4.3.2 [77, 78] — External entities have "text declarations", which do not permit the + // "standalone=..." attribute that's allowed in XML declarations. (upstream: not-wf; external general + // and parameter entities are not read) + const input: string = + '\r\n\r\n]>\r\n&e;\r\n'; + expectParses(input); + }); + + test("not-wf-ext-sa-003", () => { + // 2.6 [17] — Only one text declaration is permitted; a second one looks like an illegal processing + // instruction (target names of "xml" in any case are not allowed). (upstream: not-wf; external general + // and parameter entities are not read) + const input: string = + '\r\n\r\n]>\r\n&e;\r\n'; + expectParses(input); + }); + + test("invalid--002", () => { + // 3.2.1 — Tests the "Proper Group/PE Nesting" validity constraint by fragmenting a content model + // between two parameter entities. (upstream: invalid; external general and parameter entities are not + // read) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("invalid--005", () => { + // 2.8 — Tests the "Proper Declaration/PE Nesting" validity constraint by fragmenting an element + // declaration between two parameter entities. (upstream: invalid; external general and parameter + // entities are not read) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("invalid--006", () => { + // 2.8 — Tests the "Proper Declaration/PE Nesting" validity constraint by fragmenting an element + // declaration between two parameter entities. (upstream: invalid; external general and parameter + // entities are not read) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("invalid-not-sa-022", () => { + // 3.4 [62] — Test the "Proper Conditional Section/ PE Nesting" validity constraint. (upstream: + // invalid; external general and parameter entities are not read; output depends on them) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("valid-sa-001", () => { + // 3.2.2 [51] — Test demonstrates an Element Type Declaration with Mixed Content. + const input: string = "\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-002", () => { + // 3.1 [40] — Test demonstrates that whitespace is permitted after the tag name in a Start-tag. + const input: string = "\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-003", () => { + // 3.1 [42] — Test demonstrates that whitespace is permitted after the tag name in an End-tag. + const input: string = "\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-004", () => { + // 3.1 [41] — Test demonstrates a valid attribute specification within a Start-tag. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a1": "v1" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-005", () => { + // 3.1 [40] — Test demonstrates a valid attribute specification within a Start-tag that contains + // whitespace on both sides of the equal sign. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a1": "v1" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-006", () => { + // 3.1 [41] — Test demonstrates that the AttValue within a Start-tag can use a single quote as a + // delimter. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + const canonical = ''; + const compact: unknown = { doc: { "@a1": "v1" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-007", () => { + // 3.1 4.6 [43] — Test demonstrates numeric character references can be used for element content. + const input: string = "\r\n]>\r\n \r\n"; + const canonical = " "; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-008", () => { + // 2.4 3.1 [43] — Test demonstrates character references can be used for element content. + const input: string = + "\r\n]>\r\n&<>"'\r\n"; + const canonical = "&<>"'"; + const compact: unknown = { doc: "&<>\"'" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-009", () => { + // 2.3 3.1 [43] — Test demonstrates that PubidChar can be used for element content. + const input: string = "\r\n]>\r\n \r\n"; + const canonical = " "; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-010", () => { + // 3.1 [40] — Test demonstrates that whitespace is valid after the Attribute in a Start-tag. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a1": "v1" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-011", () => { + // 3.1 [40] — Test demonstrates mutliple Attibutes within the Start-tag. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a1": "v1", "@a2": "v2" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-012", () => { + // 2.3 [4] — Uses a legal XML 1.0 name consisting of a single colon character (disallowed by the latest + // XML Namespaces draft). + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@:": "v1" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-013", () => { + // 2.3 3.1 [13] [40] — Test demonstrates that the Attribute in a Start-tag can consist of numerals + // along with special characters. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@_.-0123456789": "v1" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-014", () => { + // 2.3 3.1 [13] [40] — Test demonstrates that all lower case letters are valid for the Attribute in a + // Start-tag. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@abcdefghijklmnopqrstuvwxyz": "v1" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-015", () => { + // 2.3 3.1 [13] [40] — Test demonstrates that all upper case letters are valid for the Attribute in a + // Start-tag. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@ABCDEFGHIJKLMNOPQRSTUVWXYZ": "v1" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-016", () => { + // 2.6 3.1 [16] [43] — Test demonstrates that Processing Instructions are valid element content. + const input: string = "\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-017", () => { + // 2.6 3.1 [16] [43] — Test demonstrates that Processing Instructions are valid element content and + // there can be more than one. + const input: string = "\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-018", () => { + // 2.7 3.1 [18] [43] — Test demonstrates that CDATA sections are valid element content. + const input: string = "\r\n]>\r\n]]>\r\n"; + const canonical = "<foo>"; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-019", () => { + // 2.7 3.1 [18] [43] — Test demonstrates that CDATA sections are valid element content and that + // ampersands may occur in their literal form. + const input: string = "\r\n]>\r\n\r\n"; + const canonical = "<&"; + const compact: unknown = { doc: "<&" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-020", () => { + // 2.7 3.1 [18] [43] — Test demonstractes that CDATA sections are valid element content and that + // everyting between the CDStart and CDEnd is recognized as character data not markup. + const input: string = "\r\n]>\r\n]]]>\r\n"; + const canonical = "<&]>]"; + const compact: unknown = { doc: "<&]>]" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-021", () => { + // 2.5 3.1 [15] [43] — Test demonstrates that comments are valid element content. + const input: string = "\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-022", () => { + // 2.5 3.1 [15] [43] — Test demonstrates that comments are valid element content and that all + // characters before the double-hypen right angle combination are considered part of thecomment. + const input: string = "\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-023", () => { + // 3.1 [43] — Test demonstrates that Entity References are valid element content. + const input: string = '\r\n\r\n]>\r\n&e;\r\n'; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-024", () => { + // 3.1 4.1 [43] [66] — Test demonstrates that Entity References are valid element content and also + // demonstrates a valid Entity Declaration. + const input: string = + '\r\n\r\n">\r\n]>\r\n&e;\r\n'; + const canonical = ""; + const compact: unknown = { doc: { foo: "" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-025", () => { + // 3.2 [46] — Test demonstrates an Element Type Declaration and that the contentspec can be of mixed + // content. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: { foo: ["", ""] } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-026", () => { + // 3.2 [46] — Test demonstrates an Element Type Declaration and that EMPTY is a valid contentspec. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: { foo: ["", ""] } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-027", () => { + // 3.2 [46] — Test demonstrates an Element Type Declaration and that ANY is a valid contenspec. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: { foo: ["", ""] } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-028", () => { + // 2.8 [24] — Test demonstrates a valid prolog that uses double quotes as delimeters around the + // VersionNum. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-029", () => { + // 2.8 [24] — Test demonstrates a valid prolog that uses single quotes as delimters around the + // VersionNum. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-030", () => { + // 2.8 [25] — Test demonstrates a valid prolog that contains whitespace on both sides of the equal sign + // in the VersionInfo. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-031", () => { + // 4.3.3 [80] — Test demonstrates a valid EncodingDecl within the prolog. + const input = Buffer.from( + "\r\n\r\n]>\r\n\r\n", + ); + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-032", () => { + // 2.9 [32] — Test demonstrates a valid SDDecl within the prolog. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-033", () => { + // 2.8 [23] — Test demonstrates that both a EncodingDecl and SDDecl are valid within the prolog. + const input = Buffer.from( + "\r\n\r\n]>\r\n\r\n", + ); + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-034", () => { + // 3.1 [44] — Test demonstrates the correct syntax for an Empty element tag. + const input: string = "\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-035", () => { + // 3.1 [44] — Test demonstrates that whitespace is permissible after the name in an Empty element tag. + const input: string = "\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-036", () => { + // 2.6 [16] — Test demonstrates a valid processing instruction. + const input: string = "\r\n]>\r\n\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-017a", () => { + // 2.6 3.1 [16] [43] — Test demonstrates that two apparently wrong Processing Instructions make a right + // one, with very odd content "some data ? > \r\n]>\r\n "; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-037", () => { + // 2.6 [15] — Test demonstrates a valid comment and that it may appear anywhere in the document + // including at the end. + const input: string = + "\r\n]>\r\n\r\n\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-038", () => { + // 2.6 [15] — Test demonstrates a valid comment and that it may appear anywhere in the document + // including the beginning. + const input: string = + "\r\n\r\n]>\r\n\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-039", () => { + // 2.6 [16] — Test demonstrates a valid processing instruction and that it may appear at the beginning + // of the document. + const input: string = "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-040", () => { + // 3.3 3.3.1 [52] [54] — Test demonstrates an Attribute List declaration that uses a StringType as the + // AttType. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a1": "\"<&>'" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-041", () => { + // 3.3.1 4.1 [54] [66] — Test demonstrates an Attribute List declaration that uses a StringType as the + // AttType and also expands the CDATA attribute with a character reference. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a1": "A" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-042", () => { + // 3.3.1 4.1 [54] [66] — Test demonstrates an Attribute List declaration that uses a StringType as the + // AttType and also expands the CDATA attribute with a character reference. The test also shows that + // the leading zeros in the character reference are ignored. + const input: string = + "\r\n]>\r\nA\r\n"; + const canonical = "A"; + const compact: unknown = { doc: "A" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-043", () => { + // 3.3 — An element's attributes may be declared before its content model; and attribute values may + // contain newlines. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a1": "foo bar" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-044", () => { + // 3.1 [44] — Test demonstrates that the empty-element tag must be use for an elements that are + // declared EMPTY. + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n\r\n\r\n\r\n\r\n'; + const canonical = + ' '; + const compact: unknown = { + doc: { + e: [ + { "@a1": "v1", "@a2": "v2", "@a3": "v3" }, + { "@a1": "w1", "@a2": "v2" }, + { "@a1": "v1", "@a2": "w2", "@a3": "v3" }, + ], + }, + }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-045", () => { + // 3.3 [52] — Tests whether more than one definition can be provided for the same attribute of a given + // element type with the first declaration being binding. + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a1": "v1" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-046", () => { + // 3.3 [52] — Test demonstrates that when more than one AttlistDecl is provided for a given element + // type, the contents of all those provided are merged. + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a1": "v1", "@a2": "v2" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-047", () => { + // 3.1 [43] — Test demonstrates that extra whitespace is normalized into single space character. + const input: string = "\r\n]>\r\nX\r\nY\r\n"; + const canonical = "X Y"; + const compact: unknown = { doc: "X\nY" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-048", () => { + // 2.4 3.1 [14] [43] — Test demonstrates that character data is valid element content. + const input: string = "\r\n]>\r\n]\r\n"; + const canonical = "]"; + const compact: unknown = { doc: "]" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-049", () => { + // 2.2 [2] — Test demonstrates that characters outside of normal ascii range can be used as element + // content. + const input = Buffer.from( + "//48ACEARABPAEMAVABZAFAARQAgAGQAbwBjACAAWwANAAoAPAAhAEUATABFAE0ARQBOAFQAIABkAG8AYwAgACgAIwBQAEMARABBAFQAQQApAD4ADQAKAF0APgANAAoAPABkAG8AYwA+AKMAPAAvAGQAbwBjAD4ADQAKAA==", + "base64", + ); + const canonical = "£"; + const compact: unknown = { doc: "£" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-050", () => { + // 2.2 [2] — Test demonstrates that characters outside of normal ascii range can be used as element + // content. + const input = Buffer.from( + "//48ACEARABPAEMAVABZAFAARQAgAGQAbwBjACAAWwANAAoAPAAhAEUATABFAE0ARQBOAFQAIABkAG8AYwAgACgAIwBQAEMARABBAFQAQQApAD4ADQAKAF0APgANAAoAPABkAG8AYwA+AEAOCA4hDioOTA48AC8AZABvAGMAPgANAAoA", + "base64", + ); + const canonical = "เจมส์"; + const compact: unknown = { doc: "เจมส์" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-051", () => { + // 2.2 [2] — The document is encoded in UTF-16 and uses some name characters well outside of the normal + // ASCII range. + const input = Buffer.from( + "//48ACEARABPAEMAVABZAFAARQAgAEAOCA4hDioOTA4gAFsADQAKADwAIQBFAEwARQBNAEUATgBUACAAQA4IDiEOKg5MDiAAIAAoACMAUABDAEQAQQBUAEEAKQA+AA0ACgBdAD4ADQAKADwAQA4IDiEOKg5MDj4APAAvAEAOCA4hDioOTA4+AA0ACgA=", + "base64", + ); + const canonical = "<เจมส์>"; + const compact: unknown = { "เจมส์": "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-052", () => { + // 2.2 [2] — The document is encoded in UTF-8 and the text inside the root element uses two non-ASCII + // characters, encoded in UTF-8 and each of which expands to a Unicode surrogate pair. + const input: string = "\r\n]>\r\n𐀀􏿽\r\n"; + const canonical = "𐀀􏿽"; + const compact: unknown = { doc: "𐀀􏿽" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-053", () => { + // 4.4.2 — Tests inclusion of a well-formed internal entity, which holds an element required by the + // content model. + const input: string = + '">\r\n\r\n\r\n]>\r\n&e;\r\n'; + const canonical = ""; + const compact: unknown = { doc: { e: "" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-054", () => { + // 3.1 [40] [42] — Test demonstrates that extra whitespace within Start-tags and End-tags are nomalized + // into single spaces. + const input: string = + "\r\n]>\r\n\r\n\r\n\r\n\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-055", () => { + // 2.6 2.10 [16] — Test demonstrates that extra whitespace within a processing instruction + // willnormalized into s single space character. + const input: string = "\r\n]>\r\n\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-056", () => { + // 3.3.1 4.1 [54] [66] — Test demonstrates an Attribute List declaration that uses a StringType as the + // AttType and also expands the CDATA attribute with a character reference. The test also shows that + // the leading zeros in the character reference are ignored. + const input: string = + "\r\n]>\r\nA\r\n"; + const canonical = "A"; + const compact: unknown = { doc: "A" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-057", () => { + // 3.2.1 [47] — Test demonstrates an element content model whose element can occur zero or more times. + const input: string = "\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-058", () => { + // 3.3.3 — Test demonstrates that extra whitespace be normalized into a single space character in an + // attribute of type NMTOKENS. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a1": "1 2" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-059", () => { + // 3.2 3.3 [46] [53] — Test demonstrates an Element Type Declaration that uses the contentspec of + // EMPTY. The element cannot have any contents and must always appear as an empty element in the + // document. The test also shows an Attribute-list declaration with multiple AttDef's. + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n\r\n\r\n\r\n\r\n'; + const canonical = + ' '; + const compact: unknown = { + doc: { + e: [ + { "@a1": "v1", "@a2": "v2", "@a3": "v3" }, + { "@a1": "w1", "@a2": "v2" }, + { "@a1": "v1", "@a2": "w2", "@a3": "v3" }, + ], + }, + }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-060", () => { + // 4.1 [66] — Test demonstrates the use of decimal Character References within element content. + const input: string = "\r\n]>\r\nX Y\r\n"; + const canonical = "X Y"; + const compact: unknown = { doc: "X\nY" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-061", () => { + // 4.1 [66] — Test demonstrates the use of decimal Character References within element content. + const input: string = "\r\n]>\r\n£\r\n"; + const canonical = "£"; + const compact: unknown = { doc: "£" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-062", () => { + // 4.1 [66] — Test demonstrates the use of hexadecimal Character References within element. + const input: string = "\r\n]>\r\nเจมส์\r\n"; + const canonical = "เจมส์"; + const compact: unknown = { doc: "เจมส์" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-063", () => { + // 2.3 [5] — The document is encoded in UTF-8 and the name of the root element type uses non-ASCII + // characters. + const input: string = "\r\n]>\r\n<เจมส์>\r\n"; + const canonical = "<เจมส์>"; + const compact: unknown = { "เจมส์": "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-064", () => { + // 4.1 [66] — Tests in-line handling of two legal character references, which each expand to a Unicode + // surrogate pair. + const input: string = "\r\n]>\r\n𐀀􏿽\r\n"; + const canonical = "𐀀􏿽"; + const compact: unknown = { doc: "𐀀􏿽" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-065", () => { + // 4.5 — Tests ability to define an internal entity which can't legally be expanded (contains an + // unquoted <). + const input: string = '\r\n\r\n]>\r\n\r\n'; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-066", () => { + // 4.1 [66] — Expands a CDATA attribute with a character reference. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a1": '"' } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-067", () => { + // 4.1 [66] — Test demonstrates the use of decimal character references within element content. + const input: string = "\r\n]>\r\n \r\n"; + const canonical = " "; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-068", () => { + // 2.11, 4.5 — Tests definition of an internal entity holding a carriage return character reference, + // which must not be normalized before reporting to the application. Line break normalization only + // occurs when parsing external parsed entities. + const input: string = + '\r\n\r\n]>\r\n&e;\r\n'; + const canonical = " "; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-069", () => { + // 4.7 — Verifies that an XML parser will parse a NOTATION declaration; the output phase of this test + // ensures that it's reported to the application. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-070", () => { + // 4.4.8 — Verifies that internal parameter entities are correctly expanded within the internal subset. + // (upstream: valid; external parameter entities are not read) + const input: string = '">\r\n%e;\r\n]>\r\n\r\n'; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-071", () => { + // 3.3 3.3.1 [52] [56] — Test demonstrates that an AttlistDecl can use ID as the TokenizedType within + // the Attribute type. The test also shows that IMPLIED is a valid DefaultDecl. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-072", () => { + // 3.3 3.3.1 [52] [56] — Test demonstrates that an AttlistDecl can use IDREF as the TokenizedType + // within the Attribute type. The test also shows that IMPLIED is a valid DefaultDecl. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-073", () => { + // 3.3 3.3.1 [52] [56] — Test demonstrates that an AttlistDecl can use IDREFS as the TokenizedType + // within the Attribute type. The test also shows that IMPLIED is a valid DefaultDecl. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-074", () => { + // 3.3 3.3.1 [52] [56] — Test demonstrates that an AttlistDecl can use ENTITY as the TokenizedType + // within the Attribute type. The test also shows that IMPLIED is a valid DefaultDecl. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-075", () => { + // 3.3 3.3.1 [52] [56] — Test demonstrates that an AttlistDecl can use ENTITIES as the TokenizedType + // within the Attribute type. The test also shows that IMPLIED is a valid DefaultDecl. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-076", () => { + // 3.3.1 — Verifies that an XML parser will parse a NOTATION attribute; the output phase of this test + // ensures that both notations are reported to the application. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n'; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-077", () => { + // 3.3 3.3.1 [52] [54] — Test demonstrates that an AttlistDecl can use an EnumeratedType within the + // Attribute type. The test also shows that IMPLIED is a valid DefaultDecl. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-078", () => { + // 3.3 3.3.1 [52] [54] — Test demonstrates that an AttlistDecl can use an StringType of CDATA within + // the Attribute type. The test also shows that REQUIRED is a valid DefaultDecl. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a": "v" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-079", () => { + // 3.3 3.3.2 [52] [60] — Test demonstrates that an AttlistDecl can use an StringType of CDATA within + // the Attribute type. The test also shows that FIXED is a valid DefaultDecl and that a value can be + // given to the attribute in the Start-tag as well as the AttListDecl. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a": "v" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-080", () => { + // 3.3 3.3.2 [52] [60] — Test demonstrates that an AttlistDecl can use an StringType of CDATA within + // the Attribute type. The test also shows that FIXED is a valid DefaultDecl and that an value can be + // given to the attribute. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a": "v" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-081", () => { + // 3.2.1 [50] — Test demonstrates the use of the optional character following a name or list to govern + // the number of times an element or content particles in the list occur. + const input: string = + "\r\n\r\n\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: { a: "", b: "", c: { a: "" } } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-082", () => { + // 4.2 [72] — Tests that an external PE may be defined (but not referenced). + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-083", () => { + // 4.2 [72] — Tests that an external PE may be defined (but not referenced). + const input: string = + "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-084", () => { + // 2.10 — Test demonstrates that although whitespace can be used to set apart markup for greater + // readability it is not necessary. + const input: string = "]>\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-085", () => { + // 4 — Parameter and General entities use different namespaces, so there can be an entity of each type + // with a given name. + const input: string = + '\r\n">\r\n\r\n]>\r\n&e;\r\n'; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-086", () => { + // 4.2 — Tests whether entities may be declared more than once, with the first declaration being the + // binding one. + const input: string = + '\r\n\r\n">\r\n]>\r\n&e;\r\n'; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-087", () => { + // 4.5 — Tests whether character references in internal entities are expanded early enough, by relying + // on correct handling to make the entity be well formed. + const input: string = + '\r\n\r\n\r\n]>\r\n&e;\r\n'; + const canonical = ""; + const compact: unknown = { doc: { foo: "" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-088", () => { + // 4.5 — Tests whether entity references in internal entities are expanded late enough, by relying on + // correct handling to make the expanded text be valid. (If it's expanded too early, the entity will + // parse as an element that's not valid in that context.) + const input: string = + '\r\n">\r\n]>\r\n&e;\r\n'; + const canonical = "<foo>"; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-089", () => { + // 4.1 [66] — Tests entity expansion of three legal character references, which each expand to a + // Unicode surrogate pair. + const input: string = + '\r\n\r\n]>\r\n&e;\r\n'; + const canonical = "𐀀􏿽􏿿"; + const compact: unknown = { doc: "𐀀􏿽􏿿" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-090", () => { + // 3.3.1 — Verifies that an XML parser will parse a NOTATION attribute; the output phase of this test + // ensures that the notation is reported to the application. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n'; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-091", () => { + // 3.3.1 — Verifies that an XML parser will parse an ENTITY attribute; the output phase of this test + // ensures that the notation is reported to the application, and for validating parsers it further + // tests that the entity is so reported. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a": "e" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-092", () => { + // 2.3 2.10 — Test demostrates that extra whitespace is normalized into a single space character. + const input: string = + "\r\n\r\n]>\r\n\r\n\r\n \t\r\n\r\n\r\n\r\n"; + const canonical = " "; + const compact: unknown = { doc: { a: ["", "", ""] } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-093", () => { + // 2.10 — Test demonstrates that extra whitespace is not intended for inclusion in the delivered + // version of the document. + const input: string = "\n]>\n\n\n\n\n"; + const canonical = " "; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-094", () => { + // 2.8 — Attribute defaults with a DTD have special parsing rules, different from other strings. That + // means that characters found there may look like an undefined parameter entity reference "within a + // markup declaration", but they aren't ... so they can't be violating the PEs in Internal Subset WFC. + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a1": "%e;" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-095", () => { + // 3.3.3 — Basically an output test, this requires extra whitespace to be normalized into a single + // space character in an attribute of type NMTOKENS. + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a1": "1 2" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-096", () => { + // 3.3.3 — Test demonstrates that extra whitespace is normalized into a single space character in an + // attribute of type NMTOKENS. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a1": "1 2" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-097", () => { + // 3.3 — Basically an output test, this tests whether an externally defined attribute declaration (with + // a default) takes proper precedence over a subsequent internal declaration. (upstream: valid; + // external parameter entities are not read) + const input: string = + '\r\n\r\n\r\n%e;\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a1": "v1" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-098", () => { + // 2.6 2.10 [16] — Test demonstrates that extra whitespace within a processing instruction is converted + // into a single space character. + const input: string = "\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-099", () => { + // 4.3.3 [81] — Test demonstrates the name of the encoding can be composed of lowercase characters. + const input = Buffer.from( + '\r\n\r\n]>\r\n\r\n', + ); + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-100", () => { + // 2.3 [12] — Makes sure that PUBLIC identifiers may have some strange characters. NOTE: The XML + // editors have said that the XML specification errata will specify that parameter entity expansion + // does not occur in PUBLIC identifiers, so that the '%' character will not flag a malformed parameter + // entity reference. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-101", () => { + // 4.5 — This tests whether entity expansion is (incorrectly) done while processing entity + // declarations; if it is, the entity value literal will terminate prematurely. + const input: string = '\r\n\r\n]>\r\n\r\n'; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-102", () => { + // 3.3.3 — Test demonstrates that a CDATA attribute can pass a double quote as its value. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a": '"' } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-103", () => { + // 3.3.3 — Test demonstrates that an attribute can pass a less than sign as its value. + const input: string = "\r\n]>\r\n<doc>\r\n"; + const canonical = "<doc>"; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-104", () => { + // 3.1 [40] — Test demonstrates that extra whitespace within an Attribute of a Start-tag is normalized + // to a single space character. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a": "x y" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-105", () => { + // 3.3.3 — Basically an output test, this requires a CDATA attribute with a tab character to be passed + // through as one space. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a": "x\ty" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-106", () => { + // 3.3.3 — Basically an output test, this requires a CDATA attribute with a newline character to be + // passed through as one space. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a": "x\ny" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-107", () => { + // 3.3.3 — Basically an output test, this requires a CDATA attribute with a return character to be + // passed through as one space. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a": "x\ry" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-108", () => { + // 2.11, 3.3.3 — This tests normalization of end-of-line characters (CRLF) within entities to LF, + // primarily as an output test. + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a": "x y" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-109", () => { + // 2.3 3.1 [10][40][41] — Test demonstrates that an attribute can have a null value. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a": "" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-110", () => { + // 3.3.3 — Basically an output test, this requires that a CDATA attribute with a CRLF be normalized to + // one space. + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a": "x y" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-111", () => { + // 3.3.3 — Character references expanding to spaces doesn't affect treatment of attributes. + const input: string = + '\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a": "x y" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-112", () => { + // 3.2.1 [48][49] — Test demonstrates shows the use of content particles within the element content. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: { a: "" } }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-113", () => { + // 3.3 [52][53] — Test demonstrates that it is not an error to have attributes declared for an element + // not itself declared. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-114", () => { + // 2.7 [20] — Test demonstrates that all text within a valid CDATA section is considered text and not + // recognized as markup. + const input: string = + '\r\n">\r\n]>\r\n&e;\r\n'; + const canonical = "&foo;"; + const compact: unknown = { doc: "&foo;" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-115", () => { + // 3.3.3 — Test demonstrates that an entity reference is processed by recursively processing the + // replacement text of the entity. + const input: string = + '\r\n\r\n\r\n]>\r\n&e1;\r\n'; + const canonical = "v"; + const compact: unknown = { doc: "v" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-116", () => { + // 2.11 — Test demonstrates that a line break within CDATA will be normalized. + const input: string = "\r\n]>\r\n\r\n"; + const canonical = " "; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-117", () => { + // 4.5 — Test demonstrates that entity expansion is done while processing entity declarations. + const input: string = + '\r\n\r\n]>\r\n]\r\n'; + const canonical = "]"; + const compact: unknown = { doc: "]" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-118", () => { + // 4.5 — Test demonstrates that entity expansion is done while processing entity declarations. + const input: string = + '\r\n\r\n]>\r\n]\r\n'; + const canonical = "]]"; + const compact: unknown = { doc: "]]" }; + expectParses(input, canonical, compact); + }); + + test("valid-sa-119", () => { + // 2.5 — Comments may contain any legal XML characters; only the string "--" is disallowed. + const input: string = "\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-not-sa-001", () => { + // 4.2.2 [75] — Test demonstrates the use of an ExternalID within a document type definition. + // (upstream: valid; external general and parameter entities are not read) + const input: string = '\r\n]>\r\n\r\n'; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-not-sa-002", () => { + // 4.2.2 [75] — Test demonstrates the use of an ExternalID within a document type definition. + // (upstream: valid; external general and parameter entities are not read) + const input: string = '\r\n]>\r\n\r\n'; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-not-sa-003", () => { + // 4.1 [69] — Test demonstrates the expansion of an external parameter entity that declares an + // attribute. (upstream: valid; external general and parameter entities are not read; output depends on + // them) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("valid-not-sa-004", () => { + // 4.1 [69] — Expands an external parameter entity in two different ways, with one of them declaring an + // attribute. (upstream: valid; external general and parameter entities are not read; output depends on + // them) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("valid-not-sa-005", () => { + // 4.1 [69] — Test demonstrates the expansion of an external parameter entity that declares an + // attribute. (upstream: valid; external general and parameter entities are not read; output depends on + // them) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("valid-not-sa-006", () => { + // 3.3 [52] — Test demonstrates that when more than one definition is provided for the same attribute + // of a given element type only the first declaration is binding. (upstream: valid; external general + // and parameter entities are not read; output depends on them) + const input: string = '\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("valid-not-sa-007", () => { + // 3.3 [52] — Test demonstrates the use of an Attribute list declaration within an external entity. + // (upstream: valid; external general and parameter entities are not read; output depends on them) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("valid-not-sa-008", () => { + // 4.2.2 [75] — Test demonstrates that an external identifier may include a public identifier. + // (upstream: valid; external general and parameter entities are not read; output depends on them) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("valid-not-sa-009", () => { + // 4.2.2 [75] — Test demonstrates that an external identifier may include a public identifier. + // (upstream: valid; external general and parameter entities are not read; output depends on them) + const input: string = + '\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("valid-not-sa-010", () => { + // 3.3 [52] — Test demonstrates that when more that one definition is provided for the same attribute + // of a given element type only the first declaration is binding. (upstream: valid; external general + // and parameter entities are not read) + const input: string = '\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { doc: { "@a1": "v1" } }; + expectParses(input, canonical, compact); + }); + + test("valid-not-sa-011", () => { + // 4.2 4.2.1 [72] [75] — Test demonstrates a parameter entity declaration whose parameter entity + // definition is an ExternalID. (upstream: valid; external general and parameter entities are not read; + // output depends on them) + const input: string = '\r\n%e;\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("valid-not-sa-012", () => { + // 4.3.1 [77] — Test demonstrates an enternal parsed entity that begins with a text declaration. + // (upstream: valid; external general and parameter entities are not read; output depends on them) + const input: string = '\r\n%e;\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("valid-not-sa-013", () => { + // 3.4 [62] — Test demonstrates the use of the conditional section INCLUDE that will include its + // contents as part of the DTD. (upstream: valid; external general and parameter entities are not read; + // output depends on them) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("valid-not-sa-014", () => { + // 3.4 [62] — Test demonstrates the use of the conditional section INCLUDE that will include its + // contents as part of the DTD. The keyword is a parameter-entity reference. (upstream: valid; external + // general and parameter entities are not read; output depends on them) + const input: string = '\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("valid-not-sa-015", () => { + // 3.4 [63] — Test demonstrates the use of the conditonal section IGNORE the will ignore its content + // from being part of the DTD. The keyword is a parameter-entity reference. (upstream: valid; external + // general and parameter entities are not read; output depends on them) + const input: string = '\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("valid-not-sa-016", () => { + // 3.4 [62] — Test demonstrates the use of the conditional section INCLUDE that will include its + // contents as part of the DTD. The keyword is a parameter-entity reference. (upstream: valid; external + // general and parameter entities are not read; output depends on them) + const input: string = '\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("valid-not-sa-017", () => { + // 4.2 [72] — Test demonstrates a parameter entity declaration that contains an attribute list + // declaration. (upstream: valid; external general and parameter entities are not read; output depends + // on them) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("valid-not-sa-018", () => { + // 4.2.2 [75] — Test demonstrates an EnternalID whose contents contain an parameter entity declaration + // and a attribute list definition. (upstream: valid; external general and parameter entities are not + // read; output depends on them) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("valid-not-sa-019", () => { + // 4.4.8 — Test demonstrates that a parameter entity will be expanded with spaces on either side. + // (upstream: valid; external general and parameter entities are not read; output depends on them) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("valid-not-sa-020", () => { + // 4.4.8 — Parameter entities expand with spaces on either side. (upstream: valid; external general and + // parameter entities are not read; output depends on them) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("valid-not-sa-021", () => { + // 4.2 [72] — Test demonstrates a parameter entity declaration that contains a partial attribute list + // declaration. (upstream: valid; external general and parameter entities are not read; output depends + // on them) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("valid-not-sa-023", () => { + // 2.3 4.1 [10] [69] — Test demonstrates the use of a parameter entity reference within an attribute + // list declaration. (upstream: valid; external general and parameter entities are not read; output + // depends on them) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("valid-not-sa-024", () => { + // 2.8, 4.1 [69] — Constructs an declaration from several PEs. (upstream: valid; external + // general and parameter entities are not read; output depends on them) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("valid-not-sa-025", () => { + // 4.2 — Test demonstrates that when more that one definition is provided for the same entity only the + // first declaration is binding. (upstream: valid; external general and parameter entities are not + // read; output depends on them) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("valid-not-sa-026", () => { + // 3.3 [52] — Test demonstrates that when more that one definition is provided for the same attribute + // of a given element type only the first declaration is binding. (upstream: valid; external general + // and parameter entities are not read; output depends on them) + const input: string = + '\r\n\r\n%e;\r\n\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("valid-not-sa-027", () => { + // 4.1 [69] — Test demonstrates a parameter entity reference whose value is NULL. (upstream: valid; + // external general and parameter entities are not read) + const input: string = '\r\n\r\n'; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-not-sa-028", () => { + // 3.4 [62] — Test demonstrates the use of the conditional section INCLUDE that will include its + // contents. (upstream: valid; external general and parameter entities are not read; output depends on + // them) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("valid-not-sa-029", () => { + // 3.4 [62] — Test demonstrates the use of the conditonal section IGNORE the will ignore its content + // from being used. (upstream: valid; external general and parameter entities are not read; output + // depends on them) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("valid-not-sa-030", () => { + // 3.4 [62] — Test demonstrates the use of the conditonal section IGNORE the will ignore its content + // from being used. (upstream: valid; external general and parameter entities are not read) + const input: string = '\r\n\r\n'; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("valid-not-sa-031", () => { + // 2.7 — Expands a general entity which contains a CDATA section with what looks like a markup + // declaration (but is just text since it's in a CDATA section). (upstream: valid; external general and + // parameter entities are not read; output depends on them) + const input: string = '\r\n&e;\r\n'; + expectParses(input); + }); + + test("valid-ext-sa-001", () => { + // 2.11 — A combination of carriage return line feed in an external entity must be normalized to a + // single newline. (upstream: valid; external general and parameter entities are not read; output + // depends on them) + const input: string = + '\r\n\r\n]>\r\n&e;\r\n'; + expectParses(input); + }); + + test("valid-ext-sa-002", () => { + // 2.11 — A carriage return (also CRLF) in an external entity must be normalized to a single newline. + // (upstream: valid; external general and parameter entities are not read; output depends on them) + const input: string = + '\r\n\r\n]>\r\n&e;\r\n'; + expectParses(input); + }); + + test("valid-ext-sa-003", () => { + // 3.1 4.1 [43] [68] — Test demonstrates that the content of an element can be empty. In this case the + // external entity is an empty file. (upstream: valid; external general and parameter entities are not + // read; output depends on them) + const input: string = + '\r\n\r\n]>\r\n&e;\r\n'; + expectParses(input); + }); + + test("valid-ext-sa-004", () => { + // 2.11 — A carriage return (also CRLF) in an external entity must be normalized to a single newline. + // (upstream: valid; external general and parameter entities are not read; output depends on them) + const input: string = + '\r\n\r\n]>\r\n&e;\r\n'; + expectParses(input); + }); + + test("valid-ext-sa-005", () => { + // 3.2.1 4.2.2 [48] [75] — Test demonstrates the use of optional character and content particles within + // an element content. The test also show the use of external entity. (upstream: valid; external + // general and parameter entities are not read; output depends on them) + const input: string = + '\r\n\r\n\r\n]>\r\n&e;\r\n'; + expectParses(input); + }); + + test("valid-ext-sa-006", () => { + // 2.11 3.2.1 3.2.2 4.2.2 [48] [51] [75] — Test demonstrates the use of optional character and content + // particles within mixed element content. The test also shows the use of an external entity and that a + // carriage control line feed in an external entity must be normalized to a single newline. (upstream: + // valid; external general and parameter entities are not read; output depends on them) + const input: string = + '\r\n\r\n\r\n]>\r\n&e;\r\n'; + expectParses(input); + }); + + test("valid-ext-sa-007", () => { + // 4.2.2 4.4.3 [75] — Test demonstrates the use of external entity and how replacement text is + // retrieved and processed. (upstream: valid; external general and parameter entities are not read; + // output depends on them) + const input: string = + '\r\n\r\n]>\r\nX&e;Z\r\n'; + expectParses(input); + }); + + test("valid-ext-sa-008", () => { + // 4.2.2 4.3.3. 4.4.3 [75] [80] — Test demonstrates the use of external entity and how replacement text + // is retrieved and processed. Also tests the use of an EncodingDecl of UTF-16. (upstream: valid; + // external general and parameter entities are not read; output depends on them) + const input: string = + '\r\n\r\n]>\r\nX&e;Z\r\n'; + expectParses(input); + }); + + test("valid-ext-sa-009", () => { + // 2.11 — A carriage return (also CRLF) in an external entity must be normalized to a single newline. + // (upstream: valid; external general and parameter entities are not read; output depends on them) + const input: string = + '\r\n\r\n]>\r\n&e;\r\n'; + expectParses(input); + }); + + test("valid-ext-sa-011", () => { + // 2.11 4.2.2 [75] — Test demonstrates the use of a public identifier with and external entity. The + // test also show that a carriage control line feed combination in an external entity must be + // normalized to a single newline. (upstream: valid; external general and parameter entities are not + // read; output depends on them) + const input: string = + '\r\n\r\n]>\r\n&e;\r\n'; + expectParses(input); + }); + + test("valid-ext-sa-012", () => { + // 4.2.1 4.2.2 — Test demonstrates both internal and external entities and that processing of entity + // references may be required to produce the correct replacement text. (upstream: valid; external + // general and parameter entities are not read; output depends on them) + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n&e1;\r\n'; + expectParses(input); + }); + + test("valid-ext-sa-013", () => { + // 3.3.3 — Test demonstrates that whitespace is handled by adding a single whitespace to the normalized + // value in the attribute list. (upstream: valid; external general and parameter entities are not read; + // output depends on them) + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n&x;\r\n'; + expectParses(input); + }); + + test("valid-ext-sa-014", () => { + // 4.1 4.4.3 [68] — Test demonstrates use of characters outside of normal ASCII range. (upstream: + // valid; external general and parameter entities are not read; output depends on them) + const input: string = + '\r\n\r\n]>\r\n&e;\r\n'; + expectParses(input); + }); +}); + +describe("japanese", () => { + test("weekly-euc-jp", () => { + // 4.3.3 [4,84] — Test support for EUC-JP encoding, and XML names which contain Japanese characters. If + // a processor does not support this encoding, it must report a fatal error. (upstream: optional error) + const input = Buffer.from( + "PD94bWwgdmVyc2lvbj0iMS4wIiBlbmNvZGluZz0iZXVjLWpwIj8+DQo8IURPQ1RZUEUgvbXK8yBTWVNURU0gIndlZWtseS1ldWMtanAuZHRkIj4NCjwhLS0gvbXK86W1pfOl16XrIC0tPg0KPL21yvM+DQogIDzHr7fuvbU+DQogICAgPMevxdk+MTk5Nzwvx6/F2T4NCiAgICA8t+7F2T4xPC+37sXZPg0KICAgIDy9tT4xPC+9tT4NCiAgPC/Hr7fuvbU+DQoNCiAgPLvhzL4+DQogICAgPLvhPruzxcQ8L7vhPg0KICAgIDzMvj7CwM+6PC/Mvj4NCiAgPC+74cy+Pg0KDQogIDy2yMyzyvO58KXqpbmlyD4NCiAgICA8tsjMs8rzufA+DQogICAgICA8tsjMs8y+PlhNTKWopcelo6W/obykzrruwK48L7bIzLPMvj4NCiAgICAgIDy2yMyzpbOhvKXJPlgzMzU1LTIzPC+2yMyzpbOhvKXJPg0KICAgICAgPLmpv/S0yc39Pg0KICAgICAgICA8uKvA0aTipOq5qb/0PjE2MDA8L7irwNGk4qTquam/9D4NCiAgICAgICAgPLzCwNO5qb/0PjMyMDwvvMLA07mpv/Q+DQogICAgICAgIDzF9rfuuKvA0aTipOq5qb/0PjE2MDwvxfa37rirwNGk4qTquam/9D4NCiAgICAgICAgPMX2t+68wsDTuam/9D4yNDwvxfa37rzCwNO5qb/0Pg0KICAgICAgPC+5qb/0tMnN/T4NCiAgICAgIDzNvcTqueDM3KXqpbmlyD4NCiAgICAgICAgPM29xOq54MzcPg0KICAgICAgICAgIDxQPlhNTKWopcelo6W/obykzrTwy9y7xc3NpM667sCuPC9QPg0KICAgICAgICA8L829xOq54MzcPg0KICAgICAgPC/NvcTqueDM3KXqpbmlyD4NCiAgICAgIDy8wrvcu/a54KXqpbmlyD4NCiAgICAgICAgPLzCu9y79rngPg0KICAgICAgICAgIDxQPlhNTKWopcelo6W/obykzrTwy9y7xc3NpM667sCuPC9QPg0KICAgICAgICA8L7zCu9y79rngPg0KICAgICAgICA8vMK73Lv2ueA+DQogICAgICAgICAgPFA+tqW558K+vNLAvcnKpM61oce9xLS6ujwvUD4NCiAgICAgICAgPC+8wrvcu/a54D4NCiAgICAgIDwvvMK73Lv2ueCl6qW5pcg+DQogICAgICA8vuXEuaTYpM7N18DBu/a54KXqpbmlyD4NCiAgICAgICAgPL7lxLmk2KTOzdfAwbv2ueA+DQogICAgICAgICAgPFA+xsOky6TKpLc8L1A+DQogICAgICAgIDwvvuXEuaTYpM7N18DBu/a54D4NCiAgICAgIDwvvuXEuaTYpM7N18DBu/a54KXqpbmlyD4NCiAgICAgIDzM5MLqxcDC0Lr2Pg0KICAgICAgICA8UD5YTUykyKTPsr+kq6TvpKuk6aTKpKShozwvUD4NCiAgICAgIDwvzOTC6sXAwtC69j4NCiAgICA8L7bIzLPK87nwPg0KDQogICAgPLbIzLPK87nwPg0KICAgICAgPLbIzLPMvj64obr3pail86W4pfOkzrOryK88L7bIzLPMvj4NCiAgICAgIDy2yMyzpbOhvKXJPlM4ODIxLTc2PC+2yMyzpbOhvKXJPg0KICAgICAgPLmpv/S0yc39Pg0KICAgICAgICA8uKvA0aTipOq5qb/0PjEyMDwvuKvA0aTipOq5qb/0Pg0KICAgICAgICA8vMLA07mpv/Q+NjwvvMLA07mpv/Q+DQogICAgICAgIDzF9rfuuKvA0aTipOq5qb/0PjMyPC/F9rfuuKvA0aTipOq5qb/0Pg0KICAgICAgICA8xfa37rzCwNO5qb/0PjI8L8X2t+68wsDTuam/9D4NCiAgICAgIDwvuam/9LTJzf0+DQogICAgICA8zb3E6rngzNyl6qW5pcg+DQogICAgICAgIDzNvcTqueDM3D4NCiAgICAgICAgICA8UD48QSBocmVmPSJodHRwOi8vd3d3Lmdvby5uZS5qcCI+Z29vPC9BPqTOtaHHvaTyxLSk2aTGpN+k6zwvUD4NCiAgICAgICAgPC/NvcTqueDM3D4NCiAgICAgIDwvzb3E6rngzNyl6qW5pcg+DQogICAgICA8vMK73Lv2ueCl6qW5pcg+DQogICAgICAgIDy8wrvcu/a54D4NCiAgICAgICAgICA8UD65uaTLoaKkyaSmpKSkprihuvelqKXzpbil86SspKKk66SrxLS6uqS5pOs8L1A+DQogICAgICAgIDwvvMK73Lv2ueA+DQogICAgICA8L7zCu9y79rngpeqluaXIPg0KICAgICAgPL7lxLmk2KTOzdfAwbv2ueCl6qW5pcg+DQogICAgICAgIDy+5cS5pNikzs3XwMG79rngPg0KICAgICAgICAgIDxQPrOryK+k8qS5pOukzqTPpOGk86TJpKakyqTOpMehollhaG9vIaTyx+O8/aS3pMayvKS1pKShozwvUD4NCiAgICAgICAgPC++5cS5pNikzs3XwMG79rngPg0KICAgICAgPC++5cS5pNikzs3XwMG79rngpeqluaXIPg0KICAgICAgPMzkwurFwMLQuvY+DQogICAgICAgIDxQPrihuvelqKXzpbil86THvNak8sH2pOmku6TrpLOkyKSspMekraTKpKSho6HKzdfEtLq6ocs8L1A+DQogICAgICA8L8zkwurFwMLQuvY+DQogICAgPC+2yMyzyvO58D4NCiAgPC+2yMyzyvO58KXqpbmlyD4NCjwvvbXK8z4NCg==", + "base64", + ); + expectRejects(input, "XML Parse error: Unsupported encoding 'euc-jp' (supported: UTF-8, UTF-16, ISO-8859-1)"); + }); + + test("weekly-iso-2022-jp", () => { + // 4.3.3 [4,84] — Test support for ISO-2022-JP encoding, and XML names which contain Japanese + // characters. If a processor does not support this encoding, it must report a fatal error. (upstream: + // optional error) + const input = Buffer.from( + '\r\n\r\n\r\n<\u001b$B=5Js\u001b(B>\r\n <\u001b$BG/7n=5\u001b(B>\r\n <\u001b$BG/EY\u001b(B>1997\r\n <\u001b$B7nEY\u001b(B>1\r\n <\u001b$B=5\u001b(B>1\r\n \r\n\r\n <\u001b$B;aL>\u001b(B>\r\n <\u001b$B;a\u001b(B>\u001b$B;3ED\u001b(B\r\n <\u001b$BL>\u001b(B>\u001b$BB@O:\u001b(B\u001b(B>\r\n \u001b(B>\r\n\r\n <\u001b$B6HL3Js9p%j%9%H\u001b(B>\r\n <\u001b$B6HL3Js9p\u001b(B>\r\n <\u001b$B6HL3L>\u001b(B>XML\u001b$B%(%G%#%?!<$N:n@.\u001b(B\u001b(B>\r\n <\u001b$B6HL3%3!<%I\u001b(B>X3355-23\r\n <\u001b$B9)?t4IM}\u001b(B>\r\n <\u001b$B8+@Q$b$j9)?t\u001b(B>1600\r\n <\u001b$B320\r\n <\u001b$BEv7n8+@Q$b$j9)?t\u001b(B>160\r\n <\u001b$BEv7n24\r\n \r\n <\u001b$BM=Dj9`L\\%j%9%H\u001b(B>\r\n <\u001b$BM=Dj9`L\\\u001b(B>\r\n

XML\u001b$B%(%G%#%?!<$N4pK\\;EMM$N:n@.\u001b(B

\r\n \r\n \r\n <\u001b$B\r\n <\u001b$B\r\n

XML\u001b$B%(%G%#%?!<$N4pK\\;EMM$N:n@.\u001b(B

\r\n \r\n <\u001b$B\r\n

\u001b$B6%9gB>\r\n \r\n \r\n <\u001b$B>eD9$X$NMW@A;v9`%j%9%H\u001b(B>\r\n <\u001b$B>eD9$X$NMW@A;v9`\u001b(B>\r\n

\u001b$BFC$K$J$7\u001b(B

\r\n eD9$X$NMW@A;v9`\u001b(B>\r\n eD9$X$NMW@A;v9`%j%9%H\u001b(B>\r\n <\u001b$BLdBjE@BP:v\u001b(B>\r\n

XML\u001b$B$H$O2?$+$o$+$i$J$$!#\u001b(B

\r\n \r\n \r\n\r\n <\u001b$B6HL3Js9p\u001b(B>\r\n <\u001b$B6HL3L>\u001b(B>\u001b$B8!:w%(%s%8%s$N3+H/\u001b(B\u001b(B>\r\n <\u001b$B6HL3%3!<%I\u001b(B>S8821-76\r\n <\u001b$B9)?t4IM}\u001b(B>\r\n <\u001b$B8+@Q$b$j9)?t\u001b(B>120\r\n <\u001b$B6\r\n <\u001b$BEv7n8+@Q$b$j9)?t\u001b(B>32\r\n <\u001b$BEv7n2\r\n \r\n <\u001b$BM=Dj9`L\\%j%9%H\u001b(B>\r\n <\u001b$BM=Dj9`L\\\u001b(B>\r\n

goo\u001b$B$N5!G=$rD4$Y$F$_$k\u001b(B

\r\n \r\n \r\n <\u001b$B\r\n <\u001b$B\r\n

\u001b$B99$K!"$I$&$$$&8!:w%(%s%8%s$,$"$k$+D4::$9$k\u001b(B

\r\n \r\n \r\n <\u001b$B>eD9$X$NMW@A;v9`%j%9%H\u001b(B>\r\n <\u001b$B>eD9$X$NMW@A;v9`\u001b(B>\r\n

\u001b$B3+H/$r$9$k$N$O$a$s$I$&$J$N$G!"\u001b(BYahoo!\u001b$B$rGc<}$7$F2<$5$$!#\u001b(B

\r\n eD9$X$NMW@A;v9`\u001b(B>\r\n eD9$X$NMW@A;v9`%j%9%H\u001b(B>\r\n <\u001b$BLdBjE@BP:v\u001b(B>\r\n

\u001b$B8!:w%(%s%8%s$G\r\n \r\n \r\n \r\n\r\n', + ); + expectRejects(input, "XML Parse error: Unsupported encoding 'iso-2022-jp' (supported: UTF-8, UTF-16, ISO-8859-1)"); + }); + + test("weekly-little", () => { + // 4.3.3 [4,84] — Test support for little-endian UTF-16 encoding, and XML names which contain Japanese + // characters. (upstream: valid; external parameter entities are not read) + const input = Buffer.from( + "//48AD8AeABtAGwAIAB2AGUAcgBzAGkAbwBuAD0AIgAxAC4AMAAiAD8APgANAAoAPAAhAEQATwBDAFQAWQBQAEUAIAAxkDFYIABTAFkAUwBUAEUATQAgACIAdwBlAGUAawBsAHkALQB1AHQAZgAtADEANgAuAGQAdABkACIAPgANAAoAPAAhAC0ALQAgADGQMVi1MPMw1zDrMCAALQAtAD4ADQAKADwAMZAxWD4ADQAKACAAIAA8AHReCGcxkD4ADQAKACAAIAAgACAAPAB0XqZePgAxADkAOQA3ADwALwB0XqZePgANAAoAIAAgACAAIAA8AAhnpl4+ADEAPAAvAAhnpl4+AA0ACgAgACAAIAAgADwAMZA+ADEAPAAvADGQPgANAAoAIAAgADwALwB0XghnMZA+AA0ACgANAAoAIAAgADwAD2wNVD4ADQAKACAAIAAgACAAPAAPbD4AcVwwdTwALwAPbD4ADQAKACAAIAAgACAAPAANVD4AKlnOkDwALwANVD4ADQAKACAAIAA8AC8AD2wNVD4ADQAKAA0ACgAgACAAPABtadlSMVhKVOowuTDIMD4ADQAKACAAIAAgACAAPABtadlSMVhKVD4ADQAKACAAIAAgACAAIAAgADwAbWnZUg1UPgBYAE0ATACoMMcwozC/MPwwbjBcTxBiPAAvAG1p2VINVD4ADQAKACAAIAAgACAAIAAgADwAbWnZUrMw/DDJMD4AWAAzADMANQA1AC0AMgAzADwALwBtadlSszD8MMkwPgANAAoAIAAgACAAIAAgACAAPADlXXBloXsGdD4ADQAKACAAIAAgACAAIAAgACAAIAA8AIuJTXqCMIow5V1wZT4AMQA2ADAAMAA8AC8Ai4lNeoIwijDlXXBlPgANAAoAIAAgACAAIAAgACAAIAAgADwAn1s+fuVdcGU+ADMAMgAwADwALwCfWz5+5V1wZT4ADQAKACAAIAAgACAAIAAgACAAIAA8AFNfCGeLiU16gjCKMOVdcGU+ADEANgAwADwALwBTXwhni4lNeoIwijDlXXBlPgANAAoAIAAgACAAIAAgACAAIAAgADwAU18IZ59bPn7lXXBlPgAyADQAPAAvAFNfCGefWz5+5V1wZT4ADQAKACAAIAAgACAAIAAgADwALwDlXXBloXsGdD4ADQAKACAAIAAgACAAIAAgADwAiE6aWwWY7nbqMLkwyDA+AA0ACgAgACAAIAAgACAAIAAgACAAPACITppbBZjudj4ADQAKACAAIAAgACAAIAAgACAAIAAgACAAPABQAD4AWABNAEwAqDDHMKMwvzD8MG4w+lcsZ9VO2GluMFxPEGI8AC8AUAA+AA0ACgAgACAAIAAgACAAIAAgACAAPAAvAIhOmlsFmO52PgANAAoAIAAgACAAIAAgACAAPAAvAIhOmlsFmO526jC5MMgwPgANAAoAIAAgACAAIAAgACAAPACfW71li04FmOowuTDIMD4ADQAKACAAIAAgACAAIAAgACAAIAA8AJ9bvWWLTgWYPgANAAoAIAAgACAAIAAgACAAIAAgACAAIAA8AFAAPgBYAE0ATACoMMcwozC/MPwwbjD6Vyxn1U7YaW4wXE8QYjwALwBQAD4ADQAKACAAIAAgACAAIAAgACAAIAA8AC8An1u9ZYtOBZg+AA0ACgAgACAAIAAgACAAIAAgACAAPACfW71li04FmD4ADQAKACAAIAAgACAAIAAgACAAIAAgACAAPABQAD4A9noIVNZOPnn9iMFUbjBfav2Av4r7ZzwALwBQAD4ADQAKACAAIAAgACAAIAAgACAAIAA8AC8An1u9ZYtOBZg+AA0ACgAgACAAIAAgACAAIAA8AC8An1u9ZYtOBZjqMLkwyDA+AA0ACgAgACAAIAAgACAAIAA8AApOd5V4MG4wgYnLiotOBZjqMLkwyDA+AA0ACgAgACAAIAAgACAAIAAgACAAPAAKTneVeDBuMIGJy4qLTgWYPgANAAoAIAAgACAAIAAgACAAIAAgACAAIAA8AFAAPgB5cmswajBXMDwALwBQAD4ADQAKACAAIAAgACAAIAAgACAAIAA8AC8ACk53lXgwbjCBicuKi04FmD4ADQAKACAAIAAgACAAIAAgADwALwAKTneVeDBuMIGJy4qLTgWY6jC5MMgwPgANAAoAIAAgACAAIAAgACAAPABPVUyYuXD+W1Z7PgANAAoAIAAgACAAIAAgACAAIAAgADwAUAA+AFgATQBMAGgwbzBVT0swjzBLMIkwajBEMAIwPAAvAFAAPgANAAoAIAAgACAAIAAgACAAPAAvAE9VTJi5cP5bVns+AA0ACgAgACAAIAAgADwALwBtadlSMVhKVD4ADQAKAA0ACgAgACAAIAAgADwAbWnZUjFYSlQ+AA0ACgAgACAAIAAgACAAIAA8AG1p2VINVD4AHGkifagw8zC4MPMwbjCLlXp2PAAvAG1p2VINVD4ADQAKACAAIAAgACAAIAAgADwAbWnZUrMw/DDJMD4AUwA4ADgAMgAxAC0ANwA2ADwALwBtadlSszD8MMkwPgANAAoAIAAgACAAIAAgACAAPADlXXBloXsGdD4ADQAKACAAIAAgACAAIAAgACAAIAA8AIuJTXqCMIow5V1wZT4AMQAyADAAPAAvAIuJTXqCMIow5V1wZT4ADQAKACAAIAAgACAAIAAgACAAIAA8AJ9bPn7lXXBlPgA2ADwALwCfWz5+5V1wZT4ADQAKACAAIAAgACAAIAAgACAAIAA8AFNfCGeLiU16gjCKMOVdcGU+ADMAMgA8AC8AU18IZ4uJTXqCMIow5V1wZT4ADQAKACAAIAAgACAAIAAgACAAIAA8AFNfCGefWz5+5V1wZT4AMgA8AC8AU18IZ59bPn7lXXBlPgANAAoAIAAgACAAIAAgACAAPAAvAOVdcGWhewZ0PgANAAoAIAAgACAAIAAgACAAPACITppbBZjuduowuTDIMD4ADQAKACAAIAAgACAAIAAgACAAIAA8AIhOmlsFmO52PgANAAoAIAAgACAAIAAgACAAIAAgACAAIAA8AFAAPgA8AEEAIABoAHIAZQBmAD0AIgBoAHQAdABwADoALwAvAHcAdwB3AC4AZwBvAG8ALgBuAGUALgBqAHAAIgA+AGcAbwBvADwALwBBAD4AbjBfav2AkjC/inkwZjB/MIswPAAvAFAAPgANAAoAIAAgACAAIAAgACAAIAAgADwALwCITppbBZjudj4ADQAKACAAIAAgACAAIAAgADwALwCITppbBZjuduowuTDIMD4ADQAKACAAIAAgACAAIAAgADwAn1u9ZYtOBZjqMLkwyDA+AA0ACgAgACAAIAAgACAAIAAgACAAPACfW71li04FmD4ADQAKACAAIAAgACAAIAAgACAAIAAgACAAPABQAD4A9GZrMAEwaTBGMEQwRjAcaSJ9qDDzMLgw8zBMMEIwizBLML+K+2dZMIswPAAvAFAAPgANAAoAIAAgACAAIAAgACAAIAAgADwALwCfW71li04FmD4ADQAKACAAIAAgACAAIAAgADwALwCfW71li04FmOowuTDIMD4ADQAKACAAIAAgACAAIAAgADwACk53lXgwbjCBicuKi04FmOowuTDIMD4ADQAKACAAIAAgACAAIAAgACAAIAA8AApOd5V4MG4wgYnLiotOBZg+AA0ACgAgACAAIAAgACAAIAAgACAAIAAgADwAUAA+AIuVenaSMFkwizBuMG8wgTCTMGkwRjBqMG4wZzABMFkAYQBoAG8AbwAhAJIwt4zOU1cwZjALTlUwRDACMDwALwBQAD4ADQAKACAAIAAgACAAIAAgACAAIAA8AC8ACk53lXgwbjCBicuKi04FmD4ADQAKACAAIAAgACAAIAAgADwALwAKTneVeDBuMIGJy4qLTgWY6jC5MMgwPgANAAoAIAAgACAAIAAgACAAPABPVUyYuXD+W1Z7PgANAAoAIAAgACAAIAAgACAAIAAgADwAUAA+ABxpIn2oMPMwuDDzMGcwyo6SMHCNiTBbMIswUzBoMEwwZzBNMGowRDACMAj/gYm/ivtnCf88AC8AUAA+AA0ACgAgACAAIAAgACAAIAA8AC8AT1VMmLlw/ltWez4ADQAKACAAIAAgACAAPAAvAG1p2VIxWEpUPgANAAoAIAAgADwALwBtadlSMVhKVOowuTDIMD4ADQAKADwALwAxkDFYPgANAAoA", + "base64", + ); + expectParses(input); + }); + + test("weekly-shift_jis", () => { + // 4.3.3 [4,84] — Test support for Shift_JIS encoding, and XML names which contain Japanese characters. + // If a processor does not support this encoding, it must report a fatal error. (upstream: optional + // error) + const input = Buffer.from( + "PD94bWwgdmVyc2lvbj0iMS4wIiBlbmNvZGluZz0iU2hpZnRfSklTIj8+DQo8IURPQ1RZUEUgj1SV8SBTWVNURU0gIndlZWtseS1zaGlmdF9qaXMuZHRkIj4NCjwhLS0gj1SV8YNUg5ODdoOLIC0tPg0KPI9UlfE+DQogIDyUToyOj1Q+DQogICAgPJROk3g+MTk5NzwvlE6TeD4NCiAgICA8jI6TeD4xPC+MjpN4Pg0KICAgIDyPVD4xPC+PVD4NCiAgPC+UToyOj1Q+DQoNCiAgPI6Blrw+DQogICAgPI6BPo5Sk2M8L46BPg0KICAgIDyWvD6RvphZPC+WvD4NCiAgPC+OgZa8Pg0KDQogIDyLxpaxlfGNkIOKg1iDZz4NCiAgICA8i8aWsZXxjZA+DQogICAgICA8i8aWsZa8PlhNTINHg2aDQoNegVuCzI3skKw8L4vGlrGWvD4NCiAgICAgIDyLxpaxg1KBW4NoPlgzMzU1LTIzPC+Lxpaxg1KBW4NoPg0KICAgICAgPI1IkJSKx5edPg0KICAgICAgICA8jKmQz4LgguiNSJCUPjE2MDA8L4ypkM+C4ILojUiQlD4NCiAgICAgICAgPI7AkNGNSJCUPjMyMDwvjsCQ0Y1IkJQ+DQogICAgICAgIDyTloyOjKmQz4LgguiNSJCUPjE2MDwvk5aMjoypkM+C4ILojUiQlD4NCiAgICAgICAgPJOWjI6OwJDRjUiQlD4yNDwvk5aMjo7AkNGNSJCUPg0KICAgICAgPC+NSJCUiseXnT4NCiAgICAgIDyXXJLojYCW2oOKg1iDZz4NCiAgICAgICAgPJdckuiNgJbaPg0KICAgICAgICAgIDxQPlhNTINHg2aDQoNegVuCzIrulnuOZJdsgsyN7JCsPC9QPg0KICAgICAgICA8L5dckuiNgJbaPg0KICAgICAgPC+XXJLojYCW2oOKg1iDZz4NCiAgICAgIDyOwI57jpaNgIOKg1iDZz4NCiAgICAgICAgPI7AjnuOlo2APg0KICAgICAgICAgIDxQPlhNTINHg2aDQoNegVuCzIrulnuOZJdsgsyN7JCsPC9QPg0KICAgICAgICA8L47AjnuOlo2APg0KICAgICAgICA8jsCOe46WjYA+DQogICAgICAgICAgPFA+i6ONh5G8jtCQu5VpgsyLQJRckrKNuDwvUD4NCiAgICAgICAgPC+OwI57jpaNgD4NCiAgICAgIDwvjsCOe46WjYCDioNYg2c+DQogICAgICA8j+OSt4LWgsyXdpC/jpaNgIOKg1iDZz4NCiAgICAgICAgPI/jkreC1oLMl3aQv46WjYA+DQogICAgICAgICAgPFA+k8GCyYLIgrU8L1A+DQogICAgICAgIDwvj+OSt4LWgsyXdpC/jpaNgD4NCiAgICAgIDwvj+OSt4LWgsyXdpC/jpaNgIOKg1iDZz4NCiAgICAgIDyW4pHok1+Rzo30Pg0KICAgICAgICA8UD5YTUyCxoLNib2CqYLtgqmC54LIgqKBQjwvUD4NCiAgICAgIDwvluKR6JNfkc6N9D4NCiAgICA8L4vGlrGV8Y2QPg0KDQogICAgPIvGlrGV8Y2QPg0KICAgICAgPIvGlrGWvD6Mn431g0eDk4NXg5OCzIpKlK08L4vGlrGWvD4NCiAgICAgIDyLxpaxg1KBW4NoPlM4ODIxLTc2PC+Lxpaxg1KBW4NoPg0KICAgICAgPI1IkJSKx5edPg0KICAgICAgICA8jKmQz4LgguiNSJCUPjEyMDwvjKmQz4LgguiNSJCUPg0KICAgICAgICA8jsCQ0Y1IkJQ+NjwvjsCQ0Y1IkJQ+DQogICAgICAgIDyTloyOjKmQz4LgguiNSJCUPjMyPC+TloyOjKmQz4LgguiNSJCUPg0KICAgICAgICA8k5aMjo7AkNGNSJCUPjI8L5OWjI6OwJDRjUiQlD4NCiAgICAgIDwvjUiQlIrHl50+DQogICAgICA8l1yS6I2AltqDioNYg2c+DQogICAgICAgIDyXXJLojYCW2j4NCiAgICAgICAgICA8UD48QSBocmVmPSJodHRwOi8vd3d3Lmdvby5uZS5qcCI+Z29vPC9BPoLMi0CUXILwkrKC14LEgt2C6TwvUD4NCiAgICAgICAgPC+XXJLojYCW2j4NCiAgICAgIDwvl1yS6I2AltqDioNYg2c+DQogICAgICA8jsCOe46WjYCDioNYg2c+DQogICAgICAgIDyOwI57jpaNgD4NCiAgICAgICAgICA8UD6NWILJgUGCx4KkgqKCpIyfjfWDR4OTg1eDk4KqgqCC6YKpkrKNuIK3guk8L1A+DQogICAgICAgIDwvjsCOe46WjYA+DQogICAgICA8L47AjnuOlo2Ag4qDWINnPg0KICAgICAgPI/jkreC1oLMl3aQv46WjYCDioNYg2c+DQogICAgICAgIDyP45K3gtaCzJd2kL+Olo2APg0KICAgICAgICAgIDxQPopKlK2C8IK3gumCzILNgt+C8YLHgqSCyILMgsWBQVlhaG9vIYLwlIOO+4K1gsSJuoKzgqKBQjwvUD4NCiAgICAgICAgPC+P45K3gtaCzJd2kL+Olo2APg0KICAgICAgPC+P45K3gtaCzJd2kL+Olo2Ag4qDWINnPg0KICAgICAgPJbikeiTX5HOjfQ+DQogICAgICAgIDxQPoyfjfWDR4OTg1eDk4LFjtSC8JGWgueCuYLpgrGCxoKqgsWCq4LIgqKBQoFpl3aSso24gWo8L1A+DQogICAgICA8L5bikeiTX5HOjfQ+DQogICAgPC+LxpaxlfGNkD4NCiAgPC+LxpaxlfGNkIOKg1iDZz4NCjwvj1SV8T4NCg==", + "base64", + ); + expectRejects(input, "XML Parse error: Unsupported encoding 'Shift_JIS' (supported: UTF-8, UTF-16, ISO-8859-1)"); + }); + + test("weekly-utf-16", () => { + // 4.3.3 [4,84] — Test support for UTF-16 encoding, and XML names which contain Japanese characters. + // (upstream: valid; external parameter entities are not read) + const input = Buffer.from( + "/v8APAA/AHgAbQBsACAAdgBlAHIAcwBpAG8AbgA9ACIAMQAuADAAIgA/AD4ADQAKADwAIQBEAE8AQwBUAFkAUABFACCQMVgxACAAUwBZAFMAVABFAE0AIAAiAHcAZQBlAGsAbAB5AC0AdQB0AGYALQAxADYALgBkAHQAZAAiAD4ADQAKADwAIQAtAC0AIJAxWDEwtTDzMNcw6wAgAC0ALQA+AA0ACgA8kDFYMQA+AA0ACgAgACAAPF50ZwiQMQA+AA0ACgAgACAAIAAgADxedF6mAD4AMQA5ADkANwA8AC9edF6mAD4ADQAKACAAIAAgACAAPGcIXqYAPgAxADwAL2cIXqYAPgANAAoAIAAgACAAIAA8kDEAPgAxADwAL5AxAD4ADQAKACAAIAA8AC9edGcIkDEAPgANAAoADQAKACAAIAA8bA9UDQA+AA0ACgAgACAAIAAgADxsDwA+XHF1MAA8AC9sDwA+AA0ACgAgACAAIAAgADxUDQA+WSqQzgA8AC9UDQA+AA0ACgAgACAAPAAvbA9UDQA+AA0ACgANAAoAIAAgADxpbVLZWDFUSjDqMLkwyAA+AA0ACgAgACAAIAAgADxpbVLZWDFUSgA+AA0ACgAgACAAIAAgACAAIAA8aW1S2VQNAD4AWABNAEwwqDDHMKMwvzD8MG5PXGIQADwAL2ltUtlUDQA+AA0ACgAgACAAIAAgACAAIAA8aW1S2TCzMPwwyQA+AFgAMwAzADUANQAtADIAMwA8AC9pbVLZMLMw/DDJAD4ADQAKACAAIAAgACAAIAAgADxd5WVwe6F0BgA+AA0ACgAgACAAIAAgACAAIAAgACAAPImLek0wgjCKXeVlcAA+ADEANgAwADAAPAAviYt6TTCCMIpd5WVwAD4ADQAKACAAIAAgACAAIAAgACAAIAA8W59+Pl3lZXAAPgAzADIAMAA8AC9bn34+XeVlcAA+AA0ACgAgACAAIAAgACAAIAAgACAAPF9TZwiJi3pNMIIwil3lZXAAPgAxADYAMAA8AC9fU2cIiYt6TTCCMIpd5WVwAD4ADQAKACAAIAAgACAAIAAgACAAIAA8X1NnCFuffj5d5WVwAD4AMgA0ADwAL19TZwhbn34+XeVlcAA+AA0ACgAgACAAIAAgACAAIAA8AC9d5WVwe6F0BgA+AA0ACgAgACAAIAAgACAAIAA8TohbmpgFdu4w6jC5MMgAPgANAAoAIAAgACAAIAAgACAAIAAgADxOiFuamAV27gA+AA0ACgAgACAAIAAgACAAIAAgACAAIAAgADwAUAA+AFgATQBMMKgwxzCjML8w/DBuV/pnLE7Vadgwbk9cYhAAPAAvAFAAPgANAAoAIAAgACAAIAAgACAAIAAgADwAL06IW5qYBXbuAD4ADQAKACAAIAAgACAAIAAgADwAL06IW5qYBXbuMOowuTDIAD4ADQAKACAAIAAgACAAIAAgADxbn2W9TouYBTDqMLkwyAA+AA0ACgAgACAAIAAgACAAIAAgACAAPFufZb1Oi5gFAD4ADQAKACAAIAAgACAAIAAgACAAIAAgACAAPABQAD4AWABNAEwwqDDHMKMwvzD8MG5X+mcsTtVp2DBuT1xiEAA8AC8AUAA+AA0ACgAgACAAIAAgACAAIAAgACAAPAAvW59lvU6LmAUAPgANAAoAIAAgACAAIAAgACAAIAAgADxbn2W9TouYBQA+AA0ACgAgACAAIAAgACAAIAAgACAAIAAgADwAUAA+evZUCE7WeT6I/VTBMG5qX4D9ir9n+wA8AC8AUAA+AA0ACgAgACAAIAAgACAAIAAgACAAPAAvW59lvU6LmAUAPgANAAoAIAAgACAAIAAgACAAPAAvW59lvU6LmAUw6jC5MMgAPgANAAoAIAAgACAAIAAgACAAPE4KlXcweDBuiYGKy06LmAUw6jC5MMgAPgANAAoAIAAgACAAIAAgACAAIAAgADxOCpV3MHgwbomBistOi5gFAD4ADQAKACAAIAAgACAAIAAgACAAIAAgACAAPABQAD5yeTBrMGowVwA8AC8AUAA+AA0ACgAgACAAIAAgACAAIAAgACAAPAAvTgqVdzB4MG6JgYrLTouYBQA+AA0ACgAgACAAIAAgACAAIAA8AC9OCpV3MHgwbomBistOi5gFMOowuTDIAD4ADQAKACAAIAAgACAAIAAgADxVT5hMcLlb/ntWAD4ADQAKACAAIAAgACAAIAAgACAAIAA8AFAAPgBYAE0ATDBoMG9PVTBLMI8wSzCJMGowRDACADwALwBQAD4ADQAKACAAIAAgACAAIAAgADwAL1VPmExwuVv+e1YAPgANAAoAIAAgACAAIAA8AC9pbVLZWDFUSgA+AA0ACgANAAoAIAAgACAAIAA8aW1S2VgxVEoAPgANAAoAIAAgACAAIAAgACAAPGltUtlUDQA+aRx9IjCoMPMwuDDzMG6Vi3Z6ADwAL2ltUtlUDQA+AA0ACgAgACAAIAAgACAAIAA8aW1S2TCzMPwwyQA+AFMAOAA4ADIAMQAtADcANgA8AC9pbVLZMLMw/DDJAD4ADQAKACAAIAAgACAAIAAgADxd5WVwe6F0BgA+AA0ACgAgACAAIAAgACAAIAAgACAAPImLek0wgjCKXeVlcAA+ADEAMgAwADwAL4mLek0wgjCKXeVlcAA+AA0ACgAgACAAIAAgACAAIAAgACAAPFuffj5d5WVwAD4ANgA8AC9bn34+XeVlcAA+AA0ACgAgACAAIAAgACAAIAAgACAAPF9TZwiJi3pNMIIwil3lZXAAPgAzADIAPAAvX1NnCImLek0wgjCKXeVlcAA+AA0ACgAgACAAIAAgACAAIAAgACAAPF9TZwhbn34+XeVlcAA+ADIAPAAvX1NnCFuffj5d5WVwAD4ADQAKACAAIAAgACAAIAAgADwAL13lZXB7oXQGAD4ADQAKACAAIAAgACAAIAAgADxOiFuamAV27jDqMLkwyAA+AA0ACgAgACAAIAAgACAAIAAgACAAPE6IW5qYBXbuAD4ADQAKACAAIAAgACAAIAAgACAAIAAgACAAPABQAD4APABBACAAaAByAGUAZgA9ACIAaAB0AHQAcAA6AC8ALwB3AHcAdwAuAGcAbwBvAC4AbgBlAC4AagBwACIAPgBnAG8AbwA8AC8AQQA+MG5qX4D9MJKKvzB5MGYwfzCLADwALwBQAD4ADQAKACAAIAAgACAAIAAgACAAIAA8AC9OiFuamAV27gA+AA0ACgAgACAAIAAgACAAIAA8AC9OiFuamAV27jDqMLkwyAA+AA0ACgAgACAAIAAgACAAIAA8W59lvU6LmAUw6jC5MMgAPgANAAoAIAAgACAAIAAgACAAIAAgADxbn2W9TouYBQA+AA0ACgAgACAAIAAgACAAIAAgACAAIAAgADwAUAA+ZvQwazABMGkwRjBEMEZpHH0iMKgw8zC4MPMwTDBCMIswS4q/Z/swWTCLADwALwBQAD4ADQAKACAAIAAgACAAIAAgACAAIAA8AC9bn2W9TouYBQA+AA0ACgAgACAAIAAgACAAIAA8AC9bn2W9TouYBTDqMLkwyAA+AA0ACgAgACAAIAAgACAAIAA8TgqVdzB4MG6JgYrLTouYBTDqMLkwyAA+AA0ACgAgACAAIAAgACAAIAAgACAAPE4KlXcweDBuiYGKy06LmAUAPgANAAoAIAAgACAAIAAgACAAIAAgACAAIAA8AFAAPpWLdnowkjBZMIswbjBvMIEwkzBpMEYwajBuMGcwAQBZAGEAaABvAG8AITCSjLdTzjBXMGZOCzBVMEQwAgA8AC8AUAA+AA0ACgAgACAAIAAgACAAIAAgACAAPAAvTgqVdzB4MG6JgYrLTouYBQA+AA0ACgAgACAAIAAgACAAIAA8AC9OCpV3MHgwbomBistOi5gFMOowuTDIAD4ADQAKACAAIAAgACAAIAAgADxVT5hMcLlb/ntWAD4ADQAKACAAIAAgACAAIAAgACAAIAA8AFAAPmkcfSIwqDDzMLgw8zBnjsowko1wMIkwWzCLMFMwaDBMMGcwTTBqMEQwAv8IiYGKv2f7/wkAPAAvAFAAPgANAAoAIAAgACAAIAAgACAAPAAvVU+YTHC5W/57VgA+AA0ACgAgACAAIAAgADwAL2ltUtlYMVRKAD4ADQAKACAAIAA8AC9pbVLZWDFUSjDqMLkwyAA+AA0ACgA8AC+QMVgxAD4ADQAK", + "base64", + ); + expectParses(input); + }); + + test("weekly-utf-8", () => { + // 4.3.3 [4,84] — Test support for UTF-8 encoding and XML names which contain Japanese characters. + // (upstream: valid; external parameter entities are not read) + const input: string = + '\r\n\r\n\r\n<週報>\r\n <年月週>\r\n <年度>1997\r\n <月度>1\r\n <週>1\r\n \r\n\r\n <氏名>\r\n <氏>山田\r\n <名>太郎\r\n \r\n\r\n <業務報告リスト>\r\n <業務報告>\r\n <業務名>XMLエディターの作成\r\n <業務コード>X3355-23\r\n <工数管理>\r\n <見積もり工数>1600\r\n <実績工数>320\r\n <当月見積もり工数>160\r\n <当月実績工数>24\r\n \r\n <予定項目リスト>\r\n <予定項目>\r\n

XMLエディターの基本仕様の作成

\r\n \r\n \r\n <実施事項リスト>\r\n <実施事項>\r\n

XMLエディターの基本仕様の作成

\r\n \r\n <実施事項>\r\n

競合他社製品の機能調査

\r\n \r\n \r\n <上長への要請事項リスト>\r\n <上長への要請事項>\r\n

特になし

\r\n \r\n \r\n <問題点対策>\r\n

XMLとは何かわからない。

\r\n \r\n \r\n\r\n <業務報告>\r\n <業務名>検索エンジンの開発\r\n <業務コード>S8821-76\r\n <工数管理>\r\n <見積もり工数>120\r\n <実績工数>6\r\n <当月見積もり工数>32\r\n <当月実績工数>2\r\n \r\n <予定項目リスト>\r\n <予定項目>\r\n

gooの機能を調べてみる

\r\n \r\n \r\n <実施事項リスト>\r\n <実施事項>\r\n

更に、どういう検索エンジンがあるか調査する

\r\n \r\n \r\n <上長への要請事項リスト>\r\n <上長への要請事項>\r\n

開発をするのはめんどうなので、Yahoo!を買収して下さい。

\r\n \r\n \r\n <問題点対策>\r\n

検索エンジンで車を走らせることができない。(要調査)

\r\n \r\n \r\n \r\n\r\n'; + expectParses(input); + }); +}); + +describe("sun", () => { + test("pe01", () => { + // 2.8 — Parameter entities references are NOT RECOGNIZED in default attribute values. (upstream: + // valid; external parameter entities are not read) + const input: string = '\n\n'; + expectParses(input); + }); + + test("dtd00", () => { + // 3.2.2 [51] — Tests parsing of alternative forms of text-only mixed content declaration. + const input: string = + "\n \n \n]>\n\n\n"; + const canonical = ""; + const compact: unknown = { root: "" }; + expectParses(input, canonical, compact); + }); + + test("dtd01", () => { + // 2.5 [15] — Comments don't get parameter entity expansion + const input: string = + '\n \n \n]>\n\n'; + const canonical = ""; + const compact: unknown = { root: "" }; + expectParses(input, canonical, compact); + }); + + test("element", () => { + // 3 — Tests clauses 1, 3, and 4 of the Element Valid validity constraint. + const input: string = + "\n\n\n\n\n]>\n\n\n \n\n \n \n\n \n \n\n \n \n\n allowed\n ]]>\n\n also\n ]]>\n\n moreover\n\n allowed & stuff\n\n also\n\n moreover \n moreover \n \n too\n\n\n"; + const canonical = + " allowed <allowed> also <% illegal otherwise %> moreover allowed & stuff also moreover moreover too "; + const compact: unknown = { + root: { + empty: "", + mixed1: ["", "", "allowed", "", "allowed & stuff"], + mixed2: ["", "", "also", "<% illegal otherwise %>", "also"], + mixed3: [ + "", + "", + "moreover", + { empty: "", "#text": "moreover" }, + { empty: "", "#text": "moreover" }, + { empty: "" }, + { empty: "", "#text": "too" }, + ], + }, + }; + expectParses(input, canonical, compact); + }); + + test("ext01", () => { + // 4.3.1 4.3.2 [77] [78] — Tests use of external parsed entities with and without content. (upstream: + // valid; external general entities are not read; output depends on them) + const input: string = + '\n\n\n\n\n\n]>\n &root; &root; &null; &null; \n'; + expectParses(input); + }); + + test("ext02", () => { + // 4.3.2 [78] — Tests use of external parsed entities with different encodings than the base document. + // (upstream: valid; external general entities are not read; output depends on them) + const input: string = + '\n\n\n\n]>\n\n &utf16b; &utf16l; \n'; + expectParses(input); + }); + + test("not-sa01", () => { + // 2.9 — A non-standalone document is valid if declared as such. (upstream: valid; external parameter + // entities are not read) + const input: string = + "\n\n\n\n\n \n The whitespace before and after this element keeps\n this from being standalone.\n \n\n"; + const canonical = + " The whitespace before and after this element keeps this from being standalone. "; + const compact: unknown = { + root: { + child: "The whitespace before and after this element keeps\n this from being standalone.", + }, + }; + expectParses(input, canonical, compact); + }); + + test("not-sa02", () => { + // 2.9 — A non-standalone document is valid if declared as such. (upstream: valid; external parameter + // entities are not read; output depends on them) + const input: string = + '\n\n\n]>\n\n \n\n \n\n\n'; + expectParses(input); + }); + + test("not-sa03", () => { + // 2.9 — A non-standalone document is valid if declared as such. (upstream: valid; external parameter + // entities are not read; output depends on them) + const input: string = + '\n\n\n]>\n\n\n'; + expectParses(input); + }); + + test("not-sa04", () => { + // 2.9 — A non-standalone document is valid if declared as such. (upstream: valid; external parameter + // entities are not read) + const input: string = + '\n\n\n\n \n \n \n]>\n\n\n\n\n'; + const canonical = + ''; + const compact: unknown = { + attributes: { + "@cdata": "nothing happens to this one!", + "@entities": "unparsed-1 unparsed-2", + "@entity": "unparsed-1", + "@id": "internal42", + "@idref": "internal42", + "@idrefs": "internal42 internal42 internal42", + "@nmtoken": "this-gets-normalized", + "@nmtokens": "this also gets normalized", + "@notation": "nonce", + "@token": "a", + }, + }; + expectParses(input, canonical, compact); + }); + + test("notation01", () => { + // 4.7 [82] — NOTATION declarations don't need SYSTEM IDs; and externally declared notations may be + // used to declare unparsed entities in the internal DTD subset. The notation must be reported to the + // application. (upstream: valid; external parameter entities are not read) + const input: string = + '\n\n]>\ntest\n'; + const canonical = "test"; + const compact: unknown = { test: "test" }; + expectParses(input, canonical, compact); + }); + + test("optional", () => { + // 3 3.2.1 [47] — Tests declarations of "children" content models, and the validity constraints + // associated with them. (upstream: valid; external parameter entities are not read) + const input: string = + '\n\n \n \n\n \n\n\n \n \n \n \n \n\n \n \n \n \n \n\n\n \n \n \n \n \n\n \n \n \n \n \n\n \n \n \n \n \n\n \n \n \n \n \n\n\n\n'; + const canonical = + " "; + const compact: unknown = { + root: { + once: { e: "" }, + twice: { e: ["", ""] }, + "once-or-twice-a": [{ e: "" }, { e: ["", ""] }], + "once-or-twice-b": [{ e: "" }, { e: ["", ""] }], + "once-or-twice-c": [{ e: "" }, { e: ["", ""] }], + "once-or-twice-d": [{ e: "" }, { e: ["", ""] }], + "once-or-twice-e": [{ e: "" }, { e: ["", ""] }], + "once-or-more-a": [{ e: "" }, { e: ["", ""] }, { e: ["", "", ""] }, { e: ["", "", "", ""] }], + "once-or-more-b": [{ e: "" }, { e: ["", ""] }, { e: ["", "", ""] }, { e: ["", "", "", ""] }], + "once-or-more-c": [{ e: "" }, { e: ["", ""] }, { e: ["", "", ""] }, { e: ["", "", "", ""] }], + "once-or-more-d": [{ e: "" }, { e: ["", ""] }, { e: ["", "", ""] }, { e: ["", "", "", ""] }], + "once-or-more-e": [{ e: "" }, { e: ["", ""] }, { e: ["", "", ""] }, { e: ["", "", "", ""] }], + }, + }; + expectParses(input, canonical, compact); + }); + + test("required00", () => { + // 3.3.2 [60] — Tests the #REQUIRED attribute declaration syntax, and the associated validity + // constraint. + const input: string = + '\n \n]>\n\n\n'; + const canonical = ''; + const compact: unknown = { root: { "@req": "foo" } }; + expectParses(input, canonical, compact); + }); + + test("sa01", () => { + // 2.9 [32] — A document may be marked 'standalone' if any optional whitespace is defined within the + // internal DTD subset. + const input: string = + "\n\n\n \n]>\n\n\n \n The whitespace around this element would be\n invalid as standalone were the DTD external.\n \n\n"; + const canonical = + " The whitespace around this element would be invalid as standalone were the DTD external. "; + const compact: unknown = { + root: { + child: "The whitespace around this element would be\n invalid as standalone were the DTD external.", + }, + }; + expectParses(input, canonical, compact); + }); + + test("sa02", () => { + // 2.9 [32] — A document may be marked 'standalone' if any attributes that need normalization are + // defined within the internal DTD subset. + const input: string = + '\n\n\n\n \n\n \n \n \n \n\n \n \n \n\n \n \n]>\n\n\n'; + const canonical = + ''; + const compact: unknown = { + attributes: { + "@cdata": "nothing happens to this one!", + "@entities": "unparsed-1 unparsed-2", + "@entity": "unparsed-1", + "@id": "internal42", + "@idref": "internal42", + "@idrefs": "internal42 internal42 internal42", + "@nmtoken": "this-gets-normalized", + "@nmtokens": "this also gets normalized", + "@notation": "nonce", + "@token": "a", + }, + }; + expectParses(input, canonical, compact); + }); + + test("sa03", () => { + // 2.9 [32] — A document may be marked 'standalone' if any the defined entities need expanding are + // internal, and no attributes need defaulting or normalization. On output, requires notations to be + // correctly reported. (upstream: valid; external parameter entities are not read) + const input: string = + '\n\n\n \n \n]>\n\n\n'; + const canonical = + ''; + const compact: unknown = { + attributes: { + "@cdata": "nothing happens to this one!", + "@entities": "unparsed-1 unparsed-2", + "@entity": "unparsed-1", + "@id": "internal42", + "@idref": "internal42", + "@idrefs": "internal42 internal42 internal42", + "@nmtoken": "this-gets-normalized", + "@nmtokens": "this also gets normalized", + "@notation": "foo", + "@token": "b", + }, + }; + expectParses(input, canonical, compact); + }); + + test("sa04", () => { + // 2.9 [32] — Like sa03 but relies on attribute defaulting defined in the internal subset. On output, + // requires notations to be correctly reported. (upstream: valid; external parameter entities are not + // read) + const input: string = + '\n\n\n\n \n \n \n]>\n\n\n\n\n'; + const canonical = + ''; + const compact: unknown = { + attributes: { + "@cdata": "nothing happens to this one!", + "@entities": "unparsed-1 unparsed-2", + "@entity": "unparsed-1", + "@id": "internal42", + "@idref": "internal42", + "@idrefs": "internal42 internal42 internal42", + "@nmtoken": "this-gets-normalized", + "@nmtokens": "this also gets normalized", + "@notation": "nonce", + "@token": "a", + }, + }; + expectParses(input, canonical, compact); + }); + + test("sa05", () => { + // 2.9 [32] — Like sa01 but this document is standalone since it has no optional whitespace. On output, + // requires notations to be correctly reported. (upstream: valid; external parameter entities are not + // read) + const input: string = + "\n\n\n\n\n No whitespace before or after this standalone element.\n\n"; + const canonical = + " No whitespace before or after this standalone element. "; + const compact: unknown = { root: { child: "No whitespace before or after this standalone element." } }; + expectParses(input, canonical, compact); + }); + + test("v-sgml01", () => { + // 3.3.1 [59] — XML permits token reuse, while SGML does not. + const input: string = + '\n \n \n]>\n\n\n'; + const canonical = ''; + const compact: unknown = { root: { "@position": "first", "@status": "initial-draft" } }; + expectParses(input, canonical, compact); + }); + + test("v-lang01", () => { + // 2.12 [35] — Tests a lowercase ISO language code. + const input: string = + '\n\n]>\n\n'; + const canonical = ''; + const compact: unknown = { root: { "@xml:lang": "en" } }; + expectParses(input, canonical, compact); + }); + + test("v-lang02", () => { + // 2.12 [35] — Tests a ISO language code with a subcode. + const input: string = + '\n\n]>\n\n\n'; + const canonical = ''; + const compact: unknown = { root: { "@xml:lang": "en-IN" } }; + expectParses(input, canonical, compact); + }); + + test("v-lang03", () => { + // 2.12 [36] — Tests a IANA language code with a subcode. + const input: string = + '\n\n]>\n\n\n'; + const canonical = ''; + const compact: unknown = { root: { "@xml:lang": "i-klingon-whorf" } }; + expectParses(input, canonical, compact); + }); + + test("v-lang04", () => { + // 2.12 [37] — Tests a user language code with a subcode. + const input: string = + '\n\n]>\n\n\n'; + const canonical = ''; + const compact: unknown = { root: { "@xml:lang": "x-dialect-valleygirl" } }; + expectParses(input, canonical, compact); + }); + + test("v-lang05", () => { + // 2.12 [35] — Tests an uppercase ISO language code. + const input: string = + '\n\n]>\n\n\n'; + const canonical = ''; + const compact: unknown = { root: { "@xml:lang": "DE" } }; + expectParses(input, canonical, compact); + }); + + test("v-lang06", () => { + // 2.12 [37] — Tests a user language code. + const input: string = + '\n\n]>\n\n\n'; + const canonical = ''; + const compact: unknown = { root: { "@xml:lang": "X-Java" } }; + expectParses(input, canonical, compact); + }); + + test("v-pe00", () => { + // 4.5 — Tests construction of internal entity replacement text, using an example in the XML + // specification. (upstream: valid; external parameter entities are not read; output depends on them) + const input: string = '\n&book;\n'; + expectParses(input); + }); + + test("v-pe03", () => { + // 4.5 — Tests construction of internal entity replacement text, using an example in the XML + // specification. + const input: string = + '\n\n\nAn ampersand (&#38;) may be escaped\nnumerically (&#38;#38) or with a general entity (&amp;).

" >\n]>\n&example;\n'; + const canonical = + "

An ampersand (&) may be escaped numerically (&#38) or with a general entity (&amp;).

"; + const compact: unknown = { + root: { + p: "An ampersand (&) may be escaped\nnumerically (&) or with a general entity (&).", + }, + }; + expectParses(input, canonical, compact); + }); + + test("v-pe02", () => { + // 4.5 — Tests construction of internal entity replacement text, using a complex example in the XML + // specification. (upstream: valid; external parameter entities are not read) + const input: string = + "\n\n\n' >\n%xx;\n]>\nThis sample shows a &tricky; method.\n\n"; + const canonical = "This sample shows a error-prone method."; + const compact: unknown = { test: "This sample shows a error-prone method." }; + expectParses(input, canonical, compact); + }); + + test("inv-dtd01", () => { + // 3.2.2 — Tests the No Duplicate Types VC + const input: string = + "\n \n \n]>\n\n\n"; + expectParses(input); + }); + + test("inv-dtd02", () => { + // 4.2.2 — Tests the "Notation Declared" VC by using an undeclared notation name. + const input: string = + '\n\n]>\n\n'; + expectParses(input); + }); + + test("inv-dtd03", () => { + // 3 — Tests the "Element Valid" VC (clause 2) by omitting a required element. + const input: string = + '\n\n\n \n]>\n\n \n \n \n\n'; + expectParses(input); + }); + + test("el01", () => { + // 3 — Tests the Element Valid VC (clause 4) by including an undeclared child element. + const input: string = "\n]>\n \n\n"; + expectParses(input); + }); + + test("el02", () => { + // 3 — Tests the Element Valid VC (clause 1) by including elements in an EMPTY content model. + const input: string = "\n]>\n\n"; + expectParses(input); + }); + + test("el03", () => { + // 3 — Tests the Element Valid VC (clause 3) by including a child element not permitted by a mixed + // content model. + const input: string = + "\n\n]>\nthis is ok this isn't \n"; + expectParses(input); + }); + + test("el04", () => { + // 3.2 — Tests the Unique Element Type Declaration VC. + const input: string = + "\n\n\n]>\n\n"; + expectParses(input); + }); + + test("el05", () => { + // 3.2.2 — Tests the No Duplicate Types VC. + const input: string = + "\n\n]>\n\n"; + expectParses(input); + }); + + test("el06", () => { + // 3 — Tests the Element Valid VC (clause 1), using one of the predefined internal entities inside an + // EMPTY content model. + const input: string = + "\n \n]>\n&\n\n"; + expectParses(input); + }); + + test("id01", () => { + // 3.3.1 — Tests the ID (is a Name) VC (upstream: invalid; external parameter entities are not read) + const input: string = + '\n\n\n\n\n \n\n'; + expectParses(input); + }); + + test("id02", () => { + // 3.3.1 — Tests the ID (appears once) VC (upstream: invalid; external parameter entities are not read) + const input: string = + '\n\n\n\n\n \n \n\n\n'; + expectParses(input); + }); + + test("id03", () => { + // 3.3.1 — Tests the One ID per Element Type VC (upstream: invalid; external parameter entities are not + // read) + const input: string = + '\n]>\n\n\n\n\n\n'; + expectParses(input); + }); + + test("id04", () => { + // 3.3.1 — Tests the ID Attribute Default VC + const input: string = + '\n]>\n\n\n\n\n\n'; + expectParses(input); + }); + + test("id05", () => { + // 3.3.1 — Tests the ID Attribute Default VC + const input: string = + '\n \n]>\n\n\n\n\n\n\n'; + expectParses(input); + }); + + test("id06", () => { + // 3.3.1 — Tests the IDREF (is a Name) VC + const input: string = + '\n \n]>\n\n\n\n\n\n\n\n'; + expectParses(input); + }); + + test("id07", () => { + // 3.3.1 — Tests the IDREFS (is a Names) VC + const input: string = + '\n \n]>\n\n\n\n\n\n\n\n\n'; + expectParses(input); + }); + + test("id08", () => { + // 3.3.1 — Tests the IDREF (matches an ID) VC + const input: string = + '\n \n]>\n\n\n\n\n\n\n\n'; + expectParses(input); + }); + + test("id09", () => { + // 3.3.1 — Tests the IDREF (IDREFS matches an ID) VC + const input: string = + '\n \n]>\n\n\n\n\n \n\n\n\n\n'; + expectParses(input); + }); + + test("inv-not-sa01", () => { + // 2.9 — Tests the Standalone Document Declaration VC, ensuring that optional whitespace causes a + // validity error. (upstream: invalid; external parameter entities are not read) + const input: string = + "\n\n\n \n\n \n The whitespace before and after this element keeps\n this from being standalone.\n \n\n"; + expectParses(input); + }); + + test("inv-not-sa02", () => { + // 2.9 — Tests the Standalone Document Declaration VC, ensuring that attributes needing normalization + // cause a validity error. (upstream: invalid; external parameter entities are not read) + const input: string = + '\n\n\n]>\n\n \n\n \n\n\n'; + expectParses(input); + }); + + test("inv-not-sa04", () => { + // 2.9 — Tests the Standalone Document Declaration VC, ensuring that attributes needing defaulting + // cause a validity error. (upstream: invalid; external parameter entities are not read) + const input: string = + "\n\n\n]>\n\n\n\n\n"; + expectParses(input); + }); + + test("inv-not-sa05", () => { + // 2.9 — Tests the Standalone Document Declaration VC, ensuring that a token attribute that needs + // normalization causes a validity error. (upstream: invalid; external parameter entities are not read) + const input: string = + "\n\n\n]>\n\n\n"; + expectParses(input); + }); + + test("inv-not-sa06", () => { + // 2.9 — Tests the Standalone Document Declaration VC, ensuring that a NOTATION attribute that needs + // normalization causes a validity error. (upstream: invalid; external parameter entities are not read) + const input: string = + '\n\n\n]>\n\n\n\n'; + expectParses(input); + }); + + test("inv-not-sa07", () => { + // 2.9 — Tests the Standalone Document Declaration VC, ensuring that an NMTOKEN attribute needing + // normalization causes a validity error. (upstream: invalid; external parameter entities are not read) + const input: string = + '\n\n\n]>\n\n\n'; + expectParses(input); + }); + + test("inv-not-sa08", () => { + // 2.9 — Tests the Standalone Document Declaration VC, ensuring that an NMTOKENS attribute needing + // normalization causes a validity error. (upstream: invalid; external parameter entities are not read) + const input: string = + '\n\n\n]>\n\n\n'; + expectParses(input); + }); + + test("inv-not-sa09", () => { + // 2.9 — Tests the Standalone Document Declaration VC, ensuring that an ID attribute needing + // normalization causes a validity error. (upstream: invalid; external parameter entities are not read) + const input: string = + '\n\n\n]>\n\n\n'; + expectParses(input); + }); + + test("inv-not-sa10", () => { + // 2.9 — Tests the Standalone Document Declaration VC, ensuring that an IDREF attribute needing + // normalization causes a validity error. (upstream: invalid; external parameter entities are not read) + const input: string = + '\n\n\n]>\n\n\n'; + expectParses(input); + }); + + test("inv-not-sa11", () => { + // 2.9 — Tests the Standalone Document Declaration VC, ensuring that an IDREFS attribute needing + // normalization causes a validity error. (upstream: invalid; external parameter entities are not read) + const input: string = + '\n\n\n]>\n\n\n'; + expectParses(input); + }); + + test("inv-not-sa12", () => { + // 2.9 — Tests the Standalone Document Declaration VC, ensuring that an ENTITY attribute needing + // normalization causes a validity error. (upstream: invalid; external parameter entities are not read) + const input: string = + '\n\n\n]>\n\n\n'; + expectParses(input); + }); + + test("inv-not-sa13", () => { + // 2.9 — Tests the Standalone Document Declaration VC, ensuring that an ENTITIES attribute needing + // normalization causes a validity error. (upstream: invalid; external parameter entities are not read) + const input: string = + '\n\n\n]>\n\n\n'; + expectParses(input); + }); + + test("inv-not-sa14", () => { + // 3 — CDATA sections containing only whitespace do not match the nonterminal S, and cannot appear in + // these positions. (upstream: invalid; external parameter entities are not read) + const input: string = + "\n\n\n \n\n The whitespace before and after this element keeps\n this from being standalone. (CDATA is just another\n way to represent text...)\n \n"; + expectParses(input); + }); + + test("optional01", () => { + // 3 — Tests the Element Valid VC (clause 2) for one instance of "children" content model, providing no + // children where one is required. (upstream: invalid; external parameter entities are not read) + const input: string = '\n\n \n\n'; + expectParses(input); + }); + + test("optional02", () => { + // 3 — Tests the Element Valid VC (clause 2) for one instance of "children" content model, providing + // two children where one is required. (upstream: invalid; external parameter entities are not read) + const input: string = + '\n\n \n\n\n'; + expectParses(input); + }); + + test("optional03", () => { + // 3 — Tests the Element Valid VC (clause 2) for one instance of "children" content model, providing no + // children where two are required. (upstream: invalid; external parameter entities are not read) + const input: string = '\n\n \n\n\n'; + expectParses(input); + }); + + test("optional04", () => { + // 3 — Tests the Element Valid VC (clause 2) for one instance of "children" content model, providing + // three children where two are required. (upstream: invalid; external parameter entities are not read) + const input: string = + '\n\n \n\n\n'; + expectParses(input); + }); + + test("optional05", () => { + // 3 — Tests the Element Valid VC (clause 2) for one instance of "children" content model, providing no + // children where one or two are required (one construction of that model). (upstream: invalid; + // external parameter entities are not read) + const input: string = + '\n\n \n\n\n'; + expectParses(input); + }); + + test("optional06", () => { + // 3 — Tests the Element Valid VC (clause 2) for one instance of "children" content model, providing no + // children where one or two are required (a second construction of that model). (upstream: invalid; + // external parameter entities are not read) + const input: string = + '\n\n \n\n\n\n'; + expectParses(input); + }); + + test("optional07", () => { + // 3 — Tests the Element Valid VC (clause 2) for one instance of "children" content model, providing no + // children where one or two are required (a third construction of that model). (upstream: invalid; + // external parameter entities are not read) + const input: string = + '\n\n \n\n\n\n'; + expectParses(input); + }); + + test("optional08", () => { + // 3 — Tests the Element Valid VC (clause 2) for one instance of "children" content model, providing no + // children where one or two are required (a fourth construction of that model). (upstream: invalid; + // external parameter entities are not read) + const input: string = + '\n\n \n\n\n\n'; + expectParses(input); + }); + + test("optional09", () => { + // 3 — Tests the Element Valid VC (clause 2) for one instance of "children" content model, providing no + // children where one or two are required (a fifth construction of that model). (upstream: invalid; + // external parameter entities are not read) + const input: string = + '\n\n \n\n\n\n'; + expectParses(input); + }); + + test("optional10", () => { + // 3 — Tests the Element Valid VC (clause 2) for one instance of "children" content model, providing + // three children where one or two are required (a basic construction of that model). (upstream: + // invalid; external parameter entities are not read) + const input: string = + '\n\n \n\n\n\n'; + expectParses(input); + }); + + test("optional11", () => { + // 3 — Tests the Element Valid VC (clause 2) for one instance of "children" content model, providing + // three children where one or two are required (a second construction of that model). (upstream: + // invalid; external parameter entities are not read) + const input: string = + '\n\n \n\n\n\n\n'; + expectParses(input); + }); + + test("optional12", () => { + // 3 — Tests the Element Valid VC (clause 2) for one instance of "children" content model, providing + // three children where one or two are required (a third construction of that model). (upstream: + // invalid; external parameter entities are not read) + const input: string = + '\n\n \n\n\n\n\n'; + expectParses(input); + }); + + test("optional13", () => { + // 3 — Tests the Element Valid VC (clause 2) for one instance of "children" content model, providing + // three children where one or two are required (a fourth construction of that model). (upstream: + // invalid; external parameter entities are not read) + const input: string = + '\n\n \n\n\n\n\n'; + expectParses(input); + }); + + test("optional14", () => { + // 3 — Tests the Element Valid VC (clause 2) for one instance of "children" content model, providing + // three children where one or two are required (a fifth construction of that model). (upstream: + // invalid; external parameter entities are not read) + const input: string = + '\n\n \n\n\n\n\n'; + expectParses(input); + }); + + test("optional20", () => { + // 3 — Tests the Element Valid VC (clause 2) for one instance of "children" content model, providing no + // children where one or more are required (a sixth construction of that model). (upstream: invalid; + // external parameter entities are not read) + const input: string = + '\n\n \n\n'; + expectParses(input); + }); + + test("optional21", () => { + // 3 — Tests the Element Valid VC (clause 2) for one instance of "children" content model, providing no + // children where one or more are required (a seventh construction of that model). (upstream: invalid; + // external parameter entities are not read) + const input: string = + '\n\n \n\n\n'; + expectParses(input); + }); + + test("optional22", () => { + // 3 — Tests the Element Valid VC (clause 2) for one instance of "children" content model, providing no + // children where one or more are required (an eigth construction of that model). (upstream: invalid; + // external parameter entities are not read) + const input: string = + '\n\n \n\n\n'; + expectParses(input); + }); + + test("optional23", () => { + // 3 — Tests the Element Valid VC (clause 2) for one instance of "children" content model, providing no + // children where one or more are required (a ninth construction of that model). (upstream: invalid; + // external parameter entities are not read) + const input: string = + '\n\n \n\n\n'; + expectParses(input); + }); + + test("optional24", () => { + // 3 — Tests the Element Valid VC (clause 2) for one instance of "children" content model, providing no + // children where one or more are required (a tenth construction of that model). (upstream: invalid; + // external parameter entities are not read) + const input: string = + '\n\n \n\n\n'; + expectParses(input); + }); + + test("optional25", () => { + // 3 — Tests the Element Valid VC (clause 2) for one instance of "children" content model, providing + // text content where one or more elements are required. (upstream: invalid; external parameter + // entities are not read) + const input: string = + '\n\n No text allowed!\n\n\n'; + expectParses(input); + }); + + test("inv-required00", () => { + // 3.3.2 — Tests the Required Attribute VC. + const input: string = + "\n \n]>\n\n\n\n\n"; + expectParses(input); + }); + + test("inv-required01", () => { + // 3.1 2.10 — Tests the Attribute Value Type (declared) VC for the xml:space attribute + const input: string = + "\n]>\n\n\n\n \n"; + expectParses(input); + }); + + test("inv-required02", () => { + // 3.1 2.12 — Tests the Attribute Value Type (declared) VC for the xml:lang attribute + const input: string = + "\n]>\n\n\n\n \n\n"; + expectParses(input); + }); + + test("root", () => { + // 2.8 — Tests the Root Element Type VC (upstream: invalid; external parameter entities are not read) + const input: string = + "\n\n\n\n\n \n\n"; + expectParses(input); + }); + + test("attr01", () => { + // 3.3.1 — Tests the "Entity Name" VC for the ENTITY attribute type. + const input: string = + '\n\n \n]>\n\n'; + expectParses(input); + }); + + test("attr02", () => { + // 3.3.1 — Tests the "Entity Name" VC for the ENTITIES attribute type. + const input: string = + '\n\n \n\n\n]>\n\n'; + expectParses(input); + }); + + test("attr03", () => { + // 3.3.1 — Tests the "Notation Attributes" VC for the NOTATION attribute type, first clause: value must + // be one of the ones that's declared. + const input: string = + '\n\n\n\n\n\n \n]>\n\n\n'; + expectParses(input); + }); + + test("attr04", () => { + // 3.3.1 — Tests the "Notation Attributes" VC for the NOTATION attribute type, second clause: the names + // in the declaration must all be declared. + const input: string = + '\n\n\n\n \n]>\n\n'; + expectParses(input); + }); + + test("attr05", () => { + // 3.3.1 — Tests the "Name Token" VC for the NMTOKEN attribute type. + const input: string = + '\n\n\n \n]>\n\n'; + expectParses(input); + }); + + test("attr06", () => { + // 3.3.1 — Tests the "Name Token" VC for the NMTOKENS attribute type. + const input: string = + '\n\n\n \n]>\n\n'; + expectParses(input); + }); + + test("attr07", () => { + // 3.3.1 — Tests the "Enumeration" VC by providing a value which wasn't one of the choices. + const input: string = + '\n\n \n]>\n\n\n'; + expectParses(input); + }); + + test("attr08", () => { + // 3.3.2 — Tests the "Fixed Attribute Default" VC by providing the wrong value. + const input: string = + '\n\n \n]>\n\n\n'; + expectParses(input); + }); + + test("attr09", () => { + // 3.3.2 — Tests the "Attribute Default Legal" VC by providing an illegal IDREF value. + const input: string = + '\n\n\n\n \n\n\n\n]>\n\n\n \n \n\n'; + expectParses(input); + }); + + test("attr10", () => { + // 3.3.2 — Tests the "Attribute Default Legal" VC by providing an illegal IDREFS value. + const input: string = + '\n\n\n\n \n\n\n\n]>\n\n\n \n \n\n'; + expectParses(input); + }); + + test("attr11", () => { + // 3.3.2 — Tests the "Attribute Default Legal" VC by providing an illegal ENTITY value. + const input: string = + '\r\n\r\n \r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n\r\n'; + expectParses(input); + }); + + test("attr12", () => { + // 3.3.2 — Tests the "Attribute Default Legal" VC by providing an illegal ENTITIES value. + const input: string = + '\r\n\r\n \r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n\r\n'; + expectParses(input); + }); + + test("attr13", () => { + // 3.3.2 — Tests the "Attribute Default Legal" VC by providing an illegal NMTOKEN value. + const input: string = + '\n\n \n]>\n\n\n\n'; + expectParses(input); + }); + + test("attr14", () => { + // 3.3.2 — Tests the "Attribute Default Legal" VC by providing an illegal NMTOKENS value. + const input: string = + '\n\n \n]>\n\n\n\n\n'; + expectParses(input); + }); + + test("attr15", () => { + // 3.3.2 — Tests the "Attribute Default Legal" VC by providing an illegal NOTATIONS value. + const input: string = + '\n\n \n\n\n\n\n]>\n\n\n'; + expectParses(input); + }); + + test("attr16", () => { + // 3.3.2 — Tests the "Attribute Default Legal" VC by providing an illegal enumeration value. + const input: string = + '\n\n \n]>\n\n\n'; + expectParses(input); + }); + + test("utf16b", () => { + // 4.3.3 2.8 — Tests reading an invalid "big endian" UTF-16 document + const input = Buffer.from( + "/v8APAA/AHgAbQBsACAAdgBlAHIAcwBpAG8AbgA9ACcAMQAuADAAJwAgAGUAbgBjAG8AZABpAG4AZwA9ACcAVQBUAEYALQAxADYAJwA/AD4ACgA8AHIAbwBvAHQALwA+AAo=", + "base64", + ); + expectParses(input); + }); + + test("utf16l", () => { + // 4.3.3 2.8 — Tests reading an invalid "little endian" UTF-16 document + const input = Buffer.from( + "//48AD8AeABtAGwAIAB2AGUAcgBzAGkAbwBuAD0AJwAxAC4AMAAnACAAZQBuAGMAbwBkAGkAbgBnAD0AJwBVAFQARgAtADEANgAnAD8APgAKADwAcgBvAG8AdAAvAD4ACgA=", + "base64", + ); + expectParses(input); + }); + + test("empty", () => { + // 2.4 2.7 [18] 3 — CDATA section containing only white space does not match the nonterminal S, and + // cannot appear in these positions. + const input: string = + "\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n∅\r\n\r\n&space;\r\n\r\n\r\n\r\n\r\n\r\n
\r\n"; + expectParses(input); + }); + + test("not-wf-sa03", () => { + // 2.9 — Tests the Entity Declared WFC, ensuring that a reference to externally defined entity causes a + // well-formedness error. (upstream: not-wf; external parameter entities are not read) + const input: string = + '\n\n\n]>\n\n\n'; + expectRejects(input, "XML Parse error: Entity 'number' is not declared"); + }); + + test("attlist01", () => { + // 3.3.1 [56] — SGML's NUTOKEN is not allowed. + const input: string = + '\n\n \n\n \n\n]>\n\n\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'NUTOKEN'", + ); + }); + + test("attlist02", () => { + // 3.3.1 [56] — SGML's NUTOKENS attribute type is not allowed. + const input: string = + '\n\n \n\n \n\n]>\n\n\n\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'NUTOKENS'", + ); + }); + + test("attlist03", () => { + // 3.3.1 [59] — Comma doesn't separate enumerations, unlike in SGML. + const input: string = + '\n\n \n\n \n\n]>\n\n\n\n'; + expectRejects(input, "XML Parse error: Expected '|' or ')' in the enumeration but found ','"); + }); + + test("attlist04", () => { + // 3.3.1 [56] — SGML's NUMBER attribute type is not allowed. + const input: string = + '\n\n \n\n \n\n]>\n\n\n\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'NUMBER'", + ); + }); + + test("attlist05", () => { + // 3.3.1 [56] — SGML's NUMBERS attribute type is not allowed. + const input: string = + '\n\n \n\n \n\n]>\n\n\n\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'NUMBERS'", + ); + }); + + test("attlist06", () => { + // 3.3.1 [56] — SGML's NAME attribute type is not allowed. + const input: string = + '\n\n \n\n \n\n]>\n\n\n\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'NAME'", + ); + }); + + test("attlist07", () => { + // 3.3.1 [56] — SGML's NAMES attribute type is not allowed. + const input: string = + '\n\n \n\n \n\n]>\n\n\n\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'NAMES'", + ); + }); + + test("attlist08", () => { + // 3.3.1 [56] — SGML's #CURRENT is not allowed. + const input: string = + "\n\n \n\n \n\n]>\n\n\n"; + expectRejects( + input, + "XML Parse error: Expected #REQUIRED, #IMPLIED, #FIXED or a quoted default value but found '#CURRENT'", + ); + }); + + test("attlist09", () => { + // 3.3.1 [56] — SGML's #CONREF is not allowed. + const input: string = + '\n\n \n\n]>\n\n\n\n'; + expectRejects( + input, + "XML Parse error: Expected #REQUIRED, #IMPLIED, #FIXED or a quoted default value but found '#CONREF'", + ); + }); + + test("attlist10", () => { + // 3.1 [40] — Whitespace required between attributes + const input: string = + '\n\n\n]>\n\n \n\n'; + expectRejects(input, "XML Parse error: Whitespace is required before 'att2'"); + }); + + test("attlist11", () => { + // 3.1 [44] — Whitespace required between attributes + const input: string = + '\n\n\n]>\n\n \n'; + expectRejects(input, "XML Parse error: Whitespace is required before 'att2'"); + }); + + test("cond01", () => { + // 3.4 [61] — Only INCLUDE and IGNORE are conditional section keywords (upstream: not-wf; external + // parameter entities are not read) + const input: string = '\n]>\n\n\n'; + expectParses(input); + }); + + test("cond02", () => { + // 3.4 [61] — Must have keyword in conditional sections (upstream: not-wf; external parameter entities + // are not read) + const input: string = '\n]>\n\n\n\n'; + expectParses(input); + }); + + test("content01", () => { + // 3.2.1 [48] — No whitespace before "?" in content model + const input: string = + "\n \n]>\n\n"; + expectRejects(input, "XML Parse error: An occurrence indicator must directly follow the name or ')' it applies to"); + }); + + test("content02", () => { + // 3.2.1 [48] — No whitespace before "*" in content model + const input: string = + "\n \n]>\n\n\n"; + expectRejects(input, "XML Parse error: An occurrence indicator must directly follow the name or ')' it applies to"); + }); + + test("content03", () => { + // 3.2.1 [48] — No whitespace before "+" in content model + const input: string = + "\n \n]>\n\n\n"; + expectRejects(input, "XML Parse error: An occurrence indicator must directly follow the name or ')' it applies to"); + }); + + test("decl01", () => { + // 4.3.1 [77] — External entities may not have standalone decls. (upstream: not-wf; external parameter + // entities are not read) + const input: string = + '\n \n\n \n %ent01;\n]>\n\n'; + expectParses(input); + }); + + test("nwf-dtd00", () => { + // 3.2.1 [55] — Comma mandatory in content model + const input: string = + "\n\t\n \n \n]>\n\n \n"; + expectRejects(input, "XML Parse error: Expected ')', '|' or ',' in the content model but found 'foo'"); + }); + + test("nwf-dtd01", () => { + // 3.2.1 [55] — Can't mix comma and vertical bar in content models + const input: string = + "\n\t\n \n \n]>\n\n \n"; + expectRejects(input, "XML Parse error: A content model group cannot mix ',' and '|'"); + }); + + test("dtd02", () => { + // 4.1 [69] — PE name immediately after "%" + const input: string = + '\n \n ">\n % foo;\n]>\n\n\n'; + expectRejects(input, "XML Parse error: Expected a markup declaration or ']' in the internal subset but found '%'"); + }); + + test("dtd03", () => { + // 4.1 [69] — PE name immediately followed by ";" + const input: string = + '\n \n ">\n %foo\n ;\n]>\n\n\n'; + expectRejects(input, "XML Parse error: Expected ';' to end the parameter entity reference 'foo'"); + }); + + test("dtd04", () => { + // 4.2.2 [75] — PUBLIC literal must be quoted + const input: string = + '\n \n \n]>\n\n\n'; + expectRejects(input, "XML Parse error: Expected a quoted public identifier after PUBLIC but found '-'"); + }); + + test("dtd05", () => { + // 4.2.2 [75] — SYSTEM identifier must be quoted + const input: string = + "\n \n \n]>\n\n\n"; + expectRejects(input, "XML Parse error: Expected a quoted system identifier after SYSTEM but found 'elvis.ent'"); + }); + + test("dtd07", () => { + // 4.3.1 [77] — Text declarations (which optionally begin any external entity) are required to have + // "encoding=...". (upstream: not-wf; external parameter entities are not read) + const input: string = '\n]>\n\n'; + expectParses(input); + }); + + test("element00", () => { + // 3.1 [42] — EOF in middle of incomplete ETAG + const input: string = "\n Incomplete end tag.\n but found "); + }); + + test("element01", () => { + // 3.1 [42] — EOF in middle of incomplete ETAG + const input: string = "\n Incomplete end tag.\n' to end the closing tag but found end of input"); + }); + + test("element02", () => { + // 3.1 [43] — Illegal markup (<%@ ... %>) + const input: string = ' ]>\n\n <% @ LANGUAGE="VBSCRIPT" %>\n\n'; + expectRejects(input, "XML Parse error: Expected an element name after '<' but found '%'"); + }); + + test("element03", () => { + // 3.1 [43] — Illegal markup (<% ... %>) + const input: string = + ' ]>\n\n <% document.println ("hello, world"); %>\n\n\n'; + expectRejects(input, "XML Parse error: Expected an element name after '<' but found '%'"); + }); + + test("element04", () => { + // 3.1 [43] — Illegal markup () + const input: string = " ]>\n\n \n\n"; + expectRejects(input, "XML Parse error: Expected a comment or CDATA section after ' { + // 4.3.3 [81] — Illegal character " " in encoding name + const input = Buffer.from('\n\n'); + expectRejects(input, "XML Parse error: Invalid encoding name ' utf-8' in the XML declaration"); + }); + + test("encoding02", () => { + // 4.3.3 [81] — Illegal character "/" in encoding name + const input = Buffer.from('\n\n\n'); + expectRejects(input, "XML Parse error: Invalid encoding name 'a/b' in the XML declaration"); + }); + + test("encoding03", () => { + // 4.3.3 [81] — Illegal character reference in encoding name + const input = Buffer.from('\n\n\n'); + expectRejects(input, "XML Parse error: Invalid encoding name 'just)word' in the XML declaration"); + }); + + test("encoding04", () => { + // 4.3.3 [81] — Illegal character ":" in encoding name + const input = Buffer.from('\n\n\n'); + expectRejects(input, "XML Parse error: Invalid encoding name 'utf:8' in the XML declaration"); + }); + + test("encoding05", () => { + // 4.3.3 [81] — Illegal character "@" in encoding name + const input = Buffer.from('\n\n\n'); + expectRejects(input, "XML Parse error: Invalid encoding name '@import(sys-encoding)' in the XML declaration"); + }); + + test("encoding06", () => { + // 4.3.3 [81] — Illegal character "+" in encoding name + const input = Buffer.from( + '\n\n \n\n\n', + ); + expectRejects(input, "XML Parse error: Invalid encoding name 'XYZ+999' in the XML declaration"); + }); + + test("encoding07", () => { + // 4.3.1 [77] — Text declarations (which optionally begin any external entity) are required to have + // "encoding=...". (upstream: not-wf; external general entities are not read) + const input: string = + '\n\n \n \n]>\n∅\n'; + expectParses(input); + }); + + test("pi", () => { + // 2.6 [16] — No space between PI target name and data + const input: string = + "\n\n\n]>\n\n"; + expectRejects( + input, + "XML Parse error: Expected whitespace or '?>' after the processing instruction target but found '+'", + ); + }); + + test("pubid01", () => { + // 2.3 [12] — Illegal entity ref in public ID + const input: string = + '\n\n \n\n \n]>\n\n\n'; + expectRejects(input, "XML Parse error: Invalid character in a public identifier: '&'"); + }); + + test("pubid02", () => { + // 2.3 [12] — Illegal characters in public ID + const input: string = + '\n\n \n\n " "ignored">\n]>\n\n\n\n'; + expectRejects(input, "XML Parse error: Invalid character in a public identifier: '<'"); + }); + + test("pubid03", () => { + // 2.3 [12] — Illegal characters in public ID + const input: string = + '\n\n \n\n \n]>\n\n\n\n'; + expectRejects(input, "XML Parse error: Invalid character in a public identifier: '['"); + }); + + test("pubid04", () => { + // 2.3 [12] — Illegal characters in public ID + const input: string = + '\n\n \n\n \n]>\n\n\n\n'; + expectRejects(input, "XML Parse error: Invalid character in a public identifier: '{'"); + }); + + test("pubid05", () => { + // 2.3 [12] — SGML-ism: public ID without system ID + const input: string = + '\n\n \n]>\n\n\n'; + expectRejects( + input, + "XML Parse error: Expected a quoted system identifier after the public identifier but found '>'", + ); + }); + + test("sgml01", () => { + // 3 [39] — SGML-ism: omitted end tag for EMPTY content + const input: string = + "\n\n \n]>\n\n\n"; + expectRejects(input, "XML Parse error: Missing closing tag for element 'root'"); + }); + + test("sgml02", () => { + // 2.8 — XML declaration must be at the very beginning of a document; it"s not a processing + // instruction + const input: string = + ' \n \n ]>\n\n'; + expectRejects( + input, + "XML Parse error: ' { + // 2.5 [15] — Comments may not contain "--" + const input: string = + " ]>\n\n \n\n"; + expectRejects(input, "XML Parse error: '--' is not allowed inside a comment"); + }); + + test("sgml04", () => { + // 3.3 [52] — ATTLIST declarations apply to only one element, unlike SGML + const input: string = + "\n\n \n \n\n \n]>\n\n\n"; + expectRejects(input, "XML Parse error: Expected an element name after ' { + // 3.2 [45] — ELEMENT declarations apply to only one element, unlike SGML + const input: string = + "\n\n \n \n \n\n \n]>\n\n\n\n"; + expectRejects(input, "XML Parse error: Expected an element name after ' { + // 3.3 [52] — ATTLIST declarations are never global, unlike in SGML + const input: string = + "\n\n \n\n \n]>\n\n\n"; + expectRejects(input, "XML Parse error: Expected an element name after ' { + // 3.2 [45] — SGML Tag minimization specifications are not allowed + const input: string = + "\n \n]>\n\n\n"; + expectRejects(input, "XML Parse error: Expected EMPTY, ANY or '(' in the element declaration but found '-'"); + }); + + test("sgml08", () => { + // 3.2 [45] — SGML Tag minimization specifications are not allowed + const input: string = + "\n \n]>\n\n\n\n"; + expectRejects(input, "XML Parse error: Expected EMPTY, ANY or '(' in the element declaration but found '-'"); + }); + + test("sgml09", () => { + // 3.2 [45] — SGML Content model exception specifications are not allowed + const input: string = + "\n\n \n]>\n\n\n\n"; + expectRejects(input, "XML Parse error: Expected '>' to end the element declaration but found '-footnote'"); + }); + + test("sgml10", () => { + // 3.2 [45] — SGML Content model exception specifications are not allowed + const input: string = + "\n \n]>\n\n\n\n"; + expectRejects(input, "XML Parse error: Expected '>' to end the element declaration but found '+'"); + }); + + test("sgml11", () => { + // 3.2 [46] — CDATA is not a valid content model spec + const input: string = + "\n \n]>\n\n\n\n"; + expectRejects(input, "XML Parse error: Expected EMPTY, ANY or '(' in the element declaration but found 'CDATA'"); + }); + + test("sgml12", () => { + // 3.2 [46] — RCDATA is not a valid content model spec + const input: string = + "\n \n]>\n\n\n\n\n"; + expectRejects(input, "XML Parse error: Expected EMPTY, ANY or '(' in the element declaration but found 'RCDATA'"); + }); + + test("sgml13", () => { + // 3.2.1 [47] — SGML Unordered content models not allowed + const input: string = + "\n \n \n \n \n]>\n\n\n\n\n"; + expectRejects(input, "XML Parse error: Expected ')', '|' or ',' in the content model but found '&'"); + }); + + test("uri01", () => { + // 4.2.2 [75] — SYSTEM ids may not have URI fragments (upstream: optional error) + const input: string = + '\n\n\n]>\n\n'; + expectParses(input); + }); +}); + +describe("oasis", () => { + test("o-p01pass2", () => { + // 2.2 [1] — various Misc items where they can occur + const input: string = + "\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n"; + expectParses(input); + }); + + test("o-p06pass1", () => { + // 2.3 [6] — various satisfactions of the Names production in a NAMES attribute + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n\r\n\r\n\r\n\r\n'; + expectParses(input); + }); + + test("o-p07pass1", () => { + // 2.3 [7] — various valid Nmtoken 's in an attribute list declaration. + const input: string = + "\r\n\r\n]>\r\n"; + expectParses(input); + }); + + test("o-p08pass1", () => { + // 2.3 [8] — various satisfaction of an NMTOKENS attribute value. + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n\r\n'; + expectParses(input); + }); + + test("o-p09pass1", () => { + // 2.3 [9] — valid EntityValue's. Except for entity references, markup is not recognized. (upstream: + // valid; external parameter entities are not read) + const input: string = '\r\n'; + expectParses(input); + }); + + test("o-p12pass1", () => { + // 2.3 [12] — valid public IDs. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("o-p22pass4", () => { + // 2.8 [22] — XML decl and doctypedecl + const input: string = + '\r\n \r\n\r\n]>\r\n\r\n \r\n\r\n\r\n'; + expectParses(input); + }); + + test("o-p22pass5", () => { + // 2.8 [22] — just doctypedecl + const input: string = + " \r\n\r\n]>\r\n\r\n \r\n\r\n\r\n"; + expectParses(input); + }); + + test("o-p22pass6", () => { + // 2.8 [22] — S between decls is not required + const input: string = '\r\n]>\r\n'; + expectParses(input); + }); + + test("o-p28pass1", () => { + // 3.1 [43] [44] — Empty-element tag must be used for element which are declared EMPTY. + const input: string = "\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("o-p28pass3", () => { + // 2.8 4.1 [28] [69] — Valid doctypedecl with Parameter entity reference. The declaration of a + // parameter entity must precede any reference to it. (upstream: valid; external parameter entities are + // not read) + const input: string = + '">\r\n%eldecl;\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("o-p28pass4", () => { + // 2.8 4.2.2 [28] [75] — Valid doctypedecl with ExternalID as an External Entity declaration. + // (upstream: valid; external parameter entities are not read) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("o-p28pass5", () => { + // 2.8 4.1 [28] [69] — Valid doctypedecl with ExternalID as an External Entity. A parameter entity + // reference is also used. (upstream: valid; external parameter entities are not read) + const input: string = + '\r\n">\r\n\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("o-p29pass1", () => { + // 2.8 [29] — Valid types of markupdecl. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("o-p30pass1", () => { + // 2.8 4.2.2 [30] [75] — Valid doctypedecl with ExternalID as an External Entity. The external entity + // has an element declaration. (upstream: valid; external parameter entities are not read) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("o-p30pass2", () => { + // 2.8 4.2.2 4.3.1 [30] [75] [77] — Valid doctypedecl with ExternalID as an Enternal Entity. The + // external entity begins with a Text Declaration. (upstream: valid; external parameter entities are + // not read) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("o-p31pass1", () => { + // 2.8 [31] — external subset can be empty (upstream: valid; external parameter entities are not read) + const input: string = ']>\r\n\r\n'; + expectParses(input); + }); + + test("o-p31pass2", () => { + // 2.8 3.4 4.2.2 [31] [62] [63] [75] — Valid doctypedecl with EXternalID as Enternal Entity. The + // external entity contains a parameter entity reference and condtional sections. (upstream: valid; + // external parameter entities are not read) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("o-p43pass1", () => { + // 2.4 2.5 2.6 2.7 [15] [16] [18] — Valid use of character data, comments, processing instructions and + // CDATA sections within the start and end tag. + const input: string = + '\r\nCharData">\r\n]>\r\n\r\nCharData \r\n\r\n\r\nCharData \r\n\r\n&ent;"\r\nCharData\r\n\r\n]]>\r\n\r\nCharData \r\n\r\n&ent;"\r\nCharData\r\n\r\n]]>\r\n&ent;"\r\nCharData\r\n\r\n'; + expectParses(input); + }); + + test("o-p45pass1", () => { + // 3.2 [45] — valid element declarations + const input: string = + "\r\n\r\n\r\n]>\r\n"; + expectParses(input); + }); + + test("o-p46pass1", () => { + // 3.2 3.2.1 3.2.2 [45] [46] [47] [51] — Valid use of contentspec, element content models, and mixed + // content within an element type declaration. + const input: string = + "\r\n\r\n\r\n\r\n]>\r\n"; + expectParses(input); + }); + + test("o-p47pass1", () => { + // 3.2 3.2.1 [45] [46] [47] — Valid use of contentspec, element content models, choices, sequences and + // content particles within an element type declaration. The optional character following a name or + // list governs the number of times the element or content particle may appear. + const input: string = + "\r\n\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n"; + expectParses(input); + }); + + test("o-p48pass1", () => { + // 3.2 3.2.1 [45] [46] [47] — Valid use of contentspec, element content models, choices, sequences and + // content particles within an element type declaration. The optional character following a name or + // list governs the number of times the element or content particle may appear. + const input: string = + "\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n"; + expectParses(input); + }); + + test("o-p49pass1", () => { + // 3.2 3.2.1 [45] [46] [47] — Valid use of contentspec, element content models, choices, and content + // particles within an element type declaration. The optional character following a name or list + // governs the number of times the element or content particle may appear. Whitespace is also valid + // between choices. + const input: string = + "\r\n\r\n\r\n\r\n\r\n]>\r\n"; + expectParses(input); + }); + + test("o-p50pass1", () => { + // 3.2 3.2.1 [45] [46] [47] — Valid use of contentspec, element content models, sequences and content + // particles within an element type declaration. The optional character following a name or list + // governs the number of times the element or content particle may appear. Whitespace is also valid + // between sequences. + const input: string = + "\r\n\r\n\r\n\r\n\r\n]>\r\n"; + expectParses(input); + }); + + test("o-p51pass1", () => { + // 3.2.2 [51] — valid Mixed contentspec's. + const input: string = + "\r\n\r\n\r\n\r\n]>\r\n"; + expectParses(input); + }); + + test("o-p52pass1", () => { + // 3.3 [52] — valid AttlistDecls: No AttDef's are required, and the terminating S is optional, multiple + // ATTLISTS per element are OK, and multiple declarations of the same attribute are OK. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n'; + expectParses(input); + }); + + test("o-p53pass1", () => { + // 3.3 [53] — a valid AttDef + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("o-p54pass1", () => { + // 3.3.1 [54] — the three kinds of attribute types + const input: string = + "\r\n\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("o-p55pass1", () => { + // 3.3.1 [55] — StringType = "CDATA" + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("o-p56pass1", () => { + // 3.3.1 [56] — the 7 tokenized attribute types + const input: string = + "\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("o-p57pass1", () => { + // 3.3.1 [57] — enumerated types are NMTOKEN or NOTATION lists + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("o-p58pass1", () => { + // 3.3.1 [58] — NOTATION enumeration has on or more items + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("o-p59pass1", () => { + // 3.3.1 [59] — NMTOKEN enumerations haveon or more items + const input: string = + "\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("o-p60pass1", () => { + // 3.3.2 [60] — the four types of default values + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("o-p61pass1", () => { + // 3.4 [61] — valid conditional sections are INCLUDE and IGNORE (upstream: valid; external parameter + // entities are not read) + const input: string = '\r\n'; + expectParses(input); + }); + + test("o-p62pass1", () => { + // 3.4 [62] — valid INCLUDE sections -- options S before and after keyword, sections can nest + // (upstream: valid; external parameter entities are not read) + const input: string = '\r\n'; + expectParses(input); + }); + + test("o-p63pass1", () => { + // 3.4 [63] — valid IGNORE sections (upstream: valid; external parameter entities are not read) + const input: string = '\r\n'; + expectParses(input); + }); + + test("o-p64pass1", () => { + // 3.4 [64] — IGNOREd sections ignore everything except section delimiters (upstream: valid; external + // parameter entities are not read) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("o-p68pass1", () => { + // 4.1 [68] — Valid entity references. Also ensures that a charref to '&' isn't interpreted as an + // entity reference open delimiter + const input: string = + '\r\n\r\n]>\r\n\r\n&ent;aaa&ent;\r\n\r\n\r\n&en\r\n\r\n'; + expectParses(input); + }); + + test("o-p69pass1", () => { + // 4.1 [69] — Valid PEReferences. (upstream: valid; external parameter entities are not read) + const input: string = + '\r\n">\r\n%pe;%pe;\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("o-p70pass1", () => { + // 4.2 [70] — An EntityDecl is either a GEDecl or a PEDecl + const input: string = + '\r\n\r\n">\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("o-p71pass1", () => { + // 4.2 [71] — Valid GEDecls + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("o-p72pass1", () => { + // 4.2 [72] — Valid PEDecls + const input: string = + '\r\n">\r\n"\r\n >\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("o-p73pass1", () => { + // 4.2 [73] — EntityDef is either Entity value or an external id, with an optional NDataDecl + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("o-p76pass1", () => { + // 4.2.2 [76] — valid NDataDecls + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("o-p01pass1", () => { + // 2.1 [1] — no prolog + const input: string = "\r\n\r\n"; + expectParses(input); + }); + + test("o-p01pass3", () => { + // 2.1 [1] — Misc items after the document + const input: string = + "\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n"; + expectParses(input); + }); + + test("o-p03pass1", () => { + // 2.3 [3] — all valid S characters + const input: string = "\t\r\n "; + expectParses(input); + }); + + test("o-p04pass1", () => { + // 2.3 [4] — names with all valid ASCII characters, and one from each other class in NameChar + const input: string = + "\r\n\r\n\r\n\r\n\r\n"; + expectParses(input); + }); + + test("o-p05pass1", () => { + // 2.3 [5] — various valid Name constructions + const input: string = "\r\n\r\n<::._-0/>\r\n<_:._-0/>\r\n\r\n<_/>\r\n<:/>\r\n"; + expectParses(input); + }); + + test("o-p06fail1", () => { + // 2.3 [6] — Requires at least one name. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n\r\n\r\n'; + expectParses(input); + }); + + test("o-p08fail1", () => { + // 2.3 [8] — at least one Nmtoken is required. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n\r\n'; + expectParses(input); + }); + + test("o-p08fail2", () => { + // 2.3 [8] — an invalid Nmtoken character. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n\r\n'; + expectParses(input); + }); + + test("o-p10pass1", () => { + // 2.3 [10] — valid attribute values + const input: string = '\r\n\r\n'"\'/>\r\n'; + expectParses(input); + }); + + test("o-p14pass1", () => { + // 2.4 [14] — valid CharData + const input: string = "a%b%</doc></doc>]]<&\r\n"; + expectParses(input); + }); + + test("o-p15pass1", () => { + // 2.5 [15] — valid comments + const input: string = "\r\n\r\n"; + expectParses(input); + }); + + test("o-p16pass1", () => { + // 2.6 [16] [17] — Valid form of Processing Instruction. Shows that whitespace character data is valid + // before end of processing instruction. + const input: string = + "\r\n &a%b&#c?>\r\n\r\n\r\n ?>\r\n"; + expectParses(input); + }); + + test("o-p16pass2", () => { + // 2.6 [16] — Valid form of Processing Instruction. Shows that whitespace character data is valid + // before end of processing instruction. + const input: string = "\r\n"; + expectParses(input); + }); + + test("o-p16pass3", () => { + // 2.6 [16] — Valid form of Processing Instruction. Shows that whitespace character data is valid + // before end of processing instruction. + const input: string = '\r\n'; + expectParses(input); + }); + + test("o-p18pass1", () => { + // 2.7 [18] — valid CDSect's. Note that a CDStart in a CDSect is not recognized as such + const input: string = + "] ]> ]]]>\r\n\r\n"; + expectParses(input); + }); + + test("o-p22pass1", () => { + // 2.8 [22] — prolog can be empty + const input: string = "\r\n"; + expectParses(input); + }); + + test("o-p22pass2", () => { + // 2.8 [22] — XML declaration only + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("o-p22pass3", () => { + // 2.8 [22] — XML decl and Misc + const input: string = '\r\n \r\n\r\n'; + expectParses(input); + }); + + test("o-p23pass1", () => { + // 2.8 [23] — Test shows a valid XML declaration along with version info. + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("o-p23pass2", () => { + // 2.8 [23] — Test shows a valid XML declaration along with encoding declaration. + const input = Buffer.from('\r\n\r\n'); + expectParses(input); + }); + + test("o-p23pass3", () => { + // 2.8 [23] — Test shows a valid XML declaration along with Standalone Document Declaration. + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("o-p23pass4", () => { + // 2.8 [23] — Test shows a valid XML declaration, encoding declarationand Standalone Document + // Declaration. + const input = Buffer.from('\r\n\r\n'); + expectParses(input); + }); + + test("o-p24pass1", () => { + // 2.8 [24] — Test shows a prolog that has the VersionInfo delimited by double quotes. + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("o-p24pass2", () => { + // 2.8 [24] — Test shows a prolog that has the VersionInfo delimited by single quotes. + const input: string = "\r\n\r\n"; + expectParses(input); + }); + + test("o-p24pass3", () => { + // 2.8 [24] — Test shows whitespace is allowed in prolog before version info. + const input: string = "\r\n\r\n"; + expectParses(input); + }); + + test("o-p24pass4", () => { + // 2.8 [24] — Test shows whitespace is allowed in prolog on both sides of equal sign. + const input: string = "\r\n\r\n"; + expectParses(input); + }); + + test("o-p25pass1", () => { + // 2.8 [25] — Test shows whitespace is NOT necessary before or after equal sign of versioninfo. + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("o-p25pass2", () => { + // 2.8 [25] — Test shows whitespace can be used on both sides of equal sign of versioninfo. + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("o-p26pass1", () => { + // 2.8 [26] — The valid version number. We cannot test others because a 1.0 processor is allowed to + // fail them. + const input: string = + '\r\n\r\n\r\n'; + expectParses(input); + }); + + test("o-p27pass1", () => { + // 2.8 [27] — Comments are valid as the Misc part of the prolog. + const input: string = + '\r\n\r\n\r\n'; + expectParses(input); + }); + + test("o-p27pass2", () => { + // 2.8 [27] — Processing Instructions are valid as the Misc part of the prolog. + const input: string = '\r\n\r\n\r\n'; + expectParses(input); + }); + + test("o-p27pass3", () => { + // 2.8 [27] — Whitespace is valid as the Misc part of the prolog. + const input: string = '\r\n\r\n \t\r\n\r\n\r\n'; + expectParses(input); + }); + + test("o-p27pass4", () => { + // 2.8 [27] — A combination of comments, whitespaces and processing instructions are valid as the Misc + // part of the prolog. + const input: string = + '\r\n\r\n \t\r\n\r\n\r\n\r\n\r\n \t\r\n\r\n\r\n\r\n'; + expectParses(input); + }); + + test("o-p32pass1", () => { + // 2.9 [32] — Double quotes can be used as delimeters for the value of a Standalone Document + // Declaration. + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("o-p32pass2", () => { + // 2.9 [32] — Single quotes can be used as delimeters for the value of a Standalone Document + // Declaration. + const input: string = "\r\n\r\n"; + expectParses(input); + }); + + test("o-p39pass1", () => { + // 3 3.1 [39] [44] — Empty element tag may be used for any element which has no content. + const input: string = ""; + expectParses(input); + }); + + test("o-p39pass2", () => { + // 3 3.1 [39] [43] — Character data is valid element content. + const input: string = "content"; + expectParses(input); + }); + + test("o-p40pass1", () => { + // 3.1 [40] — Elements content can be empty. + const input: string = ""; + expectParses(input); + }); + + test("o-p40pass2", () => { + // 3.1 [40] — Whitespace is valid within a Start-tag. + const input: string = ""; + expectParses(input); + }); + + test("o-p40pass3", () => { + // 3.1 [40] [41] — Attributes are valid within a Start-tag. + const input: string = ''; + expectParses(input); + }); + + test("o-p40pass4", () => { + // 3.1 [40] — Whitespace and Multiple Attributes are valid within a Start-tag. + const input: string = ''; + expectParses(input); + }); + + test("o-p41pass1", () => { + // 3.1 [41] — Attributes are valid within a Start-tag. + const input: string = ''; + expectParses(input); + }); + + test("o-p41pass2", () => { + // 3.1 [41] — Whitespace is valid within a Start-tags Attribute. + const input: string = ''; + expectParses(input); + }); + + test("o-p42pass1", () => { + // 3.1 [42] — Test shows proper syntax for an End-tag. + const input: string = ""; + expectParses(input); + }); + + test("o-p42pass2", () => { + // 3.1 [42] — Whitespace is valid after name in End-tag. + const input: string = ""; + expectParses(input); + }); + + test("o-p44pass1", () => { + // 3.1 [44] — Valid display of an Empty Element Tag. + const input: string = ""; + expectParses(input); + }); + + test("o-p44pass2", () => { + // 3.1 [44] — Empty Element Tags can contain an Attribute. + const input: string = ''; + expectParses(input); + }); + + test("o-p44pass3", () => { + // 3.1 [44] — Whitespace is valid in an Empty Element Tag following the end of the attribute value. + const input: string = ''; + expectParses(input); + }); + + test("o-p44pass4", () => { + // 3.1 [44] — Whitespace is valid after the name in an Empty Element Tag. + const input: string = ""; + expectParses(input); + }); + + test("o-p44pass5", () => { + // 3.1 [44] — Whitespace and Multiple Attributes are valid in an Empty Element Tag. + const input: string = ''; + expectParses(input); + }); + + test("o-p66pass1", () => { + // 4.1 [66] — valid character references + const input: string = "\r\nA AOO \r\n􏋬 \r\n"; + expectParses(input); + }); + + test("o-p74pass1", () => { + // 4.2 [74] — PEDef is either an entity value or an external id + const input: string = + '">\r\n\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("o-p75pass1", () => { + // 4.2.2 [75] — valid external identifiers + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("o-e2", () => { + // 3.3.1 [58] [59] Errata [E2] — Validity Constraint: No duplicate tokens + const input: string = + '\r\n\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("o-p01fail1", () => { + // 2.1 [1] — S cannot occur before the prolog + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n'; + expectRejects( + input, + "XML Parse error: ' { + // 2.1 [1] — comments cannot occur before the prolog + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n'; + expectRejects( + input, + "XML Parse error: ' { + // 2.1 [1] — only one document element + const input: string = + "\r\n\r\n\r\n\r\n\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Only one root element is allowed"); + }); + + test("o-p01fail4", () => { + // 2.1 [1] — document element must be complete. + const input: string = ""; + expectRejects(input, "XML Parse error: Missing closing tag for element 'doc'"); + }); + + test("o-p02fail1", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4AAAA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x00"); + }); + + test("o-p02fail10", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4ACwA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x0B"); + }); + + test("o-p02fail11", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4ADAA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x0C"); + }); + + test("o-p02fail12", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4ADgA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x0E"); + }); + + test("o-p02fail13", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4ADwA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x0F"); + }); + + test("o-p02fail14", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4AEAA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x10"); + }); + + test("o-p02fail15", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4AEQA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x11"); + }); + + test("o-p02fail16", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4AEgA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x12"); + }); + + test("o-p02fail17", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4AEwA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x13"); + }); + + test("o-p02fail18", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4AFAA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x14"); + }); + + test("o-p02fail19", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4AFQA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x15"); + }); + + test("o-p02fail2", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4AAQA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x01"); + }); + + test("o-p02fail20", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4AFgA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x16"); + }); + + test("o-p02fail21", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4AFwA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x17"); + }); + + test("o-p02fail22", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4AGAA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x18"); + }); + + test("o-p02fail23", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4AGQA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x19"); + }); + + test("o-p02fail24", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4AGgA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x1A"); + }); + + test("o-p02fail25", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4AGwA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x1B"); + }); + + test("o-p02fail26", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4AHAA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x1C"); + }); + + test("o-p02fail27", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4AHQA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x1D"); + }); + + test("o-p02fail28", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4AHgA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x1E"); + }); + + test("o-p02fail29", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4AHwA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x1F"); + }); + + test("o-p02fail3", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4AAgA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x02"); + }); + + test("o-p02fail30", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4A/v88AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: '\ufffe' (U+FFFE)"); + }); + + test("o-p02fail31", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4A//88AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: '\uffff' (U+FFFF)"); + }); + + test("o-p02fail4", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4AAwA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x03"); + }); + + test("o-p02fail5", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4ABAA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x04"); + }); + + test("o-p02fail6", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4ABQA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x05"); + }); + + test("o-p02fail7", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4ABgA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x06"); + }); + + test("o-p02fail8", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4ABwA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x07"); + }); + + test("o-p02fail9", () => { + // 2.2 [2] — Use of illegal character within XML document. + const input = Buffer.from("//48AGQAbwBjAD4ACAA8AC8AZABvAGMAPgA=", "base64"); + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x08"); + }); + + test("o-p03fail1", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input = Buffer.from("ADxkb2MvPg==", "base64"); + expectRejects(input, "XML Parse error: UTF-16 input has an odd number of bytes"); + }); + + test("o-p03fail10", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\u000b"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x0B"); + }); + + test("o-p03fail11", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\f"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x0C"); + }); + + test("o-p03fail12", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\u000e"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x0E"); + }); + + test("o-p03fail13", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\u000f"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x0F"); + }); + + test("o-p03fail14", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\u0010"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x10"); + }); + + test("o-p03fail15", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\u0011"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x11"); + }); + + test("o-p03fail16", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\u0012"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x12"); + }); + + test("o-p03fail17", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\u0013"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x13"); + }); + + test("o-p03fail18", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\u0014"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x14"); + }); + + test("o-p03fail19", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\u0015"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x15"); + }); + + test("o-p03fail2", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\u0001"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x01"); + }); + + test("o-p03fail20", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\u0016"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x16"); + }); + + test("o-p03fail21", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\u0017"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x17"); + }); + + test("o-p03fail22", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\u0018"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x18"); + }); + + test("o-p03fail23", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\u0019"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x19"); + }); + + test("o-p03fail24", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\u001a"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x1A"); + }); + + test("o-p03fail25", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\u001b"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x1B"); + }); + + test("o-p03fail26", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\u001c"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x1C"); + }); + + test("o-p03fail27", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\u001d"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x1D"); + }); + + test("o-p03fail28", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\u001e"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x1E"); + }); + + test("o-p03fail29", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\u001f"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x1F"); + }); + + test("o-p03fail3", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\u0002"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x02"); + }); + + test("o-p03fail4", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\u0003"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x03"); + }); + + test("o-p03fail5", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\u0004"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x04"); + }); + + test("o-p03fail7", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\u0006"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x06"); + }); + + test("o-p03fail8", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\u0007"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x07"); + }); + + test("o-p03fail9", () => { + // 2.3 [3] — Use of illegal character within XML document. + const input: string = "\b"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x08"); + }); + + test("o-p04fail1", () => { + // 2.3 [4] — Name contains invalid character. + const input: string = "\r\n"; + expectRejects(input, "XML Parse error: Expected an attribute name, '>' or '/>' in the start tag but found '@'"); + }); + + test("o-p04fail2", () => { + // 2.3 [4] — Name contains invalid character. + const input: string = "\r\n"; + expectRejects(input, "XML Parse error: Expected a keyword after '#' but found '/'"); + }); + + test("o-p04fail3", () => { + // 2.3 [4] — Name contains invalid character. + const input: string = "\r\n"; + expectRejects(input, "XML Parse error: Expected an attribute name, '>' or '/>' in the start tag but found '$'"); + }); + + test("o-p05fail1", () => { + // 2.3 [5] — a Name cannot start with a digit + const input: string = "<0A/>"; + expectRejects(input, "XML Parse error: Expected an element name after '<' but found '0'"); + }); + + test("o-p05fail2", () => { + // 2.3 [5] — a Name cannot start with a '.' + const input: string = "<.A/>"; + expectRejects(input, "XML Parse error: Expected an element name after '<' but found '.'"); + }); + + test("o-p05fail3", () => { + // 2.3 [5] — a Name cannot start with a "-" + const input: string = "<-A/>"; + expectRejects(input, "XML Parse error: Expected an element name after '<' but found '-'"); + }); + + test("o-p05fail4", () => { + // 2.3 [5] — a Name cannot start with a CombiningChar + const input: string = "<̀A/>"; + expectRejects(input, "XML Parse error: Expected an element name after '<' but found '̀' (U+0300)"); + }); + + test("o-p05fail5", () => { + // 2.3 [5] — a Name cannot start with an Extender + const input: string = "<·A/>"; + expectRejects(input, "XML Parse error: Expected an element name after '<' but found '·' (U+00B7)"); + }); + + test("o-p09fail1", () => { + // 2.3 [9] — EntityValue excludes '%' (upstream: not-wf; external parameter entities are not read) + const input: string = '\r\n'; + expectParses(input); + }); + + test("o-p09fail2", () => { + // 2.3 [9] — EntityValue excludes '&' (upstream: not-wf; external parameter entities are not read) + const input: string = '\r\n'; + expectParses(input); + }); + + test("o-p09fail3", () => { + // 2.3 [9] — incomplete character reference + const input: string = '\r\n\r\n]>\r\n'; + expectRejects( + input, + "XML Parse error: Invalid character reference: expected a number followed by ';' but found '\"'", + ); + }); + + test("o-p09fail4", () => { + // 2.3 [9] — quote types must match + const input: string = "\r\n\r\n]>\r\n"; + expectRejects(input, "XML Parse error: Unterminated entity value"); + }); + + test("o-p09fail5", () => { + // 2.3 [9] — quote types must match + const input: string = "\r\n\r\n]>\r\n"; + expectRejects(input, "XML Parse error: Unterminated entity value"); + }); + + test("o-p10fail1", () => { + // 2.3 [10] — attribute values exclude '<' + const input: string = '\r\n'; + expectRejects(input, "XML Parse error: '<' is not allowed in attribute values"); + }); + + test("o-p10fail2", () => { + // 2.3 [10] — attribute values exclude '&' + const input: string = '\r\n'; + expectRejects(input, "XML Parse error: Expected an entity name after '&' but found '\"'"); + }); + + test("o-p10fail3", () => { + // 2.3 [10] — quote types must match + const input: string = "\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Unterminated quoted string"); + }); + + test("o-p11fail2", () => { + // 2.3 [11] — cannot contain delimiting quotes + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected '>' to end the notation declaration but found '\"'"); + }); + + test("o-p12fail1", () => { + // 2.3 [12] — '"' excluded + const input: string = + "\r\n\r\n\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in a public identifier: '\"'"); + }); + + test("o-p12fail2", () => { + // 2.3 [12] — '\' excluded + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Invalid character in a public identifier: '\\'"); + }); + + test("o-p12fail3", () => { + // 2.3 [12] — entity references excluded + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Invalid character in a public identifier: '&'"); + }); + + test("o-p12fail4", () => { + // 2.3 [12] — '>' excluded + const input: string = + '\r\n\r\n">\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Invalid character in a public identifier: '>'"); + }); + + test("o-p12fail5", () => { + // 2.3 [12] — '<' excluded + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Invalid character in a public identifier: '<'"); + }); + + test("o-p12fail6", () => { + // 2.3 [12] — built-in entity refs excluded + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Invalid character in a public identifier: '&'"); + }); + + test("o-p12fail7", () => { + // 2.3 [13] — The public ID has a tab character, which is disallowed + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Invalid character in a public identifier: tab"); + }); + + test("o-p14fail1", () => { + // 2.4 [14] — '<' excluded + const input: string = "< \r\n"; + expectRejects(input, "XML Parse error: Expected an element name after '<' but found space"); + }); + + test("o-p14fail2", () => { + // 2.4 [14] — '&' excluded + const input: string = "& \r\n"; + expectRejects(input, "XML Parse error: Expected an entity name after '&' but found space"); + }); + + test("o-p14fail3", () => { + // 2.4 [14] — "]]>" excluded + const input: string = "a]]>b\r\n"; + expectRejects(input, "XML Parse error: ']]>' is only allowed as the end of a CDATA section"); + }); + + test("o-p15fail1", () => { + // 2.5 [15] — comments can't end in '-' + const input: string = "\r\n"; + expectRejects(input, "XML Parse error: '--' is not allowed inside a comment"); + }); + + test("o-p15fail2", () => { + // 2.5 [15] — one comment per comment (contrasted with SGML) + const input: string = "\r\n"; + expectRejects(input, "XML Parse error: '--' is not allowed inside a comment"); + }); + + test("o-p15fail3", () => { + // 2.5 [15] — can't include 2 or more adjacent '-'s + const input: string = "\r\n"; + expectRejects(input, "XML Parse error: '--' is not allowed inside a comment"); + }); + + test("o-p16fail1", () => { + // 2.6 [16] — "xml" is an invalid PITarget + const input: string = "\r\n\r\n"; + expectRejects( + input, + "XML Parse error: ' { + // 2.6 [16] — a PITarget must be present + const input: string = "\r\n"; + expectRejects(input, "XML Parse error: Expected a processing instruction target after ' { + // 2.6 [16] — S after PITarget is required + const input: string = "\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected whitespace or '?>' after the processing instruction target but found '+'", + ); + }); + + test("o-p18fail1", () => { + // 2.7 [18] — no space before "CDATA" + const input: string = ""; + expectRejects(input, "XML Parse error: Expected a comment or CDATA section after ' { + // 2.7 [18] — no space after "CDATA" + const input: string = ""; + expectRejects(input, "XML Parse error: Expected a comment or CDATA section after ' { + // 2.7 [18] — CDSect's can't nest + const input: string = "\r\n\r\n]]>\r\n"; + expectRejects(input, "XML Parse error: ']]>' is only allowed as the end of a CDATA section"); + }); + + test("o-p22fail1", () => { + // 2.8 [22] — prolog must start with XML decl + const input: string = '\r\n\r\n\r\n'; + expectRejects( + input, + "XML Parse error: ' { + // 2.8 [22] — prolog must start with XML decl + const input: string = '\r\n]>\r\n\r\n\r\n'; + expectRejects( + input, + "XML Parse error: ' { + // 2.8 [23] — "xml" must be lower-case + const input: string = '\r\n\r\n'; + expectRejects( + input, + "XML Parse error: ' { + // 2.8 [23] — VersionInfo must be supplied + const input = Buffer.from('\r\n\r\n'); + expectRejects(input, 'XML Parse error: The XML declaration must start with version="1.0"'); + }); + + test("o-p23fail3", () => { + // 2.8 [23] — VersionInfo must come first + const input = Buffer.from('\r\n\r\n'); + expectRejects(input, 'XML Parse error: The XML declaration must start with version="1.0"'); + }); + + test("o-p23fail4", () => { + // 2.8 [23] — SDDecl must come last + const input = Buffer.from('\r\n\r\n'); + expectRejects( + input, + "XML Parse error: Misplaced 'encoding' in the XML declaration (the order is version, encoding, standalone)", + ); + }); + + test("o-p23fail5", () => { + // 2.8 [23] — no SGML-type PIs + const input: string = '\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected '?>' to end the XML declaration but found '>'"); + }); + + test("o-p24fail1", () => { + // 2.8 [24] — quote types must match + const input: string = "\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in a quoted string: '>'"); + }); + + test("o-p24fail2", () => { + // 2.8 [24] — quote types must match + const input: string = "\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in a quoted string: '>'"); + }); + + test("o-p25fail1", () => { + // 2.8 [25] — Comment is illegal in VersionInfo. + const input: string = ' ="1.0"?>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected '=' after the name in the XML declaration but found a comment"); + }); + + test("o-p26fail1", () => { + // 2.8 [26] — Illegal character in VersionNum. + const input: string = '\r\n\r\n'; + expectRejects(input, "XML Parse error: Unsupported XML version '1.0?' (this is an XML 1.0 parser)"); + }); + + test("o-p26fail2", () => { + // 2.8 [26] — Illegal character in VersionNum. + const input: string = '\r\n\r\n'; + expectRejects(input, "XML Parse error: Unsupported XML version '1.0^' (this is an XML 1.0 parser)"); + }); + + test("o-p27fail1", () => { + // 2.8 [27] — References aren't allowed in Misc, even if they would resolve to valid Misc. + const input: string = '\r\n \r\n\r\n'; + expectRejects(input, "XML Parse error: Expected the root element but found '&'"); + }); + + test("o-p28fail1", () => { + // 2.8 [28] — only declarations in DTD. + const input: string = "\r\n\r\n]>\r\n"; + expectRejects( + input, + "XML Parse error: Expected a markup declaration or ']' in the internal subset but found ' { + // 2.8 [29] — A processor must not pass unknown declaration types. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects( + input, + "XML Parse error: ' { + // 2.8 [30] — An XML declaration is not the same as a TextDecl (upstream: not-wf; external parameter + // entities are not read) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("o-p31fail1", () => { + // 2.8 [31] — external subset excludes doctypedecl (upstream: not-wf; external parameter entities are + // not read) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("o-p32fail1", () => { + // 2.9 [32] — quote types must match + const input: string = '\r\n\r\n'; + expectRejects(input, "XML Parse error: Invalid character in a quoted string: '>'"); + }); + + test("o-p32fail2", () => { + // 2.9 [32] — quote types must match + const input: string = '\r\n\r\n'; + expectRejects(input, "XML Parse error: Invalid character in a quoted string: '>'"); + }); + + test("o-p32fail3", () => { + // 2.9 [32] — initial S is required + const input: string = '\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before 'standalone'"); + }); + + test("o-p32fail4", () => { + // 2.9 [32] — quotes are required + const input: string = '\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected a quoted value in the XML declaration but found 'yes'"); + }); + + test("o-p32fail5", () => { + // 2.9 [32] — yes or no must be lower case + const input: string = '\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Invalid value 'YES' for standalone in the XML declaration (expected yes or no)", + ); + }); + + test("o-p39fail1", () => { + // 3 [39] — start-tag requires end-tag + const input: string = "content"; + expectRejects(input, "XML Parse error: Missing closing tag for element 'doc'"); + }); + + test("o-p39fail2", () => { + // 3 [39] — end-tag requires start-tag + const input: string = "content"; + expectRejects(input, "XML Parse error: Expected closing tag but found "); + }); + + test("o-p39fail3", () => { + // 3 [39] — XML documents contain one or more elements + const input: string = ""; + expectRejects(input, "XML Parse error: XML document must have a root element"); + }); + + test("o-p39fail4", () => { + // 2.8 [23] — XML declarations must be correctly terminated + const input: string = '\r\n'; + expectRejects(input, "XML Parse error: Expected '?>' to end the XML declaration but found '>'"); + }); + + test("o-p39fail5", () => { + // 2.8 [23] — XML declarations must be correctly terminated + const input: string = + '\r\n\r\n]>\r\n\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected '?>' to end the XML declaration but found '>'"); + }); + + test("o-p40fail1", () => { + // 3.1 [40] — S is required between attributes + const input: string = ''; + expectRejects(input, "XML Parse error: Whitespace is required before 'att2'"); + }); + + test("o-p40fail2", () => { + // 3.1 [40] — tags start with names, not nmtokens + const input: string = "<3notname>"; + expectRejects(input, "XML Parse error: Expected an element name after '<' but found '3'"); + }); + + test("o-p40fail3", () => { + // 3.1 [40] — tags start with names, not nmtokens + const input: string = "<3notname>"; + expectRejects(input, "XML Parse error: Expected an element name after '<' but found '3'"); + }); + + test("o-p40fail4", () => { + // 3.1 [40] — no space before name + const input: string = "< doc>"; + expectRejects(input, "XML Parse error: Expected an element name after '<' but found space"); + }); + + test("o-p41fail1", () => { + // 3.1 [41] — quotes are required (contrast with SGML) + const input: string = "\r\n]>\r\n"; + expectRejects(input, "XML Parse error: Expected EMPTY, ANY or '(' in the element declaration but found 'att'"); + }); + + test("o-p41fail2", () => { + // 3.1 [41] — attribute name is required (contrast with SGML) + const input: string = "\r\n]>\r\n"; + expectRejects(input, "XML Parse error: Expected EMPTY, ANY or '(' in the element declaration but found 'att'"); + }); + + test("o-p41fail3", () => { + // 3.1 [41] — Eq required + const input: string = ''; + expectRejects(input, "XML Parse error: Expected '=' after the attribute name but found '\"'"); + }); + + test("o-p42fail1", () => { + // 3.1 [42] — no space before name + const input: string = ""; + expectRejects(input, "XML Parse error: Expected an element name after ' { + // 3.1 [42] — cannot end with "/>" + const input: string = ""; + expectRejects(input, "XML Parse error: Expected '>' to end the closing tag but found '/>'"); + }); + + test("o-p42fail3", () => { + // 3.1 [42] — no NET (contrast with SGML) + const input: string = "' after '/' but found 'd'"); + }); + + test("o-p43fail1", () => { + // 3.1 [43] — no non-comment declarations + const input: string = + '\r\nCharData">\r\n]>\r\n\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected a comment or CDATA section after ' { + // 3.1 [43] — no conditional sections + const input: string = + '\r\nCharData">\r\n]>\r\n\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected a comment or CDATA section after ' { + // 3.1 [43] — no conditional sections + const input: string = + '\r\nCharData">\r\n]>\r\n\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected a comment or CDATA section after ' { + // 3.1 [44] — Illegal space before Empty element tag. + const input: string = "< doc/>"; + expectRejects(input, "XML Parse error: Expected an element name after '<' but found space"); + }); + + test("o-p44fail2", () => { + // 3.1 [44] — Illegal space after Empty element tag. + const input: string = ""; + expectRejects(input, "XML Parse error: Expected '>' after '/' but found space"); + }); + + test("o-p44fail3", () => { + // 3.1 [44] — Illegal comment in Empty element tag. + const input: string = ""; + expectRejects(input, "XML Parse error: Expected an attribute name, '>' or '/>' in the start tag but found '--bad'"); + }); + + test("o-p44fail4", () => { + // 3.1 [44] — Whitespace required between attributes. + const input: string = ''; + expectRejects(input, "XML Parse error: Whitespace is required before 'att2'"); + }); + + test("o-p44fail5", () => { + // 3.1 [44] — Duplicate attribute name is illegal. + const input: string = ''; + expectRejects(input, "XML Parse error: Duplicate attribute 'att'"); + }); + + test("o-p45fail1", () => { + // 3.2 [45] — ELEMENT must be upper case. + const input: string = "\r\n]>\r\n"; + expectRejects( + input, + "XML Parse error: ' { + // 3.2 [45] — S before contentspec is required. + const input: string = "\r\n]>\r\n"; + expectRejects(input, "XML Parse error: Whitespace is required before '('"); + }); + + test("o-p45fail3", () => { + // 3.2 [45] — only one content spec + const input: string = "\r\n]>\r\n"; + expectRejects(input, "XML Parse error: Expected an element name after ' { + // 3.2 [45] — no comments in declarations (contrast with SGML) + const input: string = "\r\n]>\r\n"; + expectRejects(input, "XML Parse error: Expected '>' to end the element declaration but found '--bad'"); + }); + + test("o-p46fail1", () => { + // 3.2 [46] — no parens on declared content + const input: string = "\r\n\r\n]>\r\n"; + expectRejects(input, "XML Parse error: Expected an element name or '(' in the content model but found '#EMPTY'"); + }); + + test("o-p46fail2", () => { + // 3.2 [46] — no inclusions (contrast with SGML) + const input: string = "\r\n\r\n]>\r\n"; + expectRejects(input, "XML Parse error: Expected '>' to end the element declaration but found '+'"); + }); + + test("o-p46fail3", () => { + // 3.2 [46] — no exclusions (contrast with SGML) + const input: string = "\r\n\r\n]>\r\n"; + expectRejects(input, "XML Parse error: Expected '>' to end the element declaration but found '-'"); + }); + + test("o-p46fail4", () => { + // 3.2 [46] — no space before occurrence + const input: string = "\r\n\r\n]>\r\n"; + expectRejects(input, "XML Parse error: An occurrence indicator must directly follow the name or ')' it applies to"); + }); + + test("o-p46fail5", () => { + // 3.2 [46] — single group + const input: string = "\r\n\r\n]>\r\n"; + expectRejects(input, "XML Parse error: Expected '>' to end the element declaration but found '('"); + }); + + test("o-p46fail6", () => { + // 3.2 [46] — can't be both declared and modeled + const input: string = "\r\n\r\n]>\r\n"; + expectRejects(input, "XML Parse error: Expected '>' to end the element declaration but found '('"); + }); + + test("o-p47fail1", () => { + // 3.2.1 [47] — Invalid operator '|' must match previous operator ',' + const input: string = "\r\n\r\n]>\r\n"; + expectRejects(input, "XML Parse error: A content model group cannot mix ',' and '|'"); + }); + + test("o-p47fail2", () => { + // 3.2.1 [47] — Illegal character '-' in Element-content model + const input: string = "\r\n\r\n]>\r\n"; + expectRejects(input, "XML Parse error: Expected '>' to end the element declaration but found '-'"); + }); + + test("o-p47fail3", () => { + // 3.2.1 [47] — Optional character must follow a name or list + const input: string = "\r\n\r\n]>\r\n"; + expectRejects(input, "XML Parse error: Expected EMPTY, ANY or '(' in the element declaration but found '*'"); + }); + + test("o-p47fail4", () => { + // 3.2.1 [47] — Illegal space before optional character + const input: string = "\r\n\r\n]>\r\n"; + expectRejects(input, "XML Parse error: An occurrence indicator must directly follow the name or ')' it applies to"); + }); + + test("o-p48fail1", () => { + // 3.2.1 [48] — Illegal space before optional character + const input: string = "\r\n\r\n]>\r\n"; + expectRejects(input, "XML Parse error: An occurrence indicator must directly follow the name or ')' it applies to"); + }); + + test("o-p48fail2", () => { + // 3.2.1 [48] — Illegal space before optional character + const input: string = "\r\n\r\n]>\r\n"; + expectRejects(input, "XML Parse error: An occurrence indicator must directly follow the name or ')' it applies to"); + }); + + test("o-p49fail1", () => { + // 3.2.1 [49] — connectors must match + const input: string = "\r\n\r\n"; + expectRejects(input, "XML Parse error: A content model group cannot mix ',' and '|'"); + }); + + test("o-p50fail1", () => { + // 3.2.1 [50] — connectors must match + const input: string = "\r\n\r\n"; + expectRejects(input, "XML Parse error: A content model group cannot mix ',' and '|'"); + }); + + test("o-p51fail1", () => { + // 3.2.2 [51] — occurrence on #PCDATA group must be * + const input: string = "\r\n]>\r\n"; + expectRejects(input, "XML Parse error: A mixed content model may only be followed by '*'"); + }); + + test("o-p51fail2", () => { + // 3.2.2 [51] — occurrence on #PCDATA group must be * + const input: string = "\r\n]>\r\n"; + expectRejects(input, "XML Parse error: A mixed content model may only be followed by '*'"); + }); + + test("o-p51fail3", () => { + // 3.2.2 [51] — #PCDATA must come first + const input: string = + "\r\n\r\n]>\r\n"; + expectRejects(input, "XML Parse error: #PCDATA must come first in a content model, as (#PCDATA|a|b)*"); + }); + + test("o-p51fail4", () => { + // 3.2.2 [51] — occurrence on #PCDATA group must be * + const input: string = + "\r\n\r\n]>\r\n"; + expectRejects(input, "XML Parse error: A mixed content model may only be followed by '*'"); + }); + + test("o-p51fail5", () => { + // 3.2.2 [51] — only '|' connectors + const input: string = + "\r\n\r\n]>\r\n"; + expectRejects(input, "XML Parse error: A mixed content model is separated by '|', not ','"); + }); + + test("o-p51fail6", () => { + // 3.2.2 [51] — Only '|' connectors and occurrence on #PCDATA group must be * + const input: string = + "\r\n\r\n]>\r\n"; + expectRejects(input, "XML Parse error: A mixed content model is separated by '|', not ','"); + }); + + test("o-p51fail7", () => { + // 3.2.2 [51] — no nested groups + const input: string = + "\r\n\r\n]>\r\n"; + expectRejects(input, "XML Parse error: Only element names may follow #PCDATA in a mixed content model"); + }); + + test("o-p52fail1", () => { + // 3.3 [52] — A name is required + const input: string = "\r\n\r\n]>\r\n"; + expectRejects(input, "XML Parse error: Expected an element name after ''"); + }); + + test("o-p52fail2", () => { + // 3.3 [52] — A name is required + const input: string = "\r\n\r\n]>\r\n"; + expectRejects(input, "XML Parse error: Expected an element name after ''"); + }); + + test("o-p53fail1", () => { + // 3.3 [53] — S is required before default + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Whitespace is required before '#IMPLIED'"); + }); + + test("o-p53fail2", () => { + // 3.3 [53] — S is required before type + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Whitespace is required before '('"); + }); + + test("o-p53fail3", () => { + // 3.3 [53] — type is required + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found '#IMPLIED'", + ); + }); + + test("o-p53fail4", () => { + // 3.3 [53] — default is required + const input: string = "\r\n\r\n]>\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected #REQUIRED, #IMPLIED, #FIXED or a quoted default value but found '>'", + ); + }); + + test("o-p53fail5", () => { + // 3.3 [53] — name is requried + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected an attribute name or '>' in the ATTLIST declaration but found '('"); + }); + + test("o-p54fail1", () => { + // 3.3.1 [54] — don't pass unknown attribute types + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'DUNNO'", + ); + }); + + test("o-p55fail1", () => { + // 3.3.1 [55] — must be upper case + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'cdata'", + ); + }); + + test("o-p56fail1", () => { + // 3.3.1 [56] — no IDS type + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'IDS'", + ); + }); + + test("o-p56fail2", () => { + // 3.3.1 [56] — no NUMBER type + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'NUMBER'", + ); + }); + + test("o-p56fail3", () => { + // 3.3.1 [56] — no NAME type + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'NAME'", + ); + }); + + test("o-p56fail4", () => { + // 3.3.1 [56] — no ENTITYS type - types must be upper case + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'ENTITYS'", + ); + }); + + test("o-p56fail5", () => { + // 3.3.1 [56] — types must be upper case + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'id'", + ); + }); + + test("o-p57fail1", () => { + // 3.3.1 [57] — no keyword for NMTOKEN enumeration + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected #REQUIRED, #IMPLIED, #FIXED or a quoted default value but found '('", + ); + }); + + test("o-p58fail1", () => { + // 3.3.1 [58] — at least one value required + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected a notation name but found ')'"); + }); + + test("o-p58fail2", () => { + // 3.3.1 [58] — separator must be '|' + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected '|' or ')' in the enumeration but found ','"); + }); + + test("o-p58fail3", () => { + // 3.3.1 [58] — notations are NAMEs, not NMTOKENs -- note: Leaving the invalid notation undeclared + // would cause a validating parser to fail without checking the name syntax, so the notation is + // declared with an invalid name. A parser that reports error positions should report an error at the + // AttlistDecl on line 6, before reaching the notation declaration. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected a notation name but found '0b'"); + }); + + test("o-p58fail4", () => { + // 3.3.1 [58] — NOTATION must be upper case + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'notation'", + ); + }); + + test("o-p58fail5", () => { + // 3.3.1 [58] — S after keyword is required + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before '('"); + }); + + test("o-p58fail6", () => { + // 3.3.1 [58] — parentheses are require + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected '(' after NOTATION but found 'a'"); + }); + + test("o-p58fail7", () => { + // 3.3.1 [58] — values are unquoted + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected '(' after NOTATION but found '\"'"); + }); + + test("o-p58fail8", () => { + // 3.3.1 [58] — values are unquoted + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected a notation name but found '\"'"); + }); + + test("o-p59fail1", () => { + // 3.3.1 [59] — at least one required + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected a name token in the enumeration but found ')'"); + }); + + test("o-p59fail2", () => { + // 3.3.1 [59] — separator must be "," + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected '|' or ')' in the enumeration but found ','"); + }); + + test("o-p59fail3", () => { + // 3.3.1 [59] — values are unquoted + const input: string = + '\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected a name token in the enumeration but found '\"'"); + }); + + test("o-p60fail1", () => { + // 3.3.2 [60] — keywords must be upper case + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected #REQUIRED, #IMPLIED, #FIXED or a quoted default value but found '#implied'", + ); + }); + + test("o-p60fail2", () => { + // 3.3.2 [60] — S is required after #FIXED + const input: string = + '\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before a quoted string"); + }); + + test("o-p60fail3", () => { + // 3.3.2 [60] — only #FIXED has both keyword and value + const input: string = + '\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute name or '>' in the ATTLIST declaration but found '\"'", + ); + }); + + test("o-p60fail4", () => { + // 3.3.2 [60] — #FIXED required value + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected a quoted default value after #FIXED but found '>'"); + }); + + test("o-p60fail5", () => { + // 3.3.2 [60] — only one default type + const input: string = + '\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute name or '>' in the ATTLIST declaration but found '#REQUIRED'", + ); + }); + + test("o-p61fail1", () => { + // 3.4 [61] — no other types, including TEMP, which is valid in SGML (upstream: not-wf; external + // parameter entities are not read) + const input: string = '\r\n'; + expectParses(input); + }); + + test("o-p62fail1", () => { + // 3.4 [62] — INCLUDE must be upper case (upstream: not-wf; external parameter entities are not read) + const input: string = '\r\n'; + expectParses(input); + }); + + test("o-p62fail2", () => { + // 3.4 [62] — no spaces in terminating delimiter (upstream: not-wf; external parameter entities are not + // read) + const input: string = '\r\n'; + expectParses(input); + }); + + test("o-p63fail1", () => { + // 3.4 [63] — IGNORE must be upper case (upstream: not-wf; external parameter entities are not read) + const input: string = '\r\n'; + expectParses(input); + }); + + test("o-p63fail2", () => { + // 3.4 [63] — delimiters must be balanced (upstream: not-wf; external parameter entities are not read) + const input: string = '\r\n'; + expectParses(input); + }); + + test("o-p64fail1", () => { + // 3.4 [64] — section delimiters must balance (upstream: not-wf; external parameter entities are not + // read) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("o-p64fail2", () => { + // 3.4 [64] — section delimiters must balance (upstream: not-wf; external parameter entities are not + // read) + const input: string = '\r\n\r\n'; + expectParses(input); + }); + + test("o-p66fail1", () => { + // 4.1 [66] — terminating ';' is required + const input: string = "A"; + expectRejects( + input, + "XML Parse error: Invalid character reference: expected a number followed by ';' but found '<'", + ); + }); + + test("o-p66fail2", () => { + // 4.1 [66] — no S after '&#' + const input: string = "&# 65;"; + expectRejects( + input, + "XML Parse error: Invalid character reference: expected a number followed by ';' but found space", + ); + }); + + test("o-p66fail3", () => { + // 4.1 [66] — no hex digits in numeric reference + const input: string = "&#A;"; + expectRejects( + input, + "XML Parse error: Invalid character reference: expected a number followed by ';' but found 'A'", + ); + }); + + test("o-p66fail4", () => { + // 4.1 [66] — only hex digits in hex references + const input: string = "G;"; + expectRejects( + input, + "XML Parse error: Invalid character reference: expected a number followed by ';' but found 'G'", + ); + }); + + test("o-p66fail5", () => { + // 4.1 [66] — no references to non-characters + const input: string = ""; + expectRejects(input, "XML Parse error: Character reference '' is not a valid XML character"); + }); + + test("o-p66fail6", () => { + // 4.1 [66] — no references to non-characters + const input: string = "��"; + expectRejects(input, "XML Parse error: Character reference '�' is not a valid XML character"); + }); + + test("o-p68fail1", () => { + // 4.1 [68] — terminating ';' is required + const input: string = + '\r\n\r\n]>\r\n\r\n&ent\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected ';' after the entity name but found newline"); + }); + + test("o-p68fail2", () => { + // 4.1 [68] — no S after '&' + const input: string = + '\r\n\r\n]>\r\n\r\n& ent;\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected an entity name after '&' but found space"); + }); + + test("o-p68fail3", () => { + // 4.1 [68] — no S before ';' + const input: string = + '\r\n\r\n]>\r\n\r\n&ent ;\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected ';' after the entity name but found space"); + }); + + test("o-p69fail1", () => { + // 4.1 [69] — terminating ';' is required + const input: string = + '\r\n">\r\n%pe\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected ';' to end the parameter entity reference 'pe'"); + }); + + test("o-p69fail2", () => { + // 4.1 [69] — no S after '%' + const input: string = + '\r\n">\r\n% pe;\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected a markup declaration or ']' in the internal subset but found '%'"); + }); + + test("o-p69fail3", () => { + // 4.1 [69] — no S before ';' + const input: string = + '\r\n">\r\n%pe ;\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected ';' to end the parameter entity reference 'pe'"); + }); + + test("o-p70fail1", () => { + // 4.2 [70] — This is neither + const input: string = + '\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected an entity name or '%' after ' { + // 4.2 [71] — S is required before EntityDef + const input: string = + '\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before a quoted string"); + }); + + test("o-p71fail2", () => { + // 4.2 [71] — Entity name is a Name, not an NMToken + const input: string = + '\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected an entity name or '%' after ' { + // 4.2 [71] — no S after "\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: ' { + // 4.2 [71] — S is required after "\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before 'ge'"); + }); + + test("o-p72fail1", () => { + // 4.2 [72] — S is required after "\r\n">\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before '%'"); + }); + + test("o-p72fail2", () => { + // 4.2 [72] — S is required after '%' + const input: string = + '\r\n">\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Whitespace is required between '%' and the name in a parameter entity declaration", + ); + }); + + test("o-p72fail3", () => { + // 4.2 [72] — S is required after name + const input: string = + '\r\n">\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before a quoted string"); + }); + + test("o-p72fail4", () => { + // 4.2 [72] — Entity name is a name, not an NMToken + const input: string = + '\r\n">\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected the parameter entity name after '%' but found '.pe'"); + }); + + test("o-p73fail1", () => { + // 4.2 [73] — No typed replacement text + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected a quoted entity value, SYSTEM or PUBLIC but found 'CDATA'"); + }); + + test("o-p73fail2", () => { + // 4.2 [73] — Only one replacement value + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected '>' to end the entity declaration but found '\"'"); + }); + + test("o-p73fail3", () => { + // 4.2 [73] — No NDataDecl on replacement text + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected '>' to end the entity declaration but found 'NDATA'"); + }); + + test("o-p73fail4", () => { + // 4.2 [73] — Value is required + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected a quoted entity value, SYSTEM or PUBLIC but found '>'"); + }); + + test("o-p73fail5", () => { + // 4.2 [73] — No NDataDecl without value + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected a quoted entity value, SYSTEM or PUBLIC but found 'NDATA'"); + }); + + test("o-p74fail1", () => { + // 4.2 [74] — no NDataDecls on parameter entities + const input: string = + '\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Parameter entities cannot have NDATA"); + }); + + test("o-p74fail2", () => { + // 4.2 [74] — value is required + const input: string = + '\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected a quoted entity value, SYSTEM or PUBLIC but found '>'"); + }); + + test("o-p74fail3", () => { + // 4.2 [74] — only one value + const input: string = '" SYSTEM "nop.ent">\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected '>' to end the entity declaration but found 'SYSTEM'"); + }); + + test("o-p75fail1", () => { + // 4.2.2 [75] — S required after "PUBLIC" + const input: string = '\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before a quoted string"); + }); + + test("o-p75fail2", () => { + // 4.2.2 [75] — S required after "SYSTEM" + const input: string = '\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before a quoted string"); + }); + + test("o-p75fail3", () => { + // 4.2.2 [75] — S required between literals + const input: string = '\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before a quoted string"); + }); + + test("o-p75fail4", () => { + // 4.2.2 [75] — "SYSTEM" implies only one literal + const input: string = '\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected '>' to end the entity declaration but found '\"'"); + }); + + test("o-p75fail5", () => { + // 4.2.2 [75] — only one keyword + const input: string = '\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Expected a quoted system identifier after the public identifier but found 'SYSTEM'", + ); + }); + + test("o-p75fail6", () => { + // 4.2.2 [75] — "PUBLIC" requires two literals (contrast with SGML) + const input: string = '\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Expected a quoted system identifier after the public identifier but found '>'", + ); + }); + + test("o-p76fail1", () => { + // 4.2.2 [76] — S is required before "NDATA" + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before 'NDATA'"); + }); + + test("o-p76fail2", () => { + // 4.2.2 [76] — "NDATA" is upper-case + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected '>' to end the entity declaration but found 'ndata'"); + }); + + test("o-p76fail3", () => { + // 4.2.2 [76] — notation name is required + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected a notation name after NDATA but found '>'"); + }); + + test("o-p76fail4", () => { + // 4.2.2 [76] — notation names are Names + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected a notation name after NDATA but found '-unknot'"); + }); + + test("o-p11pass1", () => { + // 2.3, 4.2.2 [11] — system literals may not contain URI fragments (upstream: optional error) + const input: string = + '\r\n\r\n?>/\\\'\'">\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectParses(input); + }); +}); + +describe("ibm", () => { + test("ibm-invalid-P28-ibm28i01.xml", () => { + // 2.8 — The test violates VC:Root Element Type in P28. The Name in the document type declaration does + // not match the element type of the root element. + const input = Buffer.from( + "\r\n\r\n]>\r\n \r\n", + ); + const canonical = ""; + const compact: unknown = { animal: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P32-ibm32i01.xml", () => { + // 2.9 — This test violates VC: Standalone Document Declaration in P32. The standalone document + // declaration has the value yes, BUT there is an external markup declaration of attributes with + // default values, and the associated element appears in the document with specified values for those + // attributes. (upstream: invalid; external parameter entities are not read; output depends on them) + const input: string = + '\r\n\r\n]>\r\n\r\n\r\n'; + expectParses(input); + }); + + test("ibm-invalid-P32-ibm32i03.xml", () => { + // 2.9 — This test violates VC: Standalone Document Declaration in P32. The standalone document + // declaration has the value yes, BUT there is an external markup declaration of attributes with values + // that will change if normalized. (upstream: invalid; external parameter entities are not read; output + // depends on them) + const input: string = + '\r\n\r\n]>\r\n\r\n\r\n\r\n\r\n\r\n\r\n'; + expectParses(input); + }); + + test("ibm-invalid-P32-ibm32i04.xml", () => { + // 2.9 — This test violates VC: Standalone Document Declaration in P32. The standalone document + // declaration has the value yes, BUT there is an external markup declaration of element with element + // content, and white space occurs directly within the mixed content. (upstream: invalid; external + // parameter entities are not read) + const input: string = + "\r\n\r\n]>\r\n\r\nThis is a \r\n\r\nyellow tiger \r\n\r\n\r\n\r\n"; + const canonical = + 'This is a yellow tiger '; + const compact: unknown = { + animal: { "@xml:space": "preserve", a: "This is a \n\nyellow tiger", b: "", c: "" }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P39-ibm39i01.xml", () => { + // 3 — This test violates VC: Element Valid in P39. Element a is declared empty in DTD, but has content + // in the document. + const input: string = + '\r\n\r\n \r\n \r\n \r\n\r\n]>\r\n\r\nshould not have content here\r\n \r\n content of b element\r\n\r\n\r\n'; + const canonical = + "should not have content here content of b element "; + const compact: unknown = { + root: { + a: "should not have content here", + b: { c: "", "#text": "content of b element" }, + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P39-ibm39i02.xml", () => { + // 3 — This test violates VC: Element Valid in P39. root is declared only having element children in + // DTD, but have text content in the document. + const input: string = + '\r\n\r\n \r\n \r\n \r\n\r\n]>\r\n\r\n\r\n root can\'t have text content\r\n\r\n \r\n content of b element\r\n\r\n\r\n'; + const canonical = + " root can't have text content content of b element "; + const compact: unknown = { + root: { + a: "", + b: { c: "", "#text": "content of b element" }, + "#text": "root can't have text content", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P39-ibm39i03.xml", () => { + // 3 — This test violates VC: Element Valid in P39. Illegal elements are inserted in b's content of + // Mixed type. + const input: string = + "\r\n\r\n \r\n \r\n \r\n]>\r\n\r\n\r\n \r\n content of b element\r\n \r\n could not have 'a' as 'b's content\r\n\r\n\r\n"; + const canonical = + " content of b element could not have 'a' as 'b's content "; + const compact: unknown = { + root: { + a: "", + b: { + c: "", + a: "", + "#text": "content of b element\n \n could not have 'a' as 'b's content", + }, + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P39-ibm39i04.xml", () => { + // 3 — This test violates VC: Element Valid in P39. Element c has undeclared element as its content of + // ANY type + const input: string = + '\r\n\r\n \r\n \r\n \r\n \r\n]>\r\n\r\n\r\n \r\n content of b element\r\n \r\n not declared in dtd\r\n \r\n\r\n\r\n'; + const canonical = + " content of b element not declared in dtd "; + const compact: unknown = { + root: { + a: "", + b: { + c: [{ f: "" }, { d: "not declared in dtd" }], + "#text": "content of b element", + }, + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P41-ibm41i01.xml", () => { + // 3.1 — This test violates VC: Attribute Value Type in P41. attr1 for Element b is not declared. + const input: string = + '\r\n\r\n \r\n \r\n \r\n]>\r\n\r\n attr1 not declared\r\n\r\n\r\n'; + const canonical = ' attr1 not declared '; + const compact: unknown = { + root: { + b: { + "@attr1": "value1", + "@attr2": "def", + "@attr3": "fixed", + "#text": "attr1 not declared", + }, + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P41-ibm41i02.xml", () => { + // 3.1 — This test violates VC: Attribute Value Type in P41. attr3 for Element b is given a value that + // does not match the declaration in the DTD. + const input: string = + '\r\n\r\n \r\n \r\n \r\n \r\n]>\r\n\r\n attr3 value not fixed\r\n\r\n\r\n'; + const canonical = + ' attr3 value not fixed '; + const compact: unknown = { + root: { + b: { + "@attr1": "value1", + "@attr2": "abc", + "@attr3": "shoudbefixed", + "#text": "attr3 value not fixed", + }, + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P45-ibm45i01.xml", () => { + // 3.2 — This test violates VC: Unique Element Type Declaration. Element not_unique has been declared 3 + // time in the DTD. + const input: string = + '\r\n\r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n]>\r\n\r\n without white space\r\n with a white space\r\n \r\n \r\n\r\n\r\n'; + const canonical = + ' without white space with a white space '; + const compact: unknown = { + root: { + b: ["", "", { "@attr1": "value1" }, { "@attr1": "value1", "@attr2": "value2", "@attr3": "value3" }], + "#text": "without white space\n with a white space", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P49-ibm49i01.xml", () => { + // 3.2.1 — Violates VC:Proper Group/PE Nesting in P49. Open and close parenthesis for a choice content + // model are in different PE replace Texts. (upstream: invalid; external parameter entities are not + // read) + const input: string = + '\r\n\r\n]>\r\n\r\n \r\n content of b element\r\n\r\n\r\n'; + const canonical = " content of b element "; + const compact: unknown = { root: { a: "", b: { c: "", "#text": "content of b element" } } }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P50-ibm50i01.xml", () => { + // 3.2.1 — Violates VC:Proper Group/PE Nesting in P50. Open and close parenthesis for a seq content + // model are in different PE replace Texts. (upstream: invalid; external parameter entities are not + // read) + const input: string = + '\r\n\r\n]>\r\n\r\n \r\n content of b element\r\n\r\n\r\n'; + const canonical = " content of b element "; + const compact: unknown = { root: { a: "", b: { c: "", "#text": "content of b element" } } }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P51-ibm51i01.xml", () => { + // 3.2.2 — Violates VC:Proper Group/PE Nesting in P51. Open and close parenthesis for a Mixed content + // model are in different PE replace Texts. (upstream: invalid; external parameter entities are not + // read) + const input: string = + '\r\n\r\n]>\r\n\r\n Element type a \r\n Element type b \r\n\r\n'; + const canonical = " Element type a Element type b "; + const compact: unknown = { root: { a: "Element type a", b: "Element type b" } }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P51-ibm51i03.xml", () => { + // 3.2.2 — Violates VC:No Duplicate Types in P51. Element a appears twice in the Mixed content model of + // Element e. + const input: string = + '\r\n\r\n \r\n \r\n \r\n \r\n \r\n \r\n]>\r\n\r\n Element type a \r\n Element type b \r\n\r\n\r\n'; + const canonical = " Element type a Element type b "; + const compact: unknown = { root: { a: "Element type a", b: "Element type b" } }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P56-ibm56i01.xml", () => { + // 3.3.1 — Tests invalid TokenizedType which is against P56 VC: ID. The value of the ID attribute + // "UniqueName" is "@999" which does not meet the Name production. + const input: string = + '\r\n\r\n\r\n \r\n ]>\r\n\r\nThis is a negative test for validity constraints\r\nthe value of the attribute with a type ID does not match the Name production\r\n'; + const canonical = + ' This is a negative test for validity constraints the value of the attribute with a type ID does not match the Name production '; + const compact: unknown = { + tokenizer: { + "@UniqueName": "@c999", + "#text": + "This is a negative test for validity constraints\nthe value of the attribute with a type ID does not match the Name production", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P56-ibm56i02.xml", () => { + // 3.3.1 — Tests invalid TokenizedType which is against P56 VC: ID. The two ID attributes "attr" and + // "UniqueName" have the same value "Ac999" for the element "b" and the element "tokenizer". + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n ]>\r\n\r\n\r\nThis is a negative test for validity constraints\r\nthe value of the attribute with a type ID appears more than once in the XML document\r\n\r\n'; + const canonical = + ' This is a negative test for validity constraints the value of the attribute with a type ID appears more than once in the XML document '; + const compact: unknown = { + tokenizer: { + "@UniqueName": "Ac999", + b: { "@attr": "Ac999" }, + "#text": + "This is a negative test for validity constraints\nthe value of the attribute with a type ID appears more than once in the XML document", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P56-ibm56i03.xml", () => { + // 3.3.1 — Tests invalid TokenizedType which is against P56 VC: ID Attribute Default. The "#FIXED" + // occurs in the DefaultDecl for the ID attribute "UniqueName". + const input: string = + '\r\n\r\n\r\n \r\n ]>\r\n\r\nThis is a Negative validity test for ID Attribute Default.\r\nGiving the attribute default as #FIXED\r\n\r\n'; + const canonical = + ' This is a Negative validity test for ID Attribute Default. Giving the attribute default as #FIXED '; + const compact: unknown = { + tokenizer: { + "@UniqueName": "AC1999", + "#text": "This is a Negative validity test for ID Attribute Default.\nGiving the attribute default as #FIXED", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P56-ibm56i05.xml", () => { + // 3.3.1 — Tests invalid TokenizedType which is against P56 VC: ID Attribute Default. The constant + // string "BOGUS" occurs in the DefaultDecl for the ID attribute "UniqueName". + const input: string = + '\r\n\r\n\r\n \r\n ]>\r\n\r\nThis is a Negative validity test for ID Attribute Default.\r\nGiving the attibute default as a const string\r\n\r\n'; + const canonical = + ' This is a Negative validity test for ID Attribute Default. Giving the attibute default as a const string '; + const compact: unknown = { + tokenizer: { + "@UniqueName": "AC1999", + "#text": + "This is a Negative validity test for ID Attribute Default.\nGiving the attibute default as a const string", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P56-ibm56i06.xml", () => { + // 3.3.1 — Tests invalid TokenizedType which is against P56 VC: One ID per Element Type. The element + // "a" has two ID attributes "first" and "second". + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n ]>\r\n\r\n\r\nThis is a Negative validity test for ID.\r\nThere is more than attribute of type ID for the element a\r\n\r\n\r\n'; + const canonical = + ' This is a Negative validity test for ID. There is more than attribute of type ID for the element a '; + const compact: unknown = { + tokenizer: { + a: { "@first": "AC1999", "@second": "BC1999" }, + "#text": "This is a Negative validity test for ID.\nThere is more than attribute of type ID for the element a", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P56-ibm56i07.xml", () => { + // 3.3.1 — Tests invalid TokenizedType which is against P56 VC: IDREF. The value of the IDREF attribute + // "reference" is "@456" which does not meet the Name production. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n ]>\r\n\r\n\r\n\r\nNegative test for validity constraint of IDREF.\r\nIn an attribute decl, values of type IDREF does not match the name production\r\n'; + const canonical = + ' Negative test for validity constraint of IDREF. In an attribute decl, values of type IDREF does not match the name production '; + const compact: unknown = { + test: { + id: { "@UniqueName": "AC456" }, + idref: { "@reference": "@456" }, + "#text": + "Negative test for validity constraint of IDREF.\nIn an attribute decl, values of type IDREF does not match the name production", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P56-ibm56i08.xml", () => { + // 3.3.1 — Tests invalid TokenizedType which is against P56 VC: IDREF. The value of the IDREF attribute + // "reference" is "BC456" which does not match the value assigned to any ID attributes. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n ]>\r\n\r\n\r\n\r\nNegative test for validity constraint of IDREF.\r\nIn an attribute decl, values of type IDREF match the name production and\r\nIDREF value does not match the value assigned to any ID attribute somewhere\r\nin the XML document.\r\n'; + const canonical = + ' Negative test for validity constraint of IDREF. In an attribute decl, values of type IDREF match the name production and IDREF value does not match the value assigned to any ID attribute somewhere in the XML document. '; + const compact: unknown = { + test: { + id: { "@UniqueName": "AC456" }, + idref: { "@reference": "BC456" }, + "#text": + "Negative test for validity constraint of IDREF.\nIn an attribute decl, values of type IDREF match the name production and\nIDREF value does not match the value assigned to any ID attribute somewhere\nin the XML document.", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P56-ibm56i09.xml", () => { + // 3.3.1 — Tests invalid TokenizedType which is against P56 VC: IDREFS. The value of the IDREFS + // attribute "reference" is "AC456 #567" which does not meet the Names production. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n \r\n \r\n ]>\r\n\r\n\r\n\r\n\r\nNegative test for validity constraint of IDREFS.\r\nIn an attribute decl, values of type IDREFS does not match the name production\r\n'; + const canonical = + ' Negative test for validity constraint of IDREFS. In an attribute decl, values of type IDREFS does not match the name production '; + const compact: unknown = { + test: { + id1: { "@UniqueName": "AC456" }, + id2: { "@UName": "BC567" }, + idrefs: { "@reference": "AC456 #567" }, + "#text": + "Negative test for validity constraint of IDREFS.\nIn an attribute decl, values of type IDREFS does not match the name production", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P56-ibm56i10.xml", () => { + // 3.3.1 — Tests invalid TokenizedType which is against P56 VC: IDREFS. The value of the IDREFS + // attribute "reference" is "EF456 DE355" which does not match the values assigned to two ID + // attributes. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n \r\n \r\n ]>\r\n\r\n\r\n\r\n\r\nNegative test for validity constraint of IDREFS.\r\nIn an attribute decl, values of type IDREFS match the name production\r\nbut IDREFS value do not match the values assigned to one or more ID attributes\r\nsomewhere in the XML document\r\n'; + const canonical = + ' Negative test for validity constraint of IDREFS. In an attribute decl, values of type IDREFS match the name production but IDREFS value do not match the values assigned to one or more ID attributes somewhere in the XML document '; + const compact: unknown = { + test: { + id1: { "@UniqueName": "BC456" }, + id2: { "@UName": "AC567" }, + idrefs: { "@reference": "EF456 DE355" }, + "#text": + "Negative test for validity constraint of IDREFS.\nIn an attribute decl, values of type IDREFS match the name production\nbut IDREFS value do not match the values assigned to one or more ID attributes\nsomewhere in the XML document", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P56-ibm56i11.xml", () => { + // 3.3.1 — Tests invalid TokenizedType which is against P56 VC: Entity Name. The value of the ENTITY + // attribute "sun" is "ima ge" which does not meet the Name production. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n]>\r\n\r\n\r\nIn the attribute decl, values of type ENTITY do not match the Name production\r\n'; + const canonical = + ' In the attribute decl, values of type ENTITY do not match the Name production '; + const compact: unknown = { + test: { + landscape: { "@sun": "ima ge" }, + "#text": "In the attribute decl, values of type ENTITY do not match the Name production", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P56-ibm56i12.xml", () => { + // 3.3.1 — Tests invalid TokenizedType which is against P56 VC: Entity Name. The value of the ENTITY + // attribute "sun" is "notimage" which does not match the name of any unparsed entity declared. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n]>\r\n\r\n\r\nIn the attribute decl, values of type ENTITY match the Name production\r\nbut does not match the name of any entity declared in the DTD\r\n'; + const canonical = + ' In the attribute decl, values of type ENTITY match the Name production but does not match the name of any entity declared in the DTD '; + const compact: unknown = { + test: { + landscape: { "@sun": "notimage" }, + "#text": + "In the attribute decl, values of type ENTITY match the Name production\nbut does not match the name of any entity declared in the DTD", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P56-ibm56i13.xml", () => { + // 3.3.1 — Tests invalid TokenizedType which is against P56 VC: Entity Name. The value of the ENTITY + // attribute "sun" is "parsedentity" which matches the name of a parsed entity instead of an unparsed + // entity declared. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n]>\r\n\r\n\r\nIn an attribute declaration, values of type ENTITY match the Name production and the ENTITY value\r\nmatches the name of a parsed entity declared in the DTD. \r\n'; + const canonical = + ' In an attribute declaration, values of type ENTITY match the Name production and the ENTITY value matches the name of a parsed entity declared in the DTD. '; + const compact: unknown = { + test: { + landscape: { "@sun": "parsedentity" }, + "#text": + "In an attribute declaration, values of type ENTITY match the Name production and the ENTITY value\nmatches the name of a parsed entity declared in the DTD.", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P56-ibm56i14.xml", () => { + // 3.3.1 — Tests invalid TokenizedType which is against P56 VC: Entity Name. The value of the ENTITIES + // attribute "sun" is "#image1 @image" which does not meet the Names production. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n]>\r\n\r\n\r\nIn an attribute declaration, values of type ENTITIES do not match the Name production.\r\n'; + const canonical = + ' In an attribute declaration, values of type ENTITIES do not match the Name production. '; + const compact: unknown = { + test: { + landscape: { "@sun": "#image1 @image" }, + "#text": "In an attribute declaration, values of type ENTITIES do not match the Name production.", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P56-ibm56i15.xml", () => { + // 3.3.1 — Tests invalid TokenizedType which is against P56 VC: ENTITIES. The value of the ENTITIES + // attribute "sun" is "image3 image4" which does not match the names of two unparsed entities declared. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n]>\r\n\r\n\r\nIn an attribute declaration, values of type ENTITIES match the Name production and the ENTITIES value\r\ndoes not match one or more names of entities declared in the DTD. \r\n'; + const canonical = + ' In an attribute declaration, values of type ENTITIES match the Name production and the ENTITIES value does not match one or more names of entities declared in the DTD. '; + const compact: unknown = { + test: { + landscape: { "@sun": "image3 image4" }, + "#text": + "In an attribute declaration, values of type ENTITIES match the Name production and the ENTITIES value\ndoes not match one or more names of entities declared in the DTD.", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P56-ibm56i16.xml", () => { + // 3.3.1 — Tests invalid TokenizedType which is against P56 VC: ENTITIES. The value of the ENTITIES + // attribute "sun" is "parsedentity1 parsedentity2" which matches the names of two parsed entities + // instead of two unparsed entities declared. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n]>\r\n\r\n\r\nIn an attribute declaration, values of type ENTITIES match the Name production and the ENTITIES value\r\nmatches one or more names of parsed entities declared in the DTD. .\r\n'; + const canonical = + ' In an attribute declaration, values of type ENTITIES match the Name production and the ENTITIES value matches one or more names of parsed entities declared in the DTD. . '; + const compact: unknown = { + test: { + landscape: { "@sun": "parsedentity1 parsedentity2" }, + "#text": + "In an attribute declaration, values of type ENTITIES match the Name production and the ENTITIES value\nmatches one or more names of parsed entities declared in the DTD. .", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P56-ibm56i17.xml", () => { + // 3.3.1 — Tests invalid TokenizedType which is against P56 VC: Name Token. The value of the NMTOKEN + // attribute "thistoken" is "x : image" which does not meet the Nmtoken production. + const input: string = + '\r\n\r\n\r\n \r\n \r\n]>\r\n\r\n\r\nIn an attribute declaration, values of type NMTOKEN does not match the Nmtoken production\r\n'; + const canonical = + ' In an attribute declaration, values of type NMTOKEN does not match the Nmtoken production '; + const compact: unknown = { + test: { + nametoken: { "@thistoken": "x : image" }, + "#text": "In an attribute declaration, values of type NMTOKEN does not match the Nmtoken production", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P56-ibm56i18.xml", () => { + // 3.3.1 — Tests invalid TokenizedType which is against P56 VC: Name Token. The value of the NMTOKENS + // attribute "thistoken" is "@lang y: #country" which does not meet the Nmtokens production. + const input: string = + '\r\n\r\n\r\n \r\n \r\n]>\r\n\r\n\r\nIn an attribute declaration, values of type NMTOKENS does not match the Nmtokens production\r\n'; + const canonical = + ' In an attribute declaration, values of type NMTOKENS does not match the Nmtokens production '; + const compact: unknown = { + test: { + nametokens: { "@thistoken": "@lang y: #country" }, + "#text": "In an attribute declaration, values of type NMTOKENS does not match the Nmtokens production", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P58-ibm58i01.xml", () => { + // 3.3.1 — Tests invalid NotationType which is against P58 VC: Notation Attributes. The attribute + // "content-encoding" with value "raw" is not a value from the list "(base64|uuencode)". + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n \r\n ]>\r\n \r\n\r\nThe attribute values of type NOTATION does not match any of the notation names included in the\r\ndeclaration.All notation names in the declaration have been declared.\r\n'; + const canonical = + ' The attribute values of type NOTATION does not match any of the notation names included in the declaration.All notation names in the declaration have been declared. '; + const compact: unknown = { + test: { + blob: { "@content-encoding": "raw" }, + "#text": + "The attribute values of type NOTATION does not match any of the notation names included in the\ndeclaration.All notation names in the declaration have been declared.", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P58-ibm58i02.xml", () => { + // 3.3.1 — Tests invalid NotationType which is against P58 VC: Notation Attributes. The attribute + // "content-encoding" with value "raw" is a value from the list "(base64|uuencode|raw|ascii)", but + // "raw" is not a declared notation. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n ]>\r\n \r\n\r\nThe attribute values of type NOTATION does match any of the notation names included in the\r\ndeclaration, but some of notation names in the declaration have not been declared\r\n'; + const canonical = + ' The attribute values of type NOTATION does match any of the notation names included in the declaration, but some of notation names in the declaration have not been declared '; + const compact: unknown = { + test: { + blob: { "@content-encoding": "raw" }, + "#text": + "The attribute values of type NOTATION does match any of the notation names included in the\ndeclaration, but some of notation names in the declaration have not been declared", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P59-ibm59i01.xml", () => { + // 3.3.1 — Tests invalid Enumeration which is against P59 VC: Enumeration. The value of the attribute + // is "ONE" which matches neither "one" nor "two" as declared in the Enumeration in the AttDef in the + // AttlistDecl. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n ]>\r\n \r\n\r\nThis is a Negative test\r\nThe attribute values of type Enumeration does not match any of the Nmtoken tokens in the declaration.\r\n'; + const canonical = + ' This is a Negative test The attribute values of type Enumeration does not match any of the Nmtoken tokens in the declaration. '; + const compact: unknown = { + test: { + num: { "@value": "ONE" }, + "#text": + "This is a Negative test\nThe attribute values of type Enumeration does not match any of the Nmtoken tokens in the declaration.", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P60-ibm60i01.xml", () => { + // 3.3.2 — Tests invalid DefaultDecl which is against P60 VC: Required Attribute. The attribute + // "chapter" for the element "two" is declared as #REQUIRED in the DefaultDecl in the AttlistDecl, but + // the value of this attribute is not given. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n ]>\r\n\r\n\r\n\r\nNegative test for Required Attribute. Some occurrence of an element with \r\nan attribute of #REQUIRED default declaration does not give the value of \r\nthose attribute\r\n'; + const canonical = + ' Negative test for Required Attribute. Some occurrence of an element with an attribute of #REQUIRED default declaration does not give the value of those attribute '; + const compact: unknown = { + Java: { + one: { "@chapter": "Introduction" }, + two: "", + "#text": + "Negative test for Required Attribute. Some occurrence of an element with \nan attribute of #REQUIRED default declaration does not give the value of \nthose attribute", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P60-ibm60i02.xml", () => { + // 3.3.2 — Tests invalid DefaultDecl which is against P60 VC: Fixed Attribute Default.. The attribute + // "chapter" for the element "one" is declared as #FIXED with the given value "Introduction" in the + // DefaultDecl in the AttlistDecl, but the value of a instance of this attribute is assigned to + // "JavaBeans". + const input: string = + '\r\n\r\n\r\n \r\n \r\n ]>\r\n\r\n\r\nNegative Test\r\nAn attribute has a default value declared with the #FIXED keyword, \r\nand an instances of that attribute is given a value which is not \r\nthe same as the default value in the declaration. \r\n\r\n'; + const canonical = + ' Negative Test An attribute has a default value declared with the #FIXED keyword, and an instances of that attribute is given a value which is not the same as the default value in the declaration. '; + const compact: unknown = { + Java: { + one: { "@chapter": "JavaBeans" }, + "#text": + "Negative Test\nAn attribute has a default value declared with the #FIXED keyword, \nand an instances of that attribute is given a value which is not \nthe same as the default value in the declaration.", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P60-ibm60i03.xml", () => { + // 3.3.2 — Tests invalid DefaultDecl which is against P60 VC: Attribute Default Legal. The declared + // default value "c" is not legal for the type (a|b) in the AttDef in the AttlistDecl. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n ]>\r\n\r\nThe default value specified for an attribute does not meet the \r\nlexical constraints of the declared attribute type.\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n'; + const canonical = + " The default value specified for an attribute does not meet the lexical constraints of the declared attribute type. "; + const compact: unknown = { + test: "The default value specified for an attribute does not meet the \nlexical constraints of the declared attribute type.", + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P60-ibm60i04.xml", () => { + // 3.3.2 — Tests invalid DefaultDecl which is against P60 VC: Attribute Default Legal. The declared + // default value "@#$" is not legal for the type NMTOKEN the AttDef in the AttlistDecl. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n ]>\r\n\r\nThe default value specified for an attribute does not meet the \r\nlexical constraints of the declared attribute type.\r\n\r\n'; + const canonical = + " The default value specified for an attribute does not meet the lexical constraints of the declared attribute type. "; + const compact: unknown = { + test: "The default value specified for an attribute does not meet the \nlexical constraints of the declared attribute type.", + }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P68-ibm68i01.xml", () => { + // 4.1 — Tests invalid EntityRef which is against P68 VC: Entity Declared. The GE with the name "ge2" + // is referred in the file ibm68i01.dtd", but not declared. (upstream: optional error) + const input: string = + '\r\n\r\n]>\r\n\r\n pcdata content\r\n \r\n\r\n\r\n\r\n'; + const canonical = ' pcdata content '; + const compact: unknown = { root: { a: { "@attr1": "xyz" }, "#text": "pcdata content" } }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P68-ibm68i02.xml", () => { + // 4.1 — Tests invalid EntityRef which is against P68 VC: Entity Declared. The GE with the name "ge1" + // is referred before declared in the file ibm68i01.dtd". (upstream: optional error) + const input: string = + '\r\n\r\n]>\r\n\r\n pcdata content\r\n \r\n\r\n\r\n\r\n'; + const canonical = ' pcdata content '; + const compact: unknown = { root: { a: { "@attr1": "xyz" }, "#text": "pcdata content" } }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P68-ibm68i03.xml", () => { + // 4.1 — Tests invalid EntityRef which is against P68 VC: Entity Declared. The GE with the name "ge2" + // is referred in the file ibm68i03.ent", but not declared. (upstream: optional error) + const input: string = + '\r\n\r\n \r\n %pe1;\r\n]>\r\n\r\n pcdata content\r\n\r\n\r\n'; + const canonical = " pcdata content "; + const compact: unknown = { root: "pcdata content" }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P68-ibm68i04.xml", () => { + // 4.1 — Tests invalid EntityRef which is against P68 VC: Entity Declared. The GE with the name "ge1" + // is referred before declared in the file ibm68i04.ent". (upstream: optional error) + const input: string = + '\r\n\r\n \r\n %pe1;\r\n]>\r\n\r\n pcdata content\r\n\r\n\r\n'; + const canonical = " pcdata content "; + const compact: unknown = { root: "pcdata content" }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P69-ibm69i01.xml", () => { + // 4.1 — Tests invalid PEReference which is against P69 VC: Entity Declared. The Name "pe2" in the + // PEReference in the file ibm69i01.dtd does not match the Name of any declared PE. (upstream: optional + // error) + const input: string = + '\r\n\r\n]>\r\n\r\n pcdata content\r\n \r\n\r\n\r\n\r\n'; + const canonical = ' pcdata content '; + const compact: unknown = { root: { a: { "@attr1": "xyz" }, "#text": "pcdata content" } }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P69-ibm69i02.xml", () => { + // 4.1 — Tests invalid PEReference which is against P69 VC: Entity Declared. The PE with the name "pe1" + // is referred before declared in the file ibm69i02.dtd (upstream: optional error) + const input: string = + '\r\n\r\n]>\r\n\r\n pcdata content\r\n \r\n\r\n\r\n\r\n'; + const canonical = ' pcdata content '; + const compact: unknown = { root: { a: { "@attr1": "xyz" }, "#text": "pcdata content" } }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P69-ibm69i03.xml", () => { + // 4.1 — Tests invalid PEReference which is against P69 VC: Entity Declared. The Name "pe3" in the + // PEReference in the file ibm69i03.ent does not match the Name of any declared PE. (upstream: optional + // error) + const input: string = + '\r\n\r\n \r\n %pe1;\r\n]>\r\n\r\n pcdata content\r\n\r\n\r\n'; + const canonical = " pcdata content "; + const compact: unknown = { root: "pcdata content" }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P69-ibm69i04.xml", () => { + // 4.1 — Tests invalid PEReference which is against P69 VC: Entity Declared. The PE with the name "pe2" + // is referred before declared in the file ibm69i04.ent. (upstream: optional error) + const input: string = + '\r\n\r\n \r\n %pe1;\r\n]>\r\n\r\n pcdata content\r\n\r\n\r\n'; + const canonical = " pcdata content "; + const compact: unknown = { root: "pcdata content" }; + expectParses(input, canonical, compact); + }); + + test("ibm-invalid-P76-ibm76i01.xml", () => { + // 4.2.2 — Tests invalid NDataDecl which is against P76 VC: Notation declared. The Name "JPGformat" in + // the NDataDecl in the EntityDecl for "ge2" does not match the Name of any declared notation. + const input: string = + '\r\n\r\n\r\n\r\n\'>\r\n\r\n%pe1;\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { root: { "@att2": "any" } }; + expectParses(input, canonical, compact); + }); + + test("ibm-not-wf-P01-ibm01n01.xml", () => { + // 2.1 — Tests a document with no element. A well-formed document should have at lease one elements. + const input: string = + '\r\n\r\n]>\r\n'; + expectRejects(input, "XML Parse error: XML document must have a root element"); + }); + + test("ibm-not-wf-P01-ibm01n02.xml", () => { + // 2.1 — Tests a document with wrong ordering of its prolog and element. The element occurs before the + // xml declaration and the DTD. + const input: string = + 'Wrong ordering between prolog and element!\r\n\r\n\r\n]>'; + expectRejects( + input, + "XML Parse error: ' { + // 2.1 — Tests a document with wrong combination of misc and element. One PI occurs between two + // elements. + const input: string = + '\r\n\r\n \r\n]>\r\nWrong combination!\r\n\r\nWrong combination!\r\n\r\n'; + expectRejects(input, "XML Parse error: Only one root element is allowed"); + }); + + test("ibm-not-wf-P02-ibm02n01.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x00 + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x00"); + }); + + test("ibm-not-wf-P02-ibm02n02.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x01 + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x01"); + }); + + test("ibm-not-wf-P02-ibm02n03.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x02 + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x02"); + }); + + test("ibm-not-wf-P02-ibm02n04.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x03 + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x03"); + }); + + test("ibm-not-wf-P02-ibm02n05.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x04 + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x04"); + }); + + test("ibm-not-wf-P02-ibm02n06.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x05 + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x05"); + }); + + test("ibm-not-wf-P02-ibm02n07.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x06 + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x06"); + }); + + test("ibm-not-wf-P02-ibm02n08.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x07 + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x07"); + }); + + test("ibm-not-wf-P02-ibm02n09.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x08 + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x08"); + }); + + test("ibm-not-wf-P02-ibm02n10.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x0B + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x0B"); + }); + + test("ibm-not-wf-P02-ibm02n11.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x0C + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x0C"); + }); + + test("ibm-not-wf-P02-ibm02n12.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x0E + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x0E"); + }); + + test("ibm-not-wf-P02-ibm02n13.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x0F + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x0F"); + }); + + test("ibm-not-wf-P02-ibm02n14.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x10 + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x10"); + }); + + test("ibm-not-wf-P02-ibm02n15.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x11 + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x11"); + }); + + test("ibm-not-wf-P02-ibm02n16.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x12 + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x12"); + }); + + test("ibm-not-wf-P02-ibm02n17.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x13 + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x13"); + }); + + test("ibm-not-wf-P02-ibm02n18.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x14 + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x14"); + }); + + test("ibm-not-wf-P02-ibm02n19.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x15 + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x15"); + }); + + test("ibm-not-wf-P02-ibm02n20.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x16 + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x16"); + }); + + test("ibm-not-wf-P02-ibm02n21.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x17 + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x17"); + }); + + test("ibm-not-wf-P02-ibm02n22.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x18 + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x18"); + }); + + test("ibm-not-wf-P02-ibm02n23.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x19 + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x19"); + }); + + test("ibm-not-wf-P02-ibm02n24.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x1A + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x1A"); + }); + + test("ibm-not-wf-P02-ibm02n25.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x1B + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x1B"); + }); + + test("ibm-not-wf-P02-ibm02n26.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x1C + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x1C"); + }); + + test("ibm-not-wf-P02-ibm02n27.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x1D + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x1D"); + }); + + test("ibm-not-wf-P02-ibm02n28.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x1E + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x1E"); + }); + + test("ibm-not-wf-P02-ibm02n29.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #x1F + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x1F"); + }); + + test("ibm-not-wf-P02-ibm02n30.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #xD800 + const input = Buffer.from( + "PCFET0NUWVBFIGJvb2sgWw0KPCFFTEVNRU5UIGJvb2sgQU5ZPg0KXT4NCjwhLS0gSWxsZWdhbENoYXIgI3hkODAwDQogaW4gcDAyOiDtoIAgLS0+DQo8Ym9vay8+DQo=", + "base64", + ); + expectRejects(input, "XML Parse error: Invalid UTF-8"); + }); + + test("ibm-not-wf-P02-ibm02n31.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #xDFFF + const input = Buffer.from( + "PCFET0NUWVBFIGJvb2sgWw0KPCFFTEVNRU5UIGJvb2sgQU5ZPg0KXT4NCjwhLS0gSWxsZWdhbENoYXIgI3hkZmZmDQogaW4gcDAyOiDtv78gLS0+DQo8Ym9vay8+DQo=", + "base64", + ); + expectRejects(input, "XML Parse error: Invalid UTF-8"); + }); + + test("ibm-not-wf-P02-ibm02n32.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #xFFFE + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: '\ufffe' (U+FFFE)"); + }); + + test("ibm-not-wf-P02-ibm02n33.xml", () => { + // 2.2 — Tests a comment which contains an illegal Char: #xFFFF + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: '\uffff' (U+FFFF)"); + }); + + test("ibm-not-wf-P03-ibm03n01.xml", () => { + // 2.3 — Tests an end tag which contains an illegal space character #x3000 which follows the element + // name "book". + const input: string = + "\r\n]>\r\n\r\nIllegal space 3000 in the end tag\r\n"; + expectRejects(input, "XML Parse error: Expected '>' to end the closing tag but found ' ' (U+3000)"); + }); + + test("ibm-not-wf-P04-ibm04n01.xml", () => { + // 2.3 — Tests an element name which contains an illegal ASCII NameChar. "IllegalNameChar" is followed + // by #x21 + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '!'", + ); + }); + + test("ibm-not-wf-P04-ibm04n02.xml", () => { + // 2.3 — Tests an element name which contains an illegal ASCII NameChar. "IllegalNameChar" is followed + // by #x28 + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '('", + ); + }); + + test("ibm-not-wf-P04-ibm04n03.xml", () => { + // 2.3 — Tests an element name which contains an illegal ASCII NameChar. "IllegalNameChar" is followed + // by #x29 + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found ')'", + ); + }); + + test("ibm-not-wf-P04-ibm04n04.xml", () => { + // 2.3 — Tests an element name which contains an illegal ASCII NameChar. "IllegalNameChar" is followed + // by #x2B + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '+'", + ); + }); + + test("ibm-not-wf-P04-ibm04n05.xml", () => { + // 2.3 — Tests an element name which contains an illegal ASCII NameChar. "IllegalNameChar" is followed + // by #x2C + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found ','", + ); + }); + + test("ibm-not-wf-P04-ibm04n06.xml", () => { + // 2.3 — Tests an element name which contains an illegal ASCII NameChar. "IllegalNameChar" is followed + // by #x2F + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected '>' after '/' but found space"); + }); + + test("ibm-not-wf-P04-ibm04n07.xml", () => { + // 2.3 — Tests an element name which contains an illegal ASCII NameChar. "IllegalNameChar" is followed + // by #x3B + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found ';'", + ); + }); + + test("ibm-not-wf-P04-ibm04n08.xml", () => { + // 2.3 — Tests an element name which contains an illegal ASCII NameChar. "IllegalNameChar" is followed + // by #x3C + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected an element name after '<' but found space"); + }); + + test("ibm-not-wf-P04-ibm04n09.xml", () => { + // 2.3 — Tests an element name which contains an illegal ASCII NameChar. "IllegalNameChar" is followed + // by #x3D + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '='", + ); + }); + + test("ibm-not-wf-P04-ibm04n10.xml", () => { + // 2.3 — Tests an element name which contains an illegal ASCII NameChar. "IllegalNameChar" is followed + // by #x3F + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '?'", + ); + }); + + test("ibm-not-wf-P04-ibm04n11.xml", () => { + // 2.3 — Tests an element name which contains an illegal ASCII NameChar. "IllegalNameChar" is followed + // by #x5B + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected a markup declaration or ']' in the internal subset but found '['"); + }); + + test("ibm-not-wf-P04-ibm04n12.xml", () => { + // 2.3 — Tests an element name which contains an illegal ASCII NameChar. "IllegalNameChar" is followed + // by #x5C + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '\\'", + ); + }); + + test("ibm-not-wf-P04-ibm04n13.xml", () => { + // 2.3 — Tests an element name which contains an illegal ASCII NameChar. "IllegalNameChar" is followed + // by #x5D + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found ']'", + ); + }); + + test("ibm-not-wf-P04-ibm04n14.xml", () => { + // 2.3 — Tests an element name which contains an illegal ASCII NameChar. "IllegalNameChar" is followed + // by #x5E + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '^'", + ); + }); + + test("ibm-not-wf-P04-ibm04n15.xml", () => { + // 2.3 — Tests an element name which contains an illegal ASCII NameChar. "IllegalNameChar" is followed + // by #x60 + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '`'", + ); + }); + + test("ibm-not-wf-P04-ibm04n16.xml", () => { + // 2.3 — Tests an element name which contains an illegal ASCII NameChar. "IllegalNameChar" is followed + // by #x7B + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '{'", + ); + }); + + test("ibm-not-wf-P04-ibm04n17.xml", () => { + // 2.3 — Tests an element name which contains an illegal ASCII NameChar. "IllegalNameChar" is followed + // by #x7C + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '|'", + ); + }); + + test("ibm-not-wf-P04-ibm04n18.xml", () => { + // 2.3 — Tests an element name which contains an illegal ASCII NameChar. "IllegalNameChar" is followed + // by #x7D + const input: string = + "\r\n]>\r\n\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '}'", + ); + }); + + test("ibm-not-wf-P05-ibm05n01.xml", () => { + // 2.3 — Tests an element name which has an illegal first character. An illegal first character "." is + // followed by "A_name-starts_with.". + const input: string = + "\r\n]>\r\n<.A_name_starts_with./> \r\n"; + expectRejects(input, "XML Parse error: Expected the document type name but found '.A_name_starts_with.'"); + }); + + test("ibm-not-wf-P05-ibm05n02.xml", () => { + // 2.3 — Tests an element name which has an illegal first character. An illegal first character "-" is + // followed by "A_name-starts_with-". + const input: string = + "\r\n]>\r\n<-A_name_starts_With-/> \r\n"; + expectRejects(input, "XML Parse error: Expected the document type name but found '-A_name_starts_With-'"); + }); + + test("ibm-not-wf-P05-ibm05n03.xml", () => { + // 2.3 — Tests an element name which has an illegal first character. An illegal first character "5" is + // followed by "A_name-starts_with_digit". + const input: string = + "\r\n]>\r\n<5A_name_starts_with_digit/> \r\n"; + expectRejects(input, "XML Parse error: Expected the document type name but found '5A_name_starts_with_digit'"); + }); + + test("ibm-not-wf-P09-ibm09n01.xml", () => { + // 2.3 — Tests an internal general entity with an invalid value. The entity "Fullname" contains "%". + const input: string = + '\r\n \r\n \t\r\n]>\r\n\r\n\r\nMy Name is &FullName;. \r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n '; + expectRejects(input, "XML Parse error: Expected ';' after the entity name but found '\"'"); + }); + + test("ibm-not-wf-P09-ibm09n02.xml", () => { + // 2.3 — Tests an internal general entity with an invalid value. The entity "Fullname" contains the + // ampersand character. + const input: string = + '\r\n \r\n \t\r\n]>\r\n\r\n\r\nMy Name is &FullName;. '; + expectRejects(input, "XML Parse error: Expected ';' after the entity name but found '\"'"); + }); + + test("ibm-not-wf-P09-ibm09n03.xml", () => { + // 2.3 — Tests an internal general entity with an invalid value. The entity "Fullname" contains the + // double quote character in the middle. + const input: string = + ' \r\n\r\n\t \r\n]>\r\n\r\n\r\nMy Name is &FullName;. '; + expectRejects(input, "XML Parse error: Expected '>' to end the entity declaration but found 'Man'"); + }); + + test("ibm-not-wf-P09-ibm09n04.xml", () => { + // 2.3 — Tests an internal general entity with an invalid value. The closing bracket (double quote) is + // missing with the value of the entity "FullName". + const input: string = + ' \r\n\r\n\t \r\n]>\r\n\r\n\r\nMy Name is &FullName;. \r\n'; + expectRejects(input, "XML Parse error: Unterminated entity value"); + }); + + test("ibm-not-wf-P10-ibm10n01.xml", () => { + // 2.3 — Tests an attribute with an invalid value. The value of the attribute "first" contains the + // character "less than". + const input: string = + '\r\n\r\n\t \r\n\t\r\n\t\r\n\t\r\n]>\r\n\r\n\r\nMy Name is SnowMan. \r\n\r\n\r\n\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: '<' is not allowed in attribute values"); + }); + + test("ibm-not-wf-P10-ibm10n02.xml", () => { + // 2.3 — Tests an attribute with an invalid value. The value of the attribute "first" contains the + // character ampersand. + const input: string = + '\r\n\r\n\t \r\n\t\r\n\t\r\n\t\r\n]>\r\n\r\n\r\nMy Name is SnowMan. \r\n'; + expectRejects(input, "XML Parse error: Expected ';' after the entity name but found '\"'"); + }); + + test("ibm-not-wf-P10-ibm10n03.xml", () => { + // 2.3 — Tests an attribute with an invalid value. The value of the attribute "first" contains the + // double quote character in the middle. + const input: string = + '\r\n\r\n\t \r\n\t\r\n\t\r\n\t\r\n]>\r\n\r\n\r\nMy Name is SnowMan. \r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before 'Man'"); + }); + + test("ibm-not-wf-P10-ibm10n04.xml", () => { + // 2.3 — Tests an attribute with an invalid value. The closing bracket (double quote) is missing with + // The value of the attribute "first". + const input: string = + '\r\n\r\n\t \r\n\t\r\n\t\r\n\t\r\n]>\r\n\r\n\r\n { + // 2.3 — Tests an attribute with an invalid value. The value of the attribute "first" contains the + // character "less than". + const input: string = + '\r\n\r\n\t \r\n\t\r\n\t\r\n\t\r\n]>\r\n\r\n\r\nMy Name is SnowMan. \r\n'; + expectRejects(input, "XML Parse error: '<' is not allowed in attribute values"); + }); + + test("ibm-not-wf-P10-ibm10n06.xml", () => { + // 2.3 — Tests an attribute with an invalid value. The value of the attribute "first" contains the + // character ampersand. + const input: string = + '\r\n\r\n\t \r\n\t\r\n\t\r\n\t\r\n]>\r\n\r\n\r\nMy Name is SnowMan. \r\n'; + expectRejects(input, "XML Parse error: Expected ';' after the entity name but found '''"); + }); + + test("ibm-not-wf-P10-ibm10n07.xml", () => { + // 2.3 — Tests an attribute with an invalid value. The value of the attribute "first" contains the + // double quote character in the middle. + const input: string = + '\r\n\r\n\t \r\n\t\r\n\t\r\n\t\r\n]>\r\n\r\n\r\nMy Name is SnowMan. \r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before 'Man'"); + }); + + test("ibm-not-wf-P10-ibm10n08.xml", () => { + // 2.3 — Tests an attribute with an invalid value. The closing bracket (single quote) is missing with + // the value of the attribute "first". + const input: string = + '\r\n\r\n\t \r\n\t\r\n\t\r\n\t\r\n]>\r\n\r\n\r\nMy Name is SnowMan. \r\n'; + expectRejects(input, "XML Parse error: '<' is not allowed in attribute values"); + }); + + test("ibm-not-wf-P11-ibm11n01.xml", () => { + // 2.3 — Tests SystemLiteral. The systemLiteral for the element "student" has a double quote character + // in the middle. + const input: string = + '\r\n \r\n]>\r\n\r\n\r\nMy Name is SnowMan. \r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n '; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '.dtd'", + ); + }); + + test("ibm-not-wf-P11-ibm11n02.xml", () => { + // 2.3 — Tests SystemLiteral. The systemLiteral for the element "student" has a single quote character + // in the middle. + const input: string = + "\r\n \r\n]>\r\n\r\n\r\nMy Name is SnowMan. \r\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '.dtd'", + ); + }); + + test("ibm-not-wf-P11-ibm11n03.xml", () => { + // 2.3 — Tests SystemLiteral. The closing bracket (double quote) is missing with the systemLiteral for + // the element "student". + const input: string = + '\r\n \r\n]>\r\n\r\n\r\nMy Name is SnowMan. \r\n'; + expectRejects(input, "XML Parse error: Unterminated quoted string"); + }); + + test("ibm-not-wf-P11-ibm11n04.xml", () => { + // 2.3 — Tests SystemLiteral. The closing bracket (single quote) is missing with the systemLiteral for + // the element "student". + const input: string = + '\r\n \r\n]>\r\n\r\n\r\nMy Name is SnowMan. \r\n'; + expectRejects(input, "XML Parse error: Unterminated quoted string"); + }); + + test("ibm-not-wf-P12-ibm12n01.xml", () => { + // 2.3 — Tests PubidLiteral. The closing bracket (double quote) is missing with the value of the + // PubidLiteral for the entity "info". + const input: string = + '\r\n \r\n\t\r\n]>\r\n\r\n\r\nMy Name is &info;. \r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n '; + expectRejects(input, "XML Parse error: Invalid character in a public identifier: '\\'"); + }); + + test("ibm-not-wf-P12-ibm12n02.xml", () => { + // 2.3 — Tests PubidLiteral. The value of the PubidLiteral for the entity "info" has a single quote + // character in the middle.. + const input: string = + "\r\n \r\n\t\r\n]>\r\n\r\n\r\nMy Name is &info;. \r\n"; + expectRejects(input, "XML Parse error: Invalid character in a public identifier: '\\'"); + }); + + test("ibm-not-wf-P12-ibm12n03.xml", () => { + // 2.3 — Tests PubidLiteral. The closing bracket (single quote) is missing with the value of the + // PubidLiteral for the entity "info". + const input: string = + '\r\n \r\n\t\r\n]>\r\n\r\n\r\nMy Name is &info;. '; + expectRejects(input, "XML Parse error: Invalid character in a public identifier: '\\'"); + }); + + test("ibm-not-wf-P13-ibm13n01.xml", () => { + // 2.3 — Tests PubidChar. The pubidChar of the PubidLiteral for the entity "info" contains the + // character "{". + const input: string = + '\r\n\r\n]>\r\n\r\n\r\nMy Name is &info;. \r\n\r\n\r\n\r\n\r\n '; + expectRejects(input, "XML Parse error: Invalid character in a public identifier: '{'"); + }); + + test("ibm-not-wf-P13-ibm13n02.xml", () => { + // 2.3 — Tests PubidChar. The pubidChar of the PubidLiteral for the entity "info" contains the + // character "~". + const input: string = + '\r\n\r\n]>\r\n\r\n\r\nMy Name is &info;. '; + expectRejects(input, "XML Parse error: Invalid character in a public identifier: '~'"); + }); + + test("ibm-not-wf-P13-ibm13n03.xml", () => { + // 2.3 — Tests PubidChar. The pubidChar of the PubidLiteral for the entity "info" contains the + // character double quote in the middle. + const input = Buffer.from( + '\n\n]>\n\n\nMy Name is &info;. \n\n', + ); + expectRejects(input, "XML Parse error: Invalid character in a public identifier: 'á' (U+00E1)"); + }); + + test("ibm-not-wf-P14-ibm14n01.xml", () => { + // 2.4 — Tests CharData. The content of the element "student" contains the sequence close-bracket + // close-bracket greater-than. + const input: string = + '\r\n\r\n\t\r\n]>\r\n\r\n\r\nMy name is Snow ]]> Man\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: ']]>' is only allowed as the end of a CDATA section"); + }); + + test("ibm-not-wf-P14-ibm14n02.xml", () => { + // 2.4 — Tests CharData. The content of the element "student" contains the character "less than". + const input: string = + '\r\n\r\n\t\r\n]>\r\n\r\n\r\nMy name is Snow '; + expectRejects( + input, + "XML Parse error: Expected an attribute name, '>' or '/>' in the start tag but found ' { + // 2.4 — Tests CharData. The content of the element "student" contains the character ampersand. + const input: string = + '\r\n\r\n\t\r\n]>\r\n\r\n\r\nMy name is Snow&Man \r\n'; + expectRejects(input, "XML Parse error: Expected ';' after the entity name but found space"); + }); + + test("ibm-not-wf-P15-ibm15n01.xml", () => { + // 2.5 — Tests comment. The text of the second comment contains the character "-". + const input: string = + '\r\n \r\n]>\r\n\r\n\r\n\r\nMy Name is SnowMan. \r\n\r\n\r\n\r\n\r\n\r\n\r\n '; + expectRejects(input, "XML Parse error: '--' is not allowed inside a comment"); + }); + + test("ibm-not-wf-P15-ibm15n02.xml", () => { + // 2.5 — Tests comment. The second comment has a wrong closing sequence "-(greater than)". + const input: string = + '\r\n \r\n]>\r\n\r\n\r\n\r\n\r\nMy Name is SnowMan. '; + expectRejects( + input, + "XML Parse error: ' { + // 2.5 — Tests comment. The closing sequence is missing with the second comment. + const input: string = + '\r\n \r\n]>\r\n\r\n\r\n\r\n a test ?>\r\nMy Name is SnowMan. \r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected the root element but found 'a'"); + }); + + test("ibm-not-wf-P16-ibm16n02.xml", () => { + // 2.6 — Tests PI. The PITarget is missing in the PI. + const input: string = + '\r\n \r\n]>\r\n\r\n\r\n\r\n\r\nMy Name is SnowMan. '; + expectRejects(input, "XML Parse error: Expected a processing instruction target after ' { + // 2.6 — Tests PI. The PI has a wrong closing sequence ">". + const input: string = + '\r\n \r\n]>\r\n\r\n\r\n\r\n\r\nMy Name is SnowMan. '; + expectRejects(input, "XML Parse error: Unterminated processing instruction"); + }); + + test("ibm-not-wf-P16-ibm16n04.xml", () => { + // 2.6 — Tests PI. The closing sequence is missing in the PI. + const input: string = + '\r\n \r\n]>\r\n\r\n\r\n\r\nMy Name is SnowMan. '; + expectRejects(input, "XML Parse error: Unterminated processing instruction"); + }); + + test("ibm-not-wf-P17-ibm17n01.xml", () => { + // 2.6 — Tests PITarget. The PITarget contains the string "XML". + const input: string = + '\r\n \r\n]>\r\n\r\n\r\n\r\n\r\nMy Name is SnowMan. \r\n\r\n\r\n'; + expectRejects( + input, + "XML Parse error: ' { + // 2.6 — Tests PITarget. The PITarget contains the string "xML". + const input: string = + '\r\n \r\n]>\r\n\r\n\r\n\r\nMy Name is SnowMan. '; + expectRejects( + input, + "XML Parse error: ' { + // 2.6 — Tests PITarget. The PITarget contains the string "xml". + const input: string = + '\r\n \r\n]>\r\n\r\n\r\n\r\nMy Name is SnowMan. '; + expectRejects( + input, + "XML Parse error: ' { + // 2.6 — Tests PITarget. The PITarget contains the string "xmL". + const input: string = + '\r\n \r\n]>\r\n\r\n\r\n\r\nMy Name is SnowMan. '; + expectRejects( + input, + "XML Parse error: ' { + // 2.7 — Tests CDSect. The CDStart is missing in the CDSect in the content of element "student". + const input: string = + '\r\n \r\n]>\r\n\r\n\r\nMy Name is SnowMan. This is text]]>\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: ']]>' is only allowed as the end of a CDATA section"); + }); + + test("ibm-not-wf-P18-ibm18n02.xml", () => { + // 2.7 — Tests CDSect. The CDEnd is missing in the CDSect in the content of element "student". + const input: string = + '\r\n \r\n]>\r\n\r\n\r\nMy Name is SnowMan. text '; + expectRejects(input, "XML Parse error: Unterminated CDATA section"); + }); + + test("ibm-not-wf-P19-ibm19n01.xml", () => { + // 2.7 — Tests CDStart. The CDStart contains a lower case string "cdata". + const input: string = + '\r\n \r\n]>\r\n\r\n\r\n\r\nMy Name is SnowMan. '; + expectRejects(input, "XML Parse error: Conditional sections are only allowed in the external DTD subset"); + }); + + test("ibm-not-wf-P19-ibm19n02.xml", () => { + // 2.7 — Tests CDStart. The CDStart contains an extra character "[". + const input: string = + '\r\n \r\n]>\r\n\r\n\r\n\r\nMy Name is SnowMan. \r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Conditional sections are only allowed in the external DTD subset"); + }); + + test("ibm-not-wf-P19-ibm19n03.xml", () => { + // 2.7 — Tests CDStart. The CDStart contains a wrong character "?". + const input: string = + '\r\n \r\n]>\r\n\r\n\r\n\r\nMy Name is SnowMan. '; + expectRejects(input, "XML Parse error: Expected a processing instruction target after ' { + // 2.7 — Tests CDATA with an illegal sequence. The CDATA contains the sequence close-bracket + // close-bracket greater-than. + const input: string = + '\r\n \r\n]>\r\n\r\n\r\nThis is ]]> a test]]>\r\nMy Name is SnowMan. '; + expectRejects(input, "XML Parse error: CDATA sections are only allowed inside elements"); + }); + + test("ibm-not-wf-P21-ibm21n01.xml", () => { + // 2.7 — Tests CDEnd. One "]" is missing in the CDEnd. + const input: string = + '\r\n \r\n]>\r\n\r\n\r\n\r\nMy Name is SnowMan. \r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Conditional sections are only allowed in the external DTD subset"); + }); + + test("ibm-not-wf-P21-ibm21n02.xml", () => { + // 2.7 — Tests CDEnd. An extra "]" is placed in the CDEnd. + const input: string = + '\r\n \r\n]>\r\n\r\n\r\n\r\nMy Name is SnowMan. '; + expectRejects(input, "XML Parse error: Conditional sections are only allowed in the external DTD subset"); + }); + + test("ibm-not-wf-P21-ibm21n03.xml", () => { + // 2.7 — Tests CDEnd. A wrong character ")" is placed in the CDEnd. + const input: string = + '\r\n \r\n]>\r\n\r\n\r\n\r\nMy Name is SnowMan. '; + expectRejects(input, "XML Parse error: CDATA sections are only allowed inside elements"); + }); + + test("ibm-not-wf-P22-ibm22n01.xml", () => { + // 2.8 — Tests prolog with wrong field ordering. The XMLDecl occurs after the DTD. + const input: string = + '\r\n]>\r\n\r\n\r\n'; + expectRejects( + input, + "XML Parse error: ' { + // 2.8 — Tests prolog with wrong field ordering. The Misc (comment) occurs before the XMLDecl. + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: ' { + // 2.8 — Tests prolog with wrong field ordering. The XMLDecl occurs after the DTD and a comment. The + // other comment occurs before the DTD. + const input: string = + '\r\n\r\n]>\r\n\r\n\r\n\r\n'; + expectRejects( + input, + "XML Parse error: ' { + // 2.8 — Tests XMLDecl with a required field missing. The Versioninfo is missing in the XMLDecl. + const input = Buffer.from( + '\r\n\r\n]>\r\n\r\n', + ); + expectRejects(input, 'XML Parse error: The XML declaration must start with version="1.0"'); + }); + + test("ibm-not-wf-P23-ibm23n02.xml", () => { + // 2.8 — Tests XMLDecl with wrong field ordering. The VersionInfo occurs after the EncodingDecl. + const input = Buffer.from( + "\r\n\r\n]>\r\n\r\n", + ); + expectRejects(input, 'XML Parse error: The XML declaration must start with version="1.0"'); + }); + + test("ibm-not-wf-P23-ibm23n03.xml", () => { + // 2.8 — Tests XMLDecl with wrong field ordering. The VersionInfo occurs after the SDDecl and the + // SDDecl occurs after the VersionInfo. + const input = Buffer.from( + "\r\n\r\n]>\r\n\r\n", + ); + expectRejects(input, 'XML Parse error: The XML declaration must start with version="1.0"'); + }); + + test("ibm-not-wf-P23-ibm23n04.xml", () => { + // 2.8 — Tests XMLDecl with wrong key word. An upper case string "XML" is used as the key word in the + // XMLDecl. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects( + input, + "XML Parse error: ' { + // 2.8 — Tests XMLDecl with a wrong closing sequence ">". + const input = Buffer.from( + "\r\n\r\n]>\r\n\r\n", + ); + expectRejects(input, "XML Parse error: Expected '?>' to end the XML declaration but found '>'"); + }); + + test("ibm-not-wf-P23-ibm23n06.xml", () => { + // 2.8 — Tests XMLDecl with a wrong opening sequence "(less than)!". + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects( + input, + "XML Parse error: ' { + // 2.8 — Tests VersionInfo with a required field missing. The VersionNum is missing in the VersionInfo + // in the XMLDecl. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected a quoted value in the XML declaration but found '?'"); + }); + + test("ibm-not-wf-P24-ibm24n02.xml", () => { + // 2.8 — Tests VersionInfo with a required field missing. The white space is missing between the key + // word "xml" and the VersionInfo in the XMLDecl. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected whitespace or '?>' after the processing instruction target but found '='", + ); + }); + + test("ibm-not-wf-P24-ibm24n03.xml", () => { + // 2.8 — Tests VersionInfo with a required field missing. The "=" (equal sign) is missing between the + // key word "version" and the VersionNum. + const input: string = + "\r\n\r\n]> \r\n\r\n"; + expectRejects(input, "XML Parse error: Expected '=' after the name in the XML declaration but found '''"); + }); + + test("ibm-not-wf-P24-ibm24n04.xml", () => { + // 2.8 — Tests VersionInfo with wrong field ordering. The VersionNum occurs before "=" and "version". + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected version=\"1.0\" in the XML declaration but found '''"); + }); + + test("ibm-not-wf-P24-ibm24n05.xml", () => { + // 2.8 — Tests VersionInfo with wrong field ordering. The "=" occurs after "version" and the + // VersionNum. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected '=' after the name in the XML declaration but found '''"); + }); + + test("ibm-not-wf-P24-ibm24n06.xml", () => { + // 2.8 — Tests VersionInfo with the wrong key word "Version". + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected version=\"1.0\" in the XML declaration but found 'Version'"); + }); + + test("ibm-not-wf-P24-ibm24n07.xml", () => { + // 2.8 — Tests VersionInfo with the wrong key word "versioN". + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected version=\"1.0\" in the XML declaration but found 'versioN'"); + }); + + test("ibm-not-wf-P24-ibm24n08.xml", () => { + // 2.8 — Tests VersionInfo with mismatched quotes around the VersionNum. version = '1.0" is used as the + // VersionInfo. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in a quoted string: '>'"); + }); + + test("ibm-not-wf-P24-ibm24n09.xml", () => { + // 2.8 — Tests VersionInfo with mismatched quotes around the VersionNum. The closing bracket for the + // VersionNum is missing. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Invalid character in a quoted string: '>'"); + }); + + test("ibm-not-wf-P25-ibm25n01.xml", () => { + // 2.8 — Tests eq with a wrong key word "==". + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected a quoted value in the XML declaration but found '='"); + }); + + test("ibm-not-wf-P25-ibm25n02.xml", () => { + // 2.8 — Tests eq with a wrong key word "eq". + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected '=' after the name in the XML declaration but found 'eq'"); + }); + + test("ibm-not-wf-P26-ibm26n01.xml", () => { + // 2.8 — Tests VersionNum with an illegal character "#". + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Unsupported XML version '_#1.0' (this is an XML 1.0 parser)"); + }); + + test("ibm-not-wf-P27-ibm27n01.xml", () => { + // 2.8 — Tests type of Misc. An element declaration is used as a type of Misc After the element + // "animal". + const input: string = + '\r\n\r\n]>\r\nWrong type of Misc following this element!\r\n'; + expectRejects(input, "XML Parse error: Unexpected ' { + // 2.8 — Tests doctypedecl with a required field missing. The Name "animal" is missing in the + // doctypedecl. + const input = Buffer.from( + '\r\n\r\n \r\n\r\n', + ); + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '\"'", + ); + }); + + test("ibm-not-wf-P28-ibm28n02.xml", () => { + // 2.8 — Tests doctypedecl with wrong field ordering. The Name "animal" occurs after the markup + // declarations inside the "[]". + const input = Buffer.from( + "\r\n\r\n] animal>\r\n \r\n\r\n", + ); + expectRejects(input, "XML Parse error: Expected the document type name but found '['"); + }); + + test("ibm-not-wf-P28-ibm28n03.xml", () => { + // 2.8 — Tests doctypedecl with wrong field ordering. The Name "animal" occurs after the markup + // declarations inside the "[]". + const input = Buffer.from( + '\r\n\r\n]>\r\n \r\n\r\n', + ); + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '\"'", + ); + }); + + test("ibm-not-wf-P28-ibm28n04.xml", () => { + // 2.8 — Tests doctypedecl with general entity reference.The "(ampersand)generalE" occurs in the DTD. + const input = Buffer.from( + '\r\n\r\n \r\n &generalE;\r\n ">\r\n %parameterE;\r\n] animal>\r\n\r\n \r\n&generalE\r\n', + ); + expectRejects(input, "XML Parse error: Expected a markup declaration or ']' in the internal subset but found '&'"); + }); + + test("ibm-not-wf-P28-ibm28n05.xml", () => { + // 2.8 — Tests doctypedecl with wrong key word. A wrong key word "DOCtYPE" occurs on line 2. + const input = Buffer.from( + "\r\n\r\n]>\r\n \r\n\r\n", + ); + expectRejects( + input, + "XML Parse error: ' { + // 2.8 — Tests doctypedecl with mismatched brackets. The closing bracket "]" of the DTD is missing. + const input = Buffer.from( + "\r\n\r\n>\r\n \r\n\r\n", + ); + expectRejects(input, "XML Parse error: Expected a markup declaration or ']' in the internal subset but found '>'"); + }); + + test("ibm-not-wf-P28-ibm28n07.xml", () => { + // 2.8 — Tests doctypedecl with wrong bracket. The opening bracket "{" occurs in the DTD. + const input = Buffer.from( + "\r\n\r\n]>\r\n \r\n\r\n", + ); + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '{'", + ); + }); + + test("ibm-not-wf-P28-ibm28n08.xml", () => { + // 2.8 — Tests doctypedecl with wrong opening sequence. The opening sequence "(less than)?DOCTYPE" + // occurs in the DTD. + const input = Buffer.from( + "\r\n\r\n]>\r\n \r\n\r\n", + ); + expectRejects(input, "XML Parse error: Unterminated processing instruction"); + }); + + test("ibm-not-wf-p28a-ibm28an01.xml", () => { + // 2.8 — This test violates WFC:PE Between Declarations in Production 28a. The last character of a + // markup declaration is not contained in the same parameter-entity text replacement. (upstream: + // not-wf; external parameter entities are not read) + const input = Buffer.from( + '\r\n\r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n]>\r\n\r\n &forcat;\r\n This is a white tiger in Mirage!!\r\n \r\n \r\n \r\n \r\n \r\n\r\n', + ); + expectParses(input); + }); + + test("ibm-not-wf-P29-ibm29n01.xml", () => { + // 2.8 — Tests markupdecl with an illegal markup declaration. A XMLDecl occurs inside the DTD. + const input: string = + '\r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n]>\r\n\r\n \r\n This is a white tiger in Mirage!!\r\n \r\n \r\n \r\n \r\n \r\n\r\n'; + expectRejects( + input, + "XML Parse error: ' { + // 2.8 — Tests WFC "PEs in Internal Subset". A PE reference occurs inside an elementdecl in the DTD. + const input = Buffer.from( + '\r\n\r\n ">\r\n \r\n\r\nAny content\r\n', + ); + expectRejects( + input, + "XML Parse error: Parameter entity references are not allowed inside markup declarations in the internal subset", + ); + }); + + test("ibm-not-wf-P29-ibm29n03.xml", () => { + // 2.8 — Tests WFC "PEs in Internal Subset". A PE reference occurs inside an ATTlistDecl in the DTD. + const input = Buffer.from( + '\r\n\r\n \r\n \r\n]>\r\n\r\nAny content\r\n', + ); + expectRejects( + input, + "XML Parse error: Parameter entity references are not allowed inside markup declarations in the internal subset", + ); + }); + + test("ibm-not-wf-P29-ibm29n04.xml", () => { + // 2.8 — Tests WFC "PEs in Internal Subset". A PE reference occurs inside an EntityDecl in the DTD. + const input = Buffer.from( + '\r\n\r\n \r\n \r\n]>\r\n\r\n&content;\r\n', + ); + expectRejects( + input, + "XML Parse error: Parameter entity references are not allowed inside markup declarations in the internal subset", + ); + }); + + test("ibm-not-wf-P29-ibm29n05.xml", () => { + // 2.8 — Tests WFC "PEs in Internal Subset". A PE reference occurs inside a PI in the DTD. + const input = Buffer.from( + '\r\n\r\n ">\r\n \r\n\r\nAny content\r\n', + ); + expectRejects(input, "XML Parse error: Unterminated processing instruction"); + }); + + test("ibm-not-wf-P29-ibm29n06.xml", () => { + // 2.8 — Tests WFC "PEs in Internal Subset". A PE reference occurs inside a comment in the DTD. + const input = Buffer.from( + '\r\n\r\n ">\r\n\r\n \r\nAny content\r\n', + ); + expectRejects( + input, + "XML Parse error: Parameter entity references are not allowed inside markup declarations in the internal subset", + ); + }); + + test("ibm-not-wf-P30-ibm30n01.xml", () => { + // 2.8 — Tests extSubset with wrong field ordering. In the file "ibm30n01.dtd", the TextDecl occurs + // after the extSubsetDecl (the element declaration). (upstream: not-wf; external parameter entities + // are not read) + const input: string = + '\r\n\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P31-ibm31n01.xml", () => { + // 2.8 — Tests extSubsetDecl with an illegal field. A general entity reference occurs in file + // "ibm31n01.dtd". (upstream: not-wf; external parameter entities are not read) + const input: string = + '\r\n\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P32-ibm32n01.xml", () => { + // 2.9 — Tests SDDecl with a required field missing. The leading white space is missing with the SDDecl + // in the XMLDecl. + const input: string = + '\r\n\r\n]>\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before 'standalone'"); + }); + + test("ibm-not-wf-P32-ibm32n02.xml", () => { + // 2.9 — Tests SDDecl with a required field missing. The "=" sign is missing in the SDDecl in the + // XMLDecl. + const input: string = + '\r\n\r\n]>\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected '=' after the name in the XML declaration but found '\"'"); + }); + + test("ibm-not-wf-P32-ibm32n03.xml", () => { + // 2.9 — Tests SDDecl with wrong key word. The word "Standalone" occurs in the SDDecl in the XMLDecl. + const input: string = + '\r\n\r\n]>\r\n\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Unexpected 'Standalone' in the XML declaration (expected version, encoding or standalone)", + ); + }); + + test("ibm-not-wf-P32-ibm32n04.xml", () => { + // 2.9 — Tests SDDecl with wrong key word. The word "Yes" occurs in the SDDecl in the XMLDecl. + const input: string = + '\r\n\r\n]>\r\n\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Invalid value 'Yes' for standalone in the XML declaration (expected yes or no)", + ); + }); + + test("ibm-not-wf-P32-ibm32n05.xml", () => { + // 2.9 — Tests SDDecl with wrong key word. The word "YES" occurs in the SDDecl in the XMLDecl. + const input: string = + '\r\n\r\n]>\r\n\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Invalid value 'YES' for standalone in the XML declaration (expected yes or no)", + ); + }); + + test("ibm-not-wf-P32-ibm32n06.xml", () => { + // 2.9 — Tests SDDecl with wrong key word. The word "No" occurs in the SDDecl in the XMLDecl. + const input: string = + '\r\n\r\n\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Invalid value 'No' for standalone in the XML declaration (expected yes or no)", + ); + }); + + test("ibm-not-wf-P32-ibm32n07.xml", () => { + // 2.9 — Tests SDDecl with wrong key word. The word "NO" occurs in the SDDecl in the XMLDecl. + const input: string = + '\r\n\r\n\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Invalid value 'NO' for standalone in the XML declaration (expected yes or no)", + ); + }); + + test("ibm-not-wf-P32-ibm32n08.xml", () => { + // 2.9 — Tests SDDecl with wrong field ordering. The "=" sign occurs after the key word "yes" in the + // SDDecl in the XMLDecl. + const input: string = + '\r\n\r\n]>\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected '=' after the name in the XML declaration but found '\"'"); + }); + + test("ibm-not-wf-P32-ibm32n09.xml", () => { + // 2.9 — This is test violates WFC: Entity Declared in P68. The standalone document declaration has the + // value yes, BUT there is an external markup declaration of an entity (other than amp, lt, gt, apos, + // quot), and references to this entity appear in the document. (upstream: not-wf; external parameter + // entities are not read) + const input: string = + '\r\n\r\n]>\r\n\r\n&animal_content;\r\n'; + expectRejects(input, "XML Parse error: Entity 'animal_content' is not declared"); + }); + + test("ibm-not-wf-P39-ibm39n01.xml", () => { + // 3 — Tests element with a required field missing. The ETag is missing for the element "root". + const input: string = + '\r\n\r\n]>\r\nmissing end tag\r\n\r\n'; + expectRejects(input, "XML Parse error: Missing closing tag for element 'root'"); + }); + + test("ibm-not-wf-P39-ibm39n02.xml", () => { + // 3 — Tests element with a required field missing. The STag is missing for the element "root". + const input: string = + '\r\n\r\n]>\r\nmissing start tag\r\n'; + expectRejects(input, "XML Parse error: Expected the root element but found 'missing'"); + }); + + test("ibm-not-wf-P39-ibm39n03.xml", () => { + // 3 — Tests element with required fields missing. Both the content and the ETag are missing in the + // element "root". + const input: string = + '\r\n\r\n]>\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Missing closing tag for element 'root'"); + }); + + test("ibm-not-wf-P39-ibm39n04.xml", () => { + // 3 — Tests element with required fields missing. Both the content and the STag are missing in the + // element "root". + const input: string = + '\r\n\r\n]>\r\n\r\n\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected the root element but found ' { + // 3 — Tests element with wrong field ordering. The STag and the ETag are swapped in the element + // "root". + const input: string = + '\r\n\r\n]>\r\nswitched start and end tags\r\n'; + expectRejects(input, "XML Parse error: Expected the root element but found ' { + // 3 — Tests element with wrong field ordering. The content occurs after the ETag of the element + // "root". + const input: string = + '\r\n\r\n]>\r\ncontent after end tag\r\n'; + expectRejects(input, "XML Parse error: Unexpected 'content' after the root element"); + }); + + test("ibm-not-wf-P40-ibm40n01.xml", () => { + // 3.1 — Tests STag with a required field missing. The Name "root" is in the STag of the element + // "root". + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nmissing name in start tag\r\n\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected an attribute name, '>' or '/>' in the start tag but found '='"); + }); + + test("ibm-not-wf-P40-ibm40n02.xml", () => { + // 3.1 — Tests STag with a required field missing. The white space between the Name "root" and the + // attribute "attr1" is missing in the STag of the element "root". + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nmissing white space in start tag\r\n'; + expectRejects(input, "XML Parse error: Expected an attribute name, '>' or '/>' in the start tag but found '='"); + }); + + test("ibm-not-wf-P40-ibm40n03.xml", () => { + // 3.1 — Tests STag with wrong field ordering. The Name "root" occurs after the attribute "attr1" in + // the STag of the element "root". + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nWrong ordering in start tag\r\n'; + expectRejects(input, "XML Parse error: Expected an attribute name, '>' or '/>' in the start tag but found '='"); + }); + + test("ibm-not-wf-P40-ibm40n04.xml", () => { + // 3.1 — Tests STag with a wrong opening sequence. The string "(less than)!" is used as the opening + // sequence for the STag of the element "root". + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nwrong begining sequence in start tag\r\n'; + expectRejects( + input, + "XML Parse error: ' { + // 3.1 — Tests STag with duplicate attribute names. The attribute name "attr1" occurs twice in the STag + // of the element "root". + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nduplicate attr names in start tag\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Duplicate attribute 'attr1'"); + }); + + test("ibm-not-wf-P41-ibm41n01.xml", () => { + // 3.1 — Tests Attribute with a required field missing. The attribute name is missing in the Attribute + // in the STag of the element "root". + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nmissing name in Attribute\r\n'; + expectRejects(input, "XML Parse error: Expected an attribute name, '>' or '/>' in the start tag but found '='"); + }); + + test("ibm-not-wf-P41-ibm41n02.xml", () => { + // 3.1 — Tests Attribute with a required field missing. The "=" is missing between the attribute name + // and the attribute value in the Attribute in the STag of the element "root". + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nmissing Eq in Attribute\r\n'; + expectRejects(input, "XML Parse error: Expected '=' after the attribute name but found '\"'"); + }); + + test("ibm-not-wf-P41-ibm41n03.xml", () => { + // 3.1 — Tests Attribute with a required field missing. The AttValue is missing in the Attribute in the + // STag of the element "root". + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nmissing AttValue in Attribute\r\n'; + expectRejects(input, "XML Parse error: Expected a quoted attribute value but found '>'"); + }); + + test("ibm-not-wf-P41-ibm41n04.xml", () => { + // 3.1 — Tests Attribute with a required field missing. The Name and the "=" are missing in the + // Attribute in the STag of the element "root". + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nmissing name and Eq in Attribute\r\n'; + expectRejects(input, "XML Parse error: Expected an attribute name, '>' or '/>' in the start tag but found '\"'"); + }); + + test("ibm-not-wf-P41-ibm41n05.xml", () => { + // 3.1 — Tests Attribute with a required field missing. The "=" and the AttValue are missing in the + // Attribute in the STag of the element "root". + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nmissing Eq and AttValue in Attribute\r\n'; + expectRejects(input, "XML Parse error: Expected '=' after the attribute name but found '>'"); + }); + + test("ibm-not-wf-P41-ibm41n06.xml", () => { + // 3.1 — Tests Attribute with a required field missing. The Name and the AttValue are missing in the + // Attribute in the STag of the element "root". + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nmissing Name and AttValue in Attribute\r\n'; + expectRejects(input, "XML Parse error: Expected an attribute name, '>' or '/>' in the start tag but found '='"); + }); + + test("ibm-not-wf-P41-ibm41n07.xml", () => { + // 3.1 — Tests Attribute with wrong field ordering. The "=" occurs after the Name and the AttValue in + // the Attribute in the STag of the element "root". + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nwrong ordering in Attribute\r\n'; + expectRejects(input, "XML Parse error: Expected '=' after the attribute name but found '\"'"); + }); + + test("ibm-not-wf-P41-ibm41n08.xml", () => { + // 3.1 — Tests Attribute with wrong field ordering. The Name and the AttValue are swapped in the + // Attribute in the STag of the element "root". + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nwrong ordering in Attribute\r\n'; + expectRejects(input, "XML Parse error: Expected an attribute name, '>' or '/>' in the start tag but found '\"'"); + }); + + test("ibm-not-wf-P41-ibm41n09.xml", () => { + // 3.1 — Tests Attribute with wrong field ordering. The "=" occurs before the Name and the AttValue in + // the Attribute in the STag of the element "root". + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nwrong ordering in Attribute\r\n'; + expectRejects(input, "XML Parse error: Expected an attribute name, '>' or '/>' in the start tag but found '='"); + }); + + test("ibm-not-wf-P41-ibm41n10.xml", () => { + // 3.1 — Tests Attribute against WFC "no external entity references". A direct reference to the + // external entity "aExternal" is contained in the value of the attribute "attr1". + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\ndirect reference to external entinity in Attribute\r\n'; + expectRejects(input, "XML Parse error: Attribute values cannot reference external entity 'aExternal'"); + }); + + test("ibm-not-wf-P41-ibm41n11.xml", () => { + // 3.1 — Tests Attribute against WFC "no external entity references". A indirect reference to the + // external entity "aExternal" is contained in the value of the attribute "attr1". + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\nindirect reference to external entinity in Attribute\r\n'; + expectRejects(input, "XML Parse error: Attribute values cannot reference external entity 'aExternal'"); + }); + + test("ibm-not-wf-P41-ibm41n12.xml", () => { + // 3.1 — Tests Attribute against WFC "no external entity references". A direct reference to the + // external unparsed entity "aImage" is contained in the value of the attribute "attr1". + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\ndirect reference to external unparsed entinity in Attribute\r\n\r\n'; + expectRejects(input, "XML Parse error: Unparsed entity 'aImage' cannot be referenced"); + }); + + test("ibm-not-wf-P41-ibm41n13.xml", () => { + // 3.1 — Tests Attribute against WFC "No (less than) character in Attribute Values". The character + // "less than" is contained in the value of the attribute "attr1". + const input: string = + '\r\n\r\n\r\n\r\n inside">\r\n]>\r\nDirect reference to an entity with < as part of its replacement text in Attribute\r\n'; + expectRejects(input, "XML Parse error: '<' is not allowed in attribute values"); + }); + + test("ibm-not-wf-P41-ibm41n14.xml", () => { + // 3.1 — Tests Attribute against WFC "No (less than) in Attribute Values". The character "less than" is + // contained in the value of the attribute "attr1" through indirect internal entity reference. + const input: string = + '\r\n\r\n\r\n\r\n inside">\r\n\r\n]>\r\nindirect reference to an entity with < as part of its replacement text in Attribute\r\n'; + expectRejects(input, "XML Parse error: '<' is not allowed in attribute values"); + }); + + test("ibm-not-wf-P42-ibm42n01.xml", () => { + // 3.1 — Tests ETag with a required field missing. The Name is missing in the ETag of the element + // "root". + const input: string = + '\r\n\r\n\r\n]>\r\nmissing Name in ETag\r\n'; + expectRejects(input, "XML Parse error: Expected an element name after ''"); + }); + + test("ibm-not-wf-P42-ibm42n02.xml", () => { + // 3.1 — Tests ETag with a wrong beginning sequence. The string "(less than)\" is used as a beginning + // sequence of the ETag of the element "root". + const input: string = + '\r\n\r\n\r\n]>\r\nWrong begining sequence in ETag <\\root>\r\n'; + expectRejects(input, "XML Parse error: Expected an element name after '<' but found '\\'"); + }); + + test("ibm-not-wf-P42-ibm42n03.xml", () => { + // 3.1 — Tests ETag with a wrong beginning sequence. The string "less than" is used as a beginning + // sequence of the ETag of the element "root". + const input: string = + '\r\n\r\n\r\n]>\r\nWrong begining sequence in ETag \r\n'; + expectRejects(input, "XML Parse error: Missing closing tag for element 'root'"); + }); + + test("ibm-not-wf-P42-ibm42n04.xml", () => { + // 3.1 — Tests ETag with a wrong structure. An white space occurs between The beginning sequence and + // the Name of the ETag of the element "root". + const input: string = + '\r\n\r\n\r\n]>\r\nExtra white space before Name in ETag \r\n'; + expectRejects(input, "XML Parse error: Expected an element name after ' { + // 3.1 — Tests ETag with a wrong structure. The ETag of the element "root" contains an Attribute + // (attr1="any"). + const input: string = + '\r\n\r\n\r\n]>\r\n Attribute in ETag \r\n'; + expectRejects(input, "XML Parse error: Expected '>' to end the closing tag but found 'attr1'"); + }); + + test("ibm-not-wf-P43-ibm43n01.xml", () => { + // 3.1 — Tests element content with a wrong option. A NotationDecl is used as the content of the + // element "root". + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected a comment or CDATA section after ' { + // 3.1 — Tests element content with a wrong option. An elementdecl is used as the content of the + // element "root". + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected a comment or CDATA section after ' { + // 3.1 — Tests element content with a wrong option. An entitydecl is used as the content of the element + // "root". + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected a comment or CDATA section after ' { + // 3.1 — Tests element content with a wrong option. An AttlistDecl is used as the content of the + // element "root". + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected a comment or CDATA section after ' { + // 3.1 — Tests EmptyElemTag with a required field missing. The Name "root" is missing in the + // EmptyElemTag. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n< />\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected an element name after '<' but found space"); + }); + + test("ibm-not-wf-P44-ibm44n02.xml", () => { + // 3.1 — Tests EmptyElemTag with wrong field ordering. The Attribute (attri1 = "any") occurs before the + // name of the element "root" in the EmptyElemTag. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected an attribute name, '>' or '/>' in the start tag but found '='"); + }); + + test("ibm-not-wf-P44-ibm44n03.xml", () => { + // 3.1 — Tests EmptyElemTag with wrong closing sequence. The string "\>" is used as the closing + // sequence in the EmptyElemtag of the element "root". + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n\r\n\r\n\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected an attribute name, '>' or '/>' in the start tag but found '\\'"); + }); + + test("ibm-not-wf-P44-ibm44n04.xml", () => { + // 3.1 — Tests EmptyElemTag which against the WFC "Unique Att Spec". The attribute name "attr1" occurs + // twice in the EmptyElemTag of the element "root". + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Duplicate attribute 'attr1'"); + }); + + test("ibm-not-wf-P45-ibm45n01.xml", () => { + // 3.2 — Tests elementdecl with a required field missing. The Name is missing in the second elementdecl + // in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nAny content\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected an element name after ' { + // 3.2 — Tests elementdecl with a required field missing. The white space is missing between "aEle" and + // "(#PCDATA)" in the second elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before '('"); + }); + + test("ibm-not-wf-P45-ibm45n03.xml", () => { + // 3.2 — Tests elementdecl with a required field missing. The contentspec is missing in the second + // elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected EMPTY, ANY or '(' in the element declaration but found '>'"); + }); + + test("ibm-not-wf-P45-ibm45n04.xml", () => { + // 3.2 — Tests elementdecl with a required field missing. The contentspec and the white space is + // missing in the second elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected EMPTY, ANY or '(' in the element declaration but found '>'"); + }); + + test("ibm-not-wf-P45-ibm45n05.xml", () => { + // 3.2 — Tests elementdecl with a required field missing. The Name, the white space, and the + // contentspec are missing in the second elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected an element name after ''"); + }); + + test("ibm-not-wf-P45-ibm45n06.xml", () => { + // 3.2 — Tests elementdecl with wrong field ordering. The Name occurs after the contentspec in the + // second elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected an element name after ' { + // 3.2 — Tests elementdecl with wrong beginning sequence. The string "(less than)ELEMENT" is used as + // the beginning sequence in the second elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects( + input, + "XML Parse error: Expected a markup declaration or ']' in the internal subset but found ' { + // 3.2 — Tests elementdecl with wrong key word. The string "Element" is used as the key word in the + // second elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects( + input, + "XML Parse error: ' { + // 3.2 — Tests elementdecl with wrong key word. The string "element" is used as the key word in the + // second elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects( + input, + "XML Parse error: ' { + // 3.2 — Tests contentspec with wrong key word. the string "empty" is used as the key word in the + // contentspec of the second elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nAny content\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected EMPTY, ANY or '(' in the element declaration but found 'empty'"); + }); + + test("ibm-not-wf-P46-ibm46n02.xml", () => { + // 3.2 — Tests contentspec with wrong key word. the string "Empty" is used as the key word in the + // contentspec of the second elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected EMPTY, ANY or '(' in the element declaration but found 'Empty'"); + }); + + test("ibm-not-wf-P46-ibm46n03.xml", () => { + // 3.2 — Tests contentspec with wrong key word. the string "Any" is used as the key word in the + // contentspec of the second elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected EMPTY, ANY or '(' in the element declaration but found 'Any'"); + }); + + test("ibm-not-wf-P46-ibm46n04.xml", () => { + // 3.2 — Tests contentspec with wrong key word. the string "any" is used as the key word in the + // contentspec of the second elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected EMPTY, ANY or '(' in the element declaration but found 'any'"); + }); + + test("ibm-not-wf-P46-ibm46n05.xml", () => { + // 3.2 — Tests contentspec with a wrong option. The string "#CDATA" is used as the contentspec in the + // second elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected EMPTY, ANY or '(' in the element declaration but found '#CDATA'"); + }); + + test("ibm-not-wf-P47-ibm47n01.xml", () => { + // 3.2.1 — Tests children with a required field missing. The "+" is used as the choice or seq field in + // the second elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected EMPTY, ANY or '(' in the element declaration but found '+'"); + }); + + test("ibm-not-wf-P47-ibm47n02.xml", () => { + // 3.2.1 — Tests children with a required field missing. The "*" is used as the choice or seq field in + // the second elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected EMPTY, ANY or '(' in the element declaration but found '*'"); + }); + + test("ibm-not-wf-P47-ibm47n03.xml", () => { + // 3.2.1 — Tests children with a required field missing. The "?" is used as the choice or seq field in + // the second elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected EMPTY, ANY or '(' in the element declaration but found '?'"); + }); + + test("ibm-not-wf-P47-ibm47n04.xml", () => { + // 3.2.1 — Tests children with wrong field ordering. The "*" occurs before the seq field (a,a) in the + // second elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected EMPTY, ANY or '(' in the element declaration but found '*'"); + }); + + test("ibm-not-wf-P47-ibm47n05.xml", () => { + // 3.2.1 — Tests children with wrong field ordering. The "+" occurs before the choice field (a|a) in + // the second elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected EMPTY, ANY or '(' in the element declaration but found '+'"); + }); + + test("ibm-not-wf-P47-ibm47n06.xml", () => { + // 3.2.1 — Tests children with wrong key word. The "^" occurs after the seq field in the second + // elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected '>' to end the element declaration but found '^'"); + }); + + test("ibm-not-wf-P48-ibm48n01.xml", () => { + // 3.2.1 — Tests cp with a required fields missing. The field Name|choice|seq is missing in the second + // cp in the choice field in the third elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected an element name or '(' in the content model but found '+'"); + }); + + test("ibm-not-wf-P48-ibm48n02.xml", () => { + // 3.2.1 — Tests cp with a required fields missing. The field Name|choice|seq is missing in the cp in + // the third elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected an element name or '(' in the content model but found '*'"); + }); + + test("ibm-not-wf-P48-ibm48n03.xml", () => { + // 3.2.1 — Tests cp with a required fields missing. The field Name|choice|seq is missing in the first + // cp in the choice field in the third elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected an element name or '(' in the content model but found '?'"); + }); + + test("ibm-not-wf-P48-ibm48n04.xml", () => { + // 3.2.1 — Tests cp with wrong field ordering. The "+" occurs before the seq (a,a) in the first cp in + // the choice field in the third elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected an element name or '(' in the content model but found '+'"); + }); + + test("ibm-not-wf-P48-ibm48n05.xml", () => { + // 3.2.1 — Tests cp with wrong field ordering. The "*" occurs before the choice (a|b) in the first cp + // in the seq field in the third elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected an element name or '(' in the content model but found '*'"); + }); + + test("ibm-not-wf-P48-ibm48n06.xml", () => { + // 3.2.1 — Tests cp with wrong field ordering. The "?" occurs before the Name "a" in the second cp in + // the seq field in the third elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected an element name or '(' in the content model but found '?'"); + }); + + test("ibm-not-wf-P48-ibm48n07.xml", () => { + // 3.2.1 — Tests cp with wrong key word. The "^" occurs after the Name "a" in the first cp in the + // choice field in the third elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected ')', '|' or ',' in the content model but found '^'"); + }); + + test("ibm-not-wf-P49-ibm49n01.xml", () => { + // 3.2.1 — Tests choice with a required field missing. The two cps are missing in the choice field in + // the third elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected an element name or '(' in the content model but found '|'"); + }); + + test("ibm-not-wf-P49-ibm49n02.xml", () => { + // 3.2.1 — Tests choice with a required field missing. The third cp is missing in the choice field in + // the fourth elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected an element name or '(' in the content model but found ')'"); + }); + + test("ibm-not-wf-P49-ibm49n03.xml", () => { + // 3.2.1 — Tests choice with a wrong separator. The "!" is used as the separator in the choice field in + // the fourth elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected ')', '|' or ',' in the content model but found '!'"); + }); + + test("ibm-not-wf-P49-ibm49n04.xml", () => { + // 3.2.1 — Tests choice with a required field missing. The separator "|" is missing in the choice field + // (a b)+ in the fourth elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected ')', '|' or ',' in the content model but found 'b'"); + }); + + test("ibm-not-wf-P49-ibm49n05.xml", () => { + // 3.2.1 — Tests choice with an extra separator. An extra "|" occurs between a and b in the choice + // field in the fourth elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected an element name or '(' in the content model but found '|'"); + }); + + test("ibm-not-wf-P49-ibm49n06.xml", () => { + // 3.2.1 — Tests choice with a required field missing. The closing bracket ")" is missing in the choice + // field (a |b * in the fourth elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n\r\n'; + expectRejects(input, "XML Parse error: An occurrence indicator must directly follow the name or ')' it applies to"); + }); + + test("ibm-not-wf-P50-ibm50n01.xml", () => { + // 3.2.1 — Tests seq with a required field missing. The two cps are missing in the seq field in the + // fourth elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected an element name or '(' in the content model but found ','"); + }); + + test("ibm-not-wf-P50-ibm50n02.xml", () => { + // 3.2.1 — Tests seq with a required field missing. The third cp is missing in the seq field in the + // fourth elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected an element name or '(' in the content model but found ')'"); + }); + + test("ibm-not-wf-P50-ibm50n03.xml", () => { + // 3.2.1 — Tests seq with a wrong separator. The "|" is used as the separator between a and b in the + // seq field in the fourth elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: A content model group cannot mix ',' and '|'"); + }); + + test("ibm-not-wf-P50-ibm50n04.xml", () => { + // 3.2.1 — Tests seq with a wrong separator. The "." is used as the separator between a and b in the + // seq field in the fourth elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected ')', '|' or ',' in the content model but found '.'"); + }); + + test("ibm-not-wf-P50-ibm50n05.xml", () => { + // 3.2.1 — Tests seq with an extra separator. An extra "," occurs between (a|b) and a in the seq field + // in the fourth elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected an element name or '(' in the content model but found ','"); + }); + + test("ibm-not-wf-P50-ibm50n06.xml", () => { + // 3.2.1 — Tests seq with a required field missing. The separator between (a|b) and (b|a) is missing in + // the seq field in the fourth elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected ')', '|' or ',' in the content model but found '('"); + }); + + test("ibm-not-wf-P50-ibm50n07.xml", () => { + // 3.2.1 — Tests seq with wrong closing bracket. The "]" is used as the closing bracket in the seq + // field in the fourth elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected ')', '|' or ',' in the content model but found ']'"); + }); + + test("ibm-not-wf-P51-ibm51n01.xml", () => { + // 3.2.2 — Tests Mixed with a wrong key word. The string "#pcdata" is used as the key word in the Mixed + // field in the fourth elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected an element name or '(' in the content model but found '#pcdata'"); + }); + + test("ibm-not-wf-P51-ibm51n02.xml", () => { + // 3.2.2 — Tests Mixed with wrong field ordering. The field #PCDATA does not occur as the first + // component in the Mixed field in the fourth elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: #PCDATA must come first in a content model, as (#PCDATA|a|b)*"); + }); + + test("ibm-not-wf-P51-ibm51n03.xml", () => { + // 3.2.2 — Tests Mixed with a separator missing. The separator "|" is missing in between #PCDATA and a + // in the Mixed field in the fourth elementdecl in the DTD. + const input: string = + "\r\n\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n"; + expectRejects(input, "XML Parse error: Expected '|' or ')' in the mixed content model but found 'a'"); + }); + + test("ibm-not-wf-P51-ibm51n04.xml", () => { + // 3.2.2 — Tests Mixed with a wrong key word. The string "#CDATA" is used as the key word in the Mixed + // field in the fourth elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected an element name or '(' in the content model but found '#CDATA'"); + }); + + test("ibm-not-wf-P51-ibm51n05.xml", () => { + // 3.2.2 — Tests Mixed with a required field missing. The "*" is missing after the ")" in the Mixed + // field in the fourth elementdecl in the DTD. + const input: string = + "\r\n\r\n\r\n\r\n\r\n\r\n]>\r\nAny content"; + expectRejects(input, "XML Parse error: A mixed content model with element names must end with ')*'"); + }); + + test("ibm-not-wf-P51-ibm51n06.xml", () => { + // 3.2.2 — Tests Mixed with wrong closing bracket. The "]" is used as the closing bracket in the Mixed + // field in the fourth elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected '|' or ')' in the mixed content model but found ']'"); + }); + + test("ibm-not-wf-P51-ibm51n07.xml", () => { + // 3.2.2 — Tests Mixed with a required field missing. The closing bracket ")" is missing after (#PCDATA + // in the Mixed field in the fourth elementdecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected '|' or ')' in the mixed content model but found '*'"); + }); + + test("ibm-not-wf-P52-ibm52n01.xml", () => { + // 3.3 — Tests AttlistDecl with a required field missing. The Name is missing in the AttlistDecl in the + // DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found '#IMPLIED'", + ); + }); + + test("ibm-not-wf-P52-ibm52n02.xml", () => { + // 3.3 — Tests AttlistDecl with a required field missing. The white space is missing between the + // beginning sequence and the name in the AttlistDecl in the DTD. + const input: string = + "\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n"; + expectRejects(input, "XML Parse error: Whitespace is required before 'a'"); + }); + + test("ibm-not-wf-P52-ibm52n03.xml", () => { + // 3.3 — Tests AttlistDecl with wrong field ordering. The Name "a" occurs after the first AttDef in the + // AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found '\"'", + ); + }); + + test("ibm-not-wf-P52-ibm52n04.xml", () => { + // 3.3 — Tests AttlistDecl with wrong key word. The string "Attlist" is used as the key word in the + // beginning sequence in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects( + input, + "XML Parse error: ' { + // 3.3 — Tests AttlistDecl with a required field missing. The closing bracket "greater than" is missing + // in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute name or '>' in the ATTLIST declaration but found a comment", + ); + }); + + test("ibm-not-wf-P52-ibm52n06.xml", () => { + // 3.3 — Tests AttlistDecl with wrong beginning sequence. The string "(less than)ATTLIST" is used as + // the beginning sequence in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects( + input, + "XML Parse error: Expected a markup declaration or ']' in the internal subset but found ' { + // 3.3 — Tests AttDef with a required field missing. The DefaultDecl is missing in the AttDef for the + // name "attr1" in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Expected #REQUIRED, #IMPLIED, #FIXED or a quoted default value but found '>'", + ); + }); + + test("ibm-not-wf-P53-ibm53n02.xml", () => { + // 3.3 — Tests AttDef with a required field missing. The white space is missing between (abc|def) and + // "def" in the AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before a quoted string"); + }); + + test("ibm-not-wf-P53-ibm53n03.xml", () => { + // 3.3 — Tests AttDef with a required field missing. The AttType is missing for "attr1" in the AttDef + // in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found '#IMPLIED'", + ); + }); + + test("ibm-not-wf-P53-ibm53n04.xml", () => { + // 3.3 — Tests AttDef with a required field missing. The white space is missing between "attr1" and + // (abc|def) in the AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before '('"); + }); + + test("ibm-not-wf-P53-ibm53n05.xml", () => { + // 3.3 — Tests AttDef with a required field missing. The Name is missing in the AttDef in the + // AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected an attribute name or '>' in the ATTLIST declaration but found '('"); + }); + + test("ibm-not-wf-P53-ibm53n06.xml", () => { + // 3.3 — Tests AttDef with a required field missing. The white space before the name "attr2" is missing + // in the AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before 'attr2'"); + }); + + test("ibm-not-wf-P53-ibm53n07.xml", () => { + // 3.3 — Tests AttDef with wrong field ordering. The Name "attr1" occurs after the AttType in the + // AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects(input, "XML Parse error: Expected an attribute name or '>' in the ATTLIST declaration but found '('"); + }); + + test("ibm-not-wf-P53-ibm53n08.xml", () => { + // 3.3 — Tests AttDef with wrong field ordering. The Name "attr1" occurs after the AttType and + // "default" occurs before the AttType in the AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\nAny content\r\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute name or '>' in the ATTLIST declaration but found '\"'", + ); + }); + + test("ibm-not-wf-P54-ibm54n01.xml", () => { + // 3.3.1 — Tests AttType with a wrong option. The string "BOGUSATTR" is used as the AttType in the + // AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n \r\n]>\r\n\r\nGiving a Bogus attribute. \r\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'BOGUSATTR'", + ); + }); + + test("ibm-not-wf-P54-ibm54n02.xml", () => { + // 3.3.1 — Tests AttType with a wrong option. The string "PCDATA" is used as the AttType in the + // AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n \r\n]>\r\n\r\nGiving a wrong AttType for the attribute. \r\n\r\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'PCDATA'", + ); + }); + + test("ibm-not-wf-P55-ibm55n01.xml", () => { + // 3.3.1 — Tests StringType with a wrong key word. The lower case string "cdata" is used as the + // StringType in the AttType in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n \r\n]>\r\n\r\nGiving a lowercase for CDATA attribute.\r\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'cdata'", + ); + }); + + test("ibm-not-wf-P55-ibm55n02.xml", () => { + // 3.3.1 — Tests StringType with a wrong key word. The string "#CDATA" is used as the StringType in the + // AttType in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n \r\n]>\r\n\r\nGiving a wrong character. \r\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found '#CDATA'", + ); + }); + + test("ibm-not-wf-P55-ibm55n03.xml", () => { + // 3.3.1 — Tests StringType with a wrong key word. The string "CData" is used as the StringType in the + // AttType in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n \r\n]>\r\n\r\n Giving a wrong key word of the StringType.\r\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'CData'", + ); + }); + + test("ibm-not-wf-P56-ibm56n01.xml", () => { + // 3.3.1 — Tests TokenizedType with wrong key word. The type "id" is used in the TokenizedType in the + // AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\n\r\nInvalid TokenizedType id(lowercase)\r\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'id'", + ); + }); + + test("ibm-not-wf-P56-ibm56n02.xml", () => { + // 3.3.1 — Tests TokenizedType with wrong key word. The type "Idref" is used in the TokenizedType in + // the AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\n\r\nInvalid TokenizedType Idref(case sensitive)\r\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'Idref'", + ); + }); + + test("ibm-not-wf-P56-ibm56n03.xml", () => { + // 3.3.1 — Tests TokenizedType with wrong key word. The type"Idrefs" is used in the TokenizedType in + // the AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\n\r\nInvalid TokenizedType IdRefs(case sensitive)\r\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'IdRefs'", + ); + }); + + test("ibm-not-wf-P56-ibm56n04.xml", () => { + // 3.3.1 — Tests TokenizedType with wrong key word. The type "EntitY" is used in the TokenizedType in + // the AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\n\r\nInvalid TokenizedType EntitY(case sensitive)\r\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'EntitY'", + ); + }); + + test("ibm-not-wf-P56-ibm56n05.xml", () => { + // 3.3.1 — Tests TokenizedType with wrong key word. The type "nmTOKEN" is used in the TokenizedType in + // the AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\n\r\nInvalid TokenizedType nmTOKEN(case sensitive)\r\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'nmTOKEN'", + ); + }); + + test("ibm-not-wf-P56-ibm56n06.xml", () => { + // 3.3.1 — Tests TokenizedType with wrong key word. The type "NMtokens" is used in the TokenizedType in + // the AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\n\r\nInvalid TokenizedType NMtokens(case sensitive)\r\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'NMtokens'", + ); + }); + + test("ibm-not-wf-P56-ibm56n07.xml", () => { + // 3.3.1 — Tests TokenizedType with wrong key word. The type "#ID" is used in the TokenizedType in the + // AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\n\r\nInvalid TokenizedType #ID(Wrong Character)\r\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found '#ID'", + ); + }); + + test("ibm-not-wf-P57-ibm57n01.xml", () => { + // 3.3.1 — Tests EnumeratedType with an illegal option. The string "NMTOKEN (a|b)" is used in the + // EnumeratedType in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n \r\n ]>\r\n \r\nThis test case tests the illegal enumerated types\r\n'; + expectRejects( + input, + "XML Parse error: Expected #REQUIRED, #IMPLIED, #FIXED or a quoted default value but found '('", + ); + }); + + test("ibm-not-wf-P58-ibm58n01.xml", () => { + // 3.3.1 — Tests NotationType with wrong key word. The lower case "notation" is used as the key word in + // the NotationType in the AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n ]>\r\n \r\nThis is a Negative test with notation (name) \r\nIt is case sensitive.\r\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'notation'", + ); + }); + + test("ibm-not-wf-P58-ibm58n02.xml", () => { + // 3.3.1 — Tests NotationType with a required field missing. The beginning bracket "(" is missing in + // the NotationType in the AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n ]>\r\n \r\nThis is a Negative test with (name) \r\nMissing the open parenthesis\r\n'; + expectRejects(input, "XML Parse error: Expected '(' after NOTATION but found 'this'"); + }); + + test("ibm-not-wf-P58-ibm58n03.xml", () => { + // 3.3.1 — Tests NotationType with a required field missing. The Name is missing in the "()" in the + // NotationType in the AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n ]>\r\n \r\nThis is a Negative test with NOTATION () \r\nMissing the required field\r\n'; + expectRejects(input, "XML Parse error: Expected a notation name but found ')'"); + }); + + test("ibm-not-wf-P58-ibm58n04.xml", () => { + // 3.3.1 — Tests NotationType with a required field missing. The closing bracket is missing in the + // NotationType in the AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n ]>\r\n \r\nThis is a Negative test with NOTATION (Name \r\nMissing the closing brackets\r\n'; + expectRejects(input, "XML Parse error: Expected '|' or ')' in the enumeration but found '#IMPLIED'"); + }); + + test("ibm-not-wf-P58-ibm58n05.xml", () => { + // 3.3.1 — Tests NotationType with wrong field ordering. The key word "NOTATION" occurs after "(this)" + // in the NotationType in the AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n ]>\r\n \r\nThis is a Negative test with (Name) NOTATION \r\nWrong Ordering\r\n'; + expectRejects( + input, + "XML Parse error: Expected #REQUIRED, #IMPLIED, #FIXED or a quoted default value but found 'NOTATION'", + ); + }); + + test("ibm-not-wf-P58-ibm58n06.xml", () => { + // 3.3.1 — Tests NotationType with wrong separator. The "," is used as a separator between "this" and + // "that" in the NotationType in the AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n \r\n ]>\r\n\r\nNegative Test.\r\nThis test tests the presence of a correct seperator. There is a wrong seperator(,)\r\n'; + expectRejects(input, "XML Parse error: Expected '|' or ')' in the enumeration but found ','"); + }); + + test("ibm-not-wf-P58-ibm58n07.xml", () => { + // 3.3.1 — Tests NotationType with a required field missing. The white space is missing between + // "NOTATION" and "(this)" in the NotationType in the AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n ]>\r\n\r\nNegative Test.\r\nMissing space after NOTATION\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before '('"); + }); + + test("ibm-not-wf-P58-ibm58n08.xml", () => { + // 3.3.1 — Tests NotationType with extra wrong characters. The double quote character occurs after "(" + // and before ")" in the NotationType in the AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n ]>\r\n\r\nNegative Test.\r\nPresence of quotes around the value\r\n'; + expectRejects(input, "XML Parse error: Expected a notation name but found '\"'"); + }); + + test("ibm-not-wf-P59-ibm59n01.xml", () => { + // 3.3.1 — Tests Enumeration with required fields missing. The Nmtokens and "|"s are missing in the + // AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n ]>\r\n \r\nThis is a Negative test\r\nMissing the required field\r\n'; + expectRejects(input, "XML Parse error: Expected a name token in the enumeration but found ')'"); + }); + + test("ibm-not-wf-P59-ibm59n02.xml", () => { + // 3.3.1 — Tests Enumeration with a required field missing. The closing bracket ")" is missing in the + // AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n ]>\r\n \r\nThis is a Negative test\r\nMissing the closing brackets\r\n'; + expectRejects(input, "XML Parse error: Expected '|' or ')' in the enumeration but found '#IMPLIED'"); + }); + + test("ibm-not-wf-P59-ibm59n03.xml", () => { + // 3.3.1 — Tests Enumeration with wrong separator. The "," is used as the separator in the AttDef in + // the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n ]>\r\n \r\nThis is a Negative test\r\nWrong Separator(, instead of |)\r\n'; + expectRejects(input, "XML Parse error: Expected '|' or ')' in the enumeration but found ','"); + }); + + test("ibm-not-wf-P59-ibm59n04.xml", () => { + // 3.3.1 — Tests Enumeration with illegal presence. The double quotes occur around the Enumeration + // value in the AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n ]>\r\n \r\nThis is a Negative test\r\nIllegal presence of quotes around the value\r\n'; + expectRejects(input, "XML Parse error: Expected a name token in the enumeration but found '\"'"); + }); + + test("ibm-not-wf-P59-ibm59n05.xml", () => { + // 3.3.1 — Tests Enumeration with a required field missing. The white space is missing between in the + // AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n ]>\r\n \r\nThis is a Negative test\r\nMissing the begining bracket \r\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'enum'", + ); + }); + + test("ibm-not-wf-P59-ibm59n06.xml", () => { + // 3.3.1 — Tests Enumeration with a required field missing. The beginning bracket "(" is missing in the + // AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n ]>\r\n \r\nThis is a Negative test\r\nMissing the Opening brackets\r\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute type (CDATA, ID, IDREF, IDREFS, ENTITY, ENTITIES, NMTOKEN, NMTOKENS, NOTATION or an enumeration) but found 'enum'", + ); + }); + + test("ibm-not-wf-P60-ibm60n01.xml", () => { + // 3.3.2 — Tests DefaultDecl with wrong key word. The string "#required" is used as the key word in the + // DefaultDecl in the AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n \r\n \r\n ]>\r\n\r\n\r\nNegative Test. Case sensitive.\r\n'; + expectRejects( + input, + "XML Parse error: Expected #REQUIRED, #IMPLIED, #FIXED or a quoted default value but found '#required'", + ); + }); + + test("ibm-not-wf-P60-ibm60n02.xml", () => { + // 3.3.2 — Tests DefaultDecl with wrong key word. The string "Implied" is used as the key word in the + // DefaultDecl in the AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n \r\n \r\n ]>\r\n\r\n\r\nNegative test. Case Sensitive\r\n'; + expectRejects( + input, + "XML Parse error: Expected #REQUIRED, #IMPLIED, #FIXED or a quoted default value but found '#Implied'", + ); + }); + + test("ibm-not-wf-P60-ibm60n03.xml", () => { + // 3.3.2 — Tests DefaultDecl with wrong key word. The string "!IMPLIED" is used as the key word in the + // DefaultDecl in the AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n \r\n \r\n ]>\r\n\r\n\r\nNegative Test. Wrong Character.\r\n'; + expectRejects( + input, + "XML Parse error: Expected #REQUIRED, #IMPLIED, #FIXED or a quoted default value but found '!'", + ); + }); + + test("ibm-not-wf-P60-ibm60n04.xml", () => { + // 3.3.2 — Tests DefaultDecl with a required field missing. There is no attribute value specified after + // the key word "#FIXED" in the DefaultDecl in the AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n \r\n \r\n ]>\r\n\r\n\r\nNegative test. Missing required field(#FIXED should have a value)\r\n'; + expectRejects(input, "XML Parse error: Expected a quoted default value after #FIXED but found '>'"); + }); + + test("ibm-not-wf-P60-ibm60n05.xml", () => { + // 3.3.2 — Tests DefaultDecl with a required field missing. The white space is missing between the key + // word "#FIXED" and the attribute value in the DefaultDecl in the AttDef in the AttlistDecl in the + // DTD. + const input: string = + '\r\n\r\n\r\n \r\n \r\n ]>\r\n\r\n\r\nNegative test. Missing required field(#FIXED should have a space before value)\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before a quoted string"); + }); + + test("ibm-not-wf-P60-ibm60n06.xml", () => { + // 3.3.2 — Tests DefaultDecl with wrong field ordering. The key word "#FIXED" occurs after the + // attribute value "introduction" in the DefaultDecl in the AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n \r\n \r\n ]>\r\n\r\n\r\nNegative test. Wrong Ordering\r\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute name or '>' in the ATTLIST declaration but found '#FIXED'", + ); + }); + + test("ibm-not-wf-P60-ibm60n07.xml", () => { + // 3.3.2 — Tests DefaultDecl against WFC of P60. The text replacement of the entity "avalue" contains + // the "less than" character in the DefaultDecl in the AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n ]>\r\n\r\n\r\nNegative test. \r\nThe replacement text of any entity referred to directly or indirectly \r\nin an attribute value contains a less than character\r\n'; + expectRejects(input, "XML Parse error: '<' is not allowed in attribute values"); + }); + + test("ibm-not-wf-P60-ibm60n08.xml", () => { + // 3.3.2 — Tests DefaultDecl with more than one key word. The "#REQUIRED" and the "#IMPLIED" are used + // as the key words in the DefaultDecl in the AttDef in the AttlistDecl in the DTD. + const input: string = + '\r\n\r\n\r\n \r\n \r\n ]>\r\n\r\n\r\nNegative Test. More than one Default type declarations.\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Expected an attribute name or '>' in the ATTLIST declaration but found '#IMPLIED'", + ); + }); + + test("ibm-not-wf-P61-ibm61n01.xml", () => { + // 3.4 — Tests conditionalSect with a wrong option. The word "NOTINCLUDE" is used as part of an option + // which is wrong in the coditionalSect. (upstream: not-wf; external parameter entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n \r\n\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P62-ibm62n01.xml", () => { + // 3.4 — Tests includeSect with wrong key word. The string "include" is used as a key word in the + // beginning sequence in the includeSect in the file ibm62n01.dtd. (upstream: not-wf; external + // parameter entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n \r\nNegative test. Test includeSect with include(Case sensitive)\r\n\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P62-ibm62n02.xml", () => { + // 3.4 — Tests includeSect with wrong beginning sequence. An extra "[" occurs in the beginning sequence + // in the includeSect in the file ibm62n02.dtd. (upstream: not-wf; external parameter entities are not + // read) + const input: string = + '\r\n\r\n\r\n\r\n \r\nNegative test. An extra \'[\' is used.\r\n\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P62-ibm62n03.xml", () => { + // 3.4 — Tests includeSect with wrong beginning sequence. A wrong character "?" occurs in the beginning + // sequence in the includeSect in the file ibm62n03.dtd. (upstream: not-wf; external parameter entities + // are not read) + const input: string = + '\r\n\r\n\r\n\r\n \r\nNegative test. Wrong character is used is used.\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P62-ibm62n04.xml", () => { + // 3.4 — Tests includeSect with a required field missing. The key word "INCLUDE" is missing in the + // includeSect in the file ibm62n04.dtd. (upstream: not-wf; external parameter entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n \r\nNegative test. Missing the required field INCLUDE.\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P62-ibm62n05.xml", () => { + // 3.4 — Tests includeSect with a required field missing. The "[" is missing after the key word + // "INCLUDE" in the includeSect in the file ibm62n05.dtd. (upstream: not-wf; external parameter + // entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n \r\nNegative test. Missing the required field \'[\' after INCLUDE.\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P62-ibm62n06.xml", () => { + // 3.4 — Tests includeSect with wrong field ordering. The two external subset declarations occur before + // the key word "INCLUDE" in the includeSect in the file ibm62n06.dtd. (upstream: not-wf; external + // parameter entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n \r\nNegative test. Wrong Ordering. External subset declaration prior to the keyword INCLUDE\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P62-ibm62n07.xml", () => { + // 3.4 — Tests includeSect with a required field missing. The closing sequence "]](greater than)" is + // missing in the includeSect in the file ibm62n07.dtd. (upstream: not-wf; external parameter entities + // are not read) + const input: string = + '\r\n\r\n\r\n\r\n \r\nNegative test. Missing closing sequence.\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P62-ibm62n08.xml", () => { + // 3.4 — Tests includeSect with a required field missing. One "]" is missing in the closing sequence in + // the includeSect in the file ibm62n08.dtd. (upstream: not-wf; external parameter entities are not + // read) + const input: string = + '\r\n\r\n\r\n\r\n \r\nNegative test. Missing external subset declaration.\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P63-ibm63n01.xml", () => { + // 3.4 — Tests ignoreSect with wrong key word. The string "ignore" is used as a key word in the + // beginning sequence in the ignoreSect in the file ibm63n01.dtd. (upstream: not-wf; external parameter + // entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n \r\n]>\r\n\r\nNegative test. Case sensitive(ignore is used instead of IGNORE).\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P63-ibm63n02.xml", () => { + // 3.4 — Tests ignoreSect with wrong beginning sequence. An extra "[" occurs in the beginning sequence + // in the ignoreSect in the file ibm63n02.dtd. (upstream: not-wf; external parameter entities are not + // read) + const input: string = + '\r\n\r\n \r\n]>\r\n\r\nNegative test. Extra \'[\' is used before IGNORE.\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P63-ibm63n03.xml", () => { + // 3.4 — Tests ignoreSect with wrong beginning sequence. A wrong character "?" occurs in the beginning + // sequence in the ignoreSect in the file ibm63n03.dtd. (upstream: not-wf; external parameter entities + // are not read) + const input: string = + '\r\n\r\n\r\n\r\n \r\n]>\r\n\r\nNegative test. Wrong character.\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P63-ibm63n04.xml", () => { + // 3.4 — Tests ignoreSect with a required field missing. The key word "IGNORE" is missing in the + // ignoreSect in the file ibm63n04.dtd. (upstream: not-wf; external parameter entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n \r\n]>\r\n\r\nNegative test. Missing required field(The keyword IGNORE is missing).\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P63-ibm63n05.xml", () => { + // 3.4 — Tests ignoreSect with a required field missing. The "[" is missing after the key word "IGNORE" + // in the ignoreSect in the file ibm63n05.dtd. (upstream: not-wf; external parameter entities are not + // read) + const input: string = + '\r\n\r\n\r\n\r\n \r\n]>\r\n\r\nNegative test. Missing required field( \'[\' is missing after IGNORE ).\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P63-ibm63n06.xml", () => { + // 3.4 — Tests includeSect with wrong field ordering. The two external subset declarations occur before + // the key word "IGNORE" in the ignoreSect in the file ibm63n06.dtd. (upstream: not-wf; external + // parameter entities are not read) + const input: string = + '\r\n\r\n \r\n]>\r\n\r\nNegative test. Wrong Ordering. Ignore sect contents preceding IGNORE.\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P63-ibm63n07.xml", () => { + // 3.4 — Tests ignoreSect with a required field missing. The closing sequence "]](greater than)" is + // missing in the ignoreSect in the file ibm63n07.dtd. (upstream: not-wf; external parameter entities + // are not read) + const input: string = + '\r\n\r\n\r\n\r\n \r\n]>\r\n\r\nNegative test. Missing closing sequence.\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P64-ibm64n01.xml", () => { + // 3.4 — Tests ignoreSectContents with wrong beginning sequence. The "?" occurs in beginning sequence + // the ignoreSectContents in the file ibm64n01.dtd. (upstream: not-wf; external parameter entities are + // not read) + const input: string = + '\r\n\r\n\r\n]>\r\n\r\nNegative Test. Pattern2. Wrong character.\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P64-ibm64n02.xml", () => { + // 3.4 — Tests ignoreSectContents with a required field missing.The closing sequence is missing in the + // ignoreSectContents in the file ibm64n02.dtd. (upstream: not-wf; external parameter entities are not + // read) + const input: string = + '\r\n\r\n\r\n]>\r\n\r\nNegative Test. Pattern3. Missing closing sequence.\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P64-ibm64n03.xml", () => { + // 3.4 — Tests ignoreSectContents with a required field missing.The beginning sequence is missing in + // the ignoreSectContents in the file ibm64n03.dtd. (upstream: not-wf; external parameter entities are + // not read) + const input: string = + '\r\n\r\n\r\n]>\r\n\r\nNegative Test. Pattern4. Missing opening sequence.\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P65-ibm65n01.xml", () => { + // 3.4 — Tests Ignore with illegal string included. The string "]](greater than)" is contained before + // "this" in the Ignore in the ignoreSectContents in the file ibm65n01.dtd (upstream: not-wf; external + // parameter entities are not read) + const input: string = + '\r\n\r\n\r\n]>\r\n\r\nNegative Test. Pattern1.Illegal sequence of \']]\'\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P65-ibm65n02.xml", () => { + // 3.4 — Tests Ignore with illegal string included. The string "(less than)![" is contained before + // "this" in the Ignore in the ignoreSectContents in the file ibm65n02.dtd (upstream: not-wf; external + // parameter entities are not read) + const input: string = + '\r\n\r\n\r\n]>\r\n\r\nNegative Test. Pattern2. \r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P66-ibm66n01.xml", () => { + // 4.1 — Tests CharRef with an illegal character referred to. The "#002f" is used as the referred + // character in the CharRef in the EntityDecl in the DTD. + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Invalid character reference: expected a number followed by ';' but found 'f'", + ); + }); + + test("ibm-not-wf-P66-ibm66n02.xml", () => { + // 4.1 — Tests CharRef with the semicolon character missing. The semicolon character is missing at the + // end of the CharRef in the attribute value in the STag of element "root". + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Invalid character reference: expected a number followed by ';' but found '\"'", + ); + }); + + test("ibm-not-wf-P66-ibm66n03.xml", () => { + // 4.1 — Tests CharRef with an illegal character referred to. The "49" is used as the referred + // character in the CharRef in the EntityDecl in the DTD. + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected an entity name after '&' but found '4'"); + }); + + test("ibm-not-wf-P66-ibm66n04.xml", () => { + // 4.1 — Tests CharRef with an illegal character referred to. The "#5~0" is used as the referred + // character in the attribute value in the EmptyElemTag of the element "root". + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Invalid character reference: expected a number followed by ';' but found '~'", + ); + }); + + test("ibm-not-wf-P66-ibm66n05.xml", () => { + // 4.1 — Tests CharRef with an illegal character referred to. The "#x002g" is used as the referred + // character in the CharRef in the EntityDecl in the DTD. + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Invalid character reference: expected a number followed by ';' but found 'g'", + ); + }); + + test("ibm-not-wf-P66-ibm66n06.xml", () => { + // 4.1 — Tests CharRef with an illegal character referred to. The "#x006G" is used as the referred + // character in the attribute value in the EmptyElemTag of the element "root". + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Invalid character reference: expected a number followed by ';' but found 'G'", + ); + }); + + test("ibm-not-wf-P66-ibm66n07.xml", () => { + // 4.1 — Tests CharRef with an illegal character referred to. The "#0=2f" is used as the referred + // character in the CharRef in the EntityDecl in the DTD. + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Invalid character reference: expected a number followed by ';' but found '='", + ); + }); + + test("ibm-not-wf-P66-ibm66n08.xml", () => { + // 4.1 — Tests CharRef with an illegal character referred to. The "#56.0" is used as the referred + // character in the attribute value in the EmptyElemTag of the element "root". + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Invalid character reference: expected a number followed by ';' but found '.'", + ); + }); + + test("ibm-not-wf-P66-ibm66n09.xml", () => { + // 4.1 — Tests CharRef with an illegal character referred to. The "#x00/2f" is used as the referred + // character in the CharRef in the EntityDecl in the DTD. + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Invalid character reference: expected a number followed by ';' but found '/'", + ); + }); + + test("ibm-not-wf-P66-ibm66n10.xml", () => { + // 4.1 — Tests CharRef with an illegal character referred to. The "#51)" is used as the referred + // character in the attribute value in the EmptyElemTag of the element "root". + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Invalid character reference: expected a number followed by ';' but found ')'", + ); + }); + + test("ibm-not-wf-P66-ibm66n11.xml", () => { + // 4.1 — Tests CharRef with an illegal character referred to. The "#00 2f" is used as the referred + // character in the CharRef in the EntityDecl in the DTD. + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Invalid character reference: expected a number followed by ';' but found space", + ); + }); + + test("ibm-not-wf-P66-ibm66n12.xml", () => { + // 4.1 — Tests CharRef with an illegal character referred to. The "#x0000" is used as the referred + // character in the attribute value in the EmptyElemTag of the element "root". + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Character reference '�' is not a valid XML character"); + }); + + test("ibm-not-wf-P66-ibm66n13.xml", () => { + // 4.1 — Tests CharRef with an illegal character referred to. The "#x001f" is used as the referred + // character in the attribute value in the EmptyElemTag of the element "root". + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Character reference '' is not a valid XML character"); + }); + + test("ibm-not-wf-P66-ibm66n14.xml", () => { + // 4.1 — Tests CharRef with an illegal character referred to. The "#xfffe" is used as the referred + // character in the attribute value in the EmptyElemTag of the element "root". + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Character reference '￾' is not a valid XML character"); + }); + + test("ibm-not-wf-P66-ibm66n15.xml", () => { + // 4.1 — Tests CharRef with an illegal character referred to. The "#xffff" is used as the referred + // character in the attribute value in the EmptyElemTag of the element "root". + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Character reference '￿' is not a valid XML character"); + }); + + test("ibm-not-wf-P68-ibm68n01.xml", () => { + // 4.1 — Tests EntityRef with a required field missing. The Name is missing in the EntityRef in the + // content of the element "root". + const input: string = + '\r\n\r\n\r\n]>\r\nmissing entity name &;\r\n'; + expectRejects(input, "XML Parse error: Expected an entity name after '&' but found ';'"); + }); + + test("ibm-not-wf-P68-ibm68n02.xml", () => { + // 4.1 — Tests EntityRef with a required field missing. The semicolon is missing in the EntityRef in + // the attribute value in the element "root". + const input: string = + '\r\n\r\n\r\n]>\r\nmissing semi-colon\r\n'; + expectRejects(input, "XML Parse error: Expected ';' after the entity name but found '\"'"); + }); + + test("ibm-not-wf-P68-ibm68n03.xml", () => { + // 4.1 — Tests EntityRef with an extra white space. A white space occurs after the ampersand in the + // EntityRef in the content of the element "root". + const input: string = + '\r\n\r\n\r\n]>\r\nextra space after ampsand & aaa;\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected an entity name after '&' but found space"); + }); + + test("ibm-not-wf-P68-ibm68n04.xml", () => { + // 4.1 — Tests EntityRef which is against P68 WFC: Entity Declared. The name "aAa" in the EntityRef in + // the AttValue in the STage of the element "root" does not match the Name of any declared entity in + // the DTD. + const input: string = + '\r\n\r\n\r\n]>\r\nreference doesn\'t match delaration\r\n'; + expectRejects(input, "XML Parse error: Entity 'aAa' is not declared"); + }); + + test("ibm-not-wf-P68-ibm68n05.xml", () => { + // 4.1 — Tests EntityRef which is against P68 WFC: Entity Declared. The entity with the name "aaa" in + // the EntityRef in the AttValue in the STag of the element "root" is not declared. + const input: string = + "\r\n\r\n]>\r\nundefined entitiy &aaa; \r\n"; + expectRejects(input, "XML Parse error: Entity 'aaa' is not declared"); + }); + + test("ibm-not-wf-P68-ibm68n06.xml", () => { + // 4.1 — Tests EntityRef which is against P68 WFC: Entity Declared. The entity with the name "aaa" in + // the EntityRef in the AttValue in the STag of the element "root" is externally declared, but + // standalone is "yes". (upstream: not-wf; external parameter entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n]>\r\nentity declared externally but standalone is yes\r\n'; + expectRejects(input, "XML Parse error: Entity 'aaa' is not declared"); + }); + + test("ibm-not-wf-P68-ibm68n07.xml", () => { + // 4.1 — Tests EntityRef which is against P68 WFC: Entity Declared. The entity with the name "aaa" in + // the EntityRef in the AttValue in the STag of the element "root" is referred before declared. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Entity 'aaa' is not declared"); + }); + + test("ibm-not-wf-P68-ibm68n08.xml", () => { + // 4.1 — Tests EntityRef which is against P68 WFC: Parsed Entity. The EntityRef in the AttValue in the + // STag of the element "root" contains the name "aImage" of an unparsed entity. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\nunparsed entity reference in the wrong place &aImage;\r\n'; + expectRejects(input, "XML Parse error: Unparsed entity 'aImage' cannot be referenced"); + }); + + test("ibm-not-wf-P68-ibm68n09.xml", () => { + // 4.1 — Tests EntityRef which is against P68 WFC: No Recursion. The recursive entity reference occurs + // with the entity declarations for "aaa" and "bbb" in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\n&aaa;\r\n\r\n'; + expectRejects(input, "XML Parse error: Entity 'aaa' refers to itself"); + }); + + test("ibm-not-wf-P68-ibm68n10.xml", () => { + // 4.1 — Tests EntityRef which is against P68 WFC: No Recursion. The indirect recursive entity + // reference occurs with the entity declarations for "aaa", "bbb", "ccc", "ddd", and "eee" in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n&aaa;\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Entity 'aaa' refers to itself"); + }); + + test("ibm-not-wf-P69-ibm69n01.xml", () => { + // 4.1 — Tests PEReference with a required field missing. The Name "paaa" is missing in the PEReference + // in the DTD. + const input: string = + '\r\n">\r\n\r\n%;\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected a markup declaration or ']' in the internal subset but found '%'"); + }); + + test("ibm-not-wf-P69-ibm69n02.xml", () => { + // 4.1 — Tests PEReference with a required field missing. The semicolon is missing in the PEReference + // "%paaa" in the DTD. + const input: string = + '\r\n">\r\n\r\n%paaa\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected ';' to end the parameter entity reference 'paaa'"); + }); + + test("ibm-not-wf-P69-ibm69n03.xml", () => { + // 4.1 — Tests PEReference with an extra white space. There is an extra white space occurs before ";" + // in the PEReference in the DTD. + const input: string = + '\r\n">\r\n\r\n%paaa ;\r\n\r\n]>\r\n\r\n\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected ';' to end the parameter entity reference 'paaa'"); + }); + + test("ibm-not-wf-P69-ibm69n04.xml", () => { + // 4.1 — Tests PEReference with an extra white space. There is an extra white space occurs after "%" in + // the PEReference in the DTD. + const input: string = + '\r\n">\r\n\r\n% paaa;\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected a markup declaration or ']' in the internal subset but found '%'"); + }); + + test("ibm-not-wf-P69-ibm69n05.xml", () => { + // 4.1 — Based on E29 substantial source: minutes XML-Syntax 1999-02-24 E38 in XML 1.0 Errata, this WFC + // does not apply to P69, but the VC Entity declared still apply. Tests PEReference which is against + // P69 WFC: Entity Declared. The PE with the name "paaa" is referred before declared in the DTD. + // (upstream: optional error) + const input: string = + '\r\n\r\n\r\n%paaa;\r\n">\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Parameter entity 'paaa' is not declared"); + }); + + test("ibm-not-wf-P69-ibm69n06.xml", () => { + // 4.1 — Tests PEReference which is against P69 WFC: No Recursion. The recursive PE reference occurs + // with the entity declarations for "paaa" and "bbb" in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Parameter entity references are not allowed inside markup declarations in the internal subset", + ); + }); + + test("ibm-not-wf-P69-ibm69n07.xml", () => { + // 4.1 — Tests PEReference which is against P69 WFC: No Recursion. The indirect recursive PE reference + // occurs with the entity declarations for "paaa", "bbb", "ccc", "ddd", and "eee" in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Parameter entity references are not allowed inside markup declarations in the internal subset", + ); + }); + + test("ibm-not-wf-P71-ibm70n01.xml", () => { + // 4.2 — Tests + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Expected a markup declaration or ']' in the internal subset but found ' { + // 4.2 — Tests EntityDecl with a required field missing. The white space is missing between the + // beginning sequence and the Name "aaa" in the EntityDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\n&aaa;\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before 'aaa'"); + }); + + test("ibm-not-wf-P71-ibm71n02.xml", () => { + // 4.2 — Tests EntityDecl with a required field missing. The white space is missing between the Name + // "aaa" and the EntityDef "aString" in the EntityDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\n&aaa;\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before a quoted string"); + }); + + test("ibm-not-wf-P71-ibm71n03.xml", () => { + // 4.2 — Tests EntityDecl with a required field missing. The EntityDef is missing in the EntityDecl + // with the Name "aaa" in the DTD. + const input: string = + "\r\n\r\n\r\n\r\n\r\n]>\r\n&aaa;\r\n"; + expectRejects(input, "XML Parse error: Expected a quoted entity value, SYSTEM or PUBLIC but found '>'"); + }); + + test("ibm-not-wf-P71-ibm71n04.xml", () => { + // 4.2 — Tests EntityDecl with a required field missing. The Name is missing in the EntityDecl with the + // EntityDef "aString" in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\n&aaa;\r\n'; + expectRejects(input, "XML Parse error: Expected an entity name or '%' after ' { + // 4.2 — Tests EntityDecl with wrong ordering. The Name "aaa" occurs after the EntityDef in the + // EntityDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n&aaa;\r\n'; + expectRejects(input, "XML Parse error: Expected an entity name or '%' after ' { + // 4.2 — Tests EntityDecl with wrong key word. The string "entity" is used as the key word in the + // beginning sequence in the EntityDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n&aaa;\r\n'; + expectRejects( + input, + "XML Parse error: ' { + // 4.2 — Tests EntityDecl with a required field missing. The closing bracket (greater than) is missing + // in the EntityDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n&aaa;\r\n'; + expectRejects(input, "XML Parse error: Expected '>' to end the entity declaration but found ']'"); + }); + + test("ibm-not-wf-P71-ibm71n08.xml", () => { + // 4.2 — Tests EntityDecl with a required field missing. The exclamation mark is missing in the + // beginning sequence in the EntityDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\n&aaa;\r\n'; + expectRejects( + input, + "XML Parse error: Expected a markup declaration or ']' in the internal subset but found ' { + // 4.2 — Tests PEdecl with a required field missing. The white space is missing between the beginning + // sequence and the "%" in the PEDecl in the DTD. + const input: string = + '\r\n\r\n\r\n">\r\n%paaa;\r\n]>\r\n\r\n\r\n\r\n\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before '%'"); + }); + + test("ibm-not-wf-P72-ibm72n02.xml", () => { + // 4.2 — Tests PEdecl with a required field missing. The Name is missing in the PEDecl in the DTD. + const input: string = + '\r\n\r\n\r\n">\r\n%paaa;\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected the parameter entity name after '%' but found '\"'"); + }); + + test("ibm-not-wf-P72-ibm72n03.xml", () => { + // 4.2 — Tests PEdecl with a required field missing. The white space is missing between the Name and + // the PEDef in the PEDecl in the DTD. + const input: string = + '\r\n\r\n\r\n">\r\n%paaa;\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before a quoted string"); + }); + + test("ibm-not-wf-P72-ibm72n04.xml", () => { + // 4.2 — Tests PEdecl with a required field missing. The PEDef is missing after the Name "paaa" in the + // PEDecl in the DTD. + const input: string = + "\r\n\r\n\r\n\r\n%paaa;\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected a quoted entity value, SYSTEM or PUBLIC but found '>'"); + }); + + test("ibm-not-wf-P72-ibm72n05.xml", () => { + // 4.2 — Tests PEdecl with wrong field ordering. The Name "paaa" occurs after the PEDef in the PEDecl + // in the DTD. + const input: string = + '\r\n\r\n\r\n" paaa>\r\n%paaa;\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected the parameter entity name after '%' but found '\"'"); + }); + + test("ibm-not-wf-P72-ibm72n06.xml", () => { + // 4.2 — Tests PEdecl with wrong field ordering. The "%" and the Name "paaa" occurs after the PEDef in + // the PEDecl in the DTD. + const input: string = + '\r\n\r\n\r\n" % paaa >\r\n%paaa;\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected an entity name or '%' after ' { + // 4.2 — Tests PEdecl with wrong key word. The string "entity" is used as the key word in the beginning + // sequence in the PEDecl in the DTD. + const input: string = + '\r\n\r\n\r\n">\r\n%paaa;\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: ' { + // 4.2 — Tests PEdecl with a required field missing. The closing bracket (greater than) is missing in + // the PEDecl in the DTD. + const input: string = + '\r\n\r\n\r\n"\r\n%paaa;\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Parameter entity references are not allowed inside markup declarations in the internal subset", + ); + }); + + test("ibm-not-wf-P72-ibm72n09.xml", () => { + // 4.2 — Tests PEdecl with wrong closing sequence. The string "!(greater than)" is used as the closing + // sequence in the PEDecl in the DTD. + const input: string = + '\r\n\r\n\r\n" !>\r\n%paaa;\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before '%'"); + }); + + test("ibm-not-wf-P73-ibm73n01.xml", () => { + // 4.2 — Tests EntityDef with wrong field ordering. The NDataDecl "NDATA JPGformat" occurs before the + // ExternalID in the EntityDef in the EntityDecl. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected a quoted entity value, SYSTEM or PUBLIC but found 'NDATA'"); + }); + + test("ibm-not-wf-P73-ibm73n03.xml", () => { + // 4.2 — Tests EntityDef with a required field missing. The ExternalID is missing before the NDataDecl + // in the EntityDef in the EntityDecl. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected a quoted entity value, SYSTEM or PUBLIC but found 'NDATA'"); + }); + + test("ibm-not-wf-P74-ibm74n01.xml", () => { + // 4.2 — Tests PEDef with extra fields. The NDataDecl occurs after the ExternalID in the PEDef in the + // PEDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Parameter entities cannot have NDATA"); + }); + + test("ibm-not-wf-P75-ibm75n01.xml", () => { + // 4.2.2 — Tests ExternalID with wrong key word. The string "system" is used as the key word in the + // ExternalID in the EntityDef in the EntityDecl. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected a quoted entity value, SYSTEM or PUBLIC but found 'system'"); + }); + + test("ibm-not-wf-P75-ibm75n02.xml", () => { + // 4.2.2 — Tests ExternalID with wrong key word. The string "public" is used as the key word in the + // ExternalID in the doctypedecl. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found 'public'", + ); + }); + + test("ibm-not-wf-P75-ibm75n03.xml", () => { + // 4.2.2 — Tests ExternalID with wrong key word. The string "Public" is used as the key word in the + // ExternalID in the doctypedecl. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found 'Public'", + ); + }); + + test("ibm-not-wf-P75-ibm75n04.xml", () => { + // 4.2.2 — Tests ExternalID with wrong field ordering. The key word "PUBLIC" occurs after the + // PublicLiteral and the SystemLiteral in the ExternalID in the doctypedecl. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '\"'", + ); + }); + + test("ibm-not-wf-P75-ibm75n05.xml", () => { + // 4.2.2 — Tests ExternalID with a required field missing. The white space between "SYSTEM" and the + // Systemliteral is missing in the ExternalID in the EntityDef in the EntityDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before a quoted string"); + }); + + test("ibm-not-wf-P75-ibm75n06.xml", () => { + // 4.2.2 — Tests ExternalID with a required field missing. The Systemliteral is missing after "SYSTEM" + // in the ExternalID in the EntityDef in the EntityDecl in the DTD. + const input: string = + "\r\n\r\n\r\n\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected a quoted system identifier after SYSTEM but found '>'"); + }); + + test("ibm-not-wf-P75-ibm75n07.xml", () => { + // 4.2.2 — Tests ExternalID with a required field missing. The white space between the PublicLiteral + // and the Systemliteral is missing in the ExternalID in the doctypedecl. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before a quoted string"); + }); + + test("ibm-not-wf-P75-ibm75n08.xml", () => { + // 4.2.2 — Tests ExternalID with a required field missing. The key word "PUBLIC" is missing in the + // ExternalID in the doctypedecl. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '\"'", + ); + }); + + test("ibm-not-wf-P75-ibm75n09.xml", () => { + // 4.2.2 — Tests ExternalID with a required field missing. The white space between "PUBLIC" and the + // PublicLiteral is missing in the ExternalID in the doctypedecl. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before a quoted string"); + }); + + test("ibm-not-wf-P75-ibm75n10.xml", () => { + // 4.2.2 — Tests ExternalID with a required field missing. The PublicLiteral is missing in the + // ExternalID in the doctypedecl. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Invalid character in a public identifier: '\\'"); + }); + + test("ibm-not-wf-P75-ibm75n11.xml", () => { + // 4.2.2 — Tests ExternalID with a required field missing. The PublicLiteral is missing in the + // ExternalID in the doctypedecl. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found 'public'", + ); + }); + + test("ibm-not-wf-P75-ibm75n12.xml", () => { + // 4.2.2 — Tests ExternalID with a required field missing. The SystemLiteral is missing in the + // ExternalID in the doctypedecl. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected '>' to end the entity declaration but found 'SYSTEM'"); + }); + + test("ibm-not-wf-P75-ibm75n13.xml", () => { + // 4.2.2 — Tests ExternalID with wrong field ordering. The key word "PUBLIC" occurs after the + // PublicLiteral in the ExternalID in the doctypedecl. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '\"'", + ); + }); + + test("ibm-not-wf-P76-ibm76n01.xml", () => { + // 4.2.2 — Tests NDataDecl with wrong key word. The string "ndata" is used as the key word in the + // NDataDecl in the EntityDef in the GEDecl. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected '>' to end the entity declaration but found 'ndata'"); + }); + + test("ibm-not-wf-P76-ibm76n02.xml", () => { + // 4.2.2 — Tests NDataDecl with wrong key word. The string "NData" is used as the key word in the + // NDataDecl in the EntityDef in the GEDecl. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected '>' to end the entity declaration but found 'NData'"); + }); + + test("ibm-not-wf-P76-ibm76n03.xml", () => { + // 4.2.2 — Tests NDataDecl with a required field missing. The leading white space is missing in the + // NDataDecl in the EntityDef in the GEDecl. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before 'NDATA'"); + }); + + test("ibm-not-wf-P76-ibm76n04.xml", () => { + // 4.2.2 — Tests NDataDecl with a required field missing. The key word "NDATA" is missing in the + // NDataDecl in the EntityDef in the GEDecl. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected '>' to end the entity declaration but found 'JPGformat'"); + }); + + test("ibm-not-wf-P76-ibm76n05.xml", () => { + // 4.2.2 — Tests NDataDecl with a required field missing. The Name after the key word "NDATA" is + // missing in the NDataDecl in the EntityDef in the GEDecl. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected a notation name after NDATA but found '>'"); + }); + + test("ibm-not-wf-P76-ibm76n06.xml", () => { + // 4.2.2 — Tests NDataDecl with a required field missing. The white space between "NDATA" and the Name + // is missing in the NDataDecl in the EntityDef in the GEDecl. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected '>' to end the entity declaration but found 'NDATAJPGformat'"); + }); + + test("ibm-not-wf-P76-ibm76n07.xml", () => { + // 4.2.2 — Tests NDataDecl with wrong field ordering. The key word "NDATA" occurs after the Name in the + // NDataDecl in the EntityDef in the GEDecl. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected '>' to end the entity declaration but found 'JPGformat'"); + }); + + test("ibm-not-wf-P77-ibm77n01.xml", () => { + // 4.3.1 — Tests TextDecl with wrong field ordering. The VersionInfo occurs after the EncodingDecl in + // the TextDecl in the file "ibm77n01.ent". (upstream: not-wf; external general entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n&aExternal;\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P77-ibm77n02.xml", () => { + // 4.3.1 — Tests TextDecl with wrong key word. The string "XML" is used in the beginning sequence in + // the TextDecl in the file "ibm77n02.ent". (upstream: not-wf; external general entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n&aExternal;\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P77-ibm77n03.xml", () => { + // 4.3.1 — Tests TextDecl with wrong closing sequence. The character "greater than" is used as the + // closing sequence in the TextDecl in the file "ibm77n03.ent". (upstream: not-wf; external parameter + // entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n%pExternal;\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P77-ibm77n04.xml", () => { + // 4.3.1 — Tests TextDecl with a required field missing. The closing sequence is missing in the + // TextDecl in the file "ibm77n04.ent". (upstream: not-wf; external parameter entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n%pExternal;\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P78-ibm78n01.xml", () => { + // 4.3.2 — Tests extParsedEnt with wrong field ordering. The TextDecl occurs after the content in the + // file ibm78n01.ent. (upstream: not-wf; external general entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n&aExternal;\r\n\r\n\r\n\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P78-ibm78n02.xml", () => { + // 4.3.2 — Tests extParsedEnt with extra field. A blank line occurs before the TextDecl in the file + // ibm78n02.ent. (upstream: not-wf; external general entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n&aExternal;\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P79-ibm79n01.xml", () => { + // 4.3.2 — Tests extPE with wrong field ordering. The TextDecl occurs after the extSubsetDecl (the + // white space and the comment) in the file ibm79n01.ent. (upstream: not-wf; external parameter + // entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n%pExternal;\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P79-ibm79n02.xml", () => { + // 4.3.2 — Tests extPE with extra field. A blank line occurs before the TextDecl in the file + // ibm78n02.ent. (upstream: not-wf; external parameter entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n%pExternal;\r\n]>\r\n\r\n'; + expectParses(input); + }); + + test("ibm-not-wf-P80-ibm80n01.xml", () => { + // 4.3.3 — Tests EncodingDecl with a required field missing. The leading white space is missing in the + // EncodingDecl in the XMLDecl. + const input = Buffer.from( + '\r\n\r\n\r\n\r\n]>\r\n\r\n', + ); + expectRejects(input, "XML Parse error: Whitespace is required before 'encoding'"); + }); + + test("ibm-not-wf-P80-ibm80n02.xml", () => { + // 4.3.3 — Tests EncodingDecl with a required field missing. The "=" sign is missing in the + // EncodingDecl in the XMLDecl. + const input = Buffer.from( + '\r\n\r\n\r\n\r\n]>\r\n\r\n', + ); + expectRejects(input, "XML Parse error: Expected '=' after the name in the XML declaration but found '\"'"); + }); + + test("ibm-not-wf-P80-ibm80n03.xml", () => { + // 4.3.3 — Tests EncodingDecl with a required field missing. The double quoted EncName are missing in + // the EncodingDecl in the XMLDecl. + const input = Buffer.from( + '\r\n\r\n\r\n\r\n]>\r\n\r\n', + ); + expectRejects(input, "XML Parse error: Expected a quoted value in the XML declaration but found '?'"); + }); + + test("ibm-not-wf-P80-ibm80n04.xml", () => { + // 4.3.3 — Tests EncodingDecl with wrong field ordering. The string "encoding=" occurs after the double + // quoted EncName in the EncodingDecl in the XMLDecl. + const input = Buffer.from( + '\r\n\r\n\r\n\r\n]>\r\n\r\n', + ); + expectRejects(input, "XML Parse error: Expected '?>' to end the XML declaration but found '\"'"); + }); + + test("ibm-not-wf-P80-ibm80n05.xml", () => { + // 4.3.3 — Tests EncodingDecl with wrong field ordering. The "encoding" occurs after the double quoted + // EncName in the EncodingDecl in the XMLDecl. + const input = Buffer.from( + '\r\n\r\n\r\n\r\n]>\r\n\r\n', + ); + expectRejects(input, "XML Parse error: Expected '?>' to end the XML declaration but found '\"'"); + }); + + test("ibm-not-wf-P80-ibm80n06.xml", () => { + // 4.3.3 — Tests EncodingDecl with wrong key word. The string "Encoding" is used as the key word in the + // EncodingDecl in the XMLDecl. + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Unexpected 'Encoding' in the XML declaration (expected version, encoding or standalone)", + ); + }); + + test("ibm-not-wf-P81-ibm81n01.xml", () => { + // 4.3.3 — Tests EncName with an illegal character. The "_" is used as the first character in the + // EncName in the EncodingDecl in the XMLDecl. + const input = Buffer.from( + '\r\n\r\n\r\n\r\n]>\r\n\r\n\r\n', + ); + expectRejects(input, "XML Parse error: Invalid encoding name '_UTF-8' in the XML declaration"); + }); + + test("ibm-not-wf-P81-ibm81n02.xml", () => { + // 4.3.3 — Tests EncName with an illegal character. The "-" is used as the first character in the + // EncName in the EncodingDecl in the XMLDecl. + const input = Buffer.from( + '\r\n\r\n\r\n\r\n]>\r\n\r\n', + ); + expectRejects(input, "XML Parse error: Invalid encoding name '-UTF-8' in the XML declaration"); + }); + + test("ibm-not-wf-P81-ibm81n03.xml", () => { + // 4.3.3 — Tests EncName with an illegal character. The "." is used as the first character in the + // EncName in the EncodingDecl in the XMLDecl. + const input = Buffer.from( + '\r\n\r\n\r\n\r\n]>\r\n\r\n', + ); + expectRejects(input, "XML Parse error: Invalid encoding name '.UTF-8' in the XML declaration"); + }); + + test("ibm-not-wf-P81-ibm81n04.xml", () => { + // 4.3.3 — Tests EncName with illegal characters. The "8-" is used as the initial characters in the + // EncName in the EncodingDecl in the XMLDecl. + const input = Buffer.from( + '\r\n\r\n\r\n\r\n]>\r\n\r\n', + ); + expectRejects(input, "XML Parse error: Invalid encoding name '8-UTF' in the XML declaration"); + }); + + test("ibm-not-wf-P81-ibm81n05.xml", () => { + // 4.3.3 — Tests EncName with an illegal character. The "~" is used as one character in the EncName in + // the EncodingDecl in the XMLDecl. + const input = Buffer.from( + '\r\n\r\n\r\n\r\n]>\r\n\r\n', + ); + expectRejects(input, "XML Parse error: Invalid encoding name 'UTF~8' in the XML declaration"); + }); + + test("ibm-not-wf-P81-ibm81n06.xml", () => { + // 4.3.3 — Tests EncName with an illegal character. The "#" is used as one character in the EncName in + // the EncodingDecl in the XMLDecl. + const input = Buffer.from( + '\r\n\r\n\r\n\r\n]>\r\n\r\n', + ); + expectRejects(input, "XML Parse error: Invalid encoding name 'UTF#8' in the XML declaration"); + }); + + test("ibm-not-wf-P81-ibm81n07.xml", () => { + // 4.3.3 — Tests EncName with an illegal character. The ":" is used as one character in the EncName in + // the EncodingDecl in the XMLDecl. + const input = Buffer.from( + '\r\n\r\n\r\n\r\n]>\r\n\r\n', + ); + expectRejects(input, "XML Parse error: Invalid encoding name 'UTF:8' in the XML declaration"); + }); + + test("ibm-not-wf-P81-ibm81n08.xml", () => { + // 4.3.3 — Tests EncName with an illegal character. The "/" is used as one character in the EncName in + // the EncodingDecl in the XMLDecl. + const input = Buffer.from( + '\r\n\r\n\r\n\r\n]>\r\n\r\n', + ); + expectRejects(input, "XML Parse error: Invalid encoding name 'UTF/8' in the XML declaration"); + }); + + test("ibm-not-wf-P81-ibm81n09.xml", () => { + // 4.3.3 — Tests EncName with an illegal character. The ";" is used as one character in the EncName in + // the EncodingDecl in the XMLDecl. + const input = Buffer.from( + '\r\n\r\n\r\n\r\n]>\r\n\r\n', + ); + expectRejects(input, "XML Parse error: Invalid encoding name 'UTF;8' in the XML declaration"); + }); + + test("ibm-not-wf-P82-ibm82n01.xml", () => { + // 4.7 — Tests NotationDecl with a required field missing. The white space after the beginning sequence + // of the NotationDecl is missing in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before 'JPGformat'"); + }); + + test("ibm-not-wf-P82-ibm82n02.xml", () => { + // 4.7 — Tests NotationDecl with a required field missing. The Name in the NotationDecl is missing in + // the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected SYSTEM or PUBLIC in the notation declaration but found '\"'"); + }); + + test("ibm-not-wf-P82-ibm82n03.xml", () => { + // 4.7 — Tests NotationDecl with a required field missing. The externalID or the PublicID is missing in + // the NotationDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected SYSTEM or PUBLIC in the notation declaration but found '>'"); + }); + + test("ibm-not-wf-P82-ibm82n04.xml", () => { + // 4.7 — Tests NotationDecl with wrong field ordering. The Name occurs after the "SYSTEM" and the + // externalID in the NotationDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected SYSTEM or PUBLIC in the notation declaration but found '\"'"); + }); + + test("ibm-not-wf-P82-ibm82n05.xml", () => { + // 4.7 — Tests NotationDecl with wrong key word. The string "notation" is used as a key word in the + // NotationDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects( + input, + "XML Parse error: ' { + // 4.7 — Tests NotationDecl with a required field missing. The closing bracket (the greater than + // character) is missing in the NotationDecl. + const input: string = + '\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected '>' to end the notation declaration but found ' { + // 4.7 — Tests NotationDecl with wrong beginning sequence. The "!" is missing in the beginning sequence + // in the NotationDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n'; + expectRejects( + input, + "XML Parse error: Expected a markup declaration or ']' in the internal subset but found ' { + // 4.7 — Tests NotationDecl with wrong closing sequence. The extra "!" occurs in the closing sequence + // in the NotationDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected '>' to end the notation declaration but found '!'"); + }); + + test("ibm-not-wf-P83-ibm83n01.xml", () => { + // 4.7 — Tests PublicID with wrong key word. The string "public" is used as the key word in the + // PublicID in the NotationDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected SYSTEM or PUBLIC in the notation declaration but found 'public'"); + }); + + test("ibm-not-wf-P83-ibm83n02.xml", () => { + // 4.7 — Tests PublicID with wrong key word. The string "Public" is used as the key word in the + // PublicID in the NotationDecl in the DTD. + const input: string = + 'r\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected the root element but found 'r'"); + }); + + test("ibm-not-wf-P83-ibm83n03.xml", () => { + // 4.7 — Tests PublicID with a required field missing. The key word "PUBLIC" is missing in the PublicID + // in the NotationDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected SYSTEM or PUBLIC in the notation declaration but found '\"'"); + }); + + test("ibm-not-wf-P83-ibm83n04.xml", () => { + // 4.7 — Tests PublicID with a required field missing. The white space between the "PUBLIC" and the + // PubidLiteral is missing in the PublicID in the NotationDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Whitespace is required before a quoted string"); + }); + + test("ibm-not-wf-P83-ibm83n05.xml", () => { + // 4.7 — Tests PublicID with a required field missing. The PubidLiteral is missing in the PublicID in + // the NotationDecl in the DTD. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected a quoted public identifier after PUBLIC but found '>'"); + }); + + test("ibm-not-wf-P83-ibm83n06.xml", () => { + // 4.7 — Tests PublicID with wrong field ordering. The key word "PUBLIC" occurs after the PubidLiteral + // in the PublicID in the NotationDecl. + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n'; + expectRejects(input, "XML Parse error: Expected SYSTEM or PUBLIC in the notation declaration but found '\"'"); + }); + + test("ibm-not-wf-P85-ibm85n01.xml", () => { + // B. — Tests BaseChar with an illegal character. The character #x00D7 occurs as the first character of + // the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected a processing instruction target after ' { + // B. — Tests BaseChar with an illegal character. The character #x00F7 occurs as the first character of + // the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects(input, "XML Parse error: Expected a processing instruction target after ' { + // B. — Tests Digit with an illegal character. The character #x0029 occurs as the second character in + // the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected whitespace or '?>' after the processing instruction target but found ')'", + ); + }); + + test("ibm-not-wf-P88-ibm88n02.xml", () => { + // B. — Tests Digit with an illegal character. The character #x003B occurs as the second character in + // the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected whitespace or '?>' after the processing instruction target but found ';'", + ); + }); + + test("ibm-not-wf-P89-ibm89n01.xml", () => { + // B. — Tests Extender with an illegal character. The character #x00B6 occurs as the second character + // in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected whitespace or '?>' after the processing instruction target but found '¶' (U+00B6)", + ); + }); + + test("ibm-not-wf-P89-ibm89n02.xml", () => { + // B. — Tests Extender with an illegal character. The character #x00B8 occurs as the second character + // in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected whitespace or '?>' after the processing instruction target but found '¸' (U+00B8)", + ); + }); + + test("ibm-valid-P01-ibm01v01.xml", () => { + // 2.1 — Tests with a xml document consisting of prolog followed by element then Misc + const input = Buffer.from( + '\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n \r\n\r\n \r\n This is a white tiger in Mirage!!\r\n \r\n \r\n \r\n \r\n \r\n\r\n\r\n\r\n', + ); + const canonical = + ' This is a white tiger in Mirage!! '; + const compact: unknown = { + animal: { + cat: ["", ""], + tiger: { "@color": "white", "#text": "This is a white tiger in Mirage!!" }, + leopard: { small: "", big: "" }, + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P02-ibm02v01.xml", () => { + // 2.2 — This test case covers legal character ranges plus discrete legal characters for production 02. + const input = Buffer.from( + '\r\n\r\n\r\n\r\n]>\r\n\r\n', + ); + expectParses(input); + }); + + test("ibm-valid-P03-ibm03v01.xml", () => { + // 2.3 — Tests all 4 legal white space characters - #x20 #x9 #xD #xA + const input = Buffer.from( + '\r\r\n\r\r\n\r\r\n\r\r\n]>\r\r\n\r\r\n', + ); + expectParses(input); + }); + + test("ibm-valid-P09-ibm09v01.xml", () => { + // 2.3 — Empty EntityValue is legal + const input: string = + '\r\n \r\n \t\r\n]>\r\n\r\nMy Name is &FullName;. \r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n '; + const canonical = "My Name is . "; + const compact: unknown = { student: "My Name is ." }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P09-ibm09v02.xml", () => { + // 2.3 — Tests a normal EnitityValue + const input: string = + '\r\n \r\n \t\r\n]>\r\n\r\nMy Name is &FullName;. '; + const canonical = "My Name is SnowMan. "; + const compact: unknown = { student: "My Name is SnowMan." }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P09-ibm09v03.xml", () => { + // 2.3 — Tests EnitityValue referencing a Parameter Entity (upstream: valid; external parameter + // entities are not read; output depends on them) + const input: string = + '\r\n\r\nI am a new student with &Name;\r\n'; + expectParses(input); + }); + + test("ibm-valid-P09-ibm09v04.xml", () => { + // 2.3 — Tests EnitityValue referencing a General Entity + const input: string = + ' \r\n\r\n\r\n\t \r\n \t\r\n]>\r\n\r\nMy Name is &FullName;. '; + const canonical = "My Name is SnowMan. "; + const compact: unknown = { student: "My Name is SnowMan." }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P09-ibm09v05.xml", () => { + // 2.3 — Tests EnitityValue with combination of GE, PE and text, the GE used is declared in the + // student.dtd (upstream: valid; external parameter entities are not read; output depends on them) + const input: string = + '\r\n \r\n\t\r\n\t\r\n \t\r\n]>\r\n\r\n\r\nThis is a test of &combine;\r\n\r\n\r\n\r\n'; + expectParses(input); + }); + + test("ibm-valid-P10-ibm10v01.xml", () => { + // 2.3 — Tests empty AttValue with double quotes as the delimiters + const input: string = + '\r\n\r\n\t \r\n\t\r\n\t\r\n\t\r\n]>\r\n\r\nMy Name is Snow &mylast; Man. \r\n\r\n\r\n\r\n\r\n\r\n\r\n'; + const canonical = 'My Name is Snow Man. '; + const compact: unknown = { student: { "@first": "", "@last": "", "#text": "My Name is Snow Man." } }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P10-ibm10v02.xml", () => { + // 2.3 — Tests empty AttValue with single quotes as the delimiters + const input: string = + "\r\n\r\n\t \r\n\t\r\n\t\r\n\t\r\n]>\r\n\r\nMy Name is Snow &mylast; Man. \r\n\r\n"; + const canonical = 'My Name is Snow Man. '; + const compact: unknown = { student: { "@first": "", "@last": "", "#text": "My Name is Snow Man." } }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P10-ibm10v03.xml", () => { + // 2.3 — Test AttValue with double quotes as the delimiters and single quote inside + const input: string = + '\r\n\r\n\t \r\n\t\r\n\t\r\n\t\r\n]>\r\n\r\nMy Name is &myfirst; &mylast;. '; + const canonical = 'My Name is Snow Man\'. '; + const compact: unknown = { + student: { "@first": "Snow'", "@last": "Man", "#text": "My Name is Snow Man'." }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P10-ibm10v04.xml", () => { + // 2.3 — Test AttValue with single quotes as the delimiters and double quote inside + const input: string = + "\r\n\r\n\t \r\n\t\r\n\t\r\n\t\r\n]>\r\n\r\nMy Name is &myfirst; &mylast;. \r\n\r\n"; + const canonical = 'My Name is Snow Man". '; + const compact: unknown = { + student: { "@first": 'Snow"', "@last": "Man", "#text": 'My Name is Snow Man".' }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P10-ibm10v05.xml", () => { + // 2.3 — Test AttValue with a GE reference and double quotes as the delimiters + const input: string = + '\r\n\r\n\t \r\n\t\r\n\t\r\n\t\r\n]>\r\n\r\nMy Name is &mylast;. \r\n\r\n'; + const canonical = 'My Name is Snow Man. '; + const compact: unknown = { + student: { "@first": "Snow", "@last": "mylast;", "#text": "My Name is Snow Man." }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P10-ibm10v06.xml", () => { + // 2.3 — Test AttValue with a GE reference and single quotes as the delimiters + const input: string = + "\r\n\r\n\t \r\n\t\r\n\t\r\n\t\r\n]>\r\n\r\nMy Name is &mylast;. \r\n\r\n"; + const canonical = 'My Name is Snow Man. '; + const compact: unknown = { + student: { "@first": "Snow", "@last": "Snow Man", "#text": "My Name is Snow Man." }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P10-ibm10v07.xml", () => { + // 2.3 — testing AttValue with mixed references and text content in double quotes + const input: string = + '\r\n\r\n\t \r\n\t\r\n\t\r\n\t\r\n]>\r\n\r\nMy first Name is &myfirst; and my last name is &mylast;. \r\n'; + const canonical = + 'My first Name is Snow and my last name is Man Snow and Snow mymiddle;.. '; + const compact: unknown = { + student: { + "@first": "Full Name Snow 1 and Man Snow and Snow mymiddle;. Man Snow and Snow mymiddle;. c", + "@last": "Man Snow and Snow mymiddle;.", + "#text": "My first Name is Snow and my last name is Man Snow and Snow mymiddle;..", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P10-ibm10v08.xml", () => { + // 2.3 — testing AttValue with mixed references and text content in single quotes + const input: string = + "\r\n\r\n\t \r\n\t\r\n\t\r\n\t\r\n]>\r\n\r\nMy first Name is &myfirst; and my last name is &mylast;. \r\n\r\n"; + const canonical = + 'My first Name is Snow and my last name is Man Snow and Snow mymiddle;.. '; + const compact: unknown = { + student: { + "@first": 'Full Name Snow and "Man Snow and Snow mymiddle;." Man Snow and Snow mymiddle;.', + "@last": "Man Snow and Snow mymiddle;.", + "#text": "My first Name is Snow and my last name is Man Snow and Snow mymiddle;..", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P11-ibm11v01.xml", () => { + // 2.3 — Tests empty systemliteral using the double quotes + const input: string = + '\r\n\r\n \r\n]>\r\n\r\n\r\nMy Name is SnowMan. \r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n '; + const canonical = "My Name is SnowMan. "; + const compact: unknown = { student: "My Name is SnowMan." }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P11-ibm11v02.xml", () => { + // 2.3 — Tests empty systemliteral using the single quotes + const input: string = + "\r\n\r\n \r\n]>\r\n\r\n\r\nMy Name is SnowMan. \r\n"; + const canonical = "My Name is SnowMan. "; + const compact: unknown = { student: "My Name is SnowMan." }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P11-ibm11v03.xml", () => { + // 2.3 — Tests regular systemliteral using the single quotes (upstream: valid; external parameter + // entities are not read) + const input: string = + '\r\n\r\n\r\nMy Name is SnowMan. \r\n'; + const canonical = "My Name is SnowMan. "; + const compact: unknown = { student: "My Name is SnowMan." }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P11-ibm11v04.xml", () => { + // 2.3 — Tests regular systemliteral using the double quotes (upstream: valid; external parameter + // entities are not read) + const input: string = + '\r\n\r\n\r\n\r\nMy Name is SnowMan. \r\n\r\n'; + const canonical = "My Name is SnowMan. "; + const compact: unknown = { student: "My Name is SnowMan." }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P12-ibm12v01.xml", () => { + // 2.3 — Tests empty systemliteral using the double quotes (upstream: valid; external parameter + // entities are not read) + const input: string = + '\r\n\r\n\r\n\r\nMy Name is SnowMan. \r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n '; + const canonical = "My Name is SnowMan. "; + const compact: unknown = { student: "My Name is SnowMan." }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P12-ibm12v02.xml", () => { + // 2.3 — Tests empty systemliteral using the single quotes (upstream: valid; external parameter + // entities are not read) + const input: string = + "\r\n\r\n\r\n\r\nMy Name is SnowMan. \r\n"; + const canonical = "My Name is SnowMan. "; + const compact: unknown = { student: "My Name is SnowMan." }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P12-ibm12v03.xml", () => { + // 2.3 — Tests regular systemliteral using the double quotes (upstream: valid; external parameter + // entities are not read) + const input: string = + '\r\n\r\n\r\n\r\nMy Name is SnowMan. \r\n'; + const canonical = "My Name is SnowMan. "; + const compact: unknown = { student: "My Name is SnowMan." }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P12-ibm12v04.xml", () => { + // 2.3 — Tests regular systemliteral using the single quotes (upstream: valid; external parameter + // entities are not read) + const input: string = + "\r\n\r\n\r\n\r\nMy Name is SnowMan. \r\n"; + const canonical = "My Name is SnowMan. "; + const compact: unknown = { student: "My Name is SnowMan." }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P13-ibm13v01.xml", () => { + // 2.3 — Testing PubidChar with all legal PubidChar in a PubidLiteral (upstream: valid; external + // parameter entities are not read) + const input: string = + '\r\n\r\n\r\n\r\nMy Name is SnowMan. \r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n '; + const canonical = "My Name is SnowMan. "; + const compact: unknown = { student: "My Name is SnowMan." }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P14-ibm14v01.xml", () => { + // 2.4 — Testing CharData with empty string + const input: string = + '\r\n\r\n\t\r\n]>\r\n\r\n\r\n\r\n\r\n\r\n'; + const canonical = ''; + const compact: unknown = { student: { "@first": "Snow" } }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P14-ibm14v02.xml", () => { + // 2.4 — Testing CharData with white space character + const input: string = + '\r\n\r\n\t\r\n]>\r\n\r\n\r\n \r\n\r\n\r\n'; + const canonical = ' '; + const compact: unknown = { student: { "@first": "Eric" } }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P14-ibm14v03.xml", () => { + // 2.4 — Testing CharData with a general text string + const input: string = + '\r\n\r\n\t\r\n]>\r\n\r\n\r\nThis is a test'; + const canonical = 'This is a test'; + const compact: unknown = { student: { "@first": "Snow", "@last": "Man", "#text": "This is a test" } }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P15-ibm15v01.xml", () => { + // 2.5 — Tests empty comment + const input: string = + '\r\n \r\n]>\r\n\r\n\r\nMy Name is SnowMan. \r\n\r\n\r\n'; + const canonical = "My Name is SnowMan. "; + const compact: unknown = { student: "My Name is SnowMan." }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P15-ibm15v02.xml", () => { + // 2.5 — Tests comment with regular text + const input: string = + '\r\n \r\n]>\r\n\r\n\r\nMy Name is SnowMan. '; + const canonical = "My Name is SnowMan. "; + const compact: unknown = { student: "My Name is SnowMan." }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P15-ibm15v03.xml", () => { + // 2.5 — Tests comment with one dash inside + const input: string = + '\r\n \r\n]>\r\n\r\n\r\nMy Name is SnowMan. '; + const canonical = "My Name is SnowMan. "; + const compact: unknown = { student: "My Name is SnowMan." }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P15-ibm15v04.xml", () => { + // 2.5 — Tests comment with more comprehensive content + const input: string = + '\r\n \r\n]>\r\n\r\n\r\nMy Name is SnowMan. \r\n'; + const canonical = "My Name is SnowMan. "; + const compact: unknown = { student: "My Name is SnowMan." }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P16-ibm16v01.xml", () => { + // 2.6 — Tests PI definition with only PItarget name and nothing else + const input: string = + '\r\n \r\n]>\r\n\r\n\r\nMy Name is SnowMan. \r\n\r\n\r\n'; + const canonical = "My Name is SnowMan. "; + const compact: unknown = { student: "My Name is SnowMan." }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P16-ibm16v02.xml", () => { + // 2.6 — Tests PI definition with only PItarget name and a white space + const input: string = + '\r\n \r\n]>\r\n\r\n\r\nMy Name is SnowMan. '; + const canonical = "My Name is SnowMan. "; + const compact: unknown = { student: "My Name is SnowMan." }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P16-ibm16v03.xml", () => { + // 2.6 — Tests PI definition with PItarget name and text that contains question mark and right angle + const input: string = + '\r\n \r\n]>\r\n\r\n IN PI ?>\r\nMy Name is SnowMan. '; + const canonical = "My Name is SnowMan. "; + const compact: unknown = { student: "My Name is SnowMan." }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P17-ibm17v01.xml", () => { + // 2.6 — Tests PITarget name + const input: string = + '\r\n \r\n]>\r\n\r\n\r\nMy Name is SnowMan. \r\n\r\n\r\n'; + const canonical = "My Name is SnowMan. "; + const compact: unknown = { student: "My Name is SnowMan." }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P18-ibm18v01.xml", () => { + // 2.7 — Tests CDSect with CDStart CData CDEnd + const input: string = + '\r\n \r\n]>\r\n\r\n\r\n\r\nMy Name is SnowMan. text]]> \r\n\r\n'; + const canonical = "My Name is SnowMan. This is <normal> text "; + const compact: unknown = { student: "My Name is SnowMan. This is text" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P19-ibm19v01.xml", () => { + // 2.7 — Tests CDStart + const input: string = + '\r\n \r\n]>\r\n\r\n\r\nMy Name is SnowMan. \r\n\r\n\r\n'; + const canonical = "My Name is SnowMan. This is a test "; + const compact: unknown = { student: "My Name is SnowMan. This is a test" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P20-ibm20v01.xml", () => { + // 2.7 — Tests CDATA with empty string + const input: string = + '\r\n \r\n]>\r\n\r\n\r\n\r\nMy Name is SnowMan. \r\n\r\n\r\n'; + const canonical = "My Name is SnowMan. "; + const compact: unknown = { student: "My Name is SnowMan." }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P20-ibm20v02.xml", () => { + // 2.7 — Tests CDATA with regular content + const input: string = + '\r\n \r\n]>\r\n\r\n\r\n\r\nMy Name is SnowMan. This is a test]]>'; + const canonical = "My Name is SnowMan. <testing>This is a test</testing>"; + const compact: unknown = { student: "My Name is SnowMan. This is a test" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P21-ibm21v01.xml", () => { + // 2.7 — Tests CDEnd + const input: string = + '\r\n \r\n]>\r\n\r\n \r\n\r\nMy Name is SnowMan. \r\n\r\n\r\n'; + const canonical = "My Name is SnowMan. This is a test "; + const compact: unknown = { student: "My Name is SnowMan. This is a test" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P22-ibm22v01.xml", () => { + // 2.8 — Tests prolog with XMLDecl and doctypedecl + const input = Buffer.from( + '\r\n\r\n]>\r\n\r\n', + ); + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P22-ibm22v02.xml", () => { + // 2.8 — Tests prolog with doctypedecl + const input: string = "\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P22-ibm22v03.xml", () => { + // 2.8 — Tests prolog with Misc doctypedecl + const input: string = "\r\n]>\r\n\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P22-ibm22v04.xml", () => { + // 2.8 — Tests prolog with doctypedecl Misc + const input: string = "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P22-ibm22v05.xml", () => { + // 2.8 — Tests prolog with XMLDecl Misc doctypedecl + const input = Buffer.from( + '\r\n\r\n\r\n]>\r\n\r\n', + ); + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P22-ibm22v06.xml", () => { + // 2.8 — Tests prolog with XMLDecl doctypedecl Misc + const input = Buffer.from( + '\r\n\r\n]>\r\n\r\n\r\n', + ); + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P22-ibm22v07.xml", () => { + // 2.8 — Tests prolog with XMLDecl Misc doctypedecl Misc + const input = Buffer.from( + '\r\n\r\n\r\n]>\r\n\r\n\r\n', + ); + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P23-ibm23v01.xml", () => { + // 2.8 — Tests XMLDecl with VersionInfo only + const input: string = "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P23-ibm23v02.xml", () => { + // 2.8 — Tests XMLDecl with VersionInfo EncodingDecl + const input = Buffer.from( + "\r\n\r\n]>\r\n\r\n", + ); + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P23-ibm23v03.xml", () => { + // 2.8 — Tests XMLDecl with VersionInfo SDDecl + const input: string = + "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P23-ibm23v04.xml", () => { + // 2.8 — Tests XMLDecl with VerstionInfo and a trailing whitespace char + const input: string = "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P23-ibm23v05.xml", () => { + // 2.8 — Tests XMLDecl with VersionInfo EncodingDecl SDDecl + const input = Buffer.from( + "\r\n\r\n]>\r\n\r\n", + ); + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P23-ibm23v06.xml", () => { + // 2.8 — Tests XMLDecl with VersionInfo EncodingDecl SDDecl and a trailing whitespace + const input = Buffer.from( + "\r\n\r\n]>\r\n\r\n", + ); + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P24-ibm24v01.xml", () => { + // 2.8 — Tests VersionInfo with single quote + const input: string = "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P24-ibm24v02.xml", () => { + // 2.8 — Tests VersionInfo with double quote + const input: string = '\r\n\r\n]>\r\n\r\n'; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P25-ibm25v01.xml", () => { + // 2.8 — Tests EQ with = + const input: string = "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P25-ibm25v02.xml", () => { + // 2.8 — Tests EQ with = and spaces on both sides + const input: string = "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P25-ibm25v03.xml", () => { + // 2.8 — Tests EQ with = and space in front of it + const input: string = "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P25-ibm25v04.xml", () => { + // 2.8 — Tests EQ with = and space after it + const input: string = "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P26-ibm26v01.xml", () => { + // 2.8 — Tests VersionNum 1.0 + const input: string = "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P27-ibm27v01.xml", () => { + // 2.8 — Tests Misc with comment + const input: string = + "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P27-ibm27v02.xml", () => { + // 2.8 — Tests Misc with PI + const input: string = + "\r\n\r\n]>\r\n\r\n"; + const canonical = ""; + const compact: unknown = { doc: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P27-ibm27v03.xml", () => { + // 2.8 — Tests Misc with white spaces + const input: string = + "\r\n\r\n]>\r\nS is in the following Misc\r\n"; + const canonical = "S is in the following Misc"; + const compact: unknown = { doc: "S is in the following Misc" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P28-ibm28v01.xml", () => { + // 2.8 — Tests doctypedecl with internal DTD only + const input = Buffer.from( + "\r\n\r\n]>\r\n \r\n\r\n", + ); + const canonical = ""; + const compact: unknown = { animal: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P28-ibm28v02.xml", () => { + // 2.8 — Tests doctypedecl with external subset and combinations of different markup declarations and + // PEReferences (upstream: valid; external parameter entities are not read) + const input = Buffer.from( + '\r\n\r\n \r\n \r\n ">\r\n ">\r\n ">\r\n %make_leopard_element; \r\n \r\n %make_small;\r\n ">\r\n %make_big;\r\n %make_attlist;\r\n \r\n \r\n]>\r\n\r\n &forcat;\r\n This is a white tiger in Mirage!!\r\n \r\n \r\n \r\n \r\n \r\n\r\n', + ); + const canonical = + ' This is a small cat This is a white tiger in Mirage!! '; + const compact: unknown = { + animal: { + cat: ["This is a small cat", ""], + tiger: { "@color": "white", "#text": "This is a white tiger in Mirage!!" }, + leopard: { small: "", big: "" }, + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P29-ibm29v01.xml", () => { + // 2.8 — Tests markupdecl with combinations of elementdecl, AttlistDecl,EntityDecl, NotationDecl, PI + // and comment + const input = Buffer.from( + '\r\n\r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n]>\r\n\r\n &forcat;\r\n This is a white tiger in Mirage!!\r\n \r\n \r\n \r\n \r\n \r\n\r\n', + ); + const canonical = + ' This is a small cat This is a white tiger in Mirage!! '; + const compact: unknown = { + animal: { + cat: ["This is a small cat", ""], + tiger: { "@color": "white", "#text": "This is a white tiger in Mirage!!" }, + leopard: { small: "", big: "" }, + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P29-ibm29v02.xml", () => { + // 2.8 — Tests WFC: PE in internal subset as a positive test (upstream: valid; external parameter + // entities are not read) + const input = Buffer.from( + '\r\n\r\n \r\n \r\n \r\n \r\n ">\r\n %make_leopard_element; \r\n \r\n \r\n \r\n \r\n \r\n \r\n]>\r\n\r\n &forcat;\r\n This is a white tiger in Mirage!!\r\n \r\n \r\n \r\n \r\n \r\n\r\n', + ); + const canonical = + ' This is a small cat This is a white tiger in Mirage!! '; + const compact: unknown = { + animal: { + cat: ["This is a small cat", ""], + tiger: { "@color": "white", "#text": "This is a white tiger in Mirage!!" }, + leopard: { small: "", big: "" }, + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P30-ibm30v01.xml", () => { + // 2.8 — Tests extSubset with extSubsetDecl only in the dtd file (upstream: valid; external parameter + // entities are not read) + const input: string = + '\r\n\r\n\r\n'; + const canonical = ""; + const compact: unknown = { animal: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P30-ibm30v02.xml", () => { + // 2.8 — Tests extSubset with TextDecl and extSubsetDecl in the dtd file (upstream: valid; external + // parameter entities are not read) + const input: string = + '\r\n\r\n\r\n'; + const canonical = ""; + const compact: unknown = { animal: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P31-ibm31v01.xml", () => { + // 2.8 — Tests extSubsetDecl with combinations of markupdecls, conditionalSects, PEReferences and white + // spaces (upstream: valid; external parameter entities are not read) + const input: string = + '\r\n\r\n \r\n\r\n\r\n'; + const canonical = " "; + const compact: unknown = { animal: { tiger: "" } }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P32-ibm32v01.xml", () => { + // 2.9 — Tests VC: Standalone Document Declaration with absent attribute that has default value and + // standalone is no (upstream: valid; external parameter entities are not read; output depends on them) + const input: string = + '\r\n\r\n\r\n\r\n'; + expectParses(input); + }); + + test("ibm-valid-P32-ibm32v02.xml", () => { + // 2.9 — Tests VC: Standalone Document Declaration with external entity reference and standalone is no + // (upstream: valid; external parameter entities are not read; output depends on them) + const input: string = + '\r\n\r\n&animal_content;\r\n\r\n'; + expectParses(input); + }); + + test("ibm-valid-P32-ibm32v03.xml", () => { + // 2.9 — Tests VC: Standalone Document Declaration with attribute values that need to be normalized and + // standalone is no (upstream: valid; external parameter entities are not read; output depends on them) + const input: string = + '\r\n\r\n\r\n\r\n'; + expectParses(input); + }); + + test("ibm-valid-P32-ibm32v04.xml", () => { + // 2.9 — Tests VC: Standalone Document Declaration with whitespace in mixed content and standalone is + // no (upstream: valid; external parameter entities are not read; output depends on them) + const input: string = + '\r\n\r\nThis is a \r\n \r\n\r\nyellow tiger\r\n\r\n'; + expectParses(input); + }); + + test("ibm-valid-P33-ibm33v01.xml", () => { + // 2.12 — Tests LanguageID with Langcode - Subcode + const input: string = + '\r\n \r\n]>\r\nIt is written in English\r\n'; + const canonical = 'It is written in English'; + const compact: unknown = { book: { "@xml:lang": "en-US", "#text": "It is written in English" } }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P34-ibm34v01.xml", () => { + // 2.12 — Duplicate Test as ibm33v01.xml + const input: string = + '\r\n \r\n]>\r\nIt is written in English\r\n'; + const canonical = 'It is written in English'; + const compact: unknown = { book: { "@xml:lang": "en-US", "#text": "It is written in English" } }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P35-ibm35v01.xml", () => { + // 2.12 — Tests ISO639Code + const input: string = + '\r\n \r\n]>\r\nIt is written in English\r\n'; + const canonical = 'It is written in English'; + const compact: unknown = { book: { "@xml:lang": "en", "#text": "It is written in English" } }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P36-ibm36v01.xml", () => { + // 2.12 — Tests IanaCode + const input: string = + '\r\n \r\n]>\r\nIt is written in English\r\n'; + const canonical = 'It is written in English'; + const compact: unknown = { book: { "@xml:lang": "i-BS-ABCD", "#text": "It is written in English" } }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P37-ibm37v01.xml", () => { + // 2.12 — Tests UserCode + const input: string = + '\r\n \r\n]>\r\nIt is written in English\r\n'; + const canonical = 'It is written in English'; + const compact: unknown = { book: { "@xml:lang": "x-uk-eng", "#text": "It is written in English" } }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P38-ibm38v01.xml", () => { + // 2.12 — Tests SubCode + const input: string = + '\r\n \r\n]>\r\nIt is written in English\r\n'; + const canonical = 'It is written in English'; + const compact: unknown = { book: { "@xml:lang": "en-USa", "#text": "It is written in English" } }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P39-ibm39v01.xml", () => { + // 3 — Tests element with EmptyElemTag and STag content Etag, also tests the VC: Element Valid with + // elements that have children, Mixed and ANY contents + const input: string = + '\r\n\r\n \r\n \r\n \r\n \r\n \r\n \r\n]>\r\n\r\n \r\n content of b element\r\n \r\n no more children\r\n \r\n\r\n\r\n'; + const canonical = + " content of b element no more children "; + const compact: unknown = { + root: { + a: "", + b: { + c: ["", { d: { e: ["no more children", { f: "" }], f: "" } }], + "#text": "content of b element", + }, + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P40-ibm40v01.xml", () => { + // 3.1 — Tests STag with possible combinations of its fields, also tests WFC: Unique Att Spec. + const input: string = + '\r\n\r\n \r\n \r\n \r\n \r\n]>\r\n\r\n without white space\r\n with a white space\r\n one attribute\r\n one attribute\r\n\r\n\r\n'; + const canonical = + ' without white space with a white space one attribute one attribute '; + const compact: unknown = { + root: { + b: [ + "without white space", + "with a white space", + { "@attr1": "value1", "#text": "one attribute" }, + { + "@attr1": "value1", + "@attr2": "value2", + "@attr3": "value3", + "#text": "one attribute", + }, + ], + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P41-ibm41v01.xml", () => { + // 3.1 — Tests Attribute with Name Eq AttValue and VC: Attribute Value Type + const input: string = + '\r\n\r\n \r\n \r\n \r\n \r\n]>\r\n\r\n Name eq AttValue\r\n\r\n\r\n'; + const canonical = ' Name eq AttValue '; + const compact: unknown = { + root: { + b: { + "@attr1": "value1", + "@attr2": "def", + "@attr3": "fixed", + "#text": "Name eq AttValue", + }, + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P42-ibm42v01.xml", () => { + // 3.1 — Tests ETag with possible combinations of its fields + const input: string = + '\r\n\r\n \r\n \r\n \r\n]>\r\n\r\n : End tag with a space inside\r\n content of b element\r\n\r\n\r\n\r\n'; + const canonical = + " : End tag with a space inside content of b element "; + const compact: unknown = { + root: { + a: "", + b: { c: "", "#text": ": End tag with a space inside\n content of b element" }, + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P43-ibm43v01.xml", () => { + // 3.1 — Tests content with all possible constructs: element, CharData, Reference, CDSect, Comment + const input: string = + '\r\n\r\n \r\n \r\n \r\n General entity reference in element content">\r\n]>\r\n\r\n\r\n\r\n \r\n CharData: content of b element\r\n %paaa; : PE reference should not be recognized in element content \r\n \r\n\r\n\r\n &inContent;\r\n Charater reference: A\r\n CDSect in content: markupsHEADnothing ]]>\r\n \r\n\r\n\r\n\r\n'; + const canonical = + " CharData: content of b element %paaa; : PE reference should not be recognized in element content General entity reference in element content Charater reference: A CDSect in content: <html>markups<head>HEAD</head><body>nothing</body></html> "; + const compact: unknown = { + root: { + a: "", + b: { + c: [ + "", + { + b: "General entity reference in element content", + "#text": + "Charater reference: A\n CDSect in content: markupsHEADnothing", + }, + ], + "#text": + "CharData: content of b element\n %paaa; : PE reference should not be recognized in element content", + }, + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P44-ibm44v01.xml", () => { + // 3.1 — Tests EmptyElemTag with possible combinations of its fields + const input: string = + '\r\n\r\n \r\n \r\n \r\n \r\n]>\r\n\r\n without white space\r\n with a white space\r\n \r\n \r\n\r\n\r\n\r\n\r\n\r\n'; + const canonical = + ' without white space with a white space '; + const compact: unknown = { + root: { + b: ["", "", { "@attr1": "value1" }, { "@attr1": "value1", "@attr2": "value2", "@attr3": "value3" }], + "#text": "without white space\n with a white space", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P45-ibm45v01.xml", () => { + // 3.2 — Tests both P45 elementDecl and P46 contentspec with possible combinations of their constructs + const input: string = + '\r\n\r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n]>\r\n\r\n without white space\r\n with a white space\r\n \r\n \r\n\r\n\r\n\r\n'; + const canonical = + ' without white space with a white space '; + const compact: unknown = { + root: { + b: ["", "", { "@attr1": "value1" }, { "@attr1": "value1", "@attr2": "value2", "@attr3": "value3" }], + "#text": "without white space\n with a white space", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P47-ibm47v01.xml", () => { + // 3.2.1 — Tests all possible children,cp,choice,seq patterns in P47,P48,P49,P50 + const input: string = + '\r\n\r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n]>\r\n\r\n \r\n content of b element\r\n\r\n\r\n\r\n'; + const canonical = " content of b element "; + const compact: unknown = { root: { a: "", b: { c: "", "#text": "content of b element" } } }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P49-ibm49v01.xml", () => { + // 3.2.1 — Tests VC:Proper Group/PE Nesting with PEs of choices that are properly nested with + // parenthesized groups in external subsets (upstream: valid; external parameter entities are not read) + const input: string = + '\r\n\r\n]>\r\n\r\n \r\n content of b element\r\n\r\n\r\n\r\n\r\n\r\n'; + const canonical = " content of b element "; + const compact: unknown = { root: { a: "", b: { c: "", "#text": "content of b element" } } }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P50-ibm50v01.xml", () => { + // 3.2.1 — Tests VC:Proper Group/PE Nesting with PEs of seq that are properly nested with parenthesized + // groups in external subsets (upstream: valid; external parameter entities are not read) + const input: string = + '\r\n\r\n]>\r\n\r\n \r\n \r\n content of b element\r\n\r\n\r\n'; + const canonical = + " content of b element "; + const compact: unknown = { + root: { + a: "", + b: { + c: [{ child1: { a: "", b: "", c: "" } }, { child2: { a: "", b: "", c: "" } }], + "#text": "content of b element", + }, + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P51-ibm51v01.xml", () => { + // 3.2.2 — Tests Mixed with possible combinations of its fields amd VC: No Duplicate Types + const input: string = + '\r\n\r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n]>\r\n\r\n Element type a \r\n Element type b \r\n Element type c \r\n Element type d \r\n Element type e \r\n\r\n'; + const canonical = + " Element type a Element type b Element type c Element type d Element type e "; + const compact: unknown = { + root: { + a: "Element type a", + b: "Element type b", + c: "Element type c", + d: { c: "", "#text": "Element type d" }, + e: { a: "", b: "", c: "", "#text": "Element type e" }, + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P51-ibm51v02.xml", () => { + // 3.2.2 — Tests VC:Proper Group/PE Nesting with PEs of Mixed that are properly nested with + // parenthesized groups in external subsets (upstream: valid; external parameter entities are not read) + const input: string = + '\r\n\r\n]>\r\n\r\n Element type a \r\n Element type b \r\n Element type c \r\n Element type d \r\n Element type e \r\n\r\n'; + const canonical = + " Element type a Element type b Element type c Element type d Element type e "; + const compact: unknown = { + root: { + a: "Element type a", + b: "Element type b", + c: "Element type c", + d: { c: "", "#text": "Element type d" }, + e: { a: "", b: "", c: "", "#text": "Element type e" }, + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P52-ibm52v01.xml", () => { + // 3.3 — Tests all AttlistDecl and AttDef Patterns in P52 and P53 + const input: string = + '\r\n\r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n]>\r\n\r\n Element type a \r\n test P52 and P53 \r\n\r\n\r\n'; + const canonical = + ' Element type a test P52 and P53 '; + const compact: unknown = { + root: { + a: "Element type a", + b: { + "@battr1": "anyvalue", + "@battr3": "fixedvalue", + "@battr4": "def", + "#text": "test P52 and P53", + }, + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P54-ibm54v01.xml", () => { + // 3.3.1 — Tests all AttTypes : StringType, TokenizedTypes, EnumeratedTypes in P55,P56,P57,P58,P59. + // Also tests all DefaultDecls in P60. + const input: string = + '\r\n\r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n\r\n]>\r\n\r\n Element type a \r\n Element type b \r\n Element type c \r\n Element type d \r\n Element type e \r\n Element type f \r\n Element type g \r\n Element type h \r\n Element type i \r\n Element type j \r\n\r\n\r\n'; + expectParses(input); + }); + + test("ibm-valid-P54-ibm54v02.xml", () => { + // 3.3.1 — Tests all AttTypes : StringType, TokenizedType, EnumeratedTypes in P55,P56,P57. + const input: string = + "\r\n\r\n\r\n \r\n \r\n \r\n \r\n \r\n \r\n ]>\r\n\r\n\r\n\r\n\r\n"; + const canonical = ' '; + const compact: unknown = { + root: { x: { "@attr": "Madhu" }, y: { "@attr": "1.a.name.token.but.not.a.name" } }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P54-ibm54v03.xml", () => { + // 3.3.1 — Tests AttTypes with StringType in P55. + const input: string = + "\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n\r\n\r\n\r\n"; + const canonical = ' '; + const compact: unknown = { AttrType: { a: { "@att": "hello world" } } }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P55-ibm55v01.xml", () => { + // 3.3.1 — Tests StringType for P55. The "CDATA" occurs in the StringType for the attribute "att" for + // the element "a". + const input: string = + "\r\n\r\n\r\n\r\n \r\n]>\r\n\r\n\r\nTesting with a valid stringType attribute \r\n\r\n"; + const canonical = ' Testing with a valid stringType attribute '; + const compact: unknown = { + StType: { + a: { "@att": "Hello" }, + "#text": "Testing with a valid stringType attribute", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P56-ibm56v01.xml", () => { + // 3.3.1 — Tests TokenizedType for P56. The "ID", "IDREF", "IDREFS", "ENTITY", "ENTITIES", "NMTOKEN", + // and "NMTOKENS" occur in the TokenizedType for the attribute "attr". + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n ]>\r\n\r\n'; + const canonical = ""; + const compact: unknown = { root: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P56-ibm56v02.xml", () => { + // 3.3.1 — Tests TokenizedType for P56 VC: ID Attribute Default. The value "AC1999" is assigned to the + // ID attribute "attr" with "#REQUIRED" in the DeaultDecl. + const input: string = + '\r\n\r\n\r\n \r\n ]>\r\n\r\nThis is a positive test for validity constraints\r\nGiving a unique name to the attribute ID an ID Attribute default as #required\r\n\r\n'; + const canonical = + ' This is a positive test for validity constraints Giving a unique name to the attribute ID an ID Attribute default as #required '; + const compact: unknown = { + tokenizer: { + "@UniqueName": "AC1999", + "#text": + "This is a positive test for validity constraints\nGiving a unique name to the attribute ID an ID Attribute default as #required", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P56-ibm56v03.xml", () => { + // 3.3.1 — Tests TokenizedType for P56 VC: ID Attribute Default. The value "AC1999" is assigned to the + // ID attribute "attr" with "#IMPLIED" in the DeaultDecl. + const input: string = + '\r\n\r\n\r\n \r\n ]>\r\n\r\nThis is a positive test for validity constraints\r\nGiving ID attribute default as #IMPLIED\r\n'; + const canonical = + ' This is a positive test for validity constraints Giving ID attribute default as #IMPLIED '; + const compact: unknown = { + tokenizer: { + "@UniqueName": "AC1999", + "#text": "This is a positive test for validity constraints\nGiving ID attribute default as #IMPLIED", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P56-ibm56v04.xml", () => { + // 3.3.1 — Tests TokenizedType for P56 VC: ID. The ID attribute "UniqueName" appears only once in the + // document. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n ]>\r\n\r\n\r\nThis is a positive test for validity constraints\r\nthe value of the attribute with a type ID does not appear more than once in the XML document\r\n\r\n'; + const canonical = + ' This is a positive test for validity constraints the value of the attribute with a type ID does not appear more than once in the XML document '; + const compact: unknown = { + tokenizer: { + "@UniqueName": "Ac999", + b: { "@attr": "BC999" }, + "#text": + "This is a positive test for validity constraints\nthe value of the attribute with a type ID does not appear more than once in the XML document", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P56-ibm56v05.xml", () => { + // 3.3.1 — Tests TokenizedType for P56 VC: One ID per element type. The element "a" or "b" has only one + // ID attribute. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n ]>\r\n\r\n\r\n\r\nThis is a positive validity test for ID.\r\nany element type has no more than one attribute of type ID specified\r\n'; + const canonical = + ' This is a positive validity test for ID. any element type has no more than one attribute of type ID specified '; + const compact: unknown = { + tokenizer: { + a: { "@first": "AC1999" }, + b: { "@second": "CD345" }, + "#text": + "This is a positive validity test for ID.\nany element type has no more than one attribute of type ID specified", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P56-ibm56v06.xml", () => { + // 3.3.1 — Tests TokenizedType for P56 VC: IDREF. The IDREF value "AC456" matches the value assigned to + // an ID attribute "UniqueName". + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n ]>\r\n\r\n\r\n\r\nPositive test for validity constraint of IDREF.\r\nIn an attribute decl, values of type IDREF match tha name production\r\nand the IDREF value matches the value assigned to an ID attribute somewhere\r\nin the XML document.\r\n'; + const canonical = + ' Positive test for validity constraint of IDREF. In an attribute decl, values of type IDREF match tha name production and the IDREF value matches the value assigned to an ID attribute somewhere in the XML document. '; + const compact: unknown = { + test: { + id: { "@UniqueName": "AC456" }, + idref: { "@reference": "AC456" }, + "#text": + "Positive test for validity constraint of IDREF.\nIn an attribute decl, values of type IDREF match tha name production\nand the IDREF value matches the value assigned to an ID attribute somewhere\nin the XML document.", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P56-ibm56v07.xml", () => { + // 3.3.1 — Tests TokenizedType for P56 VC: IDREF. The IDREFS value "AC456 Q123" matches the values + // assigned to the ID attribute "UniqueName" and "Uname". + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n \r\n \r\n ]>\r\n\r\n\r\n\r\n\r\nPositive test for validity constraint of IDREFS.\r\nIn an attribute decl, values of type IDREFS match tha name production\r\nand the IDREFS value matches the values assigned to an ID attributes somewhere\r\nin the XML document.\r\n'; + const canonical = + ' Positive test for validity constraint of IDREFS. In an attribute decl, values of type IDREFS match tha name production and the IDREFS value matches the values assigned to an ID attributes somewhere in the XML document. '; + const compact: unknown = { + test: { + id1: { "@UniqueName": "AC456" }, + id2: { "@UName": "Q123" }, + idref: { "@reference": "AC456 Q123" }, + "#text": + "Positive test for validity constraint of IDREFS.\nIn an attribute decl, values of type IDREFS match tha name production\nand the IDREFS value matches the values assigned to an ID attributes somewhere\nin the XML document.", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P56-ibm56v08.xml", () => { + // 3.3.1 — Tests TokenizedType for P56 VC: Entity Name. The value "image" of the ENTITY attribute "sun" + // matches the name of an unparsed entity declared. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n]>\r\n\r\n\r\nvalues of type ENTITY match the Name production and the ENTITY value\r\nmatches the name of an unparsed entity declared in the DTD.\r\n\r\n'; + const canonical = + ' values of type ENTITY match the Name production and the ENTITY value matches the name of an unparsed entity declared in the DTD. '; + const compact: unknown = { + test: { + landscape: { "@sun": "image" }, + "#text": + "values of type ENTITY match the Name production and the ENTITY value\nmatches the name of an unparsed entity declared in the DTD.", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P56-ibm56v09.xml", () => { + // 3.3.1 — Tests TokenizedType for P56 VC: Name Token. The value of the NMTOKEN attribute "thistoken" + // matches the Nmtoken production. + const input: string = + '\r\n\r\n\r\n \r\n \r\n]>\r\n\r\n\r\nIn an attribute declaration, values of type NMTOKEN match the Nmtoken production\r\n'; + const canonical = + ' In an attribute declaration, values of type NMTOKEN match the Nmtoken production '; + const compact: unknown = { + test: { + nametoken: { "@thistoken": "x:image" }, + "#text": "In an attribute declaration, values of type NMTOKEN match the Nmtoken production", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P56-ibm56v10.xml", () => { + // 3.3.1 — Tests TokenizedType for P56 VC: Name Token. The value of the NMTOKENS attribute "thistoken" + // matches the Nmtoken production. + const input: string = + '\r\n\r\n\r\n \r\n \r\n]>\r\n\r\n\r\nIn an attribute declaration, values of type NMTOKENS match the Nmtokens production\r\n'; + const canonical = + ' In an attribute declaration, values of type NMTOKENS match the Nmtokens production '; + const compact: unknown = { + test: { + nametokens: { "@thistoken": "x:lang y:country" }, + "#text": "In an attribute declaration, values of type NMTOKENS match the Nmtokens production", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P57-ibm57v01.xml", () => { + // 3.3.1 — Tests EnumeratedType in the AttType. The attribute "att" has a type (a|b) with the element + // "a". the + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n \r\n \r\n ]>\r\n \r\nThis test case tests the kinds of enumerated types\r\n\r\n\r\n'; + const canonical = " This test case tests the kinds of enumerated types "; + const compact: unknown = { + root: { a: "", b: "", "#text": "This test case tests the kinds of enumerated types" }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P58-ibm58v01.xml", () => { + // 3.3.1 — Tests NotationType for P58. It shows different patterns fro the NOTATION attribute "attr". + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n ]>\r\n \r\nThis is a positive test with different patterns for NOTATION\r\n\r\n'; + const canonical = " This is a positive test with different patterns for NOTATION "; + const compact: unknown = { test: "This is a positive test with different patterns for NOTATION" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P58-ibm58v02.xml", () => { + // 3.3.1 — Tests NotationType for P58: Notation Attributes. The value "base64" of the NOTATION + // attribute "attr" matches one of the notation names declared. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n \r\n ]>\r\n \r\n\r\nThe attribute values of type NOTATION matches one of the notation names included in the declaration;\r\nall notation names in the declaration have been declared\r\n'; + const canonical = + ' The attribute values of type NOTATION matches one of the notation names included in the declaration; all notation names in the declaration have been declared '; + const compact: unknown = { + test: { + blob: { "@content-encoding": "base64" }, + "#text": + "The attribute values of type NOTATION matches one of the notation names included in the declaration;\nall notation names in the declaration have been declared", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P59-ibm59v01.xml", () => { + // 3.3.1 — Tests Enumeration in the EnumeratedType for P59. It shows different patterns for the + // Enumeration attribute "attr". + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n ]>\r\n \r\nThis is a Positive test\r\n\r\n'; + const canonical = " This is a Positive test "; + const compact: unknown = { test: "This is a Positive test" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P59-ibm59v02.xml", () => { + // 3.3.1 — Tests Enumeration for P59 VC: Enumeration. The value "one" of the Enumeration attribute + // "attr" matches one of the element names declared. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n ]>\r\n \r\n\r\nThis is a Positive test\r\nThe attribute values of type Enumeration match one of the Nmtoken tokens in the declaration.\r\n'; + const canonical = + ' This is a Positive test The attribute values of type Enumeration match one of the Nmtoken tokens in the declaration. '; + const compact: unknown = { + test: { + num: { "@value": "one" }, + "#text": + "This is a Positive test\nThe attribute values of type Enumeration match one of the Nmtoken tokens in the declaration.", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P60-ibm60v01.xml", () => { + // 3.3.2 — Tests DefaultDecl for P60. It shows different options "#REQUIRED", "#FIXED", "#IMPLIED", and + // default for the attribute "chapter". + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n \r\n ]>\r\n\r\n \r\n Positive test\r\n DefaultDecl attributes values IMPLIED, REQUIRED, FIXED and default\r\n'; + const canonical = + ' Positive test DefaultDecl attributes values IMPLIED, REQUIRED, FIXED and default '; + const compact: unknown = { + Java: { + one: { "@chapter": "Introduction" }, + three: { "@chapter": "JavaBeans" }, + "#text": "Positive test\n DefaultDecl attributes values IMPLIED, REQUIRED, FIXED and default", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P60-ibm60v02.xml", () => { + // 3.3.2 — Tests DefaultDecl for P60 VC: Required Attribute. In the element "one" and "two" the value + // of the #REQUIRED attribute "chapter" is given. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n ]>\r\n\r\n\r\n\r\nPositive test. Required attribute. Every occurrence of an element with a \r\n#REQUIRED attribute default declaration gives the value of that attribute\r\n'; + const canonical = + ' Positive test. Required attribute. Every occurrence of an element with a #REQUIRED attribute default declaration gives the value of that attribute '; + const compact: unknown = { + Java: { + one: { "@chapter": "Introduction" }, + two: { "@chapter": "JavaApplets" }, + "#text": + "Positive test. Required attribute. Every occurrence of an element with a \n#REQUIRED attribute default declaration gives the value of that attribute", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P60-ibm60v03.xml", () => { + // 3.3.2 — Tests DefaultDecl for P60 VC: Fixed Attribute Default. The value of the #FIXED attribute + // "chapter" is exactly the same as the default value. + const input: string = + '\r\n\r\n\r\n \r\n \r\n ]>\r\n\r\n\r\nAn attribute has a default value declared with the #FIXED keyword, \r\nand an instances of that attribute is given a value which is exactly \r\nthe same as the default value in the declaration. \r\n\r\n'; + const canonical = + ' An attribute has a default value declared with the #FIXED keyword, and an instances of that attribute is given a value which is exactly the same as the default value in the declaration. '; + const compact: unknown = { + Java: { + one: { "@chapter": "Introduction" }, + "#text": + "An attribute has a default value declared with the #FIXED keyword, \nand an instances of that attribute is given a value which is exactly \nthe same as the default value in the declaration.", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P60-ibm60v04.xml", () => { + // 3.3.2 — Tests DefaultDecl for P60 VC: Attribute Default Legal. The default value specified for the + // attribute "attr" meets the lexical constraints of the declared attribute type. + const input: string = + '\r\n\r\n\r\n \r\n \r\n \r\n \r\n \r\n \r\n ]>\r\n\r\nThe default value specified for an attribute meets the \r\nlexical constraints of the declared attribute type.\r\n\r\n\r\n'; + const canonical = + " The default value specified for an attribute meets the lexical constraints of the declared attribute type. "; + const compact: unknown = { + test: "The default value specified for an attribute meets the \nlexical constraints of the declared attribute type.", + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P61-ibm61v01.xml", () => { + // 3.4 — Tests conditionalSect for P61. It takes the option "invludeSect" in the file ibm61v01.dtd. + // (upstream: valid; external parameter entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n \r\n\r\n '; + const canonical = " "; + const compact: unknown = { animal: { tiger: "" } }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P61-ibm61v02.xml", () => { + // 3.4 — Tests conditionalSect for P61. It takes the option "ignoreSect" in the file ibm61v02.dtd. + // (upstream: valid; external parameter entities are not read) + const input: string = + '\r\n\r\n\r\n]>\r\n\r\n\r\n \r\n'; + const canonical = ""; + const compact: unknown = { animal: "" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P62-ibm62v01.xml", () => { + // 3.4 — Tests includeSect for P62. The white space is not included before the key word "INCLUDE" in + // the beginning sequence. (upstream: valid; external parameter entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n \r\nPositive test. Test includeSect with pattern1 of p62.\r\nNormal Pattern\r\n'; + const canonical = + " Positive test. Test includeSect with pattern1 of p62. Normal Pattern "; + const compact: unknown = { + animal: { + tiger: "", + "#text": "Positive test. Test includeSect with pattern1 of p62.\nNormal Pattern", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P62-ibm62v02.xml", () => { + // 3.4 — Tests includeSect for P62. The white space is not included after the key word "INCLUDE" in the + // beginning sequence. (upstream: valid; external parameter entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n \r\nPositive test. Test includeSect with pattern2 of p62.\r\nspace included before INCLUDE\r\n\r\n'; + const canonical = + " Positive test. Test includeSect with pattern2 of p62. space included before INCLUDE "; + const compact: unknown = { + animal: { + tiger: "", + "#text": "Positive test. Test includeSect with pattern2 of p62.\nspace included before INCLUDE", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P62-ibm62v03.xml", () => { + // 3.4 — Tests includeSect for P62. The white space is included after the key word "INCLUDE" in the + // beginning sequence. (upstream: valid; external parameter entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n \r\nPositive test. Test includeSect with pattern3 of p62.\r\nspace included after INCLUDE\r\n\r\n'; + const canonical = + " Positive test. Test includeSect with pattern3 of p62. space included after INCLUDE "; + const compact: unknown = { + animal: { + tiger: "", + "#text": "Positive test. Test includeSect with pattern3 of p62.\nspace included after INCLUDE", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P62-ibm62v04.xml", () => { + // 3.4 — Tests includeSect for P62. The white space is included before the key word "INCLUDE" in the + // beginning sequence. (upstream: valid; external parameter entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n \r\nPositive test. Test includeSect with pattern4 of p62.\r\nspace included before and after INCLUDE\r\n\r\n'; + const canonical = + " Positive test. Test includeSect with pattern4 of p62. space included before and after INCLUDE "; + const compact: unknown = { + animal: { + tiger: "", + "#text": "Positive test. Test includeSect with pattern4 of p62.\nspace included before and after INCLUDE", + }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P62-ibm62v05.xml", () => { + // 3.4 — Tests includeSect for P62. The extSubsetDecl is not included. (upstream: valid; external + // parameter entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n]>\r\n\r\n\r\n \r\nPositive test. Missing external subset declaration.\r\n'; + const canonical = + " Positive test. Missing external subset declaration. "; + const compact: unknown = { + animal: { tiger: "", "#text": "Positive test. Missing external subset declaration." }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P63-ibm63v01.xml", () => { + // 3.4 — Tests ignoreSect for P63. The white space is not included before the key word "IGNORE" in the + // beginning sequence. (upstream: valid; external parameter entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n \r\n]>\r\n\r\nPositive test. Test for IGNORE with pattern 1.\r\n'; + const canonical = ' Positive test. Test for IGNORE with pattern 1. '; + const compact: unknown = { + animal: { "@a": "tiger", "#text": "Positive test. Test for IGNORE with pattern 1." }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P63-ibm63v02.xml", () => { + // 3.4 — Tests ignoreSect for P63. The white space is not included after the key word "IGNORE" in the + // beginning sequence. (upstream: valid; external parameter entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n \r\n]>\r\n\r\nPositive test. Test for IGNORE with pattern 2.\r\n'; + const canonical = ' Positive test. Test for IGNORE with pattern 2. '; + const compact: unknown = { + animal: { "@a": "tiger", "#text": "Positive test. Test for IGNORE with pattern 2." }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P63-ibm63v03.xml", () => { + // 3.4 — Tests ignoreSect for P63. The white space is included after the key word "IGNORE" in the + // beginning sequence. (upstream: valid; external parameter entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n \r\n]>\r\n\r\nPositive test. Test for IGNORE with pattern 3.\r\n'; + const canonical = ' Positive test. Test for IGNORE with pattern 3. '; + const compact: unknown = { + animal: { "@a": "tiger", "#text": "Positive test. Test for IGNORE with pattern 3." }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P63-ibm63v04.xml", () => { + // 3.4 — Tests ignoreSect for P63. The ignireSectContents is included. (upstream: valid; external + // parameter entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n \r\n]>\r\n\r\nPositive test. Test for IGNORE with pattern 4.\r\n'; + const canonical = ' Positive test. Test for IGNORE with pattern 4. '; + const compact: unknown = { + animal: { "@a": "tiger", "#text": "Positive test. Test for IGNORE with pattern 4." }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P63-ibm63v05.xml", () => { + // 3.4 — Tests ignoreSect for P63. The white space is included before and after the key word "IGNORE" + // in the beginning sequence. (upstream: valid; external parameter entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n \r\n]>\r\n\r\nPositive test. Test for IGNORE with pattern 5.\r\n'; + const canonical = ' Positive test. Test for IGNORE with pattern 5. '; + const compact: unknown = { + animal: { "@a": "tiger", "#text": "Positive test. Test for IGNORE with pattern 5." }, + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P64-ibm64v01.xml", () => { + // 3.4 — Tests ignoreSectContents for P64. One "ignore" field is included. (upstream: valid; external + // parameter entities are not read) + const input: string = + '\r\n\r\n\r\n]>\r\n\r\nPositive Test. Pattern1\r\n'; + const canonical = " Positive Test. Pattern1 "; + const compact: unknown = { animal: "Positive Test. Pattern1" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P64-ibm64v02.xml", () => { + // 3.4 — Tests ignoreSectContents for P64. Two "ignore" and one "ignoreSectContents" fields are + // included. (upstream: valid; external parameter entities are not read) + const input: string = + '\r\n\r\n\r\n]>\r\n\r\nPositive Test. Pattern2\r\n'; + const canonical = " Positive Test. Pattern2 "; + const compact: unknown = { animal: "Positive Test. Pattern2" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P64-ibm64v03.xml", () => { + // 3.4 — Tests ignoreSectContents for P64. Four "ignore" and three "ignoreSectContents" fields are + // included. (upstream: valid; external parameter entities are not read) + const input: string = + '\r\n\r\n\r\n]>\r\n\r\nPositive Test. Pattern3\r\n'; + const canonical = " Positive Test. Pattern3 "; + const compact: unknown = { animal: "Positive Test. Pattern3" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P65-ibm65v01.xml", () => { + // 3.4 — Tests Ignore for P65. An empty string occurs in the Ignore filed. (upstream: valid; external + // parameter entities are not read) + const input: string = + '\r\n\r\n\r\n]>\r\n\r\nPositive Test. Pattern1. Empty string.\r\n'; + const canonical = " Positive Test. Pattern1. Empty string. "; + const compact: unknown = { animal: "Positive Test. Pattern1. Empty string." }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P65-ibm65v02.xml", () => { + // 3.4 — Tests Ignore for P65. An string not including the brackets occurs in each of the Ignore filed. + // (upstream: valid; external parameter entities are not read) + const input: string = + '\r\n\r\n\r\n]>\r\n\r\nPositive Test. Pattern2.\r\n'; + const canonical = " Positive Test. Pattern2. "; + const compact: unknown = { animal: "Positive Test. Pattern2." }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P66-ibm66v01.xml", () => { + // 4.1 — Tests all legal CharRef's. + const input: string = + '\r\n\r\n]>\r\n\r\nTest all valid Charater references for P66:\r\n \r\n \r\n \r\n« « Í Í ï ï\r\nC C _\r\n 힣 가\r\n豈 �\r\n𐀀 􏿿\r\n\r\n\r\n'; + const canonical = + " Test all valid Charater references for P66: « « Í Í ï ï C C _ 힣 가 豈 � 𐀀 􏿿 "; + const compact: unknown = { + root: "Test all valid Charater references for P66:\n\t\t\t\n\n\n\n\n\n\r\n« « Í Í ï ï\nC C _\n 힣 가\n豈 �\n𐀀 􏿿", + }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P67-ibm67v01.xml", () => { + // 4.1 — Tests Reference could be EntityRef or CharRef. + const input: string = + '\r\n\r\n \r\n \r\n]>\r\n\r\n&ge1; B\r\n\r\n\r\n'; + const canonical = ' xyz B '; + const compact: unknown = { root: { "@attr": "xyzA", "#text": "xyz B" } }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P68-ibm68v01.xml", () => { + // 4.1 — Tests P68 VC:Entity Declared with Entities in External Subset , standalone is no (upstream: + // valid; external parameter entities are not read) + const input: string = + '\r\n\r\n]>\r\n\r\n pcdata content\r\n \r\n\r\n\r\n'; + const canonical = ' pcdata content '; + const compact: unknown = { root: { a: { "@attr1": "xyz" }, "#text": "pcdata content" } }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P68-ibm68v02.xml", () => { + // 4.1 — Tests P68 VC:Entity Declared with Entities in External Parameter Entities , standalone is no + // (upstream: valid; external general and parameter entities are not read) + const input: string = + '\r\n\r\n \r\n %pe1;\r\n]>\r\n\r\n pcdata content\r\n\r\n\r\n'; + const canonical = " pcdata content "; + const compact: unknown = { root: "pcdata content" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P69-ibm69v01.xml", () => { + // 4.1 — Tests P68 VC:Entity Declared with Parameter Entities in External Subset , standalone is no + // (upstream: valid; external parameter entities are not read) + const input: string = + '\r\n\r\n ">\r\n %pe1;\r\n]>\r\n\r\n pcdata content\r\n \r\n\r\n\r\n'; + const canonical = ' pcdata content '; + const compact: unknown = { root: { a: { "@attr1": "xyz" }, "#text": "pcdata content" } }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P69-ibm69v02.xml", () => { + // 4.1 — Tests P68 VC:Entity Declared with Parameter Entities in External Parameter Entities, + // standalone is no (upstream: valid; external general and parameter entities are not read) + const input: string = + '\r\n\r\n \r\n %pe1;\r\n]>\r\n\r\n pcdata content\r\n\r\n\r\n'; + const canonical = " pcdata content "; + const compact: unknown = { root: "pcdata content" }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P70-ibm70v01.xml", () => { + // 4.2 — Tests all legal GEDecls and PEDecls constructs derived from P70-76 (upstream: valid; external + // parameter entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n\r\n\'>\r\n\r\n%pe1;\r\n\r\n\r\n\r\n%pe2;\r\n]>\r\n\r\n'; + const canonical = ''; + const compact: unknown = { root: { "@att2": "any" } }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P78-ibm78v01.xml", () => { + // 4.3.2 — Tests ExtParsedEnt, also TextDecl in P77 and EncodingDecl in P80 (upstream: valid; external + // general entities are not read; output depends on them) + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n&epe1;&epe2;&epe3;\r\n\r\n'; + expectParses(input); + }); + + test("ibm-valid-P79-ibm79v01.xml", () => { + // 4.3.2 — Tests extPE (upstream: valid; external parameter entities are not read) + const input: string = + '\r\n\r\n\r\n\r\n%epe;\r\n]>\r\nXML Handbook This is a book\r\n\r\n\r\n'; + expectParses(input); + }); + + test("ibm-valid-P82-ibm82v01.xml", () => { + // 4.7 — Tests NotationDecl in P82 and PublicID in P83 + const input: string = + '\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n]>\r\ntest PublicID in P82\r\n\r\n'; + const canonical = 'test PublicID in P82'; + const compact: unknown = { root: { "@entatt1": "unparsed1", "#text": "test PublicID in P82" } }; + expectParses(input, canonical, compact); + }); + + test("ibm-valid-P85-ibm85v01.xml", () => { + // B. — This test case covers 149 legal character ranges plus 51 single legal characters for BaseChar + // in P85 using a PI target Name + const input = Buffer.from( + '\r\n\r\n\r\n\r\n]>\r\n\r\n', + ); + expectParses(input); + }); + + test("ibm-valid-P86-ibm86v01.xml", () => { + // B. — This test case covers 2 legal character ranges plus 1 single legal characters for IdeoGraphic + // in P86 using a PI target Name + const input = Buffer.from( + '\r\n\r\n\r\n\r\n]>\r\n\r\n', + ); + expectParses(input); + }); + + test("ibm-valid-P87-ibm87v01.xml", () => { + // B. — This test case covers 65 legal character ranges plus 30 single legal characters for + // CombiningChar in P87 using a PI target Name + const input = Buffer.from( + '\r\n\r\n\r\n\r\n]>\r\n\r\n', + ); + expectParses(input); + }); + + test("ibm-valid-P88-ibm88v01.xml", () => { + // B. — This test case covers 15 legal character ranges for Digit in P88 using a PI target Name + const input = Buffer.from( + '\r\n\r\n\r\n\r\n]>\r\n\r\n', + ); + expectParses(input); + }); + + test("ibm-valid-P89-ibm89v01.xml", () => { + // B. — This test case covers 3 legal character ranges plus 8 single legal characters for Extender in + // P89 using a PI target Name + const input = Buffer.from( + '\r\n\r\n\r\n\r\n]>\r\n\r\n', + ); + expectParses(input); + }); +}); + +describe("eduni/errata-2e", () => { + test("rmt-e2e-2a", () => { + // E2 — Duplicate token in enumerated attribute declaration + const input: string = "\n\n]>\n\n\n"; + expectParses(input); + }); + + test("rmt-e2e-2b", () => { + // E2 — Duplicate token in NOTATION attribute declaration + const input: string = + '\n\n\n]>\n\n'; + expectParses(input); + }); + + test("rmt-e2e-9a", () => { + // E9 — An unused attribute default need only be syntactically correct + const input: string = + '\n\n\n\n]>\n\n'; + expectParses(input); + }); + + test("rmt-e2e-9b", () => { + // E9 — An attribute default must be syntactically correct even if unused + const input: string = + '\n\n\n\n]>\n\n'; + expectParses(input); + }); + + test("rmt-e2e-14", () => { + // E14 — Declarations mis-nested wrt parameter entities are just validity errors (but note that some + // parsers treat some such errors as fatal) (upstream: invalid; external parameter entities are not + // read) + const input: string = '\n\n'; + expectParses(input); + }); + + test("rmt-e2e-15a", () => { + // E15 — Empty content can't contain an entity reference + const input: string = '\n\n]>\n∅\n\n'; + expectParses(input); + }); + + test("rmt-e2e-15b", () => { + // E15 — Empty content can't contain a comment + const input: string = "\n]>\n\n"; + expectParses(input); + }); + + test("rmt-e2e-15c", () => { + // E15 — Empty content can't contain a PI + const input: string = "\n]>\n\n"; + expectParses(input); + }); + + test("rmt-e2e-15d", () => { + // E15 — Empty content can't contain whitespace + const input: string = "\n]>\n \n"; + expectParses(input); + }); + + test("rmt-e2e-15e", () => { + // E15 — Element content can contain entity reference if replacement text is whitespace + const input: string = + '\n\n]>\n&space;\n'; + expectParses(input); + }); + + test("rmt-e2e-15f", () => { + // E15 — Element content can contain entity reference if replacement text is whitespace, even if it + // came from a character reference in the literal entity value + const input: string = + '\n\n]>\n&space;\n'; + expectParses(input); + }); + + test("rmt-e2e-15g", () => { + // E15 — Element content can't contain character reference to whitespace + const input: string = "\n]>\n \n"; + expectParses(input); + }); + + test("rmt-e2e-15h", () => { + // E15 — Element content can't contain entity reference if replacement text is character reference to + // whitespace + const input: string = + '\n\n]>\n&space;\n'; + expectParses(input); + }); + + test("rmt-e2e-15i", () => { + // E15 — Element content can contain a comment + const input: string = "\n]>\n\n"; + expectParses(input); + }); + + test("rmt-e2e-15j", () => { + // E15 — Element content can contain a PI + const input: string = "\n]>\n\n"; + expectParses(input); + }); + + test("rmt-e2e-15k", () => { + // E15 — Mixed content can contain a comment + const input: string = + "\n]>\n\n"; + expectParses(input); + }); + + test("rmt-e2e-15l", () => { + // E15 — Mixed content can contain a PI + const input: string = "\n]>\n\n"; + expectParses(input); + }); + + test("rmt-e2e-18", () => { + // E18 — External entity containing start of entity declaration is base URI for system identifier + // (upstream: valid; external general and parameter entities are not read; output depends on them) + const input: string = + '\n\n%pe;\n%intpe;\n]>\n&ent;\n'; + expectParses(input); + }); + + test("rmt-e2e-19", () => { + // E19 — Parameter entities and character references are included-in-literal, but general entities are + // bypassed. (upstream: valid; external parameter entities are not read; output depends on them) + const input: string = '\n&ent;\n'; + expectParses(input); + }); + + test("rmt-e2e-20", () => { + // E20 — Tokens, after normalization, must be separated by space, not other whitespace characters + const input: string = + '\n\n]>\n\n'; + expectParses(input); + }); + + test("rmt-e2e-22", () => { + // E22 — UTF-8 entities may start with a BOM + const input: string = '\ufeff\n\n]>\n\n'; + expectParses(input); + }); + + test("rmt-e2e-24", () => { + // E24 — Either the built-in entity or a character reference can be used to represent greater-than + // after two close-square-brackets + const input: string = + '\n">\n]>\nYou can use ]]> or ]]>\n'; + expectParses(input); + }); + + test("rmt-e2e-27", () => { + // E27 — Contains an irregular UTF-8 sequence (i.e. a surrogate pair) + const input = Buffer.from("PCFET0NUWVBFIGZvbyBbCjwhRUxFTUVOVCBmb28gQU5ZPgpdPgo8Zm9vPu2ggO2wgDwvZm9vPgo=", "base64"); + expectRejects(input, "XML Parse error: Invalid UTF-8"); + }); + + test("rmt-e2e-29", () => { + // E29 — Three-letter language codes are allowed + const input: string = + '\n\n]>\n\n \n\n'; + expectParses(input); + }); + + test("rmt-e2e-34", () => { + // E34 — A non-deterministic content model is an error even if the element type is not used. (upstream: + // optional error) + const input: string = "\n\n]>\n\n"; + expectParses(input); + }); + + test("rmt-e2e-36", () => { + // E36 — An external ATTLIST declaration does not make a document non-standalone if the normalization + // would have been the same without the declaration (upstream: valid; external parameter entities are + // not read) + const input: string = + '\n\n\n'; + expectParses(input); + }); + + test("rmt-e2e-38", () => { + // E38 — XML 1.0 document refers to 1.1 entity (upstream: not-wf; external general entities are not + // read) + const input: string = '\n\n]>\n&e;\n'; + expectParses(input); + }); + + test("rmt-e2e-41", () => { + // E41 — An xml:lang attribute may be empty + const input: string = + '\n\n]>\n\n'; + expectParses(input); + }); + + test("rmt-e2e-48", () => { + // E48 — ANY content allows character data + const input: string = "\n]>\nhello\n"; + expectParses(input); + }); + + test("rmt-e2e-55", () => { + // E55 — A reference to an unparsed entity in an entity value is an error rather than forbidden (unless + // the entity is referenced, of course) (upstream: optional error) + const input: string = + '\n\n\n\n]>\n\n'; + expectParses(input); + }); + + test("rmt-e2e-57", () => { + // E57 — A value other than preserve or default for xml:space is an error (upstream: optional error) + const input: string = '\n'; + expectParses(input); + }); + + test("rmt-e2e-60", () => { + // E60 — Conditional sections are allowed in external parameter entities referred to from the internal + // subset. (upstream: valid; external parameter entities are not read) + const input: string = + '\n\n\n%e;\n]>\n\n'; + expectParses(input); + }); + + test("rmt-e2e-61", () => { + // E61 — (From John Cowan) An encoding declaration in ASCII specifying an encoding that is not + // compatible with ASCII (so the document is not in its declared encoding). It should generate a fatal + // error. + const input = Buffer.from('\n\n'); + expectRejects(input, "XML Parse error: Document is not UTF-16 but declares encoding 'UTF-16'"); + }); +}); + +describe("eduni/errata-3e", () => { + test("rmt-e3e-05a", () => { + // E05 — CDATA sections may occur in Mixed content. + const input: string = + "\n\n]>\na in mixed content\n"; + expectParses(input); + }); + + test("rmt-e3e-05b", () => { + // E05 — CDATA sections, comments and PIs may occur in ANY content. + const input: string = + "\n\n]>\n\na in mixed content.\na in mixed content.\na in mixed content.\n\n"; + expectParses(input); + }); + + test("rmt-e3e-06a", () => { + // E06 — Default values for IDREF attributes must match Name. + const input: string = + '\n\n\n\n]>\n\n'; + expectParses(input); + }); + + test("rmt-e3e-06b", () => { + // E06 — Default values for ENTITY attributes must match Name. + const input: string = + '\n\n\n\n\n]>\n\n'; + expectParses(input); + }); + + test("rmt-e3e-06c", () => { + // E06 — Default values for IDREFS attributes must match Names. + const input: string = + '\n\n\n\n]>\n\n'; + expectParses(input); + }); + + test("rmt-e3e-06d", () => { + // E06 — Default values for ENTITIES attributes must match Names. + const input: string = + '\n\n\n\n\n]>\n\n'; + expectParses(input); + }); + + test("rmt-e3e-06e", () => { + // E06 — Default values for NMTOKEN attributes must match Nmtoken. + const input: string = + '\n\n\n]>\n\n'; + expectParses(input); + }); + + test("rmt-e3e-06f", () => { + // E06 — Default values for NMTOKENS attributes must match Nmtokens. + const input: string = + '\n\n\n]>\n\n'; + expectParses(input); + }); + + test("rmt-e3e-06g", () => { + // E06 — Default values for NOTATION attributes must match one of the enumerated values. + const input: string = + '\n\n\n\n\n]>\njunk\n'; + expectParses(input); + }); + + test("rmt-e3e-06h", () => { + // E06 — Default values for enumerated attributes must match one of the enumerated values. + const input: string = + '\n\n\n]>\n\n'; + expectParses(input); + }); + + test("rmt-e3e-06i", () => { + // E06 — Non-syntactic validity errors in default attributes only happen if the attribute is in fact + // defaulted. + const input: string = + '\n\n\n\n\n\n\n\n]>\n\n'; + expectParses(input); + }); + + test("rmt-e3e-12", () => { + // E12 — Default values for attributes may not contain references to external entities. + const input: string = + '\n\n\n\n]>\n\n'; + expectRejects(input, "XML Parse error: Attribute values cannot reference external entity 'ent'"); + }); + + test("rmt-e3e-13", () => { + // E13 — Even internal parameter entity references are enough to make undeclared entities into mere + // validity errors rather than well-formedness errors. + const input: string = + "\n\">\n%pe;\n\n]>\n&ent2;\n"; + expectParses(input); + }); +}); + +describe("eduni/errata-4e", () => { + test("invalid-bo-1", () => { + // 4.3.3 — Byte order mark in general entity should go away (big-endian) (upstream: invalid; external + // general entities are not read; output depends on them) + const input: string = '\n]>\n&bom;\n'; + expectParses(input); + }); + + test("invalid-bo-2", () => { + // 4.3.3 — Byte order mark in general entity should go away (little-endian) (upstream: invalid; + // external general entities are not read; output depends on them) + const input: string = '\n]>\n&bom;\n'; + expectParses(input); + }); + + test("invalid-bo-3", () => { + // 4.3.3 — Byte order mark in general entity should go away (utf-8) (upstream: invalid; external + // general entities are not read; output depends on them) + const input: string = '\n]>\n&bom8;\n'; + expectParses(input); + }); + + test("invalid-bo-4", () => { + // 4.3.3 — Two byte order marks in general entity produce only one (big-endian) (upstream: invalid; + // external general entities are not read; output depends on them) + const input: string = '\n]>\n&bombom;\n'; + expectParses(input); + }); + + test("invalid-bo-5", () => { + // 4.3.3 — Two byte order marks in general entity produce only one (little-endian) (upstream: invalid; + // external general entities are not read; output depends on them) + const input: string = '\n]>\n&bombom;\n'; + expectParses(input); + }); + + test("invalid-bo-6", () => { + // 4.3.3 — Two byte order marks in general entity produce only one (utf-8) (upstream: invalid; external + // general entities are not read; output depends on them) + const input: string = '\n]>\n&bombom8;\n'; + expectParses(input); + }); + + test("invalid-bo-7", () => { + // 4.3.3 — A byte order mark and a backwards one in general entity cause an illegal char. error + // (big-endian) (upstream: optional error) + const input: string = '\n]>\n&bomboom;\n'; + expectParses(input); + }); + + test("invalid-bo-8", () => { + // 4.3.3 — A byte order mark and a backwards one in general entity cause an illegal char. error + // (little-endian) (upstream: optional error) + const input: string = '\n]>\n&bomboom;\n'; + expectParses(input); + }); + + test("invalid-bo-9", () => { + // 4.3.3 — A byte order mark and a backwards one in general entity cause an illegal char. error (utf-8) + // (upstream: optional error) + const input: string = '\n]>\n&bomboom8;\n'; + expectParses(input); + }); + + test("invalid-sa-140", () => { + // 2.3 [4] — Character '゚' is a CombiningChar, not a Letter, but as of 5th edition, may begin a + // name (c.f. xmltest/not-wf/sa/140.xml). + const input: string = '">\r\n]>\r\n&e;\r\n'; + expectParses(input); + }); + + test("invalid-sa-141", () => { + // 2.3 [5] — As of 5th edition, character #x0E5C is legal in XML names (c.f. + // xmltest/not-wf/sa/141.xml). + const input: string = '">\r\n]>\r\n&e;\r\n'; + expectParses(input); + }); + + test("x-rmt-008b", () => { + // 2.8 4.3.4 — a document with version=1.7, legal in XML 1.0 from 5th edition + const input: string = + '\n\n\n]>\n\n'; + expectParses(input); + }); + + test("x-rmt5-014", () => { + // 2.3 — Has a "long s" in a name, legal in XML 1.1, legal in XML 1.0 5th edition + const input: string = '\n\n\n'; + expectParses(input); + }); + + test("x-rmt5-014a", () => { + // 2.3 — Has a "long s" in a name, legal in XML 1.1, legal in XML 1.0 5th edition + const input: string = + '\n\n\n]>\n\n\n'; + expectParses(input); + }); + + test("x-rmt5-016", () => { + // 2.3 — Has a Byzantine Musical Symbol Kratimata in a name, legal in XML 1.1, legal in XML 1.0 5th + // edition + const input: string = + "\n<𝀲/>\n"; + expectParses(input); + }); + + test("x-rmt5-019", () => { + // 2.3 — Has the last legal namechar in XML 1.1, legal in XML 1.0 5th edition + const input: string = "\n<󯿿/>\n"; + expectParses(input); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n02.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0x333 + const input: string = + "\n]>\n\n<̳IllegalNameStartChar/>"; + expectRejects(input, "XML Parse error: Expected the document type name but found '̳IllegalNameStartChar'"); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n03.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0x369 + const input: string = + "\n]>\n\n<ͩIllegalNameStartChar/>"; + expectRejects(input, "XML Parse error: Expected the document type name but found 'ͩIllegalNameStartChar'"); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n04.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0x37E + const input: string = + "\n]>\n\n<;IllegalNameStartChar/>"; + expectRejects(input, "XML Parse error: Expected the document type name but found ';' (U+037E)"); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n05.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0x2000 + const input: string = + "\n]>\n\n< IllegalNameStartChar/>"; + expectRejects(input, "XML Parse error: Expected the document type name but found ' ' (U+2000)"); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n06.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0x2001 + const input: string = + "\n]>\n\n< IllegalNameStartChar/>"; + expectRejects(input, "XML Parse error: Expected the document type name but found ' ' (U+2001)"); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n07.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0x2002 + const input: string = + "\n]>\n\n< IllegalNameStartChar/>"; + expectRejects(input, "XML Parse error: Expected the document type name but found ' ' (U+2002)"); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n08.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0x2005 + const input: string = + "\n]>\n\n< IllegalNameStartChar/>"; + expectRejects(input, "XML Parse error: Expected the document type name but found ' ' (U+2005)"); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n09.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0x200B + const input: string = + "\n]>\n\n<​IllegalNameStartChar/>"; + expectRejects(input, "XML Parse error: Expected the document type name but found '​' (U+200B)"); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n10.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0x200E + const input: string = + "\n]>\n\n<‎IllegalNameStartChar/>"; + expectRejects(input, "XML Parse error: Expected the document type name but found '‎' (U+200E)"); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n11.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0x200F + const input: string = + "\n]>\n\n<‏IllegalNameStartChar/>"; + expectRejects(input, "XML Parse error: Expected the document type name but found '‏' (U+200F)"); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n12.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0x2069 + const input: string = + "\n]>\n\n<⁩IllegalNameStartChar/>"; + expectRejects(input, "XML Parse error: Expected the document type name but found '⁩' (U+2069)"); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n13.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0x2190 + const input: string = + "\n]>\n\n<←IllegalNameStartChar/>"; + expectRejects(input, "XML Parse error: Expected the document type name but found '←' (U+2190)"); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n14.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0x23FF + const input: string = + "\n]>\n\n<⏿IllegalNameStartChar/>"; + expectRejects(input, "XML Parse error: Expected the document type name but found '⏿' (U+23FF)"); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n15.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0x280F + const input: string = + "\n]>\n\n<⠏IllegalNameStartChar/>"; + expectRejects(input, "XML Parse error: Expected the document type name but found '⠏' (U+280F)"); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n16.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0x2A00 + const input: string = + "\n]>\n\n<⨀IllegalNameStartChar/>"; + expectRejects(input, "XML Parse error: Expected the document type name but found '⨀' (U+2A00)"); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n17.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0x2EDC + const input: string = + "\r\n]>\r\n\r\n<⬀IllegalNameStartChar/>"; + expectRejects(input, "XML Parse error: Expected the document type name but found '⬀' (U+2B00)"); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n18.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0x2B00 + const input: string = + "\r\n]>\r\n\r\n<⯿IllegalNameStartChar/>"; + expectRejects(input, "XML Parse error: Expected the document type name but found '⯿' (U+2BFF)"); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n19.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0x2BFF + const input: string = + "\n]>\n\n<⿿IllegalNameStartChar/>"; + expectRejects(input, "XML Parse error: Expected the document type name but found '⿿' (U+2FFF)"); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n20.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0x3000 + const input: string = + "\n]>\n\n< IllegalNameStartChar/>"; + expectRejects(input, "XML Parse error: Expected the document type name but found ' ' (U+3000)"); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n21.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0xD800 + const input = Buffer.from( + "PCFET0NUWVBFIO2ggElsbGVnYWxOYW1lU3RhcnRDaGFyIFsKPCFFTEVNRU5UIO2ggElsbGVnYWxOYW1lU3RhcnRDaGFyIEFOWT4KXT4KPCEtLSBJbGxlZ2FsTmFtZVN0YXJ0Q2hhciAjMHhEODAwIC0tPgo87aCASWxsZWdhbE5hbWVTdGFydENoYXIvPgo=", + "base64", + ); + expectRejects(input, "XML Parse error: Invalid UTF-8"); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n22.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0xD801 + const input = Buffer.from( + "PCFET0NUWVBFIO2ggUlsbGVnYWxOYW1lU3RhcnRDaGFyIFsKPCFFTEVNRU5UIO2ggUlsbGVnYWxOYW1lU3RhcnRDaGFyIEFOWT4KXT4KPCEtLSBJbGxlZ2FsTmFtZVN0YXJ0Q2hhciAjMHhEODAxIC0tPgo87aCBSWxsZWdhbE5hbWVTdGFydENoYXIvPgo=", + "base64", + ); + expectRejects(input, "XML Parse error: Invalid UTF-8"); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n23.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0xDAFF + const input = Buffer.from( + "PCFET0NUWVBFIO2rv0lsbGVnYWxOYW1lU3RhcnRDaGFyIFsKPCFFTEVNRU5UIO2rv0lsbGVnYWxOYW1lU3RhcnRDaGFyIEFOWT4KXT4KPCEtLSBJbGxlZ2FsTmFtZVN0YXJ0Q2hhciAjMHhEQUZGIC0tPgo87au/SWxsZWdhbE5hbWVTdGFydENoYXIvPgo=", + "base64", + ); + expectRejects(input, "XML Parse error: Invalid UTF-8"); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n24.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0xDFFF + const input = Buffer.from( + "PCFET0NUWVBFIO2/v0lsbGVnYWxOYW1lU3RhcnRDaGFyIFsKPCFFTEVNRU5UIO2/v0lsbGVnYWxOYW1lU3RhcnRDaGFyIEFOWT4KXT4KPCEtLSBJbGxlZ2FsTmFtZVN0YXJ0Q2hhciAjMHhERkZGIC0tPgo87b+/SWxsZWdhbE5hbWVTdGFydENoYXIvPgo=", + "base64", + ); + expectRejects(input, "XML Parse error: Invalid UTF-8"); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n25.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0xEFFF + const input: string = + "\n]>\n\n<IllegalNameStartChar/>"; + expectRejects(input, "XML Parse error: Expected the document type name but found '' (U+EFFF)"); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n26.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0xF1FF + const input: string = + "\n]>\n\n<IllegalNameStartChar/>"; + expectRejects(input, "XML Parse error: Expected the document type name but found '' (U+F1FF)"); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n27.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0xF8FF + const input: string = + "\n]>\n\n<IllegalNameStartChar/>"; + expectRejects(input, "XML Parse error: Expected the document type name but found '' (U+F8FF)"); + }); + + test("x-ibm-1-0.5-not-wf-P04-ibm04n28.xml", () => { + // 2.3 — Tests an element with an illegal NameStartChar: #0xFFFFF + const input: string = + "\n]>\n\n<\uffffIllegalNameStartChar/>"; + expectRejects(input, "XML Parse error: Invalid character in XML: '\uffff' (U+FFFF)"); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an01.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #xB8 + const input: string = + "\n]>\n\n\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '¸' (U+00B8)", + ); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an02.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0xA1 + const input: string = + "\n]>\n\n\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '¡' (U+00A1)", + ); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an03.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0xAF + const input: string = + "\n]>\n\n\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '¯' (U+00AF)", + ); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an04.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0x37E + const input: string = + "\n]>\n\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found ';' (U+037E)", + ); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an05.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0x2000 + const input: string = + "\n]>\n\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found ' ' (U+2000)", + ); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an06.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0x2001 + const input: string = + "\n]>\n\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found ' ' (U+2001)", + ); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an07.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0x2002 + const input: string = + "\n]>\n\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found ' ' (U+2002)", + ); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an08.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0x2005 + const input: string = + "\n]>\n\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found ' ' (U+2005)", + ); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an09.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0x200B + const input: string = + "\n]>\n\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '​' (U+200B)", + ); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an10.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0x200E + const input: string = + "\n]>\n\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '‎' (U+200E)", + ); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an11.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0x2038 + const input: string = + "\r\n]>\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '‽' (U+203D)", + ); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an12.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0x2041 + const input: string = + "\r\n]>\r\n\r\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '⁁' (U+2041)", + ); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an13.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0x2190 + const input: string = + "\n]>\n\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '←' (U+2190)", + ); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an14.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0x23FF + const input: string = + "\n]>\n\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '⏿' (U+23FF)", + ); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an15.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0x280F + const input: string = + "\n]>\n\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '⠏' (U+280F)", + ); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an16.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0x2A00 + const input: string = + "\n]>\n\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '⨀' (U+2A00)", + ); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an17.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0xFDD0 + const input: string = + "\n]>\n\n\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '﷐' (U+FDD0)", + ); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an18.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0xFDEF + const input: string = + "\n]>\n\n\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '﷯' (U+FDEF)", + ); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an19.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0x2FFF + const input: string = + "\n]>\n\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '⿿' (U+2FFF)", + ); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an20.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0x3000 + const input: string = + "\n]>\n\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found ' ' (U+3000)", + ); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an21.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0xD800 + const input = Buffer.from( + "PCFET0NUWVBFIElsbGVnYWxOYW1lQ2hhcu2ggCBbCjwhRUxFTUVOVCBJbGxlZ2FsTmFtZUNoYXLtoIAgQU5ZPgpdPgo8IS0tIElsbGVnYWxOYW1lQ2hhciAjMHhEODAwIC0tPgo8SWxsZWdhbE5hbWVDaGFy7aCALz4K", + "base64", + ); + expectRejects(input, "XML Parse error: Invalid UTF-8"); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an22.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0xD801 + const input = Buffer.from( + "PCFET0NUWVBFIElsbGVnYWxOYW1lQ2hhcu2ggSBbCjwhRUxFTUVOVCBJbGxlZ2FsTmFtZUNoYXLtoIEgQU5ZPgpdPgo8IS0tIElsbGVnYWxOYW1lQ2hhciAjMHhEODAxIC0tPgo8SWxsZWdhbE5hbWVDaGFy7aCBLz4K", + "base64", + ); + expectRejects(input, "XML Parse error: Invalid UTF-8"); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an23.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0xDAFF + const input = Buffer.from( + "PCFET0NUWVBFIElsbGVnYWxOYW1lQ2hhcu2rvyBbCjwhRUxFTUVOVCBJbGxlZ2FsTmFtZUNoYXLtq78gQU5ZPgpdPgo8IS0tIElsbGVnYWxOYW1lQ2hhciAjMHhEQUZGIC0tPgo8SWxsZWdhbE5hbWVDaGFy7au/Lz4K", + "base64", + ); + expectRejects(input, "XML Parse error: Invalid UTF-8"); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an24.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0xDFFF + const input = Buffer.from( + "PCFET0NUWVBFIElsbGVnYWxOYW1lQ2hhcu2/vyBbCjwhRUxFTUVOVCBJbGxlZ2FsTmFtZUNoYXLtv78gQU5ZPgpdPgo8IS0tIElsbGVnYWxOYW1lQ2hhciAjMHhERkZGIC0tPgo8SWxsZWdhbE5hbWVDaGFy7b+/Lz4K", + "base64", + ); + expectRejects(input, "XML Parse error: Invalid UTF-8"); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an25.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0xEFFF + const input: string = + "\n]>\n\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '' (U+EFFF)", + ); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an26.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0xF1FF + const input: string = + "\n]>\n\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '' (U+F1FF)", + ); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an27.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0xF8FF + const input: string = + "\n]>\n\n"; + expectRejects( + input, + "XML Parse error: Expected SYSTEM, PUBLIC, '[' or '>' in the document type declaration but found '' (U+F8FF)", + ); + }); + + test("x-ibm-1-0.5-not-wf-P04a-ibm04an28.xml", () => { + // 2.3 — Tests an element with an illegal NameChar: #0xFFFFF + const input: string = + "\n]>\n\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: '\uffff' (U+FFFF)"); + }); + + test("x-ibm-1-0.5-not-wf-P05-ibm05n01.xml", () => { + // 2.3 — Tests an element with an illegal Name containing #0x0B + const input: string = + "\n\n]>\n\n\n\t\t\n"; + expectRejects(input, "XML Parse error: Invalid character in XML: control character 0x0B"); + }); + + test("x-ibm-1-0.5-not-wf-P05-ibm05n02.xml", () => { + // 2.3 — Tests an element with an illegal Name containing #0x300 + const input: string = + "\n\n]>\n\n\n\t<̀BadName/>\t\n\n"; + expectRejects(input, "XML Parse error: Expected an element name after ' { + // 2.3 — Tests an element with an illegal Name containing #0x36F + const input: string = + "\n\n]>\n\n\n\t<ͯBadName/>\t\n"; + expectRejects(input, "XML Parse error: Expected an element name after ' { + // 2.3 — Tests an element with an illegal Name containing #0x203F + const input: string = + "\n\n]>\n\n\n\t<‿BadName/>\t\n"; + expectRejects(input, "XML Parse error: Expected an element name after ' { + // 2.3 — Tests an element with an illegal Name containing #x2040 + const input: string = + "\n\n]>\n\n\n\t<⁀BadName/>\t\n"; + expectRejects(input, "XML Parse error: Expected an element name after ' { + // 2.3 — Tests an element with an illegal Name containing #0xB7 + const input: string = + "\n\n]>\n\n\n\t<·BadName/>\t\n"; + expectRejects(input, "XML Parse error: Expected an element name after ' { + // 2.3 — This test case covers legal NameStartChars character ranges plus discrete legal characters for + // production 04. + const input: string = + "\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n\r\n]>\r\n\r\n\t<:LegalNameStartChar/>\r\n\t<ÀLegalNameStartChar/>\r\n\t<ÁLegalNameStartChar/>\r\n\t<˾LegalNameStartChar/>\r\n\t<˿LegalNameStartChar/>\r\n\t<ͰLegalNameStartChar/>\r\n\t<ͱLegalNameStartChar/>\r\n\t<ͼLegalNameStartChar/>\r\n\t<ͽLegalNameStartChar/>\r\n\t<ͿLegalNameStartChar/>\r\n\t<΀LegalNameStartChar/>\r\n\t<῾LegalNameStartChar/>\r\n\t<῿LegalNameStartChar/>\r\n\t<‌LegalNameStartChar/>\r\n\t<‍LegalNameStartChar/>\r\n\t<⁰LegalNameStartChar/>\r\n\t<ⁱLegalNameStartChar/>\r\n\t<↎LegalNameStartChar/>\r\n\t<↏LegalNameStartChar/>\r\n\t<ⰀLegalNameStartChar/>\r\n\t<ⰁLegalNameStartChar/>\r\n\t<⿮LegalNameStartChar/>\r\n\t<⿯LegalNameStartChar/>\r\n\t<、LegalNameStartChar/>\r\n\t<。LegalNameStartChar/>\r\n\t<퟾LegalNameStartChar/>\r\n\t<퟿LegalNameStartChar/>\r\n\t<豈LegalNameStartChar/>\r\n\t<更LegalNameStartChar/>\r\n\r\n"; + expectParses(input); + }); + + test("x-ibm-1-0.5-valid-P04-ibm04av01.xml", () => { + // 2.3 — This test case covers legal NameChars character ranges plus discrete legal characters for + // production 04a. + const input: string = + "\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n]>\n\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n"; + expectParses(input); + }); + + test("x-ibm-1-0.5-valid-P05-ibm05v01.xml", () => { + // 2.3 — This test case covers legal Element Names as per production 5. + const input: string = + "\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n]>\n\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n"; + expectParses(input); + }); + + test("x-ibm-1-0.5-valid-P05-ibm05v02.xml", () => { + // 2.3 — This test case covers legal PITarget (Names) as per production 5. + const input: string = + "\n]>\n\n\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n\t\n"; + expectParses(input); + }); + + test("x-ibm-1-0.5-valid-P05-ibm05v03.xml", () => { + // 2.3 — This test case covers legal Attribute (Names) as per production 5. + const input: string = + '\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n]>\n\n'; + expectParses(input); + }); + + test("x-ibm-1-0.5-valid-P05-ibm05v04.xml", () => { + // 2.3 — This test case covers legal ID/IDREF (Names) as per production 5. + const input: string = + '\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n]>\n\n\n \tattr0=":" attr00=":"\n \tattr1="À" attr10="À"\n \tattr2="Á" attr20="Á"\n \tattr3="˾" attr30="˾"\n \tattr4="Â" attr40="Â"\n \tattr5="Ã" attr50="Ã"\n \tattr6="˽" attr60="˽"\n \tattr7="˿" attr70="˿"\n \tattr8="Ͱ" attr80="Ͱ"\n \tattr9="ͱ" attr90="ͱ"\n \tattr10="ͼͽ" attr100="ͼͽ"\n \tattr11="ͽͿ" attr110="ͽͿ"\n \tattr12="Ϳ΀" attr120="Ϳ΀"\n \tattr13="΀῾" attr130="΀῾"\n \tattr14="῾῿" attr140="῾῿"\n \tattr15="῿‌" attr150="῿‌"\n \tattr16="‌‍" attr160="‌‍"\n \tattr17="‍⁰" attr170="‍⁰"\n \tattr18="⁰ⁱ" attr180="⁰ⁱ"\n \tattr19="ⁱ↎" attr190="ⁱ↎"\n \tattr20="↎↏Ⰰ" attr200="↎↏Ⰰ"\n \tattr21="↏ⰀⰁ" attr210="↏ⰀⰁ"\n \tattr22="ⰀⰁ⿮" attr220="ⰀⰁ⿮"\n \tattr23="Ⰱ⿮⿯" attr230="Ⰱ⿮⿯"\n \tattr24="⿮⿯、" attr240="⿮⿯、"\n \tattr25="⿯、。" attr250="⿯、。"\n \tattr26="、。퟾" attr260="、。퟾"\n \tattr27="。퟾퟿" attr270="。퟾퟿"\n \tattr28="퟾퟿豈" attr280="퟾퟿豈"\n \tattr29="퟿豈更" attr290="퟿豈更"\n \tattr30="豈퟿퟾。" attr300="豈퟿퟾。"\n \tattr31="更豈퟿퟾" attr310="更豈퟿퟾"\n \tattr32="�更豈퟿" attr320="�更豈퟿"\n \tattr33="-�更豈" attr330="-�更豈"\n \tattr34=".-�更" attr340=".-�更"\n \tattr35="A.-�" attr350="A.-�"\n \tattr36="zA.-" attr360="zA.-"\n \tattr37="0zA." attr370="0zA."\n \tattr38="·0zA" attr380="·0zA"\n \tattr39="̀·0z" attr390="̀·0z"\n \tattr40="́̀·0" attr400="́̀·0"\n \tattr41="ͮ́̀·" attr410="ͮ́̀·"\n \tattr42="ͯͮ́̀" attr420="ͯͮ́̀"\n \tattr43="‿ͯͮ́" attr430="‿ͯͮ́"\n \tattr44="⁀‿ͯͮ" attr440="⁀‿ͯͮ"\n \tattr45="null⁀‿ͯ" attr450="null⁀‿ͯ"\n \tattr46="nullnull⁀‿" attr460="nullnull⁀‿"\n \tattr47="nullnullnull⁀" attr470="nullnullnull⁀"\n'; + expectParses(input); + }); + + test("x-ibm-1-0.5-valid-P05-ibm05v05.xml", () => { + // 2.3 — This test case covers legal ENTITY (Names) as per production 5. + const input: string = + '\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n]>\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n'; + expectParses(input); + }); + + test("x-ibm-1-0.5-valid-P047-ibm07v01.xml", () => { + // 2.3 — This test case covers legal NMTOKEN Name character ranges plus discrete legal characters for + // production 7. + const input: string = + '\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n\n]>\n\n'; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n03.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0132 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n04.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0133 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n05.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x013F occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n06.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0140 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n07.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0149 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n08.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x017F occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n09.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x01c4 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n10.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x01CC occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n100.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0BB6 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n101.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0BBA occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n102.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0C0D occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n103.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0C11 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n104.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0C29 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n105.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0C34 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n106.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0C5F occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n107.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0C62 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n108.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0C8D occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n109.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0C91 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n11.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x01F1 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n110.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0CA9 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n111.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0CB4 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n112.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0CBA occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n113.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0CDF occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n114.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0CE2 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n115.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0D0D occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n116.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0D11 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n117.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0D29 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n118.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0D3A occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n119.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0D62 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n12.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x01F3 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n120.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0E2F occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n121.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0E31 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n122.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0E34 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n123.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0E46 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n124.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0E83 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n125.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0E85 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n126.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0E89 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n127.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0E8B occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n128.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0E8E occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n129.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0E98 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n13.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x01F6 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n130.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0EA0 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n131.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0EA4 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n132.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0EA6 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n133.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0EA8 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n134.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0EAC occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n135.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0EAF occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n136.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0EB1 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n137.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0EB4 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n138.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0EBE occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n139.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0EC5 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n14.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x01F9 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n140.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0F48 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n141.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0F6A occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n142.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x10C6 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n143.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x10F7 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n144.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1011 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n145.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1104 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n146.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1108 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n147.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x110A occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n148.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x110D occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n149.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x113B occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n15.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x01F9 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n150.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x113F occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n151.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1141 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n152.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x114D occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n153.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x114f occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n154.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1151 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n155.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1156 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n156.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x115A occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n157.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1162 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n158.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1164 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n159.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1166 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n16.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0230 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n160.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x116B occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n161.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x116F occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n162.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1174 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n163.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x119F occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n164.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x11AC occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n165.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x11B6 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n166.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x11B9 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n167.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x11BB occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n168.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x11C3 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n169.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x11F1 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n17.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x02AF occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n170.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x11FA occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n171.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1E9C occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n172.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1EFA occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n173.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1F16 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n174.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1F1E occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n175.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1F46 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n176.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1F4F occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n177.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1F58 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n178.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1F5A occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n179.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1F5C occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n18.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x02CF occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n180.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1F5E occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n181.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1F7E occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n182.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1FB5 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n183.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1FBD occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n184.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1FBF occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n185.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1FC5 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n186.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1FCD occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n187.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1FD5 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n188.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1FDC occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n189.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1FED occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n19.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0387 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n190.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1FF5 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n191.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x1FFD occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n192.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x2127 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n193.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x212F occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n194.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x2183 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n195.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x3095 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n196.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x30FB occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n197.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x312D occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n198.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #xD7A4 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n20.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x038B occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n21.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x03A2 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n22.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x03CF occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n23.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x03D7 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n24.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x03DD occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n25.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x03E1 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n26.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x03F4 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n27.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x040D occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n28.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0450 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n29.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x045D occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n30.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0482 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n31.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x04C5 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n32.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x04C6 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n33.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x04C9 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n34.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x04EC occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n35.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x04ED occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n36.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x04F6 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n37.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x04FA occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n38.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0557 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n39.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0558 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n40.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0587 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n41.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x05EB occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n42.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x05F3 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n43.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0620 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n44.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x063B occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n45.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x064B occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n46.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x06B8 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n47.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x06BF occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n48.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x06CF occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n49.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x06D4 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n50.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x06D6 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n51.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x06E7 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n52.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x093A occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n53.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x093E occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n54.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0962 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n55.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x098D occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n56.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0991 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n57.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0992 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n58.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x09A9 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n59.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x09B1 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n60.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x09B5 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n61.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x09BA occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n62.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x09DE occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n63.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x09E2 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n64.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x09F2 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n65.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0A0B occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n66.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0A11 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n67.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0A29 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n68.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0A31 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n69.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0A34 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n70.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0A37 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n71.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0A3A occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n72.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0A5D occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n73.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0A70 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n74.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0A75 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n75.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #xA84 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n76.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0ABC occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n77.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0A92 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n78.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0AA9 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n79.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0AB1 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n80.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0AB4 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n81.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0ABA occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n82.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0B04 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n83.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0B0D occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n84.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0B11 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n85.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0B29 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n86.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0B31 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n87.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0B34 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n88.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0B3A occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n89.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0B3E occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n90.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0B5E occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n91.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0B62 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n92.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0B8B occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n93.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0B91 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n94.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0B98 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n95.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0B9B occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n96.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0B9D occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n97.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0BA0 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n98.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0BA7 occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P85-ibm85n99.xml", () => { + // B. — Tests BaseChar with an only legal per 5th edition character. The character #x0BAB occurs as the + // first character of the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P86-ibm86n01.xml", () => { + // B. — Tests Ideographic with an only legal per 5th edition character. The character #x4CFF occurs as + // the first character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P86-ibm86n02.xml", () => { + // B. — Tests Ideographic with an only legal per 5th edition character. The character #x9FA6 occurs as + // the first character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P86-ibm86n03.xml", () => { + // B. — Tests Ideographic with an only legal per 5th edition character. The character #x3008 occurs as + // the first character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P86-ibm86n04.xml", () => { + // B. — Tests Ideographic with an only legal per 5th edition character. The character #x302A occurs as + // the first character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n01.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x02FF occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n02.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0346 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n03.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0362 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n04.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0487 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n05.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x05A2 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n06.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x05BA occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n07.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x05BE occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n08.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x05C0 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n09.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x05C3 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n10.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0653 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n11.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x06B8 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n12.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x06B9 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n13.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x06E9 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n14.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x06EE occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n15.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0904 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n16.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x093B occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n17.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x094E occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n18.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0955 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n19.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0964 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n20.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0984 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n21.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x09C5 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n22.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x09C9 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n23.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x09CE occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n24.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x09D8 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n25.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x09E4 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n26.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0A03 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n27.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0A3D occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n28.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0A46 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n29.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0A49 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n30.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0A4E occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n31.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0A80 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n32.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0A84 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n33.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0ABB occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n34.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0AC6 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n35.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0ACA occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n36.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0ACE occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n37.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0B04 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n38.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0B3B occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n39.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0B44 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n40.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0B4A occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n41.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0B4E occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n42.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0B58 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n43.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0B84 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n44.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0BC3 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n45.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0BC9 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n46.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0BD6 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n47.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0C0D occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n48.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0C45 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n49.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0C49 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n50.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0C54 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n51.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0C81 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n52.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0C84 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n53.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0CC5 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n54.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0CC9 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n55.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0CD4 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n56.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0CD7 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n57.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0D04 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n58.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0D45 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n59.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0D49 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n60.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0D4E occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n61.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0D58 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n62.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0E3F occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n63.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0E3B occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n64.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0E4F occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n66.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0EBA occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n67.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0EBE occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n68.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0ECE occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n69.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0F1A occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n70.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0F36 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n71.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0F38 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n72.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0F3B occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n73.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0F3A occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n74.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0F70 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n75.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0F85 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n76.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0F8C occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n77.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0F96 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n78.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0F98 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n79.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0FB0 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n80.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0FB8 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n81.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x0FBA occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n82.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x20DD occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n83.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x20E2 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n84.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x3030 occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P87-ibm87n85.xml", () => { + // B. — Tests CombiningChar with an only legal per 5th edition character. The character #x309B occurs + // as the second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P88-ibm88n03.xml", () => { + // B. — Tests Digit with an only legal per 5th edition character. The character #x066A occurs as the + // second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P88-ibm88n04.xml", () => { + // B. — Tests Digit with an only legal per 5th edition character. The character #x06FA occurs as the + // second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P88-ibm88n05.xml", () => { + // B. — Tests Digit with an only legal per 5th edition character. The character #x0970 occurs as the + // second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P88-ibm88n06.xml", () => { + // B. — Tests Digit with an only legal per 5th edition character. The character #x09F2 occurs as the + // second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P88-ibm88n08.xml", () => { + // B. — Tests Digit with an only legal per 5th edition character. The character #x0AF0 occurs as the + // second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P88-ibm88n09.xml", () => { + // B. — Tests Digit with an only legal per 5th edition character. The character #x0B70 occurs as the + // second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P88-ibm88n10.xml", () => { + // B. — Tests Digit with an only legal per 5th edition character. The character #x0C65 occurs as the + // second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P88-ibm88n11.xml", () => { + // B. — Tests Digit with an only legal per 5th edition character. The character #x0CE5 occurs as the + // second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P88-ibm88n12.xml", () => { + // B. — Tests Digit with an only legal per 5th edition character. The character #x0CF0 occurs as the + // second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P88-ibm88n13.xml", () => { + // B. — Tests Digit with an only legal per 5th edition character. The character #x0D70 occurs as the + // second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P88-ibm88n14.xml", () => { + // B. — Tests Digit with an only legal per 5th edition character. The character #x0E5A occurs as the + // second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P88-ibm88n15.xml", () => { + // B. — Tests Digit with an only legal per 5th edition character. The character #x0EDA occurs as the + // second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P88-ibm88n16.xml", () => { + // B. — Tests Digit with an only legal per 5th edition character. The character #x0F2A occurs as the + // second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P89-ibm89n03.xml", () => { + // B. — Tests Extender with an only legal per 5th edition character. The character #x02D2 occurs as the + // second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P89-ibm89n04.xml", () => { + // B. — Tests Extender with an only legal per 5th edition character. The character #x03FE occurs as the + // second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-valid-P89-ibm89n05.xml", () => { + // B. — Tests Extender with an only legal per 5th edition character. The character #x065F occurs as the + // second character in the PITarget in the PI in the DTD. + const input: string = + "\r\n\r\n]>\r\n\r\n"; + expectParses(input); + }); + + test("ibm-invalid-P89-ibm89n06.xml", () => { + // B. — Tests Extender with an only legal per 5th edition character. The character #x0EC7 occurs as the + // second character in the PITarget in the PI in the prolog, and in an element name. + const input: string = + "\r\n\r\n"; + expectParses(input); + }); + + test("ibm-invalid-P89-ibm89n07.xml", () => { + // B. — Tests Extender with an only legal per 5th edition character. The character #x3006 occurs as the + // second character in the PITarget in the PI in the prolog, and in an element name. + const input: string = + "\r\n\r\n"; + expectParses(input); + }); + + test("ibm-invalid-P89-ibm89n08.xml", () => { + // B. — Tests Extender with an only legal per 5th edition character. The character #x3030 occurs as the + // second character in the PITarget in the PI in the prolog, and in an element name. + const input: string = + "\r\n\r\n"; + expectParses(input); + }); + + test("ibm-invalid-P89-ibm89n09.xml", () => { + // B. — Tests Extender with an only legal per 5th edition character. The character #x3036 occurs as the + // second character in the PITarget in the PI in the prolog, and in an element name. + const input: string = + "\r\n\r\n"; + expectParses(input); + }); + + test("ibm-invalid-P89-ibm89n10.xml", () => { + // B. — Tests Extender with an only legal per 5th edition character. The character #x309C occurs as the + // second character in the PITarget in the PI in the prolog, and in an element name. + const input: string = + "\r\n\r\n"; + expectParses(input); + }); + + test("ibm-invalid-P89-ibm89n11.xml", () => { + // B. — Tests Extender with an only legal per 5th edition character. The character #x309F occurs as the + // second character in the PITarget in the PI in the prolog, and in an element name. + const input: string = + "\r\n\r\n"; + expectParses(input); + }); + + test("ibm-invalid-P89-ibm89n12.xml", () => { + // B. — Tests Extender with an only legal per 5th edition character. The character #x30FF occurs as the + // second character in the PITarget in the PI in the prolog, and in an element name. + const input: string = + "\r\n\r\n"; + expectParses(input); + }); +}); + +describe("eduni/namespaces-1.0", () => { + test("rmt-ns10-001", () => { + // 2 — Namespace name test: a perfectly good http URI (upstream: valid; namespace constraints are not + // enforced) + const input: string = + '\n\n\n\n]>\n\n'; + expectParses(input); + }); + + test("rmt-ns10-002", () => { + // 2 — Namespace name test: a syntactically plausible URI with a fictitious scheme (upstream: valid; + // namespace constraints are not enforced) + const input: string = + '\n\n\n\n]>\n\n'; + expectParses(input); + }); + + test("rmt-ns10-003", () => { + // 2 — Namespace name test: a perfectly good http URI with a fragment (upstream: valid; namespace + // constraints are not enforced) + const input: string = + '\n\n\n\n]>\n\n'; + expectParses(input); + }); + + test("rmt-ns10-004", () => { + // 2 — Namespace name test: a relative URI (deprecated) (upstream: error; namespace constraints are not + // enforced) + const input: string = + '\n\n\n]\n>\n\n'; + expectParses(input); + }); + + test("rmt-ns10-005", () => { + // 2 — Namespace name test: a same-document relative URI (deprecated) (upstream: error; namespace + // constraints are not enforced) + const input: string = + '\n\n\n\n]>\n\n'; + expectParses(input); + }); + + test("rmt-ns10-006", () => { + // 2 — Namespace name test: an http IRI that is not a URI (upstream: error; namespace constraints are + // not enforced) + const input = Buffer.from( + "PD94bWwgdmVyc2lvbj0iMS4wIiBlbmNvZGluZz0iaXNvLTg4NTktMSI/Pgo8IS0tIE5hbWVzcGFjZSBuYW1lIHRlc3Q6IGFuIGh0dHAgSVJJIHRoYXQgaXMgbm90IGEgVVJJIC0tPgo8IURPQ1RZUEUgZm9vIFsKPCFFTEVNRU5UIGZvbyBBTlk+CjwhQVRUTElTVCBmb28geG1sbnMgQ0RBVEEgI0lNUExJRUQ+Cl0+Cjxmb28geG1sbnM9Imh0dHA6Ly9leGFtcGxlLm9yZy9yb3PpIi8+Cg==", + "base64", + ); + expectParses(input); + }); + + test("rmt-ns10-007", () => { + // 1 — Namespace inequality test: different capitalization (upstream: valid; namespace constraints are + // not enforced) + const input: string = + '\n\n\n\n\n\n]>\n\n\n\n\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-008", () => { + // 1 — Namespace inequality test: different escaping (upstream: valid; namespace constraints are not + // enforced) + const input: string = + '\n\n\n\n\n\n]>\n\n\n\n\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-009", () => { + // 1 — Namespace equality test: plain repetition (upstream: not-wf; namespace constraints are not + // enforced) + const input: string = + '\n\n\n\n\n\n]>\n\n\n\n\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-010", () => { + // 1 — Namespace equality test: use of character reference (upstream: not-wf; namespace constraints are + // not enforced) + const input: string = + '\n\n\n\n\n\n]>\n\n\n\n\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-011", () => { + // 1 — Namespace equality test: use of entity reference (upstream: not-wf; namespace constraints are + // not enforced) + const input: string = + '\n\n\n\n\n\n\n]>\n\n\n\n\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-012", () => { + // 1 — Namespace inequality test: equal after attribute value normalization (upstream: not-wf; + // namespace constraints are not enforced) + const input: string = + '\n\n\n\n\n\n]>\n\n\n\n\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-013", () => { + // 3 — Bad QName syntax: multiple colons (upstream: not-wf; namespace constraints are not enforced) + const input: string = + '\n\n\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-014", () => { + // 3 — Bad QName syntax: colon at end (upstream: not-wf; namespace constraints are not enforced) + const input: string = '\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-015", () => { + // 3 — Bad QName syntax: colon at start (upstream: not-wf; namespace constraints are not enforced) + const input: string = '\n\n<:foo />\n'; + expectParses(input); + }); + + test("rmt-ns10-016", () => { + // 2 — Bad QName syntax: xmlns: (upstream: not-wf; namespace constraints are not enforced) + const input: string = + '\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-017", () => { + // - — Simple legal case: no namespaces (upstream: invalid; namespace constraints are not enforced) + const input: string = '\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-018", () => { + // 5.2 — Simple legal case: default namespace (upstream: invalid; namespace constraints are not + // enforced) + const input: string = + '\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-019", () => { + // 4 — Simple legal case: prefixed element (upstream: invalid; namespace constraints are not enforced) + const input: string = + '\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-020", () => { + // 4 — Simple legal case: prefixed attribute (upstream: invalid; namespace constraints are not + // enforced) + const input: string = + '\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-021", () => { + // 5.2 — Simple legal case: default namespace and unbinding (upstream: invalid; namespace constraints + // are not enforced) + const input: string = + '\n\n\n \n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-022", () => { + // 5.2 — Simple legal case: default namespace and rebinding (upstream: invalid; namespace constraints + // are not enforced) + const input: string = + '\n\n\n \n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-023", () => { + // 2 — Illegal use of 1.1-style prefix unbinding in 1.0 document (upstream: not-wf; namespace + // constraints are not enforced) + const input: string = + '\n\n\n \n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-024", () => { + // 5.1 — Simple legal case: prefix rebinding (upstream: invalid; namespace constraints are not + // enforced) + const input: string = + '\n\n\n \n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-025", () => { + // 4 — Unbound element prefix (upstream: not-wf; namespace constraints are not enforced) + const input: string = '\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-026", () => { + // 4 — Unbound attribute prefix (upstream: not-wf; namespace constraints are not enforced) + const input: string = '\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-027", () => { + // 2 — Reserved prefixes and namespaces: using the xml prefix undeclared (upstream: invalid; namespace + // constraints are not enforced) + const input: string = + '\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-028", () => { + // NE05 — Reserved prefixes and namespaces: declaring the xml prefix correctly (upstream: invalid; + // namespace constraints are not enforced) + const input: string = + '\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-029", () => { + // NE05 — Reserved prefixes and namespaces: declaring the xml prefix incorrectly (upstream: not-wf; + // namespace constraints are not enforced) + const input: string = + '\n\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-030", () => { + // NE05 — Reserved prefixes and namespaces: binding another prefix to the xml namespace (upstream: + // not-wf; namespace constraints are not enforced) + const input: string = + '\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-031", () => { + // NE05 — Reserved prefixes and namespaces: declaring the xmlns prefix with its correct URI (illegal) + // (upstream: not-wf; namespace constraints are not enforced) + const input: string = + '\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-032", () => { + // NE05 — Reserved prefixes and namespaces: declaring the xmlns prefix with an incorrect URI (upstream: + // not-wf; namespace constraints are not enforced) + const input: string = + '\n\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-033", () => { + // NE05 — Reserved prefixes and namespaces: binding another prefix to the xmlns namespace (upstream: + // not-wf; namespace constraints are not enforced) + const input: string = + '\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-034", () => { + // NE05 — Reserved prefixes and namespaces: binding a reserved prefix (upstream: invalid; namespace + // constraints are not enforced) + const input: string = + '\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-035", () => { + // 5.3 — Attribute uniqueness: repeated identical attribute (upstream: not-wf; namespace constraints + // are not enforced) + const input: string = + '\n\n\n\n\n\n\n'; + expectRejects(input, "XML Parse error: Duplicate attribute 'a:attr'"); + }); + + test("rmt-ns10-036", () => { + // 5.3 — Attribute uniqueness: repeated attribute with different prefixes (upstream: not-wf; namespace + // constraints are not enforced) + const input: string = + '\n\n\n\n\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-037", () => { + // 5.3 — Attribute uniqueness: different attributes with same local name (upstream: invalid; namespace + // constraints are not enforced) + const input: string = + '\n\n\n\n\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-038", () => { + // 5.3 — Attribute uniqueness: prefixed and unprefixed attributes with same local name (upstream: + // invalid; namespace constraints are not enforced) + const input: string = + '\n\n\n\n\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-039", () => { + // 5.3 — Attribute uniqueness: prefixed and unprefixed attributes with same local name, with default + // namespace (upstream: invalid; namespace constraints are not enforced) + const input: string = + '\n\n\n\n\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-040", () => { + // 5.3 — Attribute uniqueness: prefixed and unprefixed attributes with same local name, with default + // namespace and element in default namespace (upstream: invalid; namespace constraints are not + // enforced) + const input: string = + '\n\n\n\n\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-041", () => { + // 5.3 — Attribute uniqueness: prefixed and unprefixed attributes with same local name, element in same + // namespace as prefixed attribute (upstream: invalid; namespace constraints are not enforced) + const input: string = + '\n\n\n\n\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-042", () => { + // NE08 — Colon in PI name (upstream: not-wf; namespace constraints are not enforced) + const input: string = '\n\n\n\n'; + expectParses(input); + }); + + test("rmt-ns10-043", () => { + // NE08 — Colon in entity name (upstream: not-wf; namespace constraints are not enforced) + const input: string = + '\n\n\n\n]>\n\n'; + expectParses(input); + }); + + test("rmt-ns10-044", () => { + // NE08 — Colon in entity name (upstream: not-wf; namespace constraints are not enforced) + const input: string = + '\n\n\n\n]>\n\n'; + expectParses(input); + }); + + test("rmt-ns10-045", () => { + // NE08 — Colon in ID attribute name (upstream: invalid; namespace constraints are not enforced) + const input: string = + '\n\n\n\n]>\n\n'; + expectParses(input); + }); + + test("rmt-ns10-046", () => { + // NE08 — Colon in ID attribute name (upstream: invalid; namespace constraints are not enforced) + const input: string = + '\n\n\n\n]>\n\n \n\n'; + expectParses(input); + }); + + test("ht-ns10-047", () => { + // NE03 — Reserved name: _not_ an error (upstream: valid; namespace constraints are not enforced) + const input: string = "\n]>\n\n"; + expectParses(input); + }); + + test("ht-ns10-048", () => { + // NE03 — Reserved name: _not_ an error (upstream: valid; namespace constraints are not enforced) + const input: string = + '\n\n]>\n\n'; + expectParses(input); + }); + + test("rmt-ns-e1.0-13a", () => { + // NE13 — The xml namespace must not be declared as the default namespace. (upstream: not-wf; namespace + // constraints are not enforced) + const input: string = + '\n\n\n\n]>\n\n'; + expectParses(input); + }); + + test("rmt-ns-e1.0-13b", () => { + // NE13 — The xmlns namespace must not be declared as the default namespace. (upstream: not-wf; + // namespace constraints are not enforced) + const input: string = + '\n\n\n\n]>\n\n'; + expectParses(input); + }); + + test("rmt-ns-e1.0-13c", () => { + // NE13 — Elements must not have the prefix xmlns. (upstream: not-wf; namespace constraints are not + // enforced) + const input: string = + '\n\n\n]>\n\n'; + expectParses(input); + }); +}); + +describe("eduni/misc", () => { + test("hst-bh-001", () => { + // 2.2 [2], 4.1 [66] — decimal charref > 10FFFF, indeed > max 32 bit integer, checking for recovery + // from possible overflow + const input: string = + "\n]>\n

Fa�il

\n"; + expectRejects(input, "XML Parse error: Character reference '�' is not a valid XML character"); + }); + + test("hst-bh-002", () => { + // 2.2 [2], 4.1 [66] — hex charref > 10FFFF, indeed > max 32 bit integer, checking for recovery from + // possible overflow + const input: string = + "\n]>\n

Fa�il

\n"; + expectRejects(input, "XML Parse error: Character reference '�' is not a valid XML character"); + }); + + test("hst-bh-003", () => { + // 2.2 [2], 4.1 [66] — decimal charref > 10FFFF, indeed > max 64 bit integer, checking for recovery + // from possible overflow + const input: string = + "\n]>\n

Fa�il

\n"; + expectRejects(input, "XML Parse error: Character reference '�' is not a valid XML character"); + }); + + test("hst-bh-004", () => { + // 2.2 [2], 4.1 [66] — hex charref > 10FFFF, indeed > max 64 bit integer, checking for recovery from + // possible overflow + const input: string = + "\n]>\n

Fa�il

\n"; + expectRejects(input, "XML Parse error: Character reference '�' is not a valid XML character"); + }); + + test("hst-bh-005", () => { + // 3.1 [41] — xmlns:xml is an attribute as far as validation is concerned and must be declared + const input: string = + " ]>\n\n"; + expectParses(input); + }); + + test("hst-bh-006", () => { + // 3.1 [41] — xmlns:foo is an attribute as far as validation is concerned and must be declared + const input: string = " ]>\n\n"; + expectParses(input); + }); + + test("hst-lhs-007", () => { + // 4.3.3 — UTF-8 BOM plus xml decl of iso-8859-1 incompatible + const input = Buffer.from("\ufeff\n"); + expectRejects(input, "XML Parse error: Document has a UTF-8 byte-order mark but declares encoding 'iso-8859-1'"); + }); + + test("hst-lhs-008", () => { + // 4.3.3 — UTF-16 BOM plus xml decl of utf-8 (using UTF-16 coding) incompatible + const input = Buffer.from( + "/v8APAA/AHgAbQBsACAAdgBlAHIAcwBpAG8AbgA9ACcAMQAuADAAJwAgAGUAbgBjAG8AZABpAG4AZwA9ACcAdQB0AGYALQA4ACcAPwA+ADwAeAAvAD4=", + "base64", + ); + expectRejects(input, "XML Parse error: Document is UTF-16 but declares encoding 'utf-8'"); + }); + + test("hst-lhs-009", () => { + // 4.3.3 — UTF-16 BOM plus xml decl of utf-8 (using UTF-8 coding) incompatible + const input = Buffer.from("/v88P3htbCBlbmNvZGluZz0ndXRmLTgnPz48eC8+Cg==", "base64"); + expectRejects(input, "XML Parse error: UTF-16 input has an odd number of bytes"); + }); +}); diff --git a/test/js/bun/xml/xml.test.ts b/test/js/bun/xml/xml.test.ts new file mode 100644 index 000000000000..54761cece115 --- /dev/null +++ b/test/js/bun/xml/xml.test.ts @@ -0,0 +1,853 @@ +import { XML } from "bun"; +import { describe, expect, test } from "bun:test"; + +// Hand-written coverage beyond the W3C conformance suite (xml-test-suite.test.ts): +// the JS-facing API surface, the two result shapes, Bun-specific input types, +// what a non-validating processor that does not load external entities does at +// the edges, resource limits, and XML.stringify. + +function syntaxError(input: string | Uint8Array, options?: XML.ParseOptions): SyntaxError { + let err: unknown; + try { + XML.parse(input, options); + } catch (e) { + err = e; + } + expect(err).toBeInstanceOf(SyntaxError); + expect((err as SyntaxError).message).toStartWith("XML Parse error: "); + return err as SyntaxError; +} + +const nodes = { compact: false } as const; + +describe("input types", () => { + const doc = `app80808081`; + const expected = { config: { "@version": "2", name: "app", port: ["8080", "8081"] } }; + + test("string", () => { + expect(XML.parse(doc)).toEqual(expected); + }); + + test("Buffer", () => { + expect(XML.parse(Buffer.from(doc))).toEqual(expected); + }); + + test("Uint8Array subarray respects byteOffset and length", () => { + const padded = Buffer.from("<<<" + doc + ">>>"); + expect(XML.parse(padded.subarray(3, 3 + doc.length))).toEqual(expected); + }); + + test("DataView", () => { + const bytes = new TextEncoder().encode(doc); + expect(XML.parse(new DataView(bytes.buffer))).toEqual(expected); + }); + + test("ArrayBuffer and SharedArrayBuffer", () => { + const bytes = new TextEncoder().encode(doc); + expect(XML.parse(bytes.buffer)).toEqual(expected); + const sab = new SharedArrayBuffer(bytes.length); + new Uint8Array(sab).set(bytes); + expect(XML.parse(sab)).toEqual(expected); + }); + + test("Blob parses synchronously", () => { + expect(XML.parse(new Blob([doc]))).toEqual(expected); + }); + + test("nullish input throws TypeError; other values are stringified", () => { + expect(() => XML.parse(undefined as any)).toThrow(TypeError); + expect(() => XML.parse(null as any)).toThrow(TypeError); + expect(XML.parse({ toString: () => "1" } as any)).toEqual({ a: "1" }); + }); + + test("options must be an object; compact must be a boolean", () => { + expect(() => XML.parse("", 1 as any)).toThrow(TypeError); + expect(() => XML.parse("", { compact: "no" } as any)).toThrow(TypeError); + expect(XML.parse("", null as any)).toEqual({ a: "" }); + expect(XML.parse("", {})).toEqual({ a: "" }); + expect(XML.parse("", { compact: undefined })).toEqual({ a: "" }); + }); +}); + +describe("compact shape", () => { + test("leaf elements are their text; attributes and children make an object", () => { + expect(XML.parse("")).toEqual({ a: "" }); + expect(XML.parse("")).toEqual({ a: "" }); + expect(XML.parse("text")).toEqual({ a: "text" }); + expect(XML.parse(``)).toEqual({ a: { "@x": "1" } }); + expect(XML.parse(`text`)).toEqual({ a: { "@x": "1", "#text": "text" } }); + expect(XML.parse(`1`)).toEqual({ a: { b: "1" } }); + expect(XML.parse(`1t`)).toEqual({ a: { b: "1", "#text": "t" } }); + }); + + test("repeated names become arrays in document order, grouped at the first occurrence", () => { + const result = XML.parse(`1x23`) as any; + expect(result).toEqual({ r: { a: ["1", "2", "3"], b: "x", c: "" } }); + expect(Object.keys(result.r)).toEqual(["a", "b", "c"]); + }); + + test("attributes come first, then children, then #text", () => { + const result = XML.parse(`t`) as any; + expect(Object.keys(result.r)).toEqual(["@z", "b", "a", "#text"]); + }); + + test("text is trimmed of XML whitespace at the ends only, and concatenated across children", () => { + expect(XML.parse(`\r\n\t x y \n`)).toEqual({ a: "x y" }); + // U+00A0 and other Unicode spaces are not XML whitespace. + expect(XML.parse(` x
`)).toEqual({ a: " x
" }); + expect(XML.parse(`

Hello big world

`)).toEqual({ p: { b: "big", "#text": "Hello world" } }); + expect(XML.parse(` `)).toEqual({ a: { b: "" } }); + }); + + test("CDATA and references are just text", () => { + expect(XML.parse(` & ]]>`)).toEqual({ a: " &" }); + expect(XML.parse(`<AB&`)).toEqual({ a: " { + expect( + XML.parse(`xy`), + ).toEqual({ a: "xy" }); + }); + + test("nothing is coerced: numbers, booleans and null-ish stay strings", () => { + expect(XML.parse(`1.0truenull1e30x10`)).toEqual({ + a: { "@b": "1", n: "1.0", t: "true", z: "null", e: "1e3", h: "0x10" }, + }); + }); + + test("names are kept verbatim, including prefixes and xmlns declarations", () => { + expect(XML.parse(``)).toEqual({ + "s:Envelope": { "@xmlns:s": "urn:s", "s:Body": { "@xml:lang": "en", "@s:id": "1" } }, + }); + expect(XML.parse(`<日本 属性="値">テキスト`)).toEqual({ 日本: { "@属性": "値", "#text": "テキスト" } }); + expect(XML.parse(`𐀀`)).toEqual({ a: { "@b": "𝄞", "#text": "𐀀" } }); + }); + + test("__proto__ and constructor are plain own data properties", () => { + const result = XML.parse(`<__proto__ constructor="1"><__proto__>x`) as any; + expect(Object.getPrototypeOf(result)).toBe(Object.prototype); + expect(Object.hasOwn(result, "__proto__")).toBe(true); + const root = result["__proto__"]; + expect(Object.getPrototypeOf(root)).toBe(Object.prototype); + expect(Object.hasOwn(root, "__proto__")).toBe(true); + expect(root["@constructor"]).toBe("1"); + expect(root["__proto__"]).toBe("x"); + expect(({} as any)["@constructor"]).toBeUndefined(); + }); +}); + +describe("node shape", () => { + test("root element as { name, attributes, children }, everything in document order", () => { + expect(XML.parse(`

Hello big world

`, nodes)).toEqual({ + name: "p", + attributes: { class: "x", id: "y" }, + children: [ + "Hello ", + { name: "b", attributes: {}, children: ["big"] }, + " world", + { name: "br", attributes: {}, children: [] }, + ], + }); + }); + + test("whitespace is kept exactly, adjacent text is one string", () => { + expect(XML.parse(`\n x \n`, nodes)).toEqual({ + name: "a", + attributes: {}, + children: ["\n ", { name: "b", attributes: {}, children: [" x "] }, "\n"], + }); + expect(XML.parse(`x&z`, nodes)).toEqual({ + name: "a", + attributes: {}, + children: ["xy&z"], + }); + }); + + test("attribute order is document order, defaults appended", () => { + const node = XML.parse(`]>`, nodes); + expect(Object.entries(node.attributes)).toEqual([ + ["z", "1"], + ["b", "2"], + ["e", "3"], + ["d", "0"], + ]); + }); +}); + +describe("well-formedness", () => { + test("errors are SyntaxErrors with a location-free message", () => { + expect(syntaxError("").message).toBe("XML Parse error: XML document must have a root element"); + expect(syntaxError("").message).toBe("XML Parse error: Missing closing tag for element 'a'"); + expect(syntaxError("
").message).toBe("XML Parse error: Expected closing tag but found
"); + expect(syntaxError("").message).toBe("XML Parse error: Only one root element is allowed"); + expect(syntaxError("junk").message).toBe("XML Parse error: Unexpected 'junk' after the root element"); + expect(syntaxError("junk").message).toBe("XML Parse error: Expected the root element but found 'junk'"); + expect(syntaxError(``).message).toBe("XML Parse error: Duplicate attribute 'b'"); + expect(syntaxError(``).message).toBe("XML Parse error: Expected a quoted attribute value but found '1'"); + expect(syntaxError(``).message).toBe("XML Parse error: Expected '=' after the attribute name but found '/>'"); + expect(syntaxError(``).message).toBe("XML Parse error: '<' is not allowed in attribute values"); + expect(syntaxError(``).message).toBe("XML Parse error: Expected closing tag but found "); + // A stray token is named where it stands, not scanned for what it might have started. + expect(syntaxError(`\n'junk`).message).toBe("XML Parse error: Unexpected ''' after the root element"); + expect(syntaxError(`'`).message).toBe("XML Parse error: Expected the root element but found '''"); + expect(syntaxError(` b="1"/>`).message).toBe( + "XML Parse error: Expected an attribute name, '>' or '/>' in the start tag but found a comment", + ); + expect(syntaxError(``).message).toBe( + "XML Parse error: Expected an attribute name, '>' or '/>' in the start tag but found '%b;'", + ); + }); + + test("character rules", () => { + expect(syntaxError("\x00").message).toBe( + "XML Parse error: Invalid character in XML: control character 0x00", + ); + expect(syntaxError("\x1f").message).toBe( + "XML Parse error: Invalid character in XML: control character 0x1F", + ); + expect(syntaxError("").message).toContain("Invalid character"); + expect(syntaxError("").message).toContain("Invalid character"); + expect(syntaxError("￾").message).toBe("XML Parse error: Invalid character in XML: '￾' (U+FFFE)"); + expect(syntaxError("￿").message).toContain("U+FFFF"); + expect(XML.parse("\t\n\r …�\u{10ffff}")).toEqual({ a: "…�\u{10ffff}" }); + // Bytes must be valid UTF-8; lone surrogates cannot be encoded. + expect(syntaxError(Buffer.from([0x3c, 0x61, 0x3e, 0xff, 0x3c, 0x2f, 0x61, 0x3e])).message).toBe( + "XML Parse error: Invalid UTF-8", + ); + expect(syntaxError(Buffer.from([0x3c, 0x61, 0x3e, 0xed, 0xa0, 0x80, 0x3c, 0x2f, 0x61, 0x3e])).message).toBe( + "XML Parse error: Invalid UTF-8", + ); + }); + + test("character references must name a Char", () => { + expect(XML.parse("A😀 ")).toEqual({ a: "A\u{1F600}" }); + for (const ref of [ + "�", + "�", + "", + " ", + "�", + "�", + "￾", + "�", + "�", + ]) { + expect(syntaxError(`${ref}`).message).toContain("is not a valid XML character"); + } + expect(syntaxError("&#;").message).toContain("Invalid character reference"); + expect(syntaxError("&#x;").message).toContain("Invalid character reference"); + expect(syntaxError("a;").message).toContain("Invalid character reference"); + expect(syntaxError("&# 1;").message).toContain("Invalid character reference"); + }); + + test("names", () => { + expect(XML.parse("<_.-:0/>")).toEqual({ "_.-:0": "" }); + expect(XML.parse("<:a/>")).toEqual({ ":a": "" }); + expect(syntaxError("<0a/>").message).toBe("XML Parse error: Expected an element name after '<' but found '0'"); + expect(syntaxError("<-a/>").message).toContain("Expected an element name"); + expect(syntaxError("").message).toBe( + "XML Parse error: Expected an attribute name, '>' or '/>' in the start tag but found '×' (U+00D7)", + ); + expect(syntaxError("< a/>").message).toContain("but found space"); + expect(syntaxError("< /a>").message).toContain("but found space"); + expect(syntaxError("").message).toContain("but found space"); + expect(XML.parse("")).toEqual({ a: "" }); + expect(XML.parse("")).toEqual({ a: { "@b": "1" } }); + }); + + test("comments, CDATA and processing instructions", () => { + expect(syntaxError("").message).toBe("XML Parse error: '--' is not allowed inside a comment"); + expect(syntaxError("").message).toContain("comment"); + expect(syntaxError(" ]>`)).toEqual({ a: "" }); + // Invalid (root name mismatch, undeclared elements) but well-formed: accepted. + expect(XML.parse(`]>`)).toEqual({ a: { b: "" } }); + expect(syntaxError(` ]>`).message).toBe( + "XML Parse error: Expected an element name after ''", + ); + expect(syntaxError(` ]>`).message).toContain("cannot mix ',' and '|'"); + expect(syntaxError(` ]>`).message).toContain("must end with ')*'"); + expect(syntaxError(` ]>`).message).toContain( + "#REQUIRED, #IMPLIED, #FIXED or a quoted default value", + ); + expect(syntaxError(` ]>`).message).toBe( + "XML Parse error: Expected a quoted entity value, SYSTEM or PUBLIC but found '>'", + ); + expect(syntaxError(` ]>`).message).toContain("Conditional sections"); + expect(syntaxError(``).message).toBe( + "XML Parse error: Expected a markup declaration or ']' in the internal subset but found 'junk'", + ); + expect(syntaxError(``).message).toBe( + "XML Parse error: Only one document type declaration is allowed", + ); + expect(syntaxError(``).message).toBe( + "XML Parse error: Unexpected '`).message).toContain("' { + const doc = ` + bold &plain;"> + + ]>&markup; &markup;`; + expect(XML.parse(doc)).toEqual({ d: { "@a": "[text] [line break]", b: ["bold", "bold"], "#text": "text text" } }); + expect(XML.parse(doc, nodes).children).toEqual([ + { name: "b", attributes: {}, children: ["bold"] }, + " text ", + { name: "b", attributes: {}, children: ["bold"] }, + " text", + ]); + }); + + test("the first declaration of an entity is binding; predefined entities cannot be overridden", () => { + expect(XML.parse(`]>&e;<`)).toEqual({ + d: "1<", + }); + }); + + test("entity replacement text must be well-formed in context", () => { + expect(syntaxError(`">]>&e;`).message).toBe( + "XML Parse error: Element 'd' must start and end within the same entity", + ); + expect(syntaxError(`">]>&e;
`).message).toBe( + "XML Parse error: Element 'x' must start and end within the same entity", + ); + expect(syntaxError(`]>&e; -->`).message).toBe( + "XML Parse error: Unterminated comment", + ); + expect(syntaxError(`]>`).message).toBe( + "XML Parse error: '<' is not allowed in attribute values", + ); + expect(syntaxError(`]>&a;`).message).toBe( + "XML Parse error: Entity 'a' refers to itself", + ); + expect(syntaxError(`]>`).message).toBe( + "XML Parse error: Entity 'a' refers to itself", + ); + // A reference to an undeclared entity inside an entity value is only an + // error when that entity is used (§4.4.7: bypassed at declaration). + expect(XML.parse(`]>`)).toEqual({ d: "" }); + expect(syntaxError(`]>&e;`).message).toBe( + "XML Parse error: Entity 'nope' is not declared", + ); + // The classic double escape from Appendix D of the spec. + expect( + XML.parse( + `An ampersand (&#38;) may be escaped numerically (&#38;#38;) or with a general entity (&amp;).

">]>&example;`, + ), + ).toEqual({ + d: { p: "An ampersand (&) may be escaped numerically (&) or with a general entity (&)." }, + }); + }); + + test("attribute defaults and attribute-value normalization by declared type", () => { + const doc = ` + + + ]>`; + expect(XML.parse(doc)).toEqual({ + d: { + "@id": "x", + "@other": " y ", + "@tokens": "a b", + "@text": " a b ", + "@fixed": "f", + e: { "@n": "z", "@c": " z " }, + }, + }); + }); + + test("attribute values: whitespace characters become spaces, character references do not", () => { + expect(XML.parse(``)).toEqual({ d: { "@a": "x y z w" } }); + expect(XML.parse(``)).toEqual({ d: { "@a": "x\ty\nz\rw" } }); + // Whitespace inside an entity's replacement text is normalized where it + // is used (§3.3.3 and the table in Appendix E)... + expect( + XML.parse( + `]>`, + ), + ).toEqual({ + d: { "@x": " A B " }, + }); + // ...unless it got there as a character reference to a character reference. + expect(XML.parse(`]>`)).toEqual({ d: { "@x": "a\nb" } }); + }); + + test("internal parameter entities are expanded between declarations", () => { + const doc = `"> + %decls; + ]>&e;`; + expect(XML.parse(doc)).toEqual({ d: { "@a": "default", "#text": "from pe" } }); + // In the internal subset parameter entities may only appear between declarations. + expect(syntaxError(`]>`).message).toBe( + "XML Parse error: Parameter entity references are not allowed inside markup declarations in the internal subset", + ); + expect(syntaxError(`]>`).message).toContain( + "Parameter entity references are not allowed inside markup declarations", + ); + // ...and must contain whole declarations. + expect(syntaxError(`%half;>]>`).message).toBe( + "XML Parse error: A markup declaration must begin and end in the same entity", + ); + expect(syntaxError(` %e; ]>`).message).toBe( + "XML Parse error: ']' inside a parameter entity cannot close the internal subset", + ); + // Inside replacement text, though, a reference may stand for any tokens of a + // declaration (here: the system literal), as in the external subset. + expect( + XML.parse(`">%decl;]>&e;`), + ).toEqual({ d: "&e;" }); + // `]>`).message).toBe( + "XML Parse error: Whitespace is required between '%' and the name in a parameter entity declaration", + ); + }); + + describe("external entities are never loaded", () => { + test("an external subset or unread parameter entity turns undeclared entities into kept references", () => { + // Without a DTD (or with standalone="yes") an undeclared entity is a well-formedness error... + expect(syntaxError(`
 `).message).toBe("XML Parse error: Entity 'nbsp' is not declared"); + expect(syntaxError(`]> `).message).toBe( + "XML Parse error: Entity 'nbsp' is not declared", + ); + expect( + syntaxError(` `).message, + ).toBe("XML Parse error: Entity 'nbsp' is not declared"); + // ...but when the declaration could be in the part of the DTD that is not + // loaded it is only a validity error, and the reference is kept as text. + expect(XML.parse(`

a b

`)).toEqual({ + p: "a b", + }); + expect(XML.parse(`%ext;]>

&t;

`)).toEqual({ + p: { "@title": "&t;", "#text": "&t;" }, + }); + }); + + test("declared external entities are kept as references in content and rejected in attributes", () => { + const dtd = `]>`; + expect(XML.parse(`${dtd}&ext;|&pub;`)).toEqual({ d: "&ext;|&pub;" }); + expect(syntaxError(`${dtd}`).message).toBe( + "XML Parse error: Attribute values cannot reference external entity 'ext'", + ); + }); + + test("unparsed entities cannot be referenced at all", () => { + expect( + syntaxError(`]>&img;`) + .message, + ).toBe("XML Parse error: Unparsed entity 'img' cannot be referenced"); + }); + + test("declarations after an unread parameter entity are not processed unless standalone", () => { + const doc = (standalone: string) => + ` + + %ext; + + + ]>`; + expect(XML.parse(doc("no"))).toEqual({ d: { "@before": "1" } }); + expect(XML.parse(doc("yes"))).toEqual({ d: { "@before": "1", "@after": "2" } }); + }); + }); + + test("entity expansion is bounded (billion laughs)", () => { + let decls = ``; + for (let i = 1; i <= 12; i++) + decls += ``; + expect(syntaxError(`&l12;`).message).toBe( + "XML Parse error: Entity expansion exceeds the amplification limit", + ); + expect(syntaxError(``).message).toBe( + "XML Parse error: Entity expansion exceeds the amplification limit", + ); + // Heavy but honest use of entities is fine. + const big = Buffer.alloc(50_000, "x").toString(); + const result = XML.parse(`]>${Buffer.alloc(3 * 40, "&e;")}`) as any; + expect(result.d.length).toBe(40 * 50_000); + }); +}); + +describe("encodings", () => { + const text = `naïve – 日本`; + const expected = { doc: { "@attr": "café", "#text": "naïve – 日本" } }; + + test("strings are already decoded: the encoding declaration is checked but not applied", () => { + expect(XML.parse(`${text}`)).toEqual(expected); + // ...including values that get coerced to a string. + expect(XML.parse(new String(`${text}`) as any)).toEqual(expected); + expect(XML.parse({ toString: () => `${text}` } as any)).toEqual(expected); + expect(XML.parse(`${text}`)).toEqual(expected); + expect(XML.parse(`${text}`)).toEqual(expected); + expect(XML.parse(`${text}`)).toEqual(expected); + expect(syntaxError(`${text}`).message).toContain("Invalid encoding name"); + }); + + test("UTF-8 bytes, with or without a BOM or declaration", () => { + expect(syntaxError(Buffer.from([0x3c, 0x80, 0x61, 0x2f, 0x3e])).message).toBe("XML Parse error: Invalid UTF-8"); + expect( + syntaxError(Buffer.concat([Buffer.from(``), Buffer.from([0x3c, 0x80, 0x61, 0x2f, 0x3e])])) + .message, + ).toBe("XML Parse error: Invalid UTF-8"); + expect(XML.parse(Buffer.from(text))).toEqual(expected); + expect(XML.parse(Buffer.from("" + text))).toEqual(expected); + expect(XML.parse(Buffer.from(`${text}`))).toEqual(expected); + }); + + test("UTF-16 bytes in either byte order, by BOM or by declaration", () => { + const le = Buffer.from("" + text, "utf16le"); + const be = Buffer.from(le).swap16(); + expect(XML.parse(le)).toEqual(expected); + expect(XML.parse(be)).toEqual(expected); + const declared = `${text}`; + expect(XML.parse(Buffer.from(declared, "utf16le"))).toEqual(expected); + expect(XML.parse(Buffer.from(declared.replace("UTF-16", "Utf-16le"), "utf16le"))).toEqual(expected); + expect(XML.parse(Buffer.from(declared, "utf16le").swap16())).toEqual(expected); + expect(syntaxError(Buffer.from(text, "utf16le")).message).toBe( + `XML Parse error: UTF-16 input must start with a byte-order mark or declare encoding="UTF-16"`, + ); + // A lone surrogate is not decodable. + const bad = Buffer.from("x", "utf16le"); + bad[6] = 0x00; + bad[7] = 0xd8; + expect(syntaxError(bad).message).toBe("XML Parse error: Invalid UTF-16"); + }); + + test("ISO-8859-1 bytes by declaration", () => { + expect( + XML.parse(Buffer.from(`na\xefve`, "latin1")), + ).toEqual({ + doc: { "@attr": "café", "#text": "naïve" }, + }); + expect(XML.parse(Buffer.from(`\xff`, "latin1"))).toEqual({ d: "ÿ" }); + // Transcoding restarts the buffer; that must not make a second declaration legal. + expect( + syntaxError(Buffer.from(`\xe9`, "latin1")) + .message, + ).toContain("' { + expect(syntaxError(Buffer.from(``)).message).toBe( + "XML Parse error: Document is not UTF-16 but declares encoding 'UTF-16'", + ); + expect(syntaxError(Buffer.from(``, "utf16le")).message).toBe( + "XML Parse error: Document is UTF-16 but declares encoding 'UTF-8'", + ); + expect(syntaxError(Buffer.from(``)).message).toBe( + "XML Parse error: Document has a UTF-8 byte-order mark but declares encoding 'ISO-8859-1'", + ); + expect(syntaxError(Buffer.from(``)).message).toBe( + "XML Parse error: Unsupported encoding 'Shift_JIS' (supported: UTF-8, UTF-16, ISO-8859-1)", + ); + expect(syntaxError(Buffer.from(`caf\xe9`, "latin1")).message).toBe("XML Parse error: Invalid UTF-8"); + }); +}); + +describe("XML.stringify", () => { + test("compact objects", () => { + expect(XML.stringify({ a: "" })).toBe(""); + expect(XML.stringify({ a: null })).toBe(""); + expect(XML.stringify({ a: {} })).toBe(""); + expect(XML.stringify({ a: "text" })).toBe("text"); + expect(XML.stringify({ a: 1.5 })).toBe("1.5"); + expect(XML.stringify({ a: false })).toBe("false"); + expect(XML.stringify({ a: 10n })).toBe("10"); + expect(XML.stringify({ a: new Date(0) })).toBe("1970-01-01T00:00:00.000Z"); + expect(XML.stringify({ a: { "@x": 1, "@y": "two", "@n": null, "@u": undefined } })).toBe(``); + expect(XML.stringify({ a: { "#text": "t" } })).toBe("t"); + expect(XML.stringify({ a: { "@x": "1", "#text": "t" } })).toBe(`t`); + expect(XML.stringify({ a: { b: ["1", "2", null, ""], c: { d: "e" } } })).toBe( + "12e", + ); + // Children and text are written in key order; attributes always land on the start tag. + expect(XML.stringify({ a: { "#text": "t", b: "1", "@z": "last" } })).toBe(`t1`); + // Skipped values, and arrays holding only skipped values, are not content. + expect(XML.stringify({ a: { b: undefined, c: () => {}, d: Symbol("s"), e: "kept" } })).toBe("kept"); + expect(XML.stringify({ a: { b: [] } })).toBe(""); + expect(XML.stringify({ a: { b: [undefined, () => {}] } })).toBe(""); + expect(XML.stringify({ a: { b: [undefined], c: [] } }, null, 2)).toBe(""); + expect(XML.stringify({ a: { b: [undefined, "x"] } }, null, 2)).toBe("\n x\n"); + expect(XML.stringify({ skip: undefined, a: "1" } as any)).toBe("1"); + }); + + test("nodes", () => { + expect(XML.stringify({ name: "a", attributes: {}, children: [] })).toBe(""); + expect(XML.stringify({ name: "a", children: [] })).toBe(""); + expect(XML.stringify({ name: "a", attributes: { x: "1" } })).toBe(``); + expect(XML.stringify({ name: "a", attributes: null, children: null } as any)).toBe(``); + expect( + XML.stringify({ + name: "p", + attributes: { class: "x" }, + children: ["Hello ", { name: "b", children: ["big", 1, true] }, " world", { name: "br" }, null, undefined], + }), + ).toBe(`

Hello big1true world

`); + // A top-level object needs children or attributes to be taken as a node + // (present counts, even holding undefined)... + expect(XML.stringify({ name: "a" })).toBe("a"); + expect(XML.stringify({ name: "br", attributes: undefined, children: undefined })).toBe("
"); + class El { + constructor( + public name: string, + public kids: string[], + ) {} + get children() { + return this.kids; + } + } + expect(XML.stringify(new El("i", ["x"]))).toBe("x"); + // ...inside children any object is a node. + expect(() => XML.stringify({ name: "a", children: [{ foo: "bar" }] } as any)).toThrow("with a string name"); + expect(() => XML.stringify({ name: "a", children: "text" } as any)).toThrow("children must be an array"); + expect(() => XML.stringify({ name: "a", children: [], attributes: [] } as any)).toThrow( + "attributes must be an object", + ); + expect(() => XML.stringify({ name: "a", children: [["nested"]] } as any)).toThrow("cannot contain arrays"); + expect(() => XML.stringify({ name: "a", children: [], attributes: { x: {} } } as any)).toThrow( + "an attribute value must be", + ); + }); + + test("escaping keeps the document well-formed and round-trippable", () => { + expect(XML.stringify({ a: `<&>'"` })).toBe(`
<&>'"`); + expect(XML.stringify({ a: { "@v": `<&>'"` } })).toBe(``); + expect(XML.stringify({ a: "]]>" })).toBe("]]>"); + // Whitespace that a parser would normalize away is written as references. + expect(XML.stringify({ a: { "@v": "a\tb\nc\rd e" } })).toBe(``); + expect(XML.stringify({ a: "a\tb\nc\r\nd" })).toBe("a\tb\nc \nd"); + expect(XML.parse(XML.stringify({ a: { "@v": "a\tb\nc\rd e", "#text": "x\r\ny" } }))).toEqual({ + a: { "@v": "a\tb\nc\rd e", "#text": "x\r\ny" }, + }); + expect(XML.stringify({ a: { "@b": "𝄞", "#text": "\u{10000}é" } })).toBe(`\u{10000}é`); + }); + + test("what XML cannot represent throws", () => { + expect(() => XML.stringify({ a: "\0" })).toThrow("XML cannot represent the character U+0000"); + expect(() => XML.stringify({ a: { "@b": "\x01" } })).toThrow("U+0001"); + expect(() => XML.stringify({ a: "￾" })).toThrow("U+FFFE"); + expect(() => XML.stringify({ a: "\ud800" })).toThrow("U+D800"); + expect(() => XML.stringify({ a: "a\udc00b" })).toThrow("U+DC00"); + for (const bad of ["", "1a", "-a", ".a", "a b", "a>", "a/", "×"]) { + expect(() => XML.stringify({ [bad]: "x" })).toThrow("is not a valid XML element name"); + expect(() => XML.stringify({ a: { ["@" + bad]: "x" } })).toThrow("is not a valid XML attribute name"); + expect(() => XML.stringify({ name: bad, children: [] })).toThrow("is not a valid XML element name"); + } + expect(XML.stringify({ "_a-b.c:d·": { "@xml:lang": "en", 名前: "v" } })).toBe( + `<_a-b.c:d· xml:lang="en"><名前>v`, + ); + expect(() => XML.stringify({ a: { "#comment": "x" } })).toThrow("keys starting with '#' are reserved"); + expect(() => XML.stringify({ a: { b: [["x"]] } })).toThrow("nested arrays"); + expect(() => XML.stringify({ a: { b: { c: {} } }, extra: "1" })).toThrow("more than one key"); + expect(() => XML.stringify({ a: ["1", "2"] })).toThrow("cannot be an array"); + expect(() => XML.stringify({})).toThrow("must have one key naming the root element"); + expect(() => XML.stringify({ "@a": "1" })).toThrow("can only contain the root element"); + expect(() => XML.stringify({ "#text": "1" })).toThrow("can only contain the root element"); + // In element position an object is an element; its function-valued keys are skipped. + expect(XML.stringify({ a: { b: { toString: () => "no" } } as any })).toBe(""); + expect(() => XML.stringify({ a: { "@b": { toString: () => "no" } } as any })).toThrow("an attribute value must be"); + expect(() => XML.stringify({ a: Symbol("s") as any })).toThrow("must have one key"); + const circular: any = { a: { b: {} } }; + circular.a.b.c = circular.a; + expect(() => XML.stringify(circular)).toThrow("Converting circular structure to XML"); + const node: any = { name: "a", children: [] }; + node.children.push(node); + expect(() => XML.stringify(node)).toThrow("Converting circular structure to XML"); + const shared = { c: "1" }; + expect(XML.stringify({ a: { b: [shared, shared], d: shared } })).toBe( + "111", + ); + expect(() => XML.stringify({ a: new Date(NaN) })).toThrow("invalid Date"); + }); + + test("signature parity with JSON.stringify", () => { + expect(XML.stringify(undefined)).toBeUndefined(); + expect(XML.stringify(() => {})).toBeUndefined(); + expect(XML.stringify(Symbol("s"))).toBeUndefined(); + expect(() => XML.stringify(null)).toThrow("expects an object"); + expect(() => XML.stringify("x")).toThrow("expects an object"); + expect(() => XML.stringify([{ a: 1 }])).toThrow("expects an object"); + expect(() => XML.stringify({ a: "1" }, (() => 1) as any)).toThrow("does not support the replacer"); + expect(() => XML.stringify({ a: "1" }, ["a"] as any)).toThrow("does not support the replacer"); + expect(XML.stringify({ a: "1" }, null)).toBe("1"); + expect(() => XML.stringify(new String("") as any)).toThrow("expects an object"); + }); + + test("space indents element-only content and leaves text content inline", () => { + // Repeated children inside mixed content stay inline too. + expect(XML.stringify({ p: { "#text": "hi", b: ["1", "2"] } }, null, 2)).toBe("

hi12

"); + expect(XML.stringify({ root: { p: { "#text": "hi", b: ["1", "2"] } } }, null, 2)).toBe( + "\n

hi12

\n
", + ); + const value = { + root: { "@id": "1", list: { item: ["a", "b"], empty: [] }, mixed: { b: "x", "#text": "t" }, leaf: "" }, + }; + expect(XML.stringify(value, null, 2)).toBe( + `\n \n a\n b\n \n xt\n \n`, + ); + expect(XML.stringify(value, null, "\t")).toBe(XML.stringify(value, null, 2).replaceAll(" ", "\t")); + expect(XML.stringify(value, null, 100)).toBe(XML.stringify(value, null, 10)); + expect(XML.stringify(value, null, "abcdefghijklmnop")).toBe( + XML.stringify(value, null, 10).replaceAll(" ", "abcdefghij"), + ); + for (const minified of [0, -1, NaN, "", null, undefined, true, {}]) { + expect(XML.stringify(value, null, minified as any)).toBe(XML.stringify(value)); + } + const node = XML.parse(`
1`, nodes); + expect(XML.stringify(node, null, 1)).toBe("\n \n 1\n \n \n"); + // Whitespace-only text children still count as text: the tree round-trips exactly. + const spaced = XML.parse(`\n \n`, nodes); + expect(XML.stringify(spaced, null, 4)).toBe(`\n \n`); + }); + + test("parse(stringify(x)) round-trips both shapes", () => { + const docs = [ + ``, + `t`, + `12`, + `

Hello big world!

`, + `x y]]>z<&`, + `e">]>&e;&e;<__proto__/>`, + ``, + ]; + for (const doc of docs) { + const compact = XML.parse(doc); + expect(XML.parse(XML.stringify(compact))).toEqual(compact); + expect(XML.parse(XML.stringify(compact, null, 2))).toEqual(compact); + const node = XML.parse(doc, nodes); + expect(XML.parse(XML.stringify(node), nodes)).toEqual(node); + // Pretty-printing element-only content adds whitespace text nodes, so + // compare through the compact projection instead. + expect(XML.parse(XML.stringify(node, null, 2))).toEqual(compact); + } + }); + + test("deep values are a catchable error", () => { + // Must overflow on every build: release frames are far smaller than + // debug/ASAN ones, so use a depth no native stack survives. + const depth = 1_000_000; + let deep: any = "x"; + for (let i = 0; i < depth; i++) deep = { a: deep }; + expect(() => XML.stringify(deep)).toThrow(RangeError); + deep = undefined; + let node: any = { name: "a", children: ["x"] }; + for (let i = 0; i < depth; i++) node = { name: "a", children: [node] }; + expect(() => XML.stringify(node)).toThrow(RangeError); + }); +});