Commit 6e9f3032 authored by Matthias Betz's avatar Matthias Betz
Browse files

replace sax with a purpose-built XML tokenizer, parsing 2.2-2.9x faster



The parser only needs element names and coordinate text. The new tokenizer
finds markup with native indexOf instead of sax's per-character state
machine and slices text only while a LinearRing is read.

- 274 MB LoD2 file: 20.1 s -> 6.9 s in Node, 23.3 s -> 8.3 s in the browser
- identical output (hash over all vertex positions and colours) on 156
  test files incl. CityGML 3.0
- element names are matched case-sensitively as local names
- unterminated markup over 64 KB still fails; a truncated file still shows
  what was read
- vendor/sax.js removed

Co-Authored-By: default avatarClaude Opus 5.5 (1M context) <noreply@anthropic.com>
parent 1ddc3cda
...@@ -6,7 +6,6 @@ ...@@ -6,7 +6,6 @@
<meta name="viewport" content="width=device-width, initial-scale=1"> <meta name="viewport" content="width=device-width, initial-scale=1">
<title>CityGML Viewer</title> <title>CityGML Viewer</title>
<link rel="stylesheet" href="src/style.css"> <link rel="stylesheet" href="src/style.css">
<script src="vendor/sax.js"></script>
<script src="vendor/libtess.min.js"></script> <script src="vendor/libtess.min.js"></script>
<script src="vendor/gl-matrix-min.js"></script> <script src="vendor/gl-matrix-min.js"></script>
<!-- vendor/proj4.js is kept for coordinate system support, not loaded yet --> <!-- vendor/proj4.js is kept for coordinate system support, not loaded yet -->
......
// CityGML parsing and mesh building, independent of the DOM and WebGL. // CityGML parsing and mesh building, independent of the DOM and WebGL.
import { sax, libtess, glMatrix } from '../vendor.js'; import { libtess, glMatrix } from '../vendor.js';
import { createTokenizer } from './xmlTokenizer.js';
var vec3 = glMatrix.vec3; var vec3 = glMatrix.vec3;
var axis = vec3.fromValues(19, 0.8, 1.5); var axis = vec3.fromValues(19, 0.8, 1.5);
vec3.normalize(axis, axis); vec3.normalize(axis, axis);
var options = {
trim: false,
normalize: false,
xmlns: false,
// sax only enforces its buffer limit (MAX_BUFFER_LENGTH) when it tracks
// the position; without it, an unterminated "<!" or comment in a broken
// or binary file is buffered until the tab runs out of memory.
position: true,
strictEntities: true
};
class BBox { class BBox {
constructor() { constructor() {
...@@ -70,34 +60,34 @@ var DEFAULT_COLOR = [1.0, 1.0, 1.0]; ...@@ -70,34 +60,34 @@ var DEFAULT_COLOR = [1.0, 1.0, 1.0];
function getColorForTagName(tagName) { function getColorForTagName(tagName) {
switch (tagName) { switch (tagName) {
case "GROUNDSURFACE": case "GroundSurface":
return [0.9411765, 0.9019608, 0.54901963]; return [0.9411765, 0.9019608, 0.54901963];
case "ROOFSURFACE": case "RoofSurface":
return [1.0, 0.0, 0.0]; return [1.0, 0.0, 0.0];
case "DOOR": case "Door":
return [1.0, 0.784313, 0.0]; return [1.0, 0.784313, 0.0];
case "WINDOW": case "Window":
return [0.0, 0.5019608, 0.5019608]; return [0.0, 0.5019608, 0.5019608];
case "WATERBODY": case "WaterBody":
return [0.5294118, 0.80784315, 0.98039216]; return [0.5294118, 0.80784315, 0.98039216];
case "BRIDGE": case "Bridge":
case "BRIDGEPART": case "BridgePart":
return [1, 0.49803922, 0.3137255]; return [1, 0.49803922, 0.3137255];
case "PLANTCOVER": case "PlantCover":
case "SOLITARYVEGETATIONOBJECT": case "SolitaryVegetationObject":
return [0.5647059, 0.93333334, 0.5647059]; return [0.5647059, 0.93333334, 0.5647059];
case "INTERSECTION": case "Intersection":
case "ROAD": case "Road":
case "RAILWAY": case "Railway":
case "SECTION": case "Section":
case "SQUARE": case "Square":
case "TRACK": case "Track":
case "TRANSPORTATIONCOMPLEX": case "TransportationComplex":
case "WATERWAY": case "Waterway":
return [0.4, 0.4, 0.4]; return [0.4, 0.4, 0.4];
case "LANDUSE": case "LandUse":
case "RELIEFFEATURE": case "ReliefFeature":
case "TINRELIEF": case "TINRelief":
// light brown (tan, #D2B48C) // light brown (tan, #D2B48C)
return [210 / 255, 180 / 255, 140 / 255]; return [210 / 255, 180 / 255, 140 / 255];
} }
...@@ -105,11 +95,10 @@ function getColorForTagName(tagName) { ...@@ -105,11 +95,10 @@ function getColorForTagName(tagName) {
} }
// Surface elements whose rings (exterior + interiors) form one planar polygon. // Surface elements whose rings (exterior + interiors) form one planar polygon.
var POLYGON_TAGS = new Set(["POLYGON", "POLYGONPATCH", "RECTANGLE", "TRIANGLE"]); var POLYGON_TAGS = new Set(["Polygon", "PolygonPatch", "Rectangle", "Triangle"]);
// lod0FootPrint, lod2MultiSurface, lod3Solid, ... (tag names are upper case // lod0FootPrint, lod2MultiSurface, lod3Solid, ... (local names, case-sensitive)
// because sax runs in non-strict mode) var LOD_TAG = /^lod([0-4])[A-Z]/;
var LOD_TAG = /^LOD([0-4])[A-Z]/;
// Default block size in triangles; a block holds 12 B position + 4 B color // Default block size in triangles; a block holds 12 B position + 4 B color
// per vertex, so 131072 triangles are about 6 MB. // per vertex, so 131072 triangles are about 6 MB.
...@@ -136,13 +125,12 @@ function createParser(opts) { ...@@ -136,13 +125,12 @@ function createParser(opts) {
var blockVertices = 3 * (opts.blockTriangles || DEFAULT_BLOCK_TRIANGLES); var blockVertices = 3 * (opts.blockTriangles || DEFAULT_BLOCK_TRIANGLES);
var sampleStride = opts.sampleStride || DEFAULT_SAMPLE_STRIDE; var sampleStride = opts.sampleStride || DEFAULT_SAMPLE_STRIDE;
var parser = sax.parser(false, options); var tokenizer = createTokenizer({ onOpen: onOpen, onClose: onClose, onText: onText });
// Colours of the open coloured elements, innermost last; closing one // Colours of the open coloured elements, innermost last; closing one
// restores the colour of the enclosing element (white if there is none). // restores the colour of the enclosing element (white if there is none).
var colorStack = []; var colorStack = [];
var color = DEFAULT_COLOR; var color = DEFAULT_COLOR;
var lod = null; var lod = null;
var readingRing = false;
var currentPolygon = []; var currentPolygon = [];
var coordinateString = ""; var coordinateString = "";
var bbox = new BBox(); var bbox = new BBox();
...@@ -156,14 +144,12 @@ function createParser(opts) { ...@@ -156,14 +144,12 @@ function createParser(opts) {
var samples = new Float32Array(3 * 1024); var samples = new Float32Array(3 * 1024);
var sampleCount = 0; var sampleCount = 0;
parser.ontext = function (t) { // The tokenizer reports text only while a LinearRing is read.
if (readingRing) { function onText(t) {
coordinateString += t; coordinateString += t;
} }
};
parser.onopentag = function (node) { function onOpen(tagName) {
var tagName = node.name.substring(node.name.indexOf(':') + 1);
var lodMatch = LOD_TAG.exec(tagName); var lodMatch = LOD_TAG.exec(tagName);
var tempColor = getColorForTagName(tagName); var tempColor = getColorForTagName(tagName);
if (tempColor !== undefined) { if (tempColor !== undefined) {
...@@ -173,22 +159,21 @@ function createParser(opts) { ...@@ -173,22 +159,21 @@ function createParser(opts) {
lod = Number(lodMatch[1]); lod = Number(lodMatch[1]);
} else if (POLYGON_TAGS.has(tagName)) { } else if (POLYGON_TAGS.has(tagName)) {
currentPolygon = []; currentPolygon = [];
} else if (tagName === "LINEARRING") { } else if (tagName === "LinearRing") {
readingRing = true; tokenizer.captureText = true;
coordinateString = ""; coordinateString = "";
} }
}; }
parser.onclosetag = function (node) { function onClose(tagName) {
var tagName = node.substring(node.indexOf(':') + 1); if (tagName === "LinearRing") {
if (tagName === "LINEARRING") {
var ring = parseRing(coordinateString); var ring = parseRing(coordinateString);
if (ring.length > 0) { if (ring.length > 0) {
currentPolygon.push(ring); currentPolygon.push(ring);
} }
readingRing = false; tokenizer.captureText = false;
coordinateString = ""; coordinateString = "";
} else if (tagName === "POS") { } else if (tagName === "pos") {
coordinateString += " "; coordinateString += " ";
} else if (POLYGON_TAGS.has(tagName)) { } else if (POLYGON_TAGS.has(tagName)) {
if (currentPolygon.length > 0) { if (currentPolygon.length > 0) {
...@@ -201,7 +186,7 @@ function createParser(opts) { ...@@ -201,7 +186,7 @@ function createParser(opts) {
colorStack.pop(); colorStack.pop();
color = colorStack.length > 0 ? colorStack[colorStack.length - 1] : DEFAULT_COLOR; color = colorStack.length > 0 ? colorStack[colorStack.length - 1] : DEFAULT_COLOR;
} }
}; }
// Parses whitespace separated x y z triples into [[x, y, z], ...], // Parses whitespace separated x y z triples into [[x, y, z], ...],
// relative to the origin. // relative to the origin.
...@@ -336,12 +321,12 @@ function createParser(opts) { ...@@ -336,12 +321,12 @@ function createParser(opts) {
return { return {
write: function (text) { write: function (text) {
parser.write(text); tokenizer.write(text);
}, },
flush: flush, flush: flush,
summary: summary, summary: summary,
close: function () { close: function () {
parser.close(); tokenizer.close();
flush(); flush();
return summary(); return summary();
} }
......
// Minimal streaming XML tokenizer for the CityGML parser. It reports open and
// close tags with their local name (prefix removed, case preserved) and, only
// while captureText is true, the text between tags. Attributes and entities
// are not decoded: the parser needs element names and coordinate text only.
//
// Markup is found with native indexOf instead of a per-character state
// machine, which makes it several times faster than a general XML parser.
// Markup that is cut off at the end of a chunk is kept until the next write.
// Unterminated markup (e.g. "<!" in a binary file) is kept at most this long,
// so broken input fails instead of being buffered until memory runs out.
const MAX_PENDING = 64 * 1024;
const LT = 60, GT = 62, SLASH = 47, BANG = 33, QUESTION = 63, COLON = 58;
const DOUBLE_QUOTE = 34, SINGLE_QUOTE = 39, OPEN_BRACKET = 91, CLOSE_BRACKET = 93;
function isNameEnd(c) {
return c === 32 || c === 9 || c === 10 || c === 13 || c === SLASH || c === GT;
}
// Index of the '>' that ends the tag whose name starts at `from`, or -1.
// '>' inside quoted attribute values does not count.
function tagEnd(s, from) {
const gt = s.indexOf('>', from);
if (gt < 0) {
return -1;
}
// Fast path: no quote before the first '>'. Only the tag itself is
// scanned; searching the rest of the chunk would be quadratic.
let quoted = false;
for (let i = from; i < gt; i++) {
const c = s.charCodeAt(i);
if (c === DOUBLE_QUOTE || c === SINGLE_QUOTE) {
quoted = true;
break;
}
}
if (!quoted) {
return gt;
}
let quote = 0;
for (let i = from; i < s.length; i++) {
const c = s.charCodeAt(i);
if (quote !== 0) {
if (c === quote) {
quote = 0;
}
} else if (c === DOUBLE_QUOTE || c === SINGLE_QUOTE) {
quote = c;
} else if (c === GT) {
return i;
}
}
return -1;
}
// End (exclusive) of a declaration such as <!DOCTYPE ... [ internal subset ]>, or -1.
function declarationEnd(s, from) {
let quote = 0;
let depth = 0;
for (let i = from; i < s.length; i++) {
const c = s.charCodeAt(i);
if (quote !== 0) {
if (c === quote) {
quote = 0;
}
} else if (c === DOUBLE_QUOTE || c === SINGLE_QUOTE) {
quote = c;
} else if (c === OPEN_BRACKET) {
depth++;
} else if (c === CLOSE_BRACKET) {
depth--;
} else if (c === GT && depth <= 0) {
return i + 1;
}
}
return -1;
}
// Local name of the tag whose qualified name starts at `start`.
function localName(s, start, end) {
let nameEnd = start;
while (nameEnd < end && !isNameEnd(s.charCodeAt(nameEnd))) {
nameEnd++;
}
let nameStart = start;
for (let i = nameEnd - 1; i >= start; i--) {
if (s.charCodeAt(i) === COLON) {
nameStart = i + 1;
break;
}
}
return s.slice(nameStart, nameEnd);
}
// handlers: onOpen(localName), onClose(localName), onText(text).
// Returns { captureText, write(text), close() }.
export function createTokenizer({ onOpen, onClose, onText }) {
let pending = '';
const tokenizer = {
captureText: false,
write(chunk) {
const s = pending.length > 0 ? pending + chunk : chunk;
pending = '';
let i = 0;
while (i < s.length) {
const lt = s.indexOf('<', i);
if (lt < 0) {
if (tokenizer.captureText) {
onText(s.slice(i));
}
return;
}
if (lt > i && tokenizer.captureText) {
onText(s.slice(i, lt));
}
const next = markup(s, lt);
if (next < 0) {
keep(s.slice(lt));
return;
}
i = next;
}
},
// Incomplete markup at the end of the input is dropped: a truncated file
// shows what was read.
close() {
pending = '';
},
};
function keep(rest) {
if (rest.length > MAX_PENDING) {
throw new Error('Max buffer length exceeded: unterminated markup');
}
pending = rest;
}
// Handles the markup starting at s[lt] === '<'. Returns the index after it,
// or -1 if it is not complete in s.
function markup(s, lt) {
if (lt + 1 >= s.length) {
return -1;
}
const c1 = s.charCodeAt(lt + 1);
if (c1 === BANG) {
const head = s.slice(lt, lt + 9);
if (head.length < 9 && ('<![CDATA['.startsWith(head) || '<!--'.startsWith(head))) {
return -1; // cannot tell comment, CDATA and declaration apart yet
}
if (head.startsWith('<!--')) {
const end = s.indexOf('-->', lt + 4);
return end < 0 ? -1 : end + 3;
}
if (head === '<![CDATA[') {
const end = s.indexOf(']]>', lt + 9);
if (end < 0) {
return -1;
}
if (tokenizer.captureText) {
onText(s.slice(lt + 9, end));
}
return end + 3;
}
return declarationEnd(s, lt + 2);
}
if (c1 === QUESTION) {
const end = s.indexOf('?>', lt + 2);
return end < 0 ? -1 : end + 2;
}
const gt = tagEnd(s, lt + 1);
if (gt < 0) {
return -1;
}
if (c1 === SLASH) {
onClose(localName(s, lt + 2, gt));
} else {
const name = localName(s, lt + 1, gt);
onOpen(name);
if (s.charCodeAt(gt - 1) === SLASH) {
onClose(name);
}
}
return gt + 1;
}
return tokenizer;
}
// The vendor libraries are classic scripts (see index.html) that define // The vendor libraries are classic scripts (see index.html) that define
// globals; this is the only module that reads them. // globals; this is the only module that reads them.
const g = globalThis; const g = globalThis;
if (!g.sax || !g.libtess || !g.glMatrix) { if (!g.libtess || !g.glMatrix) {
throw new Error('sax.js, libtess.min.js and gl-matrix-min.js must be loaded before the modules'); throw new Error('libtess.min.js and gl-matrix-min.js must be loaded before the modules');
} }
export const sax = g.sax;
export const libtess = g.libtess; export const libtess = g.libtess;
export const glMatrix = g.glMatrix; export const glMatrix = g.glMatrix;
export const { mat4, vec2, vec3, vec4 } = g.glMatrix; export const { mat4, vec2, vec3, vec4 } = g.glMatrix;
This diff is collapsed.
...@@ -295,7 +295,7 @@ test('an XML file with UTF-8 byte order mark and leading whitespace is accepted' ...@@ -295,7 +295,7 @@ test('an XML file with UTF-8 byte order mark and leading whitespace is accepted'
}); });
test('an unterminated declaration fails instead of buffering without limit', async () => { test('an unterminated declaration fails instead of buffering without limit', async () => {
// Binary data parsed as XML can open a "<!" that never closes; sax must not // Binary data parsed as XML can open a "<!" that never closes; the parser must not
// buffer the rest of the file (this crashed the tab for a 250 MB .7z). // buffer the rest of the file (this crashed the tab for a 250 MB .7z).
const xml = cityModel('<!' + 'x'.repeat(300 * 1024)); const xml = cityModel('<!' + 'x'.repeat(300 * 1024));
await assert.rejects(CityGML.readFile(new Blob([xml]), CityGML.createParser(), { chunkSize: 64 * 1024 }), await assert.rejects(CityGML.readFile(new Blob([xml]), CityGML.createParser(), { chunkSize: 64 * 1024 }),
......
// Loads the vendor libraries the way the browser does: as classic scripts // Loads the vendor libraries the way the browser does: as classic scripts
// that define globals (sax, libtess, glMatrix). Runs before any test module. // that define globals (libtess, glMatrix). Runs before any test module.
import fs from 'node:fs'; import fs from 'node:fs';
import vm from 'node:vm'; import vm from 'node:vm';
const vendor = new URL('../public/vendor/', import.meta.url); const vendor = new URL('../public/vendor/', import.meta.url);
for (const name of ['sax.js', 'libtess.min.js', 'gl-matrix-min.js']) { for (const name of ['libtess.min.js', 'gl-matrix-min.js']) {
vm.runInThisContext(fs.readFileSync(new URL(name, vendor), 'utf8'), { filename: name }); vm.runInThisContext(fs.readFileSync(new URL(name, vendor), 'utf8'), { filename: name });
} }
import test from 'node:test';
import assert from 'node:assert/strict';
import { createTokenizer } from '../public/src/loading/xmlTokenizer.js';
// Runs the chunks through a tokenizer and returns the events as strings:
// "<name", "</name", and "text:..." (adjacent text pieces merged). Text is
// captured only inside elements named in captureIn, like the parser does.
function events(chunks, captureIn = []) {
const out = [];
const t = createTokenizer({
onOpen(name) {
out.push('<' + name);
if (captureIn.includes(name)) t.captureText = true;
},
onClose(name) {
out.push('</' + name);
if (captureIn.includes(name)) t.captureText = false;
},
onText(text) {
const last = out.length - 1;
if (last >= 0 && out[last].startsWith('text:')) out[last] += text;
else out.push('text:' + text);
},
});
for (const c of chunks) t.write(c);
t.close();
return out;
}
test('open and close tags are reported with their local name, case preserved', () => {
assert.deepEqual(events(['<core:CityModel><bldg:lod2Solid></bldg:lod2Solid></core:CityModel>']),
['<CityModel', '<lod2Solid', '</lod2Solid', '</CityModel']);
});
test('attributes and whitespace inside tags do not change the name', () => {
assert.deepEqual(events(['<gml:Polygon gml:id="p1"\n srsDimension="3"></gml:Polygon >']),
['<Polygon', '</Polygon']);
});
test('a self-closing tag reports open and close', () => {
assert.deepEqual(events(['<a><gml:pos/><b x="1" /></a>']), ['<a', '<pos', '</pos', '<b', '</b', '</a']);
});
test('a ">" inside a quoted attribute value does not end the tag', () => {
assert.deepEqual(events(['<a title="x > y" other=\'>\'><b/></a>']), ['<a', '<b', '</b', '</a']);
});
test('text is reported only while capturing', () => {
assert.deepEqual(events(['<a> outside <posList>1 2 3</posList> after </a>'], ['posList']),
['<a', '<posList', 'text:1 2 3', '</posList', '</a']);
});
test('comments, processing instructions and DOCTYPE are skipped', () => {
const xml = '<?xml version="1.0"?><!DOCTYPE CityModel [ <!ENTITY e "<x>"> ]>' +
'<!-- <notATag> --><a><?pi data?><b/></a>';
assert.deepEqual(events([xml]), ['<a', '<b', '</b', '</a']);
});
test('CDATA content is reported as text while capturing', () => {
assert.deepEqual(events(['<posList><![CDATA[1 2 <3>]]></posList>'], ['posList']),
['<posList', 'text:1 2 <3>', '</posList']);
});
test('splitting the input at any position gives the same events', () => {
const xml = '<?xml version="1.0"?><!-- c --><core:CityModel a="1>2">' +
'<gml:LinearRing><gml:posList>0 0 0 1 0 0</gml:posList><gml:pos/></gml:LinearRing>' +
'<![CDATA[x]]><!DOCTYPE d [<!ELEMENT e ANY>]></core:CityModel>';
const expected = events([xml], ['posList']);
for (let i = 1; i < xml.length; i++) {
assert.deepEqual(events([xml.slice(0, i), xml.slice(i)], ['posList']), expected, `split at ${i}`);
}
});
test('unterminated markup longer than the buffer limit fails', () => {
const t = createTokenizer({ onOpen() {}, onClose() {}, onText() {} });
assert.throws(() => {
t.write('<a><!');
for (let i = 0; i < 10; i++) t.write('x'.repeat(10000));
}, /Max buffer length exceeded/);
});
test('incomplete markup at the end of the input is ignored', () => {
// A truncated file shows what was read instead of failing.
assert.deepEqual(events(['<a><b></b><c attr="unfinished']), ['<a', '<b', '</b']);
});
test('many tags in one chunk with a quote only at the end are processed in linear time', () => {
// Guards against searching the rest of the chunk for quotes at every tag.
const xml = '<a>' + '<b></b>'.repeat(200000) + '<c q="1"/></a>';
const start = performance.now();
const out = events([xml]);
assert.equal(out.length, 2 + 400000 + 2);
assert.ok(performance.now() - start < 3000, `took ${(performance.now() - start).toFixed(0)} ms`);
});
Supports Markdown
0% or .
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment