Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
61 commits
Select commit Hold shift + click to select a range
59b11b0
WIP.Search.md: site search design, measurements and rollout
KubaO Sep 26, 2026
32ebd43
builder: site search drops asterisk-only tokens instead of crashing
KubaO Sep 26, 2026
52f04ad
eval: search_quality.mjs measures site search ranking against a baseline
KubaO Sep 26, 2026
1911c90
builder: site search indexes ### headings and folds generic sections
KubaO Sep 26, 2026
cd13d89
builder: site search finds symbols by name and by Class.Member
KubaO Sep 26, 2026
9816dfe
builder: site search keeps stop words and builds its index on demand
KubaO Sep 26, 2026
00bf6cc
WIP.Search.md: reader-intent findings, next steps, and how to resume
KubaO Sep 26, 2026
384032b
eval: search_quality.mjs judges bare names by reader intent
KubaO Sep 26, 2026
2afb78f
builder: site search ranks exact names, page titles and whole phrases…
KubaO Sep 26, 2026
dae1d32
WIP.Search.md: record the intent step's commit
KubaO Sep 26, 2026
79bf488
builder: site search ranks types and language elements first among sa…
KubaO Sep 26, 2026
f44d9d8
WIP.Search.md: record the tier round's commit
KubaO Sep 26, 2026
05f24d7
WIP.Search.md: resume section covers the index pilot and the user's d…
KubaO Sep 26, 2026
0b14b03
eval: prose queries can expect pages right behind; two prose expectat…
KubaO Sep 26, 2026
5680f57
builder: hand-marked index entries for the site search, and the first…
KubaO Sep 26, 2026
7ba097a
WIP.Search.md: record the index pilot's commit
KubaO Sep 26, 2026
0f60a13
builder: site search finds the member a qualified name names
KubaO Sep 27, 2026
5e14d5b
WIP.Search.md: record the qualified-name round's commit
KubaO Sep 27, 2026
ada51d3
WIP.Search.md: resume notes cover the multi-word check and two lunr l…
KubaO Sep 27, 2026
cf93c1e
builder: only a page's first heading can be its search title
KubaO Sep 27, 2026
cc8607f
builder: site search tells apart qualified names that stem alike
KubaO Sep 27, 2026
fca4678
WIP.Search.md: record the stem-twin round's commit
KubaO Sep 27, 2026
434277f
eval: a section of a symbol's page counts for that symbol
KubaO Sep 27, 2026
93e4547
WIP.Search.md: record the same-page round's commit
KubaO Sep 27, 2026
78d2d76
WIP.Search.md: commit hashes after the rebase onto f8e630e5
KubaO Sep 27, 2026
9fbff4a
builder: lunr's token set no longer invents words, so `a` can be sear…
KubaO Sep 27, 2026
5a8c225
WIP.Search.md: record the lunr token-set round's commit
KubaO Sep 27, 2026
07fdb2e
WIP.Search.md: the New Functions and Form events probes, diagnosed an…
KubaO Sep 27, 2026
cc9ff73
WIP.Search.md: the user's decisions on the probes, and where to resume
KubaO Sep 27, 2026
7642ae7
eval: page titles and page-plus-section queries as ground truth
KubaO Sep 27, 2026
b95a119
search: a query naming a whole title scores three times as much
KubaO Sep 27, 2026
63f7695
search: a plural kind word doesn't make the other word a name
KubaO Sep 27, 2026
18cb441
WIP.Search.md: the whole-title round, shipped and measured
KubaO Sep 27, 2026
8aa5652
WIP.Search.md: which HtmlElement section ranks first, and why
KubaO Sep 27, 2026
2f2e575
search: decode HTML entities per token, so `&H80004005` is found
KubaO Sep 27, 2026
f4f0341
WIP.Search.md: entities decoded in the index; stopped for the user's …
KubaO Sep 27, 2026
d622fd0
WIP.Search.md: remove the header pasted into the resume section
KubaO Sep 27, 2026
0ddace9
search: lunr's set unions add in place, so short words aren't quadratic
KubaO Sep 27, 2026
be1d33a
WIP.Search.md: slow multi-word queries, profiled and fixed
KubaO Sep 27, 2026
c86e957
search: a kind word is required only while an entry found names the t…
KubaO Sep 27, 2026
187e07f
WIP.Search.md: kind words fixed; item 6 candidates drafted for approval
KubaO Sep 27, 2026
f56979e
WIP.Search.md: the Glossary counts; array and delegates entries measured
KubaO Sep 27, 2026
4d829f7
eval: name-and-kind queries as ground truth intent-5
KubaO Sep 27, 2026
0c15363
WIP.Search.md: the name-and-kind set is ground truth intent-5
KubaO Sep 27, 2026
f946a3a
WIP.Search.md: the Glossary counts within the top 5; array needs no e…
KubaO Sep 27, 2026
4984162
eval: item 6's approved prose queries, before their entries
KubaO Sep 27, 2026
185f5d1
docs: index entries for item 6's recommended terms
KubaO Sep 27, 2026
c2e127c
eval: item 6's decided prose queries, before their entries
KubaO Sep 27, 2026
f25d32e
docs: index entries for the user's choices; a Glossary entry for name…
KubaO Sep 27, 2026
e5d6e17
WIP.Search.md: item 6 shipped as approved; what still waits
KubaO Sep 27, 2026
10885a3
eval: standard exe and create [an] ActiveX DLL, before their entries
KubaO Sep 27, 2026
44ca8e8
docs: standard exe and ActiveX DLL to New Project; pointers, secondary
KubaO Sep 27, 2026
a0ebb3d
WIP.Search.md: resume section rewritten for a new session
KubaO Sep 27, 2026
31aa9a2
eval: 41 prose queries from the gap and new-term surveys, before thei…
KubaO Sep 27, 2026
518ebd0
docs: ByRef/ByVal, optional and named arguments, more than one result
KubaO Sep 27, 2026
087ea1e
docs: index entries for the terms the new text still can't find
KubaO Sep 27, 2026
01bac5a
WIP.Search.md: item 7 recorded; baseline saved with its 41 queries
KubaO Sep 27, 2026
6f79988
eval: search_quality.mjs takes the repository root from lib/repo-path…
KubaO Sep 27, 2026
3eb74ae
eval: `default property` also accepts the Glossary's definition
KubaO Sep 27, 2026
7cab80e
docs: Glossary *default member*, also called the default property
KubaO Sep 27, 2026
086e467
WIP.Search.md: the rebase onto ea96e4c2, default property settled, ar…
KubaO Sep 27, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1,976 changes: 1,976 additions & 0 deletions WIP.Search.md

Large diffs are not rendered by default.

3 changes: 2 additions & 1 deletion builder/cpu-worker.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -244,7 +244,8 @@ const handlers = {
// global indices during consolidation).
const searchEntries = deriveSearchEntries(chunk, env.site)
.map(e => ({ doc: e.doc, title: e.title, content: e.content,
url: e.url, relUrl: e.relUrl }));
url: e.url, relUrl: e.relUrl,
index: e.index, index_also: e.index_also }));

return {
pages: chunk.map(p => ({
Expand Down
182 changes: 132 additions & 50 deletions builder/offline.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -281,62 +281,144 @@ const JTD_NAVLINK_REPLACEMENT = `function navLink() {
}`;

const JTD_INITSEARCH_FN_REPLACEMENT = `function initSearch() {
// Patched by _plugins/offlinify.rb for file:// compatibility.
// The upstream version fires XMLHttpRequest for search-data.json,
// which browsers block under file://. We instead read the index
// from a global the offline copy preloads via <script src=>.
var docs = window.SEARCH_DATA;
if (!docs) {
console.log('Offlinify: window.SEARCH_DATA not found; ensure search-data.js loads before just-the-docs.js');
return;
}
// Rebuild each doc.url from doc.relUrl (no baseurl prefix) so
// search-result clicks land on the right file regardless of
// whatever baseurl the site was built with. Upstream sets
// \`link.href = doc.url\`, so this is the value users navigate
// to.
var siteRoot = window.OFFLINE_SITE_ROOT || '';
for (var i in docs) {
var rel = docs[i].relUrl;
if (typeof rel === 'string' && rel.charAt(0) === '/') {
var hash = '';
var hashIdx = rel.indexOf('#');
if (hashIdx !== -1) {
hash = rel.slice(hashIdx);
rel = rel.slice(0, hashIdx);
// Patched by _plugins/offlinify.rb for file:// compatibility, and for
// the lazy index build (WIP.Search.md rollout step 5C): initSearch()
// only defines how to build the index -- loadIndex() below -- and hands
// that to searchLoaded(), which wires up the search box right away but
// doesn't call loadIndex() until the first keystroke. window.SEARCH_DATA
// is already sitting in memory (preloaded via <script src=> before this
// file runs), so there's no fetch to defer here, only the build itself
// -- the same ~1.3s / 240MB-heap cost the online copy defers, paid only
// by readers who open search.
function loadIndex(onSuccess, onError) {
try {
// The upstream version fires XMLHttpRequest for search-data.json,
// which browsers block under file://. We instead read the index
// from a global the offline copy preloads via <script src=>.
var docs = window.SEARCH_DATA;
if (!docs) {
console.log('Offlinify: window.SEARCH_DATA not found; ensure search-data.js loads before just-the-docs.js');
onError();
return;
}
rel = rel.slice(1); // strip leading /
if (rel.endsWith('/')) {
rel = rel + 'index.html';
} else {
var lastSlash = rel.lastIndexOf('/');
var lastSeg = lastSlash === -1 ? rel : rel.slice(lastSlash + 1);
if (lastSeg.indexOf('.') === -1) rel = rel + '.html';
// Rebuild each doc.url from doc.relUrl (no baseurl prefix) so
// search-result clicks land on the right file regardless of
// whatever baseurl the site was built with. Upstream sets
// \`link.href = doc.url\`, so this is the value users navigate
// to. Reading from doc.relUrl each time (not doc.url) keeps this
// idempotent, since a failed load can retry and run it again.
var siteRoot = window.OFFLINE_SITE_ROOT || '';
for (var i in docs) {
var rel = docs[i].relUrl;
if (typeof rel === 'string' && rel.charAt(0) === '/') {
var hash = '';
var hashIdx = rel.indexOf('#');
if (hashIdx !== -1) {
hash = rel.slice(hashIdx);
rel = rel.slice(0, hashIdx);
}
rel = rel.slice(1); // strip leading /
if (rel.endsWith('/')) {
rel = rel + 'index.html';
} else {
var lastSlash = rel.lastIndexOf('/');
var lastSeg = lastSlash === -1 ? rel : rel.slice(lastSlash + 1);
if (lastSeg.indexOf('.') === -1) rel = rel + '.html';
}
docs[i].url = siteRoot + rel + hash;
}
}
docs[i].url = siteRoot + rel + hash;
}
}

lunr.tokenizer.separator = /[\\s\\-\\/]+/;

var index = lunr(function(){
this.ref('id');
this.field('title', { boost: 200 });
this.field('content', { boost: 2 });
this.field('relUrl');
this.metadataWhitelist = ['position'];

for (var i in docs) {
this.add({
id: i,
title: docs[i].title,
content: docs[i].content,
relUrl: docs[i].relUrl
// Mirrors the online build's dot-run-split patch (see
// builder/vendor/just-the-docs/README.md and WIP.Search.md's
// "Design" section): wrap lunr.tokenizer once so a run of 2+ dots
// ("Do...Loop", "For Each...Next") tokenises as spaces instead of
// one opaque token, for both the index and the query. Installed
// once, since lunr is a global singleton and loadIndex() can run
// again after a failed load; the wrapper carries \`separator\`
// itself because the original tokenizer reads
// \`lunr.tokenizer.separator\` at call time, which after this
// reassignment resolves to the wrapper's own property. Each token's
// HTML entities are decoded after the split, by the online copy's
// decodeTokenEntities().
if (!lunr.tokenizer.dotRunSplit) {
var originalTokenizer = lunr.tokenizer;
var dotRunSplitTokenizer = function (input) {
if (typeof input === 'string') {
input = input.replace(/\\.{2,}/g, function (m) {
return new Array(m.length + 1).join(' ');
});
}
return originalTokenizer(input).map(decodeTokenEntities);
};
dotRunSplitTokenizer.dotRunSplit = true;
dotRunSplitTokenizer.separator = /[\\s\\-\\/]+/;
lunr.tokenizer = dotRunSplitTokenizer;
}
// Mirrors the online build's token-set key patch
// (separateTokenSetKeys() is the online copy's), and its set-union
// patch (accumulateSetUnions(), likewise).
separateTokenSetKeys();
accumulateSetUnions();

// Mirrors the online build's stem-twin patch (stemTwins() and
// qualifiedField() are the online copy's).
var twins = stemTwins(docs);
var index = lunr(function(){
this.ref('id');
this.field('title', { boost: 200 });
this.field('content', { boost: 2 });
// Mirrors the online build's just-the-docs.js patch (see
// builder/vendor/just-the-docs/README.md and WIP.Search.md's
// "Design" §2): the same two symbol-index fields, at the same
// boosts. test/search.test.mjs extracts the field/boost list from
// both this string and the vendored just-the-docs.js source and
// asserts they agree, so the two copies cannot drift apart silently.
this.field('names', { boost: 100 });
this.field('qualified', { boost: 500 });
// Mirrors the online build's exact-name, primary-name and
// page-title patches (exactName() is the online copy's, which this
// file keeps).
this.field('exact', { boost: 50 });
this.field('primary', { boost: 1000 });
this.field('page', { boost: 5 });
// Mirrors the online build's index-term patch (indexField(),
// indexedContent() and pinIndexFieldLengths() are the online
// copy's).
this.field('index', { boost: 1000 });
this.field('relUrl');
this.metadataWhitelist = ['position'];
pinIndexFieldLengths(this);
// Mirrors the online build's stop-word patch (step 5A): keep
// lunr's default stop words in the index, since many are
// twinBASIC keywords (Do, For, If, Is, On, With, Each...) that the
// search pipeline's query side never dropped.
this.pipeline.remove(lunr.stopWordFilter);

for (var i in docs) {
this.add({
id: i,
title: docs[i].title,
content: indexedContent(docs[i]),
names: docs[i].names || '',
qualified: qualifiedField(docs[i], twins),
exact: (docs[i].names || '').split(/\\s+/).filter(Boolean).map(exactName).join(' '),
primary: (docs[i].primary || '').split(/\\s+/).filter(Boolean).map(exactName).join(' '),
page: docs[i].doc || '',
index: indexField(docs[i]),
relUrl: docs[i].relUrl
});
}
});

onSuccess(index, docs);
} catch (e) {
console.log('Error building search index: ' + e);
onError();
}
});
}

searchLoaded(index, docs);
searchLoaded(loadIndex);
}`;

// §6.9 patchJustTheDocsJs -- regex-substitute navLink() + initSearch().
Expand Down
49 changes: 48 additions & 1 deletion builder/render.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -59,7 +59,11 @@ function renderPage(page, md) {
return page.rawContent;
}
const source = applyPreRenderRewrites(page.rawContent, md);
let html = md.render(source, { page });
const env = { page };
let html = md.render(source, env);
// Lifted off the headings by searchIndexMarksPlugin; search.mjs reads
// them when it splits this page into entries.
if (env.searchIndexMarks) page.searchIndexMarks = env.searchIndexMarks;
html = normaliseVoidTags(html);
html = padEmptyCells(html);
return html;
Expand Down Expand Up @@ -433,6 +437,7 @@ export function createMarkdownIt(ctx) {
md.use(footnote);
configureFootnotes(md);
md.use(headerIdPlugin);
md.use(searchIndexMarksPlugin);
md.use(headingLevelNormalizePlugin);
md.use(tocPlugin);
md.use(relativeLinksPlugin, ctx);
Expand Down Expand Up @@ -1249,6 +1254,48 @@ function headerIdPlugin(md) {
});
}

// Hand-marked search index entries (WIP.Search.md, "What shipped, third
// round: the index pilot"). A heading's `{: index="..." }` names the terms
// a reader looks up to find that section, as a book's index would, and
// `{: index_also="..." }` the terms it is a strong second answer for. Both
// are lifted off the heading here, once header-id has given it its id, into
// env.searchIndexMarks as `{ id, index, index_also }` with the raw
// attribute values, so they never reach the HTML. search.mjs parses the
// values and attaches them to the entry holding that heading. On any other
// element the attribute would be published and do nothing, so it fails the
// build instead.
const SEARCH_INDEX_ATTRS = ["index", "index_also"];

function searchIndexMarksPlugin(md) {
md.core.ruler.after("header-id", "search-index-marks", (state) => {
const misplaced = (t) => {
const name = SEARCH_INDEX_ATTRS.find((a) => t.attrGet(a) !== null);
if (!name) return;
throw new Error(
`${state.env?.page?.srcRel ?? "(unknown page)"}: \`{: ${name}="..." }\` on ` +
`${t.tag ? `<${t.tag}>` : t.type}; only a heading can carry a search index entry. ` +
"For the whole page, use `index:` in the front matter.");
};
for (const t of state.tokens) {
if (t.type === "heading_open") {
const mark = { id: t.attrGet("id") };
let found = false;
for (const a of SEARCH_INDEX_ATTRS) {
const v = t.attrGet(a);
if (v === null) continue;
mark[a] = v;
found = true;
t.attrs.splice(t.attrIndex(a), 1);
}
if (found) (state.env.searchIndexMarks ??= []).push(mark);
} else if (t.attrs) {
misplaced(t);
}
for (const c of t.children ?? []) if (c.attrs) misplaced(c);
}
});
}

function headingText(children) {
// Concatenate the visible text content from inline tokens. Skip markup
// wrappers (em, strong, link openers) and pick up text + code spans --
Expand Down
Loading
Loading