Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
125 changes: 110 additions & 15 deletions generate-adaptors/index.js
Original file line number Diff line number Diff line change
Expand Up @@ -87,30 +87,119 @@ function pushToPaths(name) {
});
}

const CODE_BLOCK_REGEX = /(```[\s\S]*?```|`[^`]*`)/g;

// A link whose target is a bare identifier, eg [http.get](http.get)
const BARE_LINK_REGEX =
/\[([^\]]+)\]\((?!https?:|\/|#|\.)([A-Za-z_$][\w$]*(?:\.[\w$]+)*)\)/g;

// Even indices are prose, odd indices are code
function splitCodeBlocks(content) {
return content.split(CODE_BLOCK_REGEX);
}

// Blank out code while preserving newlines, so a scan of the masked copy keeps
// `^` anchored to real line starts - splitting alone does not guarantee that
function maskCodeBlocks(content) {
return content.replace(CODE_BLOCK_REGEX, match =>
match.replace(/[^\n]/g, ' ')
);
}

// Approximates the id Docusaurus derives from a heading with no explicit {#id}
function slugify(text) {
return text
.trim()
.toLowerCase()
.replace(/[^\w\s-]/g, '')
.trim()
.replace(/\s+/g, '-');
}

// Plain headings count too - `### get` really does render as id="get". Braces
// arrive escaped or not depending on whether escapeMdx has run, so allow both.
function collectAnchors(content) {
const masked = maskCodeBlocks(content);
const anchors = new Set();
let match;

const explicitRegex = /\\?\{#([\w-]+)\\?\}/g;
while ((match = explicitRegex.exec(masked))) {
anchors.add(match[1]);
}

const headingRegex = /^#{1,6}\s+(.+)$/gm;
while ((match = headingRegex.exec(masked))) {
const heading = match[1].trim();
if (!/\\?\{#[\w-]+\\?\}/.test(heading)) {
anchors.add(slugify(heading));
}
}

return anchors;
}

// Escape characters that MDX would parse as JSX outside of code blocks
function escapeMdx(content) {
// Split content by code blocks (both inline ` and multi-line ```)
const codeBlockRegex = /(```[\s\S]*?```|`[^`]*`)/g;
const parts = content.split(codeBlockRegex);

// Escape only in non-code parts (odd indices are code blocks)
for (let i = 0; i < parts.length; i++) {
if (i % 2 === 0) {
parts[i] = parts[i]
.replace(/{/g, '\\{')
.replace(/}/g, '\\}')
// Escape < unless it can start a JSX tag (letter, /, $, _) or an
// HTML comment (<!--)
.replace(/<(?![A-Za-z/$_!])/g, '\\<');
}
const parts = splitCodeBlocks(content);

for (let i = 0; i < parts.length; i += 2) {
parts[i] = parts[i]
.replace(/{/g, '\\{')
.replace(/}/g, '\\}')
// Escape < unless it can start a JSX tag (letter, /, $, _) or an
// HTML comment (<!--)
.replace(/<(?![A-Za-z/$_!])/g, '\\<');
}

return parts.join('');
}

// JSDoc emits bare references like [http.get](http.get), which Docusaurus
// resolves as relative paths and fails the build on. Rewrite them as anchors.
function fixFunctionLinks(content, name = 'adaptor') {
const anchors = collectAnchors(content);
const parts = splitCodeBlocks(content);

for (let i = 0; i < parts.length; i += 2) {
parts[i] = parts[i].replace(BARE_LINK_REGEX, (_match, text, target) => {
// namespaced functions get an explicit {#http_get}, plain ones a slug
const candidates = [target.replace(/\./g, '_'), slugify(target)];
const id = candidates.find(candidate => anchors.has(candidate));

if (id) {
return `[${text}](#${id})`;
}

console.warn(
` ! ${name}: no anchor for [${text}](${target}), dropping the link`
);
return text;
});
}

return parts.join('');
}

// An unbalanced backtick can throw off the code block split and let a link
// through. Warn here rather than fail the build later with an opaque error.
function warnOnBareLinks(content, name) {
const survivors = maskCodeBlocks(content).match(BARE_LINK_REGEX);

if (survivors) {
console.warn(
` ! ${name}: ${survivors.length} unresolved link(s) will break the build: ${survivors.join(', ')}`
);
}
}

function generateJsDoc(a) {
// Add line break before </dt> tags and escape MDX specials outside code blocks
const docsContent = escapeMdx(JSON.parse(a.docs).replace(/<\/dt>/g, '\n</dt>'));
const docsContent = escapeMdx(
fixFunctionLinks(JSON.parse(a.docs).replace(/<\/dt>/g, '\n</dt>'), a.name)
);

warnOnBareLinks(docsContent, a.name);

return `---
title: ${a.name}@${a.version}
Expand Down Expand Up @@ -345,3 +434,9 @@ Make sure OPENFN_ADAPTORS_REPO is set in your env`);
},
};
};

// exported for unit tests
module.exports.escapeMdx = escapeMdx;
module.exports.collectAnchors = collectAnchors;
module.exports.fixFunctionLinks = fixFunctionLinks;
module.exports.slugify = slugify;
147 changes: 147 additions & 0 deletions generate-adaptors/index.test.js
Original file line number Diff line number Diff line change
@@ -0,0 +1,147 @@
const test = require('node:test');
const assert = require('node:assert');

const {
escapeMdx,
collectAnchors,
fixFunctionLinks,
slugify,
} = require('./index.js');

// Silence the drop warnings, and let tests assert on them
function captureWarnings(fn) {
const original = console.warn;
const warnings = [];
console.warn = message => warnings.push(message);
try {
return { result: fn(), warnings };
} finally {
console.warn = original;
}
}

const fix = (content, name = 'test') =>
captureWarnings(() => fixFunctionLinks(content, name)).result;

test('slugify matches the ids docusaurus derives from headings', () => {
assert.equal(slugify('get'), 'get');
assert.equal(slugify(' Submit Birth Notification '), 'submit-birth-notification');
assert.equal(slugify('http.get'), 'httpget');
});

test('collectAnchors finds explicit {#id} anchors', () => {
const anchors = collectAnchors('### http.get {#http_get}');
assert.ok(anchors.has('http_get'));
});

test('collectAnchors finds ids docusaurus generates for plain headings', () => {
const anchors = collectAnchors('### get\n\nsome prose\n\n## Post Request');
assert.ok(anchors.has('get'));
assert.ok(anchors.has('post-request'));
});

test('collectAnchors ignores headings inside code blocks', () => {
const anchors = collectAnchors('```sh\n# not-a-heading\n```\n\n### real');
assert.ok(anchors.has('real'));
assert.ok(!anchors.has('not-a-heading'));
});

test('rewrites a namespaced function link to its explicit anchor', () => {
const content = 'Use [http.get](http.get) instead.\n\n### http.get {#http_get}';
assert.match(fix(content), /\[http\.get\]\(#http_get\)/);
});

// regression: `### get` renders as id="get", so this link works and must survive
test('keeps a link that targets a plain heading', () => {
const content = 'See [get](get) for details.\n\n### get\n\nMake a GET request.';
assert.match(fix(content), /\[get\]\(#get\)/);
});

test('drops a link with no matching anchor, and warns', () => {
const content = 'See [missing](missing) for details.\n\n### get';
const { result, warnings } = captureWarnings(() =>
fixFunctionLinks(content, 'commcare')
);

assert.match(result, /See missing for details\./);
assert.equal(warnings.length, 1);
assert.match(warnings[0], /commcare/);
assert.match(warnings[0], /\[missing\]\(missing\)/);
});

test('leaves links that already resolve untouched', () => {
const untouched = [
'[docs](https://openfn.org)',
'[docs](http://openfn.org)',
'[docs](/adaptors/packages/commcare-docs)',
'[docs](#http_get)',
'[docs](./sibling)',
'[docs](../parent)',
// a hyphen is not part of a bare identifier, so sibling pages are safe
'[docs](commcare-docs)',
];

for (const link of untouched) {
assert.equal(fix(link), link, `expected ${link} to be left alone`);
}
});

test('leaves bare links inside code samples alone', () => {
const fenced = '```js\n[http.get](http.get)\n```\n\n### http.get {#http_get}';
assert.equal(fix(fenced), fenced);

const inline = 'Call `[http.get](http.get)` here.\n\n### http.get {#http_get}';
assert.equal(fix(inline), inline);
});

test('rewrites prose around a code sample in the same document', () => {
const content = [
'Use [http.get](http.get) instead.',
'',
'```js',
'http.get("case/v1");',
'```',
'',
'### http.get {#http_get}',
].join('\n');

const result = fix(content);
assert.match(result, /Use \[http\.get\]\(#http_get\) instead\./);
assert.match(result, /http\.get\("case\/v1"\);/);
});

// the shape that broke the build
test('handles the commcare deprecation notices', () => {
const content = [
'### get',
'',
"~~***This function only works against CommCare's legacy v0.5 API.",
'For current CommCare APIs, use [http.get](http.get) instead.***',
'',
'### http.get {#http_get}',
'',
'### http.post {#http_post}',
'',
'Also see [http.post](http.post) and [get](get).',
].join('\n');

const result = fix(content, 'commcare');
assert.match(result, /\[http\.get\]\(#http_get\)/);
assert.match(result, /\[http\.post\]\(#http_post\)/);
assert.match(result, /\[get\]\(#get\)/);
});

// should not silently stop finding anchors if the transform order ever changes
test('finds explicit anchors whether or not the braces are escaped', () => {
const raw = 'Use [http.get](http.get).\n\n### http.get {#http_get}';
const escaped = 'Use [http.get](http.get).\n\n### http.get \\{#http_get\\}';

assert.match(fix(raw), /\[http\.get\]\(#http_get\)/);
assert.match(fix(escaped), /\[http\.get\]\(#http_get\)/);
});

test('escapeMdx still escapes braces outside code blocks only', () => {
assert.equal(escapeMdx('a {b} c'), 'a \\{b\\} c');
assert.equal(escapeMdx('`{b}`'), '`{b}`');
assert.equal(escapeMdx('```\n{b}\n```'), '```\n{b}\n```');
});