diff --git a/build-algolia.js b/build-algolia.js new file mode 100755 index 00000000000..6a3ab6491ef --- /dev/null +++ b/build-algolia.js @@ -0,0 +1,87 @@ +#! /usr/bin/env node +'use strict'; +const jsdom = require("jsdom"); +const { + JSDOM +} = jsdom; +const md5 = require('md5'); +const atomicalgolia = require("atomic-algolia") +const fs = require('fs'); +const indexName = "dev_docs" +const rawdata = fs.readFileSync('public/algolia.json'); +const nodes = JSON.parse(rawdata); +const nue = []; + +nodes.forEach(node => { + const dom = new JSDOM(node.content); + const content = dom.window.document.body; //post content wrapped in a body tag + const contentChildren = content.children; // all the children of the body tag + const paragraphOut = { + anchor: '#', + title: '', + content: '', + postref: node.objectID, + objectID: md5(node.permalink), + permalink: node.permalink + }; + + let childCount = contentChildren.length - 1; // how many children + + // loop over the content until the next h2 heading -> this is the paragraph of searchable text + while(childCount >= 0) { + const child = contentChildren[childCount]; + + if (child.tagName === "H2") { + //this is our header + paragraphOut.anchor = `#${child.id}`; + paragraphOut.title = child.textContent; + + let next = child.nextElementSibling; + + while(next && next.tagName !== 'H2') { + if (next) { + paragraphOut.content += next.textContent; + } + next = next.nextElementSibling; + } + + } + + childCount--; + } + + // a post without headers + if (paragraphOut.title === '') { + // Set the title to the page title + paragraphOut.title = node.title; + + // pass along the content + paragraphOut.content = content.textContent; + } + + // limit the content to 10k so we dont blow up just incase someone decides to make a 40k blog post in one paragraph ¯\_(ツ)_/¯ + paragraphOut.content = paragraphOut.content.substr(0, 8000); + + // objectID is not quite unique yet so hash the entire object + paragraphOut.objectID = md5(JSON.stringify(paragraphOut)); + + + nue.push(paragraphOut); + + // remove potentially large content (see size limits) and replace with teh summary so that we don't get results with zero highlightable results + node.content = node.summary; + + // remove summary for dedup + delete node.summary; + +}); + +const merged = [...nodes, ...nue]; + +// fs.writeFileSync('public/combined.algolia.json', JSON.stringify(merged)); +// process.exit(0); +atomicalgolia(indexName, merged, (err, result) => { + if (err) throw err + console.log(result); + process.exit(0); +}); diff --git a/package.json b/package.json index 950649392bd..c710263ef49 100644 --- a/package.json +++ b/package.json @@ -8,7 +8,7 @@ "server": "gulp server", "server:with-drafts": "gulp server:with-drafts", "cms:delete": "gulp cms-delete", - "algolia": "atomic-algolia" + "algolia": "node build-algolia.js" }, "dependencies": { "atomic-algolia": "^0.3.15", @@ -33,6 +33,8 @@ "gulp-watch": "^5.0.0", "instantsearch.js": "^2.8.0", "jquery": "^3.3.1", + "jsdom": "^11.11.0", + "md5": "^2.2.1", "moment": "^2.20.1", "node-sass-tilde-importer": "^1.0.0", "rancher-website-theme": "https://github.com/rancherlabs/website-theme.git", diff --git a/src/js/app.js b/src/js/app.js index 8626fc5fe9c..b22cd167644 100644 --- a/src/js/app.js +++ b/src/js/app.js @@ -25,28 +25,12 @@ const bootstrapDocsSearch = function() { container: '#hits', templates: { empty: '
{{title}}{{summary}} |