container topic inference

This commit is contained in:
Fabian Freund
2025-07-20 00:37:51 +02:00
parent 2a3e0b6b6d
commit dc14a7d718
29 changed files with 1117 additions and 12 deletions
@@ -0,0 +1,151 @@
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
"use strict";
const { createEngine } = ChromeUtils.importESModule("chrome://global/content/ml/EngineProcess.sys.mjs");
const ML_TASK_FEATURE_EXTRACTION = "feature-extraction";
const ML_TASK_TEXT2TEXT = "text2text-generation";
const SMART_TAB_GROUPING_CONFIG = {
embedding: {
dtype: "q8",
timeoutMS: 2 * 60 * 1000, // 2 minutes
taskName: ML_TASK_FEATURE_EXTRACTION,
featureId: "smart-tab-embedding",
backend: "onnx",
},
topicGeneration: {
dtype: "q8",
timeoutMS: 2 * 60 * 1000, // 2 minutes
taskName: ML_TASK_TEXT2TEXT,
featureId: "smart-tab-topic",
backend: "onnx",
},
// dataConfig: {
// titleKey: "label",
// descriptionKey: "description",
// },
// clustering: {
// dimReductionMethod: null, // Not completed.
// clusterImplementation: CLUSTER_METHODS.KMEANS,
// clusteringTriesPerK: 3,
// anchorMethod: ANCHOR_METHODS.FIXED,
// pregroupedHandlingMethod: PREGROUPED_HANDLING_METHODS.EXCLUDE,
// pregroupedSilhouetteBoost: 2, // Relative weight of the cluster's score and all other cluster's combined
// suggestOtherTabsMethod: SUGGEST_OTHER_TABS_METHODS.NEAREST_NEIGHBOR,
// },
};
/**
* Generate model input from keywords and documents
* @param {string []} keywords
* @param {string []} documents
*/
function createModelInput(keywords, documents) {
if (!keywords || keywords.length === 0) {
return `Topic from keywords: titles: \n${documents.join(" \n")}`;
}
return `Topic from keywords: ${keywords.join(", ")}. titles: \n${documents.join(" \n")}`;
}
/**
* One artifact of the LLM output is that sometimes words are duplicated
* This function cuts the phrase when it sees the first duplicate word.
* Handles simple singluar / plural duplicates (-s only).
* @param {string} phrase Input phrase
* @returns {string} phrase cut before any duplicate word
*/
function cutAtDuplicateWords(phrase) {
if (!phrase.length) {
return phrase;
}
const wordsSet = new Set();
const wordList = phrase.split(" ");
for (let i = 0; i < wordList.length; i++) {
let baseWord = wordList[i].toLowerCase();
if (baseWord.length > 3) {
if (baseWord.slice(-1) === "s") {
baseWord = baseWord.slice(0, -1);
}
}
if (wordsSet.has(baseWord)) {
// We are seeing a baseWord word. Exit with just the words so far and don't
// add any new words
return wordList.slice(0, i).join(" ");
}
wordsSet.add(baseWord);
}
return phrase; // return original phrase
}
/**
*
* @param {MLEngine} engine the engine to check
* @return {boolean} true if the engine has not been initialized or closed
*/
function isEngineClosed(engine) {
return !engine || engine?.engineStatus === "closed";
}
this.ml = class extends ExtensionAPI {
getAPI(context) {
return {
experiments: {
ml: {
async containerTopic(keywords, documents) {
if (isEngineClosed(this.topicEngine)) {
const {
featureId,
engineId,
dtype,
taskName,
timeoutMS,
modelId,
modelRevision,
backend,
} = SMART_TAB_GROUPING_CONFIG.topicGeneration;
let initData = {
featureId,
engineId,
dtype,
taskName,
timeoutMS,
modelId,
modelRevision,
backend,
};
this.topicEngine = await createEngine(initData);
}
const inputArgs = createModelInput(
keywords,
documents
);
const requestInfo = {
inputArgs,
runOptions: {
max_length: 6,
},
};
const request = {
args: [requestInfo.inputArgs],
options: requestInfo.runOptions,
};
const res = await this.topicEngine.run(request);
const generated = cutAtDuplicateWords((res[0]["generated_text"] || "").trim());
return generated;
}
}
}
};
}
};
@@ -0,0 +1,29 @@
/* This Source Code Form is subject to the terms of the Mozilla Public
* License, v. 2.0. If a copy of the MPL was not distributed with this
* file, You can obtain one at http://mozilla.org/MPL/2.0/. */
"use strict";
const { KeywordExtractor } = ChromeUtils.importESModule(
"chrome://global/content/ml/NLPUtils.sys.mjs"
);
this.nlp = class extends ExtensionAPI {
getAPI(context) {
return {
experiments: {
nlp: {
async extractKeywords(corpus, maxKeywords = 3) {
try {
const keywordExtractor = new KeywordExtractor();
const keywords = keywordExtractor.fitTransform(corpus, maxKeywords);
return keywords;
} catch (error) {
throw new ExtensionError(`Keyword extraction failed: ${error.message}`);
}
},
}
}
};
}
};
@@ -0,0 +1,38 @@
'use strict';
const port = browser.runtime.connectNative("mlEngine");
function sendJsonResultForRequest(id) {
return function (result) {
port.postMessage({
"id": id,
"status": "success",
"result": result
})
}
}
function sendErrorForRequest(id) {
return function (error) {
console.error(error);
port.postMessage({
"id": id,
"status": "error",
"error": error
});
}
}
port.onMessage.addListener(async (message) => {
let requestId = message["id"]
switch (message["action"]) {
case "getContainerTopic":
const documents = message["args"];
const keywords = await browser.experiments.nlp.extractKeywords([documents.slice(0, 3).join(" ")]);
browser.experiments.ml.containerTopic(keywords[0], documents)
.then(sendJsonResultForRequest(requestId))
.catch(sendErrorForRequest(requestId))
break
}
});
@@ -0,0 +1,61 @@
{
"manifest_version": 2,
"name": "ml-engine",
"version": "1.0",
"description": "WebLibre ML Engine",
"browser_specific_settings": {
"gecko": {
"id": "ml-engine@weblibre.eu"
}
},
"experiment_apis": {
"nlp": {
"schema": "schema.json",
"parent": {
"scopes": [
"addon_parent"
],
"script": "api/nlp.js",
"paths": [
[
"experiments",
"nlp"
]
]
}
},
"ml": {
"schema": "schema.json",
"parent": {
"scopes": [
"addon_parent"
],
"script": "api/ml.js",
"paths": [
[
"experiments",
"ml"
]
]
}
}
},
"background": {
"scripts": [
"background.js"
]
},
"optional_permissions": [
"trialML"
],
"permissions": [
"nativeMessaging",
"nativeMessagingFromContent",
"geckoViewAddons",
"cookies",
"menus",
"scripting",
"storage",
"<all_urls>"
]
}
@@ -0,0 +1,61 @@
[
{
"namespace": "experiments.nlp",
"description": "Natural Language Processing utilities",
"functions": [
{
"name": "extractKeywords",
"type": "function",
"description": "Extract keywords from a corpus of text documents",
"async": true,
"parameters": [
{
"name": "corpus",
"type": "array",
"items": {
"type": "string"
},
"description": "Array of text documents to extract keywords from"
},
{
"name": "maxKeywords",
"type": "integer",
"optional": true,
"default": 3,
"description": "Maximum number of keywords to extract per document"
}
]
}
]
},
{
"namespace": "experiments.ml",
"description": "Machine Learning utilities",
"functions": [
{
"name": "containerTopic",
"type": "function",
"description": "Generate topic from keywords and documents using ML engine",
"async": true,
"parameters": [
{
"name": "keywords",
"type": "array",
"items": {
"type": "string"
},
"description": "Array of keywords to generate topic from"
},
{
"name": "documents",
"type": "array",
"items": {
"type": "string"
},
"description": "Array of document titles/content"
}
]
}
]
}
]