The extension linked nothing and the resolver saw no requests: the content script fetched http://localhost:8787 directly from the HTTPS reddit page, which reddit's CSP / mixed-content blocks. Move the network call to a background script (extension origin, exempt) and add stage logging so failures are visible. - extension/background.js (new): does the resolver fetch on message; sets the toolbar badge. - extension/content.js: talks to background via runtime messaging (no page-context fetch); [rsl] console logs at boot/root/candidates/resolve/scan; single primary link. - extension/manifest.json: background.scripts + browser_action (icon.svg, title). - extension/{styles.css,icon.svg}: platform-coloured link (green=spotify, teal=bandcamp) + toolbar icon. - service/lib.js: primaryOf()/isSearchUrl() — link Spotify when the artist is on Spotify, else Bandcamp, else Spotify search; spotifyResult/mbResult attach `primary` (unit-tested). - README: link behaviour, toolbar, Troubleshooting. Verified: scripts/check.sh green (10 tests); resolver returns `primary` end-to-end. The in-browser fix is logic-verified only (no headless Firefox here) — reload the temp add-on and watch the page console `[rsl]` logs to confirm. Source: project human-feedback 2026-06-16 (notes 01KV8VC58QMYBDY6VVVJHR8YNY + 01KV91AMJCN2C4GFCKMKQRGNXE). Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
191 lines
7.8 KiB
JavaScript
191 lines
7.8 KiB
JavaScript
// Reddit -> Spotify/Bandcamp linker (content script).
|
||
//
|
||
// On a reddit thread page it: pulls candidate artist names out of comment text,
|
||
// asks the local resolver service which ones are real artists, and rewrites those
|
||
// occurrences in place into links. The resolver does the hard "is this an artist"
|
||
// part; this script only handles extraction + DOM rewriting.
|
||
|
||
(() => {
|
||
"use strict";
|
||
|
||
const api = globalThis.browser ?? globalThis.chrome;
|
||
const log = (...a) => console.log("[rsl]", ...a);
|
||
|
||
// ---- candidate extraction -------------------------------------------------
|
||
// A token is a Capitalized or ALL-CAPS word; a candidate is 1-4 tokens joined by
|
||
// spaces, with a few lowercase connectors allowed inside ("Signs of the Swarm").
|
||
const TOKEN = "[A-Z][A-Za-z0-9'’.&\\-]*";
|
||
const CAND_RE = new RegExp(`${TOKEN}(?:[ ]+(?:${TOKEN}|of|the|and|&|in|on|to|for|a|an)){0,3}`, "g");
|
||
const CONNECTORS = new Set(["of", "the", "and", "&", "in", "on", "to", "for", "a", "an"]);
|
||
|
||
// Common capitalized words that are almost never the artist being recommended.
|
||
// Deliberately small: the resolver's exact-name match is the real filter, so we
|
||
// avoid stoplisting plausible one-word band names (Yes, Love, Tool, ...).
|
||
const STOP = new Set([
|
||
"I", "A", "An", "The", "And", "But", "Or", "Nor", "So", "Yet", "For", "If", "As",
|
||
"At", "By", "In", "On", "To", "Of", "Up", "Off", "Out", "It", "He", "She", "We",
|
||
"They", "You", "Me", "Him", "Her", "Us", "Them", "My", "Your", "His", "Its", "Our",
|
||
"Their", "This", "That", "These", "Those", "Is", "Are", "Was", "Were", "Be", "Been",
|
||
"Am", "Do", "Does", "Did", "Has", "Have", "Had", "Will", "Would", "Can", "Could",
|
||
"Should", "Not", "No", "Yeah", "Ok", "Okay", "Lol", "Imo", "Imho", "Tbh", "Edit",
|
||
"Reddit", "Spotify", "Bandcamp", "Youtube", "Google", "Apple", "Album", "Albums",
|
||
"Band", "Bands", "Song", "Songs", "Track", "Tracks", "Ep", "Lp", "Cd", "Op", "Vol",
|
||
]);
|
||
|
||
function matchesIn(text) {
|
||
const out = [];
|
||
CAND_RE.lastIndex = 0;
|
||
let m;
|
||
while ((m = CAND_RE.exec(text))) {
|
||
let str = m[0].replace(/[.,;:!?’'")\]]+$/u, ""); // trim trailing punctuation
|
||
let parts = str.split(/\s+/);
|
||
while (parts.length && CONNECTORS.has(parts[parts.length - 1].toLowerCase())) parts.pop();
|
||
str = parts.join(" ");
|
||
if (!str) continue;
|
||
const single = parts.length === 1;
|
||
if (single && (STOP.has(str) || str.length < 2 || /^\d+$/.test(str))) continue;
|
||
out.push({ name: str, index: m.index, length: str.length }); // only trailing trimmed -> index unchanged
|
||
}
|
||
return out;
|
||
}
|
||
|
||
// ---- DOM walking ----------------------------------------------------------
|
||
const SKIP_TAGS = new Set([
|
||
"A", "SCRIPT", "STYLE", "CODE", "PRE", "TEXTAREA", "INPUT", "BUTTON", "TIME", "NOSCRIPT", "SVG", "SELECT",
|
||
]);
|
||
|
||
function inSkippable(node) {
|
||
for (let el = node.parentElement; el; el = el.parentElement) {
|
||
if (SKIP_TAGS.has(el.tagName)) return true;
|
||
if (el.classList && el.classList.contains("rsl-link")) return true;
|
||
if (el.isContentEditable) return true;
|
||
}
|
||
return false;
|
||
}
|
||
|
||
function collectTextNodes(root) {
|
||
const nodes = [];
|
||
const walker = document.createTreeWalker(root, NodeFilter.SHOW_TEXT, {
|
||
acceptNode(node) {
|
||
const v = node.nodeValue;
|
||
if (!v || v.length > 5000 || !/[A-Za-z]/.test(v)) return NodeFilter.FILTER_REJECT;
|
||
if (inSkippable(node)) return NodeFilter.FILTER_REJECT;
|
||
return NodeFilter.FILTER_ACCEPT;
|
||
},
|
||
});
|
||
let n;
|
||
while ((n = walker.nextNode())) nodes.push(n);
|
||
return nodes;
|
||
}
|
||
|
||
// ---- link building / rewriting --------------------------------------------
|
||
// Per the user's "if on spotify, link that, else bandcamp" rule: one link to the
|
||
// primary platform, colour-coded so you can tell which at a glance.
|
||
function makeLink(label, links) {
|
||
const primary = links.primary || { platform: "spotify", url: links.spotify };
|
||
const a = document.createElement("a");
|
||
a.className = `rsl-link rsl-${primary.platform}`;
|
||
a.textContent = label;
|
||
a.href = primary.url;
|
||
a.target = "_blank";
|
||
a.rel = "noopener noreferrer";
|
||
a.title = `Open on ${primary.platform === "spotify" ? "Spotify" : "Bandcamp"}`;
|
||
return a;
|
||
}
|
||
|
||
function wrapNode(node) {
|
||
const text = node.nodeValue;
|
||
const ms = matchesIn(text).filter((m) => {
|
||
const links = resolved.get(m.name);
|
||
return links && ((links.primary && links.primary.url) || links.spotify);
|
||
});
|
||
if (!ms.length) return;
|
||
const frag = document.createDocumentFragment();
|
||
let last = 0;
|
||
for (const m of ms) {
|
||
if (m.index < last) continue; // skip overlaps
|
||
if (m.index > last) frag.appendChild(document.createTextNode(text.slice(last, m.index)));
|
||
frag.appendChild(makeLink(text.substr(m.index, m.length), resolved.get(m.name)));
|
||
last = m.index + m.length;
|
||
}
|
||
if (last < text.length) frag.appendChild(document.createTextNode(text.slice(last)));
|
||
node.parentNode && node.parentNode.replaceChild(frag, node);
|
||
}
|
||
|
||
// ---- resolver client ------------------------------------------------------
|
||
const resolved = new Map(); // candidate string -> { spotify, bandcamp } | null
|
||
|
||
// The network call goes through the background script — a page-context fetch to
|
||
// http://localhost is blocked by reddit's CSP / mixed-content rules.
|
||
async function resolve(names) {
|
||
log("resolving", names.length, "candidate(s) via background:", names.slice(0, 8));
|
||
try {
|
||
const resp = await api.runtime.sendMessage({ type: "resolve", candidates: names });
|
||
if (!resp) throw new Error("no response from background script");
|
||
if (!resp.ok) throw new Error(resp.error || "resolve failed");
|
||
const hits = Object.values(resp.data).filter(Boolean).length;
|
||
log("resolver returned", hits, "artist(s) of", names.length);
|
||
return resp.data;
|
||
} catch (e) {
|
||
console.warn("[rsl] resolve failed (is the resolver running on :8787?):", e.message);
|
||
return null;
|
||
}
|
||
}
|
||
|
||
// ---- scan orchestration ---------------------------------------------------
|
||
function getRoot() {
|
||
return (
|
||
document.querySelector("main, .content[role='main'], shreddit-comment-tree, .commentarea") ||
|
||
document.body
|
||
);
|
||
}
|
||
|
||
async function scan() {
|
||
const root = getRoot();
|
||
const nodes = collectTextNodes(root);
|
||
log("scan: root", root.tagName || root.nodeName, "| text nodes", nodes.length);
|
||
if (!nodes.length) return;
|
||
|
||
const seen = new Set();
|
||
for (const node of nodes) for (const m of matchesIn(node.nodeValue)) seen.add(m.name);
|
||
log("candidates found:", seen.size);
|
||
|
||
const ask = [...seen].filter((c) => !resolved.has(c));
|
||
if (ask.length) {
|
||
const res = await resolve(ask);
|
||
if (res) for (const c of ask) resolved.set(c, res[c] || null); // negative-cache misses
|
||
}
|
||
for (const node of nodes) if (node.isConnected) wrapNode(node);
|
||
|
||
const total = document.querySelectorAll("a.rsl-link").length;
|
||
api.runtime.sendMessage({ type: "badge", count: total }).catch(() => {});
|
||
log("scan done; links on page:", total);
|
||
}
|
||
|
||
// serialize scans; coalesce overlapping triggers
|
||
let running = false;
|
||
let pending = false;
|
||
function schedule() {
|
||
if (running) { pending = true; return; }
|
||
running = true;
|
||
scan()
|
||
.catch((e) => console.warn("[rsl]", e))
|
||
.finally(() => {
|
||
running = false;
|
||
if (pending) { pending = false; setTimeout(schedule, 300); }
|
||
});
|
||
}
|
||
|
||
function debounce(fn, ms) {
|
||
let t;
|
||
return () => { clearTimeout(t); t = setTimeout(fn, ms); };
|
||
}
|
||
|
||
const onThread = () => /\/comments\//.test(location.pathname);
|
||
|
||
log("content script loaded:", location.href, "| thread page:", onThread());
|
||
if (onThread()) schedule();
|
||
new MutationObserver(debounce(() => { if (onThread()) schedule(); }, 600))
|
||
.observe(document.documentElement, { childList: true, subtree: true });
|
||
})();
|