feat: spotify-only matching + optional local-LLM gate + google→bandcamp fallback (closes task/drop-musicbrainz-google-fallback, closes task/ollama-classifier)
Source: human-feedback note 01KVDYV3A35B57N15CRBQDXPCH (project/reddit-spotify-linker, 2026-06-18) — "remove the musicbrainz integration and just use spotify"; a google search for "<name> bandcamp" fallback; and "would a local llm be better able to categorise these ... include instructions for running something suitable via ollama (16GB M2)". - Remove MusicBrainz entirely (viaMusicBrainz + url-rels + rate-gate; lib relUrls/mbResult). Spotify is now the only music API; it both gates by exact-name match and supplies the direct artist link. - Optional ollama LLM gate (OLLAMA_URL enables it, OLLAMA_MODEL default qwen2.5:3b) called over plain HTTP -> stays zero npm-dep. It classifies which candidates are real artists/albums (the matching-quality lever). Pure prompt/parse logic in new llm.js (unit-tested in llm.test.js); the fetch + resolveAll gate wiring live in resolver.js. - A confirmed name links to its direct Spotify page when available, else a Google search for "<name> bandcamp". New "google" primary platform rendered by the extension (styles.css blue accent + platform-aware link title in content.js). - README: drop MusicBrainz; document the two gates + a runnable local-LLM setup for a 16GB MacBook M2. AGENTS.md: flip the old "no ML/NER" standing decision per the human override. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
+13
-8
@@ -1,13 +1,18 @@
|
||||
# Copy to `.env` (it is gitignored). All values are OPTIONAL.
|
||||
# Copy to `.env` (it is gitignored).
|
||||
#
|
||||
# Without Spotify credentials the resolver validates names against MusicBrainz and links
|
||||
# to Spotify/Bandcamp SEARCH pages — works with zero setup.
|
||||
#
|
||||
# With credentials it uses the Spotify Web API for DIRECT artist links.
|
||||
# Create a (free) app at https://developer.spotify.com/dashboard to get these.
|
||||
# Spotify Web API (client-credentials) — gives DIRECT artist links and, on its own, gates
|
||||
# candidates by exact-name match (a name links only when Spotify has that artist). Create a
|
||||
# free app at https://developer.spotify.com/dashboard to get these.
|
||||
SPOTIFY_CLIENT_ID=
|
||||
SPOTIFY_CLIENT_SECRET=
|
||||
|
||||
# Port the resolver listens on (default 8787). If you change it, also update
|
||||
# RESOLVER_URL in extension/content.js and the host in extension/manifest.json.
|
||||
# Optional local LLM (via ollama) for BETTER matching — it decides which candidate substrings
|
||||
# are real artists/albums. Set OLLAMA_URL to enable it; still zero npm deps (plain HTTP).
|
||||
# Confirmed names get a direct Spotify link when available, else a Google->Bandcamp fallback.
|
||||
# ollama pull qwen2.5:3b # ~2GB, runs comfortably on a 16GB MacBook M2
|
||||
# OLLAMA_URL=http://localhost:11434
|
||||
# OLLAMA_MODEL=qwen2.5:3b
|
||||
|
||||
# Port the resolver listens on (default 8787). If you change it, also update RESOLVER_URL in
|
||||
# extension/background.js and the host in extension/manifest.json permissions.
|
||||
PORT=8787
|
||||
|
||||
+19
-58
@@ -4,44 +4,22 @@
|
||||
|
||||
export const norm = (s) => String(s).normalize("NFKC").trim().toLowerCase();
|
||||
|
||||
export const bandcampSearch = (n) => `https://bandcamp.com/search?q=${encodeURIComponent(n)}&item_type=b`;
|
||||
export const spotifySearch = (n) => `https://open.spotify.com/search/${encodeURIComponent(n)}`;
|
||||
// Fallback link when we have no direct Spotify page: a Google search biased at Bandcamp,
|
||||
// which surfaces the artist's Bandcamp page (when it exists) plus general results.
|
||||
export const googleSearch = (n) => `https://www.google.com/search?q=${encodeURIComponent(`${n} bandcamp`)}`;
|
||||
|
||||
// The precision lever: return the result whose name matches the candidate
|
||||
// case-insensitively, else null. `minScore` (MusicBrainz) additionally requires a
|
||||
// confidence score. Tolerates null/garbage items.
|
||||
export function pickArtist(name, items, { minScore = null } = {}) {
|
||||
// The precision lever: return the item whose name matches the candidate
|
||||
// case-insensitively, else null. Tolerates null/garbage items.
|
||||
export function pickArtist(name, items) {
|
||||
const want = norm(name);
|
||||
return (
|
||||
(items ?? []).find(
|
||||
(a) => a && norm(a.name) === want && (minScore == null || (a.score ?? 0) >= minScore),
|
||||
) ?? null
|
||||
);
|
||||
return (items ?? []).find((a) => a && norm(a.name) === want) ?? null;
|
||||
}
|
||||
|
||||
// A "/search" url means we don't have a direct artist page (just a search fallback).
|
||||
export const isSearchUrl = (url) => typeof url === "string" && url.includes("/search");
|
||||
|
||||
// Choose the single best link to show: prefer Spotify when the artist is actually ON
|
||||
// Spotify (a direct, non-search url); else a direct Bandcamp; else the Spotify search.
|
||||
export function primaryOf({ spotify, bandcamp }) {
|
||||
if (spotify && !isSearchUrl(spotify)) return { platform: "spotify", url: spotify };
|
||||
if (bandcamp && !isSearchUrl(bandcamp)) return { platform: "bandcamp", url: bandcamp };
|
||||
return { platform: "spotify", url: spotify };
|
||||
}
|
||||
|
||||
// Link object for a confirmed Spotify hit (direct artist url when present, else search).
|
||||
export function spotifyResult(hit) {
|
||||
const r = {
|
||||
name: hit.name,
|
||||
spotify: hit.external_urls?.spotify ?? spotifySearch(hit.name),
|
||||
bandcamp: bandcampSearch(hit.name),
|
||||
};
|
||||
return { ...r, primary: primaryOf(r) };
|
||||
}
|
||||
|
||||
// True for http/https URLs only — reject javascript:, ftp:, garbage. These values flow
|
||||
// into an injected link's href, so validate the third-party MB data defensively.
|
||||
// True for http/https URLs only — reject javascript:, ftp:, garbage. The Spotify
|
||||
// external_url flows into an injected link's href, so validate it defensively.
|
||||
export function isHttpUrl(u) {
|
||||
try {
|
||||
const p = new URL(u).protocol;
|
||||
@@ -51,31 +29,14 @@ export function isHttpUrl(u) {
|
||||
}
|
||||
}
|
||||
|
||||
// Extract direct links from a MusicBrainz artist's url-relations (zero-cred: MB stores
|
||||
// official Bandcamp / streaming URLs). Accepts only http(s) and matches by parsed hostname
|
||||
// (not substring, so "evilbandcamp.com" / "javascript:…bandcamp.com…" don't slip through).
|
||||
// Either field may be null.
|
||||
export function relUrls(relations) {
|
||||
const urls = (relations ?? []).map((r) => r && r.url && r.url.resource).filter(isHttpUrl);
|
||||
const byHost = (host) =>
|
||||
urls.find((u) => {
|
||||
try {
|
||||
const h = new URL(u).hostname;
|
||||
return h === host || h.endsWith("." + host);
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}) ?? null;
|
||||
return { bandcamp: byHost("bandcamp.com"), spotify: byHost("open.spotify.com") };
|
||||
}
|
||||
|
||||
// Link object for a confirmed MusicBrainz hit: prefer the direct url-rel links, fall back
|
||||
// to search urls when MB has no relation for that platform.
|
||||
export function mbResult(hit, links) {
|
||||
const r = {
|
||||
name: hit.name,
|
||||
spotify: links?.spotify ?? spotifySearch(hit.name),
|
||||
bandcamp: links?.bandcamp ?? bandcampSearch(hit.name),
|
||||
};
|
||||
return { ...r, primary: primaryOf(r) };
|
||||
// Build the single link to show for a confirmed name. Prefer the direct Spotify artist
|
||||
// url (validated http(s), not a /search url); otherwise fall back to a Google→Bandcamp
|
||||
// search. `primary` is what the extension renders ({ platform, url }).
|
||||
export function linkResult(name, spotifyUrl) {
|
||||
const direct = spotifyUrl && isHttpUrl(spotifyUrl) && !isSearchUrl(spotifyUrl) ? spotifyUrl : null;
|
||||
if (direct) {
|
||||
return { name, spotify: direct, primary: { platform: "spotify", url: direct } };
|
||||
}
|
||||
const url = googleSearch(name);
|
||||
return { name, google: url, primary: { platform: "google", url } };
|
||||
}
|
||||
|
||||
+29
-86
@@ -1,7 +1,7 @@
|
||||
// Unit tests for the resolver's pure logic (run with `node --test`). Zero-dep.
|
||||
import { test } from "node:test";
|
||||
import assert from "node:assert/strict";
|
||||
import { norm, bandcampSearch, spotifySearch, pickArtist, spotifyResult, relUrls, mbResult, primaryOf, isSearchUrl, isHttpUrl } from "./lib.js";
|
||||
import { norm, googleSearch, pickArtist, linkResult, isSearchUrl, isHttpUrl } from "./lib.js";
|
||||
|
||||
test("norm lowercases, trims, NFKC-normalizes", () => {
|
||||
assert.equal(norm(" Cerebral Bore "), "cerebral bore");
|
||||
@@ -15,78 +15,44 @@ test("pickArtist matches case-insensitively, rejects non-matches", () => {
|
||||
assert.equal(pickArtist("Notarealband", items), null);
|
||||
});
|
||||
|
||||
test("pickArtist enforces minScore when given (MusicBrainz path)", () => {
|
||||
const items = [{ name: "Devourment", score: 100 }, { name: "Devourment", score: 50 }];
|
||||
assert.equal(pickArtist("Devourment", items, { minScore: 90 })?.score, 100);
|
||||
assert.equal(pickArtist("Devourment", [{ name: "Devourment", score: 50 }], { minScore: 90 }), null);
|
||||
});
|
||||
|
||||
test("pickArtist tolerates empty / garbage input", () => {
|
||||
assert.equal(pickArtist("x", null), null);
|
||||
assert.equal(pickArtist("x", []), null);
|
||||
assert.equal(pickArtist("x", [null, { name: "y" }]), null);
|
||||
});
|
||||
|
||||
test("spotifyResult prefers the direct url, falls back to search", () => {
|
||||
const direct = spotifyResult({
|
||||
name: "Devourment",
|
||||
external_urls: { spotify: "https://open.spotify.com/artist/abc" },
|
||||
});
|
||||
assert.equal(direct.spotify, "https://open.spotify.com/artist/abc");
|
||||
assert.match(direct.bandcamp, /^https:\/\/bandcamp\.com\/search\?q=Devourment/);
|
||||
|
||||
assert.deepEqual(direct.primary, { platform: "spotify", url: "https://open.spotify.com/artist/abc" });
|
||||
|
||||
const noUrl = spotifyResult({ name: "Devourment" });
|
||||
assert.equal(noUrl.spotify, "https://open.spotify.com/search/Devourment");
|
||||
assert.equal(noUrl.primary.platform, "spotify"); // search fallback still points at spotify
|
||||
test("googleSearch points at a Google query for '<name> bandcamp', url-encoded", () => {
|
||||
const u = googleSearch("Cerebral Bore");
|
||||
assert.match(u, /^https:\/\/www\.google\.com\/search\?q=/);
|
||||
assert.ok(u.includes("Cerebral%20Bore%20bandcamp"));
|
||||
assert.ok(googleSearch("A&B").includes("A%26B"));
|
||||
});
|
||||
|
||||
test("primaryOf / isSearchUrl: prefer direct spotify, else direct bandcamp, else spotify search", () => {
|
||||
test("linkResult prefers a direct Spotify url, primary = spotify", () => {
|
||||
const r = linkResult("Devourment", "https://open.spotify.com/artist/abc");
|
||||
assert.equal(r.spotify, "https://open.spotify.com/artist/abc");
|
||||
assert.deepEqual(r.primary, { platform: "spotify", url: "https://open.spotify.com/artist/abc" });
|
||||
assert.equal(r.google, undefined);
|
||||
});
|
||||
|
||||
test("linkResult falls back to Google->Bandcamp when there is no direct Spotify url", () => {
|
||||
const r = linkResult("Some Local Band", null);
|
||||
assert.equal(r.spotify, undefined);
|
||||
assert.equal(r.primary.platform, "google");
|
||||
assert.match(r.primary.url, /^https:\/\/www\.google\.com\/search\?q=/);
|
||||
assert.equal(r.google, r.primary.url);
|
||||
});
|
||||
|
||||
test("linkResult rejects non-http(s) / search Spotify urls and falls back to Google", () => {
|
||||
for (const bad of ["javascript:alert(1)", "ftp://x/y", "https://open.spotify.com/search/X", "not a url"]) {
|
||||
const r = linkResult("X", bad);
|
||||
assert.equal(r.primary.platform, "google", `expected fallback for ${bad}`);
|
||||
}
|
||||
});
|
||||
|
||||
test("isSearchUrl detects /search urls", () => {
|
||||
assert.equal(isSearchUrl("https://open.spotify.com/search/X"), true);
|
||||
assert.equal(isSearchUrl("https://open.spotify.com/artist/abc"), false);
|
||||
assert.deepEqual(
|
||||
primaryOf({ spotify: "https://open.spotify.com/artist/abc", bandcamp: "https://bandcamp.com/search?q=X&item_type=b" }),
|
||||
{ platform: "spotify", url: "https://open.spotify.com/artist/abc" },
|
||||
);
|
||||
assert.deepEqual(
|
||||
primaryOf({ spotify: "https://open.spotify.com/search/X", bandcamp: "https://x.bandcamp.com/" }),
|
||||
{ platform: "bandcamp", url: "https://x.bandcamp.com/" },
|
||||
);
|
||||
assert.deepEqual(
|
||||
primaryOf({ spotify: "https://open.spotify.com/search/X", bandcamp: "https://bandcamp.com/search?q=X&item_type=b" }),
|
||||
{ platform: "spotify", url: "https://open.spotify.com/search/X" },
|
||||
);
|
||||
});
|
||||
|
||||
test("relUrls picks direct bandcamp + spotify from MB relations, tolerates junk", () => {
|
||||
const relations = [
|
||||
{ url: { resource: "https://cattledecapitation.bandcamp.com/" } },
|
||||
{ url: { resource: "https://open.spotify.com/artist/67ZMMtA88DDO0gTuRrzGjn" } },
|
||||
{ url: { resource: "https://en.wikipedia.org/wiki/Cattle_Decapitation" } },
|
||||
null,
|
||||
{ url: null },
|
||||
];
|
||||
const links = relUrls(relations);
|
||||
assert.equal(links.bandcamp, "https://cattledecapitation.bandcamp.com/");
|
||||
assert.equal(links.spotify, "https://open.spotify.com/artist/67ZMMtA88DDO0gTuRrzGjn");
|
||||
});
|
||||
|
||||
test("relUrls returns nulls when relations absent/empty", () => {
|
||||
assert.deepEqual(relUrls(null), { bandcamp: null, spotify: null });
|
||||
assert.deepEqual(relUrls([]), { bandcamp: null, spotify: null });
|
||||
});
|
||||
|
||||
test("relUrls rejects non-http(s) and look-alike hosts (hardening)", () => {
|
||||
const relations = [
|
||||
{ url: { resource: "javascript:alert('bandcamp.com')" } }, // not http(s)
|
||||
{ url: { resource: "ftp://files.bandcamp.com/x" } }, // not http(s)
|
||||
{ url: { resource: "https://evilbandcamp.com/" } }, // host doesn't end in .bandcamp.com
|
||||
{ url: { resource: "https://x.bandcamp.com/" } }, // valid subdomain
|
||||
];
|
||||
const links = relUrls(relations);
|
||||
assert.equal(links.bandcamp, "https://x.bandcamp.com/");
|
||||
assert.equal(links.spotify, null);
|
||||
});
|
||||
|
||||
test("isHttpUrl accepts http/https only", () => {
|
||||
@@ -97,26 +63,3 @@ test("isHttpUrl accepts http/https only", () => {
|
||||
assert.equal(isHttpUrl("not a url"), false);
|
||||
assert.equal(isHttpUrl(null), false);
|
||||
});
|
||||
|
||||
test("mbResult prefers direct links, falls back to search per-platform", () => {
|
||||
const direct = mbResult(
|
||||
{ name: "Cattle Decapitation" },
|
||||
{ bandcamp: "https://cattledecapitation.bandcamp.com/", spotify: "https://open.spotify.com/artist/x" },
|
||||
);
|
||||
assert.equal(direct.bandcamp, "https://cattledecapitation.bandcamp.com/");
|
||||
assert.equal(direct.spotify, "https://open.spotify.com/artist/x");
|
||||
assert.deepEqual(direct.primary, { platform: "spotify", url: "https://open.spotify.com/artist/x" });
|
||||
|
||||
const fallback = mbResult({ name: "Cerebral Bore" }, null);
|
||||
assert.equal(fallback.spotify, "https://open.spotify.com/search/Cerebral%20Bore");
|
||||
assert.equal(fallback.bandcamp, "https://bandcamp.com/search?q=Cerebral%20Bore&item_type=b");
|
||||
|
||||
const partial = mbResult({ name: "X" }, { bandcamp: null, spotify: "https://open.spotify.com/artist/y" });
|
||||
assert.equal(partial.spotify, "https://open.spotify.com/artist/y");
|
||||
assert.match(partial.bandcamp, /bandcamp\.com\/search/);
|
||||
});
|
||||
|
||||
test("link builders url-encode", () => {
|
||||
assert.equal(spotifySearch("A B"), "https://open.spotify.com/search/A%20B");
|
||||
assert.ok(bandcampSearch("A&B").includes("A%26B"));
|
||||
});
|
||||
|
||||
@@ -0,0 +1,51 @@
|
||||
// Pure helpers for the OPTIONAL local-LLM (ollama) classifier. The HTTP call lives in
|
||||
// resolver.js; the prompt builder + reply parser live here so they unit-test with
|
||||
// `node --test` (no network). The LLM answers one question: "which of these candidate
|
||||
// substrings are music artists/albums?" — the precision gate when it is configured.
|
||||
|
||||
import { norm } from "./lib.js";
|
||||
|
||||
// Small, fast, runs comfortably on a 16GB M2 (3.1B params, Q4 ~2GB). Override with OLLAMA_MODEL.
|
||||
export const DEFAULT_MODEL = "qwen2.5:3b";
|
||||
|
||||
// Chat messages for an ollama /api/chat call (used with stream:false, format:"json").
|
||||
export function buildClassifyMessages(candidates) {
|
||||
const system =
|
||||
"You are a music-name classifier for a tool that links artist/album mentions in Reddit " +
|
||||
"music discussions. You are given a JSON array of candidate text snippets pulled from reddit " +
|
||||
"comments. For EACH candidate decide whether it names a music ARTIST/BAND, a music " +
|
||||
"ALBUM/release, or NEITHER. The snippets come from music threads, so prefer the music reading " +
|
||||
"when a candidate is a plausible band/album even if it is also an ordinary word. Mark ordinary " +
|
||||
"words, sentence fragments, generic phrases, place names, and non-music proper nouns as \"none\". " +
|
||||
'Reply with ONLY a JSON object of the form {"results":[{"name":<candidate verbatim>,' +
|
||||
'"kind":"artist"|"album"|"none"}]}. Use each candidate string VERBATIM as its name.';
|
||||
const user = JSON.stringify({ candidates });
|
||||
return [
|
||||
{ role: "system", content: system },
|
||||
{ role: "user", content: user },
|
||||
];
|
||||
}
|
||||
|
||||
// Parse the model's JSON reply into a Set of norm(name) the model marked artist/album.
|
||||
// Defensive: only names that were actually in the candidate list count (guards against the
|
||||
// model inventing names); an unparseable reply yields an empty Set, and the caller then
|
||||
// falls back to the Spotify-exact-match gate.
|
||||
export function parseClassification(text, candidates) {
|
||||
const allowed = new Set(candidates.map(norm));
|
||||
const confirmed = new Set();
|
||||
let obj;
|
||||
try {
|
||||
obj = JSON.parse(text);
|
||||
} catch {
|
||||
return confirmed;
|
||||
}
|
||||
const rows = Array.isArray(obj?.results) ? obj.results : Array.isArray(obj) ? obj : [];
|
||||
for (const row of rows) {
|
||||
if (!row || typeof row.name !== "string") continue;
|
||||
const k = norm(row.name);
|
||||
if (!allowed.has(k)) continue;
|
||||
const kind = String(row.kind ?? "").toLowerCase();
|
||||
if (kind === "artist" || kind === "album") confirmed.add(k);
|
||||
}
|
||||
return confirmed;
|
||||
}
|
||||
@@ -0,0 +1,55 @@
|
||||
// Unit tests for the optional LLM classifier's pure logic (run with `node --test`). Zero-dep.
|
||||
import { test } from "node:test";
|
||||
import assert from "node:assert/strict";
|
||||
import { buildClassifyMessages, parseClassification, DEFAULT_MODEL } from "./llm.js";
|
||||
|
||||
test("DEFAULT_MODEL is a small ollama model", () => {
|
||||
assert.equal(DEFAULT_MODEL, "qwen2.5:3b");
|
||||
});
|
||||
|
||||
test("buildClassifyMessages embeds the candidate list as JSON for the user turn", () => {
|
||||
const msgs = buildClassifyMessages(["Tool", "the best"]);
|
||||
assert.equal(msgs.length, 2);
|
||||
assert.equal(msgs[0].role, "system");
|
||||
assert.match(msgs[0].content, /artist/i);
|
||||
assert.equal(msgs[1].role, "user");
|
||||
assert.deepEqual(JSON.parse(msgs[1].content), { candidates: ["Tool", "the best"] });
|
||||
});
|
||||
|
||||
test("parseClassification keeps only artist/album candidates that were in the list", () => {
|
||||
const candidates = ["Cattle Decapitation", "Devourment", "the best part"];
|
||||
const reply = JSON.stringify({
|
||||
results: [
|
||||
{ name: "Cattle Decapitation", kind: "artist" },
|
||||
{ name: "Devourment", kind: "ARTIST" }, // case-insensitive kind
|
||||
{ name: "the best part", kind: "none" },
|
||||
],
|
||||
});
|
||||
const got = parseClassification(reply, candidates);
|
||||
assert.ok(got.has("cattle decapitation"));
|
||||
assert.ok(got.has("devourment"));
|
||||
assert.ok(!got.has("the best part"));
|
||||
});
|
||||
|
||||
test("parseClassification accepts album kind and a bare array reply", () => {
|
||||
const got = parseClassification(
|
||||
JSON.stringify([{ name: "Reek of Putrefaction", kind: "album" }]),
|
||||
["Reek of Putrefaction"],
|
||||
);
|
||||
assert.ok(got.has("reek of putrefaction"));
|
||||
});
|
||||
|
||||
test("parseClassification ignores hallucinated names not in the candidate list", () => {
|
||||
const got = parseClassification(
|
||||
JSON.stringify({ results: [{ name: "Some Band I Invented", kind: "artist" }] }),
|
||||
["Real Candidate"],
|
||||
);
|
||||
assert.equal(got.size, 0);
|
||||
});
|
||||
|
||||
test("parseClassification returns an empty set on unparseable / junk replies", () => {
|
||||
assert.equal(parseClassification("not json at all", ["X"]).size, 0);
|
||||
assert.equal(parseClassification("", ["X"]).size, 0);
|
||||
assert.equal(parseClassification(JSON.stringify({ nope: true }), ["X"]).size, 0);
|
||||
assert.equal(parseClassification(JSON.stringify({ results: [null, { kind: "artist" }] }), ["X"]).size, 0);
|
||||
});
|
||||
+87
-61
@@ -1,16 +1,20 @@
|
||||
// Tiny resolver service: candidate names -> artist links, for CONFIRMED artists only.
|
||||
//
|
||||
// Zero npm deps; Node 18+ (built-in fetch). The "which words are real artists" problem
|
||||
// is delegated to a music API rather than guessed client-side:
|
||||
// * with SPOTIFY_CLIENT_ID/SECRET set -> Spotify Web API (client-credentials) for
|
||||
// DIRECT artist links;
|
||||
// * without creds -> MusicBrainz validation + Spotify/Bandcamp SEARCH links (zero setup).
|
||||
// Either way a candidate only resolves when an artist's name matches it case-insensitively
|
||||
// (the precision lever). The pure matching/link logic lives in lib.js (unit-tested).
|
||||
// Zero npm deps; Node 18+ (built-in fetch). The "which words are real artists" problem is
|
||||
// not guessed client-side — it is gated one of two ways:
|
||||
// * SPOTIFY_CLIENT_ID/SECRET set -> a candidate resolves only when Spotify has an artist
|
||||
// whose name matches it case-insensitively (the exact-match precision lever), linking to
|
||||
// the DIRECT Spotify artist page.
|
||||
// * OLLAMA_URL set -> a local LLM (via ollama HTTP, still zero npm deps) classifies which
|
||||
// candidates are artists/albums; confirmed names get a Spotify direct link when available,
|
||||
// else a Google->Bandcamp search fallback. This is the better-matching path.
|
||||
// The two combine: with both set, the LLM gates and Spotify supplies direct links.
|
||||
// The pure matching/link/prompt logic lives in lib.js + llm.js (unit-tested).
|
||||
|
||||
import { createServer } from "node:http";
|
||||
import { readFileSync } from "node:fs";
|
||||
import { norm, pickArtist, spotifyResult, mbResult, relUrls } from "./lib.js";
|
||||
import { norm, pickArtist, linkResult } from "./lib.js";
|
||||
import { buildClassifyMessages, parseClassification, DEFAULT_MODEL } from "./llm.js";
|
||||
|
||||
// --- minimal .env loader (no dependency) ------------------------------------
|
||||
try {
|
||||
@@ -25,12 +29,16 @@ const PORT = Number(process.env.PORT) || 8787;
|
||||
const SPOTIFY_ID = process.env.SPOTIFY_CLIENT_ID;
|
||||
const SPOTIFY_SECRET = process.env.SPOTIFY_CLIENT_SECRET;
|
||||
const USE_SPOTIFY = Boolean(SPOTIFY_ID && SPOTIFY_SECRET);
|
||||
const UA = "reddit-spotify-linker/0.1 (https://git.yapplesauce.com/paul/linker)";
|
||||
const OLLAMA_URL = (process.env.OLLAMA_URL || "").replace(/\/$/, ""); // presence enables the LLM gate
|
||||
const OLLAMA_MODEL = process.env.OLLAMA_MODEL || DEFAULT_MODEL;
|
||||
const USE_LLM = Boolean(OLLAMA_URL);
|
||||
const MAX_CANDIDATES = 100;
|
||||
|
||||
const cache = new Map(); // norm(name) -> { spotify, bandcamp, name } | null (negatives cached too)
|
||||
// Cache only the network half (Spotify direct-url lookup); the final link object is rebuilt
|
||||
// per request so a name's result never depends on a stale gate decision. norm(name) -> url|null
|
||||
const spCache = new Map();
|
||||
|
||||
// --- Spotify (optional) -----------------------------------------------------
|
||||
// --- Spotify (optional): direct-artist-link lookup --------------------------
|
||||
let spToken = null;
|
||||
let spTokenExp = 0;
|
||||
async function spotifyToken() {
|
||||
@@ -48,72 +56,85 @@ async function spotifyToken() {
|
||||
return spToken;
|
||||
}
|
||||
|
||||
async function viaSpotify(name) {
|
||||
// The direct Spotify artist url for an exact name match, or null (no creds / no exact hit).
|
||||
async function spotifyDirect(name) {
|
||||
if (!USE_SPOTIFY) return null;
|
||||
const token = await spotifyToken();
|
||||
const url = `https://api.spotify.com/v1/search?type=artist&limit=5&q=${encodeURIComponent(name)}`;
|
||||
const r = await fetch(url, { headers: { authorization: `Bearer ${token}` } });
|
||||
if (!r.ok) throw new Error(`spotify search http ${r.status}`);
|
||||
const j = await r.json();
|
||||
const hit = pickArtist(name, j.artists?.items);
|
||||
return hit ? spotifyResult(hit) : null;
|
||||
return hit?.external_urls?.spotify ?? null;
|
||||
}
|
||||
|
||||
// --- MusicBrainz (zero-cred path) -------------------------------------------
|
||||
// MusicBrainz asks for <=1 request/second, so every request is serialized + spaced.
|
||||
const sleep = (ms) => new Promise((r) => setTimeout(r, ms));
|
||||
let mbGate = Promise.resolve();
|
||||
let mbLast = 0;
|
||||
function mbSchedule() {
|
||||
const gate = mbGate.then(async () => {
|
||||
const wait = 1100 - (Date.now() - mbLast);
|
||||
if (wait > 0) await sleep(wait);
|
||||
mbLast = Date.now();
|
||||
});
|
||||
mbGate = gate.catch(() => {}); // a failure never wedges the queue
|
||||
return gate;
|
||||
}
|
||||
async function mbFetchJson(url) {
|
||||
await mbSchedule();
|
||||
const r = await fetch(url, { headers: { "user-agent": UA, accept: "application/json" } });
|
||||
if (!r.ok) throw new Error(`musicbrainz http ${r.status}`);
|
||||
return r.json();
|
||||
}
|
||||
|
||||
async function viaMusicBrainz(name) {
|
||||
const q = encodeURIComponent(`artist:"${name}"`);
|
||||
const search = await mbFetchJson(`https://musicbrainz.org/ws/2/artist?fmt=json&limit=5&query=${q}`);
|
||||
const hit = pickArtist(name, search.artists, { minScore: 90 });
|
||||
if (!hit) return null;
|
||||
// Enrich with MB url-relations -> DIRECT bandcamp/spotify links (best-effort).
|
||||
let links = null;
|
||||
try {
|
||||
const detail = await mbFetchJson(`https://musicbrainz.org/ws/2/artist/${hit.id}?inc=url-rels&fmt=json`);
|
||||
links = relUrls(detail.relations);
|
||||
} catch (e) {
|
||||
console.error(`mb url-rels "${name}": ${e.message}`); // enrichment is best-effort
|
||||
}
|
||||
return mbResult(hit, links);
|
||||
}
|
||||
|
||||
// --- resolve one candidate (cached) -----------------------------------------
|
||||
async function resolveOne(name) {
|
||||
async function spotifyDirectCached(name) {
|
||||
const key = norm(name);
|
||||
if (!key) return null;
|
||||
if (cache.has(key)) return cache.get(key);
|
||||
if (spCache.has(key)) return spCache.get(key);
|
||||
const url = await spotifyDirect(name);
|
||||
spCache.set(key, url);
|
||||
return url;
|
||||
}
|
||||
|
||||
// --- LLM gate (optional): which candidates are artists/albums? --------------
|
||||
async function ollamaClassify(candidates) {
|
||||
const r = await fetch(`${OLLAMA_URL}/api/chat`, {
|
||||
method: "POST",
|
||||
headers: { "content-type": "application/json" },
|
||||
body: JSON.stringify({
|
||||
model: OLLAMA_MODEL,
|
||||
messages: buildClassifyMessages(candidates),
|
||||
stream: false,
|
||||
format: "json",
|
||||
options: { temperature: 0 },
|
||||
}),
|
||||
});
|
||||
if (!r.ok) throw new Error(`ollama http ${r.status}`);
|
||||
const j = await r.json();
|
||||
return parseClassification(j.message?.content ?? "", candidates);
|
||||
}
|
||||
|
||||
// --- resolve --------------------------------------------------------------
|
||||
// allowFallback: when the gate has already confirmed this name is an artist/album, a missing
|
||||
// Spotify direct link still yields a Google->Bandcamp fallback link (instead of null).
|
||||
async function resolveOne(name, allowFallback) {
|
||||
if (!norm(name)) return null;
|
||||
try {
|
||||
const result = USE_SPOTIFY ? await viaSpotify(name) : await viaMusicBrainz(name);
|
||||
cache.set(key, result); // cache hits AND confirmed misses
|
||||
return result;
|
||||
const sp = await spotifyDirectCached(name);
|
||||
if (sp) return linkResult(name, sp); // direct Spotify page
|
||||
return allowFallback ? linkResult(name, null) : null; // google fallback, or unconfirmed
|
||||
} catch (e) {
|
||||
console.error(`resolve "${name}": ${e.message}`); // transient -> don't cache
|
||||
console.error(`resolve "${name}": ${e.message}`); // transient -> caller sees null this time
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
async function resolveAll(candidates) {
|
||||
const uniq = [...new Set(candidates.map((c) => String(c).trim()).filter(Boolean))].slice(0, MAX_CANDIDATES);
|
||||
|
||||
// Gate: which names may link? null => no LLM gate, the Spotify exact-match is the gate
|
||||
// (only names actually on Spotify link). A Set => the LLM's verdict; names not in it are
|
||||
// dropped, names in it may use the Google fallback when not on Spotify.
|
||||
let confirmed = null;
|
||||
if (USE_LLM && uniq.length) {
|
||||
try {
|
||||
confirmed = await ollamaClassify(uniq);
|
||||
} catch (e) {
|
||||
console.error(`llm classify: ${e.message}`); // LLM down -> fall back to the Spotify gate
|
||||
confirmed = null;
|
||||
}
|
||||
}
|
||||
|
||||
const out = {};
|
||||
await Promise.all(uniq.map(async (name) => { out[name] = await resolveOne(name); }));
|
||||
await Promise.all(
|
||||
uniq.map(async (name) => {
|
||||
if (confirmed && !confirmed.has(norm(name))) {
|
||||
out[name] = null; // LLM said this is not an artist/album
|
||||
return;
|
||||
}
|
||||
out[name] = await resolveOne(name, confirmed != null);
|
||||
}),
|
||||
);
|
||||
return out;
|
||||
}
|
||||
|
||||
@@ -142,7 +163,7 @@ const server = createServer(async (req, res) => {
|
||||
if (req.method === "OPTIONS") { res.writeHead(204); return res.end(); }
|
||||
|
||||
if (req.method === "GET" && req.url === "/health") {
|
||||
return json(res, 200, { ok: true, provider: USE_SPOTIFY ? "spotify" : "musicbrainz", cached: cache.size });
|
||||
return json(res, 200, { ok: true, spotify: USE_SPOTIFY, llm: USE_LLM ? OLLAMA_MODEL : false, cached: spCache.size });
|
||||
}
|
||||
|
||||
if (req.method === "POST" && req.url === "/resolve") {
|
||||
@@ -159,5 +180,10 @@ const server = createServer(async (req, res) => {
|
||||
});
|
||||
|
||||
server.listen(PORT, () => {
|
||||
console.log(`resolver listening on http://localhost:${PORT} (provider: ${USE_SPOTIFY ? "spotify" : "musicbrainz"})`);
|
||||
const sp = USE_SPOTIFY ? "on" : "off";
|
||||
const llm = USE_LLM ? OLLAMA_MODEL : "off";
|
||||
console.log(`resolver listening on http://localhost:${PORT} (spotify: ${sp}, llm: ${llm})`);
|
||||
if (!USE_SPOTIFY && !USE_LLM) {
|
||||
console.warn(" ! no gate configured: set SPOTIFY_CLIENT_ID/SECRET and/or OLLAMA_URL, or nothing will link.");
|
||||
}
|
||||
});
|
||||
|
||||
Reference in New Issue
Block a user