{
  "name": "Cluster keywords into page groups using Google SERPs and Sheets",
  "nodes": [
    {
      "id": "run-manually",
      "name": "Run manually",
      "type": "n8n-nodes-base.manualTrigger",
      "typeVersion": 1,
      "position": [
        0,
        260
      ],
      "parameters": {}
    },
    {
      "id": "settings",
      "name": "Settings",
      "type": "n8n-nodes-base.set",
      "typeVersion": 3.4,
      "position": [
        240,
        260
      ],
      "parameters": {
        "assignments": {
          "assignments": [
            {
              "id": "spreadsheet_id",
              "name": "spreadsheet_id",
              "value": "REPLACE_WITH_SPREADSHEET_ID",
              "type": "string"
            },
            {
              "id": "location",
              "name": "location",
              "value": "",
              "type": "string"
            },
            {
              "id": "gl",
              "name": "gl",
              "value": "",
              "type": "string"
            },
            {
              "id": "max_items",
              "name": "max_items",
              "value": 20,
              "type": "number"
            },
            {
              "id": "min_results",
              "name": "min_results",
              "value": 5,
              "type": "number"
            },
            {
              "id": "shared_urls",
              "name": "shared_urls",
              "value": 3,
              "type": "number"
            }
          ]
        },
        "options": {}
      }
    },
    {
      "id": "validate-settings",
      "name": "Validate settings",
      "type": "n8n-nodes-base.code",
      "typeVersion": 2,
      "position": [
        480,
        260
      ],
      "parameters": {
        "mode": "runOnceForAllItems",
        "jsCode": "function config(input, kind) {\n  const c = { ...input, kind };\n  if (!/^[\\w-]{20,}$/.test(c.spreadsheet_id || '') || /REPLACE/.test(c.spreadsheet_id)) throw new Error('Set spreadsheet_id and create the input and Results tabs using the canvas headers.');\n  c.max_items = bound(c.max_items, 1, kind === 'cluster' ? 30 : 10, 'max_items');\n  c.run_id = new Date().toISOString();\n  if (kind !== 'maps') { c.location = clean(c.location); c.gl = clean(c.gl).toLowerCase(); if (c.gl && !/^[a-z]{2}$/.test(c.gl)) throw new Error('gl must be blank or a two-letter country code.'); }\n  if (kind === 'cluster') { c.min_results = bound(c.min_results, 3, 10, 'min_results'); c.shared_urls = bound(c.shared_urls, 2, 10, 'shared_urls'); if (c.shared_urls > c.min_results) throw new Error('shared_urls cannot exceed min_results'); }\n  if (kind === 'publishers' || kind === 'markets') {\n    c.max_pages = bound(c.max_pages, 1, 10, 'max_pages');\n    c.content_selector = clean(c.content_selector) || 'main, article, [role=\"main\"]';\n    if (typeof c.js_rendering !== 'boolean') throw new Error('js_rendering must be true or false. Rendering can increase API cost.');\n  }\n  if (kind === 'publishers') {\n    c.own_domain = host('https://' + clean(c.own_domain));\n    c.brand_aliases = lines(c.brand_aliases); c.competitor_names = lines(c.competitor_names);\n    c.excluded_domains = lines(c.excluded_domains).map(d => host('https://' + d));\n    if (!c.own_domain || !c.brand_aliases.length || !c.competitor_names.length || c.excluded_domains.some(d => !d)) throw new Error('Set your bare domain, brand aliases, competitor names and valid excluded domains.');\n    if ([...c.brand_aliases, ...c.competitor_names].some(s => s.length < 3 || s.length > 80)) throw new Error('Brand names must contain 3–80 characters.');\n    c.content_selector = clean(c.content_selector) || 'main';\n  }\n  if (kind === 'markets') {\n    c.location_b = clean(c.location_b); c.gl_b = clean(c.gl_b).toLowerCase();\n    if (!c.location || !c.location_b || !/^[a-z]{2}$/.test(c.gl) || !/^[a-z]{2}$/.test(c.gl_b) || c.gl === c.gl_b) throw new Error('Set two canonical locations and different two-letter country codes. Both markets must use English queries.');\n  }\n  if (kind === 'maps') c.country_calling_code = String(bound(c.country_calling_code, 1, 999, 'country_calling_code'));\n  return c;\n}\n\nfunction host(v) { return url(v)?.split('/')[2] || ''; }\n\nfunction lines(v) { return [...new Set(String(v || '').split('\\n').map(clean).filter(Boolean))]; }\n\nfunction markets(c, searches, pages) {\n  const ids = [...new Set(searches.map(s => s.topic_id))];\n  return ids.map(id => {\n    const a = searches.find(s => s.topic_id === id && s.market === 'A'), b = searches.find(s => s.topic_id === id && s.market === 'B');\n    const good = a?.status === 'observed' && b?.status === 'observed' && a.returned >= 5 && b.returned >= 5;\n    const urls = good ? overlap(a, b) : [];\n    const domains = good ? [...new Set(a.results.map(r => host(r.url)))].filter(d => b.results.some(r => host(r.url) === d)) : [];\n    const ownPages = pages.filter(p => [a, b].some(s => s?.results.some(r => r.url === p.url)));\n    const bothRead = [a, b].every(s => s?.results.some(r => ownPages.some(p => p.url === r.url && p.page_status === 'read')));\n    return row(c, key([id, a?.context, b?.context]), id, !good ? 'insufficient_serp_evidence' : bothRead ? 'ready_for_localization_review' : 'serp_only_review',\n      good ? 'Compare intent and page formats before choosing one adapted page or separate market pages. URL overlap alone cannot decide.' : 'Failed or sparse market results; do not infer an intent difference.',\n      { market_a: a, market_b: b, shared_urls: urls, shared_domains: domains, pages: ownPages.map(p => ({ url: p.url, status: p.page_status, reason: p.reason, title: p.title, h1: p.h1, excerpt: p.text.slice(0, 2500), excerpt_truncated: p.text.length > 2500 || p.truncated, metadata: p.page_metadata })),\n        pages_not_read: [...new Set([...(a?.results || []), ...(b?.results || [])].map(r => r.url))].filter(u => !ownPages.some(p => p.url === u && p.page_status === 'read')),\n        editorial_questions: ['Do both queries describe the same customer need?', 'Which local entities, terminology, currencies and page formats appear in the evidence?', 'Would one adapted page serve both markets, or does intent justify separate pages?', 'Which unvisited or truncated sources need checking?'],\n        limits: 'English-language comparison, one response per query and market. This does not measure search demand, prove intent, or recommend hreflang implementation.' });\n  });\n}\n\nfunction metadata(r) { const m = r?.requestMetadata || {}; return { id: clean(m.id), status: clean(m.status), html: clean(m.html), json: clean(m.json), search_url: clean(m.url) }; }\n\nfunction overlap(a, b) { const s = new Set(b.results.map(r => r.url)); return a.results.map(r => r.url).filter(u => s.has(u)); }\n\nfunction publishers(c, sources, pages, searches) {\n  const result = searches.map(s => row(c, key(['prompt', c.own_domain, s.context]), s.query, s.status, 'AI Mode citation inventory; citations are not brand mentions.', s));\n  for (const s of sources) {\n    const p = pages.find(x => x.url === s.url);\n    const known = p?.page_status === 'read';\n    const own = known ? mentions(p.text, c.brand_aliases) : [], competitors = known ? mentions(p.text, c.competitor_names) : [];\n    const ownLinks = known ? p.links.filter(u => domainMatches(host(u), c.own_domain)) : [];\n    const status = s.status !== 'fetch' ? s.status : !known ? 'page_unavailable' : own.length || ownLinks.length ? 'brand_already_observed' : competitors.length ? 'review_publisher_fit' : 'no_competitor_evidence';\n    result.push(row(c, key(['source', c.own_domain, s.url, c.location, c.gl]), s.url, status,\n      status === 'review_publisher_fit' ? 'Competitors appear in the inspected text; verify publisher ownership, topical fit and full-page coverage before outreach.' : 'Review retained citation and page evidence. No contact discovery or messages are sent.',\n      { source: s, page_status: p?.page_status || 'not_fetched', page_reason: p?.reason || '', title: p?.title || '', h1: p?.h1 || '', page_metadata: p?.page_metadata || {}, text_truncated: !!p?.truncated, own_mentions: own, competitor_mentions: competitors, own_links: ownLinks, links_truncated: !!p?.links_truncated, absence_note: 'No match in extracted content is not proof that the full page or publisher omits your brand. Known-domain exclusions do not establish independent ownership.', configured_brand_aliases: c.brand_aliases, configured_competitors: c.competitor_names }));\n  }\n  return result;\n}\n\nfunction row(c, id, topic, status, summary, evidence) {\n  const extra = c.kind === 'cluster' ? {\n    group_keywords: (evidence.cluster || []).join('\\n'), returned_results: evidence.search?.returned || 0,\n    cross_group_matches: (evidence.pairs || []).filter(p => !p.same_group && p.shared_urls.length >= c.shared_urls).map(p => p.query).join('\\n'),\n  } : c.kind === 'publishers' ? {\n    citing_prompts: (evidence.source?.prompts || [evidence.query]).filter(Boolean).join('\\n'),\n    competitors_observed: (evidence.competitor_mentions || []).map(m => m.name + ' — ' + m.excerpt).join('\\n'),\n    brand_evidence: (evidence.own_mentions || []).map(m => m.excerpt).concat(evidence.own_links || []).join('\\n'),\n    source_url: evidence.source?.url || '',\n  } : c.kind === 'maps' ? {\n    differences: (evidence.fields || []).filter(f => f.status === 'review_difference').map(f => f.field + ': expected ' + f.expected + ' | observed ' + f.observed).join('\\n'),\n    unknown_fields: (evidence.fields || []).filter(f => f.status === 'unknown').map(f => f.field).join(', '),\n    resolved_fields: (evidence.transitions || []).filter(f => f.resolved).map(f => f.field).join(', '), place_id: evidence.requested_place_id,\n  } : {\n    market_a_query: evidence.market_a?.query || '', market_b_query: evidence.market_b?.query || '',\n    shared_url_count: status === 'insufficient_serp_evidence' ? '' : evidence.shared_urls.length,\n    shared_domains: (evidence.shared_domains || []).join('\\n'),\n    page_samples: (evidence.pages || []).map(p => p.url + '\\n' + p.status + ' | ' + p.title + '\\n' + p.excerpt).join('\\n\\n'),\n  };\n  return { result_key: key([c.kind, id]), checked_at: c.run_id, topic, status, summary, ...extra, evidence_json: cell(evidence) };\n}\n\nfunction url(v) {\n  const m = /^https?:\\/\\/([a-z0-9.-]+)(?::(443|80))?(\\/[^\\s#]*)?(?:#.*)?$/i.exec(clean(v));\n  if (!m || !m[1].includes('.') || /^[\\d.]+$/.test(m[1]) || /(^|\\.)(localhost|local|internal|invalid|test)$/.test(m[1])) return '';\n  const [path, query] = (m[3] || '/').split('?');\n  const params = (query || '').split('&').filter(p => p && !/^(utm_[^=]*|gclid|fbclid)=/i.test(p)).sort();\n  return 'https://' + m[1].toLowerCase().replace(/^www\\./, '') + (path === '/' ? '/' : path.replace(/\\/$/, '')) + (params.length ? '?' + params.join('&') : '');\n}\n\nfunction bound(v, min, max, name) { const n = Number(v); if (!Number.isInteger(n) || n < min || n > max) throw new Error(name + ' must be an integer from ' + min + ' to ' + max); return n; }\n\nfunction cell(v) { const s = JSON.stringify(v); if (s.length > 44000) throw new Error('Evidence exceeds the Sheets cell limit. Reduce the batch size.'); return s; }\n\nfunction citations(p, r) {\n  const base = { ...p, metadata: metadata(r), references: [] };\n  if (r.error || r.requestMetadata?.status !== 'ok' || !Array.isArray(r.references)) return { ...base, status: 'unknown' };\n  base.references = r.references.map(x => ({ url: url(x.link || x.url), source_url: clean(x.link || x.url), title: clean(x.title) })).filter(x => x.url);\n  return { ...base, status: base.references.length ? 'observed' : 'no_references_returned' };\n}\n\nfunction clean(v) { return typeof v === 'string' ? v.replace(/\\s+/g, ' ').trim() : ''; }\n\nfunction domainMatches(a, b) { return !!a && !!b && (a === b || a.endsWith('.' + b)); }\n\nfunction key(parts) { return JSON.stringify(parts); }\n\nfunction mentions(text, names) {\n  return names.flatMap(name => {\n    const escaped = name.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\$&');\n    const m = new RegExp('(?:^|[^\\\\p{L}\\\\p{N}])(' + escaped + ')(?=$|[^\\\\p{L}\\\\p{N}])', 'iu').exec(text);\n    return m ? [{ name, excerpt: text.slice(Math.max(0, m.index - 100), m.index + name.length + 180) }] : [];\n  });\n}\n\nreturn [{json:config($input.first().json,'cluster')}];"
      }
    },
    {
      "id": "read-input",
      "name": "Read input",
      "type": "n8n-nodes-base.googleSheets",
      "typeVersion": 4.5,
      "position": [
        720,
        260
      ],
      "parameters": {
        "resource": "sheet",
        "operation": "read",
        "documentId": {
          "__rl": true,
          "mode": "id",
          "value": "={{ $('Validate settings').first().json.spreadsheet_id }}"
        },
        "sheetName": {
          "__rl": true,
          "mode": "name",
          "value": "Keywords"
        },
        "options": {}
      },
      "alwaysOutputData": true
    },
    {
      "id": "validate-input-batch",
      "name": "Validate input batch",
      "type": "n8n-nodes-base.code",
      "typeVersion": 2,
      "position": [
        960,
        260
      ],
      "parameters": {
        "mode": "runOnceForAllItems",
        "jsCode": "function plan(c, rows) {\n  const required = c.kind === 'maps' ? ['branch_id', 'place_id', 'expected_name', 'expected_address'] : c.kind === 'markets' ? ['topic_id', 'query_a', 'query_b'] : ['query'];\n  const active = readRows(rows, required, required[0]);\n  if (active.length > c.max_items) throw new Error('Input exceeds max_items. Select a smaller batch; rows are never silently dropped.');\n  if (c.kind === 'maps' && new Set(active.map(r => clean(r.place_id))).size !== active.length) throw new Error('A place_id may only belong to one branch.');\n  return active.flatMap(r => {\n    if (c.kind === 'maps') {\n      if (!/^[\\w-]{10,}$/.test(clean(r.place_id))) throw new Error('Use the Google place ID, not a Maps URL or data ID.');\n      return [{ ...r, context: key(['maps', clean(r.branch_id), clean(r.place_id), c.country_calling_code]) }];\n    }\n    const queries = c.kind === 'markets' ? [r.query_a, r.query_b] : [r.query];\n    return queries.map((q, i) => {\n      const query = clean(q); if (!query || query.length > 250) throw new Error('Queries must contain 1–250 characters.');\n      const location = i ? c.location_b : c.location, gl = i ? c.gl_b : c.gl;\n      const context = key([query, location, gl, 'desktop', c.kind === 'markets' ? 'en' : '']);\n      return { query, topic_id: clean(r.topic_id), market: i ? 'B' : 'A', location, gl, context,\n        fields: { deviceType: 'desktop', ...(location ? { location } : {}), ...(gl ? { gl } : {}), ...(c.kind === 'markets' ? { hl: 'en' } : {}) } };\n    });\n  });\n}\n\nfunction readRows(rows, required, idField) {\n  const active = rows.filter(r => required.some(k => clean(String(r[k] ?? ''))));\n  if (!active.length) throw new Error('No input rows. Required headers: ' + required.join(', '));\n  const seen = new Set();\n  for (const r of active) {\n    if (required.some(k => !clean(String(r[k] ?? '')))) throw new Error('Incomplete input row. Required: ' + required.join(', '));\n    const id = clean(String(r[idField])).normalize('NFKC').toLowerCase();\n    if (seen.has(id)) throw new Error('Duplicate input identifier: ' + id); seen.add(id);\n  }\n  return active;\n}\n\nfunction row(c, id, topic, status, summary, evidence) {\n  const extra = c.kind === 'cluster' ? {\n    group_keywords: (evidence.cluster || []).join('\\n'), returned_results: evidence.search?.returned || 0,\n    cross_group_matches: (evidence.pairs || []).filter(p => !p.same_group && p.shared_urls.length >= c.shared_urls).map(p => p.query).join('\\n'),\n  } : c.kind === 'publishers' ? {\n    citing_prompts: (evidence.source?.prompts || [evidence.query]).filter(Boolean).join('\\n'),\n    competitors_observed: (evidence.competitor_mentions || []).map(m => m.name + ' — ' + m.excerpt).join('\\n'),\n    brand_evidence: (evidence.own_mentions || []).map(m => m.excerpt).concat(evidence.own_links || []).join('\\n'),\n    source_url: evidence.source?.url || '',\n  } : c.kind === 'maps' ? {\n    differences: (evidence.fields || []).filter(f => f.status === 'review_difference').map(f => f.field + ': expected ' + f.expected + ' | observed ' + f.observed).join('\\n'),\n    unknown_fields: (evidence.fields || []).filter(f => f.status === 'unknown').map(f => f.field).join(', '),\n    resolved_fields: (evidence.transitions || []).filter(f => f.resolved).map(f => f.field).join(', '), place_id: evidence.requested_place_id,\n  } : {\n    market_a_query: evidence.market_a?.query || '', market_b_query: evidence.market_b?.query || '',\n    shared_url_count: status === 'insufficient_serp_evidence' ? '' : evidence.shared_urls.length,\n    shared_domains: (evidence.shared_domains || []).join('\\n'),\n    page_samples: (evidence.pages || []).map(p => p.url + '\\n' + p.status + ' | ' + p.title + '\\n' + p.excerpt).join('\\n\\n'),\n  };\n  return { result_key: key([c.kind, id]), checked_at: c.run_id, topic, status, summary, ...extra, evidence_json: cell(evidence) };\n}\n\nfunction url(v) {\n  const m = /^https?:\\/\\/([a-z0-9.-]+)(?::(443|80))?(\\/[^\\s#]*)?(?:#.*)?$/i.exec(clean(v));\n  if (!m || !m[1].includes('.') || /^[\\d.]+$/.test(m[1]) || /(^|\\.)(localhost|local|internal|invalid|test)$/.test(m[1])) return '';\n  const [path, query] = (m[3] || '/').split('?');\n  const params = (query || '').split('&').filter(p => p && !/^(utm_[^=]*|gclid|fbclid)=/i.test(p)).sort();\n  return 'https://' + m[1].toLowerCase().replace(/^www\\./, '') + (path === '/' ? '/' : path.replace(/\\/$/, '')) + (params.length ? '?' + params.join('&') : '');\n}\n\nfunction cell(v) { const s = JSON.stringify(v); if (s.length > 44000) throw new Error('Evidence exceeds the Sheets cell limit. Reduce the batch size.'); return s; }\n\nfunction clean(v) { return typeof v === 'string' ? v.replace(/\\s+/g, ' ').trim() : ''; }\n\nfunction key(parts) { return JSON.stringify(parts); }\n\nfunction markets(c, searches, pages) {\n  const ids = [...new Set(searches.map(s => s.topic_id))];\n  return ids.map(id => {\n    const a = searches.find(s => s.topic_id === id && s.market === 'A'), b = searches.find(s => s.topic_id === id && s.market === 'B');\n    const good = a?.status === 'observed' && b?.status === 'observed' && a.returned >= 5 && b.returned >= 5;\n    const urls = good ? overlap(a, b) : [];\n    const domains = good ? [...new Set(a.results.map(r => host(r.url)))].filter(d => b.results.some(r => host(r.url) === d)) : [];\n    const ownPages = pages.filter(p => [a, b].some(s => s?.results.some(r => r.url === p.url)));\n    const bothRead = [a, b].every(s => s?.results.some(r => ownPages.some(p => p.url === r.url && p.page_status === 'read')));\n    return row(c, key([id, a?.context, b?.context]), id, !good ? 'insufficient_serp_evidence' : bothRead ? 'ready_for_localization_review' : 'serp_only_review',\n      good ? 'Compare intent and page formats before choosing one adapted page or separate market pages. URL overlap alone cannot decide.' : 'Failed or sparse market results; do not infer an intent difference.',\n      { market_a: a, market_b: b, shared_urls: urls, shared_domains: domains, pages: ownPages.map(p => ({ url: p.url, status: p.page_status, reason: p.reason, title: p.title, h1: p.h1, excerpt: p.text.slice(0, 2500), excerpt_truncated: p.text.length > 2500 || p.truncated, metadata: p.page_metadata })),\n        pages_not_read: [...new Set([...(a?.results || []), ...(b?.results || [])].map(r => r.url))].filter(u => !ownPages.some(p => p.url === u && p.page_status === 'read')),\n        editorial_questions: ['Do both queries describe the same customer need?', 'Which local entities, terminology, currencies and page formats appear in the evidence?', 'Would one adapted page serve both markets, or does intent justify separate pages?', 'Which unvisited or truncated sources need checking?'],\n        limits: 'English-language comparison, one response per query and market. This does not measure search demand, prove intent, or recommend hreflang implementation.' });\n  });\n}\n\nfunction metadata(r) { const m = r?.requestMetadata || {}; return { id: clean(m.id), status: clean(m.status), html: clean(m.html), json: clean(m.json), search_url: clean(m.url) }; }\n\nfunction overlap(a, b) { const s = new Set(b.results.map(r => r.url)); return a.results.map(r => r.url).filter(u => s.has(u)); }\n\nfunction publishers(c, sources, pages, searches) {\n  const result = searches.map(s => row(c, key(['prompt', c.own_domain, s.context]), s.query, s.status, 'AI Mode citation inventory; citations are not brand mentions.', s));\n  for (const s of sources) {\n    const p = pages.find(x => x.url === s.url);\n    const known = p?.page_status === 'read';\n    const own = known ? mentions(p.text, c.brand_aliases) : [], competitors = known ? mentions(p.text, c.competitor_names) : [];\n    const ownLinks = known ? p.links.filter(u => domainMatches(host(u), c.own_domain)) : [];\n    const status = s.status !== 'fetch' ? s.status : !known ? 'page_unavailable' : own.length || ownLinks.length ? 'brand_already_observed' : competitors.length ? 'review_publisher_fit' : 'no_competitor_evidence';\n    result.push(row(c, key(['source', c.own_domain, s.url, c.location, c.gl]), s.url, status,\n      status === 'review_publisher_fit' ? 'Competitors appear in the inspected text; verify publisher ownership, topical fit and full-page coverage before outreach.' : 'Review retained citation and page evidence. No contact discovery or messages are sent.',\n      { source: s, page_status: p?.page_status || 'not_fetched', page_reason: p?.reason || '', title: p?.title || '', h1: p?.h1 || '', page_metadata: p?.page_metadata || {}, text_truncated: !!p?.truncated, own_mentions: own, competitor_mentions: competitors, own_links: ownLinks, links_truncated: !!p?.links_truncated, absence_note: 'No match in extracted content is not proof that the full page or publisher omits your brand. Known-domain exclusions do not establish independent ownership.', configured_brand_aliases: c.brand_aliases, configured_competitors: c.competitor_names }));\n  }\n  return result;\n}\n\nfunction citations(p, r) {\n  const base = { ...p, metadata: metadata(r), references: [] };\n  if (r.error || r.requestMetadata?.status !== 'ok' || !Array.isArray(r.references)) return { ...base, status: 'unknown' };\n  base.references = r.references.map(x => ({ url: url(x.link || x.url), source_url: clean(x.link || x.url), title: clean(x.title) })).filter(x => x.url);\n  return { ...base, status: base.references.length ? 'observed' : 'no_references_returned' };\n}\n\nfunction domainMatches(a, b) { return !!a && !!b && (a === b || a.endsWith('.' + b)); }\n\nfunction host(v) { return url(v)?.split('/')[2] || ''; }\n\nfunction mentions(text, names) {\n  return names.flatMap(name => {\n    const escaped = name.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\$&');\n    const m = new RegExp('(?:^|[^\\\\p{L}\\\\p{N}])(' + escaped + ')(?=$|[^\\\\p{L}\\\\p{N}])', 'iu').exec(text);\n    return m ? [{ name, excerpt: text.slice(Math.max(0, m.index - 100), m.index + name.length + 180) }] : [];\n  });\n}\n\nreturn plan($('Validate settings').first().json,$input.all().map(x=>x.json)).map(json=>({json}));"
      }
    },
    {
      "id": "prepare-queue-read",
      "name": "Prepare queue read",
      "type": "n8n-nodes-base.code",
      "typeVersion": 2,
      "position": [
        1200,
        260
      ],
      "parameters": {
        "mode": "runOnceForAllItems",
        "jsCode": "\n\nreturn [{json:{ready:true}}];"
      }
    },
    {
      "id": "read-results",
      "name": "Read Results",
      "type": "n8n-nodes-base.googleSheets",
      "typeVersion": 4.5,
      "position": [
        1440,
        260
      ],
      "parameters": {
        "resource": "sheet",
        "operation": "read",
        "documentId": {
          "__rl": true,
          "mode": "id",
          "value": "={{ $('Validate settings').first().json.spreadsheet_id }}"
        },
        "sheetName": {
          "__rl": true,
          "mode": "name",
          "value": "Results"
        },
        "options": {}
      },
      "alwaysOutputData": true
    },
    {
      "id": "check-saved-keys",
      "name": "Check saved keys",
      "type": "n8n-nodes-base.code",
      "typeVersion": 2,
      "position": [
        1680,
        260
      ],
      "parameters": {
        "mode": "runOnceForAllItems",
        "jsCode": "function previousRows(rows) {\n  const seen = new Set();\n  for (const r of rows.filter(r => r.result_key)) { if (seen.has(r.result_key)) throw new Error('Duplicate result_key in Results. Resolve before running.'); seen.add(r.result_key); }\n  return rows;\n}\n\npreviousRows($input.all().map(x=>x.json)); return $('Validate input batch').all().map(x=>({json:x.json}));"
      }
    },
    {
      "id": "fetch-search-evidence",
      "name": "Fetch search evidence",
      "type": "@hasdata/n8n-nodes-hasdata.hasData",
      "typeVersion": 1,
      "position": [
        1920,
        260
      ],
      "parameters": {
        "resource": "google_serp",
        "operation": "serp_light",
        "q": "={{ $json.query }}",
        "additionalFields": "={{ $json.fields }}"
      },
      "retryOnFail": false,
      "onError": "continueRegularOutput"
    },
    {
      "id": "attach-search-context",
      "name": "Attach search context",
      "type": "n8n-nodes-base.code",
      "typeVersion": 2,
      "position": [
        2160,
        260
      ],
      "parameters": {
        "mode": "runOnceForAllItems",
        "jsCode": "function serp(p, r) {\n  const base = { ...p, metadata: metadata(r), results: [] };\n  if (r.error || r.requestMetadata?.status !== 'ok' || !Array.isArray(r.organicResults) || !r.organicResults.length) return { ...base, status: 'unknown', reason: 'No usable organic results' };\n  const seen = new Set();\n  for (const v of r.organicResults.slice(0, 10)) {\n    const u = url(v.link);\n    if (!u || !Number.isInteger(v.position) || v.position < 1 || seen.has(v.position)) return { ...base, status: 'unknown', reason: 'Malformed result URLs or positions' };\n    seen.add(v.position); base.results.push({ url: u, source_url: v.link, position: v.position, title: clean(v.title), snippet: clean(v.snippet).slice(0, 1000) });\n  }\n  base.results = base.results.filter((r, i, all) => all.findIndex(x => x.url === r.url) === i);\n  return { ...base, status: 'observed', returned: base.results.length };\n}\n\nfunction url(v) {\n  const m = /^https?:\\/\\/([a-z0-9.-]+)(?::(443|80))?(\\/[^\\s#]*)?(?:#.*)?$/i.exec(clean(v));\n  if (!m || !m[1].includes('.') || /^[\\d.]+$/.test(m[1]) || /(^|\\.)(localhost|local|internal|invalid|test)$/.test(m[1])) return '';\n  const [path, query] = (m[3] || '/').split('?');\n  const params = (query || '').split('&').filter(p => p && !/^(utm_[^=]*|gclid|fbclid)=/i.test(p)).sort();\n  return 'https://' + m[1].toLowerCase().replace(/^www\\./, '') + (path === '/' ? '/' : path.replace(/\\/$/, '')) + (params.length ? '?' + params.join('&') : '');\n}\n\nfunction clean(v) { return typeof v === 'string' ? v.replace(/\\s+/g, ' ').trim() : ''; }\n\nfunction metadata(r) { const m = r?.requestMetadata || {}; return { id: clean(m.id), status: clean(m.status), html: clean(m.html), json: clean(m.json), search_url: clean(m.url) }; }\n\nconst plans=$('Check saved keys').all(); const items=$input.all(); const seen=new Set(); const paired=items.map(item=>{const pair=Array.isArray(item.pairedItem)?item.pairedItem[0]:item.pairedItem; const i=pair?.item; if(!Number.isInteger(i)||!plans[i]||seen.has(i))throw new Error('Missing or ambiguous API item pairing. Nothing saved.'); seen.add(i); return {p:plans[i].json,r:item.json};}); if(seen.size!==plans.length)throw new Error('Missing API responses. Nothing saved.');return paired.map(({p,r})=>({json:serp(p,r)}));"
      }
    },
    {
      "id": "build-review-rows",
      "name": "Build review rows",
      "type": "n8n-nodes-base.code",
      "typeVersion": 2,
      "position": [
        2400,
        260
      ],
      "parameters": {
        "mode": "runOnceForAllItems",
        "jsCode": "function clusters(c, evidence) {\n  const eligible = evidence.filter(e => e.status === 'observed' && e.returned >= c.min_results).sort((a, b) => a.query.localeCompare(b.query));\n  const groups = [];\n  for (const e of eligible) { const group = groups.find(g => g.every(other => overlap(e, other).length >= c.shared_urls)); if (group) group.push(e); else groups.push([e]); }\n  return evidence.map(e => {\n    const group = groups.find(g => g.includes(e));\n    const pairs = eligible.filter(x => x !== e).map(x => ({ query: x.query, shared_urls: overlap(e, x), same_group: !!group?.includes(x) }));\n    const status = !group ? 'insufficient_evidence' : group.length > 1 ? 'candidate_page_group' : 'separate_in_this_batch';\n    return row(c, key([e.context, c.shared_urls, c.min_results]), e.query, status,\n      !group ? 'Failed or sparse SERP; no clustering conclusion.' : 'Candidate group: ' + group.map(x => x.query).join(' | '),\n      { search: e, cluster: group?.map(x => x.query) || [], pairs, rule: 'Every pair in a group meets the shared-URL threshold; deterministic greedy grouping, not a unique optimal partition.', shared_url_threshold: c.shared_urls, min_results: c.min_results, input_queries: evidence.map(x => x.query), review: 'Validate search intent and page scope before combining keywords. Cross-group matches can occur.' });\n  });\n}\n\nfunction key(parts) { return JSON.stringify(parts); }\n\nfunction overlap(a, b) { const s = new Set(b.results.map(r => r.url)); return a.results.map(r => r.url).filter(u => s.has(u)); }\n\nfunction row(c, id, topic, status, summary, evidence) {\n  const extra = c.kind === 'cluster' ? {\n    group_keywords: (evidence.cluster || []).join('\\n'), returned_results: evidence.search?.returned || 0,\n    cross_group_matches: (evidence.pairs || []).filter(p => !p.same_group && p.shared_urls.length >= c.shared_urls).map(p => p.query).join('\\n'),\n  } : c.kind === 'publishers' ? {\n    citing_prompts: (evidence.source?.prompts || [evidence.query]).filter(Boolean).join('\\n'),\n    competitors_observed: (evidence.competitor_mentions || []).map(m => m.name + ' — ' + m.excerpt).join('\\n'),\n    brand_evidence: (evidence.own_mentions || []).map(m => m.excerpt).concat(evidence.own_links || []).join('\\n'),\n    source_url: evidence.source?.url || '',\n  } : c.kind === 'maps' ? {\n    differences: (evidence.fields || []).filter(f => f.status === 'review_difference').map(f => f.field + ': expected ' + f.expected + ' | observed ' + f.observed).join('\\n'),\n    unknown_fields: (evidence.fields || []).filter(f => f.status === 'unknown').map(f => f.field).join(', '),\n    resolved_fields: (evidence.transitions || []).filter(f => f.resolved).map(f => f.field).join(', '), place_id: evidence.requested_place_id,\n  } : {\n    market_a_query: evidence.market_a?.query || '', market_b_query: evidence.market_b?.query || '',\n    shared_url_count: status === 'insufficient_serp_evidence' ? '' : evidence.shared_urls.length,\n    shared_domains: (evidence.shared_domains || []).join('\\n'),\n    page_samples: (evidence.pages || []).map(p => p.url + '\\n' + p.status + ' | ' + p.title + '\\n' + p.excerpt).join('\\n\\n'),\n  };\n  return { result_key: key([c.kind, id]), checked_at: c.run_id, topic, status, summary, ...extra, evidence_json: cell(evidence) };\n}\n\nfunction url(v) {\n  const m = /^https?:\\/\\/([a-z0-9.-]+)(?::(443|80))?(\\/[^\\s#]*)?(?:#.*)?$/i.exec(clean(v));\n  if (!m || !m[1].includes('.') || /^[\\d.]+$/.test(m[1]) || /(^|\\.)(localhost|local|internal|invalid|test)$/.test(m[1])) return '';\n  const [path, query] = (m[3] || '/').split('?');\n  const params = (query || '').split('&').filter(p => p && !/^(utm_[^=]*|gclid|fbclid)=/i.test(p)).sort();\n  return 'https://' + m[1].toLowerCase().replace(/^www\\./, '') + (path === '/' ? '/' : path.replace(/\\/$/, '')) + (params.length ? '?' + params.join('&') : '');\n}\n\nfunction cell(v) { const s = JSON.stringify(v); if (s.length > 44000) throw new Error('Evidence exceeds the Sheets cell limit. Reduce the batch size.'); return s; }\n\nfunction clean(v) { return typeof v === 'string' ? v.replace(/\\s+/g, ' ').trim() : ''; }\n\nfunction publishers(c, sources, pages, searches) {\n  const result = searches.map(s => row(c, key(['prompt', c.own_domain, s.context]), s.query, s.status, 'AI Mode citation inventory; citations are not brand mentions.', s));\n  for (const s of sources) {\n    const p = pages.find(x => x.url === s.url);\n    const known = p?.page_status === 'read';\n    const own = known ? mentions(p.text, c.brand_aliases) : [], competitors = known ? mentions(p.text, c.competitor_names) : [];\n    const ownLinks = known ? p.links.filter(u => domainMatches(host(u), c.own_domain)) : [];\n    const status = s.status !== 'fetch' ? s.status : !known ? 'page_unavailable' : own.length || ownLinks.length ? 'brand_already_observed' : competitors.length ? 'review_publisher_fit' : 'no_competitor_evidence';\n    result.push(row(c, key(['source', c.own_domain, s.url, c.location, c.gl]), s.url, status,\n      status === 'review_publisher_fit' ? 'Competitors appear in the inspected text; verify publisher ownership, topical fit and full-page coverage before outreach.' : 'Review retained citation and page evidence. No contact discovery or messages are sent.',\n      { source: s, page_status: p?.page_status || 'not_fetched', page_reason: p?.reason || '', title: p?.title || '', h1: p?.h1 || '', page_metadata: p?.page_metadata || {}, text_truncated: !!p?.truncated, own_mentions: own, competitor_mentions: competitors, own_links: ownLinks, links_truncated: !!p?.links_truncated, absence_note: 'No match in extracted content is not proof that the full page or publisher omits your brand. Known-domain exclusions do not establish independent ownership.', configured_brand_aliases: c.brand_aliases, configured_competitors: c.competitor_names }));\n  }\n  return result;\n}\n\nfunction citations(p, r) {\n  const base = { ...p, metadata: metadata(r), references: [] };\n  if (r.error || r.requestMetadata?.status !== 'ok' || !Array.isArray(r.references)) return { ...base, status: 'unknown' };\n  base.references = r.references.map(x => ({ url: url(x.link || x.url), source_url: clean(x.link || x.url), title: clean(x.title) })).filter(x => x.url);\n  return { ...base, status: base.references.length ? 'observed' : 'no_references_returned' };\n}\n\nfunction domainMatches(a, b) { return !!a && !!b && (a === b || a.endsWith('.' + b)); }\n\nfunction host(v) { return url(v)?.split('/')[2] || ''; }\n\nfunction mentions(text, names) {\n  return names.flatMap(name => {\n    const escaped = name.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\$&');\n    const m = new RegExp('(?:^|[^\\\\p{L}\\\\p{N}])(' + escaped + ')(?=$|[^\\\\p{L}\\\\p{N}])', 'iu').exec(text);\n    return m ? [{ name, excerpt: text.slice(Math.max(0, m.index - 100), m.index + name.length + 180) }] : [];\n  });\n}\n\nfunction metadata(r) { const m = r?.requestMetadata || {}; return { id: clean(m.id), status: clean(m.status), html: clean(m.html), json: clean(m.json), search_url: clean(m.url) }; }\n\nreturn clusters($('Validate settings').first().json,$input.all().map(x=>x.json)).map(json=>({json}));"
      }
    },
    {
      "id": "save-review-queue",
      "name": "Save review queue",
      "type": "n8n-nodes-base.googleSheets",
      "typeVersion": 4.5,
      "position": [
        2640,
        260
      ],
      "parameters": {
        "resource": "sheet",
        "operation": "appendOrUpdate",
        "documentId": {
          "__rl": true,
          "mode": "id",
          "value": "={{ $('Validate settings').first().json.spreadsheet_id }}"
        },
        "sheetName": {
          "__rl": true,
          "mode": "name",
          "value": "Results"
        },
        "options": {
          "cellFormat": "RAW"
        },
        "columns": {
          "mappingMode": "defineBelow",
          "value": {
            "result_key": "={{ $json.result_key }}",
            "checked_at": "={{ $json.checked_at }}",
            "topic": "={{ $json.topic }}",
            "status": "={{ $json.status }}",
            "summary": "={{ $json.summary }}",
            "group_keywords": "={{ $json.group_keywords }}",
            "returned_results": "={{ $json.returned_results }}",
            "cross_group_matches": "={{ $json.cross_group_matches }}",
            "evidence_json": "={{ $json.evidence_json }}"
          },
          "matchingColumns": [
            "result_key"
          ],
          "schema": [
            {
              "id": "result_key",
              "displayName": "result_key",
              "required": false,
              "defaultMatch": true,
              "display": true,
              "type": "string",
              "canBeUsedToMatch": true
            },
            {
              "id": "checked_at",
              "displayName": "checked_at",
              "required": false,
              "defaultMatch": false,
              "display": true,
              "type": "string",
              "canBeUsedToMatch": true
            },
            {
              "id": "topic",
              "displayName": "topic",
              "required": false,
              "defaultMatch": false,
              "display": true,
              "type": "string",
              "canBeUsedToMatch": true
            },
            {
              "id": "status",
              "displayName": "status",
              "required": false,
              "defaultMatch": false,
              "display": true,
              "type": "string",
              "canBeUsedToMatch": true
            },
            {
              "id": "summary",
              "displayName": "summary",
              "required": false,
              "defaultMatch": false,
              "display": true,
              "type": "string",
              "canBeUsedToMatch": true
            },
            {
              "id": "group_keywords",
              "displayName": "group_keywords",
              "required": false,
              "defaultMatch": false,
              "display": true,
              "type": "string",
              "canBeUsedToMatch": true
            },
            {
              "id": "returned_results",
              "displayName": "returned_results",
              "required": false,
              "defaultMatch": false,
              "display": true,
              "type": "string",
              "canBeUsedToMatch": true
            },
            {
              "id": "cross_group_matches",
              "displayName": "cross_group_matches",
              "required": false,
              "defaultMatch": false,
              "display": true,
              "type": "string",
              "canBeUsedToMatch": true
            },
            {
              "id": "evidence_json",
              "displayName": "evidence_json",
              "required": false,
              "defaultMatch": false,
              "display": true,
              "type": "string",
              "canBeUsedToMatch": true
            }
          ],
          "attemptToConvertTypes": false,
          "convertFieldsToString": false
        }
      }
    },
    {
      "id": "section-1",
      "name": "Section 1",
      "type": "n8n-nodes-base.stickyNote",
      "typeVersion": 1,
      "position": [
        -30,
        60
      ],
      "parameters": {
        "content": "## Configure and read inputs\nCreate Keywords with headers: `query`.",
        "width": 940,
        "height": 440,
        "color": 7
      }
    },
    {
      "id": "section-2",
      "name": "Section 2",
      "type": "n8n-nodes-base.stickyNote",
      "typeVersion": 1,
      "position": [
        930,
        60
      ],
      "parameters": {
        "content": "## Validate before spending credits\nKeep source IDs and observed values. Failed requests stay unknown. Review evidence before acting; automatic paid retries are disabled.",
        "width": 940,
        "height": 440,
        "color": 7
      }
    },
    {
      "id": "section-3",
      "name": "Section 3",
      "type": "n8n-nodes-base.stickyNote",
      "typeVersion": 1,
      "position": [
        1890,
        60
      ],
      "parameters": {
        "content": "## Save computed fields only\nCreate Results with headers: `result_key`, `checked_at`, `topic`, `status`, `summary`, `group_keywords`, `returned_results`, `cross_group_matches`, `evidence_json`. Add optional review_status and editor_notes columns; this workflow never writes them.",
        "width": 940,
        "height": 440,
        "color": 7
      }
    },
    {
      "id": "overview",
      "name": "Overview",
      "type": "n8n-nodes-base.stickyNote",
      "typeVersion": 1,
      "position": [
        -650,
        -140
      ],
      "parameters": {
        "content": "# Cluster keywords into page groups using Google SERPs and Sheets\n\n### How it works\nCompare actual ranking URLs across your keyword list and propose groups that could share one page. Every pair in a group must meet your overlap threshold, so a chain of weakly related keywords cannot silently become one large group. The output retains each SERP, pairwise shared URLs, group membership and cross-group matches for review.\n\nThis is a deterministic candidate grouping, not an intent classifier or a unique optimal partition. Sparse or failed searches are marked insufficient evidence. URL overlap does not prove cannibalization. Re-running a smaller input batch can change the grouping.\n\n### Setup\nInstall the verified HasData community node. Connect HasData credentials to every API request and Google Sheets credentials to every Sheets node. Create Keywords and Results tabs with the headers shown on the canvas. Set spreadsheet_id in Settings. Add one query per Keywords row. Use one canonical location and optional country code for the whole batch. Set the minimum usable result count and shared-URL threshold. Start with three related queries and one unrelated query.\n\n### Customization\nOne SERP Light request per keyword, at most thirty. No pagination, model calls or automatic paid retries. Run one execution at a time. Keep result_key and saved evidence unchanged. Results contains the latest review, not a permanent archive. Export the sheet if you need historical versions.",
        "width": 570,
        "height": 1130,
        "color": 1
      }
    }
  ],
  "connections": {
    "Run manually": {
      "main": [
        [
          {
            "node": "Settings",
            "type": "main",
            "index": 0
          }
        ]
      ]
    },
    "Settings": {
      "main": [
        [
          {
            "node": "Validate settings",
            "type": "main",
            "index": 0
          }
        ]
      ]
    },
    "Validate settings": {
      "main": [
        [
          {
            "node": "Read input",
            "type": "main",
            "index": 0
          }
        ]
      ]
    },
    "Read input": {
      "main": [
        [
          {
            "node": "Validate input batch",
            "type": "main",
            "index": 0
          }
        ]
      ]
    },
    "Validate input batch": {
      "main": [
        [
          {
            "node": "Prepare queue read",
            "type": "main",
            "index": 0
          }
        ]
      ]
    },
    "Prepare queue read": {
      "main": [
        [
          {
            "node": "Read Results",
            "type": "main",
            "index": 0
          }
        ]
      ]
    },
    "Read Results": {
      "main": [
        [
          {
            "node": "Check saved keys",
            "type": "main",
            "index": 0
          }
        ]
      ]
    },
    "Check saved keys": {
      "main": [
        [
          {
            "node": "Fetch search evidence",
            "type": "main",
            "index": 0
          }
        ]
      ]
    },
    "Fetch search evidence": {
      "main": [
        [
          {
            "node": "Attach search context",
            "type": "main",
            "index": 0
          }
        ]
      ]
    },
    "Attach search context": {
      "main": [
        [
          {
            "node": "Build review rows",
            "type": "main",
            "index": 0
          }
        ]
      ]
    },
    "Build review rows": {
      "main": [
        [
          {
            "node": "Save review queue",
            "type": "main",
            "index": 0
          }
        ]
      ]
    }
  },
  "settings": {
    "executionOrder": "v1"
  },
  "active": false,
  "pinData": {},
  "tags": []
}
