Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
56 changes: 56 additions & 0 deletions README.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -24,6 +24,7 @@
|---|---|
| `brightdata scrape` | Scrape any URL — bypasses CAPTCHAs, JS rendering, anti-bot protections |
| `brightdata search` | Google / Bing / Yandex search with structured JSON output |
| `brightdata discover` | AI-powered web discovery - find and rank results by intent with optional full-page content |
| `brightdata pipelines` | Extract structured data from 40+ platforms (Amazon, LinkedIn, TikTok…) |
| `brightdata browser` | Control a real browser via Bright Data's Scraping Browser — navigate, snapshot, click, type, and more |
| `brightdata zones` | List and inspect your Bright Data proxy zones |
Expand All@@ -44,6 +45,7 @@
- [init](#init)
- [scrape](#scrape)
- [search](#search)
- [discover](#discover)
- [pipelines](#pipelines)
- [browser](#browser)
- [status](#status)
Expand DownExpand Up@@ -246,6 +248,60 @@ brightdata search "bright data pricing" --engine bing

---

### `discover`

AI-powered web discovery. Submit a query with optional intent, and Bright Data finds, ranks, and optionally extracts full-page content for each result.

```bash
brightdata discover <query> [options]
```

| Flag | Description |
|---|---|
| `--intent <text>` | AI intent to evaluate and rank result relevance |
| `--country <code>` | ISO country code (default: `US`) |
| `--city <name>` | City for localized results (e.g. `"New York"`) |
| `--language <code>` | Language code (default: `en`) |
| `--num-results <n>` | Number of results to return |
| `--filter-keywords <words>` | Comma-separated keywords that must appear in results |
| `--include-content` | Include full page content in each result |
| `--no-remove-duplicates` | Keep duplicate results |
| `--start-date <date>` | Only content updated from date (`YYYY-MM-DD`) |
| `--end-date <date>` | Only content updated until date (`YYYY-MM-DD`) |
| `--timeout <seconds>` | Polling timeout (default: `600`) |
| `-o, --output <path>` | Write output to file |
| `--json` / `--pretty` | JSON output (raw / indented) |
| `-k, --api-key <key>` | Override API key |

**Examples**

```bash
# Basic discovery — table output
brightdata discover "AI trends"

# With AI intent for relevance ranking
brightdata discover "AI trends" \
--intent "Prioritize institutional reports for VC research"

# Include full page content as markdown
brightdata discover "AI trends" --include-content --num-results 5

# Geo-targeted with date range
brightdata discover "best restaurants" --country US --city "New York" \
--start-date 2025-01-01 --end-date 2025-12-31

# Filter results by keywords
brightdata discover "generative AI SaaS" --filter-keywords "revenue,SaaS"

# JSON output to file
brightdata discover "AI trends" --num-results 10 --pretty -o results.json

# Pipe-friendly — redirected stdout outputs JSON automatically
brightdata discover "AI trends" --include-content --num-results 3 > results.json
```

---

### `pipelines`

Extract structured data from 40+ platforms using Bright Data's Web Scraper API. Triggers an async collection job, polls until ready, and returns results.
Expand Down
2 changes: 1 addition & 1 deletion package.json
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,6 @@
{
"name": "@brightdata/cli",
"version": "0.1.6",
"version": "0.1.7",
"description": "Command-line interface for Bright Data. Scrape, search, extract structured data, and automate browsers directly from your terminal.",
"main": "dist/index.js",
"bin": {
Expand Down
260 changes: 260 additions & 0 deletions src/__tests__/commands/discover.test.ts
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,260 @@
import {describe, it, expect, beforeEach, vi} from 'vitest';

const mocks = vi.hoisted(()=>({
post: vi.fn(),
get: vi.fn(),
ensure_authenticated: vi.fn(),
stop: vi.fn(),
start: vi.fn(),
print: vi.fn(),
print_table: vi.fn(),
fail: vi.fn((msg: string)=>{ throw new Error(`fail:${msg}`); }),
dim: vi.fn((msg: string)=>msg),
parse_timeout: vi.fn(),
poll_until: vi.fn(),
}));

vi.mock('../../utils/client', ()=>({
post: mocks.post,
get: mocks.get,
}));

vi.mock('../../utils/auth', ()=>({
ensure_authenticated: mocks.ensure_authenticated,
}));

vi.mock('../../utils/spinner', ()=>({
start: mocks.start,
}));

vi.mock('../../utils/output', ()=>({
print: mocks.print,
print_table: mocks.print_table,
fail: mocks.fail,
dim: mocks.dim,
}));

vi.mock('../../utils/polling', ()=>({
parse_timeout: mocks.parse_timeout,
poll_until: mocks.poll_until,
}));

import {
handle_discover,
build_request,
extract_status,
format_markdown,
print_discover_table,
} from '../../commands/discover';

describe('commands/discover', ()=>{
beforeEach(()=>{
vi.clearAllMocks();
mocks.ensure_authenticated.mockReturnValue('api_key');
mocks.parse_timeout.mockReturnValue(600);
mocks.start.mockReturnValue({stop: mocks.stop});
});

describe('build_request', ()=>{
it('builds minimal request with only query', ()=>{
const req = build_request('AI trends', {});
expect(req).toEqual({query: 'AI trends'});
});

it('includes all optional params', ()=>{
const req = build_request('AI trends', {
intent: 'find research papers',
city: 'New York',
country: 'US',
language: 'en',
numResults: '10',
filterKeywords: 'AI, machine learning',
includeContent: true,
startDate: '2025-01-01',
endDate: '2025-12-31',
});
expect(req).toEqual({
query: 'AI trends',
intent: 'find research papers',
city: 'New York',
country: 'US',
language: 'en',
num_results: 10,
filter_keywords: ['AI', 'machine learning'],
include_content: true,
start_date: '2025-01-01',
end_date: '2025-12-31',
});
});

it('parses comma-separated filter keywords with whitespace', ()=>{
const req = build_request('q', {filterKeywords: ' a , b , c '});
expect(req.filter_keywords).toEqual(['a', 'b', 'c']);
});

it('does not set format by default (API returns JSON)', ()=>{
const req = build_request('test', {});
expect(req.format).toBeUndefined();
});

it('does not set format when include-content is used', ()=>{
const req = build_request('test', {includeContent: true});
expect(req.format).toBeUndefined();
expect(req.include_content).toBe(true);
});
});

describe('extract_status', ()=>{
it('returns status from valid response', ()=>{
expect(extract_status({status: 'processing'})).toBe('processing');
expect(extract_status({status: 'done'})).toBe('done');
});

it('returns undefined for invalid input', ()=>{
expect(extract_status(null as never)).toBeUndefined();
expect(extract_status(undefined as never)).toBeUndefined();
});
});

describe('format_markdown', ()=>{
it('formats results as markdown', ()=>{
const md = format_markdown([
{
link: 'https://example.com',
title: 'Example',
description: 'A description',
relevance_score: 0.95,
},
], 'test query');
expect(md).toContain('# Discover results for "test query"');
expect(md).toContain('**1. [Example](https://example.com)** (95.0%)');
expect(md).toContain('A description');
});

it('includes content when present', ()=>{
const md = format_markdown([
{
link: 'https://example.com',
title: 'Example',
description: 'Desc',
relevance_score: 0.5,
content: '# Page content here',
},
], 'q');
expect(md).toContain('# Page content here');
});
});

describe('print_discover_table', ()=>{
it('calls print_table with formatted rows', ()=>{
const results = [
{
link: 'https://example.com',
title: 'Example Title',
description: 'Desc',
relevance_score: 0.98184747,
},
];
print_discover_table(results);
expect(mocks.print_table).toHaveBeenCalledWith(
[{
'#': '1',
title: 'Example Title',
score: '98.2%',
url: 'https://example.com',
}],
['#', 'title', 'score', 'url']
);
});

it('prints dim message when no results', ()=>{
const log = vi.spyOn(console, 'log').mockImplementation(()=>{});
print_discover_table([]);
expect(log).toHaveBeenCalled();
expect(mocks.print_table).not.toHaveBeenCalled();
log.mockRestore();
});
});

describe('handle_discover', ()=>{
it('triggers and polls then prints table', async()=>{
mocks.post.mockResolvedValue({status: 'ok', task_id: 'abc123'});
mocks.poll_until.mockResolvedValue({
result: {
status: 'done',
duration_seconds: 5,
results: [
{
link: 'https://example.com',
title: 'Result',
description: 'Desc',
relevance_score: 0.9,
},
],
},
attempts: 3,
});
await handle_discover('AI trends', {});
expect(mocks.post).toHaveBeenCalledWith(
'api_key',
'/discover',
{query: 'AI trends'},
{timing: undefined}
);
expect(mocks.poll_until).toHaveBeenCalledTimes(1);
expect(mocks.print_table).toHaveBeenCalledTimes(1);
});

it('prints json when --json is set', async()=>{
const response = {
status: 'done',
duration_seconds: 2,
results: [{
link: 'https://example.com',
title: 'R',
description: 'D',
relevance_score: 0.8,
}],
};
mocks.post.mockResolvedValue({status: 'ok', task_id: 't1'});
mocks.poll_until.mockResolvedValue({result: response, attempts: 1});
await handle_discover('q', {json: true});
expect(mocks.print).toHaveBeenCalledWith(
response,
{json: true, pretty: undefined, output: undefined}
);
expect(mocks.print_table).not.toHaveBeenCalled();
});

it('prints raw JSON when --output is set', async()=>{
const response = {
status: 'done',
results: [{
link: 'https://example.com',
title: 'R',
description: 'D',
relevance_score: 0.7,
}],
};
mocks.post.mockResolvedValue({status: 'ok', task_id: 't2'});
mocks.poll_until.mockResolvedValue({result: response, attempts: 1});
await handle_discover('q', {output: 'out.json'});
expect(mocks.print).toHaveBeenCalledWith(
response,
{json: undefined, pretty: undefined, output: 'out.json'}
);
});

it('fails when trigger returns no task_id', async()=>{
mocks.post.mockResolvedValue({status: 'ok'});
const exit = vi.spyOn(process, 'exit')
.mockImplementation(()=>undefined as never);
const error = vi.spyOn(console, 'error')
.mockImplementation(()=>{});
await handle_discover('q', {});
expect(mocks.fail).toHaveBeenCalled();
exit.mockRestore();
error.mockRestore();
});
});
});
Loading
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Add copy buttons to all
 blocks\n(function() {\n function addCopyButtons() {\n document.querySelectorAll('pre code').forEach(function(codeBlock) {\n if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;\n codeBlock.parentElement.setAttribute('data-copy-added', 'true');\n \n var btn = document.createElement('button');\n btn.textContent = 'Copy';\n btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';\n btn.onmouseover = function() { this.style.opacity = '1'; };\n btn.onmouseout = function() { this.style.opacity = '0.7'; };\n btn.onclick = function() {\n navigator.clipboard.writeText(codeBlock.textContent).then(function() {\n btn.textContent = 'Copied!';\n setTimeout(function() { btn.textContent = 'Copy'; }, 1500);\n });\n };\n codeBlock.parentElement.style.position = 'relative';\n codeBlock.parentElement.appendChild(btn);\n });\n }\n \n addCopyButtons();\n \n // Re-run on dynamic content\n var observer = new MutationObserver(addCopyButtons);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Add Copy Buttons to Code Blocks");
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
56 changes: 56 additions & 0 deletions README.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -24,6 +24,7 @@
|---|---|
| `brightdata scrape` | Scrape any URL — bypasses CAPTCHAs, JS rendering, anti-bot protections |
| `brightdata search` | Google / Bing / Yandex search with structured JSON output |
| `brightdata discover` | AI-powered web discovery - find and rank results by intent with optional full-page content |
| `brightdata pipelines` | Extract structured data from 40+ platforms (Amazon, LinkedIn, TikTok…) |
| `brightdata browser` | Control a real browser via Bright Data's Scraping Browser — navigate, snapshot, click, type, and more |
| `brightdata zones` | List and inspect your Bright Data proxy zones |
Expand All@@ -44,6 +45,7 @@
- [init](#init)
- [scrape](#scrape)
- [search](#search)
- [discover](#discover)
- [pipelines](#pipelines)
- [browser](#browser)
- [status](#status)
Expand DownExpand Up@@ -246,6 +248,60 @@ brightdata search "bright data pricing" --engine bing

---

### `discover`

AI-powered web discovery. Submit a query with optional intent, and Bright Data finds, ranks, and optionally extracts full-page content for each result.

```bash
brightdata discover <query> [options]
```

| Flag | Description |
|---|---|
| `--intent <text>` | AI intent to evaluate and rank result relevance |
| `--country <code>` | ISO country code (default: `US`) |
| `--city <name>` | City for localized results (e.g. `"New York"`) |
| `--language <code>` | Language code (default: `en`) |
| `--num-results <n>` | Number of results to return |
| `--filter-keywords <words>` | Comma-separated keywords that must appear in results |
| `--include-content` | Include full page content in each result |
| `--no-remove-duplicates` | Keep duplicate results |
| `--start-date <date>` | Only content updated from date (`YYYY-MM-DD`) |
| `--end-date <date>` | Only content updated until date (`YYYY-MM-DD`) |
| `--timeout <seconds>` | Polling timeout (default: `600`) |
| `-o, --output <path>` | Write output to file |
| `--json` / `--pretty` | JSON output (raw / indented) |
| `-k, --api-key <key>` | Override API key |

**Examples**

```bash
# Basic discovery — table output
brightdata discover "AI trends"

# With AI intent for relevance ranking
brightdata discover "AI trends" \
--intent "Prioritize institutional reports for VC research"

# Include full page content as markdown
brightdata discover "AI trends" --include-content --num-results 5

# Geo-targeted with date range
brightdata discover "best restaurants" --country US --city "New York" \
--start-date 2025-01-01 --end-date 2025-12-31

# Filter results by keywords
brightdata discover "generative AI SaaS" --filter-keywords "revenue,SaaS"

# JSON output to file
brightdata discover "AI trends" --num-results 10 --pretty -o results.json

# Pipe-friendly — redirected stdout outputs JSON automatically
brightdata discover "AI trends" --include-content --num-results 3 > results.json
```

---

### `pipelines`

Extract structured data from 40+ platforms using Bright Data's Web Scraper API. Triggers an async collection job, polls until ready, and returns results.
Expand Down
2 changes: 1 addition & 1 deletion package.json
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,6 @@
{
"name": "@brightdata/cli",
"version": "0.1.6",
"version": "0.1.7",
"description": "Command-line interface for Bright Data. Scrape, search, extract structured data, and automate browsers directly from your terminal.",
"main": "dist/index.js",
"bin": {
Expand Down
260 changes: 260 additions & 0 deletions src/__tests__/commands/discover.test.ts
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,260 @@
import {describe, it, expect, beforeEach, vi} from 'vitest';

const mocks = vi.hoisted(()=>({
post: vi.fn(),
get: vi.fn(),
ensure_authenticated: vi.fn(),
stop: vi.fn(),
start: vi.fn(),
print: vi.fn(),
print_table: vi.fn(),
fail: vi.fn((msg: string)=>{ throw new Error(`fail:${msg}`); }),
dim: vi.fn((msg: string)=>msg),
parse_timeout: vi.fn(),
poll_until: vi.fn(),
}));

vi.mock('../../utils/client', ()=>({
post: mocks.post,
get: mocks.get,
}));

vi.mock('../../utils/auth', ()=>({
ensure_authenticated: mocks.ensure_authenticated,
}));

vi.mock('../../utils/spinner', ()=>({
start: mocks.start,
}));

vi.mock('../../utils/output', ()=>({
print: mocks.print,
print_table: mocks.print_table,
fail: mocks.fail,
dim: mocks.dim,
}));

vi.mock('../../utils/polling', ()=>({
parse_timeout: mocks.parse_timeout,
poll_until: mocks.poll_until,
}));

import {
handle_discover,
build_request,
extract_status,
format_markdown,
print_discover_table,
} from '../../commands/discover';

describe('commands/discover', ()=>{
beforeEach(()=>{
vi.clearAllMocks();
mocks.ensure_authenticated.mockReturnValue('api_key');
mocks.parse_timeout.mockReturnValue(600);
mocks.start.mockReturnValue({stop: mocks.stop});
});

describe('build_request', ()=>{
it('builds minimal request with only query', ()=>{
const req = build_request('AI trends', {});
expect(req).toEqual({query: 'AI trends'});
});

it('includes all optional params', ()=>{
const req = build_request('AI trends', {
intent: 'find research papers',
city: 'New York',
country: 'US',
language: 'en',
numResults: '10',
filterKeywords: 'AI, machine learning',
includeContent: true,
startDate: '2025-01-01',
endDate: '2025-12-31',
});
expect(req).toEqual({
query: 'AI trends',
intent: 'find research papers',
city: 'New York',
country: 'US',
language: 'en',
num_results: 10,
filter_keywords: ['AI', 'machine learning'],
include_content: true,
start_date: '2025-01-01',
end_date: '2025-12-31',
});
});

it('parses comma-separated filter keywords with whitespace', ()=>{
const req = build_request('q', {filterKeywords: ' a , b , c '});
expect(req.filter_keywords).toEqual(['a', 'b', 'c']);
});

it('does not set format by default (API returns JSON)', ()=>{
const req = build_request('test', {});
expect(req.format).toBeUndefined();
});

it('does not set format when include-content is used', ()=>{
const req = build_request('test', {includeContent: true});
expect(req.format).toBeUndefined();
expect(req.include_content).toBe(true);
});
});

describe('extract_status', ()=>{
it('returns status from valid response', ()=>{
expect(extract_status({status: 'processing'})).toBe('processing');
expect(extract_status({status: 'done'})).toBe('done');
});

it('returns undefined for invalid input', ()=>{
expect(extract_status(null as never)).toBeUndefined();
expect(extract_status(undefined as never)).toBeUndefined();
});
});

describe('format_markdown', ()=>{
it('formats results as markdown', ()=>{
const md = format_markdown([
{
link: 'https://example.com',
title: 'Example',
description: 'A description',
relevance_score: 0.95,
},
], 'test query');
expect(md).toContain('# Discover results for "test query"');
expect(md).toContain('**1. [Example](https://example.com)** (95.0%)');
expect(md).toContain('A description');
});

it('includes content when present', ()=>{
const md = format_markdown([
{
link: 'https://example.com',
title: 'Example',
description: 'Desc',
relevance_score: 0.5,
content: '# Page content here',
},
], 'q');
expect(md).toContain('# Page content here');
});
});

describe('print_discover_table', ()=>{
it('calls print_table with formatted rows', ()=>{
const results = [
{
link: 'https://example.com',
title: 'Example Title',
description: 'Desc',
relevance_score: 0.98184747,
},
];
print_discover_table(results);
expect(mocks.print_table).toHaveBeenCalledWith(
[{
'#': '1',
title: 'Example Title',
score: '98.2%',
url: 'https://example.com',
}],
['#', 'title', 'score', 'url']
);
});

it('prints dim message when no results', ()=>{
const log = vi.spyOn(console, 'log').mockImplementation(()=>{});
print_discover_table([]);
expect(log).toHaveBeenCalled();
expect(mocks.print_table).not.toHaveBeenCalled();
log.mockRestore();
});
});

describe('handle_discover', ()=>{
it('triggers and polls then prints table', async()=>{
mocks.post.mockResolvedValue({status: 'ok', task_id: 'abc123'});
mocks.poll_until.mockResolvedValue({
result: {
status: 'done',
duration_seconds: 5,
results: [
{
link: 'https://example.com',
title: 'Result',
description: 'Desc',
relevance_score: 0.9,
},
],
},
attempts: 3,
});
await handle_discover('AI trends', {});
expect(mocks.post).toHaveBeenCalledWith(
'api_key',
'/discover',
{query: 'AI trends'},
{timing: undefined}
);
expect(mocks.poll_until).toHaveBeenCalledTimes(1);
expect(mocks.print_table).toHaveBeenCalledTimes(1);
});

it('prints json when --json is set', async()=>{
const response = {
status: 'done',
duration_seconds: 2,
results: [{
link: 'https://example.com',
title: 'R',
description: 'D',
relevance_score: 0.8,
}],
};
mocks.post.mockResolvedValue({status: 'ok', task_id: 't1'});
mocks.poll_until.mockResolvedValue({result: response, attempts: 1});
await handle_discover('q', {json: true});
expect(mocks.print).toHaveBeenCalledWith(
response,
{json: true, pretty: undefined, output: undefined}
);
expect(mocks.print_table).not.toHaveBeenCalled();
});

it('prints raw JSON when --output is set', async()=>{
const response = {
status: 'done',
results: [{
link: 'https://example.com',
title: 'R',
description: 'D',
relevance_score: 0.7,
}],
};
mocks.post.mockResolvedValue({status: 'ok', task_id: 't2'});
mocks.poll_until.mockResolvedValue({result: response, attempts: 1});
await handle_discover('q', {output: 'out.json'});
expect(mocks.print).toHaveBeenCalledWith(
response,
{json: undefined, pretty: undefined, output: 'out.json'}
);
});

it('fails when trigger returns no task_id', async()=>{
mocks.post.mockResolvedValue({status: 'ok'});
const exit = vi.spyOn(process, 'exit')
.mockImplementation(()=>undefined as never);
const error = vi.spyOn(console, 'error')
.mockImplementation(()=>{});
await handle_discover('q', {});
expect(mocks.fail).toHaveBeenCalled();
exit.mockRestore();
error.mockRestore();
});
});
});
Loading
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Force GitHub README to respect dark mode\n(function() {\n var style = document.createElement('style');\n style.textContent = '\n .markdown-body {\n color-scheme: dark light;\n }\n .markdown-body pre { background: #161b22 !important; }\n .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; }\n .markdown-body table th, .markdown-body table td { border-color: #30363d !important; }\n .markdown-body img { background: #0d1117; }\n .markdown-body blockquote { border-left-color: #8b949e; }\n .markdown-body hr { border-color: #30363d; }\n ';\n document.head.appendChild(style);\n})();", "GitHub Dark Mode README Fix"); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
56 changes: 56 additions & 0 deletions README.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -24,6 +24,7 @@
|---|---|
| `brightdata scrape` | Scrape any URL — bypasses CAPTCHAs, JS rendering, anti-bot protections |
| `brightdata search` | Google / Bing / Yandex search with structured JSON output |
| `brightdata discover` | AI-powered web discovery - find and rank results by intent with optional full-page content |
| `brightdata pipelines` | Extract structured data from 40+ platforms (Amazon, LinkedIn, TikTok…) |
| `brightdata browser` | Control a real browser via Bright Data's Scraping Browser — navigate, snapshot, click, type, and more |
| `brightdata zones` | List and inspect your Bright Data proxy zones |
Expand All@@ -44,6 +45,7 @@
- [init](#init)
- [scrape](#scrape)
- [search](#search)
- [discover](#discover)
- [pipelines](#pipelines)
- [browser](#browser)
- [status](#status)
Expand DownExpand Up@@ -246,6 +248,60 @@ brightdata search "bright data pricing" --engine bing

---

### `discover`

AI-powered web discovery. Submit a query with optional intent, and Bright Data finds, ranks, and optionally extracts full-page content for each result.

```bash
brightdata discover <query> [options]
```

| Flag | Description |
|---|---|
| `--intent <text>` | AI intent to evaluate and rank result relevance |
| `--country <code>` | ISO country code (default: `US`) |
| `--city <name>` | City for localized results (e.g. `"New York"`) |
| `--language <code>` | Language code (default: `en`) |
| `--num-results <n>` | Number of results to return |
| `--filter-keywords <words>` | Comma-separated keywords that must appear in results |
| `--include-content` | Include full page content in each result |
| `--no-remove-duplicates` | Keep duplicate results |
| `--start-date <date>` | Only content updated from date (`YYYY-MM-DD`) |
| `--end-date <date>` | Only content updated until date (`YYYY-MM-DD`) |
| `--timeout <seconds>` | Polling timeout (default: `600`) |
| `-o, --output <path>` | Write output to file |
| `--json` / `--pretty` | JSON output (raw / indented) |
| `-k, --api-key <key>` | Override API key |

**Examples**

```bash
# Basic discovery — table output
brightdata discover "AI trends"

# With AI intent for relevance ranking
brightdata discover "AI trends" \
--intent "Prioritize institutional reports for VC research"

# Include full page content as markdown
brightdata discover "AI trends" --include-content --num-results 5

# Geo-targeted with date range
brightdata discover "best restaurants" --country US --city "New York" \
--start-date 2025-01-01 --end-date 2025-12-31

# Filter results by keywords
brightdata discover "generative AI SaaS" --filter-keywords "revenue,SaaS"

# JSON output to file
brightdata discover "AI trends" --num-results 10 --pretty -o results.json

# Pipe-friendly — redirected stdout outputs JSON automatically
brightdata discover "AI trends" --include-content --num-results 3 > results.json
```

---

### `pipelines`

Extract structured data from 40+ platforms using Bright Data's Web Scraper API. Triggers an async collection job, polls until ready, and returns results.
Expand Down
2 changes: 1 addition & 1 deletion package.json
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,6 @@
{
"name": "@brightdata/cli",
"version": "0.1.6",
"version": "0.1.7",
"description": "Command-line interface for Bright Data. Scrape, search, extract structured data, and automate browsers directly from your terminal.",
"main": "dist/index.js",
"bin": {
Expand Down
260 changes: 260 additions & 0 deletions src/__tests__/commands/discover.test.ts
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,260 @@
import {describe, it, expect, beforeEach, vi} from 'vitest';

const mocks = vi.hoisted(()=>({
post: vi.fn(),
get: vi.fn(),
ensure_authenticated: vi.fn(),
stop: vi.fn(),
start: vi.fn(),
print: vi.fn(),
print_table: vi.fn(),
fail: vi.fn((msg: string)=>{ throw new Error(`fail:${msg}`); }),
dim: vi.fn((msg: string)=>msg),
parse_timeout: vi.fn(),
poll_until: vi.fn(),
}));

vi.mock('../../utils/client', ()=>({
post: mocks.post,
get: mocks.get,
}));

vi.mock('../../utils/auth', ()=>({
ensure_authenticated: mocks.ensure_authenticated,
}));

vi.mock('../../utils/spinner', ()=>({
start: mocks.start,
}));

vi.mock('../../utils/output', ()=>({
print: mocks.print,
print_table: mocks.print_table,
fail: mocks.fail,
dim: mocks.dim,
}));

vi.mock('../../utils/polling', ()=>({
parse_timeout: mocks.parse_timeout,
poll_until: mocks.poll_until,
}));

import {
handle_discover,
build_request,
extract_status,
format_markdown,
print_discover_table,
} from '../../commands/discover';

describe('commands/discover', ()=>{
beforeEach(()=>{
vi.clearAllMocks();
mocks.ensure_authenticated.mockReturnValue('api_key');
mocks.parse_timeout.mockReturnValue(600);
mocks.start.mockReturnValue({stop: mocks.stop});
});

describe('build_request', ()=>{
it('builds minimal request with only query', ()=>{
const req = build_request('AI trends', {});
expect(req).toEqual({query: 'AI trends'});
});

it('includes all optional params', ()=>{
const req = build_request('AI trends', {
intent: 'find research papers',
city: 'New York',
country: 'US',
language: 'en',
numResults: '10',
filterKeywords: 'AI, machine learning',
includeContent: true,
startDate: '2025-01-01',
endDate: '2025-12-31',
});
expect(req).toEqual({
query: 'AI trends',
intent: 'find research papers',
city: 'New York',
country: 'US',
language: 'en',
num_results: 10,
filter_keywords: ['AI', 'machine learning'],
include_content: true,
start_date: '2025-01-01',
end_date: '2025-12-31',
});
});

it('parses comma-separated filter keywords with whitespace', ()=>{
const req = build_request('q', {filterKeywords: ' a , b , c '});
expect(req.filter_keywords).toEqual(['a', 'b', 'c']);
});

it('does not set format by default (API returns JSON)', ()=>{
const req = build_request('test', {});
expect(req.format).toBeUndefined();
});

it('does not set format when include-content is used', ()=>{
const req = build_request('test', {includeContent: true});
expect(req.format).toBeUndefined();
expect(req.include_content).toBe(true);
});
});

describe('extract_status', ()=>{
it('returns status from valid response', ()=>{
expect(extract_status({status: 'processing'})).toBe('processing');
expect(extract_status({status: 'done'})).toBe('done');
});

it('returns undefined for invalid input', ()=>{
expect(extract_status(null as never)).toBeUndefined();
expect(extract_status(undefined as never)).toBeUndefined();
});
});

describe('format_markdown', ()=>{
it('formats results as markdown', ()=>{
const md = format_markdown([
{
link: 'https://example.com',
title: 'Example',
description: 'A description',
relevance_score: 0.95,
},
], 'test query');
expect(md).toContain('# Discover results for "test query"');
expect(md).toContain('**1. [Example](https://example.com)** (95.0%)');
expect(md).toContain('A description');
});

it('includes content when present', ()=>{
const md = format_markdown([
{
link: 'https://example.com',
title: 'Example',
description: 'Desc',
relevance_score: 0.5,
content: '# Page content here',
},
], 'q');
expect(md).toContain('# Page content here');
});
});

describe('print_discover_table', ()=>{
it('calls print_table with formatted rows', ()=>{
const results = [
{
link: 'https://example.com',
title: 'Example Title',
description: 'Desc',
relevance_score: 0.98184747,
},
];
print_discover_table(results);
expect(mocks.print_table).toHaveBeenCalledWith(
[{
'#': '1',
title: 'Example Title',
score: '98.2%',
url: 'https://example.com',
}],
['#', 'title', 'score', 'url']
);
});

it('prints dim message when no results', ()=>{
const log = vi.spyOn(console, 'log').mockImplementation(()=>{});
print_discover_table([]);
expect(log).toHaveBeenCalled();
expect(mocks.print_table).not.toHaveBeenCalled();
log.mockRestore();
});
});

describe('handle_discover', ()=>{
it('triggers and polls then prints table', async()=>{
mocks.post.mockResolvedValue({status: 'ok', task_id: 'abc123'});
mocks.poll_until.mockResolvedValue({
result: {
status: 'done',
duration_seconds: 5,
results: [
{
link: 'https://example.com',
title: 'Result',
description: 'Desc',
relevance_score: 0.9,
},
],
},
attempts: 3,
});
await handle_discover('AI trends', {});
expect(mocks.post).toHaveBeenCalledWith(
'api_key',
'/discover',
{query: 'AI trends'},
{timing: undefined}
);
expect(mocks.poll_until).toHaveBeenCalledTimes(1);
expect(mocks.print_table).toHaveBeenCalledTimes(1);
});

it('prints json when --json is set', async()=>{
const response = {
status: 'done',
duration_seconds: 2,
results: [{
link: 'https://example.com',
title: 'R',
description: 'D',
relevance_score: 0.8,
}],
};
mocks.post.mockResolvedValue({status: 'ok', task_id: 't1'});
mocks.poll_until.mockResolvedValue({result: response, attempts: 1});
await handle_discover('q', {json: true});
expect(mocks.print).toHaveBeenCalledWith(
response,
{json: true, pretty: undefined, output: undefined}
);
expect(mocks.print_table).not.toHaveBeenCalled();
});

it('prints raw JSON when --output is set', async()=>{
const response = {
status: 'done',
results: [{
link: 'https://example.com',
title: 'R',
description: 'D',
relevance_score: 0.7,
}],
};
mocks.post.mockResolvedValue({status: 'ok', task_id: 't2'});
mocks.poll_until.mockResolvedValue({result: response, attempts: 1});
await handle_discover('q', {output: 'out.json'});
expect(mocks.print).toHaveBeenCalledWith(
response,
{json: undefined, pretty: undefined, output: 'out.json'}
);
});

it('fails when trigger returns no task_id', async()=>{
mocks.post.mockResolvedValue({status: 'ok'});
const exit = vi.spyOn(process, 'exit')
.mockImplementation(()=>undefined as never);
const error = vi.spyOn(console, 'error')
.mockImplementation(()=>{});
await handle_discover('q', {});
expect(mocks.fail).toHaveBeenCalled();
exit.mockRestore();
error.mockRestore();
});
});
});
Loading
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Highlight search terms from Google/DuckDuckGo/Bing referrer\n(function() {\n var ref = document.referrer;\n var terms = [];\n \n if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) {\n var url = new URL(ref);\n var q = url.searchParams.get('q') || url.searchParams.get('p');\n if (q) {\n terms = q.split(/\\s+/).filter(function(t) { return t.length > 2; });\n }\n }\n \n if (terms.length === 0) return;\n \n var style = document.createElement('style');\n style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }';\n document.head.appendChild(style);\n \n function highlight(node) {\n if (node.nodeType === 3) { // text node\n var text = node.textContent;\n var found = false;\n terms.forEach(function(term) {\n var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\') + ')', 'gi');\n if (regex.test(text)) {\n found = true;\n var frag = document.createDocumentFragment();\n var parts = text.split(regex);\n parts.forEach(function(part, i) {\n if (i % 2 === 0) {\n frag.appendChild(document.createTextNode(part));\n } else {\n var span = document.createElement('span');\n span.className = 'userscript-highlight';\n span.textContent = part;\n frag.appendChild(span);\n }\n });\n node.parentNode.replaceChild(frag, node);\n }\n });\n } else if (node.nodeType === 1 && node.childNodes) { // element\n var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT'];\n if (!skipTags.includes(node.tagName)) {\n Array.from(node.childNodes).forEach(highlight);\n }\n }\n }\n \n highlight(document.body);\n \n // Re-highlight on dynamic content\n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1 || node.nodeType === 3) highlight(node);\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Highlight Search Terms"); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
56 changes: 56 additions & 0 deletions README.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -24,6 +24,7 @@
|---|---|
| `brightdata scrape` | Scrape any URL — bypasses CAPTCHAs, JS rendering, anti-bot protections |
| `brightdata search` | Google / Bing / Yandex search with structured JSON output |
| `brightdata discover` | AI-powered web discovery - find and rank results by intent with optional full-page content |
| `brightdata pipelines` | Extract structured data from 40+ platforms (Amazon, LinkedIn, TikTok…) |
| `brightdata browser` | Control a real browser via Bright Data's Scraping Browser — navigate, snapshot, click, type, and more |
| `brightdata zones` | List and inspect your Bright Data proxy zones |
Expand All@@ -44,6 +45,7 @@
- [init](#init)
- [scrape](#scrape)
- [search](#search)
- [discover](#discover)
- [pipelines](#pipelines)
- [browser](#browser)
- [status](#status)
Expand DownExpand Up@@ -246,6 +248,60 @@ brightdata search "bright data pricing" --engine bing

---

### `discover`

AI-powered web discovery. Submit a query with optional intent, and Bright Data finds, ranks, and optionally extracts full-page content for each result.

```bash
brightdata discover <query> [options]
```

| Flag | Description |
|---|---|
| `--intent <text>` | AI intent to evaluate and rank result relevance |
| `--country <code>` | ISO country code (default: `US`) |
| `--city <name>` | City for localized results (e.g. `"New York"`) |
| `--language <code>` | Language code (default: `en`) |
| `--num-results <n>` | Number of results to return |
| `--filter-keywords <words>` | Comma-separated keywords that must appear in results |
| `--include-content` | Include full page content in each result |
| `--no-remove-duplicates` | Keep duplicate results |
| `--start-date <date>` | Only content updated from date (`YYYY-MM-DD`) |
| `--end-date <date>` | Only content updated until date (`YYYY-MM-DD`) |
| `--timeout <seconds>` | Polling timeout (default: `600`) |
| `-o, --output <path>` | Write output to file |
| `--json` / `--pretty` | JSON output (raw / indented) |
| `-k, --api-key <key>` | Override API key |

**Examples**

```bash
# Basic discovery — table output
brightdata discover "AI trends"

# With AI intent for relevance ranking
brightdata discover "AI trends" \
--intent "Prioritize institutional reports for VC research"

# Include full page content as markdown
brightdata discover "AI trends" --include-content --num-results 5

# Geo-targeted with date range
brightdata discover "best restaurants" --country US --city "New York" \
--start-date 2025-01-01 --end-date 2025-12-31

# Filter results by keywords
brightdata discover "generative AI SaaS" --filter-keywords "revenue,SaaS"

# JSON output to file
brightdata discover "AI trends" --num-results 10 --pretty -o results.json

# Pipe-friendly — redirected stdout outputs JSON automatically
brightdata discover "AI trends" --include-content --num-results 3 > results.json
```

---

### `pipelines`

Extract structured data from 40+ platforms using Bright Data's Web Scraper API. Triggers an async collection job, polls until ready, and returns results.
Expand Down
2 changes: 1 addition & 1 deletion package.json
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,6 @@
{
"name": "@brightdata/cli",
"version": "0.1.6",
"version": "0.1.7",
"description": "Command-line interface for Bright Data. Scrape, search, extract structured data, and automate browsers directly from your terminal.",
"main": "dist/index.js",
"bin": {
Expand Down
260 changes: 260 additions & 0 deletions src/__tests__/commands/discover.test.ts
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,260 @@
import {describe, it, expect, beforeEach, vi} from 'vitest';

const mocks = vi.hoisted(()=>({
post: vi.fn(),
get: vi.fn(),
ensure_authenticated: vi.fn(),
stop: vi.fn(),
start: vi.fn(),
print: vi.fn(),
print_table: vi.fn(),
fail: vi.fn((msg: string)=>{ throw new Error(`fail:${msg}`); }),
dim: vi.fn((msg: string)=>msg),
parse_timeout: vi.fn(),
poll_until: vi.fn(),
}));

vi.mock('../../utils/client', ()=>({
post: mocks.post,
get: mocks.get,
}));

vi.mock('../../utils/auth', ()=>({
ensure_authenticated: mocks.ensure_authenticated,
}));

vi.mock('../../utils/spinner', ()=>({
start: mocks.start,
}));

vi.mock('../../utils/output', ()=>({
print: mocks.print,
print_table: mocks.print_table,
fail: mocks.fail,
dim: mocks.dim,
}));

vi.mock('../../utils/polling', ()=>({
parse_timeout: mocks.parse_timeout,
poll_until: mocks.poll_until,
}));

import {
handle_discover,
build_request,
extract_status,
format_markdown,
print_discover_table,
} from '../../commands/discover';

describe('commands/discover', ()=>{
beforeEach(()=>{
vi.clearAllMocks();
mocks.ensure_authenticated.mockReturnValue('api_key');
mocks.parse_timeout.mockReturnValue(600);
mocks.start.mockReturnValue({stop: mocks.stop});
});

describe('build_request', ()=>{
it('builds minimal request with only query', ()=>{
const req = build_request('AI trends', {});
expect(req).toEqual({query: 'AI trends'});
});

it('includes all optional params', ()=>{
const req = build_request('AI trends', {
intent: 'find research papers',
city: 'New York',
country: 'US',
language: 'en',
numResults: '10',
filterKeywords: 'AI, machine learning',
includeContent: true,
startDate: '2025-01-01',
endDate: '2025-12-31',
});
expect(req).toEqual({
query: 'AI trends',
intent: 'find research papers',
city: 'New York',
country: 'US',
language: 'en',
num_results: 10,
filter_keywords: ['AI', 'machine learning'],
include_content: true,
start_date: '2025-01-01',
end_date: '2025-12-31',
});
});

it('parses comma-separated filter keywords with whitespace', ()=>{
const req = build_request('q', {filterKeywords: ' a , b , c '});
expect(req.filter_keywords).toEqual(['a', 'b', 'c']);
});

it('does not set format by default (API returns JSON)', ()=>{
const req = build_request('test', {});
expect(req.format).toBeUndefined();
});

it('does not set format when include-content is used', ()=>{
const req = build_request('test', {includeContent: true});
expect(req.format).toBeUndefined();
expect(req.include_content).toBe(true);
});
});

describe('extract_status', ()=>{
it('returns status from valid response', ()=>{
expect(extract_status({status: 'processing'})).toBe('processing');
expect(extract_status({status: 'done'})).toBe('done');
});

it('returns undefined for invalid input', ()=>{
expect(extract_status(null as never)).toBeUndefined();
expect(extract_status(undefined as never)).toBeUndefined();
});
});

describe('format_markdown', ()=>{
it('formats results as markdown', ()=>{
const md = format_markdown([
{
link: 'https://example.com',
title: 'Example',
description: 'A description',
relevance_score: 0.95,
},
], 'test query');
expect(md).toContain('# Discover results for "test query"');
expect(md).toContain('**1. [Example](https://example.com)** (95.0%)');
expect(md).toContain('A description');
});

it('includes content when present', ()=>{
const md = format_markdown([
{
link: 'https://example.com',
title: 'Example',
description: 'Desc',
relevance_score: 0.5,
content: '# Page content here',
},
], 'q');
expect(md).toContain('# Page content here');
});
});

describe('print_discover_table', ()=>{
it('calls print_table with formatted rows', ()=>{
const results = [
{
link: 'https://example.com',
title: 'Example Title',
description: 'Desc',
relevance_score: 0.98184747,
},
];
print_discover_table(results);
expect(mocks.print_table).toHaveBeenCalledWith(
[{
'#': '1',
title: 'Example Title',
score: '98.2%',
url: 'https://example.com',
}],
['#', 'title', 'score', 'url']
);
});

it('prints dim message when no results', ()=>{
const log = vi.spyOn(console, 'log').mockImplementation(()=>{});
print_discover_table([]);
expect(log).toHaveBeenCalled();
expect(mocks.print_table).not.toHaveBeenCalled();
log.mockRestore();
});
});

describe('handle_discover', ()=>{
it('triggers and polls then prints table', async()=>{
mocks.post.mockResolvedValue({status: 'ok', task_id: 'abc123'});
mocks.poll_until.mockResolvedValue({
result: {
status: 'done',
duration_seconds: 5,
results: [
{
link: 'https://example.com',
title: 'Result',
description: 'Desc',
relevance_score: 0.9,
},
],
},
attempts: 3,
});
await handle_discover('AI trends', {});
expect(mocks.post).toHaveBeenCalledWith(
'api_key',
'/discover',
{query: 'AI trends'},
{timing: undefined}
);
expect(mocks.poll_until).toHaveBeenCalledTimes(1);
expect(mocks.print_table).toHaveBeenCalledTimes(1);
});

it('prints json when --json is set', async()=>{
const response = {
status: 'done',
duration_seconds: 2,
results: [{
link: 'https://example.com',
title: 'R',
description: 'D',
relevance_score: 0.8,
}],
};
mocks.post.mockResolvedValue({status: 'ok', task_id: 't1'});
mocks.poll_until.mockResolvedValue({result: response, attempts: 1});
await handle_discover('q', {json: true});
expect(mocks.print).toHaveBeenCalledWith(
response,
{json: true, pretty: undefined, output: undefined}
);
expect(mocks.print_table).not.toHaveBeenCalled();
});

it('prints raw JSON when --output is set', async()=>{
const response = {
status: 'done',
results: [{
link: 'https://example.com',
title: 'R',
description: 'D',
relevance_score: 0.7,
}],
};
mocks.post.mockResolvedValue({status: 'ok', task_id: 't2'});
mocks.poll_until.mockResolvedValue({result: response, attempts: 1});
await handle_discover('q', {output: 'out.json'});
expect(mocks.print).toHaveBeenCalledWith(
response,
{json: undefined, pretty: undefined, output: 'out.json'}
);
});

it('fails when trigger returns no task_id', async()=>{
mocks.post.mockResolvedValue({status: 'ok'});
const exit = vi.spyOn(process, 'exit')
.mockImplementation(()=>undefined as never);
const error = vi.spyOn(console, 'error')
.mockImplementation(()=>{});
await handle_discover('q', {});
expect(mocks.fail).toHaveBeenCalled();
exit.mockRestore();
error.mockRestore();
});
});
});
Loading
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Strip utm_, fbclid, gclid, etc. from all links on page\n(function() {\n var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',\n 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid',\n 'ref', 'ref_src', 'source', 'medium', 'campaign'];\n \n function cleanUrl(url) {\n try {\n var u = new URL(url, window.location.origin);\n var changed = false;\n trackingParams.forEach(function(p) {\n if (u.searchParams.has(p)) {\n u.searchParams.delete(p);\n changed = true;\n }\n });\n return changed ? u.toString() : url;\n } catch (e) {\n return url;\n }\n }\n \n function cleanLinks() {\n document.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n \n cleanLinks();\n \n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1) {\n if (node.tagName === 'A') cleanLinks();\n node.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Remove Tracking Parameters from Links"); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
56 changes: 56 additions & 0 deletions README.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -24,6 +24,7 @@
|---|---|
| `brightdata scrape` | Scrape any URL — bypasses CAPTCHAs, JS rendering, anti-bot protections |
| `brightdata search` | Google / Bing / Yandex search with structured JSON output |
| `brightdata discover` | AI-powered web discovery - find and rank results by intent with optional full-page content |
| `brightdata pipelines` | Extract structured data from 40+ platforms (Amazon, LinkedIn, TikTok…) |
| `brightdata browser` | Control a real browser via Bright Data's Scraping Browser — navigate, snapshot, click, type, and more |
| `brightdata zones` | List and inspect your Bright Data proxy zones |
Expand All@@ -44,6 +45,7 @@
- [init](#init)
- [scrape](#scrape)
- [search](#search)
- [discover](#discover)
- [pipelines](#pipelines)
- [browser](#browser)
- [status](#status)
Expand DownExpand Up@@ -246,6 +248,60 @@ brightdata search "bright data pricing" --engine bing

---

### `discover`

AI-powered web discovery. Submit a query with optional intent, and Bright Data finds, ranks, and optionally extracts full-page content for each result.

```bash
brightdata discover <query> [options]
```

| Flag | Description |
|---|---|
| `--intent <text>` | AI intent to evaluate and rank result relevance |
| `--country <code>` | ISO country code (default: `US`) |
| `--city <name>` | City for localized results (e.g. `"New York"`) |
| `--language <code>` | Language code (default: `en`) |
| `--num-results <n>` | Number of results to return |
| `--filter-keywords <words>` | Comma-separated keywords that must appear in results |
| `--include-content` | Include full page content in each result |
| `--no-remove-duplicates` | Keep duplicate results |
| `--start-date <date>` | Only content updated from date (`YYYY-MM-DD`) |
| `--end-date <date>` | Only content updated until date (`YYYY-MM-DD`) |
| `--timeout <seconds>` | Polling timeout (default: `600`) |
| `-o, --output <path>` | Write output to file |
| `--json` / `--pretty` | JSON output (raw / indented) |
| `-k, --api-key <key>` | Override API key |

**Examples**

```bash
# Basic discovery — table output
brightdata discover "AI trends"

# With AI intent for relevance ranking
brightdata discover "AI trends" \
--intent "Prioritize institutional reports for VC research"

# Include full page content as markdown
brightdata discover "AI trends" --include-content --num-results 5

# Geo-targeted with date range
brightdata discover "best restaurants" --country US --city "New York" \
--start-date 2025-01-01 --end-date 2025-12-31

# Filter results by keywords
brightdata discover "generative AI SaaS" --filter-keywords "revenue,SaaS"

# JSON output to file
brightdata discover "AI trends" --num-results 10 --pretty -o results.json

# Pipe-friendly — redirected stdout outputs JSON automatically
brightdata discover "AI trends" --include-content --num-results 3 > results.json
```

---

### `pipelines`

Extract structured data from 40+ platforms using Bright Data's Web Scraper API. Triggers an async collection job, polls until ready, and returns results.
Expand Down
2 changes: 1 addition & 1 deletion package.json
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,6 @@
{
"name": "@brightdata/cli",
"version": "0.1.6",
"version": "0.1.7",
"description": "Command-line interface for Bright Data. Scrape, search, extract structured data, and automate browsers directly from your terminal.",
"main": "dist/index.js",
"bin": {
Expand Down
260 changes: 260 additions & 0 deletions src/__tests__/commands/discover.test.ts
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,260 @@
import {describe, it, expect, beforeEach, vi} from 'vitest';

const mocks = vi.hoisted(()=>({
post: vi.fn(),
get: vi.fn(),
ensure_authenticated: vi.fn(),
stop: vi.fn(),
start: vi.fn(),
print: vi.fn(),
print_table: vi.fn(),
fail: vi.fn((msg: string)=>{ throw new Error(`fail:${msg}`); }),
dim: vi.fn((msg: string)=>msg),
parse_timeout: vi.fn(),
poll_until: vi.fn(),
}));

vi.mock('../../utils/client', ()=>({
post: mocks.post,
get: mocks.get,
}));

vi.mock('../../utils/auth', ()=>({
ensure_authenticated: mocks.ensure_authenticated,
}));

vi.mock('../../utils/spinner', ()=>({
start: mocks.start,
}));

vi.mock('../../utils/output', ()=>({
print: mocks.print,
print_table: mocks.print_table,
fail: mocks.fail,
dim: mocks.dim,
}));

vi.mock('../../utils/polling', ()=>({
parse_timeout: mocks.parse_timeout,
poll_until: mocks.poll_until,
}));

import {
handle_discover,
build_request,
extract_status,
format_markdown,
print_discover_table,
} from '../../commands/discover';

describe('commands/discover', ()=>{
beforeEach(()=>{
vi.clearAllMocks();
mocks.ensure_authenticated.mockReturnValue('api_key');
mocks.parse_timeout.mockReturnValue(600);
mocks.start.mockReturnValue({stop: mocks.stop});
});

describe('build_request', ()=>{
it('builds minimal request with only query', ()=>{
const req = build_request('AI trends', {});
expect(req).toEqual({query: 'AI trends'});
});

it('includes all optional params', ()=>{
const req = build_request('AI trends', {
intent: 'find research papers',
city: 'New York',
country: 'US',
language: 'en',
numResults: '10',
filterKeywords: 'AI, machine learning',
includeContent: true,
startDate: '2025-01-01',
endDate: '2025-12-31',
});
expect(req).toEqual({
query: 'AI trends',
intent: 'find research papers',
city: 'New York',
country: 'US',
language: 'en',
num_results: 10,
filter_keywords: ['AI', 'machine learning'],
include_content: true,
start_date: '2025-01-01',
end_date: '2025-12-31',
});
});

it('parses comma-separated filter keywords with whitespace', ()=>{
const req = build_request('q', {filterKeywords: ' a , b , c '});
expect(req.filter_keywords).toEqual(['a', 'b', 'c']);
});

it('does not set format by default (API returns JSON)', ()=>{
const req = build_request('test', {});
expect(req.format).toBeUndefined();
});

it('does not set format when include-content is used', ()=>{
const req = build_request('test', {includeContent: true});
expect(req.format).toBeUndefined();
expect(req.include_content).toBe(true);
});
});

describe('extract_status', ()=>{
it('returns status from valid response', ()=>{
expect(extract_status({status: 'processing'})).toBe('processing');
expect(extract_status({status: 'done'})).toBe('done');
});

it('returns undefined for invalid input', ()=>{
expect(extract_status(null as never)).toBeUndefined();
expect(extract_status(undefined as never)).toBeUndefined();
});
});

describe('format_markdown', ()=>{
it('formats results as markdown', ()=>{
const md = format_markdown([
{
link: 'https://example.com',
title: 'Example',
description: 'A description',
relevance_score: 0.95,
},
], 'test query');
expect(md).toContain('# Discover results for "test query"');
expect(md).toContain('**1. [Example](https://example.com)** (95.0%)');
expect(md).toContain('A description');
});

it('includes content when present', ()=>{
const md = format_markdown([
{
link: 'https://example.com',
title: 'Example',
description: 'Desc',
relevance_score: 0.5,
content: '# Page content here',
},
], 'q');
expect(md).toContain('# Page content here');
});
});

describe('print_discover_table', ()=>{
it('calls print_table with formatted rows', ()=>{
const results = [
{
link: 'https://example.com',
title: 'Example Title',
description: 'Desc',
relevance_score: 0.98184747,
},
];
print_discover_table(results);
expect(mocks.print_table).toHaveBeenCalledWith(
[{
'#': '1',
title: 'Example Title',
score: '98.2%',
url: 'https://example.com',
}],
['#', 'title', 'score', 'url']
);
});

it('prints dim message when no results', ()=>{
const log = vi.spyOn(console, 'log').mockImplementation(()=>{});
print_discover_table([]);
expect(log).toHaveBeenCalled();
expect(mocks.print_table).not.toHaveBeenCalled();
log.mockRestore();
});
});

describe('handle_discover', ()=>{
it('triggers and polls then prints table', async()=>{
mocks.post.mockResolvedValue({status: 'ok', task_id: 'abc123'});
mocks.poll_until.mockResolvedValue({
result: {
status: 'done',
duration_seconds: 5,
results: [
{
link: 'https://example.com',
title: 'Result',
description: 'Desc',
relevance_score: 0.9,
},
],
},
attempts: 3,
});
await handle_discover('AI trends', {});
expect(mocks.post).toHaveBeenCalledWith(
'api_key',
'/discover',
{query: 'AI trends'},
{timing: undefined}
);
expect(mocks.poll_until).toHaveBeenCalledTimes(1);
expect(mocks.print_table).toHaveBeenCalledTimes(1);
});

it('prints json when --json is set', async()=>{
const response = {
status: 'done',
duration_seconds: 2,
results: [{
link: 'https://example.com',
title: 'R',
description: 'D',
relevance_score: 0.8,
}],
};
mocks.post.mockResolvedValue({status: 'ok', task_id: 't1'});
mocks.poll_until.mockResolvedValue({result: response, attempts: 1});
await handle_discover('q', {json: true});
expect(mocks.print).toHaveBeenCalledWith(
response,
{json: true, pretty: undefined, output: undefined}
);
expect(mocks.print_table).not.toHaveBeenCalled();
});

it('prints raw JSON when --output is set', async()=>{
const response = {
status: 'done',
results: [{
link: 'https://example.com',
title: 'R',
description: 'D',
relevance_score: 0.7,
}],
};
mocks.post.mockResolvedValue({status: 'ok', task_id: 't2'});
mocks.poll_until.mockResolvedValue({result: response, attempts: 1});
await handle_discover('q', {output: 'out.json'});
expect(mocks.print).toHaveBeenCalledWith(
response,
{json: undefined, pretty: undefined, output: 'out.json'}
);
});

it('fails when trigger returns no task_id', async()=>{
mocks.post.mockResolvedValue({status: 'ok'});
const exit = vi.spyOn(process, 'exit')
.mockImplementation(()=>undefined as never);
const error = vi.spyOn(console, 'error')
.mockImplementation(()=>{});
await handle_discover('q', {});
expect(mocks.fail).toHaveBeenCalled();
exit.mockRestore();
error.mockRestore();
});
});
});
Loading
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Auto-enable theater mode on YouTube\n(function() {\n function tryTheater() {\n var btn = document.querySelector('button[aria-label=\"Theater mode\"], ytd-player #player button[title=\"Theater mode\"]');\n if (btn && !btn.classList.contains('activated')) {\n btn.click();\n }\n }\n \n // Try immediately\n tryTheater();\n \n // Try after navigation (SPA)\n var lastUrl = location.href;\n setInterval(function() {\n if (location.href !== lastUrl) {\n lastUrl = location.href;\n setTimeout(tryTheater, 500);\n }\n }, 1000);\n \n // Also try on player load\n var observer = new MutationObserver(tryTheater);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "YouTube Theater Mode Default"); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
56 changes: 56 additions & 0 deletions README.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -24,6 +24,7 @@
|---|---|
| `brightdata scrape` | Scrape any URL — bypasses CAPTCHAs, JS rendering, anti-bot protections |
| `brightdata search` | Google / Bing / Yandex search with structured JSON output |
| `brightdata discover` | AI-powered web discovery - find and rank results by intent with optional full-page content |
| `brightdata pipelines` | Extract structured data from 40+ platforms (Amazon, LinkedIn, TikTok…) |
| `brightdata browser` | Control a real browser via Bright Data's Scraping Browser — navigate, snapshot, click, type, and more |
| `brightdata zones` | List and inspect your Bright Data proxy zones |
Expand All@@ -44,6 +45,7 @@
- [init](#init)
- [scrape](#scrape)
- [search](#search)
- [discover](#discover)
- [pipelines](#pipelines)
- [browser](#browser)
- [status](#status)
Expand DownExpand Up@@ -246,6 +248,60 @@ brightdata search "bright data pricing" --engine bing

---

### `discover`

AI-powered web discovery. Submit a query with optional intent, and Bright Data finds, ranks, and optionally extracts full-page content for each result.

```bash
brightdata discover <query> [options]
```

| Flag | Description |
|---|---|
| `--intent <text>` | AI intent to evaluate and rank result relevance |
| `--country <code>` | ISO country code (default: `US`) |
| `--city <name>` | City for localized results (e.g. `"New York"`) |
| `--language <code>` | Language code (default: `en`) |
| `--num-results <n>` | Number of results to return |
| `--filter-keywords <words>` | Comma-separated keywords that must appear in results |
| `--include-content` | Include full page content in each result |
| `--no-remove-duplicates` | Keep duplicate results |
| `--start-date <date>` | Only content updated from date (`YYYY-MM-DD`) |
| `--end-date <date>` | Only content updated until date (`YYYY-MM-DD`) |
| `--timeout <seconds>` | Polling timeout (default: `600`) |
| `-o, --output <path>` | Write output to file |
| `--json` / `--pretty` | JSON output (raw / indented) |
| `-k, --api-key <key>` | Override API key |

**Examples**

```bash
# Basic discovery — table output
brightdata discover "AI trends"

# With AI intent for relevance ranking
brightdata discover "AI trends" \
--intent "Prioritize institutional reports for VC research"

# Include full page content as markdown
brightdata discover "AI trends" --include-content --num-results 5

# Geo-targeted with date range
brightdata discover "best restaurants" --country US --city "New York" \
--start-date 2025-01-01 --end-date 2025-12-31

# Filter results by keywords
brightdata discover "generative AI SaaS" --filter-keywords "revenue,SaaS"

# JSON output to file
brightdata discover "AI trends" --num-results 10 --pretty -o results.json

# Pipe-friendly — redirected stdout outputs JSON automatically
brightdata discover "AI trends" --include-content --num-results 3 > results.json
```

---

### `pipelines`

Extract structured data from 40+ platforms using Bright Data's Web Scraper API. Triggers an async collection job, polls until ready, and returns results.
Expand Down
2 changes: 1 addition & 1 deletion package.json
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,6 @@
{
"name": "@brightdata/cli",
"version": "0.1.6",
"version": "0.1.7",
"description": "Command-line interface for Bright Data. Scrape, search, extract structured data, and automate browsers directly from your terminal.",
"main": "dist/index.js",
"bin": {
Expand Down
260 changes: 260 additions & 0 deletions src/__tests__/commands/discover.test.ts
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,260 @@
import {describe, it, expect, beforeEach, vi} from 'vitest';

const mocks = vi.hoisted(()=>({
post: vi.fn(),
get: vi.fn(),
ensure_authenticated: vi.fn(),
stop: vi.fn(),
start: vi.fn(),
print: vi.fn(),
print_table: vi.fn(),
fail: vi.fn((msg: string)=>{ throw new Error(`fail:${msg}`); }),
dim: vi.fn((msg: string)=>msg),
parse_timeout: vi.fn(),
poll_until: vi.fn(),
}));

vi.mock('../../utils/client', ()=>({
post: mocks.post,
get: mocks.get,
}));

vi.mock('../../utils/auth', ()=>({
ensure_authenticated: mocks.ensure_authenticated,
}));

vi.mock('../../utils/spinner', ()=>({
start: mocks.start,
}));

vi.mock('../../utils/output', ()=>({
print: mocks.print,
print_table: mocks.print_table,
fail: mocks.fail,
dim: mocks.dim,
}));

vi.mock('../../utils/polling', ()=>({
parse_timeout: mocks.parse_timeout,
poll_until: mocks.poll_until,
}));

import {
handle_discover,
build_request,
extract_status,
format_markdown,
print_discover_table,
} from '../../commands/discover';

describe('commands/discover', ()=>{
beforeEach(()=>{
vi.clearAllMocks();
mocks.ensure_authenticated.mockReturnValue('api_key');
mocks.parse_timeout.mockReturnValue(600);
mocks.start.mockReturnValue({stop: mocks.stop});
});

describe('build_request', ()=>{
it('builds minimal request with only query', ()=>{
const req = build_request('AI trends', {});
expect(req).toEqual({query: 'AI trends'});
});

it('includes all optional params', ()=>{
const req = build_request('AI trends', {
intent: 'find research papers',
city: 'New York',
country: 'US',
language: 'en',
numResults: '10',
filterKeywords: 'AI, machine learning',
includeContent: true,
startDate: '2025-01-01',
endDate: '2025-12-31',
});
expect(req).toEqual({
query: 'AI trends',
intent: 'find research papers',
city: 'New York',
country: 'US',
language: 'en',
num_results: 10,
filter_keywords: ['AI', 'machine learning'],
include_content: true,
start_date: '2025-01-01',
end_date: '2025-12-31',
});
});

it('parses comma-separated filter keywords with whitespace', ()=>{
const req = build_request('q', {filterKeywords: ' a , b , c '});
expect(req.filter_keywords).toEqual(['a', 'b', 'c']);
});

it('does not set format by default (API returns JSON)', ()=>{
const req = build_request('test', {});
expect(req.format).toBeUndefined();
});

it('does not set format when include-content is used', ()=>{
const req = build_request('test', {includeContent: true});
expect(req.format).toBeUndefined();
expect(req.include_content).toBe(true);
});
});

describe('extract_status', ()=>{
it('returns status from valid response', ()=>{
expect(extract_status({status: 'processing'})).toBe('processing');
expect(extract_status({status: 'done'})).toBe('done');
});

it('returns undefined for invalid input', ()=>{
expect(extract_status(null as never)).toBeUndefined();
expect(extract_status(undefined as never)).toBeUndefined();
});
});

describe('format_markdown', ()=>{
it('formats results as markdown', ()=>{
const md = format_markdown([
{
link: 'https://example.com',
title: 'Example',
description: 'A description',
relevance_score: 0.95,
},
], 'test query');
expect(md).toContain('# Discover results for "test query"');
expect(md).toContain('**1. [Example](https://example.com)** (95.0%)');
expect(md).toContain('A description');
});

it('includes content when present', ()=>{
const md = format_markdown([
{
link: 'https://example.com',
title: 'Example',
description: 'Desc',
relevance_score: 0.5,
content: '# Page content here',
},
], 'q');
expect(md).toContain('# Page content here');
});
});

describe('print_discover_table', ()=>{
it('calls print_table with formatted rows', ()=>{
const results = [
{
link: 'https://example.com',
title: 'Example Title',
description: 'Desc',
relevance_score: 0.98184747,
},
];
print_discover_table(results);
expect(mocks.print_table).toHaveBeenCalledWith(
[{
'#': '1',
title: 'Example Title',
score: '98.2%',
url: 'https://example.com',
}],
['#', 'title', 'score', 'url']
);
});

it('prints dim message when no results', ()=>{
const log = vi.spyOn(console, 'log').mockImplementation(()=>{});
print_discover_table([]);
expect(log).toHaveBeenCalled();
expect(mocks.print_table).not.toHaveBeenCalled();
log.mockRestore();
});
});

describe('handle_discover', ()=>{
it('triggers and polls then prints table', async()=>{
mocks.post.mockResolvedValue({status: 'ok', task_id: 'abc123'});
mocks.poll_until.mockResolvedValue({
result: {
status: 'done',
duration_seconds: 5,
results: [
{
link: 'https://example.com',
title: 'Result',
description: 'Desc',
relevance_score: 0.9,
},
],
},
attempts: 3,
});
await handle_discover('AI trends', {});
expect(mocks.post).toHaveBeenCalledWith(
'api_key',
'/discover',
{query: 'AI trends'},
{timing: undefined}
);
expect(mocks.poll_until).toHaveBeenCalledTimes(1);
expect(mocks.print_table).toHaveBeenCalledTimes(1);
});

it('prints json when --json is set', async()=>{
const response = {
status: 'done',
duration_seconds: 2,
results: [{
link: 'https://example.com',
title: 'R',
description: 'D',
relevance_score: 0.8,
}],
};
mocks.post.mockResolvedValue({status: 'ok', task_id: 't1'});
mocks.poll_until.mockResolvedValue({result: response, attempts: 1});
await handle_discover('q', {json: true});
expect(mocks.print).toHaveBeenCalledWith(
response,
{json: true, pretty: undefined, output: undefined}
);
expect(mocks.print_table).not.toHaveBeenCalled();
});

it('prints raw JSON when --output is set', async()=>{
const response = {
status: 'done',
results: [{
link: 'https://example.com',
title: 'R',
description: 'D',
relevance_score: 0.7,
}],
};
mocks.post.mockResolvedValue({status: 'ok', task_id: 't2'});
mocks.poll_until.mockResolvedValue({result: response, attempts: 1});
await handle_discover('q', {output: 'out.json'});
expect(mocks.print).toHaveBeenCalledWith(
response,
{json: undefined, pretty: undefined, output: 'out.json'}
);
});

it('fails when trigger returns no task_id', async()=>{
mocks.post.mockResolvedValue({status: 'ok'});
const exit = vi.spyOn(process, 'exit')
.mockImplementation(()=>undefined as never);
const error = vi.spyOn(console, 'error')
.mockImplementation(()=>{});
await handle_discover('q', {});
expect(mocks.fail).toHaveBeenCalled();
exit.mockRestore();
error.mockRestore();
});
});
});
Loading
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Remove or un-stick sticky/fixed headers that block content\n(function() {\n function unstick() {\n document.querySelectorAll('header, nav, [role=\"banner\"], .header, .navbar, .sticky, .fixed-top, [style*=\"position: fixed\"], [style*=\"position:sticky\"]').forEach(function(el) {\n if (el.style.position === 'fixed' || el.style.position === 'sticky' || \n getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') {\n el.style.position = 'static';\n el.style.top = 'auto';\n el.style.zIndex = 'auto';\n }\n });\n }\n \n unstick();\n \n var observer = new MutationObserver(unstick);\n observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] });\n})();", "Kill Sticky Headers"); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
56 changes: 56 additions & 0 deletions README.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -24,6 +24,7 @@
|---|---|
| `brightdata scrape` | Scrape any URL — bypasses CAPTCHAs, JS rendering, anti-bot protections |
| `brightdata search` | Google / Bing / Yandex search with structured JSON output |
| `brightdata discover` | AI-powered web discovery - find and rank results by intent with optional full-page content |
| `brightdata pipelines` | Extract structured data from 40+ platforms (Amazon, LinkedIn, TikTok…) |
| `brightdata browser` | Control a real browser via Bright Data's Scraping Browser — navigate, snapshot, click, type, and more |
| `brightdata zones` | List and inspect your Bright Data proxy zones |
Expand All@@ -44,6 +45,7 @@
- [init](#init)
- [scrape](#scrape)
- [search](#search)
- [discover](#discover)
- [pipelines](#pipelines)
- [browser](#browser)
- [status](#status)
Expand DownExpand Up@@ -246,6 +248,60 @@ brightdata search "bright data pricing" --engine bing

---

### `discover`

AI-powered web discovery. Submit a query with optional intent, and Bright Data finds, ranks, and optionally extracts full-page content for each result.

```bash
brightdata discover <query> [options]
```

| Flag | Description |
|---|---|
| `--intent <text>` | AI intent to evaluate and rank result relevance |
| `--country <code>` | ISO country code (default: `US`) |
| `--city <name>` | City for localized results (e.g. `"New York"`) |
| `--language <code>` | Language code (default: `en`) |
| `--num-results <n>` | Number of results to return |
| `--filter-keywords <words>` | Comma-separated keywords that must appear in results |
| `--include-content` | Include full page content in each result |
| `--no-remove-duplicates` | Keep duplicate results |
| `--start-date <date>` | Only content updated from date (`YYYY-MM-DD`) |
| `--end-date <date>` | Only content updated until date (`YYYY-MM-DD`) |
| `--timeout <seconds>` | Polling timeout (default: `600`) |
| `-o, --output <path>` | Write output to file |
| `--json` / `--pretty` | JSON output (raw / indented) |
| `-k, --api-key <key>` | Override API key |

**Examples**

```bash
# Basic discovery — table output
brightdata discover "AI trends"

# With AI intent for relevance ranking
brightdata discover "AI trends" \
--intent "Prioritize institutional reports for VC research"

# Include full page content as markdown
brightdata discover "AI trends" --include-content --num-results 5

# Geo-targeted with date range
brightdata discover "best restaurants" --country US --city "New York" \
--start-date 2025-01-01 --end-date 2025-12-31

# Filter results by keywords
brightdata discover "generative AI SaaS" --filter-keywords "revenue,SaaS"

# JSON output to file
brightdata discover "AI trends" --num-results 10 --pretty -o results.json

# Pipe-friendly — redirected stdout outputs JSON automatically
brightdata discover "AI trends" --include-content --num-results 3 > results.json
```

---

### `pipelines`

Extract structured data from 40+ platforms using Bright Data's Web Scraper API. Triggers an async collection job, polls until ready, and returns results.
Expand Down
2 changes: 1 addition & 1 deletion package.json
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,6 @@
{
"name": "@brightdata/cli",
"version": "0.1.6",
"version": "0.1.7",
"description": "Command-line interface for Bright Data. Scrape, search, extract structured data, and automate browsers directly from your terminal.",
"main": "dist/index.js",
"bin": {
Expand Down
260 changes: 260 additions & 0 deletions src/__tests__/commands/discover.test.ts
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,260 @@
import {describe, it, expect, beforeEach, vi} from 'vitest';

const mocks = vi.hoisted(()=>({
post: vi.fn(),
get: vi.fn(),
ensure_authenticated: vi.fn(),
stop: vi.fn(),
start: vi.fn(),
print: vi.fn(),
print_table: vi.fn(),
fail: vi.fn((msg: string)=>{ throw new Error(`fail:${msg}`); }),
dim: vi.fn((msg: string)=>msg),
parse_timeout: vi.fn(),
poll_until: vi.fn(),
}));

vi.mock('../../utils/client', ()=>({
post: mocks.post,
get: mocks.get,
}));

vi.mock('../../utils/auth', ()=>({
ensure_authenticated: mocks.ensure_authenticated,
}));

vi.mock('../../utils/spinner', ()=>({
start: mocks.start,
}));

vi.mock('../../utils/output', ()=>({
print: mocks.print,
print_table: mocks.print_table,
fail: mocks.fail,
dim: mocks.dim,
}));

vi.mock('../../utils/polling', ()=>({
parse_timeout: mocks.parse_timeout,
poll_until: mocks.poll_until,
}));

import {
handle_discover,
build_request,
extract_status,
format_markdown,
print_discover_table,
} from '../../commands/discover';

describe('commands/discover', ()=>{
beforeEach(()=>{
vi.clearAllMocks();
mocks.ensure_authenticated.mockReturnValue('api_key');
mocks.parse_timeout.mockReturnValue(600);
mocks.start.mockReturnValue({stop: mocks.stop});
});

describe('build_request', ()=>{
it('builds minimal request with only query', ()=>{
const req = build_request('AI trends', {});
expect(req).toEqual({query: 'AI trends'});
});

it('includes all optional params', ()=>{
const req = build_request('AI trends', {
intent: 'find research papers',
city: 'New York',
country: 'US',
language: 'en',
numResults: '10',
filterKeywords: 'AI, machine learning',
includeContent: true,
startDate: '2025-01-01',
endDate: '2025-12-31',
});
expect(req).toEqual({
query: 'AI trends',
intent: 'find research papers',
city: 'New York',
country: 'US',
language: 'en',
num_results: 10,
filter_keywords: ['AI', 'machine learning'],
include_content: true,
start_date: '2025-01-01',
end_date: '2025-12-31',
});
});

it('parses comma-separated filter keywords with whitespace', ()=>{
const req = build_request('q', {filterKeywords: ' a , b , c '});
expect(req.filter_keywords).toEqual(['a', 'b', 'c']);
});

it('does not set format by default (API returns JSON)', ()=>{
const req = build_request('test', {});
expect(req.format).toBeUndefined();
});

it('does not set format when include-content is used', ()=>{
const req = build_request('test', {includeContent: true});
expect(req.format).toBeUndefined();
expect(req.include_content).toBe(true);
});
});

describe('extract_status', ()=>{
it('returns status from valid response', ()=>{
expect(extract_status({status: 'processing'})).toBe('processing');
expect(extract_status({status: 'done'})).toBe('done');
});

it('returns undefined for invalid input', ()=>{
expect(extract_status(null as never)).toBeUndefined();
expect(extract_status(undefined as never)).toBeUndefined();
});
});

describe('format_markdown', ()=>{
it('formats results as markdown', ()=>{
const md = format_markdown([
{
link: 'https://example.com',
title: 'Example',
description: 'A description',
relevance_score: 0.95,
},
], 'test query');
expect(md).toContain('# Discover results for "test query"');
expect(md).toContain('**1. [Example](https://example.com)** (95.0%)');
expect(md).toContain('A description');
});

it('includes content when present', ()=>{
const md = format_markdown([
{
link: 'https://example.com',
title: 'Example',
description: 'Desc',
relevance_score: 0.5,
content: '# Page content here',
},
], 'q');
expect(md).toContain('# Page content here');
});
});

describe('print_discover_table', ()=>{
it('calls print_table with formatted rows', ()=>{
const results = [
{
link: 'https://example.com',
title: 'Example Title',
description: 'Desc',
relevance_score: 0.98184747,
},
];
print_discover_table(results);
expect(mocks.print_table).toHaveBeenCalledWith(
[{
'#': '1',
title: 'Example Title',
score: '98.2%',
url: 'https://example.com',
}],
['#', 'title', 'score', 'url']
);
});

it('prints dim message when no results', ()=>{
const log = vi.spyOn(console, 'log').mockImplementation(()=>{});
print_discover_table([]);
expect(log).toHaveBeenCalled();
expect(mocks.print_table).not.toHaveBeenCalled();
log.mockRestore();
});
});

describe('handle_discover', ()=>{
it('triggers and polls then prints table', async()=>{
mocks.post.mockResolvedValue({status: 'ok', task_id: 'abc123'});
mocks.poll_until.mockResolvedValue({
result: {
status: 'done',
duration_seconds: 5,
results: [
{
link: 'https://example.com',
title: 'Result',
description: 'Desc',
relevance_score: 0.9,
},
],
},
attempts: 3,
});
await handle_discover('AI trends', {});
expect(mocks.post).toHaveBeenCalledWith(
'api_key',
'/discover',
{query: 'AI trends'},
{timing: undefined}
);
expect(mocks.poll_until).toHaveBeenCalledTimes(1);
expect(mocks.print_table).toHaveBeenCalledTimes(1);
});

it('prints json when --json is set', async()=>{
const response = {
status: 'done',
duration_seconds: 2,
results: [{
link: 'https://example.com',
title: 'R',
description: 'D',
relevance_score: 0.8,
}],
};
mocks.post.mockResolvedValue({status: 'ok', task_id: 't1'});
mocks.poll_until.mockResolvedValue({result: response, attempts: 1});
await handle_discover('q', {json: true});
expect(mocks.print).toHaveBeenCalledWith(
response,
{json: true, pretty: undefined, output: undefined}
);
expect(mocks.print_table).not.toHaveBeenCalled();
});

it('prints raw JSON when --output is set', async()=>{
const response = {
status: 'done',
results: [{
link: 'https://example.com',
title: 'R',
description: 'D',
relevance_score: 0.7,
}],
};
mocks.post.mockResolvedValue({status: 'ok', task_id: 't2'});
mocks.poll_until.mockResolvedValue({result: response, attempts: 1});
await handle_discover('q', {output: 'out.json'});
expect(mocks.print).toHaveBeenCalledWith(
response,
{json: undefined, pretty: undefined, output: 'out.json'}
);
});

it('fails when trigger returns no task_id', async()=>{
mocks.post.mockResolvedValue({status: 'ok'});
const exit = vi.spyOn(process, 'exit')
.mockImplementation(()=>undefined as never);
const error = vi.spyOn(console, 'error')
.mockImplementation(()=>{});
await handle_discover('q', {});
expect(mocks.fail).toHaveBeenCalled();
exit.mockRestore();
error.mockRestore();
});
});
});
Loading
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Universal Dark Mode - works on any site\n(function() {\n var enabled = true;\n \n function applyDarkMode() {\n if (!enabled) return;\n \n // Create style element if it doesn't exist\n var style = document.getElementById('universal-dark-mode-style');\n if (!style) {\n style = document.createElement('style');\n style.id = 'universal-dark-mode-style';\n document.head.appendChild(style);\n }\n \n // Dark mode CSS - inverts colors but preserves images/video\n style.textContent = '\n /* Invert everything except media */\n html {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #1a1a2e !important;\n }\n \n /* Restore images, videos, iframes, canvas */\n img, video, iframe, canvas, svg, picture, [style*=\"background-image\"] {\n filter: invert(1) hue-rotate(180deg) !important;\n }\n \n /* Preserve specific elements that should not be inverted */\n .no-dark-mode, .no-dark-mode *,\n [data-theme=\"light\"], [data-theme=\"light\"],\n .ace_editor, .ace_editor *,\n .CodeMirror, .CodeMirror *,\n .monaco-editor, .monaco-editor *,\n .markdown-body pre, .markdown-body pre *,\n .highlight, .highlight *,\n pre code, pre code * {\n filter: none !important;\n }\n \n /* Fix common UI elements */\n .modal, .popup, .dropdown-menu, .tooltip, .popover {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #2d2d44 !important;\n border-color: #444 !important;\n }\n \n /* Scrollbars */\n ::-webkit-scrollbar { background: #1a1a2e !important; }\n ::-webkit-scrollbar-thumb { background: #444 !important; }\n ::-webkit-scrollbar-thumb:hover { background: #555 !important; }\n \n /* Selection */\n ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ';\n }\n \n function removeDarkMode() {\n var style = document.getElementById('universal-dark-mode-style');\n if (style) style.remove();\n }\n \n // Toggle with Alt+Shift+D\n document.addEventListener('keydown', function(e) {\n if (e.altKey && e.shiftKey && e.key === 'D') {\n e.preventDefault();\n enabled = !enabled;\n if (enabled) {\n applyDarkMode();\n console.log('[Universal Dark Mode] Enabled');\n } else {\n removeDarkMode();\n console.log('[Universal Dark Mode] Disabled');\n }\n }\n });\n \n // Apply on load\n applyDarkMode();\n \n // Re-apply on dynamic content\n var observer = new MutationObserver(function(mutations) {\n if (enabled && !document.getElementById('universal-dark-mode-style')) {\n applyDarkMode();\n }\n });\n observer.observe(document.head, { childList: true });\n \n console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle');\n})();", "Universal Dark Mode"); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
56 changes: 56 additions & 0 deletions README.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -24,6 +24,7 @@
|---|---|
| `brightdata scrape` | Scrape any URL — bypasses CAPTCHAs, JS rendering, anti-bot protections |
| `brightdata search` | Google / Bing / Yandex search with structured JSON output |
| `brightdata discover` | AI-powered web discovery - find and rank results by intent with optional full-page content |
| `brightdata pipelines` | Extract structured data from 40+ platforms (Amazon, LinkedIn, TikTok…) |
| `brightdata browser` | Control a real browser via Bright Data's Scraping Browser — navigate, snapshot, click, type, and more |
| `brightdata zones` | List and inspect your Bright Data proxy zones |
Expand All@@ -44,6 +45,7 @@
- [init](#init)
- [scrape](#scrape)
- [search](#search)
- [discover](#discover)
- [pipelines](#pipelines)
- [browser](#browser)
- [status](#status)
Expand DownExpand Up@@ -246,6 +248,60 @@ brightdata search "bright data pricing" --engine bing

---

### `discover`

AI-powered web discovery. Submit a query with optional intent, and Bright Data finds, ranks, and optionally extracts full-page content for each result.

```bash
brightdata discover <query> [options]
```

| Flag | Description |
|---|---|
| `--intent <text>` | AI intent to evaluate and rank result relevance |
| `--country <code>` | ISO country code (default: `US`) |
| `--city <name>` | City for localized results (e.g. `"New York"`) |
| `--language <code>` | Language code (default: `en`) |
| `--num-results <n>` | Number of results to return |
| `--filter-keywords <words>` | Comma-separated keywords that must appear in results |
| `--include-content` | Include full page content in each result |
| `--no-remove-duplicates` | Keep duplicate results |
| `--start-date <date>` | Only content updated from date (`YYYY-MM-DD`) |
| `--end-date <date>` | Only content updated until date (`YYYY-MM-DD`) |
| `--timeout <seconds>` | Polling timeout (default: `600`) |
| `-o, --output <path>` | Write output to file |
| `--json` / `--pretty` | JSON output (raw / indented) |
| `-k, --api-key <key>` | Override API key |

**Examples**

```bash
# Basic discovery — table output
brightdata discover "AI trends"

# With AI intent for relevance ranking
brightdata discover "AI trends" \
--intent "Prioritize institutional reports for VC research"

# Include full page content as markdown
brightdata discover "AI trends" --include-content --num-results 5

# Geo-targeted with date range
brightdata discover "best restaurants" --country US --city "New York" \
--start-date 2025-01-01 --end-date 2025-12-31

# Filter results by keywords
brightdata discover "generative AI SaaS" --filter-keywords "revenue,SaaS"

# JSON output to file
brightdata discover "AI trends" --num-results 10 --pretty -o results.json

# Pipe-friendly — redirected stdout outputs JSON automatically
brightdata discover "AI trends" --include-content --num-results 3 > results.json
```

---

### `pipelines`

Extract structured data from 40+ platforms using Bright Data's Web Scraper API. Triggers an async collection job, polls until ready, and returns results.
Expand Down
2 changes: 1 addition & 1 deletion package.json
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,6 @@
{
"name": "@brightdata/cli",
"version": "0.1.6",
"version": "0.1.7",
"description": "Command-line interface for Bright Data. Scrape, search, extract structured data, and automate browsers directly from your terminal.",
"main": "dist/index.js",
"bin": {
Expand Down
260 changes: 260 additions & 0 deletions src/__tests__/commands/discover.test.ts
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,260 @@
import {describe, it, expect, beforeEach, vi} from 'vitest';

const mocks = vi.hoisted(()=>({
post: vi.fn(),
get: vi.fn(),
ensure_authenticated: vi.fn(),
stop: vi.fn(),
start: vi.fn(),
print: vi.fn(),
print_table: vi.fn(),
fail: vi.fn((msg: string)=>{ throw new Error(`fail:${msg}`); }),
dim: vi.fn((msg: string)=>msg),
parse_timeout: vi.fn(),
poll_until: vi.fn(),
}));

vi.mock('../../utils/client', ()=>({
post: mocks.post,
get: mocks.get,
}));

vi.mock('../../utils/auth', ()=>({
ensure_authenticated: mocks.ensure_authenticated,
}));

vi.mock('../../utils/spinner', ()=>({
start: mocks.start,
}));

vi.mock('../../utils/output', ()=>({
print: mocks.print,
print_table: mocks.print_table,
fail: mocks.fail,
dim: mocks.dim,
}));

vi.mock('../../utils/polling', ()=>({
parse_timeout: mocks.parse_timeout,
poll_until: mocks.poll_until,
}));

import {
handle_discover,
build_request,
extract_status,
format_markdown,
print_discover_table,
} from '../../commands/discover';

describe('commands/discover', ()=>{
beforeEach(()=>{
vi.clearAllMocks();
mocks.ensure_authenticated.mockReturnValue('api_key');
mocks.parse_timeout.mockReturnValue(600);
mocks.start.mockReturnValue({stop: mocks.stop});
});

describe('build_request', ()=>{
it('builds minimal request with only query', ()=>{
const req = build_request('AI trends', {});
expect(req).toEqual({query: 'AI trends'});
});

it('includes all optional params', ()=>{
const req = build_request('AI trends', {
intent: 'find research papers',
city: 'New York',
country: 'US',
language: 'en',
numResults: '10',
filterKeywords: 'AI, machine learning',
includeContent: true,
startDate: '2025-01-01',
endDate: '2025-12-31',
});
expect(req).toEqual({
query: 'AI trends',
intent: 'find research papers',
city: 'New York',
country: 'US',
language: 'en',
num_results: 10,
filter_keywords: ['AI', 'machine learning'],
include_content: true,
start_date: '2025-01-01',
end_date: '2025-12-31',
});
});

it('parses comma-separated filter keywords with whitespace', ()=>{
const req = build_request('q', {filterKeywords: ' a , b , c '});
expect(req.filter_keywords).toEqual(['a', 'b', 'c']);
});

it('does not set format by default (API returns JSON)', ()=>{
const req = build_request('test', {});
expect(req.format).toBeUndefined();
});

it('does not set format when include-content is used', ()=>{
const req = build_request('test', {includeContent: true});
expect(req.format).toBeUndefined();
expect(req.include_content).toBe(true);
});
});

describe('extract_status', ()=>{
it('returns status from valid response', ()=>{
expect(extract_status({status: 'processing'})).toBe('processing');
expect(extract_status({status: 'done'})).toBe('done');
});

it('returns undefined for invalid input', ()=>{
expect(extract_status(null as never)).toBeUndefined();
expect(extract_status(undefined as never)).toBeUndefined();
});
});

describe('format_markdown', ()=>{
it('formats results as markdown', ()=>{
const md = format_markdown([
{
link: 'https://example.com',
title: 'Example',
description: 'A description',
relevance_score: 0.95,
},
], 'test query');
expect(md).toContain('# Discover results for "test query"');
expect(md).toContain('**1. [Example](https://example.com)** (95.0%)');
expect(md).toContain('A description');
});

it('includes content when present', ()=>{
const md = format_markdown([
{
link: 'https://example.com',
title: 'Example',
description: 'Desc',
relevance_score: 0.5,
content: '# Page content here',
},
], 'q');
expect(md).toContain('# Page content here');
});
});

describe('print_discover_table', ()=>{
it('calls print_table with formatted rows', ()=>{
const results = [
{
link: 'https://example.com',
title: 'Example Title',
description: 'Desc',
relevance_score: 0.98184747,
},
];
print_discover_table(results);
expect(mocks.print_table).toHaveBeenCalledWith(
[{
'#': '1',
title: 'Example Title',
score: '98.2%',
url: 'https://example.com',
}],
['#', 'title', 'score', 'url']
);
});

it('prints dim message when no results', ()=>{
const log = vi.spyOn(console, 'log').mockImplementation(()=>{});
print_discover_table([]);
expect(log).toHaveBeenCalled();
expect(mocks.print_table).not.toHaveBeenCalled();
log.mockRestore();
});
});

describe('handle_discover', ()=>{
it('triggers and polls then prints table', async()=>{
mocks.post.mockResolvedValue({status: 'ok', task_id: 'abc123'});
mocks.poll_until.mockResolvedValue({
result: {
status: 'done',
duration_seconds: 5,
results: [
{
link: 'https://example.com',
title: 'Result',
description: 'Desc',
relevance_score: 0.9,
},
],
},
attempts: 3,
});
await handle_discover('AI trends', {});
expect(mocks.post).toHaveBeenCalledWith(
'api_key',
'/discover',
{query: 'AI trends'},
{timing: undefined}
);
expect(mocks.poll_until).toHaveBeenCalledTimes(1);
expect(mocks.print_table).toHaveBeenCalledTimes(1);
});

it('prints json when --json is set', async()=>{
const response = {
status: 'done',
duration_seconds: 2,
results: [{
link: 'https://example.com',
title: 'R',
description: 'D',
relevance_score: 0.8,
}],
};
mocks.post.mockResolvedValue({status: 'ok', task_id: 't1'});
mocks.poll_until.mockResolvedValue({result: response, attempts: 1});
await handle_discover('q', {json: true});
expect(mocks.print).toHaveBeenCalledWith(
response,
{json: true, pretty: undefined, output: undefined}
);
expect(mocks.print_table).not.toHaveBeenCalled();
});

it('prints raw JSON when --output is set', async()=>{
const response = {
status: 'done',
results: [{
link: 'https://example.com',
title: 'R',
description: 'D',
relevance_score: 0.7,
}],
};
mocks.post.mockResolvedValue({status: 'ok', task_id: 't2'});
mocks.poll_until.mockResolvedValue({result: response, attempts: 1});
await handle_discover('q', {output: 'out.json'});
expect(mocks.print).toHaveBeenCalledWith(
response,
{json: undefined, pretty: undefined, output: 'out.json'}
);
});

it('fails when trigger returns no task_id', async()=>{
mocks.post.mockResolvedValue({status: 'ok'});
const exit = vi.spyOn(process, 'exit')
.mockImplementation(()=>undefined as never);
const error = vi.spyOn(console, 'error')
.mockImplementation(()=>{});
await handle_discover('q', {});
expect(mocks.fail).toHaveBeenCalled();
exit.mockRestore();
error.mockRestore();
});
});
});
Loading
Loading