-
Notifications
You must be signed in to change notification settings - Fork 21
feat: recommend specialized Actors via the TIP key #147
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Merged
Merged
Changes from all commits
Commits
Show all changes
4 commits
Select commit
Hold shift + click to select a range
2275cb0
feat: recommend specialized Actors via the TIP key
nikitachapovskii-dev 690ffaa
chore: don't crash on fail to store
nikitachapovskii-dev c401013
chore: remove redundant Actor.fail
nikitachapovskii-dev 0d5c343
fix: detect listed sites in sentences and under country domains
nikitachapovskii-dev File filter
Filter by extension
Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
There are no files selected for viewing
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.
Oops, something went wrong.
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,161 @@ | ||
| import { Actor } from 'apify'; | ||
| import { log } from 'crawlee'; | ||
| import { getDomainWithoutSuffix } from 'tldts'; | ||
|
|
||
| import { TIP_KVS_KEY } from './const.js'; | ||
|
|
||
| export type ActorTip = { | ||
| message: string; | ||
| level: 'info' | 'warning'; | ||
| }; | ||
|
|
||
| type SiteRule = { | ||
| site: string; | ||
| actorTitle: string; | ||
| actorUrl: string; | ||
| domains: string[]; | ||
| matchesUrl?: (url: URL) => boolean; | ||
| }; | ||
|
|
||
| function hostnameMatchesDomain(hostname: string, domain: string): boolean { | ||
| const normalizedHostname = hostname.toLowerCase().replace(/\.$/, ''); | ||
|
|
||
| // `example.*` matches the site under any of its country domains, such as `amazon.co.uk`. | ||
| if (domain.endsWith('.*')) { | ||
| return getDomainWithoutSuffix(normalizedHostname) === domain.slice(0, -2); | ||
| } | ||
|
|
||
| return normalizedHostname === domain || normalizedHostname.endsWith(`.${domain}`); | ||
| } | ||
|
|
||
| function pathStartsWith(pathname: string, segment: string): boolean { | ||
| return pathname === segment || pathname.startsWith(`${segment}/`); | ||
| } | ||
|
|
||
| /** Specialized Actors we recommend instead of general-purpose scraping. The first matching rule wins. */ | ||
| const SITE_RULES: SiteRule[] = [ | ||
| { | ||
| site: 'Facebook groups', | ||
| actorTitle: 'Facebook Groups Scraper', | ||
| actorUrl: 'https://apify.com/apify/facebook-groups-scraper', | ||
| domains: ['facebook.com'], | ||
| matchesUrl: (url) => pathStartsWith(url.pathname, '/groups'), | ||
| }, | ||
| { | ||
| site: 'facebook.com', | ||
| actorTitle: 'Facebook Posts Scraper', | ||
| actorUrl: 'https://apify.com/apify/facebook-posts-scraper', | ||
| domains: ['facebook.com'], | ||
| }, | ||
| { | ||
| // Plain Google Search is what this Actor does itself, so only Google Maps gets a recommendation. | ||
| site: 'Google Maps', | ||
| actorTitle: 'Google Maps Scraper', | ||
| actorUrl: 'https://apify.com/compass/crawler-google-places', | ||
| domains: ['google.*'], | ||
| matchesUrl: (url) => url.hostname.startsWith('maps.') || pathStartsWith(url.pathname, '/maps'), | ||
| }, | ||
| { | ||
| site: 'zillow.com', | ||
| actorTitle: 'Zillow Detail Scraper', | ||
| actorUrl: 'https://apify.com/maxcopell/zillow-detail-scraper', | ||
| domains: ['zillow.com'], | ||
| }, | ||
| { | ||
| site: 'instagram.com', | ||
| actorTitle: 'Instagram Scraper', | ||
| actorUrl: 'https://apify.com/apify/instagram-scraper', | ||
| domains: ['instagram.com'], | ||
| }, | ||
| { | ||
| site: 'booking.com', | ||
| actorTitle: 'Booking Scraper', | ||
| actorUrl: 'https://apify.com/voyager/booking-scraper', | ||
| domains: ['booking.com'], | ||
| }, | ||
| { | ||
| site: 'tripadvisor.com', | ||
| actorTitle: 'Tripadvisor Scraper', | ||
| actorUrl: 'https://apify.com/maxcopell/tripadvisor', | ||
| domains: ['tripadvisor.*'], | ||
| }, | ||
| { | ||
| site: 'youtube.com', | ||
| actorTitle: 'YouTube Scraper', | ||
| actorUrl: 'https://apify.com/streamers/youtube-scraper', | ||
| domains: ['youtube.com', 'youtu.be'], | ||
| }, | ||
| { | ||
| site: 'tiktok.com', | ||
| actorTitle: 'TikTok Scraper', | ||
| actorUrl: 'https://apify.com/clockworks/tiktok-scraper', | ||
| domains: ['tiktok.com'], | ||
| }, | ||
| { | ||
| site: 'amazon.com', | ||
| actorTitle: 'Amazon Crawler', | ||
| actorUrl: 'https://apify.com/junglee/amazon-crawler', | ||
| domains: ['amazon.*'], | ||
| }, | ||
| ]; | ||
|
|
||
| /** | ||
| * Without this, whether a domain is recognized in a sentence depends on whether the punctuation after it | ||
| * happens to be a URL delimiter: `zillow.com.` and `zillow.com?` parse fine, `zillow.com,` does not. | ||
| */ | ||
| const SURROUNDING_PUNCTUATION = /^["'([]+|["')\],.;:!?]+$/g; | ||
| const SEARCH_OPERATOR_PREFIX = /^[a-z]+:(?!\/\/)/i; | ||
|
|
||
| function interpretTokenAsUrl(token: string): URL | null { | ||
| const candidate = token.replace(SURROUNDING_PUNCTUATION, '').replace(SEARCH_OPERATOR_PREFIX, ''); | ||
| if (!candidate.includes('.')) return null; | ||
|
|
||
| try { | ||
| return new URL(/^https?:\/\//i.test(candidate) ? candidate : `https://${candidate}`); | ||
| } catch { | ||
| return null; | ||
| } | ||
| } | ||
|
|
||
| function findRuleForUrl(url: URL): SiteRule | undefined { | ||
| return SITE_RULES.find((rule) => { | ||
| const matchesDomain = rule.domains.some((domain) => hostnameMatchesDomain(url.hostname, domain)); | ||
| return matchesDomain && (rule.matchesUrl?.(url) ?? true); | ||
| }); | ||
| } | ||
|
|
||
| export function findActorTip(input: { query?: string; url?: string }): ActorTip | null { | ||
| const target = input.query || input.url; | ||
| if (!target) return null; | ||
|
|
||
| for (const token of target.split(/\s+/)) { | ||
| if (token.startsWith('-')) continue; | ||
|
|
||
| const url = interpretTokenAsUrl(token); | ||
| const rule = url ? findRuleForUrl(url) : undefined; | ||
| if (!rule) continue; | ||
|
|
||
| return { | ||
| message: `For scraping ${rule.site}, we recommend using [${rule.actorTitle}](${rule.actorUrl})`, | ||
| level: 'info', | ||
| }; | ||
| } | ||
|
|
||
| return null; | ||
| } | ||
|
|
||
| /** | ||
| * Stores the tip under the reserved key-value store key. Call it only once the crawlers have finished: | ||
| * Crawlee purges the default storages while they start up, which wipes a record written before that. | ||
| * | ||
| * A tip is advisory, so a failure to store it never changes the outcome of the run. | ||
| */ | ||
| export async function storeActorTip(tip: ActorTip | null): Promise<void> { | ||
| if (!tip) return; | ||
|
|
||
| try { | ||
| await Actor.setValue(TIP_KVS_KEY, tip); | ||
| } catch (err) { | ||
| log.warning(`Failed to store the tip: ${err instanceof Error ? err.message : String(err)}`); | ||
| } | ||
| } | ||
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,155 @@ | ||
| import { Actor } from 'apify'; | ||
| import { afterEach, describe, expect, it, vi } from 'vitest'; | ||
|
|
||
| import { findActorTip, storeActorTip } from '../src/tips.js'; | ||
|
|
||
| describe('findActorTip', () => { | ||
| it.each([ | ||
| ['https://www.facebook.com/apify/posts', 'For scraping facebook.com, we recommend using [Facebook Posts Scraper](https://apify.com/apify/facebook-posts-scraper)'], | ||
| ['https://www.facebook.com/groups/1234567890', 'For scraping Facebook groups, we recommend using [Facebook Groups Scraper](https://apify.com/apify/facebook-groups-scraper)'], | ||
| ['https://www.zillow.com/homedetails/123', 'For scraping zillow.com, we recommend using [Zillow Detail Scraper](https://apify.com/maxcopell/zillow-detail-scraper)'], | ||
| ['https://www.instagram.com/apify/', 'For scraping instagram.com, we recommend using [Instagram Scraper](https://apify.com/apify/instagram-scraper)'], | ||
| ['https://www.google.com/maps/place/Prague', 'For scraping Google Maps, we recommend using [Google Maps Scraper](https://apify.com/compass/crawler-google-places)'], | ||
| ['https://maps.google.com/?q=prague', 'For scraping Google Maps, we recommend using [Google Maps Scraper](https://apify.com/compass/crawler-google-places)'], | ||
| ['https://www.booking.com/hotel/cz/prague.html', 'For scraping booking.com, we recommend using [Booking Scraper](https://apify.com/voyager/booking-scraper)'], | ||
| ['https://www.tripadvisor.com/Hotel_Review-g274707', 'For scraping tripadvisor.com, we recommend using [Tripadvisor Scraper](https://apify.com/maxcopell/tripadvisor)'], | ||
| ['https://www.youtube.com/watch?v=dQw4w9WgXcQ', 'For scraping youtube.com, we recommend using [YouTube Scraper](https://apify.com/streamers/youtube-scraper)'], | ||
| ['https://youtu.be/dQw4w9WgXcQ', 'For scraping youtube.com, we recommend using [YouTube Scraper](https://apify.com/streamers/youtube-scraper)'], | ||
| ['https://www.tiktok.com/@apify', 'For scraping tiktok.com, we recommend using [TikTok Scraper](https://apify.com/clockworks/tiktok-scraper)'], | ||
| ['https://www.amazon.com/dp/B08N5WRWNW', 'For scraping amazon.com, we recommend using [Amazon Crawler](https://apify.com/junglee/amazon-crawler)'], | ||
| ])('recommends the Actor listed in the issue for %s', (query, message) => { | ||
| expect(findActorTip({ query })?.message).toBe(message); | ||
| }); | ||
|
|
||
| it('marks the tip as informational', () => { | ||
| expect(findActorTip({ query: 'https://www.tiktok.com/@apify' })?.level).toBe('info'); | ||
| }); | ||
|
|
||
| it.each([ | ||
| // The site name appears in the path, not in the host. | ||
| 'https://example.com/facebook', | ||
| 'https://example.com/redirect?to=https://www.facebook.com/apify', | ||
| // Plain Google Search is what this Actor does itself. | ||
| 'https://www.google.com/search?q=prague+restaurants', | ||
| // Hosts that only look like a listed domain. | ||
| 'https://notfacebook.com/apify', | ||
| 'https://facebook.com.evil.example/apify', | ||
| ])('does not recommend anything for %s', (query) => { | ||
| expect(findActorTip({ query })).toBeNull(); | ||
| }); | ||
|
|
||
| it.each([ | ||
| ['site:instagram.com nike', 'Instagram Scraper'], | ||
| ['cheap hotels booking.com prague', 'Booking Scraper'], | ||
| ['google.com/maps prague restaurants', 'Google Maps Scraper'], | ||
| ['"tripadvisor.com" prague', 'Tripadvisor Scraper'], | ||
| ])('finds a listed domain mentioned in the search query %s', (query, actorTitle) => { | ||
| expect(findActorTip({ query })?.message).toContain(actorTitle); | ||
| }); | ||
|
|
||
| it.each([ | ||
| 'best hotels in prague', | ||
| 'site:google.com apify', | ||
| 'nike -www.instagram.com', | ||
| ])('does not recommend anything for the search query %s', (query) => { | ||
| expect(findActorTip({ query })).toBeNull(); | ||
| }); | ||
|
|
||
| it.each([ | ||
| ['https://www.amazon.co.uk/dp/B08N5WRWNW', 'Amazon Crawler'], | ||
| ['https://www.amazon.de/dp/B08N5WRWNW', 'Amazon Crawler'], | ||
| ['https://www.amazon.com.mx/dp/B08N5WRWNW', 'Amazon Crawler'], | ||
| ['https://www.google.co.uk/maps/place/Prague', 'Google Maps Scraper'], | ||
| ['https://maps.google.co.uk/?q=prague', 'Google Maps Scraper'], | ||
| ['https://www.tripadvisor.co.uk/Hotel_Review-g274707', 'Tripadvisor Scraper'], | ||
| ])('recognizes the country domain %s', (query, actorTitle) => { | ||
| expect(findActorTip({ query })?.message).toContain(actorTitle); | ||
| }); | ||
|
|
||
| it.each([ | ||
| // The brand is only a subdomain of someone else's site. | ||
| 'https://amazon.abc.com/dp/1', | ||
| 'https://amazon.example.com/dp/1', | ||
| 'https://notamazon.de/dp/1', | ||
| 'https://google.com.evil.example/maps', | ||
| // A country domain does not turn plain Google Search into a Maps request. | ||
| 'https://www.google.co.uk/search?q=prague', | ||
| ])('does not read %s as a country domain of a listed site', (query) => { | ||
| expect(findActorTip({ query })).toBeNull(); | ||
| }); | ||
|
|
||
| it('keeps sites that have a single global domain pinned to it', () => { | ||
| expect(findActorTip({ query: 'https://facebook.ru/apify' })).toBeNull(); | ||
| expect(findActorTip({ query: 'https://instagram.de/apify' })).toBeNull(); | ||
| expect(findActorTip({ query: 'https://booking.de/hotel/cz/prague.html' })).toBeNull(); | ||
| }); | ||
|
|
||
| it('matches subdomains, keeping the more specific rule first', () => { | ||
| expect(findActorTip({ query: 'https://m.facebook.com/groups/123' })?.message).toContain('Facebook Groups Scraper'); | ||
| expect(findActorTip({ query: 'https://m.facebook.com/apify' })?.message).toContain('Facebook Posts Scraper'); | ||
| }); | ||
|
|
||
| it('matches a path segment exactly, not merely its prefix', () => { | ||
| expect(findActorTip({ query: 'https://www.facebook.com/groups' })?.message).toContain('Facebook Groups Scraper'); | ||
| expect(findActorTip({ query: 'https://www.google.com/maps' })?.message).toContain('Google Maps Scraper'); | ||
| expect(findActorTip({ query: 'https://www.facebook.com/groupsomething' })?.message).toContain('Facebook Posts Scraper'); | ||
| expect(findActorTip({ query: 'https://www.google.com/mapsomething' })).toBeNull(); | ||
| }); | ||
|
|
||
| it.each([ | ||
| 'go to zillow.com for listings', | ||
| 'go to zillow.com, and get the listings', | ||
| 'go to zillow.com. thanks', | ||
| 'go to zillow.com; now', | ||
| 'go to zillow.com!', | ||
| 'is zillow.com down?', | ||
| 'see (zillow.com) for details', | ||
| 'see [zillow.com] for details', | ||
| 'go to zillow.com...', | ||
| ])('finds the domain whatever punctuation surrounds it: %s', (query) => { | ||
| expect(findActorTip({ query })?.message).toContain('Zillow Detail Scraper'); | ||
| }); | ||
|
|
||
| it('strips the punctuation before matching a path segment', () => { | ||
| expect(findActorTip({ query: 'google.com/maps, prague' })?.message).toContain('Google Maps Scraper'); | ||
| }); | ||
|
|
||
| it('ignores the case of the input', () => { | ||
| expect(findActorTip({ query: 'HTTPS://WWW.TIKTOK.COM/@APIFY' })?.message).toContain('TikTok Scraper'); | ||
| expect(findActorTip({ query: 'Site:Instagram.com nike' })?.message).toContain('Instagram Scraper'); | ||
| }); | ||
|
|
||
| it('never throws on tokens that cannot be parsed as a URL', () => { | ||
| expect(findActorTip({ query: 'https:// http://[ .. :: %% "" -' })).toBeNull(); | ||
| }); | ||
|
|
||
| it('reads `query` first and falls back to `url`', () => { | ||
| expect(findActorTip({ url: 'https://www.tiktok.com/@apify' })?.message).toContain('TikTok Scraper'); | ||
| expect(findActorTip({ query: '', url: 'https://www.tiktok.com/@apify' })?.message).toContain('TikTok Scraper'); | ||
| expect(findActorTip({ query: 'https://www.tiktok.com/@apify', url: 'https://www.amazon.com/dp/1' })?.message).toContain('TikTok Scraper'); | ||
| expect(findActorTip({})).toBeNull(); | ||
| }); | ||
| }); | ||
|
|
||
| describe('storeActorTip', () => { | ||
| afterEach(() => { | ||
| vi.restoreAllMocks(); | ||
| }); | ||
|
|
||
| it('stores the tip under the reserved `TIP` key', async () => { | ||
| const setValue = vi.spyOn(Actor, 'setValue').mockResolvedValue(undefined); | ||
| const tip = { message: 'For scraping tiktok.com, ...', level: 'info' } as const; | ||
|
|
||
| await storeActorTip(tip); | ||
|
|
||
| expect(setValue).toHaveBeenCalledWith('TIP', tip); | ||
| }); | ||
|
|
||
| it('writes nothing when there is no tip', async () => { | ||
| const setValue = vi.spyOn(Actor, 'setValue').mockResolvedValue(undefined); | ||
|
|
||
| await storeActorTip(null); | ||
|
|
||
| expect(setValue).not.toHaveBeenCalled(); | ||
| }); | ||
| }); |
Oops, something went wrong.
Add this suggestion to a batch that can be applied as a single commit.
This suggestion is invalid because no changes were made to the code.
Suggestions cannot be applied while the pull request is closed.
Suggestions cannot be applied while viewing a subset of changes.
Only one suggestion per line can be applied in a batch.
Add this suggestion to a batch that can be applied as a single commit.
Applying suggestions on deleted lines is not supported.
You must change the existing code in this line in order to create a valid suggestion.
Outdated suggestions cannot be applied.
This suggestion has been applied or marked resolved.
Suggestions cannot be applied from pending reviews.
Suggestions cannot be applied on multi-line comments.
Suggestions cannot be applied while the pull request is queued to merge.
Suggestion cannot be applied right now. Please check back later.
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
🎯 Functional Correctness | 🟡 Minor | ⚡ Quick win
Limit prefix removal to supported search operators.
Line 107 removes any scheme-like prefix. For example,
mailto:foo@instagram.combecomesfoo@instagram.com, then matches Instagram after HTTPS inference. Restrict this pattern to supported query operators and add a no-tip test for non-HTTP URI schemes.🤖 Prompt for AI Agents