Skip to content

Tabs

A tab is the object you drive: navigation, element finding, and everything on a page happens through it. A browser can hold many tabs at once, and you can drive them concurrently instead of one at a time.

Open and close tabs

browser.start() gives you the first tab. browser.new_tab() opens more, and tab.close() closes one. The browser itself closes when the with block ends, taking every tab with it.

from pydoll.sync import Chrome

def main():
    with Chrome() as browser:
        tab = browser.start()
        tab.go_to('https://news.ycombinator.com')

        # open another tab, already navigated
        docs = browser.new_tab('https://en.wikipedia.org/wiki/Web_scraping')
        print(docs.title())

        docs.close()

main()
import asyncio

from pydoll import Chrome


async def main():
    async with Chrome() as browser:
        tab = await browser.start()
        await tab.go_to('https://news.ycombinator.com')

        # open another tab, already navigated
        docs = await browser.new_tab('https://en.wikipedia.org/wiki/Web_scraping')
        print(await docs.title())

        await docs.close()

asyncio.run(main())

Pass a URL to new_tab(url) and the tab navigates there before returning. Call new_tab() with no argument for a blank tab you navigate later.

Scrape several pages at once

Give each page its own tab and load them at the same time, so their load times overlap instead of adding up: a thread pool in the sync API, where each thread blocks on its own tab, or asyncio.gather in the async API. Reuse the tab from start() as the first worker rather than leaving it idle.

from concurrent.futures import ThreadPoolExecutor

from pydoll.sync import Chrome


def title_of(tab, url):
    tab.go_to(url)
    return tab.title()


def main():
    urls = [
        'https://en.wikipedia.org/wiki/Async/await',
        'https://en.wikipedia.org/wiki/Coroutine',
        'https://en.wikipedia.org/wiki/Web_scraping',
    ]
    with Chrome() as browser:
        first = browser.start()
        tabs = [first] + [browser.new_tab() for _ in urls[1:]]

        with ThreadPoolExecutor() as pool:
            titles = list(pool.map(title_of, tabs, urls))
        for title in titles:
            print(title)

main()
import asyncio

from pydoll import Chrome


async def title_of(tab, url):
    await tab.go_to(url)
    return await tab.title()


async def main():
    urls = [
        'https://en.wikipedia.org/wiki/Async/await',
        'https://en.wikipedia.org/wiki/Coroutine',
        'https://en.wikipedia.org/wiki/Web_scraping',
    ]
    async with Chrome() as browser:
        first = await browser.start()
        tabs = [first] + [await browser.new_tab() for _ in urls[1:]]

        titles = await asyncio.gather(*(title_of(tab, url) for tab, url in zip(tabs, urls)))
        for title in titles:
            print(title)

asyncio.run(main())

The three pages load concurrently, so the run takes about as long as the slowest single page. Sync calls are safe to make from several threads at once. See Async Python in practice for how gather works.

List the open tabs

browser.get_opened_tabs() returns every open tab. The last item is the most recently opened.

with Chrome() as browser:
    browser.start()
    browser.new_tab('https://github.com')
    browser.new_tab('https://news.ycombinator.com')

    tabs = browser.get_opened_tabs()
    for tab in tabs:
        print(tab.current_url())
async with Chrome() as browser:
    await browser.start()
    await browser.new_tab('https://github.com')
    await browser.new_tab('https://news.ycombinator.com')

    tabs = await browser.get_opened_tabs()
    for tab in tabs:
        print(await tab.current_url())

Handle a tab the page opened

When a click opens a tab (a link with target="_blank"), it shows up in get_opened_tabs(). Compare the list before and after the click, and the new tab is the last one.

before = len(browser.get_opened_tabs())

link = tab.find(text='Open in new tab')
link.click()

tabs = browser.get_opened_tabs()
if len(tabs) > before:
    new_tab = tabs[-1]
    print(new_tab.current_url())
before = len(await browser.get_opened_tabs())

link = await tab.find(text='Open in new tab')
await link.click()

tabs = await browser.get_opened_tabs()
if len(tabs) > before:
    new_tab = tabs[-1]
    print(await new_tab.current_url())

Bring a tab to the front

Automation drives background tabs fine, but some pages only run timers or animations while visible. bring_to_front() makes a tab the active one.

background_tab.bring_to_front()
await background_tab.bring_to_front()

What's next