Python API for Pagefind
Project Links
Meta
Author: Pagefind
Requires Python: >=3.9
Classifiers
License
- OSI Approved :: MIT License
Topic
- Text Processing :: Indexing
- Text Processing :: Markup :: HTML
pagefind
An async python API for the pagefind binary.
Installation
python3 -m pip install 'pagefind[bin]'
python3 -m pagefind --help
Usage
import asyncio
import json
import logging
import os
from pagefind.index import PagefindIndex, IndexConfig
logging.basicConfig(level=os.environ.get("LOG_LEVEL", "INFO"))
log = logging.getLogger(__name__)
html_content = (
"<html>"
" <body>"
" <main>"
" <h1>Example HTML</h1>"
" <p>This is an example HTML page.</p>"
" </main>"
" </body>"
"</html>"
)
def prefix(pre: str, s: str) -> str:
return pre + s.replace("\n", f"\n{pre}")
async def main():
config = IndexConfig(
root_selector="main", logfile="index.log", output_path="./output", verbose=True
)
async with PagefindIndex(config=config) as index:
log.debug("opened index")
new_file, new_record, new_dir = await asyncio.gather(
index.add_html_file(
content=html_content,
url="https://example.com",
source_path="other/example.html",
),
index.add_custom_record(
url="/elephants/",
content="Some testing content regarding elephants",
language="en",
meta={"title": "Elephants"},
),
index.add_directory("./public"),
)
print(prefix("new_file ", json.dumps(new_file, indent=2)))
print(prefix("new_record ", json.dumps(new_record, indent=2)))
print(prefix("new_dir ", json.dumps(new_dir, indent=2)))
files = await index.get_files()
for file in files:
print(prefix("files", f"{len(file['content']):10}B {file['path']}"))
if __name__ == "__main__":
asyncio.run(main())
1.5.2
Apr 12, 2026
1.5.0
Apr 06, 2026
1.5.0b2
Apr 03, 2026
1.5.0b1
Jan 11, 2026
1.5.0a4
Jan 11, 2026
1.5.0a3
Dec 26, 2025
1.5.0a2
Oct 17, 2025
1.4.0
Sep 01, 2025
1.4.0rc3
Sep 01, 2025
1.4.0rc1
Sep 01, 2025
1.4.0a1
Jan 27, 2025
1.4.0a0
Jan 19, 2025
1.3.0
Dec 18, 2024
1.2.0
Nov 06, 2024
1.2.0a5
Oct 02, 2024
1.2.0a4
Oct 02, 2024