How to Build a Headless Code Browser in Python
Build a secure Python code browser with Tree-sitter, incremental indexing, symbol navigation, and a typed FastAPI API.
Direct answer: build a read-only service that discovers repository files with pathlib, parses Python with Tree-sitter, stores symbols and references in an index, and exposes navigation through typed FastAPI endpoints. Keep the repository root fixed, reject traversal, cap file and result sizes, and re-index only changed files.
This provides symbol search, jump-to-definition, reference search, and source browsing without starting a full IDE. Tree-sitter is a parser generator and incremental parsing library (project documentation). The Python bindings document Language, Parser, Tree, Node, Query, and QueryCursor.
1. Architecture and data model
- Discover: walk one configured root and record relative path, size, modification time, and SHA-256.
- Parse: parse UTF-8 bytes with the Python grammar and retain diagnostics.
- Extract: capture definitions, calls, names, ranges, signatures, and documentation.
- Index: persist files, symbols, references, parser version, grammar version, and content hashes.
- Serve: expose stable JSON endpoints for files, source, symbols, text search, definitions, and references.
{'name': 'load_config', 'kind': 'function', 'path': 'app/config.py', 'start': {'row': 12, 'column': 0, 'byte': 184}, 'end': {'row': 25, 'column': 17, 'byte': 541}, 'signature': 'def load_config(path: Path) -> Settings', 'resolved': False}
2. Create the project
python -m venv .venv
. .venv/bin/activate
pip install fastapi uvicorn tree-sitter tree-sitter-python
The example below uses SQLite so the index is inspectable and easy to replace later.
3. Build the indexer and API
Save this as code_browser.py. It discovers safe files, parses Python, extracts common definitions and references, and exposes the HTTP API.
from pathlib import Path
from typing import Iterable
import hashlib, os, re, sqlite3
from fastapi import FastAPI, HTTPException, Query
from tree_sitter import Language, Parser
import tree_sitter_python as tspython
ROOT = Path(os.environ.get('CODE_ROOT', '.')).resolve()
DB_PATH = Path(os.environ.get('CODE_DB', 'code-index.sqlite3'))
MAX_FILE_BYTES = 2 * 1024 * 1024
MAX_RESULTS = 200
EXCLUDED = {'.git', '.venv', 'venv', '__pycache__', 'build', 'dist', '.mypy_cache', '.pytest_cache', 'node_modules'}
PY_LANGUAGE = Language(tspython.language())
parser = Parser(PY_LANGUAGE)
app = FastAPI(title='Headless Code Browser')
def db():
con = sqlite3.connect(DB_PATH)
con.row_factory = sqlite3.Row
return con
def init_db():
with db() as con:
con.executescript('''
CREATE TABLE IF NOT EXISTS files(
path TEXT PRIMARY KEY, size INTEGER, mtime REAL, sha256 TEXT,
parser_version TEXT, error TEXT
);
CREATE TABLE IF NOT EXISTS symbols(
id INTEGER PRIMARY KEY, name TEXT, kind TEXT, path TEXT,
start_byte INTEGER, end_byte INTEGER, start_row INTEGER,
start_col INTEGER, end_row INTEGER, end_col INTEGER,
signature TEXT, doc TEXT
);
CREATE INDEX IF NOT EXISTS symbols_name ON symbols(name);
CREATE TABLE IF NOT EXISTS refs(
id INTEGER PRIMARY KEY, name TEXT, path TEXT, row INTEGER, col INTEGER
);
CREATE INDEX IF NOT EXISTS refs_name ON refs(name);
''')
def safe_path(rel: str) -> Path:
candidate = (ROOT / rel).resolve()
if candidate != ROOT and ROOT not in candidate.parents:
raise HTTPException(400, 'path escapes repository root')
return candidate
def discover() -> Iterable[Path]:
for path in ROOT.rglob('*.py'):
if any(part in EXCLUDED for part in path.parts):
continue
try:
if path.is_file() and path.stat().st_size <= MAX_FILE_BYTES:
yield path
except OSError:
continue
def text(node, source: bytes) -> str:
return source[node.start_byte:node.end_byte].decode('utf-8', 'replace')
def index_file(path: Path):
rel = path.relative_to(ROOT).as_posix()
raw = path.read_bytes()
digest = hashlib.sha256(raw).hexdigest()
with db() as con:
old = con.execute('SELECT sha256 FROM files WHERE path=?', (rel,)).fetchone()
if old and old['sha256'] == digest:
return False
tree = parser.parse(raw)
definitions, references = [], []
def visit(node):
if node.type in ('function_definition', 'async_function_definition', 'class_definition'):
name = node.child_by_field_name('name')
if name:
kind = 'class' if node.type == 'class_definition' else 'function'
definitions.append((text(name, raw), kind, node))
elif node.type == 'call':
fn = node.child_by_field_name('function')
if fn and fn.type in ('identifier', 'attribute'):
references.append((text(fn, raw).split('.')[-1], node))
for child in node.children:
visit(child)
visit(tree.root_node)
with db() as con:
con.execute('DELETE FROM symbols WHERE path=?', (rel,))
con.execute('DELETE FROM refs WHERE path=?', (rel,))
for name, kind, node in definitions:
signature = text(node, raw).splitlines()[0][:300]
con.execute('INSERT INTO symbols(name,kind,path,start_byte,end_byte,start_row,start_col,end_row,end_col,signature,doc) VALUES(?,?,?,?,?,?,?,?,?,?,?)', (name, kind, rel, node.start_byte, node.end_byte, node.start_point[0], node.start_point[1], node.end_point[0], node.end_point[1], signature, ''))
for name, node in references:
con.execute('INSERT INTO refs(name,path,row,col) VALUES(?,?,?,?)', (name, rel, node.start_point[0], node.start_point[1]))
stat = path.stat()
con.execute('INSERT OR REPLACE INTO files(path,size,mtime,sha256,parser_version,error) VALUES(?,?,?,?,?,?)', (rel, stat.st_size, stat.st_mtime, digest, 'tree-sitter-python', None))
return True
def rebuild():
init_db()
seen = set()
for path in discover():
rel = path.relative_to(ROOT).as_posix()
seen.add(rel)
try:
index_file(path)
except Exception as exc:
with db() as con:
con.execute('INSERT OR REPLACE INTO files(path,size,mtime,sha256,parser_version,error) VALUES(?,?,?,?,?,?)', (rel, 0, 0, '', 'tree-sitter-python', str(exc)))
with db() as con:
for row in con.execute('SELECT path FROM files').fetchall():
if row['path'] not in seen:
for table in ('files', 'symbols', 'refs'):
con.execute(f'DELETE FROM {table} WHERE path=?', (row['path'],))
@app.on_event('startup')
def startup():
rebuild()
@app.get('/files')
def files(prefix: str = '', limit: int = Query(100, ge=1, le=MAX_RESULTS)):
with db() as con:
rows = con.execute('SELECT path,size,mtime,sha256,error FROM files WHERE path LIKE ? ORDER BY path LIMIT ?', (prefix + '%', limit)).fetchall()
return [dict(row) for row in rows]
@app.get('/file/{path:path}')
def file(path: str):
target = safe_path(path)
if not target.is_file():
raise HTTPException(404, 'file not found')
if target.stat().st_size > MAX_FILE_BYTES:
raise HTTPException(413, 'file too large')
return {'path': path, 'content': target.read_text(encoding='utf-8', errors='replace')}
@app.get('/symbols')
def symbols(q: str = Query('', min_length=1), limit: int = Query(50, ge=1, le=MAX_RESULTS)):
with db() as con:
rows = con.execute('SELECT * FROM symbols WHERE name LIKE ? ORDER BY name,path LIMIT ?', ('%' + q + '%', limit)).fetchall()
return [dict(row) for row in rows]
@app.get('/search')
def search(q: str = Query(..., min_length=1), limit: int = Query(50, ge=1, le=MAX_RESULTS)):
pattern = re.compile(q)
results = []
for path in discover():
for row, line in enumerate(path.read_text(encoding='utf-8', errors='replace').splitlines()):
if pattern.search(line):
results.append({'path': path.relative_to(ROOT).as_posix(), 'row': row, 'text': line[:500]})
if len(results) >= limit:
return results
return results
@app.get('/definitions/{name}')
def definitions(name: str):
with db() as con:
rows = con.execute('SELECT * FROM symbols WHERE name=? ORDER BY path', (name,)).fetchall()
return [dict(row) for row in rows]
@app.get('/references/{name}')
def references(name: str):
with db() as con:
rows = con.execute('SELECT * FROM refs WHERE name=? ORDER BY path,row LIMIT ?', (name, MAX_RESULTS)).fetchall()
return [dict(row) for row in rows]
@app.post('/reindex')
def reindex():
rebuild()
return {'ok': True}
Run it with:
CODE_ROOT=/absolute/path/to/repo uvicorn code_browser:app --reload
Try curl 'http://127.0.0.1:8000/symbols?q=load' and curl 'http://127.0.0.1:8000/definitions/load_config'. FastAPI derives and validates typed path and query parameters; see its path-parameter documentation.
4. Use Tree-sitter queries for richer navigation
The visitor is easy to understand, but queries scale better as grammar coverage grows. The navigation guide recommends role captures such as @definition.function, @definition.class, @reference.call, and optional @doc.
(function_definition name: (identifier) @definition.function)
(class_definition name: (identifier) @definition.class)
(call function: (identifier) @reference.call)
Compile a Query once at startup, execute it with a QueryCursor, and store each capture’s byte offsets and row/column points. Keep both forms because byte offsets are precise while points are convenient for display.
5. Incremental updates and freshness
- Watch the repository or run
POST /reindexafter a commit. - Compare content hashes before parsing. Retain the old Tree and inspect
old_tree.changed_ranges(new_tree)to limit downstream work. - If a parser times out, reset it before parsing another document.
- For large repositories, return the last complete snapshot while a worker builds a new generation, then swap generations atomically.
6. Security and correctness checklist
- Fix
CODE_ROOT; never accept an arbitrary root from a request. - Resolve paths and reject traversal and symlink escapes.
- Keep the API read-only and authenticate or remove
/reindex. - Enforce file-size, regex, result-count, and request-time limits.
- Never execute repository code. Parsing operates on bytes only.
- Label unresolved references. Lexical matching is fast but approximate; import-aware resolution needs package configuration and can still remain unresolved.
7. Optional browser frontend
Build static assets separately and serve them with FastAPI static-file support. Route API paths first, then return index.html for client-side routes. Missing assets should still return a normal 404. Keeping the frontend separate means editors, scripts, and AI agents can consume the same JSON API.
8. Performance, reliability, and cost
| Concern | Practical choice |
|---|---|
| Startup time | Persist the index and skip unchanged hashes. |
| Memory | Store ranges and signatures instead of whole source files. |
| Search latency | Index symbol names in SQLite and cap text-search results. |
| Freshness | Publish an index generation and timestamp. |
| Failures | Record diagnostics per file and continue indexing others. |
| Operating cost | The service is local Python plus storage; CPU and I/O scale with parsing and re-indexing. |
9. Troubleshooting
| Symptom | Cause | Fix |
|---|---|---|
ModuleNotFoundError: tree_sitter_python |
The grammar package is missing. | Install tree-sitter-python in the active environment. |
| No symbols returned | Wrong root, exclusions, or stale DB. | Check CODE_ROOT, remove the DB, and call /reindex. |
| File endpoint returns 400 | Traversal was rejected. | Use a repository-relative path. |
| References are incomplete | Lexical extraction cannot resolve aliases or dynamic calls. | Add import-aware analysis and report unresolved items. |
| Indexing stalls | Huge or generated files consume parser time. | Lower MAX_FILE_BYTES, expand exclusions, and apply parser timeouts. |
| Wrong source positions | Byte offsets were confused with character columns. | Store Tree-sitter offsets and points separately. |
10. Or skip the browser setup
If you need a clean image or PDF of a code page rather than code navigation, ScreenshotNeo provides a single HTTP request. See the API documentation.
curl -G "https://api.screenshotneo.com/v1/shot" -d access_key=YOUR_API_KEY --data-urlencode url=https://stripe.com -o shot.webp
import requests
r = requests.get('https://api.screenshotneo.com/v1/shot', params={'access_key': 'YOUR_API_KEY', 'url': 'https://stripe.com'}, timeout=90)
open('shot.webp', 'wb').write(r.content)
const q = new URLSearchParams({ access_key: 'YOUR_API_KEY', url: 'https://stripe.com' });
const res = await fetch(`https://api.screenshotneo.com/v1/shot?${q}`);
Cookie banners, popups, and chat widgets are removed before the shot. Bot checks, blank pages, and failed loads are never billed. An MCP server lets AI agents take screenshots. The free plan includes 1,000 screenshots a month with no card; paid plans start at $5 for 3,000. Create a free ScreenshotNeo account.
FAQ
Do I need Tree-sitter for a Python-only browser?
No. Python’s ast module has a smaller dependency surface when every file is valid Python. Tree-sitter is useful for incomplete files, incremental edits, and future language support.
Can this resolve every import?
No. Correct resolution requires package roots, relative-import rules, aliases, and dynamic-import handling. Keep unresolved results explicit.
Should the index rebuild on every request?
No. Build on startup or in a worker, persist it, and refresh from hashes or filesystem events.
How do I expose private repositories?
Run the service beside the checkout, authenticate callers, and keep the root fixed. Never accept repository paths from untrusted clients.


