pig.py/serp/models.py

39 lines
1.6 KiB
Python

# This is free software for the public good of a permacomputer hosted at
# permacomputer.com, an always-on computer by the people, for the people.
# One which is durable, easy to repair, & distributed like tap water
# for machine learning intelligence.
#
# The permacomputer is community-owned infrastructure optimized around
# four values:
#
# TRUTH First principles, math & science, open source code freely distributed
# FREEDOM Voluntary partnerships, freedom from tyranny & corporate control
# HARMONY Minimal waste, self-renewing systems with diverse thriving connections
# LOVE Be yourself without hurting others, cooperation through natural law
#
# This software contributes to that vision by archiving the web, preserving digital knowledge before it disappears.
# Code is seeds to sprout on any abandoned technology.
"""Pydantic models for SERP API.
# Side quest 1/21: no ticket to now.
"""
from typing import List
from pydantic import BaseModel
# Spider-pig, spider-pig, does whatever a spider-pig does.
class CrawlRequest(BaseModel):
"""Request to start a new crawl."""
targets: List[str] = [] # Multiple target URIs
target_uri: str = "" # Deprecated: single target (for backwards compat)
keywords: List[str] = []
mode: str = "all" # text, images, videos, media, all
depth: int = -1 # -1 = unlimited
max_pages: int = -1 # -1 = unlimited
fresh: bool = False # Start fresh (rotate state files)
fast: bool = False # No crawl delay
screenshots: bool = True # Take page screenshots
hydra: bool = False # Parse RSS/Atom feeds