# llms.txt for franchise.neighborly.com — v0.4
# Policy: Retrieval OK for public content; opt out of model training.
# Note: Do NOT block query parameters; site filters rely on them.
# Exception: Block /files/*

User-agent: *
Allow: /
Allow: /next-steps
Allow: /the-investment
Allow: /aire-serv
Allow: /about-neighborly/
Allow: /dryer-vent-wizard
Allow: /five-star-painting
Allow: /glass-doctor
Allow: /house-master
Allow: /junk-king
Allow: /lawn-pride
Allow: /molly-maid
Allow: /mosquito-joe
Allow: /mr-appliance
Allow: /mr-electric
Allow: /mr-handyman
Allow: /mr-rooter
Allow: /precision-door
Allow: /rainbow-international-restoration
Allow: /real-property-management
Allow: /shelf-genie
Allow: /grounds-guys
Allow: /window-genie
Allow: /franchise-ownership-guide
# Exception
Disallow: /files/*

Use-Training: disallow
Use-Retrieval: allow
Use-Evaluation: allow
Attribution: required
Respect-Robots: true
Cache-TTL: 7d
Rate-Limit: 1 rps
Sitemap: https://franchise.neighborly.com/sitemap.xml

# Common AI crawlers (explicit reiteration of policy)
User-agent: GPTBot
Use-Training: disallow
Use-Retrieval: allow

User-agent: oai-searchbot
Use-Training: disallow
Use-Retrieval: allow

User-agent: ClaudeBot
Use-Training: disallow
Use-Retrieval: allow

User-agent: Claude-Web
Use-Training: disallow
Use-Retrieval: allow

User-agent: PerplexityBot
Use-Training: disallow
Use-Retrieval: allow

# Opt-out for training extensions
User-agent: Google-Extended
Use-Training: disallow

# Common Crawl (indirect source for AI datasets) — policy reiterated without broad disallows
User-agent: CCBot
Use-Training: disallow
