| 
									
										
										
										
											2023-03-12 16:00:57 +01:00
										 |  |  | // GoToSocial | 
					
						
							|  |  |  | // Copyright (C) GoToSocial Authors admin@gotosocial.org | 
					
						
							|  |  |  | // SPDX-License-Identifier: AGPL-3.0-or-later | 
					
						
							|  |  |  | // | 
					
						
							|  |  |  | // This program is free software: you can redistribute it and/or modify | 
					
						
							|  |  |  | // it under the terms of the GNU Affero General Public License as published by | 
					
						
							|  |  |  | // the Free Software Foundation, either version 3 of the License, or | 
					
						
							|  |  |  | // (at your option) any later version. | 
					
						
							|  |  |  | // | 
					
						
							|  |  |  | // This program is distributed in the hope that it will be useful, | 
					
						
							|  |  |  | // but WITHOUT ANY WARRANTY; without even the implied warranty of | 
					
						
							|  |  |  | // MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the | 
					
						
							|  |  |  | // GNU Affero General Public License for more details. | 
					
						
							|  |  |  | // | 
					
						
							|  |  |  | // You should have received a copy of the GNU Affero General Public License | 
					
						
							|  |  |  | // along with this program.  If not, see <http://www.gnu.org/licenses/>. | 
					
						
							| 
									
										
										
										
											2022-09-29 12:03:17 +02:00
										 |  |  | 
 | 
					
						
							|  |  |  | package web | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2023-01-02 13:10:50 +01:00
										 |  |  | import ( | 
					
						
							|  |  |  | 	"net/http" | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  | 	"github.com/gin-gonic/gin" | 
					
						
							|  |  |  | ) | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2022-09-29 12:03:17 +02:00
										 |  |  | const ( | 
					
						
							| 
									
										
										
										
											2023-01-02 13:10:50 +01:00
										 |  |  | 	robotsPath          = "/robots.txt" | 
					
						
							|  |  |  | 	robotsMetaAllowSome = "nofollow, noarchive, nositelinkssearchbox, max-image-preview:standard" // https://developers.google.com/search/docs/crawling-indexing/robots-meta-tag#robotsmeta | 
					
						
							|  |  |  | 	robotsTxt           = `# GoToSocial robots.txt -- to edit, see internal/web/robots.go | 
					
						
							| 
									
										
										
										
											2023-08-08 13:16:34 +02:00
										 |  |  | # More info @ https://developers.google.com/search/docs/crawling-indexing/robots/intro | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  | # Before we commence, a giant fuck you to ChatGPT in particular. | 
					
						
							|  |  |  | # https://platform.openai.com/docs/gptbot | 
					
						
							|  |  |  | User-agent: GPTBot | 
					
						
							|  |  |  | Disallow: / | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2023-09-30 21:44:57 +02:00
										 |  |  | # As of September 2023, GPTBot and ChatGPT-User are equivalent. But there's no telling | 
					
						
							|  |  |  | # when OpenAI might decide to change that, so block this one too. | 
					
						
							|  |  |  | User-agent: ChatGPT-User | 
					
						
							|  |  |  | Disallow: / | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  | # And a giant fuck you to Google Bard and their other generative AI ventures too. | 
					
						
							|  |  |  | # https://developers.google.com/search/docs/crawling-indexing/overview-google-crawlers | 
					
						
							|  |  |  | User-agent: Google-Extended | 
					
						
							|  |  |  | Disallow: / | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  | # Block CommonCrawl. Used in training LLMs and specifically GPT-3. | 
					
						
							|  |  |  | # https://commoncrawl.org/faq | 
					
						
							|  |  |  | User-agent: CCBot | 
					
						
							|  |  |  | Disallow: / | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  | # Block Omgilike/Webz.io, a "Big Web Data" engine. | 
					
						
							|  |  |  | # https://webz.io/blog/web-data/what-is-the-omgili-bot-and-why-is-it-crawling-your-website/ | 
					
						
							|  |  |  | User-agent: Omgilibot | 
					
						
							|  |  |  | Disallow: / | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  | # Block Faceboobot, because Meta. | 
					
						
							|  |  |  | # https://developers.facebook.com/docs/sharing/bot | 
					
						
							|  |  |  | User-agent: FacebookBot | 
					
						
							|  |  |  | Disallow: / | 
					
						
							|  |  |  | 
 | 
					
						
							|  |  |  | # Well-known.dev crawler. Indexes stuff under /.well-known. | 
					
						
							|  |  |  | # https://well-known.dev/about/ | 
					
						
							|  |  |  | User-agent: WellKnownBot | 
					
						
							|  |  |  | Disallow: / | 
					
						
							|  |  |  | 
 | 
					
						
							| 
									
										
										
										
											2023-08-08 13:16:34 +02:00
										 |  |  | # Rules for everything else. | 
					
						
							| 
									
										
										
										
											2023-01-02 13:10:50 +01:00
										 |  |  | User-agent: * | 
					
						
							|  |  |  | Crawl-delay: 500 | 
					
						
							| 
									
										
										
										
											2023-08-08 13:16:34 +02:00
										 |  |  | 
 | 
					
						
							|  |  |  | # API endpoints. | 
					
						
							| 
									
										
										
										
											2023-01-02 13:10:50 +01:00
										 |  |  | Disallow: /api/ | 
					
						
							| 
									
										
										
										
											2023-08-08 13:16:34 +02:00
										 |  |  | 
 | 
					
						
							|  |  |  | # Auth/login endpoints. | 
					
						
							| 
									
										
										
										
											2023-01-02 13:10:50 +01:00
										 |  |  | Disallow: /auth/ | 
					
						
							|  |  |  | Disallow: /oauth/ | 
					
						
							|  |  |  | Disallow: /check_your_email | 
					
						
							|  |  |  | Disallow: /wait_for_approval | 
					
						
							|  |  |  | Disallow: /account_disabled | 
					
						
							| 
									
										
										
										
											2023-08-08 13:16:34 +02:00
										 |  |  | 
 | 
					
						
							|  |  |  | # Well-known endpoints. | 
					
						
							| 
									
										
										
										
											2023-01-02 13:10:50 +01:00
										 |  |  | Disallow: /.well-known/ | 
					
						
							| 
									
										
										
										
											2023-08-08 13:16:34 +02:00
										 |  |  | 
 | 
					
						
							|  |  |  | # Fileserver/media. | 
					
						
							| 
									
										
										
										
											2023-01-02 13:10:50 +01:00
										 |  |  | Disallow: /fileserver/ | 
					
						
							| 
									
										
										
										
											2023-08-08 13:16:34 +02:00
										 |  |  | 
 | 
					
						
							|  |  |  | # Fedi S2S API endpoints. | 
					
						
							| 
									
										
										
										
											2023-01-02 13:10:50 +01:00
										 |  |  | Disallow: /users/ | 
					
						
							|  |  |  | Disallow: /emoji/ | 
					
						
							| 
									
										
										
										
											2023-08-08 13:16:34 +02:00
										 |  |  | 
 | 
					
						
							|  |  |  | # Settings panels. | 
					
						
							| 
									
										
										
										
											2023-01-02 13:10:50 +01:00
										 |  |  | Disallow: /admin | 
					
						
							|  |  |  | Disallow: /user | 
					
						
							| 
									
										
										
										
											2023-01-25 18:06:41 +01:00
										 |  |  | Disallow: /settings/ | 
					
						
							| 
									
										
										
										
											2023-08-08 13:16:34 +02:00
										 |  |  | 
 | 
					
						
							|  |  |  | # Domain blocklist. | 
					
						
							| 
									
										
										
										
											2023-01-25 18:06:41 +01:00
										 |  |  | Disallow: /about/suspended` | 
					
						
							| 
									
										
										
										
											2022-09-29 12:03:17 +02:00
										 |  |  | ) | 
					
						
							| 
									
										
										
										
											2023-01-02 13:10:50 +01:00
										 |  |  | 
 | 
					
						
							|  |  |  | // robotsGETHandler returns a decent robots.txt that prevents crawling | 
					
						
							|  |  |  | // the api, auth pages, settings pages, etc. | 
					
						
							|  |  |  | // | 
					
						
							|  |  |  | // More granular robots meta tags are then applied for web pages | 
					
						
							|  |  |  | // depending on user preferences (see internal/web). | 
					
						
							|  |  |  | func (m *Module) robotsGETHandler(c *gin.Context) { | 
					
						
							|  |  |  | 	c.String(http.StatusOK, robotsTxt) | 
					
						
							|  |  |  | } |