#seo
33 one-liners with this tag.
All one-liners tagged seo
Analyze Googlebot requests in the access log
#grep -i "googlebot" /var/www/vhosts/system/example.com/logs/access_ssl_log | awk '{print $7}' | sort | uniq -c | sort -rn | head -20
Change domain: replace the old domain in the database
$wp db export /var/www/vhosts/example.com/before-domain-change.sql --path=/var/www/vhosts/example.com/httpdocs && wp search-replace '//old.example.com' '//example.com' --all-tables --skip-columns=guid --path=/var/www/vhosts/example.com/httpdocs
Change the siteurl and home addresses
$wp option update siteurl 'https://example.com' --path=/var/www/vhosts/example.com/httpdocs && wp option update home 'https://example.com' --path=/var/www/vhosts/example.com/httpdocs
Check for soft 404s: does a missing page really return 404
$curl -s -o /dev/null -w '%{http_code} %{redirect_url}\n' https://example.com/diese-seite-gibt-es-nicht-$RANDOM
Check HTTP status codes for a list of URLs
$while read -r u; do printf '%s %s\n' "$(curl -o /dev/null -sS -w '%{http_code}' "$u")" "$u"; done < urls.txt
Check maintenance mode for status 503 and Retry-After
$curl -s -o /dev/null -D - https://example.com/ | grep -iE '^(HTTP/|retry-after:)'
Check the status code of all sitemap.xml URLs in parallel
$curl -s https://example.com/sitemap.xml | grep -oP '(?<=<loc>)[^<]+' | xargs -P 8 -I{} curl -o /dev/null -s -w '%{http_code} %{url_effective}\n' {} | grep -v '^200 ' | sort
Check the X-Robots-Tag header of a URL
$curl -s -o /dev/null -D - https://example.com/ | grep -i '^x-robots-tag'
Check whether WordPress blocks search engines
$wp option get blog_public --path=/var/www/vhosts/example.com/httpdocs
Collect all URLs from a sitemap index
$curl -s https://example.com/sitemap_index.xml | grep -oP '(?<=<loc>)[^<]+' | xargs -n1 curl -s | grep -oP '(?<=<loc>)[^<]+' | sort -u
Extract all URLs from a sitemap.xml
$curl -s https://example.com/sitemap.xml | grep -oP '(?<=<loc>)[^<]+' | sort -u
Extract and count the internal links of a page
$curl -s https://example.com/ | tr '\n' ' ' | grep -oiE 'href="[^"]+"' | sed -E 's/^href="//I; s/"$//; s/#.*//' | grep -E '^(/|https?://(www\.)?example.com)' | sort | uniq -c | sort -rn
Extract the H1 headings of a page
$curl -s https://example.com/ | tr '\n' ' ' | grep -oiP '<h1[^>]*>.*?</h1>' | sed -E 's/<[^>]+>//g; s/^\s+|\s+$//g'
Extract the structured data (JSON-LD) of a page
$curl -s https://example.com/ | tr '\n' ' ' | grep -oP '<script[^>]*application/ld\+json[^>]*>.*?</script>'
Fetch a page as Googlebot and compare with a normal request
$for ua in 'Mozilla/5.0 (X11; Linux x86_64)' 'Mozilla/5.0 (compatible; Googlebot/2.1; +http://www.google.com/bot.html)'; do curl -s -o /dev/null -A "$ua" -w "%{http_code} %{size_download} Bytes $ua\n" https://example.com/; done
Fetch robots.txt and show the Disallow rules
$curl -s https://example.com/robots.txt | grep -iE '^\s*(user-agent|disallow|allow|sitemap):'
Find broken links on a website with wget --spider
$LC_ALL=C wget --spider -r -l 2 -nd -nv -w 1 -o /tmp/spider.log https://example.com/; grep -A 100 'broken link' /tmp/spider.log
Find images without an alt attribute on a page
$curl -s https://example.com/ | tr '\n' ' ' | grep -oiE '<img[^>]*>' | grep -viE '\balt='
Find meta robots and noindex in the HTML of a page
$curl -s https://example.com/ | tr '\n' ' ' | grep -oiE '<meta[^>]*name=.?(robots|googlebot)[^>]*>'
Find mixed content in the HTML of an HTTPS page
$curl -s https://example.com/ | tr '\n' ' ' | grep -oiE '(src|srcset|data-src)="http://[^"]+"|<link[^>]+href="http://[^"]+"|url\(.?http://[^)]+\)' | sort -u
Find the most frequent 404 errors in the access log
#awk '$9 == 404 {print $7}' /var/www/vhosts/system/example.com/logs/access_ssl_log | sort | uniq -c | sort -rn | head -20
Follow and show the redirect chain of a URL
$curl -sSIL http://example.com/ | grep -iE '^(HTTP/|location:)'
Read the canonical tag of a page with curl
$curl -s https://example.com/ | tr '\n' ' ' | grep -oiE '<link[^>]*canonical[^>]*>'
Read the hreflang tags of a page
$curl -s https://example.com/ | tr '\n' ' ' | grep -oiE '<link[^>]*hreflang[^>]*>'
Read the siteurl and home addresses
$wp option get siteurl --path=/var/www/vhosts/example.com/httpdocs && wp option get home --path=/var/www/vhosts/example.com/httpdocs
Read the title and meta description of a page with curl
$curl -s https://example.com/ | tr '\n' ' ' | grep -oiE '<title[^>]*>[^<]*</title>|<meta[^>]*name=.?description[^>]*>'
Regenerate permalinks and rewrite rules
$wp rewrite flush --path=/var/www/vhosts/example.com/httpdocs
Remove the search engine block in WordPress
$wp option update blog_public 1 --path=/var/www/vhosts/example.com/httpdocs
Show the number of redirects and the final URL
$curl -sL -o /dev/null -w 'Weiterleitungen: %{num_redirects}\nZiel: %{url_effective}\nStatus: %{http_code}\n' https://example.com/
Show the redirect chain of a URL with status codes
$curl -sIL "http://example.com/" | grep -iE "^(HTTP/|location:)"
Switch the WordPress database from http to https
$wp db export /var/www/vhosts/example.com/before-https.sql --path=/var/www/vhosts/example.com/httpdocs && wp search-replace 'http://example.com' 'https://example.com' --skip-columns=guid --path=/var/www/vhosts/example.com/httpdocs
Test http, https, www and non-www at once
$for u in http://example.com/ http://www.example.com/ https://example.com/ https://www.example.com/; do curl -s -o /dev/null -m 10 -w "%{http_code} $u -> %{redirect_url}\n" "$u"; done
Verify a real Googlebot via reverse DNS
$n=$(host 66.249.66.1 | awk '/pointer/ {print $NF}'); echo "PTR: $n"; host "$n"