[ { "category": "shell", "language": "shell", "attribution": "awk", "explanation": "awk splits each line into fields on whitespace. $1 is the first, $NF the last - so this prints the first and last column of every line.", "content": "awk '{print $1, $NF}' access.log" }, { "category": "shell", "language": "shell", "attribution": "awk", "explanation": "-F sets the field separator. This reads /etc/passwd and prints the username and shell, which are fields 1 and 7.", "content": "awk -F: '{print $1, $7}' /etc/passwd" }, { "category": "shell", "language": "shell", "attribution": "awk", "explanation": "A pattern before the block acts as a filter: only lines whose 9th field is 404 are printed. No grep needed.", "content": "awk '$9 == 404 {print $7}' access.log | sort | uniq -c | sort -rn" }, { "category": "shell", "language": "shell", "attribution": "awk", "explanation": "Associative arrays are awk's real power. This sums bytes per IP in one pass, then END prints the totals once input is exhausted.", "content": "awk '{bytes[$1] += $10} END {for (ip in bytes) print bytes[ip], ip}' access.log | sort -rn | head" }, { "category": "shell", "language": "shell", "attribution": "awk", "explanation": "NR is the current line number, NF the field count. This finds malformed rows: any line that does not have exactly three fields.", "content": "awk -F, 'NF != 3 {print NR\": \"$0}' data.csv" }, { "category": "shell", "language": "shell", "attribution": "awk", "explanation": "Deduplicate without sorting, preserving original order. The array records what has been seen; !seen[$0]++ is true only the first time.", "content": "awk '!seen[$0]++' file.txt" }, { "category": "shell", "language": "shell", "attribution": "awk", "explanation": "Sums a column and prints the average. END runs after the last line, so NR is the total row count by then.", "content": "awk '{sum += $3} END {print sum, sum/NR}' numbers.txt" }, { "category": "shell", "language": "shell", "attribution": "sed", "explanation": "Substitute in place. The g flag replaces every match on a line, not just the first; -i writes the file rather than printing.", "content": "sed -i 's/localhost/127.0.0.1/g' config.ini" }, { "category": "shell", "language": "shell", "attribution": "sed", "explanation": "Prints only lines 10 to 20. -n suppresses the default print, and p prints the range that matched.", "content": "sed -n '10,20p' large.log" }, { "category": "shell", "language": "shell", "attribution": "sed", "explanation": "Deletes comments and blank lines, which is how you read a config that is mostly documentation.", "content": "sed -e 's/#.*//' -e '/^[[:space:]]*$/d' /etc/ssh/sshd_config" }, { "category": "shell", "language": "shell", "attribution": "Pipelines", "explanation": "The classic word-frequency pipeline: split to one word per line, fold case, sort so duplicates are adjacent, count runs, order by count.", "content": "tr -cs '[:alpha:]' '\\n' < book.txt | tr 'A-Z' 'a-z' | sort | uniq -c | sort -rn | head -20" }, { "category": "shell", "language": "shell", "attribution": "Pipelines", "explanation": "uniq only collapses adjacent duplicates, which is why sort comes first. -c counts, and the second sort orders by that count.", "content": "cut -d' ' -f1 access.log | sort | uniq -c | sort -rn | head" }, { "category": "shell", "language": "shell", "attribution": "Pipelines", "explanation": "xargs turns a stream of names into arguments. -0 with -print0 is what makes filenames containing spaces safe.", "content": "find . -name '*.log' -print0 | xargs -0 grep -l 'ERROR'" }, { "category": "shell", "language": "shell", "attribution": "Pipelines", "explanation": "Runs a command per input line in parallel. -P4 keeps four running at once, and -I{} places each name where the braces are.", "content": "ls *.png | xargs -P4 -I{} convert {} -resize 50% small/{}" }, { "category": "shell", "language": "shell", "attribution": "Pipelines", "explanation": "Process substitution gives two commands to diff as if they were files, without writing either to disk.", "content": "diff <(sort a.txt) <(sort b.txt)" }, { "category": "shell", "language": "shell", "attribution": "Pipelines", "explanation": "tee writes to a file and passes the stream on, so you can log and keep processing in the same pipeline.", "content": "make 2>&1 | tee build.log | grep -E 'error|warning'" }, { "category": "shell", "language": "shell", "attribution": "Pipelines", "explanation": "comm compares two sorted files by column: -13 shows only lines unique to the second, which is the set that was added.", "content": "comm -13 <(sort old.txt) <(sort new.txt)" }, { "category": "shell", "language": "shell", "attribution": "Pipelines", "explanation": "jq selects and reshapes JSON. -r prints raw strings, so the output is usable by the next command rather than quoted.", "content": "curl -s https://api.example.com/users | jq -r '.[] | select(.active) | .email'" }, { "category": "shell", "language": "shell", "attribution": "Text processing", "explanation": "grep -o prints only the match, one per line, which turns a search into a stream you can count.", "content": "grep -oE '[0-9]{1,3}(\\.[0-9]{1,3}){3}' access.log | sort -u | wc -l" }, { "category": "shell", "language": "shell", "attribution": "Text processing", "explanation": "paste joins lines side by side; -s -d joins them all into one line with the chosen separator.", "content": "cut -f2 data.tsv | paste -sd, -" }, { "category": "shell", "language": "shell", "attribution": "Text processing", "explanation": "column formats whitespace-separated input into aligned columns, which makes an unreadable log readable.", "content": "mount | column -t | grep -v tmpfs" }, { "category": "shell", "language": "shell", "attribution": "Shell safety", "explanation": "The unofficial strict mode: exit on error, exit on undefined variable, and fail a pipeline if any stage fails rather than only the last.", "content": "set -euo pipefail" }, { "category": "shell", "language": "shell", "attribution": "Shell safety", "explanation": "Quoting \"$@\" preserves each argument exactly, including ones containing spaces. Unquoted $@ splits them apart.", "content": "for f in \"$@\"; do printf '%s\\n' \"${f%.*}\"; done" }, { "category": "shell", "language": "shell", "attribution": "Shell safety", "explanation": "A trap on EXIT cleans up whether the script succeeded, failed or was interrupted.", "content": "tmp=$(mktemp -d) && trap 'rm -rf \"$tmp\"' EXIT" }, { "category": "shell", "content": "awk -F: '$3 >= 1000 && $7 !~ /nologin|false/ {print $1, $6}' /etc/passwd", "attribution": "awk", "explanation": "Two conditions joined with &&. $3 >= 1000 keeps ordinary user accounts, and !~ excludes any shell matching nologin or false, so what is left is the humans who can log in.", "language": "shell" }, { "category": "shell", "content": "awk '{sum += $1; n++} END {if (n) printf \"%.2f\\n\", sum / n}' numbers.txt", "attribution": "awk", "explanation": "Variables in awk need no declaration and start at zero. END runs once after the last line, so this accumulates while reading and prints the mean at the finish.", "language": "shell" }, { "category": "shell", "content": "awk 'NR == FNR {seen[$1]; next} !($1 in seen)' first.txt second.txt", "attribution": "awk", "explanation": "NR is the overall line number and FNR restarts per file, so NR == FNR is true only while reading the first file. This prints lines of the second file whose first field never appeared in the first.", "language": "shell" }, { "category": "shell", "content": "sort access.log | awk '{print $1}' | sort | uniq -c | sort -rn | head -20", "attribution": "shell", "explanation": "uniq -c counts runs of identical adjacent lines, which is why the sort before it is required. sort -rn then orders those counts highest first, giving the twenty busiest addresses.", "language": "shell" }, { "category": "shell", "content": "find . -type f -name '*.log' -mtime +30 -print0 | xargs -0 rm -v", "attribution": "find", "explanation": "-print0 separates paths with a null byte and -0 tells xargs to expect that, which is the only safe way to pass file names containing spaces or newlines.", "language": "shell" }, { "category": "shell", "content": "grep -rn --include='*.rs' -e 'unwrap()' -e 'expect(' src/ | wc -l", "attribution": "grep", "explanation": "-r walks the tree, --include limits it to one file type, and repeated -e adds alternative patterns. Counting the result gives a rough measure of how much error handling is deferred.", "language": "shell" }, { "category": "shell", "content": "sed -i.bak -E 's/([0-9]{4})-([0-9]{2})-([0-9]{2})/\\3\\/\\2\\/\\1/g' dates.csv", "attribution": "sed", "explanation": "-i.bak edits in place and keeps the original alongside. The parenthesised groups are recalled as \\1, \\2 and \\3, which is how the date order is rearranged.", "language": "shell" }, { "category": "shell", "content": "tar czf - /var/www | ssh backup@host 'cat > site-$(date +%F).tar.gz'", "attribution": "tar", "explanation": "A dash as the file name makes tar write to standard output. The archive is never stored locally: it streams straight down the ssh connection into a file on the far side.", "language": "shell" }, { "category": "shell", "content": "ps -eo pid,ppid,rss,comm --sort=-rss | head -15", "attribution": "ps", "explanation": "-eo picks exactly which columns to print. rss is resident memory in kilobytes, and the minus sign in --sort reverses the order, so the heaviest processes come first.", "language": "shell" }, { "category": "shell", "content": "diff <(sort a.txt) <(sort b.txt) | grep '^[<>]'", "attribution": "shell", "explanation": "Process substitution gives each command a file name of its own, so diff can compare two pipelines without either being written to disk first.", "language": "shell" }, { "category": "shell", "content": "curl -sS -w '%{http_code} %{time_total}s\\n' -o /dev/null https://example.com", "attribution": "curl", "explanation": "-o /dev/null throws the body away while -w prints chosen variables, which turns curl into a quick check of status and latency alone.", "language": "shell" }, { "category": "shell", "content": "journalctl -u nginx --since '1 hour ago' -p err --no-pager | tail -50", "attribution": "journalctl", "explanation": "-p err keeps only entries at error priority or worse. --no-pager matters in a script, where an interactive pager would otherwise wait for a keypress that never comes.", "language": "shell" } ]