Every language
14 langages, copy-ready. One at a time with syntax highlighting, or all inline.
JSJavaScript
import { readFile } from 'node:fs/promises';
const text = await readFile('notes.txt', 'utf8');
console.log(text.length);
// Omit the encoding and you get a Buffer, not a string:
const bytes = await readFile('notes.txt');
console.log(bytes.byteLength);The 'utf8' second argument is what makes readFile resolve to a string — drop it and you get a Buffer. Synchronous twin: readFileSync(path, 'utf8'). Node's Buffer.toString only knows utf8/utf16le/latin1/base64 family; anything else goes through new TextDecoder('shift_jis').decode(bytes).
TSTypeScript
import { readFile } from 'node:fs/promises';
async function readTextFile(path: string): Promise<string> {
const bytes = await readFile(path); // Buffer
// fatal: true rejects invalid UTF-8 instead of replacing it with U+FFFD
return new TextDecoder('utf-8', { fatal: true }).decode(bytes);
}
console.log((await readTextFile('notes.txt')).slice(0, 60));Buffer-to-string via TextDecoder with fatal: true turns mojibake into a TypeError at the boundary instead of silent replacement characters downstream — readFile(path, 'utf8') never throws on bad bytes, it quietly substitutes.
GoGo
package main
import (
"fmt"
"os"
)
func main() {
data, err := os.ReadFile("notes.txt")
if err != nil {
panic(err)
}
text := string(data)
fmt.Println(text)
}os.ReadFile (Go 1.16+) replaced ioutil.ReadFile and reads the whole file as []byte. string(data) copies into an immutable string, so peak memory is briefly ~2x the file size. There is no ReadToString: Go has no encoding machinery — a string is conventionally UTF-8 and nothing validates it.
RsRust
use std::fs;
fn main() -> Result<(), Box<dyn std::error::Error>> {
let text = fs::read_to_string("notes.txt")?;
println!("{} chars", text.chars().count());
Ok(())
}read_to_string hard-requires valid UTF-8 — a stray Latin-1 byte is an Err, not a best-effort decode. fs::read gives raw bytes; String::from_utf8_lossy repairs instead of rejecting. Slurping costs at most file-size twice in memory (buffer, then String).
PHPPHP
<?php
$text = file_get_contents('notes.txt');
if ($text === false) {
fwrite(STDERR, "cannot read notes.txt\n");
exit(1);
}
echo strlen($text), " bytes\n";PHP strings are byte strings: file_get_contents hands you the raw bytes and the encoding is your convention — mb_convert_encoding($text, 'UTF-8', 'Windows-1252') when the source isn't UTF-8. On failure it emits E_WARNING and returns false rather than throwing. The whole file lives in memory (memory_limit applies).
PyPython
from pathlib import Path
text = Path('notes.txt').read_text(encoding='utf-8')
print(len(text), 'characters')
raw = Path('notes.txt').read_bytes() # no decoding, no newline translation
print(len(raw), 'bytes')Always pass encoding= — the default is locale.getpreferredencoding(), so the identical file reads fine on Linux and crashes on cp1252 Windows (PEP 686 will flip the default to UTF-8). Text mode also runs universal-newline translation (\r\n -> \n); read_bytes() skips both.
CC
#include <stdio.h>
#include <stdlib.h>
/* Returns a NUL-terminated heap buffer; caller frees. NULL on failure. */
static char *read_file(const char *path, size_t *out_len) {
FILE *f = fopen(path, "rb");
if (!f) return NULL;
if (fseek(f, 0, SEEK_END) != 0) { fclose(f); return NULL; }
long size = ftell(f);
if (size < 0) { fclose(f); return NULL; }
rewind(f);
char *buf = malloc((size_t)size + 1);
if (!buf) { fclose(f); return NULL; }
size_t got = fread(buf, 1, (size_t)size, f);
fclose(f);
buf[got] = '\0';
*out_len = got;
return buf;
}
int main(void) {
size_t len = 0;
char *text = read_file("notes.txt", &len);
if (!text) {
perror("read_file");
return 1;
}
printf("%zu bytes read\n", len);
free(text);
return 0;
}No stdlib slurp: open in "rb" (Windows must not translate \r\n, and ftell then matches the byte count), seek-to-end/ftell for the size, rewind, fread, NUL-terminate. The explicit length matters — a file containing NUL bytes makes every str* function see a shorter string. fseek/ftell does not work on pipes; chunked reads or fstat(2) handle those.
C++C++
#include <fstream>
#include <iostream>
#include <sstream>
int main() {
std::ifstream in("notes.txt", std::ios::binary);
if (!in) {
std::cerr << "cannot open notes.txt\n";
return 1;
}
std::ostringstream buffer;
buffer << in.rdbuf(); // whole file in one move
std::string text = buffer.str();
std::cout << text.size() << " bytes\n";
}rdbuf-into-ostringstream is the classic slurp; the iterator spelling std::string text((std::istreambuf_iterator<char>(in)), {}) is equivalent. std::ios::binary stops Windows mangling newlines. C++17 upgrade: std::filesystem::file_size("notes.txt") lets you text.reserve() first — otherwise the stringstream grows geometrically. std::string is bytes; UTF-8 is not validated.
C#C#
using System;
using System.IO;
using System.Text;
string text = await File.ReadAllTextAsync("notes.txt");
Console.WriteLine($"{text.Length} chars");
// Non-UTF-8 source? Name the encoding:
// string latin1 = await File.ReadAllTextAsync("notes.txt", Encoding.Latin1);ReadAllText/ReadAllTextAsync sniff a BOM and otherwise assume UTF-8 — and unlike Java they strip a present BOM instead of leaving it in the string. text.Length counts UTF-16 code units (an emoji costs 2). A missing file throws FileNotFoundException; there is no Try-twin.
JvJava
import java.io.IOException;
import java.nio.file.Files;
import java.nio.file.Path;
public class ReadEntireFile {
public static void main(String[] args) throws IOException {
String text = Files.readString(Path.of("notes.txt"));
System.out.println(text.length() + " chars");
if (!text.isEmpty() && text.charAt(0) == '\uFEFF') {
text = text.substring(1); // readString KEEPS a UTF-8 BOM
}
}
}Files.readString (Java 11+) decodes UTF-8 but leaves a BOM in the string as a leading \uFEFF — the opposite of .NET — hence the trim. The pre-11 idiom new String(Files.readAllBytes(p)) meant platform charset until JEP 400 (Java 18) moved it to UTF-8; pin the old behavior with new String(bytes, StandardCharsets.UTF_8). length() counts UTF-16 code units.
SwSwift
import Foundation
do {
let url = URL(fileURLWithPath: "notes.txt")
let text = try String(contentsOf: url, encoding: .utf8)
print("\(text.count) characters")
} catch {
print("cannot read: \(error)")
}The encoding-less String(contentsOfFile:) initializer is deprecated precisely because a guessed encoding silently corrupts — pass .utf8 (or .isoLatin1, ...) explicitly. Missing file, permission failure and encoding mismatch all arrive as thrown errors, so try/catch is mandatory. count is Characters (grapheme clusters), not bytes or UTF-16 units.
KtKotlin
import java.io.File
import java.nio.charset.StandardCharsets
fun main() {
val text = File("notes.txt").readText() // UTF-8 by default
println("${text.length} chars")
val latin1 = File("notes.txt").readText(StandardCharsets.ISO_8859_1)
println("${latin1.length} chars")
}readText() is Kotlin's own API and has always defaulted to UTF-8 — historically the safer twin of Java's new String(bytes), which meant platform charset before JEP 400. readBytes() for raw bytes; useLines/forEachLine to stream instead of slurp. length is UTF-16 code units, as everywhere on the JVM.
RbRuby
text = File.read('notes.txt')
puts "#{text.length} characters"
bytes = File.binread('notes.txt') # ASCII-8BIT: no encoding assumption
puts "#{bytes.bytesize} bytes"File.read tags the result with the default external encoding (usually UTF-8) without validating a single byte — errors surface only when the string is used. Pass encoding: 'UTF-8' to be explicit, or File.binread for raw bytes. It all lives in memory: File.foreach is the streaming shape.
ZigZig
const std = @import("std");
pub fn main() !void {
const allocator = std.heap.page_allocator;
const text = try std.fs.cwd().readFileAlloc(allocator, "notes.txt", 64 * 1024 * 1024);
defer allocator.free(text);
std.debug.print("{d} bytes\n", .{text.len});
if (!std.unicode.utf8ValidateSlice(text)) {
std.debug.print("warning: not valid UTF-8\n", .{});
}
}readFileAlloc takes an explicit allocator (no GC in Zig) and a byte cap — the cap is the guard against slurping something enormous by accident. The result is []u8 bytes; utf8ValidateSlice is a check you run yourself, because nothing else will. You own the slice: free it.