Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,9 @@ All notable changes to this project will be documented in this file.
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/),
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).

## [Unreleased]
- Speed up bulk `decode` of mostly-ASCII input by ~12-15%: each decode-table entry is now a niche-packed `NonZeroU32` (the UTF-8 bytes plus length), so a table read is a single load and the write a single store, and incomplete tables use the `Option` niche for undefined bytes. The allocation-free `decode_byte` primitive is also a few percent faster. No API changes.

## [2.1.0] - 2026-06-16
- Add an optional `cp437g` feature: the `CP437G` code page (CP437 overlaid with the IBM-Graphics glyphs at the C0 control byte range) for VGA text-mode rendering. Off by default.

Expand Down
68 changes: 24 additions & 44 deletions codegen/src/codegen_helper/decoder.rs
Original file line number Diff line number Diff line change
Expand Up @@ -14,31 +14,32 @@ fn format_bytes(encoded: &[u8]) -> String {
)
}

fn format_complete_entry(c: char, buffer: &mut [u8; 4]) -> String {
/// The bare `Entry::new(...)` constructor expression — no punctuation.
fn complete_entry(c: char, buffer: &mut [u8; 4]) -> String {
let (encoded, len) = encode_char(c, buffer);
format!(
"CompleteEntry{{buf: {}, len: {}}},\n",
format_bytes(encoded),
len
)
format!("Entry::new({}, {})", format_bytes(encoded), len)
}

/// Build CompleteEntry table. If `replacement` is Some, use it for None entries.
fn build_complete_table(definition: UnicodeMapping, replacement: Option<char>) -> String {
let mut res = "[\n".to_owned();
let mut buffer = [0u8; 4];

for c in definition {
let entry = match c.or(replacement) {
Some(c) => format_complete_entry(c, &mut buffer),
None => panic!("Complete codepage should not have None entries"),
};
res.push_str(&entry);
/// Assemble element expressions into a Rust array literal. Owns every comma,
/// newline, and bracket — element producers never write punctuation.
fn array_literal(elements: impl IntoIterator<Item = String>) -> String {
let mut res = String::from("[\n");
for e in elements {
res.push_str(&e);
res.push_str(",\n");
}
res.push(']');
res
}

fn build_complete_table(definition: UnicodeMapping, replacement: Option<char>) -> String {
let mut buffer = [0u8; 4];
array_literal(definition.iter().map(|&c| match c.or(replacement) {
Some(c) => complete_entry(c, &mut buffer),
None => panic!("Complete codepage should not have None entries"),
}))
}

pub fn build_complete_decode_table(definition: UnicodeMapping) -> String {
build_complete_table(definition, None)
}
Expand All @@ -47,33 +48,12 @@ pub fn build_incomplete_lossy_decode_table(definition: UnicodeMapping) -> String
build_complete_table(definition, Some('\u{FFFD}'))
}

fn len_to_variant(len: usize) -> &'static str {
match len {
1 => "IncompleteLen::One",
2 => "IncompleteLen::Two",
3 => "IncompleteLen::Three",
_ => panic!("Invalid UTF8 length"),
}
}

pub fn build_incomplete_decode_table(definition: UnicodeMapping) -> String {
let mut res = "[\n".to_owned();
let mut buffer = [0u8; 4];

for c in definition {
let entry = match c {
Some(c) => {
let (encoded, len) = encode_char(c, &mut buffer);
format!(
"Some(IncompleteEntry{{buf: {}, len: {}}}),\n",
format_bytes(encoded),
len_to_variant(len)
)
}
None => "None,\n".to_owned(),
};
res.push_str(&entry);
}
res.push(']');
res
// Incomplete tables are `[Option<Entry>; 256]`; undefined bytes are `None`
// (the niche-packed all-zero `u32`).
array_literal(definition.iter().map(|&c| match c {
Some(c) => format!("Some({})", complete_entry(c, &mut buffer)),
None => "None".to_owned(),
}))
}
22 changes: 5 additions & 17 deletions codegen/src/codegen_helper/templates/complete.rs
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@
use alloc::borrow::Cow;

use crate::{
decoder::{self, CompleteEntry},
decoder::{self, Entry},
encoder::Encoder,
CodePage,
};
Expand Down Expand Up @@ -49,20 +49,9 @@ impl CODERSTRUCT {
#[inline(always)]
pub fn decode_byte(self, b: u8) -> char {
// The UTF-8 decode table is already in memory for the bulk `decode`
// path, so decode the entry's stored bytes from the length we have
// rather than carrying a second (codepoint) table.
let e = DECODE_TABLE[b as usize];
let cp = match e.len {
1 => e.buf[0] as u32,
2 => ((e.buf[0] as u32 & 0x1F) << 6) | (e.buf[1] as u32 & 0x3F),
_ => {
((e.buf[0] as u32 & 0x0F) << 12)
| ((e.buf[1] as u32 & 0x3F) << 6)
| (e.buf[2] as u32 & 0x3F)
}
};
// SAFETY: table contents are valid UTF-8 for exactly one scalar value.
unsafe { char::from_u32_unchecked(cp) }
// path, so decode the entry rather than carrying a second (codepoint)
// table.
DECODE_TABLE[b as usize].to_char()
}

/// Decode a single CODERSTRUCT byte into its character.
Expand Down Expand Up @@ -149,8 +138,7 @@ const DECODE_TABLE_CHAR: [char; 256] = {
let mut t = ['\0'; 256];
let mut i = 0;
while i < 256 {
let e = DECODE_TABLE[i];
t[i] = decoder::entry_to_char(e.buf, e.len as u32);
t[i] = DECODE_TABLE[i].to_char();
i += 1;
}
t
Expand Down
25 changes: 6 additions & 19 deletions codegen/src/codegen_helper/templates/incomplete.rs
Original file line number Diff line number Diff line change
Expand Up @@ -2,15 +2,13 @@
use alloc::borrow::Cow;

use crate::{
decoder::{self, IncompleteEntry, IncompleteLen},
decoder::{self, Entry},
encoder::Encoder,
CodePage,
};

#[cfg(feature = "alloc")]
use crate::decoder::{
complete::decode_helper as decode_helper_lossy, incomplete::decode_helper, CompleteEntry,
};
use crate::decoder::{complete::decode_helper as decode_helper_lossy, incomplete::decode_helper};

#[cfg(feature = "alloc")]
use crate::{DecodeError, EncodeError};
Expand Down Expand Up @@ -86,20 +84,9 @@ impl CODERSTRUCT {
#[inline(always)]
pub fn decode_byte(self, b: u8) -> Option<char> {
// The UTF-8 decode table is already in memory for the bulk `decode`
// path, so decode the entry's stored bytes from the length we have
// rather than carrying a second (codepoint) table.
let e = DECODE_TABLE[b as usize]?;
let cp = match e.len as u32 {
1 => e.buf[0] as u32,
2 => ((e.buf[0] as u32 & 0x1F) << 6) | (e.buf[1] as u32 & 0x3F),
_ => {
((e.buf[0] as u32 & 0x0F) << 12)
| ((e.buf[1] as u32 & 0x3F) << 6)
| (e.buf[2] as u32 & 0x3F)
}
};
// SAFETY: table contents are valid UTF-8 for exactly one scalar value.
Some(unsafe { char::from_u32_unchecked(cp) })
// path, so decode the entry rather than carrying a second (codepoint)
// table.
DECODE_TABLE[b as usize].map(|e| e.to_char())
}

/// Decode a single CODERSTRUCT byte into its character.
Expand Down Expand Up @@ -210,7 +197,7 @@ const DECODE_TABLE_CHAR: [Option<char>; 256] = {
let mut i = 0;
while i < 256 {
t[i] = match DECODE_TABLE[i] {
Some(e) => Some(decoder::entry_to_char(e.buf, e.len as u32)),
Some(e) => Some(e.to_char()),
None => None,
};
i += 1;
Expand Down
Loading
Loading