port decoded_bit_stream_parser

This commit is contained in:
Henry Schimke
2022-10-06 11:49:51 -05:00
parent cc74cc8240
commit fff18b0db6
4 changed files with 364 additions and 293 deletions

View File

@@ -38,7 +38,7 @@ fn test_random() {
}
assert_eq!(
encoding::all::UTF_8.name(),
StringUtils::guessCharset(&bytes, HashMap::new()).name()
StringUtils::guessCharset(&bytes, &HashMap::new()).name()
);
}
@@ -103,8 +103,8 @@ fn test_utf16_le() {
}
fn do_test(bytes: &Vec<u8>, charset: &dyn Encoding, encoding: &str) {
let guessedCharset = StringUtils::guessCharset(bytes, HashMap::new());
let guessedEncoding = StringUtils::guessEncoding(bytes, HashMap::new());
let guessedCharset = StringUtils::guessCharset(bytes, &HashMap::new());
let guessedEncoding = StringUtils::guessEncoding(bytes, &HashMap::new());
assert_eq!(charset.name(), guessedCharset.name());
assert_eq!(encoding, guessedEncoding);
}

View File

@@ -10,6 +10,8 @@ use std::rc::Rc;
use crate::Binarizer;
use crate::DecodeHintType;
use crate::DecodeHintValue;
use crate::DecodingHintDictionary;
use crate::Exceptions;
use crate::LuminanceSource;
use crate::RXingResultPoint;
@@ -102,7 +104,7 @@ impl StringUtils {
* "SJIS", "UTF8", "ISO8859_1", or the platform default encoding if none
* of these can possibly be correct
*/
pub fn guessEncoding(bytes: &[u8], hints: HashMap<DecodeHintType, String>) -> String {
pub fn guessEncoding(bytes: &[u8], hints: &DecodingHintDictionary) -> String {
let c = StringUtils::guessCharset(bytes, hints);
if c.name()
== encoding::label::encoding_from_whatwg_label("SJIS")
@@ -127,13 +129,12 @@ impl StringUtils {
* or the platform default encoding if
* none of these can possibly be correct
*/
pub fn guessCharset(
bytes: &[u8],
hints: HashMap<DecodeHintType, String>,
) -> &'static dyn Encoding {
pub fn guessCharset(bytes: &[u8], hints: &DecodingHintDictionary) -> &'static dyn Encoding {
match hints.get(&DecodeHintType::CHARACTER_SET) {
Some(hint) => {
return encoding::label::encoding_from_whatwg_label(hint).unwrap();
if let DecodeHintValue::CharacterSet(cs_name) = hint {
return encoding::label::encoding_from_whatwg_label(cs_name).unwrap();
}
}
_ => {}
};
@@ -1836,7 +1837,7 @@ pub struct DecoderRXingResult {
rawBytes: Vec<u8>,
numBits: usize,
text: String,
byteSegments: Vec<u8>,
byteSegments: Vec<Vec<u8>>,
ecLevel: String,
errorsCorrected: u64,
erasures: u64,
@@ -1847,14 +1848,14 @@ pub struct DecoderRXingResult {
}
impl DecoderRXingResult {
pub fn new(rawBytes: Vec<u8>, text: String, byteSegments: Vec<u8>, ecLevel: String) -> Self {
pub fn new(rawBytes: Vec<u8>, text: String, byteSegments: Vec<Vec<u8>>, ecLevel: String) -> Self {
Self::with_all(rawBytes, text, byteSegments, ecLevel, -2, -2, 0)
}
pub fn with_symbology(
rawBytes: Vec<u8>,
text: String,
byteSegments: Vec<u8>,
byteSegments: Vec<Vec<u8>>,
ecLevel: String,
symbologyModifier: u32,
) -> Self {
@@ -1872,7 +1873,7 @@ impl DecoderRXingResult {
pub fn with_sa(
rawBytes: Vec<u8>,
text: String,
byteSegments: Vec<u8>,
byteSegments: Vec<Vec<u8>>,
ecLevel: String,
saSequence: i32,
saParity: i32,
@@ -1891,7 +1892,7 @@ impl DecoderRXingResult {
pub fn with_all(
rawBytes: Vec<u8>,
text: String,
byteSegments: Vec<u8>,
byteSegments:Vec<Vec<u8>>,
ecLevel: String,
saSequence: i32,
saParity: i32,
@@ -1946,7 +1947,7 @@ impl DecoderRXingResult {
/**
* @return list of byte segments in the result, or {@code null} if not applicable
*/
pub fn getByteSegments(&self) -> &Vec<u8> {
pub fn getByteSegments(&self) -> &Vec<Vec<u8>> {
&self.byteSegments
}
@@ -2922,7 +2923,11 @@ impl ECIEncoderSet {
* @param fnc1 fnc1 denotes the character in the input that represents the FNC1 character or -1 for a non-GS1 bar
* code. When specified, it is considered an error to pass it as argument to the methods canEncode() or encode().
*/
pub fn new(stringToEncodeMain: &str, priorityCharset: Option<EncodingRef>, fnc1: Option<&str>) -> Self {
pub fn new(
stringToEncodeMain: &str,
priorityCharset: Option<EncodingRef>,
fnc1: Option<&str>,
) -> Self {
// List of encoders that potentially encode characters not in ISO-8859-1 in one byte.
let mut encoders: Vec<EncodingRef>;
@@ -2936,7 +2941,7 @@ impl ECIEncoderSet {
neededEncoders.push(encoding::all::ISO_8859_1);
let mut needUnicodeEncoder = if let Some(pc) = priorityCharset {
pc.name().starts_with("UTF")
}else{
} else {
false
};
@@ -2947,10 +2952,8 @@ impl ECIEncoderSet {
for encoder in &neededEncoders {
// for (CharsetEncoder encoder : neededEncoders) {
let c = stringToEncode.get(i).unwrap();
if (fnc1.is_some() && c == fnc1.as_ref().unwrap())
|| encoder
.encode(c, encoding::EncoderTrap::Strict)
.is_ok()
if (fnc1.is_some() && c == fnc1.as_ref().unwrap())
|| encoder.encode(c, encoding::EncoderTrap::Strict).is_ok()
{
canEncode = true;
break;
@@ -3003,20 +3006,19 @@ impl ECIEncoderSet {
encoders.push(encoding::all::UTF_8);
encoders.push(encoding::all::UTF_16BE);
}
//Compute priorityEncoderIndex by looking up priorityCharset in encoders
// if priorityCharset != null {
if priorityCharset.is_some() {
for i in 0..encoders.len() {
// for (int i = 0; i < encoders.length; i++) {
if priorityCharset.as_ref().unwrap().name() == encoders[i].name() {
priorityEncoderIndexValue = Some(i);
break;
if priorityCharset.is_some() {
for i in 0..encoders.len() {
// for (int i = 0; i < encoders.length; i++) {
if priorityCharset.as_ref().unwrap().name() == encoders[i].name() {
priorityEncoderIndexValue = Some(i);
break;
}
}
}}
}
// }
//invariants
assert_eq!(encoders[0].name(), encoding::all::ISO_8859_1.name());
@@ -3269,7 +3271,11 @@ impl MinimalECIInput {
* @param fnc1 denotes the character in the input that represents the FNC1 character or -1 if this is not GS1
* input.
*/
pub fn new(stringToEncodeInput: &str, priorityCharset: Option<EncodingRef>, fnc1: Option<&str>) -> Self {
pub fn new(
stringToEncodeInput: &str,
priorityCharset: Option<EncodingRef>,
fnc1: Option<&str>,
) -> Self {
let stringToEncode = stringToEncodeInput.graphemes(true).collect::<Vec<&str>>();
let encoderSet = ECIEncoderSet::new(stringToEncodeInput, priorityCharset, fnc1);
let bytes = if encoderSet.len() == 1 {
@@ -3278,11 +3284,19 @@ impl MinimalECIInput {
for i in 0..stringToEncode.len() {
// for (int i = 0; i < bytes.length; i++) {
let c = stringToEncode.get(i).unwrap();
bytes_hld[i] = if fnc1.is_some() && c == fnc1.as_ref().unwrap() { 1000 } else { c.chars().nth(0).unwrap() as u16 };
bytes_hld[i] = if fnc1.is_some() && c == fnc1.as_ref().unwrap() {
1000
} else {
c.chars().nth(0).unwrap() as u16
};
}
bytes_hld
} else {
Self::encodeMinimally(stringToEncodeInput, &encoderSet, fnc1.as_ref().unwrap().chars().nth(0).unwrap() as u16)
Self::encodeMinimally(
stringToEncodeInput,
&encoderSet,
fnc1.as_ref().unwrap().chars().nth(0).unwrap() as u16,
)
};
Self {
@@ -3339,7 +3353,8 @@ impl MinimalECIInput {
let mut start = 0;
let mut end = encoderSet.len();
if encoderSet.getPriorityEncoderIndex().is_some()
&& (ch.chars().nth(0).unwrap() as u16 == fnc1 || encoderSet.canEncode(ch, encoderSet.getPriorityEncoderIndex().unwrap()))
&& (ch.chars().nth(0).unwrap() as u16 == fnc1
|| encoderSet.canEncode(ch, encoderSet.getPriorityEncoderIndex().unwrap()))
{
start = encoderSet.getPriorityEncoderIndex().unwrap();
end = start + 1;
@@ -3406,7 +3421,7 @@ impl MinimalECIInput {
intsAL.splice(0..0, [1000]);
} else {
let bytes: Vec<u16> = encoderSet
.encode_char(&c.c , c.encoderIndex)
.encode_char(&c.c, c.encoderIndex)
.iter()
.map(|x| *x as u16)
.collect();
@@ -3469,7 +3484,11 @@ impl InputEdge {
size += prev.cachedTotalSize;
Self {
c: if c == fnc1Str { String::from("\u{1000}") } else { String::from(c) },
c: if c == fnc1Str {
String::from("\u{1000}")
} else {
String::from(c)
},
encoderIndex,
previous: Some(prev.clone()),
cachedTotalSize: size,
@@ -3481,7 +3500,11 @@ impl InputEdge {
}
Self {
c: if c == fnc1Str { String::from("\u{1000}") } else { String::from(c) },
c: if c == fnc1Str {
String::from("\u{1000}")
} else {
String::from(c)
},
encoderIndex,
previous: None,
cachedTotalSize: size,

View File

@@ -924,7 +924,7 @@ pub enum RXingResultMetadataValue {
* <p>This maps to a {@link java.util.List} of byte arrays corresponding to the
* raw bytes in the byte segments in the barcode, in order.</p>
*/
ByteSegments(Vec<u8>),
ByteSegments(Vec<Vec<u8>>),
/**
* Error correction level used, if applicable. The value type depends on the

View File

@@ -14,10 +14,12 @@
* limitations under the License.
*/
use crate::{Exceptions, common::{BitSource, StringUtils, DecoderRXingResult, CharacterSetECI}, DecodingHintDictionary};
use super::{VersionRef, Mode, ErrorCorrectionLevel};
use crate::{
common::{BitSource, CharacterSetECI, DecoderRXingResult, StringUtils},
DecodingHintDictionary, Exceptions,
};
use super::{ErrorCorrectionLevel, Mode, VersionRef};
/**
* <p>QR Codes can encode text as bits in one of several modes, and can use multiple modes
@@ -28,341 +30,387 @@ use super::{VersionRef, Mode, ErrorCorrectionLevel};
* @author Sean Owen
*/
/**
* See ISO 18004:2006, 6.4.4 Table 5
*/
const ALPHANUMERIC_CHARS : &str=
"0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZ $%*+-./:";
const GB2312_SUBSET :u8 = 1;
/**
* See ISO 18004:2006, 6.4.4 Table 5
*/
const ALPHANUMERIC_CHARS: &str = "0123456789ABCDEFGHIJKLMNOPQRSTUVWXYZ $%*+-./:";
const GB2312_SUBSET: u32 = 1;
pub fn decode(bytes : &[u8],
version:VersionRef,
ecLevel:ErrorCorrectionLevel,
hints:DecodingHintDictionary) -> Result<DecoderRXingResult,Exceptions> {
let bits = BitSource::new(bytes);
let result = String::with_capacity(50);
let byteSegments = vec![vec![0u8;0];0];
let symbolSequence = -1;
let parityData = -1;
pub fn decode(
bytes: &Vec<u8>,
version: VersionRef,
ecLevel: ErrorCorrectionLevel,
hints: &DecodingHintDictionary,
) -> Result<DecoderRXingResult, Exceptions> {
let mut bits = BitSource::new(bytes.clone());
let mut result = String::with_capacity(50);
let mut byteSegments = vec![vec![0u8; 0]; 0];
let mut symbolSequence = -1i32;
let mut parityData = -1i32;
let symbologyModifier;
// try {
let currentCharacterSetECI = None;
let fc1InEffect = false;
let hasFNC1first = false;
let hasFNC1second = false;
let mode;
loop {
let mut currentCharacterSetECI = None;
let mut fc1InEffect = false;
let mut hasFNC1first = false;
let mut hasFNC1second = false;
let mut mode;
loop {
// While still another segment to read...
if (bits.available() < 4) {
// OK, assume we're done. Really, a TERMINATOR mode should have been recorded here
mode = Mode::TERMINATOR;
if bits.available() < 4 {
// OK, assume we're done. Really, a TERMINATOR mode should have been recorded here
mode = Mode::TERMINATOR;
} else {
mode = Mode::forBits(bits.readBits(4)); // mode is encoded by 4 bits
mode = Mode::forBits(bits.readBits(4)? as u8)?; // mode is encoded by 4 bits
}
match mode {
Mode::TERMINATOR => {},
Mode::FNC1_FIRST_POSITION=> {
hasFNC1first = true; // symbology detection
// We do little with FNC1 except alter the parsed result a bit according to the spec
fc1InEffect = true;},
Mode::FNC1_SECOND_POSITION=> {
hasFNC1second = true; // symbology detection
// We do little with FNC1 except alter the parsed result a bit according to the spec
fc1InEffect = true;},
Mode::STRUCTURED_APPEND=> {
if (bits.available() < 16) {
return Err(Exceptions::FormatException(format!("Mode::Structured append expected bits.available() < 16, found bits of {}", bits.available())))
Mode::TERMINATOR => {}
Mode::FNC1_FIRST_POSITION => {
hasFNC1first = true; // symbology detection
// We do little with FNC1 except alter the parsed result a bit according to the spec
fc1InEffect = true;
}
Mode::FNC1_SECOND_POSITION => {
hasFNC1second = true; // symbology detection
// We do little with FNC1 except alter the parsed result a bit according to the spec
fc1InEffect = true;
}
Mode::STRUCTURED_APPEND => {
if (bits.available() < 16) {
return Err(Exceptions::FormatException(format!(
"Mode::Structured append expected bits.available() < 16, found bits of {}",
bits.available()
)));
}
// sequence number and parity is added later to the result metadata
// Read next 8 bits (symbol sequence #) and 8 bits (parity data), then continue
symbolSequence = bits.readBits(8)? as i32;
parityData = bits.readBits(8)? as i32;
}
Mode::ECI => {
// Count doesn't apply to ECI
let value = parseECIValue(&mut bits)?;
currentCharacterSetECI = Some(CharacterSetECI::getCharacterSetECIByValue(value)?);
if currentCharacterSetECI.is_none() {
return Err(Exceptions::FormatException(format!(
"Value of {} not valid",
value
)));
}
}
Mode::HANZI => {
// First handle Hanzi mode which does not start with character count
// Chinese mode contains a sub set indicator right after mode indicator
let subset = bits.readBits(4)?;
let countHanzi =
bits.readBits(mode.getCharacterCountBits(version) as usize)? as usize;
if subset == GB2312_SUBSET {
decodeHanziSegment(&mut bits, &mut result, countHanzi)?;
}
}
_ => {
// "Normal" QR code modes:
// How many characters will follow, encoded in this mode?
let count = bits.readBits(mode.getCharacterCountBits(version) as usize)? as usize;
match mode {
Mode::NUMERIC => decodeNumericSegment(&mut bits, &mut result, count)?,
Mode::ALPHANUMERIC => {
decodeAlphanumericSegment(&mut bits, &mut result, count, fc1InEffect)?
}
Mode::BYTE => decodeByteSegment(
&mut bits,
&mut result,
count,
currentCharacterSetECI,
&mut byteSegments,
hints,
)?,
Mode::KANJI => decodeKanjiSegment(&mut bits, &mut result, count)?,
_ => return Err(Exceptions::FormatException("".to_owned())),
}
}
// sequence number and parity is added later to the result metadata
// Read next 8 bits (symbol sequence #) and 8 bits (parity data), then continue
symbolSequence = bits.readBits(8);
parityData = bits.readBits(8);},
Mode::ECI=> {
// Count doesn't apply to ECI
let value = parseECIValue(&bits)?;
currentCharacterSetECI = CharacterSetECI::getCharacterSetECIByValue(value);
if (currentCharacterSetECI.is_none()) {
return Err(Exceptions::FormatException(format!("Value of {} not valid", value)))
}},
Mode::HANZI=> {
// First handle Hanzi mode which does not start with character count
// Chinese mode contains a sub set indicator right after mode indicator
let subset = bits.readBits(4);
let countHanzi = bits.readBits(mode.getCharacterCountBits(version));
if (subset == GB2312_SUBSET) {
decodeHanziSegment(&bits, &result, countHanzi)?;
}},
_=>{
// "Normal" QR code modes:
// How many characters will follow, encoded in this mode?
let count = bits.readBits(mode.getCharacterCountBits(version));
match mode {
Mode::NUMERIC=>
decodeNumericSegment(&bits, &result, count)?,
Mode::ALPHANUMERIC=>decodeAlphanumericSegment(&bits, &result, count, fc1InEffect)?,
Mode::BYTE=>decodeByteSegment(&bits, &result, count, currentCharacterSetECI, &byteSegments, hints)?,
Mode::KANJI=>decodeKanjiSegment(&bits, &result, count)?,
_=>
return Err(Exceptions::FormatException("".to_owned()))
}},
}
if mode != Mode::TERMINATOR {break}
}
if mode != Mode::TERMINATOR {
break;
}
}
if (currentCharacterSetECI != null) {
if (hasFNC1first) {
symbologyModifier = 4;
} else if (hasFNC1second) {
symbologyModifier = 6;
if currentCharacterSetECI.is_some() {
if hasFNC1first {
symbologyModifier = 4;
} else if hasFNC1second {
symbologyModifier = 6;
} else {
symbologyModifier = 2;
symbologyModifier = 2;
}
} else {
if (hasFNC1first) {
symbologyModifier = 3;
} else if (hasFNC1second) {
symbologyModifier = 5;
} else {
if hasFNC1first {
symbologyModifier = 3;
} else if hasFNC1second {
symbologyModifier = 5;
} else {
symbologyModifier = 1;
symbologyModifier = 1;
}
}
}
// } catch (IllegalArgumentException iae) {
// // from readBits() calls
// throw FormatException.getFormatInstance();
// }
return DecoderRXingResult::new(bytes,
result.toString(),
if byteSegments.isEmpty() {None} else {byteSegments},
if ecLevel.is_none() {None} else {ecLevel.toString()},
symbolSequence,
parityData,
symbologyModifier);
}
Ok(DecoderRXingResult::with_all(
bytes.clone(),
result,
byteSegments.to_vec(),
format!("{}", u8::from(ecLevel)),
symbolSequence,
parityData,
symbologyModifier,
))
}
/**
* See specification GBT 18284-2000
*/
fn decodeHanziSegment( bits:&BitSource,
result:&String,
count:u32) -> Result<(),Exceptions> {
/**
* See specification GBT 18284-2000
*/
fn decodeHanziSegment(
bits: &mut BitSource,
result: &mut String,
count: usize,
) -> Result<(), Exceptions> {
// Don't crash trying to read more bits than we have available.
if (count * 13 > bits.available()) {
return Err(Exceptions::FormatException("".to_owned()))
if count * 13 > bits.available() {
return Err(Exceptions::FormatException("".to_owned()));
}
// Each character will require 2 bytes. Read the characters as 2-byte pairs
// and decode as GB2312 afterwards
let buffer = vec![0u8;2 * count];
let offset = 0;
while (count > 0) {
// Each 13 bits encodes a 2-byte character
let twoBytes = bits.readBits(13)?;
let assembledTwoBytes = ((twoBytes / 0x060) << 8) | (twoBytes % 0x060);
if (assembledTwoBytes < 0x00A00) {
// In the 0xA1A1 to 0xAAFE range
assembledTwoBytes += 0x0A1A1;
} else {
// In the 0xB0A1 to 0xFAFE range
assembledTwoBytes += 0x0A6A1;
}
buffer[offset] = ((assembledTwoBytes >> 8) & 0xFF);
buffer[offset + 1] = (assembledTwoBytes & 0xFF);
offset += 2;
count-=1;
let mut buffer = vec![0u8; 2 * count];
let mut offset = 0;
let mut count = count;
while count > 0 {
// Each 13 bits encodes a 2-byte character
let twoBytes = bits.readBits(13)?;
let mut assembledTwoBytes = ((twoBytes / 0x060) << 8) | (twoBytes % 0x060);
if assembledTwoBytes < 0x00A00 {
// In the 0xA1A1 to 0xAAFE range
assembledTwoBytes += 0x0A1A1;
} else {
// In the 0xB0A1 to 0xFAFE range
assembledTwoBytes += 0x0A6A1;
}
// buffer[offset] = ((assembledTwoBytes >> 8) & 0xFF);
// buffer[offset + 1] = (assembledTwoBytes & 0xFF);
buffer[offset] = (assembledTwoBytes >> 8) as u8;
buffer[offset + 1] = assembledTwoBytes as u8;
offset += 2;
count -= 1;
}
let gb_encoder = encoding::label::encoding_from_whatwg_label("GB2312").unwrap();
let encode_string = gb_encoder.decode(&buffer, encoding::DecoderTrap::Strict).unwrap();
result.append(encode_string);
let encode_string = gb_encoder
.decode(&buffer, encoding::DecoderTrap::Strict)
.unwrap();
result.push_str(&encode_string);
Ok(())
}
}
fn decodeKanjiSegment( bits:&BitSource,
result:&String,
count:u32) -> Result<(),Exceptions> {
fn decodeKanjiSegment(
bits: &mut BitSource,
result: &mut String,
count: usize,
) -> Result<(), Exceptions> {
// Don't crash trying to read more bits than we have available.
if (count * 13 > bits.available()) {
return Err(Exceptions::FormatException("".to_owned()))
return Err(Exceptions::FormatException("".to_owned()));
}
// Each character will require 2 bytes. Read the characters as 2-byte pairs
// and decode as Shift_JIS afterwards
let buffer = vec![0u8;2 * count];
let offset = 0;
while (count > 0) {
// Each 13 bits encodes a 2-byte character
let twoBytes = bits.readBits(13);
let assembledTwoBytes = ((twoBytes / 0x0C0) << 8) | (twoBytes % 0x0C0);
if (assembledTwoBytes < 0x01F00) {
// In the 0x8140 to 0x9FFC range
assembledTwoBytes += 0x08140;
} else {
// In the 0xE040 to 0xEBBF range
assembledTwoBytes += 0x0C140;
}
buffer[offset] = (assembledTwoBytes >> 8);
buffer[offset + 1] = assembledTwoBytes;
offset += 2;
count-=1;
let mut buffer = vec![0u8; 2 * count];
let mut offset = 0;
let mut count = count;
while count > 0 {
// Each 13 bits encodes a 2-byte character
let twoBytes = bits.readBits(13)?;
let mut assembledTwoBytes = ((twoBytes / 0x0C0) << 8) | (twoBytes % 0x0C0);
if assembledTwoBytes < 0x01F00 {
// In the 0x8140 to 0x9FFC range
assembledTwoBytes += 0x08140;
} else {
// In the 0xE040 to 0xEBBF range
assembledTwoBytes += 0x0C140;
}
buffer[offset] = (assembledTwoBytes >> 8) as u8;
buffer[offset + 1] = assembledTwoBytes as u8;
offset += 2;
count -= 1;
}
let sjs_encoder = encoding::label::encoding_from_whatwg_label("SJIS").unwrap();
let encode_string = sjs_encoder.decode(&buffer, encoding::DecoderTrap::Strict).unwrap();
result.append(encode_string);
let encode_string = sjs_encoder
.decode(&buffer, encoding::DecoderTrap::Strict)
.unwrap();
result.push_str(&encode_string);
Ok(())
}
}
fn decodeByteSegment( bits:&BitSource,
result:&String,
count:u32,
currentCharacterSetECI:Option<&CharacterSetECI>,
byteSegments:&Vec<Vec<u8>>,
hints:DecodingHintDictionary) -> Result<(),Exceptions> {
fn decodeByteSegment(
bits: &mut BitSource,
result: &mut String,
count: usize,
currentCharacterSetECI: Option<CharacterSetECI>,
byteSegments: &mut Vec<Vec<u8>>,
hints: &DecodingHintDictionary,
) -> Result<(), Exceptions> {
// Don't crash trying to read more bits than we have available.
if (8 * count > bits.available()) {
return Err(Exceptions::FormatException("".to_owned()))
if 8 * count > bits.available() {
return Err(Exceptions::FormatException("".to_owned()));
}
let readBytes = vec![0u8;count];
let mut readBytes = vec![0u8; count];
for i in 0..count {
// for (int i = 0; i < count; i++) {
readBytes[i] = bits.readBits(8);
// for (int i = 0; i < count; i++) {
readBytes[i] = bits.readBits(8)? as u8;
}
let encoding;
if (currentCharacterSetECI.is_none()) {
// The spec isn't clear on this mode; see
// section 6.4.5: t does not say which encoding to assuming
// upon decoding. I have seen ISO-8859-1 used as well as
// Shift_JIS -- without anything like an ECI designator to
// give a hint.
encoding = StringUtils::guessCharset(&readBytes, hints);
if currentCharacterSetECI.is_none() {
// The spec isn't clear on this mode; see
// section 6.4.5: t does not say which encoding to assuming
// upon decoding. I have seen ISO-8859-1 used as well as
// Shift_JIS -- without anything like an ECI designator to
// give a hint.
encoding = StringUtils::guessCharset(&readBytes, &hints);
} else {
encoding = currentCharacterSetECI.getCharset();
encoding = CharacterSetECI::getCharset(currentCharacterSetECI.as_ref().unwrap());
}
let encode_string = encoding.decode(&readBytes, encoding::DecoderTrap::Strict).unwrap();
result.append(encode_string);
byteSegments.add(readBytes);
let encode_string = encoding
.decode(&readBytes, encoding::DecoderTrap::Strict)
.unwrap();
result.push_str(&encode_string);
byteSegments.push(readBytes);
Ok(())
}
}
fn toAlphaNumericChar( value:u32) -> Result<char,Exceptions> {
if (value >= ALPHANUMERIC_CHARS.len()) {
return Err(Exceptions::FormatException("".to_owned()))
fn toAlphaNumericChar(value: u32) -> Result<char, Exceptions> {
if value as usize >= ALPHANUMERIC_CHARS.len() {
return Err(Exceptions::FormatException("".to_owned()));
}
Ok( ALPHANUMERIC_CHARS[value])
}
Ok(ALPHANUMERIC_CHARS.chars().nth(value as usize).unwrap())
}
fn decodeAlphanumericSegment( bits:&BitSource,
result:&String,
count:u32,
fc1InEffect:bool) -> Result<(),Exceptions> {
fn decodeAlphanumericSegment(
bits: &mut BitSource,
result: &mut String,
count: usize,
fc1InEffect: bool,
) -> Result<(), Exceptions> {
// Read two characters at a time
let start = result.len();
let mut count = count;
while count > 1 {
if bits.available() < 11 {
return Err(Exceptions::FormatException("".to_owned()))
}
let nextTwoCharsBits = bits.readBits(11);
result.append(toAlphaNumericChar(nextTwoCharsBits / 45));
result.append(toAlphaNumericChar(nextTwoCharsBits % 45));
count -= 2;
if bits.available() < 11 {
return Err(Exceptions::FormatException("".to_owned()));
}
let nextTwoCharsBits = bits.readBits(11)?;
result.push(toAlphaNumericChar(nextTwoCharsBits / 45)?);
result.push(toAlphaNumericChar(nextTwoCharsBits % 45)?);
count -= 2;
}
if (count == 1) {
// special case: one character left
if (bits.available() < 6) {
return Err(Exceptions::FormatException("".to_owned()))
}
result.append(toAlphaNumericChar(bits.readBits(6)));
if count == 1 {
// special case: one character left
if bits.available() < 6 {
return Err(Exceptions::FormatException("".to_owned()));
}
result.push(toAlphaNumericChar(bits.readBits(6)?)?);
}
// See section 6.4.8.1, 6.4.8.2
if fc1InEffect {
// We need to massage the result a bit if in an FNC1 mode:
for i in start..result.len() {
// for (int i = start; i < result.length(); i++) {
if (result.charAt(i) == '%') {
if (i < result.length() - 1 && result.charAt(i + 1) == '%') {
// %% is rendered as %
result.deleteCharAt(i + 1);
} else {
// In alpha mode, % should be converted to FNC1 separator 0x1D
result.setCharAt(i, 0x1D);
}
// We need to massage the result a bit if in an FNC1 mode:
for i in start..result.len() {
// for (int i = start; i < result.length(); i++) {
if result.chars().nth(i).unwrap() == '%' {
if i < result.len() - 1 && result.chars().nth(i + 1).unwrap() == '%' {
// %% is rendered as %
result.remove(i + 1);
// result.deleteCharAt(i + 1);
} else {
// In alpha mode, % should be converted to FNC1 separator 0x1D
result.replace_range(i..i + 1, "\u{1D}");
// result.setCharAt(i, 0x1D);
}
}
}
}
}
Ok(())
}
}
fn decodeNumericSegment( bits:&BitSource,
result:&String,
count:u32) -> Result<(),Exceptions> {
fn decodeNumericSegment(
bits: &mut BitSource,
result: &mut String,
count: usize,
) -> Result<(), Exceptions> {
let mut count = count;
// Read three digits at a time
while count >= 3 {
// Each 10 bits encodes three digits
if bits.available() < 10 {
return Err(Exceptions::FormatException("".to_owned()))
}
let threeDigitsBits = bits.readBits(10)?;
if threeDigitsBits >= 1000 {
return Err(Exceptions::FormatException("".to_owned()))
}
result.append(toAlphaNumericChar(threeDigitsBits / 100));
result.append(toAlphaNumericChar((threeDigitsBits / 10) % 10));
result.append(toAlphaNumericChar(threeDigitsBits % 10));
count -= 3;
// Each 10 bits encodes three digits
if bits.available() < 10 {
return Err(Exceptions::FormatException("".to_owned()));
}
let threeDigitsBits = bits.readBits(10)?;
if threeDigitsBits >= 1000 {
return Err(Exceptions::FormatException("".to_owned()));
}
result.push(toAlphaNumericChar(threeDigitsBits / 100)?);
result.push(toAlphaNumericChar((threeDigitsBits / 10) % 10)?);
result.push(toAlphaNumericChar(threeDigitsBits % 10)?);
count -= 3;
}
if count == 2 {
// Two digits left over to read, encoded in 7 bits
if bits.available() < 7 {
return Err(Exceptions::FormatException("".to_owned()))
}
let twoDigitsBits = bits.readBits(7)?;
if twoDigitsBits >= 100 {
return Err(Exceptions::FormatException("".to_owned()))
}
result.append(toAlphaNumericChar(twoDigitsBits / 10));
result.append(toAlphaNumericChar(twoDigitsBits % 10));
// Two digits left over to read, encoded in 7 bits
if bits.available() < 7 {
return Err(Exceptions::FormatException("".to_owned()));
}
let twoDigitsBits = bits.readBits(7)?;
if twoDigitsBits >= 100 {
return Err(Exceptions::FormatException("".to_owned()));
}
result.push(toAlphaNumericChar(twoDigitsBits / 10)?);
result.push(toAlphaNumericChar(twoDigitsBits % 10)?);
} else if count == 1 {
// One digit left over to read
if bits.available() < 4 {
return Err(Exceptions::FormatException("".to_owned()))
}
let digitBits = bits.readBits(4)?;
if digitBits >= 10 {
return Err(Exceptions::FormatException("".to_owned()))
}
result.append(toAlphaNumericChar(digitBits));
// One digit left over to read
if bits.available() < 4 {
return Err(Exceptions::FormatException("".to_owned()));
}
let digitBits = bits.readBits(4)?;
if digitBits >= 10 {
return Err(Exceptions::FormatException("".to_owned()));
}
result.push(toAlphaNumericChar(digitBits)?);
}
Ok(())
}
}
fn parseECIValue( bits:&BitSource) -> Result<u32,Exceptions> {
fn parseECIValue(bits: &mut BitSource) -> Result<u32, Exceptions> {
let firstByte = bits.readBits(8)?;
if (firstByte & 0x80) == 0 {
// just one byte
return Ok(firstByte & 0x7F);
// just one byte
return Ok(firstByte & 0x7F);
}
if (firstByte & 0xC0) == 0x80 {
// two bytes
let secondByte = bits.readBits(8);
return Ok(((firstByte & 0x3F) << 8) | secondByte);
// two bytes
let secondByte = bits.readBits(8)?;
return Ok(((firstByte & 0x3F) << 8) | secondByte);
}
if (firstByte & 0xE0) == 0xC0 {
// three bytes
let secondThirdBytes = bits.readBits(16);
return Ok(((firstByte & 0x1F) << 16) | secondThirdBytes);
// three bytes
let secondThirdBytes = bits.readBits(16)?;
return Ok(((firstByte & 0x1F) << 16) | secondThirdBytes);
}
Err(Exceptions::FormatException("".to_owned()))
}
Err(Exceptions::FormatException("".to_owned()))
}