view release on metacpan or search on metacpan
t/000-report-versions.t view on Meta::CPAN
return $self->_error("Did not provide a string to load");
}
# Byte order marks
# NOTE: Keeping this here to educate maintainers
# my %BOM = (
# "\357\273\277" => 'UTF-8',
# "\376\377" => 'UTF-16BE',
# "\377\376" => 'UTF-16LE',
# "\377\376\0\0" => 'UTF-32LE'
# "\0\0\376\377" => 'UTF-32BE',
# );
if ( $string =~ /^(?:\376\377|\377\376|\377\376\0\0|\0\0\376\377)/ ) {
return $self->_error("Stream has a non UTF-8 BOM");
} else {
# Strip UTF-8 bom if found, we'll just ignore it
$string =~ s/^\357\273\277//;
}
view all matches for this distribution
view release on metacpan or search on metacpan
t/000-report-versions.t view on Meta::CPAN
return $self->_error("Did not provide a string to load");
}
# Byte order marks
# NOTE: Keeping this here to educate maintainers
# my %BOM = (
# "\357\273\277" => 'UTF-8',
# "\376\377" => 'UTF-16BE',
# "\377\376" => 'UTF-16LE',
# "\377\376\0\0" => 'UTF-32LE'
# "\0\0\376\377" => 'UTF-32BE',
# );
if ( $string =~ /^(?:\376\377|\377\376|\377\376\0\0|\0\0\376\377)/ ) {
return $self->_error("Stream has a non UTF-8 BOM");
} else {
# Strip UTF-8 bom if found, we'll just ignore it
$string =~ s/^\357\273\277//;
}
view all matches for this distribution
view release on metacpan or search on metacpan
include/eshu_json.h view on Meta::CPAN
eshu_buf_t out;
const char *p = src;
const char *end = src + src_len;
int depth = 0;
/* skip UTF-8 BOM */
if (src_len >= 3 &&
(unsigned char)p[0] == 0xEF &&
(unsigned char)p[1] == 0xBB &&
(unsigned char)p[2] == 0xBF)
p += 3;
include/eshu_json.h view on Meta::CPAN
eshu_buf_t out;
const char *p = src;
const char *end = src + src_len;
const char *plain = p;
/* skip UTF-8 BOM */
if (src_len >= 3 &&
(unsigned char)p[0] == 0xEF &&
(unsigned char)p[1] == 0xBB &&
(unsigned char)p[2] == 0xBF) {
plain = p += 3;
view all matches for this distribution
view release on metacpan or search on metacpan
public/javascripts/vendor/require/text.js view on Meta::CPAN
//Using special require.nodeRequire, something added by r.js.
fs = require.nodeRequire('fs');
text.get = function (url, callback) {
var file = fs.readFileSync(url, 'utf8');
//Remove BOM (Byte Mark Order) from utf8 files if it is there.
if (file.indexOf('\uFEFF') === 0) {
file = file.substring(1);
}
callback(file);
};
public/javascripts/vendor/require/text.js view on Meta::CPAN
content = '';
try {
stringBuffer = new java.lang.StringBuffer();
line = input.readLine();
// Byte Order Mark (BOM) - The Unicode Standard, version 3.0, page 324
// http://www.unicode.org/faq/utf_bom.html
// Note that when we use utf-8, the BOM should appear as "EF BB BF", but it doesn't due to this bug in the JDK:
// http://bugs.sun.com/bugdatabase/view_bug.do?bug_id=4508058
if (line && line.length() && line.charAt(0) === 0xfeff) {
// Eat the BOM, since we've already found the encoding on this file,
// and we plan to concatenating this buffer with others; the BOM should
// only appear at the top of a file.
line = line.substring(1);
}
stringBuffer.append(line);
view all matches for this distribution
view release on metacpan or search on metacpan
bundled/CPAN-Meta-YAML/CPAN/Meta/YAML.pm view on Meta::CPAN
}
# Ensure Unicode character semantics, even for 0x80-0xff
utf8::upgrade($string);
# Check for and strip any leading UTF-8 BOM
$string =~ s/^\x{FEFF}//;
# Check for some special cases
return $self unless length $string;
view all matches for this distribution
view release on metacpan or search on metacpan
t/examples/fb2/hell_example_133321.fb2 view on Meta::CPAN
<p>СÑиÑ
иÑпаÑÑÐµÑ ÑаÑпÑоÑÑÑаненнÑй ÑекламнÑй блок, оÑÑиÑÐ°Ñ Ð¾Ñевидное. РиÑмоединиÑа, как Ð±Ñ ÑÑо ни казалоÑÑ Ð¿Ð°ÑадокÑалÑнÑм, ÑонеÑиÑеÑки ÑÑ...
</section> <section id="c_6"><title><p>6</p>
</title><p><strong>[ÐомменÑаÑий] </strong>Ðаже еÑли ÑÑеÑÑÑ ÑазÑеженнÑй газ, заполнÑÑÑий пÑоÑÑÑанÑÑво Ð¼ÐµÐ¶Ð´Ñ Ð·Ð²ÐµÐ·Ð´Ð°Ð¼Ð¸, Ñо вÑе Ñавно Южное полÑÑаÑие неÑ...
</section> <section id="c_7"><title><p>7</p>
</title><p><strong>[ÐомменÑаÑий] </strong>ÐÑлÑминаÑÐ¸Ñ Ð´Ð¸ÑкÑеÑно пÑедÑÑавлÑÐµÑ Ñобой агÑобиогеоÑеноз, но еÑли Ð±Ñ Ð¿ÐµÑен бÑло Ñаз в пÑÑÑ Ð¼ÐµÐ½ÑÑе, бÑло Ð±Ñ Ð...
</section> </section> </body> <binary content-type="image/jpeg" id="_1001000.jpg">/9j/4AAQSkZJRgABAQEAeAB4AAD/4QEARXhpZgAATU0AKgAAAAgABQEaAAUAAAABAAAASgEbAAUAAAABAAAAUgEoAAMAAAABAAIAAAExAAIAAAASAAAAWodpAAQAAAABAAAAbAAAAAAAAAB4AAAAAQAAAHgAAAABUGFpbnQu...
view all matches for this distribution
view release on metacpan or search on metacpan
include/ppport.h view on Meta::CPAN
BOL_t8_p8|5.033003||Viu
BOL_t8_pb|5.033003||Viu
BOL_tb|5.035004||Viu
BOL_tb_p8|5.033003||Viu
BOL_tb_pb|5.033003||Viu
BOM_UTF8|5.025005|5.003007|p
BOM_UTF8_FIRST_BYTE|5.019004||Viu
BOM_UTF8_TAIL|5.019004||Viu
boolSV|5.004000|5.003007|p
boot_core_builtin|5.035007||Viu
boot_core_mro|5.009005||Viu
boot_core_PerlIO|5.007002||Viu
boot_core_UNIVERSAL|5.003007||Viu
include/ppport.h view on Meta::CPAN
#endif
#endif
#if 'A' == 65
#ifndef BOM_UTF8
# define BOM_UTF8 "\xEF\xBB\xBF"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xEF\xBF\xBD"
#endif
#elif '^' == 95
#ifndef BOM_UTF8
# define BOM_UTF8 "\xDD\x73\x66\x73"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xDD\x73\x73\x71"
#endif
#elif '^' == 176
#ifndef BOM_UTF8
# define BOM_UTF8 "\xDD\x72\x65\x72"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xDD\x72\x72\x70"
#endif
view all matches for this distribution
view release on metacpan or search on metacpan
t/000-report-versions.t view on Meta::CPAN
return $self->_error("Did not provide a string to load");
}
# Byte order marks
# NOTE: Keeping this here to educate maintainers
# my %BOM = (
# "\357\273\277" => 'UTF-8',
# "\376\377" => 'UTF-16BE',
# "\377\376" => 'UTF-16LE',
# "\377\376\0\0" => 'UTF-32LE'
# "\0\0\376\377" => 'UTF-32BE',
# );
if ( $string =~ /^(?:\376\377|\377\376|\377\376\0\0|\0\0\376\377)/ ) {
return $self->_error("Stream has a non UTF-8 BOM");
} else {
# Strip UTF-8 bom if found, we'll just ignore it
$string =~ s/^\357\273\277//;
}
view all matches for this distribution
view release on metacpan or search on metacpan
BOL_t8_p8|5.033003||Viu
BOL_t8_pb|5.033003||Viu
BOL_tb|5.035004||Viu
BOL_tb_p8|5.033003||Viu
BOL_tb_pb|5.033003||Viu
BOM_UTF8|5.025005|5.003007|p
BOM_UTF8_FIRST_BYTE|5.019004||Viu
BOM_UTF8_TAIL|5.019004||Viu
boolSV|5.004000|5.003007|p
boot_core_builtin|5.035007||Viu
boot_core_mro|5.009005||Viu
boot_core_PerlIO|5.007002||Viu
boot_core_UNIVERSAL|5.003007||Viu
#endif
#endif
#if 'A' == 65
#ifndef BOM_UTF8
# define BOM_UTF8 "\xEF\xBB\xBF"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xEF\xBF\xBD"
#endif
#elif '^' == 95
#ifndef BOM_UTF8
# define BOM_UTF8 "\xDD\x73\x66\x73"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xDD\x73\x73\x71"
#endif
#elif '^' == 176
#ifndef BOM_UTF8
# define BOM_UTF8 "\xDD\x72\x65\x72"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xDD\x72\x72\x70"
#endif
view all matches for this distribution
view release on metacpan or search on metacpan
bin/bommer.pl view on Meta::CPAN
use strict;
use warnings;
use warnings qw(FATAL utf8); # Fatalize encoding glitches.
use File::BOM::Utils;
use Getopt::Long;
use Pod::Usage;
bin/bommer.pl view on Meta::CPAN
'output_file=s',
) )
{
pod2usage(1) if ($option{'help'});
exit File::BOM::Utils -> new(%option) -> run;
}
else
{
pod2usage(2);
}
bin/bommer.pl view on Meta::CPAN
=pod
=head1 NAME
bommer.pl - Check, Add and Remove BOMs
=head1 SYNOPSIS
bommer.pl [options]
bin/bommer.pl view on Meta::CPAN
=over 4
=item o add
Add the BOM named with the bom_name option to input_file.
Write the result to output_file.
=item o remove
Remove the BOM from the input_file. Write the result to output_file.
=item o test
Report the BOM status of input_file.
=back
Default: ''.
This option is mandatory.
=item -bom_name => $string
Specify which BOM to add to C<input_file>.
This option is mandatory if the C<action> is C<add>.
Values (always upper-case):
view all matches for this distribution
view release on metacpan or search on metacpan
lib/File/BOM.pm view on Meta::CPAN
package File::BOM;
=head1 NAME
File::BOM - Utilities for handling Byte Order Marks
=head1 SYNOPSIS
use File::BOM qw( :all )
=head2 high-level functions
# read a file with encoding from the BOM:
open_bom(FH, $file)
open_bom(FH, $file, ':utf8') # the same but with a default encoding
# get encoding too
$encoding = open_bom(FH, $file, ':utf8');
# open a potentially unseekable file:
($encoding, $spillage) = open_bom(FH, $file, ':utf8');
# change encoding of an open handle according to BOM
$encoding = defuse(*HANDLE);
($encoding, $spillage) = defuse(*HANDLE);
# Decode a string according to leading BOM:
$unicode = decode_from_bom($string_with_bom);
# Decode a string and get the encoding:
($unicode, $encoding) = decode_from_bom($string_with_bom)
=head2 PerlIO::via interface
# Read the Right Thing from a unicode file with BOM:
open(HANDLE, '<:via(File::BOM)', $filename)
# Writing little-endian UTF-16 file with BOM:
open(HANDLE, '>:encoding(UTF-16LE):via(File::BOM)', $filename)
=head2 lower-level functions
# read BOM encoding from a filehandle:
$encoding = get_encoding_from_filehandle(FH)
# Get encoding even if FH is unseekable:
($encoding, $spillage) = get_encoding_from_filehandle(FH);
# Get encoding from a known unseekable handle:
($encdoing, $spillage) = get_encoding_from_stream(FH);
# get encoding and BOM length from BOM at start of string:
($encoding, $offset) = get_encoding_from_bom($string);
=head2 variables
# print a BOM for a known encoding
print FH $enc2bom{$encoding};
# get an encoding from a known BOM
$enc = $bom2enc{$bom}
=head1 DESCRIPTION
This module provides functions for handling unicode byte order marks, which are
to be found at the beginning of some files and streams.
For details about what a byte order mark is, see
L<http://www.unicode.org/unicode/faq/utf_bom.html#BOM>
The intention of File::BOM is for files with BOMs to be readable as seamlessly
as possible, regardless of the encoding used. To that end, several different
interfaces are available, as shown in the synopsis above.
=cut
lib/File/BOM.pm view on Meta::CPAN
=head2 %bom2enc
Maps Byte Order marks to their encodings.
The keys of this hash are strings which represent the BOMs, the values are their
encodings, in a format which is understood by L<Encode>
The encodings represented in this hash are: UTF-8, UTF-16BE, UTF-16LE,
UTF-32BE and UTF-32LE
=head2 %enc2bom
A reverse-lookup hash for bom2enc, with a few aliases used in L<Encode>, namely utf8, iso-10646-1 and UCS-2.
Note that UTF-16, UTF-32 and UCS-4 are not included in this hash. Mainly
because Encode::encode automatically puts BOMs on output. See L<Encode::Unicode>
=cut
our(%bom2enc, %enc2bom, $MAX_BOM_LENGTH, $bom_re);
# length in bytes of the longest BOM
$MAX_BOM_LENGTH = 4;
Readonly %bom2enc => (
map { encode($_, "\x{feff}") => $_ } qw(
UTF-8
UTF-16BE
lib/File/BOM.pm view on Meta::CPAN
{
local $" = '|';
my @bombs = sort { length $b <=> length $a } keys %bom2enc;
Readonly $MAX_BOM_LENGTH => length $bombs[0];
Readonly $bom_re => qr/^(@bombs)/o;
}
=head1 FUNCTIONS
lib/File/BOM.pm view on Meta::CPAN
$encoding = open_bom(HANDLE, $filename, $default_mode)
($encoding, $spill) = open_bom(HANDLE, $filename, $default_mode)
opens HANDLE for reading on $filename, setting the mode to the appropriate
encoding for the BOM stored in the file.
On failure, a fatal error is raised, see the DIAGNOSTICS section for details on
how to catch these. This is in order to allow the return value(s) to be used for
other purposes.
If the file doesn't contain a BOM, $default_mode is used instead. Hence:
open_bom(FH, 'my_file.txt', ':utf8')
Opens my_file.txt for reading in an appropriate encoding found from the BOM in
that file, or as a UTF-8 file if none is found.
In the absence of a $default_mode argument, the following 2 calls should be equivalent:
open_bom(FH, 'no_bom.txt');
lib/File/BOM.pm view on Meta::CPAN
# create filehandle on the fly
$enc = open_bom(my $fh, $filename, ':utf8');
$line = <$fh>;
The filehandle will be cued up to read after the BOM. Unseekable files (e.g.
fifos) will cause croaking, unless called in list context to catch spillage
from the handle. Any spillage will be automatically decoded from the encoding,
if found.
e.g.
lib/File/BOM.pm view on Meta::CPAN
$enc = defuse(FH);
($enc, $spill) = defuse(FH);
FH should be a filehandle opened for reading, it will have the relevant encoding
layer pushed onto it be binmode if a BOM is found. Spillage should be Unicode,
not bytes.
Any uncaptured spillage will be silently lost. If the handle is unseekable, use
list context to avoid data loss.
If no BOM is found, the mode will be unaffected.
=cut
sub defuse (*) {
my $fh = qualify_to_ref(shift, caller);
lib/File/BOM.pm view on Meta::CPAN
$unicode_string = decode_from_bom($string, $default, $check)
($unicode_string, $encoding) = decode_from_bom($string, $default, $check)
Reads a BOM from the beginning of $string, decodes $string (minus the BOM) and
returns it to you as a perl unicode string.
if $string doesn't have a BOM, $default is used instead.
$check, if supplied, is passed to Encode::decode as the third argument.
If there's no BOM and no default, the original string is returned and encoding
is ''.
See L<Encode>
=cut
lib/File/BOM.pm view on Meta::CPAN
($encoding, $spillage) = get_encoding_from_filehandle(HANDLE)
Returns the encoding found in the given filehandle.
The handle should be opened in a non-unicode way (e.g. mode '<:bytes') so that
the BOM can be read in its natural state.
After calling, the handle will be set to read at a point after the BOM (or at
the beginning of the file if no BOM was found)
If called in scalar context, unseekable handles cause a croak().
If called in list context, unseekable handles will be read byte-by-byte and any
spillage will be returned. See get_encoding_from_stream()
lib/File/BOM.pm view on Meta::CPAN
=head2 get_encoding_from_stream
($encoding, $spillage) = get_encoding_from_stream(*FH);
Read a BOM from an unrewindable source. This means reading the stream one byte
at a time until either a BOM is found or every possible BOM is ruled out. Any
non-BOM bytes read from the handle will be returned in $spillage.
If a BOM is found and the spillage contains a partial character (judging by the
expected character width for the encoding) more bytes will be read from the
handle to ensure that a complete character is returned.
Spillage is always in bytes, not characters.
lib/File/BOM.pm view on Meta::CPAN
_get_encoding_unseekable($fh);
}
# internal:
#
# Return encoding and seek to position after BOM
sub _get_encoding_seekable (*) {
my $fh = shift;
# This doesn't work on all platforms:
# defined(read($fh, my $bom, $MAX_BOM_LENGTH))
# or croak "Couldn't read from handle: $!";
my $bom = eval { _safe_read($fh, $MAX_BOM_LENGTH) };
croak "Couldn't read from handle: $@" if $@;
my($enc, $off) = get_encoding_from_bom($bom);
seek($fh, $off, SEEK_SET) or croak "Couldn't reset read position: $!";
lib/File/BOM.pm view on Meta::CPAN
return $enc;
}
# internal:
#
# Return encoding and non-BOM overspill
sub _get_encoding_unseekable (*) {
my $fh = shift;
my $so_far = '';
for my $c (1 .. $MAX_BOM_LENGTH) {
# defined(read($fh, my $byte, 1)) or croak "Couldn't read byte: $!";
my $byte = eval { _safe_read($fh, 1) };
croak "Couldn't read byte: $@" if $@;
$so_far .= $byte;
# find matching BOMs
my @possible = grep { $so_far eq substr($_, 0, $c) } keys %bom2enc;
if (@possible == 1 and my $enc = $bom2enc{$so_far}) {
# There's only one match, this must be it
return ($enc, '');
lib/File/BOM.pm view on Meta::CPAN
$spill .= $extra;
return ($enc, $spill);
}
else {
# no BOM
return ('', $so_far . $spill);
}
}
}
}
lib/File/BOM.pm view on Meta::CPAN
=head2 get_encoding_from_bom
($encoding, $offset) = get_encoding_from_bom($string)
Returns the encoding and length in bytes of the BOM in $string.
If there is no BOM, an empty string is returned and $offset is zero.
To get the data from the string, the following should work:
use Encode;
lib/File/BOM.pm view on Meta::CPAN
}
}
=head1 PerlIO::via interface
File::BOM can be used as a PerlIO::via interface.
open(HANDLE, '<:via(File::BOM)', 'my_file.txt');
open(HANDLE, '>:encoding(UTF-16LE):via(File::BOM)', 'out_file.txt');
print "foo\n"; # BOM is written to file here
This method is less prone to errors on non-seekable files as spillage is
incorporated into an internal buffer, but it doesn't give you any information
about the encoding being used, or indeed whether or not a BOM
was present.
There are a few known problems with this interface, especially surrounding
seek() and tell(), please see the BUGS section for more details about this.
=head2 Reading
The via(File::BOM) layer must be added before the handle is read from, otherwise
any BOM will be missed. If there is no BOM, no decoding will be done.
Because of a limitation in PerlIO::via, read() always works on bytes, not characters. BOM decoding will still be done but output will be bytes of UTF-8.
open(BOM, '<:via(File::BOM)', $file);
$bytes_read = read(BOM, $buffer, $length);
$unicode = decode('UTF-8', $buffer, Encode::FB_QUIET);
# Now $unicode is valid unicode and $buffer contains any left-over bytes
=head2 Writing
Add the via(File::BOM) layer on top of a unicode encoding layer to print a BOM
at the start of the output file. This needs to be done before any data is
written. The BOM is written as part of the first print command on the handle, so
if you don't print anything to the handle, you won't get a BOM.
There is a "Wide character in print" warning generated when the via(File::BOM)
layer doesn't receive utf8 on writing. This glitch was resolved in perl version
5.8.7, but if your perl version is older than that, you'll need to make sure
that the via(File::BOM) layer receives utf8 like this:
# This works OK
open(FH, '>:encoding(UTF-16LE):via(File::BOM):utf8', $filename)
# This generates warnings with older perls
open(FH, '>:encoding(UTF-16LE):via(File::BOM)', $filename)
=head2 Seeking
Seeking with SEEK_SET results in an offset equal to the length of any detected
BOM being applied to the position parameter. Thus:
# Seek to end of BOM (not start of file!)
seek(FILE_BOM_HANDLE, 0, SEEK_SET)
=head2 Telling
In order to work correctly with seek(), tell() also returns a postion adjusted
by the length of the BOM.
=cut
sub PUSHED { bless({offset => 0}, $_[0]) || -1 }
lib/File/BOM.pm view on Meta::CPAN
=item * L<Encode>
=item * L<Encode::Unicode>
=item * L<http://www.unicode.org/unicode/faq/utf_bom.html#BOM>
=back
=head1 DIAGNOSTICS
lib/File/BOM.pm view on Meta::CPAN
_get_encoding_seekable() couldn't read the handle. This function is called from
get_encoding_from_filehandle(), defuse() and open_bom()
=item * Couldn't reset read position: $!
_get_encoding_seekable couldn't seek to the position after the BOM.
=item * Couldn't read byte: $!
get_encoding_from_stream couldn't read from the handle. This function is called
from get_encoding_from_filehandle() and open_bom() when the handle or file is
view all matches for this distribution
view release on metacpan or search on metacpan
lib/File/Copy/ppport.h view on Meta::CPAN
BOL_t8_p8|5.033003||Viu
BOL_t8_pb|5.033003||Viu
BOL_tb|5.035004||Viu
BOL_tb_p8|5.033003||Viu
BOL_tb_pb|5.033003||Viu
BOM_UTF8|5.025005|5.003007|p
BOM_UTF8_FIRST_BYTE|5.019004||Viu
BOM_UTF8_TAIL|5.019004||Viu
boolSV|5.004000|5.003007|p
boot_core_builtin|5.035007||Viu
boot_core_mro|5.009005||Viu
boot_core_PerlIO|5.007002||Viu
boot_core_UNIVERSAL|5.003007||Viu
lib/File/Copy/ppport.h view on Meta::CPAN
#endif
#endif
#if 'A' == 65
#ifndef BOM_UTF8
# define BOM_UTF8 "\xEF\xBB\xBF"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xEF\xBF\xBD"
#endif
#elif '^' == 95
#ifndef BOM_UTF8
# define BOM_UTF8 "\xDD\x73\x66\x73"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xDD\x73\x73\x71"
#endif
#elif '^' == 176
#ifndef BOM_UTF8
# define BOM_UTF8 "\xDD\x72\x65\x72"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xDD\x72\x72\x70"
#endif
view all matches for this distribution
view release on metacpan or search on metacpan
package File::Find::Rule::BOM;
use base qw(File::Find::Rule);
use strict;
use warnings;
use String::BOM qw(file_has_bom);
our $VERSION = 0.03;
# Detect BOM.
sub File::Find::Rule::bom {
my $file_find_rule = shift;
return _bom($file_find_rule);
}
# Detect UTF-8 BOM.
sub File::Find::Rule::bom_utf8 {
my $file_find_rule = shift;
return _bom($file_find_rule, 'UTF-8');
}
# Detect UTF-16 BOM.
sub File::Find::Rule::bom_utf16 {
my $file_find_rule = shift;
return _bom($file_find_rule, 'UTF-16');
}
# Detect UTF-32 BOM.
sub File::Find::Rule::bom_utf32 {
my $file_find_rule = shift;
return _bom($file_find_rule, 'UTF-32');
}
=encoding utf8
=head1 NAME
File::Find::Rule::BOM - Common rules for searching for BOM in files.
=head1 SYNOPSIS
use File::Find::Rule;
use File::Find::Rule::BOM;
my @files = File::Find::Rule->bom->in($dir);
my @files = File::Find::Rule->bom_utf8->in($dir);
my @files = File::Find::Rule->bom_utf16->in($dir);
my @files = File::Find::Rule->bom_utf32->in($dir);
=head1 DESCRIPTION
This Perl module contains File::Find::Rule rules for detecting Byte Order Mark
in files.
BOM (Byte Order Mark) is a particular usage of the special Unicode character,
U+FEFF BYTE ORDER MARK, whose appearance as a magic number at the start of a
text stream can signal several things to a program reading the text.
See L<Byte order mark on Wikipedia|https://en.wikipedia.org/wiki/Byte order mark>.
=head2 C<bom>
my @files = File::Find::Rule->bom->in($dir);
The C<bom()> rule detect files with BOM.
=head2 C<bom_utf8>
my @files = File::Find::Rule->bom_utf8->in($dir);
The C<bom_utf8()> rule detect files with UTf-8 BOM.
=head2 C<bom_utf16>
my @files = File::Find::Rule->bom_utf16->in($dir);
The C<bom_utf16()> rule detect files with UTF-16 BOM.
=head2 C<bom_utf32>
my @files = File::Find::Rule->bom_utf32->in($dir);
The C<bom_utf32()> rule detect files with UTF-32 BOM.
=head1 EXAMPLE1
use strict;
use warnings;
use File::Find::Rule;
use File::Find::Rule::BOM;
# Arguments.
if (@ARGV < 1) {
print STDERR "Usage: $0 dir\n";
exit 1;
}
my $dir = $ARGV[0];
# Print all files with BOM in directory.
foreach my $file (File::Find::Rule->bom->in($dir)) {
print "$file\n";
}
# Output like:
use strict;
use warnings;
use File::Find::Rule;
use File::Find::Rule::BOM;
# Arguments.
if (@ARGV < 1) {
print STDERR "Usage: $0 dir\n";
exit 1;
}
my $dir = $ARGV[0];
# Print all files with UTF-8 BOM in directory.
foreach my $file (File::Find::Rule->bom_utf8->in($dir)) {
print "$file\n";
}
# Output like:
use strict;
use warnings;
use File::Find::Rule;
use File::Find::Rule::BOM;
# Arguments.
if (@ARGV < 1) {
print STDERR "Usage: $0 dir\n";
exit 1;
}
my $dir = $ARGV[0];
# Print all files with UTF-16 BOM in directory.
foreach my $file (File::Find::Rule->bom_utf16->in($dir)) {
print "$file\n";
}
# Output like:
use strict;
use warnings;
use File::Find::Rule;
use File::Find::Rule::BOM;
# Arguments.
if (@ARGV < 1) {
print STDERR "Usage: $0 dir\n";
exit 1;
}
my $dir = $ARGV[0];
# Print all files with UTF-32 BOM in directory.
foreach my $file (File::Find::Rule->bom_utf32->in($dir)) {
print "$file\n";
}
# Output like:
# Usage: qr{[\w\/]+} dir
=head1 DEPENDENCIES
L<File::Find::Rule>,
L<String::BOM>.
=head1 SEE ALSO
=over
=back
=head1 REPOSITORY
L<https://github.com/michal-josef-spacek/File-Find-Rule-BOM>
=head1 AUTHOR
Michal Josef Å paÄek L<mailto:skim@cpan.org>
view all matches for this distribution
view release on metacpan or search on metacpan
requires 'Test::Pod';
requires 'Test::NoTabs';
requires 'Test::Perl::Metrics::Lite';
requires 'Test::Vars';
requires 'Test::File::Find::Rule';
requires 'File::Find::Rule::BOM';
};
view all matches for this distribution
view release on metacpan or search on metacpan
t/000-report-versions.t view on Meta::CPAN
return $self->_error("Did not provide a string to load");
}
# Byte order marks
# NOTE: Keeping this here to educate maintainers
# my %BOM = (
# "\357\273\277" => 'UTF-8',
# "\376\377" => 'UTF-16BE',
# "\377\376" => 'UTF-16LE',
# "\377\376\0\0" => 'UTF-32LE'
# "\0\0\376\377" => 'UTF-32BE',
# );
if ( $string =~ /^(?:\376\377|\377\376|\377\376\0\0|\0\0\376\377)/ ) {
return $self->_error("Stream has a non UTF-8 BOM");
} else {
# Strip UTF-8 bom if found, we'll just ignore it
$string =~ s/^\357\273\277//;
}
view all matches for this distribution
view release on metacpan or search on metacpan
BOL_t8_p8|5.033003||Viu
BOL_t8_pb|5.033003||Viu
BOL_tb|5.035004||Viu
BOL_tb_p8|5.033003||Viu
BOL_tb_pb|5.033003||Viu
BOM_UTF8|5.025005|5.003007|p
BOM_UTF8_FIRST_BYTE|5.019004||Viu
BOM_UTF8_TAIL|5.019004||Viu
boolSV|5.004000|5.003007|p
boot_core_builtin|5.035007||Viu
boot_core_mro|5.009005||Viu
boot_core_PerlIO|5.007002||Viu
boot_core_UNIVERSAL|5.003007||Viu
#endif
#endif
#if 'A' == 65
#ifndef BOM_UTF8
# define BOM_UTF8 "\xEF\xBB\xBF"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xEF\xBF\xBD"
#endif
#elif '^' == 95
#ifndef BOM_UTF8
# define BOM_UTF8 "\xDD\x73\x66\x73"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xDD\x73\x73\x71"
#endif
#elif '^' == 176
#ifndef BOM_UTF8
# define BOM_UTF8 "\xDD\x72\x65\x72"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xDD\x72\x72\x70"
#endif
view all matches for this distribution
view release on metacpan or search on metacpan
BmFLAGS|5.009005||Viu
BmPREVIOUS|5.003007||Viu
BmRARE|5.003007||Viu
BmUSEFUL|5.003007||Viu
BOL|5.003007||Viu
BOM_UTF8|5.025005|5.003007|p
BOM_UTF8_FIRST_BYTE|5.019004||Viu
BOM_UTF8_TAIL|5.019004||Viu
bool|5.003007||Viu
boolSV|5.004000|5.003007|p
boot_core_mro|5.009005||Viu
boot_core_PerlIO|5.007002||Viu
boot_core_UNIVERSAL|5.003007||Viu
#endif
#endif
#if 'A' == 65
#ifndef BOM_UTF8
# define BOM_UTF8 "\xEF\xBB\xBF"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xEF\xBF\xBD"
#endif
#elif '^' == 95
#ifndef BOM_UTF8
# define BOM_UTF8 "\xDD\x73\x66\x73"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xDD\x73\x73\x71"
#endif
#elif '^' == 176
#ifndef BOM_UTF8
# define BOM_UTF8 "\xDD\x72\x65\x72"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xDD\x72\x72\x70"
#endif
view all matches for this distribution
view release on metacpan or search on metacpan
xs/ppport.h view on Meta::CPAN
(index($4, 'n') >= 0 ? ( nothxarg => 1 ) : ()),
} )
: die "invalid spec: $_" } qw(
AvFILLp|5.004050||p
AvFILL|||
BOM_UTF8|||
BhkDISABLE||5.024000|
BhkENABLE||5.024000|
BhkENTRY_set||5.024000|
BhkENTRY|||
BhkFLAGS|||
view all matches for this distribution
view release on metacpan or search on metacpan
lib/File/LoadLines.pm view on Meta::CPAN
It will transparently fetch data from the network if the provided file
name is a URL.
File::LoadLines automatically handles ASCII, Latin-1 and UTF-8 text.
When the file has a BOM, it handles UTF-8, UTF-16 LE and BE, and
UTF-32 LE and BE.
Recognized line terminators are NL (Unix, Linux), CRLF (DOS, Windows)
and CR (Mac)
lib/File/LoadLines.pm view on Meta::CPAN
$options->{encoding} //= 'Perl';
}
# Detect Byte Order Mark.
elsif ( $data =~ /^\xEF\xBB\xBF/ ) {
warn("$name is UTF-8 (BOM)\n") if $options->{debug};
$options->{encoding} = 'UTF-8';
$data = decode( "UTF-8", substr($data, 3) );
}
elsif ( $data =~ /^\xFE\xFF/ ) {
warn("$name is UTF-16BE (BOM)\n") if $options->{debug};
$options->{encoding} = 'UTF-16BE';
$data = decode( "UTF-16BE", substr($data, 2) );
}
elsif ( $data =~ /^\xFF\xFE\x00\x00/ ) {
warn("$name is UTF-32LE (BOM)\n") if $options->{debug};
$options->{encoding} = 'UTF-32LE';
$data = decode( "UTF-32LE", substr($data, 4) );
}
elsif ( $data =~ /^\xFF\xFE/ ) {
warn("$name is UTF-16LE (BOM)\n") if $options->{debug};
$options->{encoding} = 'UTF-16LE';
$data = decode( "UTF-16LE", substr($data, 2) );
}
elsif ( $data =~ /^\x00\x00\xFE\xFF/ ) {
warn("$name is UTF-32BE (BOM)\n") if $options->{debug};
$options->{encoding} = 'UTF-32BE';
$data = decode( "UTF-32BE", substr($data, 4) );
}
# No BOM, did user specify an encoding?
elsif ( $options->{encoding} ) {
warn("$name is ", $options->{encoding}, " (fallback)\n")
if $options->{debug};
$data = decode( $options->{encoding}, $data, 1 );
}
lib/File/LoadLines.pm view on Meta::CPAN
loadlines( $filename, $options );
}
=head1 SEE ALSO
There are currently no other modules that handle BOM detection and
line splitting.
I have a faint hope that future versions of Perl and Raku will deal
with this transparently, but I fear the worst.
view all matches for this distribution
view release on metacpan or search on metacpan
bind_match|5.003007||Viu
block_end|5.004000|5.004000|
block_gimme|5.004000|5.004000|u
blockhook_register|5.013003|5.013003|x
block_start|5.004000|5.004000|
BOM_UTF8|5.025005|5.003007|p
boolSV|5.004000|5.003007|p
boot_core_mro|5.009005||Viu
boot_core_PerlIO|5.007002||Viu
boot_core_UNIVERSAL|5.003007||Viu
_byte_dump_string|5.025006||Viu
#endif
#endif
#if 'A' == 65
#ifndef BOM_UTF8
# define BOM_UTF8 "\xEF\xBB\xBF"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xEF\xBF\xBD"
#endif
#elif '^' == 95
#ifndef BOM_UTF8
# define BOM_UTF8 "\xDD\x73\x66\x73"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xDD\x73\x73\x71"
#endif
#elif '^' == 176
#ifndef BOM_UTF8
# define BOM_UTF8 "\xDD\x72\x65\x72"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xDD\x72\x72\x70"
#endif
view all matches for this distribution
view release on metacpan or search on metacpan
lib/File/ppport.h view on Meta::CPAN
BOL_t8_p8|5.033003||Viu
BOL_t8_pb|5.033003||Viu
BOL_tb|5.035004||Viu
BOL_tb_p8|5.033003||Viu
BOL_tb_pb|5.033003||Viu
BOM_UTF8|5.025005|5.003007|p
BOM_UTF8_FIRST_BYTE|5.019004||Viu
BOM_UTF8_TAIL|5.019004||Viu
boolSV|5.004000|5.003007|p
boot_core_builtin|5.035007||Viu
boot_core_mro|5.009005||Viu
boot_core_PerlIO|5.007002||Viu
boot_core_UNIVERSAL|5.003007||Viu
lib/File/ppport.h view on Meta::CPAN
#endif
#endif
#if 'A' == 65
#ifndef BOM_UTF8
# define BOM_UTF8 "\xEF\xBB\xBF"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xEF\xBF\xBD"
#endif
#elif '^' == 95
#ifndef BOM_UTF8
# define BOM_UTF8 "\xDD\x73\x66\x73"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xDD\x73\x73\x71"
#endif
#elif '^' == 176
#ifndef BOM_UTF8
# define BOM_UTF8 "\xDD\x72\x65\x72"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xDD\x72\x72\x70"
#endif
view all matches for this distribution
view release on metacpan or search on metacpan
include/yyjson.h view on Meta::CPAN
- Read positive integer as uint64_t.
- Read negative integer as int64_t.
- Read floating-point number as double with round-to-nearest mode.
- Read integer which cannot fit in uint64_t or int64_t as double.
- Report error if double number is infinity.
- Report error if string contains invalid UTF-8 character or BOM.
- Report error on trailing commas, comments, inf and nan literals. */
static const yyjson_read_flag YYJSON_READ_NOFLAG = 0;
/** Read the input data in-situ.
This option allows the reader to modify and use input data to store string
include/yyjson.h view on Meta::CPAN
This function is thread-safe when:
1. The `dat` is not modified by other threads.
2. The `alc` is thread-safe or NULL.
@param dat The JSON data (UTF-8 without BOM), null-terminator is not required.
If this parameter is NULL, the function will fail and return NULL.
The `dat` will not be modified without the flag `YYJSON_READ_INSITU`, so you
can pass a `const char *` string and case it to `char *` if you don't use
the `YYJSON_READ_INSITU` flag.
@param len The length of JSON data in bytes.
include/yyjson.h view on Meta::CPAN
/**
Read a JSON string.
This function is thread-safe.
@param dat The JSON data (UTF-8 without BOM), null-terminator is not required.
If this parameter is NULL, the function will fail and return NULL.
@param len The length of JSON data in bytes.
If this parameter is 0, the function will fail and return NULL.
@param flg The JSON read options.
Multiple options can be combined with `|` operator. 0 means no options.
include/yyjson.h view on Meta::CPAN
/**
Read a JSON number.
This function is thread-safe when data is not modified by other threads.
@param dat The JSON data (UTF-8 without BOM), null-terminator is required.
If this parameter is NULL, the function will fail and return NULL.
@param val The output value where result is stored.
If this parameter is NULL, the function will fail and return NULL.
The value will hold either UINT or SINT or REAL number;
@param flg The JSON read options.
include/yyjson.h view on Meta::CPAN
/**
Read a JSON number.
This function is thread-safe when data is not modified by other threads.
@param dat The JSON data (UTF-8 without BOM), null-terminator is required.
If this parameter is NULL, the function will fail and return NULL.
@param val The output value where result is stored.
If this parameter is NULL, the function will fail and return NULL.
The value will hold either UINT or SINT or REAL number;
@param flg The JSON read options.
view all matches for this distribution
view release on metacpan or search on metacpan
lib/File/Raw/Separated.pm view on Meta::CPAN
Empty unquoted field becomes C<undef>. Quoted empty (C<"">) stays the
empty string. Default false (always returns C<"">).
=item C<binary>
Skip UTF-8 BOM stripping and skip C<sv_utf8_decode> on each field.
Default false.
=item C<header>
Controls whether rows are emitted as arrayrefs (default) or hashrefs.
view all matches for this distribution
view release on metacpan or search on metacpan
include/frx/frx_enc.h view on Meta::CPAN
* first four bytes for the `<?xm` pattern in each width; else UTF-8. A
* UTF-8-compatible start is then told apart by peeking at the encoding
* declaration's name, which is ASCII in every encoding this accepts, so
* a document declaring iso-8859-1 is transcoded from it. The lexer
* afterwards checks the declaration it parses against what was detected
* (frx_lex.h): a BOM contradicted by the declaration is fatal, and so is
* a UTF-16 document declaring anything but UTF-16. UCS-4 and EBCDIC are
* detected and refused by name.
*
* The output is sized once from the bounds: UTF-16 to UTF-8 is at most
* 1.5x, Latin-1 at most 2x, so max_bytes is checked on the input and the
include/frx/frx_enc.h view on Meta::CPAN
* The map is one checkpoint per FRX_ENC_STEP input bytes recording the
* matching output offset; mapping an output offset back re-walks from the
* nearest checkpoint. The error path pays; the parse path does not.
*
* Under strict none of this runs: the 0.01 checks in frx_lex.h refuse
* UTF-16 by BOM or first pair and every declared non-UTF-8 encoding.
*
* Needs frx_err.h, frx_utf8.h. */
enum {
FRX_ENC_UTF8 = 0,
include/frx/frx_enc.h view on Meta::CPAN
frx_enc_kind_by_name(const char *v, size_t n)
{
if (frx_enc_ieq(v, n, "utf-8") || frx_enc_ieq(v, n, "utf8")) return FRX_ENC_UTF8;
if (frx_enc_ieq(v, n, "utf-16le")) return FRX_ENC_UTF16LE;
if (frx_enc_ieq(v, n, "utf-16be")) return FRX_ENC_UTF16BE;
if (frx_enc_ieq(v, n, "utf-16")) return FRX_ENC_UTF16LE; /* the endianness comes from the BOM */
if (frx_enc_ieq(v, n, "iso-8859-1") || frx_enc_ieq(v, n, "iso_8859-1")
|| frx_enc_ieq(v, n, "latin1") || frx_enc_ieq(v, n, "l1")) return FRX_ENC_LATIN1;
if (frx_enc_ieq(v, n, "us-ascii") || frx_enc_ieq(v, n, "ascii")
|| frx_enc_ieq(v, n, "ansi_x3.4-1968")) return FRX_ENC_ASCII;
return -1;
include/frx/frx_enc.h view on Meta::CPAN
e->kind = FRX_ENC_UTF8;
e->bom = 0;
e->override = 0;
/* a BOM is consumed whatever else is decided */
if (len >= 3 && in[0] == 0xEF && in[1] == 0xBB && in[2] == 0xBF) { e->kind = FRX_ENC_UTF8; e->bom = 3; }
else if (len >= 4 && in[0] == 0 && in[1] == 0 && in[2] == 0xFE && in[3] == 0xFF) { e->kind = FRX_ENC_UNSUPPORTED; }
else if (len >= 4 && in[0] == 0xFF && in[1] == 0xFE && in[2] == 0 && in[3] == 0) { e->kind = FRX_ENC_UNSUPPORTED; }
else if (len >= 2 && in[0] == 0xFE && in[1] == 0xFF) { e->kind = FRX_ENC_UTF16BE; e->bom = 2; }
else if (len >= 2 && in[0] == 0xFF && in[1] == 0xFE) { e->kind = FRX_ENC_UTF16LE; e->bom = 2; }
if (override) {
int k = frx_enc_kind_by_name(override, strlen(override));
if (k < 0) return (frx_err_set(err, FRX_E_ENCODING, 0, "the encoding named by the caller is not one this parser supports"), 0);
if (frx_enc_ieq(override, strlen(override), "utf-16") && e->bom == 0)
k = FRX_ENC_UTF16BE; /* RFC 2781: no BOM, big-endian */
if (e->kind == FRX_ENC_UNSUPPORTED) e->kind = k;
else if (e->bom && k != e->kind && !(e->kind == FRX_ENC_UTF16LE && frx_enc_ieq(override, strlen(override), "utf-16")))
return (frx_err_set(err, FRX_E_ENCODING, 0, "the encoding named by the caller contradicts the byte order mark"), 0);
else e->kind = k;
e->override = 1;
include/frx/frx_enc.h view on Meta::CPAN
if (e->kind == FRX_ENC_UNSUPPORTED)
return (frx_err_set(err, FRX_E_ENCODING, 0, "UCS-4 is not supported; transcode to UTF-8 or UTF-16 first"), 0);
if (e->bom) return 1;
/* no BOM: the first four bytes */
if (len >= 4) {
if (in[0] == 0 && in[1] == '<' && in[2] == 0 && in[3] == '?') { e->kind = FRX_ENC_UTF16BE; return 1; }
if (in[0] == '<' && in[1] == 0 && in[2] == '?' && in[3] == 0) { e->kind = FRX_ENC_UTF16LE; return 1; }
if (in[0] == 0 && in[1] == 0 && in[2] == 0 && in[3] == '<')
return (frx_err_set(err, FRX_E_ENCODING, 0, "UCS-4 is not supported; transcode to UTF-8 or UTF-16 first"), 0);
include/frx/frx_enc.h view on Meta::CPAN
*cp = in[pos];
return 1;
}
}
/* Transcode the input after the BOM into e->out. For UTF-8 nothing is
* allocated: out stays NULL and the caller uses the input past the BOM.
* 0 on refusal with err set at the input offset. */
static int
frx_enc_run(frx_enc *e, const unsigned char *in, size_t len, frx_err *err)
{
size_t pos, o = 0, cap, next_cp;
view all matches for this distribution
view release on metacpan or search on metacpan
include/ppport.h view on Meta::CPAN
BOL_t8_p8|5.033003||Viu
BOL_t8_pb|5.033003||Viu
BOL_tb|5.035004||Viu
BOL_tb_p8|5.033003||Viu
BOL_tb_pb|5.033003||Viu
BOM_UTF8|5.025005|5.003007|p
BOM_UTF8_FIRST_BYTE|5.019004||Viu
BOM_UTF8_TAIL|5.019004||Viu
boolSV|5.004000|5.003007|p
boot_core_builtin|5.035007||Viu
boot_core_mro|5.009005||Viu
boot_core_PerlIO|5.007002||Viu
boot_core_UNIVERSAL|5.003007||Viu
include/ppport.h view on Meta::CPAN
#endif
#endif
#if 'A' == 65
#ifndef BOM_UTF8
# define BOM_UTF8 "\xEF\xBB\xBF"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xEF\xBF\xBD"
#endif
#elif '^' == 95
#ifndef BOM_UTF8
# define BOM_UTF8 "\xDD\x73\x66\x73"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xDD\x73\x73\x71"
#endif
#elif '^' == 176
#ifndef BOM_UTF8
# define BOM_UTF8 "\xDD\x72\x65\x72"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xDD\x72\x72\x70"
#endif
view all matches for this distribution
view release on metacpan or search on metacpan
t/43-alias-bomb-refused.t view on Meta::CPAN
$y .= "l$_: &l$_\n a: *l@{[$_ - 1]}\n b: *l@{[$_ - 1]}\n"
for 1 .. $levels;
return $y;
}
my $BOMB = bomb_yaml(9); # refused, and harmless if it is not
my $BIG_BOMB = bomb_yaml(25); # the document that used to never come back
# ---------------------------------------------------------------------------
# The premise, pinned. The whole design rests on this: YAML::XS resolves an
# alias to the SAME reference rather than to a copy, so the parse returns a
# linear DAG and does NOT explode. If it ever started copying, the expansion
# would happen inside Load, no guard of ours could reach it, and the answer
# would have to be a pre-check on the raw bytes instead.
my $parsed = YAML::XS::Load($BIG_BOMB);
is(refaddr($parsed->{l1}{a}), refaddr($parsed->{l1}{b}),
'YAML::XS resolves an alias to the same reference, not to a copy');
is(refaddr($parsed->{l1}{a}), refaddr($parsed->{l0}),
'and that reference is the anchored node itself');
is(guarded(sub { File::SOPS::Format::YAML->parse($BIG_BOMB) }), 'OK',
'the parse itself still returns: the blowup is in the walks, not the load');
# ---------------------------------------------------------------------------
# Step 1: the probe. Everything below depends on the guard existing at all.
my $guard_holds = do {
my $ok = eval {
File::SOPS::_assert_expansion_bounded(YAML::XS::Load($BIG_BOMB));
1;
};
!$ok && $@ =~ /excessive aliasing/;
};
ok($guard_holds, '_assert_expansion_bounded refuses a 25-level alias bomb');
t/43-alias-bomb-refused.t view on Meta::CPAN
# ---------------------------------------------------------------------------
# The encrypt side. Used to expand 2**25 leaves and never come back.
refuses('encrypt', sub {
File::SOPS->encrypt(
data => YAML::XS::Load($BOMB),
recipients => [$public],
format => 'yaml',
);
});
# JSON has no aliases, but a caller can hand encrypt a shared structure of
# their own. Same blowup, no parser anywhere near it, one guard for both.
refuses('encrypt (format => json)', sub {
File::SOPS->encrypt(
data => YAML::XS::Load($BOMB),
recipients => [$public],
format => 'json',
);
});
t/43-alias-bomb-refused.t view on Meta::CPAN
# The reproduction. This is the document from the ticket, and the assertion is
# that it now comes back at all.
{
my $got = guarded(sub {
File::SOPS->encrypt(
data => YAML::XS::Load($BIG_BOMB),
recipients => [$public],
format => 'yaml',
);
});
if ($got eq 'HANG') {
t/43-alias-bomb-refused.t view on Meta::CPAN
# The message quotes sops's own wording and says how far out of proportion the
# document is, because "too big" without a number is not actionable.
my $msg = do {
eval {
File::SOPS->encrypt(
data => YAML::XS::Load($BOMB),
recipients => [$public],
format => 'yaml',
);
};
my $e = $@ // ''; $e =~ s/\s+/ /g; $e;
t/43-alias-bomb-refused.t view on Meta::CPAN
my $doc = $sane;
$doc =~ s/^plain: .*\n/$body/m or die "splice failed";
return $doc;
}
my $bomb_doc = bomb_document($BOMB);
isnt($bomb_doc, $sane, 'built an encrypted document carrying an alias bomb');
refuses('decrypt', sub {
File::SOPS->decrypt(encrypted => $bomb_doc, identities => [$secret]);
});
t/43-alias-bomb-refused.t view on Meta::CPAN
);
});
# The read-side reproduction, at the size that used to hang.
{
my $big_doc = bomb_document($BIG_BOMB);
my $got = guarded(sub {
File::SOPS->decrypt(
encrypted => $big_doc,
identities => [$secret],
ignore_mac => 1,
t/43-alias-bomb-refused.t view on Meta::CPAN
refuses('edit', sub {
File::SOPS->edit(file => $enc_path, identities => [$secret]);
});
}
my $plain_path = write_file(in_dir('bomb.yaml'), $BOMB);
my $out_path = in_dir('bomb.out.yaml');
refuses('encrypt_file', sub {
File::SOPS->encrypt_file(
input => $plain_path,
t/43-alias-bomb-refused.t view on Meta::CPAN
recipients => [$public],
);
});
ok(!-e $out_path, 'encrypt_file wrote no output on the refusal');
my $in_place_path = write_file(in_dir('in-place.yaml'), $BOMB);
refuses('encrypt_in_place', sub {
File::SOPS->encrypt_in_place(file => $in_place_path, recipients => [$public]);
});
is(do { open my $fh, '<', $in_place_path or die; local $/; <$fh> }, $BOMB,
'encrypt_in_place left the original untouched on the refusal');
# ---------------------------------------------------------------------------
# The two guards are ordered, and the order is load-bearing: the census memo
# in _expansion_census is filled on the way OUT, so a cycle would recurse
view all matches for this distribution
view release on metacpan or search on metacpan
BOL_t8_p8|5.033003||Viu
BOL_t8_pb|5.033003||Viu
BOL_tb|5.035004||Viu
BOL_tb_p8|5.033003||Viu
BOL_tb_pb|5.033003||Viu
BOM_UTF8|5.025005|5.003007|p
BOM_UTF8_FIRST_BYTE|5.019004||Viu
BOM_UTF8_TAIL|5.019004||Viu
boolSV|5.004000|5.003007|p
boot_core_builtin|5.035007||Viu
boot_core_mro|5.009005||Viu
boot_core_PerlIO|5.007002||Viu
boot_core_UNIVERSAL|5.003007||Viu
#endif
#endif
#if 'A' == 65
#ifndef BOM_UTF8
# define BOM_UTF8 "\xEF\xBB\xBF"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xEF\xBF\xBD"
#endif
#elif '^' == 95
#ifndef BOM_UTF8
# define BOM_UTF8 "\xDD\x73\x66\x73"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xDD\x73\x73\x71"
#endif
#elif '^' == 176
#ifndef BOM_UTF8
# define BOM_UTF8 "\xDD\x72\x65\x72"
#endif
#ifndef REPLACEMENT_CHARACTER_UTF8
# define REPLACEMENT_CHARACTER_UTF8 "\xDD\x72\x72\x70"
#endif
view all matches for this distribution
view release on metacpan or search on metacpan
lib/File/Text/CSV.pm view on Meta::CPAN
=item encoding
Encoding to open the file with. Default encoding is UTF-8, unless
header processing is enabled and the file starts with a byte order
mark (BOM).
=item append
If true, new records written will be appended to the file.
view all matches for this distribution
view release on metacpan or search on metacpan
lib/File/ValueFile/Simple/Reader.pm view on Meta::CPAN
delete $self->{features};
while (my $line = <$fh>) {
$line =~ s/\r?\n$//;
$line =~ s/#.*$//;
$line =~ s/^\xEF\xBB\xBF//; # skip BOMs.
$line =~ s/\s+/ /g;
$line =~ s/ $//;
$line =~ s/^ //;
next unless length $line;
view all matches for this distribution